-
Total Latency
{total_latency_fmt}
+
+
Total Latency (end-to-end)
{total_latency_fmt}
+
+
Total Agent Wall
{total_agent_fmt}
+
Total Grading
{total_grading_fmt}
Assistant Turns
{total_asst}
Avg Turn Latency
{avg_turn_fmt}
diff --git a/src/coder_eval/reports/markdown.py b/src/coder_eval/reports/markdown.py
index 05c3f858..70ffc52c 100644
--- a/src/coder_eval/reports/markdown.py
+++ b/src/coder_eval/reports/markdown.py
@@ -263,6 +263,30 @@ def _pass_rate_lines(summary: RunSummary) -> list[str]:
return lines
+def _row_agent_wall_ms(row: dict[str, Any]) -> float | None:
+ """A row's agent wall: the stored ``agent_wall_ms``, else the legacy derivation.
+
+ PREFERS THE STORED KEY, written once by ``result_metrics.agent_wall_ms``. A
+ ``run.json`` written before the key existed is still renderable: its 6-key
+ ``iterations`` projection carries each turn's ``duration_seconds``, and the
+ same rule (sum the positive ones) applies. Presence is tested with ``in``, not
+ with truthiness, so a stored ``None`` (no turn timed) stays ``None``.
+
+ Rationale: .claude/notes/reporting.md § Read the stored value, do not re-derive it
+ """
+ if "agent_wall_ms" in row:
+ return row["agent_wall_ms"]
+ seconds = [
+ d for t in row.get("iterations") or [] if isinstance(d := t.get("duration_seconds"), (int, float)) and d > 0
+ ]
+ return sum(seconds) * 1000.0 if seconds else None
+
+
+def _fmt_seconds_ms(ms: float | None) -> str:
+ """``ms`` as ``12.3s``, or ``N/A`` when it was not measured (never ``0.0s``)."""
+ return f"{ms / 1000.0:.1f}s" if ms is not None else "N/A"
+
+
class ReportGenerator:
"""Generates reports from evaluation results."""
@@ -344,9 +368,9 @@ def _generate_generation_metrics_section(task_results: list[dict[str, Any]]) ->
lines = [
"## Generation Metrics",
"",
- "| Task ID | Total Latency | Turns | Asst Turns | Avg Turn Latency "
+ "| Task ID | Total Latency (end-to-end) | Agent Wall | Turns | Asst Turns | Avg Turn Latency "
+ "| Startup | Generation | Tool exec | Teardown |",
- "|---------|---------------|-------|------------|------------------"
+ "|---------|----------------------------|------------|-------|------------|------------------"
+ "|---------|------------|-----------|----------|",
]
@@ -372,7 +396,11 @@ def _generate_generation_metrics_section(task_results: list[dict[str, Any]]) ->
format_ms(task.get(key)) for key in ("startup_ms", "generation_ms", "tool_ms", "teardown_ms")
)
- lines.append(f"| {task_id} | {total_latency} | {num_turns} | {asst_turns} | {avg_turn_str} | {buckets} |")
+ agent_wall = _fmt_seconds_ms(_row_agent_wall_ms(task))
+ lines.append(
+ f"| {task_id} | {total_latency} | {agent_wall} | {num_turns} | {asst_turns} "
+ + f"| {avg_turn_str} | {buckets} |"
+ )
return lines
@@ -434,7 +462,16 @@ def _summary_section_lines(summary: RunSummary) -> list[str]:
if scores:
lines.append(f"- **Avg Reliability Score**: {sum(scores) / len(scores):.3f}")
if durations:
- lines.append(f"- **Avg Generation Latency**: {sum(durations) / len(durations):.1f}s")
+ # `duration` is END-TO-END (setup + agent + grading), so the label says so.
+ # The agent's own share and the checker's share are averaged separately,
+ # each over only the rows that measured it (#212).
+ lines.append(f"- **Avg End-to-end Latency**: {sum(durations) / len(durations):.1f}s")
+ agent_walls = [ms for t in summary.task_results if (ms := _row_agent_wall_ms(t)) is not None]
+ if agent_walls:
+ lines.append(f"- **Avg Agent Wall**: {sum(agent_walls) / len(agent_walls) / 1000.0:.1f}s")
+ gradings = [ms for t in summary.task_results if (ms := t.get("grading_ms")) is not None]
+ if gradings:
+ lines.append(f"- **Avg Grading**: {sum(gradings) / len(gradings) / 1000.0:.1f}s")
total_asst_turns = sum(
sum(t.get("assistant_turn_count", 0) for t in task.get("iterations", [])) for task in summary.task_results
@@ -467,8 +504,11 @@ def _task_details_lines(summary: RunSummary) -> list[str]:
has_tags = any(t.get("tags") for t in summary.task_results)
has_cmds_efficiency = any(t.get("commands_efficiency") is not None for t in summary.task_results)
- header = "| Task ID | Status | Reliability Score | Latency |"
- separator = "|---------|--------|-------------------|---------|"
+ # `Latency` is END-TO-END: setup + the agent's turns + grading. `Agent Wall`
+ # and `Grading` split it, so a checker that runs a live command is not read
+ # as agent time (#212). An older run.json has no `grading_ms`: N/A, not 0.
+ header = "| Task ID | Status | Reliability Score | Latency (end-to-end) | Agent Wall | Grading |"
+ separator = "|---------|--------|-------------------|----------------------|------------|---------|"
if has_model:
header += " Model |"
separator += "-------|"
@@ -489,7 +529,13 @@ def _task_details_lines(summary: RunSummary) -> list[str]:
score_str = f"{weighted_score:.3f}" if weighted_score is not None else "N/A"
duration = f"{task_result['duration']:.1f}s"
- row = f"| {task_result['task_id']} | {task_result['status']} | {score_str} | {duration} |"
+ agent_wall = _fmt_seconds_ms(_row_agent_wall_ms(task_result))
+ grading = _fmt_seconds_ms(task_result.get("grading_ms"))
+
+ row = (
+ f"| {task_result['task_id']} | {task_result['status']} | {score_str} | {duration} "
+ + f"| {agent_wall} | {grading} |"
+ )
if has_model:
model = task_result.get("model_used") or "N/A"
row += f" {model} |"
diff --git a/src/coder_eval/result_metrics.py b/src/coder_eval/result_metrics.py
index 10be401e..3dde1b30 100644
--- a/src/coder_eval/result_metrics.py
+++ b/src/coder_eval/result_metrics.py
@@ -93,6 +93,30 @@ def turn_time_buckets(result: EvaluationResult) -> TurnTimeBuckets:
return TurnTimeBuckets(startup, generation, tool, teardown, unaccounted)
+def agent_wall_ms(result: EvaluationResult) -> float | None:
+ """Wall milliseconds the agent's own turns took: the sum of ``iterations[].duration_seconds``.
+
+ ``duration_seconds`` is the task's END-TO-END wall clock. It also holds
+ ``setup_ms`` (sandbox, pre_run) and ``grading_ms`` (every success check). A
+ checker that runs a live command, such as a tenant ``flow debug``, can spend
+ 30-120 s there, so ``duration_seconds`` is not a measure of the agent.
+
+ It is the SUM, not ``duration - setup - grading``. A detached grade does not
+ carry ``grading_ms`` and ``coder-eval execute`` never sets it, so the
+ subtraction has no value on those rows. The sum does. It is also the rule
+ the evalboard's ``agentSecondsFromRaw`` applies, so the two agree: only a
+ POSITIVE turn duration counts, because ``0.0`` is the unmeasured default.
+
+ ``None`` when no turn recorded a duration, never ``0.0`` (CE058).
+ """
+ seconds = [
+ t.duration_seconds
+ for t in result.iterations or []
+ if isinstance(t.duration_seconds, (int, float)) and math.isfinite(t.duration_seconds) and t.duration_seconds > 0
+ ]
+ return sum(seconds) * 1000.0 if seconds else None
+
+
def _sum_measured(values: Iterable[float | None]) -> float | None:
"""Sum what was measured, or ``None`` when nothing was.
diff --git a/src/coder_eval/run_record.py b/src/coder_eval/run_record.py
index 8f99719d..615efc07 100644
--- a/src/coder_eval/run_record.py
+++ b/src/coder_eval/run_record.py
@@ -17,7 +17,7 @@
from coder_eval.errors import truncate_crash_message
from coder_eval.models import EvaluationResult, FinalStatus, judge_cost_usd, simulator_cost_usd, sum_costs
-from coder_eval.result_metrics import expected_turns_overage, turn_time_buckets, visible_turn_count
+from coder_eval.result_metrics import agent_wall_ms, expected_turns_overage, turn_time_buckets, visible_turn_count
from coder_eval.result_metrics import has_final_reply as _has_final_reply
@@ -135,6 +135,12 @@ def eval_result_to_task_dict(
"generation_ms": _buckets.generation_ms,
"tool_ms": _buckets.tool_ms,
"teardown_ms": _buckets.teardown_ms,
+ # `duration` is END-TO-END: setup + the agent's turns + grading. These three
+ # split it so a slow checker is not read as a slow agent (#212). Each is
+ # `float | None`; None means not measured, never 0 (CE049).
+ "agent_wall_ms": agent_wall_ms(result),
+ "setup_ms": result.setup_ms,
+ "grading_ms": result.grading_ms,
"model_used": result.model_used,
"reference_similarity": ref_similarity,
# The UNCACHED slice, not TokenUsage.input_tokens; evalboard/lib/runs.ts
diff --git a/tests/_fixtures/report_snapshots/run_full.md b/tests/_fixtures/report_snapshots/run_full.md
index e576cf41..e7dc0f7d 100644
--- a/tests/_fixtures/report_snapshots/run_full.md
+++ b/tests/_fixtures/report_snapshots/run_full.md
@@ -13,17 +13,18 @@
- **Errors**: 0
- **Pass Rate**: 50.0% (1/2)
- **Avg Reliability Score**: 0.625
-- **Avg Generation Latency**: 10.2s
+- **Avg End-to-end Latency**: 10.2s
+- **Avg Agent Wall**: 3.1s
- **Total Assistant Turns**: 4
- **Crashed Partials**: 1 (0 recovered, 1 terminal)
- **Avg Ground Truth Similarity**: 0.715
## Task Details
-| Task ID | Status | Reliability Score | Latency | Model | Tags | Similarity | Cmd Efficiency |
-|---------|--------|-------------------|---------|-------|------|------------|----------------|
-| alpha | success | 0.950 | 12.5s | claude-haiku-4-5 | smoke, fast | 0.880 | 75.0% (4/6) |
-| beta | failure | 0.300 | 8.0s | claude-sonnet-4-6 | regression | 0.550 | 100.0% (3/3) |
+| Task ID | Status | Reliability Score | Latency (end-to-end) | Agent Wall | Grading | Model | Tags | Similarity | Cmd Efficiency |
+|---------|--------|-------------------|----------------------|------------|---------|-------|------|------------|----------------|
+| alpha | success | 0.950 | 12.5s | 4.2s | N/A | claude-haiku-4-5 | smoke, fast | 0.880 | 75.0% (4/6) |
+| beta | failure | 0.300 | 8.0s | 2.0s | N/A | claude-sonnet-4-6 | regression | 0.550 | 100.0% (3/3) |
## Run-time Notes
@@ -33,10 +34,10 @@
## Generation Metrics
-| Task ID | Total Latency | Turns | Asst Turns | Avg Turn Latency | Startup | Generation | Tool exec | Teardown |
-|---------|---------------|-------|------------|------------------|---------|------------|-----------|----------|
-| alpha | 12.5s | 1 | 3 | 4.2s | — | — | — | — |
-| beta | 8.0s | 1 | 1 | 2.0s | — | — | — | — |
+| Task ID | Total Latency (end-to-end) | Agent Wall | Turns | Asst Turns | Avg Turn Latency | Startup | Generation | Tool exec | Teardown |
+|---------|----------------------------|------------|-------|------------|------------------|---------|------------|-----------|----------|
+| alpha | 12.5s | 4.2s | 1 | 3 | 4.2s | — | — | — | — |
+| beta | 8.0s | 2.0s | 1 | 1 | 2.0s | — | — | — | — |
## Token Usage
diff --git a/tests/_fixtures/report_snapshots/run_minimal.md b/tests/_fixtures/report_snapshots/run_minimal.md
index be0e8308..ef926665 100644
--- a/tests/_fixtures/report_snapshots/run_minimal.md
+++ b/tests/_fixtures/report_snapshots/run_minimal.md
@@ -14,8 +14,8 @@
## Task Details
-| Task ID | Status | Reliability Score | Latency |
-|---------|--------|-------------------|---------|
+| Task ID | Status | Reliability Score | Latency (end-to-end) | Agent Wall | Grading |
+|---------|--------|-------------------|----------------------|------------|---------|
## Environment
diff --git a/tests/test_reports.py b/tests/test_reports.py
index 18e9cb47..4989f92c 100644
--- a/tests/test_reports.py
+++ b/tests/test_reports.py
@@ -218,7 +218,7 @@ def test_generate_markdown_basic():
# Check P0 aggregate metrics
assert "Avg Reliability Score**:" in report_md
- assert "Avg Generation Latency**:" in report_md
+ assert "Avg End-to-end Latency**:" in report_md
assert "Avg Ground Truth Similarity**:" in report_md
# Check task details table has new columns
@@ -397,9 +397,9 @@ def test_generate_markdown_generation_metrics_section():
assert "## Generation Metrics" in report_md
# task1: 1 turn, 0 asst turns (no assistant_turn_count in test data)
- assert "| task1 | 50.0s | 1 | 0 | 50.0s |" in report_md
+ assert "| task1 | 50.0s | 50.0s | 1 | 0 | 50.0s |" in report_md
# task2: 3 turns, avg turn = (25+22+23)/3 = 23.3s
- assert "| task2 | 70.0s | 3 | 0 | 23.3s |" in report_md
+ assert "| task2 | 70.0s | 70.0s | 3 | 0 | 23.3s |" in report_md
def test_generate_markdown_no_generation_metrics_without_turns():
@@ -448,7 +448,7 @@ def test_generate_markdown_aggregate_metrics():
# Avg reliability = (0.8 + 0.6) / 2 = 0.7
assert "Avg Reliability Score**: 0.700" in report_md
# Avg latency = (40 + 60) / 2 = 50.0s
- assert "Avg Generation Latency**: 50.0s" in report_md
+ assert "Avg End-to-end Latency**: 50.0s" in report_md
def test_generate_markdown_backward_compatible_task_results():
@@ -1535,3 +1535,64 @@ def test_importing_reports_does_not_pull_in_criteria(self):
code = "import coder_eval.reports, sys; print('coder_eval.criteria' in sys.modules)"
out = subprocess.run([sys.executable, "-c", code], capture_output=True, text=True, check=True)
assert out.stdout.strip() == "False", "coder_eval.reports must not import coder_eval.criteria at module level"
+
+
+def test_task_details_split_end_to_end_into_agent_wall_and_grading():
+ """`Latency` is end-to-end; a slow checker must not read as a slow agent (#212).
+
+ Numbers from adhoc-2026-10-01_19-24-36, calculator flow-v2 run 00: 101.3 s
+ end to end = 55.7 s of agent turns + 13.7 s setup + 27.7 s grading + residual.
+ """
+ row = _make_task_result(
+ "calc", "SUCCESS", 1.0, 101.25, iteration_count=1, turns=[{"iteration": 1, "duration_seconds": 55.66}]
+ )
+ row.update(agent_wall_ms=55657.3, setup_ms=13677.6, grading_ms=27656.9)
+ summary = RunSummary(
+ run_id="r",
+ start_time=datetime(2026, 10, 1, 19, 24, 36),
+ end_time=datetime(2026, 10, 1, 19, 26, 17),
+ total_duration_seconds=101.25,
+ tasks_run=1,
+ tasks_succeeded=1,
+ tasks_failed=0,
+ tasks_error=0,
+ task_results=[row],
+ framework_version="0.1.0",
+ environment_info={},
+ )
+ report_md = ReportGenerator.generate_markdown(summary)
+ assert "| Task ID | Status | Reliability Score | Latency (end-to-end) | Agent Wall | Grading |" in report_md
+ assert "| calc | SUCCESS | 1.000 | 101.2s | 55.7s | 27.7s |" in report_md
+ assert "Avg Agent Wall**: 55.7s" in report_md
+ assert "Avg Grading**: 27.7s" in report_md
+
+
+def test_task_details_legacy_row_derives_agent_wall_and_never_fakes_grading():
+ """A run.json written before #212 has no `agent_wall_ms` / `grading_ms` keys.
+
+ Agent wall falls back to the row's own `iterations[].duration_seconds`; grading
+ was never recorded on the row, so it renders N/A, never 0.0s (CE058).
+ """
+ row = _make_task_result(
+ "old",
+ "SUCCESS",
+ 1.0,
+ 90.0,
+ turns=[{"iteration": 1, "duration_seconds": 30.0}, {"iteration": 2, "duration_seconds": 0.0}],
+ )
+ summary = RunSummary(
+ run_id="r",
+ start_time=datetime(2026, 10, 1, 19, 24, 36),
+ end_time=datetime(2026, 10, 1, 19, 26, 6),
+ total_duration_seconds=90.0,
+ tasks_run=1,
+ tasks_succeeded=1,
+ tasks_failed=0,
+ tasks_error=0,
+ task_results=[row],
+ framework_version="0.1.0",
+ environment_info={},
+ )
+ report_md = ReportGenerator.generate_markdown(summary)
+ assert "| old | SUCCESS | 1.000 | 90.0s | 30.0s | N/A |" in report_md
+ assert "Avg Grading" not in report_md
diff --git a/tests/test_run_record.py b/tests/test_run_record.py
index c2d0f82c..922dc3be 100644
--- a/tests/test_run_record.py
+++ b/tests/test_run_record.py
@@ -73,6 +73,9 @@
"task_id": "char-task",
"task_path": None,
"teardown_ms": None,
+ "agent_wall_ms": 2500.0,
+ "setup_ms": None,
+ "grading_ms": None,
"tool_ms": None,
"total_cost_usd": 0.5,
"total_tokens": 1000,
@@ -271,3 +274,30 @@ def test_none_when_run_limits_not_dict(self):
)
d = eval_result_to_task_dict(result)
assert d["expected_turns"] is None
+
+
+class TestAgentWallIsSplitFromEndToEnd:
+ """`duration` is end-to-end; the row also carries the agent's and the checker's share (#212)."""
+
+ def test_row_carries_setup_grading_and_agent_wall(self):
+ result = _result()
+ result.setup_ms = 14000.0
+ result.grading_ms = 28000.0
+ row = eval_result_to_task_dict(result)
+ assert row["duration"] == 3.0
+ assert row["agent_wall_ms"] == 2500.0
+ assert row["setup_ms"] == 14000.0
+ assert row["grading_ms"] == 28000.0
+
+ def test_agent_wall_is_none_when_no_turn_was_timed(self):
+ result = _result()
+ result.iterations = []
+ assert eval_result_to_task_dict(result)["agent_wall_ms"] is None
+
+ def test_agent_wall_sums_turns_and_skips_the_unmeasured_default(self):
+ result = _result()
+ timed = result.iterations[0]
+ untimed = timed.model_copy(update={"iteration": 2, "duration_seconds": 0.0})
+ third = timed.model_copy(update={"iteration": 3, "duration_seconds": 1.5})
+ result.iterations = [timed, untimed, third]
+ assert eval_result_to_task_dict(result)["agent_wall_ms"] == 4000.0