diff --git a/README.md b/README.md index 865899e..0c229cd 100644 --- a/README.md +++ b/README.md @@ -300,40 +300,28 @@ Set `BENCH_PROVIDER=claude_code` to run the pipeline on the subscription that al ## What It Costs -Every governed `Write`, `Edit`, or `MultiEdit` is three sequential model calls, and Claude Code makes many small edits. The figures below come from Bench's own operational ledger, through the helpers in `utils/stats.py` that `python -m cli stats` and the viewer use. `cli stats` prints the all-entries rows (the "Tokens per edit", "Seconds per edit", and "Seconds by stage" lines) and the viewer's dashboard plots verdicts and latency by week; the per-week token rows come from the same helpers over one week's entries, which this reproduces on any chain: - -```python -from collections import defaultdict -from ledger.chain import load_ledger -from utils.stats import billed_tokens_per_entry, seconds_by_stage, tokens_per_entry, week_of - -weeks = defaultdict(list) -for entry in load_ledger(): - weeks[week_of(entry.get("timestamp", ""))].append(entry) -for week, entries in sorted(weeks.items()): - print(week, tokens_per_entry(entries), billed_tokens_per_entry(entries), seconds_by_stage(entries)["total"]) -``` +Every governed `Write`, `Edit`, or `MultiEdit` is three sequential model calls, and Claude Code makes many small edits. The figures below come from Bench's own operational ledger. `python -m cli stats` prints each of them (the "Tokens per edit", "Tokens by week", "Seconds per edit", and "Seconds by stage" lines), and the viewer's dashboard shows the same distributions, per week and over every entry, so anyone with a chain can reproduce the table for their own ledger rather than take an estimate. Every weekly figure is built on one grouping, `utils.stats.entries_by_week`, so the tables cannot disagree about which week an entry belongs to. -Tokens per governed edit, all stages, by week of the operational chain, as of 2026-09-07: +Tokens per governed edit, all stages, by week of the operational chain. This is a fixed snapshot of the chain as it stood at 2026-09-07T06:37:19Z, 2,767 entries of which 2,743 carry usage, not a live weekly value; the release week was still open when it was taken, and the chain has moved on since. -| Week | Governed edits | Median tokens | 90th percentile | Median at cached rates | +| Week | Entries with usage | Median tokens | 90th percentile | Median at cached rates | |---|---|---|---|---| -| 2026-W36, the v2.1 roadmap week | 1,797 | 34,747 | 42,650 | 33,451 | -| 2026-W37, the release week so far | 80 | 19,483 | 36,811 | 18,124 | -| Every entry with usage recorded (2,705) | | 34,967 | 44,486 | 33,979 | +| 2026-W36, the v2.1 roadmap week | 1,777 | 34,747 | 42,650 | 33,451 | +| 2026-W37, the release week at the snapshot | 118 | 20,383 | 36,811 | 19,085 | +| Every entry with usage in the snapshot | 2,743 | 34,917 | 44,401 | 33,899 | -The release-week row is the first under every 2.1 cost change together, and it is a small sample of light edits: most were docstring changes, which the Challenger clears in seconds and the Defender then skips. It is not a settled reduction. The roadmap-week row is the number to plan on until the release week fills out, and the roadmap's target of a median under 20,000 tokens stays a target. +The release-week row is the first under every 2.1 cost change together, and it is a small sample of light edits: most were docstring and test changes, which the Challenger clears in seconds and the Defender then skips. It is not a settled reduction. On the day of the snapshot that median moved from under 20,000 to just over it as heavier edits landed. The roadmap-week row is the number to plan on until a full week of ordinary edits is in, and the roadmap's target of a median under 20,000 tokens stays a target. -Wall time, recorded per stage, on the `claude_code` provider, which cold-starts a `claude` process per stage. Over the 1,635 timed entries: +Wall time, recorded per stage, on the `claude_code` provider, which cold-starts a `claude` process per stage. Over the 1,673 timed entries in the same snapshot: | Stage | Median seconds | 90th percentile | |---|---|---| -| Challenger | 21.8 | 38.7 | -| Defender | 15.8 | 38.6 | -| Oracle | 23.1 | 34.9 | -| Whole edit | 61.9 | 107.7 | +| Challenger | 21.7 | 38.5 | +| Defender | 15.3 | 38.5 | +| Oracle | 23.1 | 34.7 | +| Whole edit | 61.1 | 106.8 | -By week, the whole-edit median was 63.5 seconds in 2026-W36 and 22.0 seconds in 2026-W37, with the same caveat on the second figure. +By week, the whole-edit median was 63.5 seconds in 2026-W36 and 26.6 seconds in 2026-W37, with the same caveat on the second figure. The Defender median is low because a CLEAR challenge skips it and records zero seconds. The direct API providers avoid the process start but not the sequential three-call shape. Both tables will drift as the ledger grows; run `python -m cli stats` for the current figures on your own chain, and `python -m cli viewer` for the per-week view. diff --git a/cli/commands.py b/cli/commands.py index a54a7e2..edc7286 100644 --- a/cli/commands.py +++ b/cli/commands.py @@ -68,6 +68,7 @@ normalized_entries, pct, seconds_by_stage, + tokens_by_week, tokens_per_entry, ) from utils.owneronly import open_owner_only @@ -986,6 +987,16 @@ def cmd_stats() -> int: f"p90 {int(billed['p90']):,} " f"(reads {CACHE_READ_RATE:g}x, writes {CACHE_WRITE_RATE:g}x)" ) + # The same distribution per ISO week, so a release week's figure + # can be read off the chain instead of recomputed by hand. + print("Tokens by week :") + for row in tokens_by_week(entries): + print( + f" {row['week']}: median {int(row['median']):,}, " + f"p90 {int(row['p90']):,}, " + f"at cached rates {int(row['billed_median']):,} " + f"({int(row['entries'])} with usage)" + ) else: print("Tokens per edit : n/a (no entry carries token usage)") seconds: dict[str, dict[str, float | int]] = seconds_by_stage(entries) diff --git a/tests/test_commands.py b/tests/test_commands.py index 60e8a06..17724f8 100644 --- a/tests/test_commands.py +++ b/tests/test_commands.py @@ -219,6 +219,37 @@ def test_stats_prints_token_and_timing_distributions(self) -> None: text, ) + def test_stats_prints_tokens_by_week(self) -> None: + entries: list[dict] = _entries() + entries[0]["timestamp"] = "2026-01-05T00:00:00+00:00" + entries[0]["oracle"] = { + **entries[0].get("oracle", {}), + "_tokens": {"input": 500, "output": 50}, + } + entries[1]["timestamp"] = "2026-01-12T00:00:00+00:00" + entries[1]["oracle"] = { + **entries[1].get("oracle", {}), + "_tokens": {"input": 1000, "output": 0, "cache_read": 1000, "cache_creation": 0}, + } + out = io.StringIO() + with patch("cli.commands.load_ledger", return_value=entries): + with patch("cli.commands.verify_chain", return_value=_valid_verify()): + with redirect_stdout(out): + code: int = cmd_stats() + self.assertEqual(code, 0) + text: str = out.getvalue() + self.assertIn("Tokens by week :", text) + self.assertIn( + " 2026-W02: median 550, p90 550, at cached rates 550 (1 with usage)", + text, + ) + # 1,000 input entirely read from cache bills at 100. + self.assertIn( + " 2026-W03: median 1,000, p90 1,000, at cached rates 100 " + "(1 with usage)", + text, + ) + def test_stats_prints_models_by_stage_with_override_counts(self) -> None: entries: list[dict] = _entries() entries[0]["oracle"] = {"verdict": "PASS", "_model": "claude-opus-4-8", "_model_override": False} diff --git a/tests/test_stats.py b/tests/test_stats.py index 004e260..cf22af5 100644 --- a/tests/test_stats.py +++ b/tests/test_stats.py @@ -20,8 +20,10 @@ MODEL_NOT_REACHED, MODEL_SKIPPED, MODEL_UNRECORDED, + UNKNOWN_WEEK, citations_by_constraint, compute_ledger_stats, + entries_by_week, entry_has_pipeline_error, models_by_stage, stage_model_label, @@ -34,6 +36,7 @@ stats_by_scope, stats_by_week, tokens_by_stage, + tokens_by_week, tokens_per_entry, billed_input, billed_tokens_per_entry, @@ -674,6 +677,120 @@ def test_empty_ledger(self) -> None: self.assertEqual(summary[stage], {"entries": 0, "median": 0.0, "p90": 0.0}) +class EntriesByWeekTests(unittest.TestCase): + """The one grouping every weekly table is built on.""" + + def test_iso_week_boundaries(self) -> None: + # ISO weeks run Monday to Sunday. Sunday 2026-01-04 closes 2026-W01 + # and Monday 2026-01-05 opens W02. The last days of 2025 belong to + # 2026-W01, and 2026 has 53 ISO weeks, so New Year's Day 2027 falls + # in 2026-W53. + entries: list[dict] = [ + {"timestamp": "2026-01-04T23:59:59+00:00", "id": "sunday"}, + {"timestamp": "2026-01-05T00:00:00+00:00", "id": "monday"}, + {"timestamp": "2025-12-29T12:00:00+00:00", "id": "year-end"}, + {"timestamp": "2027-01-01T00:00:00+00:00", "id": "year-start"}, + ] + groups: dict[str, list[dict]] = entries_by_week(entries) + self.assertEqual( + {week: [e["id"] for e in rows] for week, rows in groups.items()}, + { + "2026-W01": ["sunday", "year-end"], + "2026-W02": ["monday"], + "2026-W53": ["year-start"], + }, + ) + + def test_unparseable_and_missing_timestamps_bucket_under_unknown(self) -> None: + entries: list[dict] = [ + {"timestamp": "not a date", "id": "bad"}, + {"id": "missing"}, + {"timestamp": "2026-01-05T00:00:00+00:00", "id": "good"}, + ] + groups: dict[str, list[dict]] = entries_by_week(entries) + self.assertEqual([e["id"] for e in groups[UNKNOWN_WEEK]], ["bad", "missing"]) + self.assertEqual([e["id"] for e in groups["2026-W02"]], ["good"]) + # Renderers sort the labels; the unknown bucket sorts after any year. + self.assertEqual(sorted(groups)[-1], UNKNOWN_WEEK) + + def test_every_weekly_table_groups_the_same_way(self) -> None: + entries: list[dict] = [ + {"timestamp": "2026-01-05T00:00:00+00:00", "oracle": {"verdict": "PASS", "_seconds": 2.0, "_tokens": {"input": 10, "output": 1}}}, + {"timestamp": "2026-01-11T00:00:00+00:00", "oracle": {"verdict": "VETO"}}, + {"timestamp": "2026-01-12T00:00:00+00:00", "oracle": {"verdict": "PASS", "_seconds": 3.0}}, + {"timestamp": "broken", "oracle": {"verdict": "PASS", "_tokens": {"input": 5, "output": 5}}}, + ] + groups: dict[str, list[dict]] = entries_by_week(entries) + tallies: list[dict] = stats_by_week(entries) + self.assertEqual([row["week"] for row in tallies], sorted(groups)) + for row in tallies: + self.assertEqual(row["total"], len(groups[row["week"]])) + # The other two tables omit weeks with nothing to measure, and every + # week they do show is one of the same groups. + self.assertEqual([r["week"] for r in latency_by_week(entries)], ["2026-W02", "2026-W03"]) + self.assertEqual([r["week"] for r in tokens_by_week(entries)], ["2026-W02", UNKNOWN_WEEK]) + for table in (latency_by_week(entries), tokens_by_week(entries)): + self.assertLessEqual({r["week"] for r in table}, set(groups)) + + +class TokensByWeekTests(unittest.TestCase): + """The per-week token distribution the README quotes for a release week.""" + + def test_weeks_without_usage_are_omitted_and_malformed_usage_is_skipped( + self, + ) -> None: + entries: list[dict] = [ + {"timestamp": "2026-01-05T00:00:00+00:00", "oracle": {"_tokens": {"input": 100, "output": 10}}}, + { + "timestamp": "2026-01-06T00:00:00+00:00", + "challenger": {"_tokens": {"input": 200, "output": 20}}, + "oracle": {"_tokens": {"input": 300, "output": 30}}, + }, + {"timestamp": "2026-01-07T00:00:00+00:00", "verdict": "PASS"}, + {"timestamp": "2026-01-12T00:00:00+00:00", "oracle": {"_tokens": {"input": True, "output": "many"}}}, + {"timestamp": "2026-01-19T00:00:00+00:00", "oracle": {"_tokens": {"input": 1000, "output": 0}}}, + {"timestamp": "not a date", "oracle": {"_tokens": {"input": 5, "output": 5}}}, + ] + # W02 totals 110 and 550 (the untracked third entry is not counted); + # W03 holds only malformed usage and is omitted; W04 is 1000; the + # unparseable timestamp lands in the unknown bucket, last. + self.assertEqual( + tokens_by_week(entries), + [ + {"week": "2026-W02", "entries": 2, "median": 330.0, "p90": 550.0, "billed_median": 330.0, "billed_p90": 550.0}, + {"week": "2026-W04", "entries": 1, "median": 1000.0, "p90": 1000.0, "billed_median": 1000.0, "billed_p90": 1000.0}, + {"week": UNKNOWN_WEEK, "entries": 1, "median": 10.0, "p90": 10.0, "billed_median": 10.0, "billed_p90": 10.0}, + ], + ) + + def test_cache_reads_are_priced_in_the_billed_figures(self) -> None: + entries: list[dict] = [ + { + "timestamp": "2026-01-05T00:00:00+00:00", + "oracle": {"_tokens": {"input": 1000, "output": 10, "cache_read": 900, "cache_creation": 0}}, + } + ] + row: dict = tokens_by_week(entries)[0] + # 100 uncached + 900 at the cached rate + 10 output. + self.assertEqual(row["median"], 1010.0) + self.assertEqual(row["billed_median"], 200.0) + + def test_a_single_week_agrees_with_the_all_entries_figures(self) -> None: + entries: list[dict] = [ + {"timestamp": "2026-01-05T00:00:00+00:00", "oracle": {"_tokens": {"input": 100, "output": 10}}}, + {"timestamp": "2026-01-06T00:00:00+00:00", "oracle": {"_tokens": {"input": 400, "output": 40, "cache_read": 200}}}, + {"timestamp": "2026-01-07T00:00:00+00:00", "oracle": {"_tokens": {"input": 900, "output": 90}}}, + ] + row: dict = tokens_by_week(entries)[0] + overall: dict = tokens_per_entry(entries) + billed: dict = billed_tokens_per_entry(entries) + self.assertEqual((row["entries"], row["median"], row["p90"]), (overall["entries"], overall["median"], overall["p90"])) + self.assertEqual((row["billed_median"], row["billed_p90"]), (billed["median"], billed["p90"])) + + def test_empty_ledger(self) -> None: + self.assertEqual(tokens_by_week([]), []) + + class LatencyByWeekTests(unittest.TestCase): def test_weeks_without_a_timing_are_omitted(self) -> None: entries: list[dict] = [ diff --git a/tests/test_viewer.py b/tests/test_viewer.py index 2d097bd..5a1a744 100644 --- a/tests/test_viewer.py +++ b/tests/test_viewer.py @@ -40,6 +40,7 @@ pct, stats_by_scope, stats_by_week, + tokens_by_week, ) from utils.viewer import generate_viewer_html # noqa: E402 @@ -327,6 +328,44 @@ def test_token_table_prices_cache_reads_at_the_cached_rate(self) -> None: html_out, ) + def test_token_table_has_a_row_per_week_with_usage(self) -> None: + """The release-week token median the roadmap's R3 asks the dashboard + to show, restating utils.stats.tokens_by_week.""" + chain: list[dict] = _build_valid_chain(3) + chain[0]["oracle"]["_tokens"] = {"input": 1000, "output": 10} + chain[0]["entry_hash"] = compute_entry_hash(chain[0]) + chain[1]["previous_hash"] = chain[0]["entry_hash"] + chain[1]["timestamp"] = "2026-01-12T00:00:00+00:00" # ISO week 3 + chain[1]["oracle"]["_tokens"] = { + "input": 3000, + "output": 20, + "cache_read": 2000, + "cache_creation": 0, + } + chain[1]["entry_hash"] = compute_entry_hash(chain[1]) + chain[2]["previous_hash"] = chain[1]["entry_hash"] + chain[2]["timestamp"] = "2026-01-19T00:00:00+00:00" # week 4, no usage + chain[2]["entry_hash"] = compute_entry_hash(chain[2]) + html_out: str = self._render(chain) + + rows: list[dict] = tokens_by_week(chain) + self.assertEqual([row["week"] for row in rows], ["2026-W01", "2026-W03"]) + for row in rows: + self.assertIn( + self._row([ + row["week"], + f"{row['entries']:,}", + f"{int(row['median']):,}", + f"{int(row['p90']):,}", + f"{int(row['billed_median']):,}", + ]), + html_out, + ) + # And in plain figures: 1,010 uncached; 3,020 of which 2,000 read + # from cache bills as 1,000 + 200 + 20. + self.assertIn(self._row(["2026-W01", "1", "1,010", "1,010", "1,010"]), html_out) + self.assertIn(self._row(["2026-W03", "1", "3,020", "3,020", "1,220"]), html_out) + def test_latency_table_restates_seconds_by_stage(self) -> None: chain: list[dict] = _build_valid_chain(3) chain[0]["oracle"]["_seconds"] = 10.0 diff --git a/utils/stats.py b/utils/stats.py index df49cff..5fac00a 100644 --- a/utils/stats.py +++ b/utils/stats.py @@ -156,16 +156,31 @@ def _group_tallies(groups: dict[str, list[dict]], key: str) -> list[dict]: return rows -def stats_by_week(entries: list[dict]) -> list[dict]: - """Verdict tallies per ISO week of entry timestamp, oldest first. - - Each row is a _tally_verdicts dict plus "week". Entries whose timestamp - does not parse land in the UNKNOWN_WEEK bucket rather than vanishing. +def entries_by_week(entries: list[dict]) -> dict[str, list[dict]]: + """Entries grouped by the ISO week of their timestamp, in ledger order. + + The one grouping every per-week figure is built on (verdict tallies, + latency, tokens), so no two weekly tables can disagree about which week + an entry belongs to. A week is the ISO week, Monday to Sunday, of the + timestamp's calendar date, so the days around a year end belong to + whichever year's week they fall in. An entry whose timestamp does not + parse lands in the UNKNOWN_WEEK bucket rather than vanishing. Keys are + in first-seen order; every renderer sorts them, which puts the unknown + bucket last. """ groups: dict[str, list[dict]] = {} for entry in entries: groups.setdefault(week_of(entry.get("timestamp")), []).append(entry) - return _group_tallies(groups, "week") + return groups + + +def stats_by_week(entries: list[dict]) -> list[dict]: + """Verdict tallies per ISO week of entry timestamp, oldest first. + + Each row is a _tally_verdicts dict plus "week", over the groups + entries_by_week makes. + """ + return _group_tallies(entries_by_week(entries), "week") def stats_by_scope(entries: list[dict], project_root: str) -> list[dict]: @@ -558,22 +573,50 @@ def seconds_by_stage(entries: list[dict]) -> dict[str, dict[str, float | int]]: def latency_by_week(entries: list[dict]) -> list[dict]: """Per-entry total wall time per ISO week, oldest first. - Each row is a _distribution_summary dict plus "week". Only entries that - recorded at least one stage timing contribute, and weeks with none are - omitted rather than shown as zero, so the table cannot imply a verdict - was instant when it was merely unmeasured. + Each row is a _distribution_summary dict plus "week", over the groups + entries_by_week makes. Only entries that recorded at least one stage + timing contribute, and weeks with none are omitted rather than shown + as zero, so the table cannot imply a verdict was instant when it was + merely unmeasured. """ - groups: dict[str, list[float]] = {} - for entry in entries: - seconds: dict[str, float] = _stage_seconds(entry) - if not seconds: + rows: list[dict] = [] + groups: dict[str, list[dict]] = entries_by_week(entries) + for label in sorted(groups): + totals: list[float] = [] + for entry in groups[label]: + seconds: dict[str, float] = _stage_seconds(entry) + if seconds: + totals.append(sum(seconds.values())) + if not totals: continue - groups.setdefault(week_of(entry.get("timestamp")), []).append( - sum(seconds.values()) - ) + row: dict = _distribution_summary(totals) + row["week"] = label + rows.append(row) + return rows + + +def tokens_by_week(entries: list[dict]) -> list[dict]: + """Per-entry token totals per ISO week, oldest first. + + Each row is tokens_per_entry over that week's entries plus "week", and + "billed_median" and "billed_p90" from billed_tokens_per_entry over the + same entries, so a week's row is exactly what the all-entries lines + would print for that week alone; the groups are the ones + entries_by_week makes. Only entries that recorded usable usage + contribute, and weeks with none are omitted rather than shown as zero, + so the table cannot imply an edit was free when it was merely + unmeasured. This is the row the README quotes for the release week. + """ rows: list[dict] = [] + groups: dict[str, list[dict]] = entries_by_week(entries) for label in sorted(groups): - row: dict = _distribution_summary(groups[label]) + summary: dict[str, float | int] = tokens_per_entry(groups[label]) + if not summary["entries"]: + continue + billed: dict[str, float | int] = billed_tokens_per_entry(groups[label]) + row: dict = dict(summary) + row["billed_median"] = billed["median"] + row["billed_p90"] = billed["p90"] row["week"] = label rows.append(row) return rows diff --git a/utils/viewer.py b/utils/viewer.py index 403d8e4..111a2f1 100644 --- a/utils/viewer.py +++ b/utils/viewer.py @@ -40,6 +40,7 @@ seconds_by_stage, stats_by_week, tokens_by_stage, + tokens_by_week, tokens_per_entry, billed_tokens_per_entry, CACHE_READ_RATE, @@ -483,6 +484,16 @@ def _latency_cells(row: dict) -> list[str]: latency_week_rows: list[list[str]] = [ [str(row.get("week", ""))] + _latency_cells(row) for row in latency_weeks ] + token_week_rows: list[list[str]] = [ + [ + str(row.get("week", "")), + f"{int(row.get('entries', 0)):,}", + f"{int(row.get('median', 0)):,}", + f"{int(row.get('p90', 0)):,}", + f"{int(row.get('billed_median', 0)):,}", + ] + for row in tokens_by_week(entries) + ] week_rows: list[list[str]] = [ [str(row.get("week", ""))] + _verdict_cells(row) for row in weeks @@ -595,7 +606,15 @@ def _scope_cells(rows: list[dict]) -> list[list[str]]: ) + "\n " + _token_distribution_line(per_entry, billed_per_entry) - + "\n\n" + + "\n " + + _table( + ["Week", "Entries", "Median tokens", "p90", "Median at cached rates"], + token_week_rows, "No token figures recorded.", + ) + + '\n
Per week, over entries that carry usage, the same ' + "distribution as the line above for that week alone. A week with no " + "usage recorded is omitted rather than shown as zero.
\n" + "\n" '