From 13414aece7e0f365f1303175a057e33e72aa6ad9 Mon Sep 17 00:00:00 2001 From: Dana Burks Date: Sun, 6 Sep 2026 23:39:12 -0700 Subject: [PATCH 1/3] [bench] stats: one weekly grouping; tokens by week in cli stats and the dashboard Release check R3 requires the dashboard to show the release week's median tokens per edit, and neither cli stats nor the viewer exposed tokens by week: the CLI printed tokens over all entries, and the dashboard grouped only verdicts and latency by week. utils.stats.entries_by_week is now the one grouping every per-week figure is built on: entries by the ISO week of their timestamp, with unparseable timestamps in the unknown bucket. stats_by_week and latency_by_week are refactored onto it with their outputs unchanged, which the existing exact-equality tests confirm, and tokens_by_week builds on it too: per week, the same median and p90 the all-entries line prints plus the cached-rate medians, over entries that recorded usage, with empty weeks omitted rather than shown as zero. cli stats prints a Tokens by week block, one line per week, and the viewer's tokens card gains a per-week table. Tests cover the ISO week boundaries (Sunday closes W01 and Monday opens W02; the last days of 2025 fall in 2026-W01; New Year's Day 2027 falls in 2026-W53), the unknown bucket, all three weekly tables sharing the grouping, missing and malformed usage, cached-rate pricing, agreement with the all-entries figures for a single week, the empty ledger, the dashboard row, and the CLI line. The README's What It Costs section is refreshed as a fixed snapshot of the chain at 2026-09-07T06:37:19Z (2,767 entries, 2,743 with usage), stated as such rather than as a live weekly value, and the reproducibility paragraph now names the two interfaces that expose the weekly rows and the shared grouping behind them. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_0122bFBUvdtXT245QFoYDkMf --- README.md | 38 +++++-------- cli/commands.py | 10 ++++ tests/test_commands.py | 28 ++++++++++ tests/test_stats.py | 117 +++++++++++++++++++++++++++++++++++++++++ tests/test_viewer.py | 39 ++++++++++++++ utils/stats.py | 79 +++++++++++++++++++++------- utils/viewer.py | 21 +++++++- 7 files changed, 288 insertions(+), 44 deletions(-) diff --git a/README.md b/README.md index 865899e..0c229cd 100644 --- a/README.md +++ b/README.md @@ -300,40 +300,28 @@ Set `BENCH_PROVIDER=claude_code` to run the pipeline on the subscription that al ## What It Costs -Every governed `Write`, `Edit`, or `MultiEdit` is three sequential model calls, and Claude Code makes many small edits. The figures below come from Bench's own operational ledger, through the helpers in `utils/stats.py` that `python -m cli stats` and the viewer use. `cli stats` prints the all-entries rows (the "Tokens per edit", "Seconds per edit", and "Seconds by stage" lines) and the viewer's dashboard plots verdicts and latency by week; the per-week token rows come from the same helpers over one week's entries, which this reproduces on any chain: - -```python -from collections import defaultdict -from ledger.chain import load_ledger -from utils.stats import billed_tokens_per_entry, seconds_by_stage, tokens_per_entry, week_of - -weeks = defaultdict(list) -for entry in load_ledger(): - weeks[week_of(entry.get("timestamp", ""))].append(entry) -for week, entries in sorted(weeks.items()): - print(week, tokens_per_entry(entries), billed_tokens_per_entry(entries), seconds_by_stage(entries)["total"]) -``` +Every governed `Write`, `Edit`, or `MultiEdit` is three sequential model calls, and Claude Code makes many small edits. The figures below come from Bench's own operational ledger. `python -m cli stats` prints each of them (the "Tokens per edit", "Tokens by week", "Seconds per edit", and "Seconds by stage" lines), and the viewer's dashboard shows the same distributions, per week and over every entry, so anyone with a chain can reproduce the table for their own ledger rather than take an estimate. Every weekly figure is built on one grouping, `utils.stats.entries_by_week`, so the tables cannot disagree about which week an entry belongs to. -Tokens per governed edit, all stages, by week of the operational chain, as of 2026-09-07: +Tokens per governed edit, all stages, by week of the operational chain. This is a fixed snapshot of the chain as it stood at 2026-09-07T06:37:19Z, 2,767 entries of which 2,743 carry usage, not a live weekly value; the release week was still open when it was taken, and the chain has moved on since. -| Week | Governed edits | Median tokens | 90th percentile | Median at cached rates | +| Week | Entries with usage | Median tokens | 90th percentile | Median at cached rates | |---|---|---|---|---| -| 2026-W36, the v2.1 roadmap week | 1,797 | 34,747 | 42,650 | 33,451 | -| 2026-W37, the release week so far | 80 | 19,483 | 36,811 | 18,124 | -| Every entry with usage recorded (2,705) | | 34,967 | 44,486 | 33,979 | +| 2026-W36, the v2.1 roadmap week | 1,777 | 34,747 | 42,650 | 33,451 | +| 2026-W37, the release week at the snapshot | 118 | 20,383 | 36,811 | 19,085 | +| Every entry with usage in the snapshot | 2,743 | 34,917 | 44,401 | 33,899 | -The release-week row is the first under every 2.1 cost change together, and it is a small sample of light edits: most were docstring changes, which the Challenger clears in seconds and the Defender then skips. It is not a settled reduction. The roadmap-week row is the number to plan on until the release week fills out, and the roadmap's target of a median under 20,000 tokens stays a target. +The release-week row is the first under every 2.1 cost change together, and it is a small sample of light edits: most were docstring and test changes, which the Challenger clears in seconds and the Defender then skips. It is not a settled reduction. On the day of the snapshot that median moved from under 20,000 to just over it as heavier edits landed. The roadmap-week row is the number to plan on until a full week of ordinary edits is in, and the roadmap's target of a median under 20,000 tokens stays a target. -Wall time, recorded per stage, on the `claude_code` provider, which cold-starts a `claude` process per stage. Over the 1,635 timed entries: +Wall time, recorded per stage, on the `claude_code` provider, which cold-starts a `claude` process per stage. Over the 1,673 timed entries in the same snapshot: | Stage | Median seconds | 90th percentile | |---|---|---| -| Challenger | 21.8 | 38.7 | -| Defender | 15.8 | 38.6 | -| Oracle | 23.1 | 34.9 | -| Whole edit | 61.9 | 107.7 | +| Challenger | 21.7 | 38.5 | +| Defender | 15.3 | 38.5 | +| Oracle | 23.1 | 34.7 | +| Whole edit | 61.1 | 106.8 | -By week, the whole-edit median was 63.5 seconds in 2026-W36 and 22.0 seconds in 2026-W37, with the same caveat on the second figure. +By week, the whole-edit median was 63.5 seconds in 2026-W36 and 26.6 seconds in 2026-W37, with the same caveat on the second figure. The Defender median is low because a CLEAR challenge skips it and records zero seconds. The direct API providers avoid the process start but not the sequential three-call shape. Both tables will drift as the ledger grows; run `python -m cli stats` for the current figures on your own chain, and `python -m cli viewer` for the per-week view. diff --git a/cli/commands.py b/cli/commands.py index a54a7e2..946b3ac 100644 --- a/cli/commands.py +++ b/cli/commands.py @@ -68,6 +68,7 @@ normalized_entries, pct, seconds_by_stage, + tokens_by_week, tokens_per_entry, ) from utils.owneronly import open_owner_only @@ -986,6 +987,15 @@ def cmd_stats() -> int: f"p90 {int(billed['p90']):,} " f"(reads {CACHE_READ_RATE:g}x, writes {CACHE_WRITE_RATE:g}x)" ) + # The same distribution per ISO week, so a release week's figure + # can be read off the chain instead of recomputed by hand. + print("Tokens by week :") + for row in tokens_by_week(entries): + print( + f" {row['week']}: median {int(row['median']):,}, " + f"at cached rates {int(row['billed_median']):,} " + f"({int(row['entries'])} with usage)" + ) else: print("Tokens per edit : n/a (no entry carries token usage)") seconds: dict[str, dict[str, float | int]] = seconds_by_stage(entries) diff --git a/tests/test_commands.py b/tests/test_commands.py index 60e8a06..59bbaae 100644 --- a/tests/test_commands.py +++ b/tests/test_commands.py @@ -219,6 +219,34 @@ def test_stats_prints_token_and_timing_distributions(self) -> None: text, ) + def test_stats_prints_tokens_by_week(self) -> None: + entries: list[dict] = _entries() + entries[0]["timestamp"] = "2026-01-05T00:00:00+00:00" + entries[0]["oracle"] = { + **entries[0].get("oracle", {}), + "_tokens": {"input": 500, "output": 50}, + } + entries[1]["timestamp"] = "2026-01-12T00:00:00+00:00" + entries[1]["oracle"] = { + **entries[1].get("oracle", {}), + "_tokens": {"input": 1000, "output": 0, "cache_read": 1000, "cache_creation": 0}, + } + out = io.StringIO() + with patch("cli.commands.load_ledger", return_value=entries): + with patch("cli.commands.verify_chain", return_value=_valid_verify()): + with redirect_stdout(out): + code: int = cmd_stats() + self.assertEqual(code, 0) + text: str = out.getvalue() + self.assertIn("Tokens by week :", text) + self.assertIn( + " 2026-W02: median 550, at cached rates 550 (1 with usage)", text + ) + # 1,000 input entirely read from cache bills at 100. + self.assertIn( + " 2026-W03: median 1,000, at cached rates 100 (1 with usage)", text + ) + def test_stats_prints_models_by_stage_with_override_counts(self) -> None: entries: list[dict] = _entries() entries[0]["oracle"] = {"verdict": "PASS", "_model": "claude-opus-4-8", "_model_override": False} diff --git a/tests/test_stats.py b/tests/test_stats.py index 004e260..5c41b17 100644 --- a/tests/test_stats.py +++ b/tests/test_stats.py @@ -20,8 +20,10 @@ MODEL_NOT_REACHED, MODEL_SKIPPED, MODEL_UNRECORDED, + UNKNOWN_WEEK, citations_by_constraint, compute_ledger_stats, + entries_by_week, entry_has_pipeline_error, models_by_stage, stage_model_label, @@ -34,6 +36,7 @@ stats_by_scope, stats_by_week, tokens_by_stage, + tokens_by_week, tokens_per_entry, billed_input, billed_tokens_per_entry, @@ -674,6 +677,120 @@ def test_empty_ledger(self) -> None: self.assertEqual(summary[stage], {"entries": 0, "median": 0.0, "p90": 0.0}) +class EntriesByWeekTests(unittest.TestCase): + """The one grouping every weekly table is built on.""" + + def test_iso_week_boundaries(self) -> None: + # ISO weeks run Monday to Sunday. Sunday 2026-01-04 closes 2026-W01 + # and Monday 2026-01-05 opens W02. The last days of 2025 belong to + # 2026-W01, and 2026 has 53 ISO weeks, so New Year's Day 2027 falls + # in 2026-W53. + entries: list[dict] = [ + {"timestamp": "2026-01-04T23:59:59+00:00", "id": "sunday"}, + {"timestamp": "2026-01-05T00:00:00+00:00", "id": "monday"}, + {"timestamp": "2025-12-29T12:00:00+00:00", "id": "year-end"}, + {"timestamp": "2027-01-01T00:00:00+00:00", "id": "year-start"}, + ] + groups: dict[str, list[dict]] = entries_by_week(entries) + self.assertEqual( + {week: [e["id"] for e in rows] for week, rows in groups.items()}, + { + "2026-W01": ["sunday", "year-end"], + "2026-W02": ["monday"], + "2026-W53": ["year-start"], + }, + ) + + def test_unparseable_and_missing_timestamps_bucket_under_unknown(self) -> None: + entries: list[dict] = [ + {"timestamp": "not a date", "id": "bad"}, + {"id": "missing"}, + {"timestamp": "2026-01-05T00:00:00+00:00", "id": "good"}, + ] + groups: dict[str, list[dict]] = entries_by_week(entries) + self.assertEqual([e["id"] for e in groups[UNKNOWN_WEEK]], ["bad", "missing"]) + self.assertEqual([e["id"] for e in groups["2026-W02"]], ["good"]) + # Renderers sort the labels; the unknown bucket sorts after any year. + self.assertEqual(sorted(groups)[-1], UNKNOWN_WEEK) + + def test_every_weekly_table_groups_the_same_way(self) -> None: + entries: list[dict] = [ + {"timestamp": "2026-01-05T00:00:00+00:00", "oracle": {"verdict": "PASS", "_seconds": 2.0, "_tokens": {"input": 10, "output": 1}}}, + {"timestamp": "2026-01-11T00:00:00+00:00", "oracle": {"verdict": "VETO"}}, + {"timestamp": "2026-01-12T00:00:00+00:00", "oracle": {"verdict": "PASS", "_seconds": 3.0}}, + {"timestamp": "broken", "oracle": {"verdict": "PASS", "_tokens": {"input": 5, "output": 5}}}, + ] + groups: dict[str, list[dict]] = entries_by_week(entries) + tallies: list[dict] = stats_by_week(entries) + self.assertEqual([row["week"] for row in tallies], sorted(groups)) + for row in tallies: + self.assertEqual(row["total"], len(groups[row["week"]])) + # The other two tables omit weeks with nothing to measure, and every + # week they do show is one of the same groups. + self.assertEqual([r["week"] for r in latency_by_week(entries)], ["2026-W02", "2026-W03"]) + self.assertEqual([r["week"] for r in tokens_by_week(entries)], ["2026-W02", UNKNOWN_WEEK]) + for table in (latency_by_week(entries), tokens_by_week(entries)): + self.assertTrue({r["week"] for r in table} <= set(groups)) + + +class TokensByWeekTests(unittest.TestCase): + """The per-week token distribution the README quotes for a release week.""" + + def test_weeks_without_usage_are_omitted_and_malformed_usage_is_skipped( + self, + ) -> None: + entries: list[dict] = [ + {"timestamp": "2026-01-05T00:00:00+00:00", "oracle": {"_tokens": {"input": 100, "output": 10}}}, + { + "timestamp": "2026-01-06T00:00:00+00:00", + "challenger": {"_tokens": {"input": 200, "output": 20}}, + "oracle": {"_tokens": {"input": 300, "output": 30}}, + }, + {"timestamp": "2026-01-07T00:00:00+00:00", "verdict": "PASS"}, + {"timestamp": "2026-01-12T00:00:00+00:00", "oracle": {"_tokens": {"input": True, "output": "many"}}}, + {"timestamp": "2026-01-19T00:00:00+00:00", "oracle": {"_tokens": {"input": 1000, "output": 0}}}, + {"timestamp": "not a date", "oracle": {"_tokens": {"input": 5, "output": 5}}}, + ] + # W02 totals 110 and 550 (the untracked third entry is not counted); + # W03 holds only malformed usage and is omitted; W04 is 1000; the + # unparseable timestamp lands in the unknown bucket, last. + self.assertEqual( + tokens_by_week(entries), + [ + {"week": "2026-W02", "entries": 2, "median": 330.0, "p90": 550.0, "billed_median": 330.0, "billed_p90": 550.0}, + {"week": "2026-W04", "entries": 1, "median": 1000.0, "p90": 1000.0, "billed_median": 1000.0, "billed_p90": 1000.0}, + {"week": UNKNOWN_WEEK, "entries": 1, "median": 10.0, "p90": 10.0, "billed_median": 10.0, "billed_p90": 10.0}, + ], + ) + + def test_cache_reads_are_priced_in_the_billed_figures(self) -> None: + entries: list[dict] = [ + { + "timestamp": "2026-01-05T00:00:00+00:00", + "oracle": {"_tokens": {"input": 1000, "output": 10, "cache_read": 900, "cache_creation": 0}}, + } + ] + row: dict = tokens_by_week(entries)[0] + # 100 uncached + 900 at the cached rate + 10 output. + self.assertEqual(row["median"], 1010.0) + self.assertEqual(row["billed_median"], 200.0) + + def test_a_single_week_agrees_with_the_all_entries_figures(self) -> None: + entries: list[dict] = [ + {"timestamp": "2026-01-05T00:00:00+00:00", "oracle": {"_tokens": {"input": 100, "output": 10}}}, + {"timestamp": "2026-01-06T00:00:00+00:00", "oracle": {"_tokens": {"input": 400, "output": 40, "cache_read": 200}}}, + {"timestamp": "2026-01-07T00:00:00+00:00", "oracle": {"_tokens": {"input": 900, "output": 90}}}, + ] + row: dict = tokens_by_week(entries)[0] + overall: dict = tokens_per_entry(entries) + billed: dict = billed_tokens_per_entry(entries) + self.assertEqual((row["entries"], row["median"], row["p90"]), (overall["entries"], overall["median"], overall["p90"])) + self.assertEqual((row["billed_median"], row["billed_p90"]), (billed["median"], billed["p90"])) + + def test_empty_ledger(self) -> None: + self.assertEqual(tokens_by_week([]), []) + + class LatencyByWeekTests(unittest.TestCase): def test_weeks_without_a_timing_are_omitted(self) -> None: entries: list[dict] = [ diff --git a/tests/test_viewer.py b/tests/test_viewer.py index 2d097bd..5a1a744 100644 --- a/tests/test_viewer.py +++ b/tests/test_viewer.py @@ -40,6 +40,7 @@ pct, stats_by_scope, stats_by_week, + tokens_by_week, ) from utils.viewer import generate_viewer_html # noqa: E402 @@ -327,6 +328,44 @@ def test_token_table_prices_cache_reads_at_the_cached_rate(self) -> None: html_out, ) + def test_token_table_has_a_row_per_week_with_usage(self) -> None: + """The release-week token median the roadmap's R3 asks the dashboard + to show, restating utils.stats.tokens_by_week.""" + chain: list[dict] = _build_valid_chain(3) + chain[0]["oracle"]["_tokens"] = {"input": 1000, "output": 10} + chain[0]["entry_hash"] = compute_entry_hash(chain[0]) + chain[1]["previous_hash"] = chain[0]["entry_hash"] + chain[1]["timestamp"] = "2026-01-12T00:00:00+00:00" # ISO week 3 + chain[1]["oracle"]["_tokens"] = { + "input": 3000, + "output": 20, + "cache_read": 2000, + "cache_creation": 0, + } + chain[1]["entry_hash"] = compute_entry_hash(chain[1]) + chain[2]["previous_hash"] = chain[1]["entry_hash"] + chain[2]["timestamp"] = "2026-01-19T00:00:00+00:00" # week 4, no usage + chain[2]["entry_hash"] = compute_entry_hash(chain[2]) + html_out: str = self._render(chain) + + rows: list[dict] = tokens_by_week(chain) + self.assertEqual([row["week"] for row in rows], ["2026-W01", "2026-W03"]) + for row in rows: + self.assertIn( + self._row([ + row["week"], + f"{row['entries']:,}", + f"{int(row['median']):,}", + f"{int(row['p90']):,}", + f"{int(row['billed_median']):,}", + ]), + html_out, + ) + # And in plain figures: 1,010 uncached; 3,020 of which 2,000 read + # from cache bills as 1,000 + 200 + 20. + self.assertIn(self._row(["2026-W01", "1", "1,010", "1,010", "1,010"]), html_out) + self.assertIn(self._row(["2026-W03", "1", "3,020", "3,020", "1,220"]), html_out) + def test_latency_table_restates_seconds_by_stage(self) -> None: chain: list[dict] = _build_valid_chain(3) chain[0]["oracle"]["_seconds"] = 10.0 diff --git a/utils/stats.py b/utils/stats.py index df49cff..5fac00a 100644 --- a/utils/stats.py +++ b/utils/stats.py @@ -156,16 +156,31 @@ def _group_tallies(groups: dict[str, list[dict]], key: str) -> list[dict]: return rows -def stats_by_week(entries: list[dict]) -> list[dict]: - """Verdict tallies per ISO week of entry timestamp, oldest first. - - Each row is a _tally_verdicts dict plus "week". Entries whose timestamp - does not parse land in the UNKNOWN_WEEK bucket rather than vanishing. +def entries_by_week(entries: list[dict]) -> dict[str, list[dict]]: + """Entries grouped by the ISO week of their timestamp, in ledger order. + + The one grouping every per-week figure is built on (verdict tallies, + latency, tokens), so no two weekly tables can disagree about which week + an entry belongs to. A week is the ISO week, Monday to Sunday, of the + timestamp's calendar date, so the days around a year end belong to + whichever year's week they fall in. An entry whose timestamp does not + parse lands in the UNKNOWN_WEEK bucket rather than vanishing. Keys are + in first-seen order; every renderer sorts them, which puts the unknown + bucket last. """ groups: dict[str, list[dict]] = {} for entry in entries: groups.setdefault(week_of(entry.get("timestamp")), []).append(entry) - return _group_tallies(groups, "week") + return groups + + +def stats_by_week(entries: list[dict]) -> list[dict]: + """Verdict tallies per ISO week of entry timestamp, oldest first. + + Each row is a _tally_verdicts dict plus "week", over the groups + entries_by_week makes. + """ + return _group_tallies(entries_by_week(entries), "week") def stats_by_scope(entries: list[dict], project_root: str) -> list[dict]: @@ -558,22 +573,50 @@ def seconds_by_stage(entries: list[dict]) -> dict[str, dict[str, float | int]]: def latency_by_week(entries: list[dict]) -> list[dict]: """Per-entry total wall time per ISO week, oldest first. - Each row is a _distribution_summary dict plus "week". Only entries that - recorded at least one stage timing contribute, and weeks with none are - omitted rather than shown as zero, so the table cannot imply a verdict - was instant when it was merely unmeasured. + Each row is a _distribution_summary dict plus "week", over the groups + entries_by_week makes. Only entries that recorded at least one stage + timing contribute, and weeks with none are omitted rather than shown + as zero, so the table cannot imply a verdict was instant when it was + merely unmeasured. """ - groups: dict[str, list[float]] = {} - for entry in entries: - seconds: dict[str, float] = _stage_seconds(entry) - if not seconds: + rows: list[dict] = [] + groups: dict[str, list[dict]] = entries_by_week(entries) + for label in sorted(groups): + totals: list[float] = [] + for entry in groups[label]: + seconds: dict[str, float] = _stage_seconds(entry) + if seconds: + totals.append(sum(seconds.values())) + if not totals: continue - groups.setdefault(week_of(entry.get("timestamp")), []).append( - sum(seconds.values()) - ) + row: dict = _distribution_summary(totals) + row["week"] = label + rows.append(row) + return rows + + +def tokens_by_week(entries: list[dict]) -> list[dict]: + """Per-entry token totals per ISO week, oldest first. + + Each row is tokens_per_entry over that week's entries plus "week", and + "billed_median" and "billed_p90" from billed_tokens_per_entry over the + same entries, so a week's row is exactly what the all-entries lines + would print for that week alone; the groups are the ones + entries_by_week makes. Only entries that recorded usable usage + contribute, and weeks with none are omitted rather than shown as zero, + so the table cannot imply an edit was free when it was merely + unmeasured. This is the row the README quotes for the release week. + """ rows: list[dict] = [] + groups: dict[str, list[dict]] = entries_by_week(entries) for label in sorted(groups): - row: dict = _distribution_summary(groups[label]) + summary: dict[str, float | int] = tokens_per_entry(groups[label]) + if not summary["entries"]: + continue + billed: dict[str, float | int] = billed_tokens_per_entry(groups[label]) + row: dict = dict(summary) + row["billed_median"] = billed["median"] + row["billed_p90"] = billed["p90"] row["week"] = label rows.append(row) return rows diff --git a/utils/viewer.py b/utils/viewer.py index 403d8e4..111a2f1 100644 --- a/utils/viewer.py +++ b/utils/viewer.py @@ -40,6 +40,7 @@ seconds_by_stage, stats_by_week, tokens_by_stage, + tokens_by_week, tokens_per_entry, billed_tokens_per_entry, CACHE_READ_RATE, @@ -483,6 +484,16 @@ def _latency_cells(row: dict) -> list[str]: latency_week_rows: list[list[str]] = [ [str(row.get("week", ""))] + _latency_cells(row) for row in latency_weeks ] + token_week_rows: list[list[str]] = [ + [ + str(row.get("week", "")), + f"{int(row.get('entries', 0)):,}", + f"{int(row.get('median', 0)):,}", + f"{int(row.get('p90', 0)):,}", + f"{int(row.get('billed_median', 0)):,}", + ] + for row in tokens_by_week(entries) + ] week_rows: list[list[str]] = [ [str(row.get("week", ""))] + _verdict_cells(row) for row in weeks @@ -595,7 +606,15 @@ def _scope_cells(rows: list[dict]) -> list[list[str]]: ) + "\n " + _token_distribution_line(per_entry, billed_per_entry) - + "\n\n" + + "\n " + + _table( + ["Week", "Entries", "Median tokens", "p90", "Median at cached rates"], + token_week_rows, "No token figures recorded.", + ) + + '\n

Per week, over entries that carry usage, the same ' + "distribution as the line above for that week alone. A week with no " + "usage recorded is omitted rather than shown as zero.

\n" + "\n" '
\n' "

Seconds by stage

\n " + _table( From 93cea54a531f19a7424aafcbf55068c9170f984c Mon Sep 17 00:00:00 2001 From: Dana Burks Date: Sun, 6 Sep 2026 23:41:35 -0700 Subject: [PATCH 2/3] [bench] tests: assert the weekly subset with assertLessEqual SonarCloud S5906 on #66: a set-subset check written as assertTrue over a comparison reads better, and fails with the two sets shown, as assertLessEqual. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_0122bFBUvdtXT245QFoYDkMf --- tests/test_stats.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/test_stats.py b/tests/test_stats.py index 5c41b17..cf22af5 100644 --- a/tests/test_stats.py +++ b/tests/test_stats.py @@ -730,7 +730,7 @@ def test_every_weekly_table_groups_the_same_way(self) -> None: self.assertEqual([r["week"] for r in latency_by_week(entries)], ["2026-W02", "2026-W03"]) self.assertEqual([r["week"] for r in tokens_by_week(entries)], ["2026-W02", UNKNOWN_WEEK]) for table in (latency_by_week(entries), tokens_by_week(entries)): - self.assertTrue({r["week"] for r in table} <= set(groups)) + self.assertLessEqual({r["week"] for r in table}, set(groups)) class TokensByWeekTests(unittest.TestCase): From 0cb4faaa1324c6b2a8c1cba0f78304f2758a7d02 Mon Sep 17 00:00:00 2001 From: Dana Burks Date: Sun, 6 Sep 2026 23:44:49 -0700 Subject: [PATCH 3/3] [bench] cli: print p90 in the weekly token rows Codex review on #66: the weekly lines dropped the p90 that tokens_by_week computes, so cli stats could not reproduce the README table's 90th-percentile column. Each line now carries median, p90, and the cached-rate median beside the entry count. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_0122bFBUvdtXT245QFoYDkMf --- cli/commands.py | 1 + tests/test_commands.py | 7 +++++-- 2 files changed, 6 insertions(+), 2 deletions(-) diff --git a/cli/commands.py b/cli/commands.py index 946b3ac..edc7286 100644 --- a/cli/commands.py +++ b/cli/commands.py @@ -993,6 +993,7 @@ def cmd_stats() -> int: for row in tokens_by_week(entries): print( f" {row['week']}: median {int(row['median']):,}, " + f"p90 {int(row['p90']):,}, " f"at cached rates {int(row['billed_median']):,} " f"({int(row['entries'])} with usage)" ) diff --git a/tests/test_commands.py b/tests/test_commands.py index 59bbaae..17724f8 100644 --- a/tests/test_commands.py +++ b/tests/test_commands.py @@ -240,11 +240,14 @@ def test_stats_prints_tokens_by_week(self) -> None: text: str = out.getvalue() self.assertIn("Tokens by week :", text) self.assertIn( - " 2026-W02: median 550, at cached rates 550 (1 with usage)", text + " 2026-W02: median 550, p90 550, at cached rates 550 (1 with usage)", + text, ) # 1,000 input entirely read from cache bills at 100. self.assertIn( - " 2026-W03: median 1,000, at cached rates 100 (1 with usage)", text + " 2026-W03: median 1,000, p90 1,000, at cached rates 100 " + "(1 with usage)", + text, ) def test_stats_prints_models_by_stage_with_override_counts(self) -> None: