{
 "site": {
  "title": "Do token-saving skills cut what they claim?",
  "url": "https://skill-evals.brenhq.com"
 },
 "harness": {
  "claude_code": [
   "2.1.278",
   "2.1.280",
   "2.1.283",
   "2.1.284"
  ],
  "codex_cli": [
   "0.153.3",
   "0.156.0",
   "0.157.1",
   "0.159.0"
  ]
 },
 "models": [
  {
   "id": "claude-sonnet-5",
   "label": "Claude Sonnet 5",
   "short": "Sonnet 5",
   "host": "Claude Code 2.1.278",
   "note": "high effort, Claude Code's default"
  },
  {
   "id": "claude-opus-5-5",
   "label": "Claude Opus 5.5",
   "short": "Opus 5.5",
   "host": "Claude Code 2.1.280",
   "note": "medium effort, Claude Code's default",
   "fresh": true
  },
  {
   "id": "claude-sonnet-5-5",
   "label": "Claude Sonnet 5.5",
   "short": "Sonnet 5.5",
   "host": "Claude Code 2.1.284",
   "note": "medium effort, Claude Code's default",
   "fresh": true
  },
  {
   "id": "gpt-5.6-luna",
   "label": "GPT-5.6 Luna",
   "short": "5.6 Luna",
   "host": "Codex CLI 0.153.3",
   "note": "low effort, set by me"
  },
  {
   "id": "gpt-6-luna",
   "label": "GPT-6 Luna",
   "short": "6 Luna",
   "host": "Codex CLI 0.156.0",
   "note": "low effort, set by me",
   "fresh": true
  },
  {
   "id": "gpt-6.1-sol",
   "label": "GPT-6.1 Sol",
   "short": "6.1 Sol",
   "host": "Codex CLI 0.159.0",
   "note": "medium effort, set by me",
   "fresh": true
  }
 ],
 "tasks": [
  {
   "id": "T3",
   "kind": "prose",
   "label": "Explain task",
   "detail": "Explain in prose how the code computes monthly totals and what happens with an invalid date, without changing any file. An answer passes if it names ValueError and month.",
   "short": "Explain"
  },
  {
   "id": "T1",
   "kind": "coding",
   "label": "Feature task",
   "detail": "A small Python expense ledger gets a failing test for a new function. The agent has to make the whole suite pass without touching the tests.",
   "short": "Feature"
  },
  {
   "id": "T2",
   "kind": "coding",
   "label": "Bug fix task",
   "detail": "A failing test exposes a rounding bug in the same ledger. The agent again has to make the whole suite pass without touching the tests.",
   "short": "Bug fix"
  },
  {
   "id": "T4",
   "kind": "coding",
   "label": "Build task",
   "detail": "The agent adds a command-line report to the ledger. It reads a CSV file and prints each category's total for one month, largest first, then the grand total. Four failing tests define the output and the error cases. It is the one task where the number of lines of code can vary a lot.",
   "short": "Build"
  }
 ],
 "skills": [
  {
   "id": "control",
   "label": "No skill",
   "role": "baseline"
  },
  {
   "id": "placebo",
   "label": "my control file",
   "role": "placebo",
   "note": "A 365-word file of generic project guidelines I wrote. It asks for a review pass and edge-case handling, so it is not a neutral placebo."
  },
  {
   "id": "caveman",
   "label": "caveman",
   "stars": 107335,
   "metric": "tokens",
   "claim": "cuts 65% of tokens",
   "claim_source": "its GitHub description",
   "claims": [
    {
     "measure": "visible",
     "task": "T3",
     "n": 65
    }
   ],
   "setting": "The 65% came from a results table in the old README. That table counted output tokens per reply on ten coding prompts, against no prompt at all. The README now reports 50% fewer output tokens against an \"Answer concisely.\" prompt, which I do not test. I tested the skill file only, not the proxy that caveman now ships.",
   "test_note": "That task asks for one written answer, the closest of my tasks to a single reply.",
   "source_url": "https://github.com/JuliusBrussee/caveman",
   "claim_extra": "Its README dropped the 65% on 8 September 2026, before my runs. The description still says it.",
   "repo": "https://github.com/JuliusBrussee/caveman",
   "commit": "880114a",
   "file": "skills/caveman/SKILL.md"
  },
  {
   "id": "ponytail",
   "label": "ponytail",
   "stars": 144278,
   "metric": "code",
   "claim": "~54% less code (up to 94%) \u00b7 ~20% cheaper \u00b7 ~27% faster \u00b7 100% safe",
   "claim_source": "its README",
   "claims": [
    {
     "measure": "lines_added",
     "task": "T4",
     "n": 54
    },
    {
     "measure": "total",
     "task": "T4",
     "n": 22
    },
    {
     "measure": "cost_usd",
     "task": "T4",
     "n": 20
    },
    {
     "measure": "wall_s",
     "task": "T4",
     "n": 27
    }
   ],
   "setting": "It ran 12 feature tickets in Claude Code on Haiku 4.5, four runs each, and counted lines added in the diff, 191 a ticket without the skill. Its token count covers the whole session, input included, about 349,000 a ticket without the skill. It says cost and time can rise on some reasoning models.",
   "test_note": "That task is my biggest coding task. I don't test safety.",
   "source_url": "https://github.com/DietrichGebert/ponytail/blob/b6c04480c03e8db2f035751d7c46289779ec3362/README.md",
   "claim_extra": "Its results table also lists \u221222% tokens.",
   "repo": "https://github.com/DietrichGebert/ponytail",
   "commit": "b6c0448",
   "file": "AGENTS.md"
  },
  {
   "id": "karpathy",
   "label": "karpathy-skills",
   "stars": 214632,
   "metric": "code",
   "claim": "\"Minimum code that solves the problem\" and \"fewer unnecessary changes in diffs\"",
   "claim_source": "its CLAUDE.md",
   "claims": [
    {
     "measure": "lines_added",
     "task": "T4",
     "n": null
    }
   ],
   "setting": "It gives no number and no measurement.",
   "test_note": "",
   "source_url": "https://github.com/multica-ai/andrej-karpathy-skills/blob/8462496b34419f20b32778610571ac723e91f94c/CLAUDE.md",
   "repo": "https://github.com/multica-ai/andrej-karpathy-skills",
   "commit": "8462496",
   "file": "CLAUDE.md"
  }
 ],
 "changelog": [
  {
   "date": "2026-09-22",
   "text": "I wrote the plan before the first run.",
   "pages": [
    "skills"
   ]
  },
  {
   "date": "2026-09-22",
   "text": "After two trial runs, before any counted run, I changed the explain check to the words ValueError and month.",
   "pages": [
    "skills"
   ]
  },
  {
   "date": "2026-09-22",
   "text": "Round 1 ran 24 runs.",
   "pages": [
    "skills"
   ]
  },
  {
   "date": "2026-09-22",
   "text": "I corrected the first write-up, which pooled the tasks and misquoted caveman's claim.",
   "pages": [
    "skills"
   ]
  },
  {
   "date": "2026-09-22",
   "text": "I logged round 2 before running it: 36 runs.",
   "pages": [
    "skills"
   ]
  },
  {
   "date": "2026-09-22",
   "text": "I logged round 3 before running it. It added karpathy-skills and the build task and brought ponytail to three runs per task.",
   "pages": [
    "skills"
   ]
  },
  {
   "date": "2026-09-22",
   "text": "I logged round 4 before running it. It added my control file, which I lengthened twice before its first run.",
   "pages": [
    "skills"
   ]
  },
  {
   "date": "2026-09-22",
   "text": "Rounds 1 to 4 finished with 144 runs, no errors and every run passed its test. 120 are shown here; the other 24 tested cutweight.",
   "pages": [
    "skills"
   ]
  },
  {
   "date": "2026-09-22",
   "text": "I took cutweight, a skill I make, off this page, so the page only measures other skills against their own claims.",
   "pages": [
    "skills"
   ]
  },
  {
   "date": "2026-09-22",
   "text": "I logged round 5 before running it: 18 explain-task runs to recheck the largest effects, 12 shown here. Every run now records a hash of its instruction file, and every hash matched.",
   "pages": [
    "skills"
   ]
  },
  {
   "date": "2026-09-22",
   "text": "Two uncounted trial runs checked the model names of Claude Opus 5.5 and GPT-6 Luna. Then I logged round 6: 12 explain-task runs on those models, with the CLI versions released the same day.",
   "pages": [
    "skills"
   ]
  },
  {
   "date": "2026-09-22",
   "text": "I logged round 7 before running it: 108 runs that give Claude Opus 5.5 and GPT-6 Luna every task and setup. GPT-6 Luna failed 9 of its 15 explain runs by answering without reading the file.",
   "pages": [
    "skills"
   ]
  },
  {
   "date": "2026-09-22",
   "text": "I checked GPT-6 Luna's failed explain runs. Its shell tools worked, and none of four harness changes stopped it answering without reading the file, so the failures stay on the page.",
   "pages": [
    "skills"
   ]
  },
  {
   "date": "2026-09-22",
   "text": "Each claim is now checked with the skill's own measure: visible output tokens for caveman, and lines added, total tokens, cost and time for ponytail. I had compared ponytail's token claim with output tokens.",
   "pages": [
    "skills"
   ]
  },
  {
   "date": "2026-09-28",
   "text": "Round 8, on Claude Code 2.1.283 and Codex CLI 0.157.1 with Codex CLI at medium effort. One explain run per model first checked that each model reads the files, then 2 runs without a skill and 2 with caveman for Claude Fable 5.1, GPT-5.6 Luna, GPT-6 Luna, GPT-5.6 Sol, GPT-6 Sol and GPT-6 Astra. GPT-6 Luna read the file on every run. The earlier rounds stay as they ran, at low effort for Codex CLI. Run dates on this site are in UTC.",
   "pages": [
    "skills"
   ]
  },
  {
   "date": "2026-09-28",
   "text": "Added the Models page at /numbers. It shows every model's input tokens, cache reads and writes, output, hidden thinking, time and output per second side by side, for round 8 and for each task and skill in rounds 1 to 7.",
   "pages": [
    "skills",
    "models"
   ]
  },
  {
   "date": "2026-09-28",
   "text": "Codex CLI token counts now include the hidden first request Codex CLI leaves out of its own count, 9,751 to 12,371 tokens a run, measured for each model and CLI release. Codex CLI cache writes, which Codex CLI always reports as 0, are now the prompt tokens it did not read from cache.",
   "pages": [
    "skills",
    "models"
   ]
  },
  {
   "date": "2026-09-28",
   "text": "Round 9 ran Claude Sonnet 5.5 on its launch day, with Sonnet 5 rerun on the same Claude Code release the same day, on the explain task.",
   "pages": [
    "models"
   ]
  },
  {
   "date": "2026-09-28",
   "text": "Sonnet 5.5 ran every task and skill, so round 9 now shows each skill's own claim on it and a full results table.",
   "pages": [
    "skills"
   ]
  },
  {
   "date": "2026-09-28",
   "text": "Opus 5.5 ran the explain task twice on Claude Code 2.1.284, beside Sonnet 5.5 and Sonnet 5 on the same release. The Models page adds a total tokens chart and link anchors on every section and chart.",
   "pages": [
    "skills",
    "models"
   ]
  },
  {
   "date": "2026-09-28",
   "text": "Added an effort table: the effort Claude Code sent for each Claude model, logged from its requests, and the effort I set for each Codex CLI model.",
   "pages": [
    "models"
   ]
  },
  {
   "date": "2026-09-28",
   "text": "Timed every model on the explain task, two runs each, with the same method for Claude Code and Codex CLI: total time split into getting ready, waiting on the model and running tools, plus time to first token and when the written answer starts. The Models page now opens with a sortable leaderboard.",
   "pages": [
    "models"
   ]
  },
  {
   "date": "2026-09-28",
   "text": "New layout: section links in a bar at the top, each section's title above its intro, and tables at the full page width like the charts.",
   "pages": [
    "skills"
   ]
  },
  {
   "date": "2026-09-28",
   "text": "The site now has two pages. Skills tests each skill against the same task with no skill, and Models shows every model with no skill. The per-skill token charts moved to Skills and the effort table to Models, and old links forward to the new place.",
   "pages": [
    "skills",
    "models"
   ]
  },
  {
   "date": "2026-09-28",
   "text": "The Sonnet 5.5 chart on the Models page now names the effort each model ran at: medium for Sonnet 5.5 and Opus 5.5, high for Sonnet 5, each Claude Code's default.",
   "pages": [
    "skills",
    "models"
   ]
  },
  {
   "date": "2026-09-28",
   "text": "The Models page now opens with the leaderboard under a plain title and one sentence on the task. The table keeps time to finish, tokens used and share from cache, and fits a phone; first token and answer start are in the speed charts below it.",
   "pages": [
    "skills",
    "models"
   ]
  },
  {
   "date": "2026-09-29",
   "text": "GPT-6.1 Sol came out during OpenAI's DevDay, and I ran it that day at medium effort on Codex CLI 0.159.0, the first release that runs it on a ChatGPT login: 2 explain runs with no skill, 2 with caveman and 2 timed runs. GPT-6 Sol reran on the same release and Opus 5.5 on Claude Code 2.1.284 the same day. All three are on the leaderboard and the overall chart with these runs.",
   "pages": [
    "skills",
    "models"
   ]
  },
  {
   "date": "2026-09-29",
   "text": "Every run on the Models page now shows what its tokens cost at the model's published API prices, cached tokens at the cache rate, with the price table and its sources. For Claude it matches Claude Code's own estimate on all 194 Claude runs.",
   "pages": [
    "models"
   ]
  },
  {
   "date": "2026-09-29",
   "text": "GPT-6.1 Sol joins the skill runs: every task with every skill, three runs each, at medium effort on Codex CLI 0.159.0. It is in every chart and table on the Skills page, and the claim counts, ranges and grid counts now include it.",
   "pages": [
    "skills"
   ]
  },
  {
   "date": "2026-09-29",
   "text": "The Models page now says what the test is, under the leaderboard: the explain task, how it passes, and that it writes no code. The coding tasks are on the Skills page.",
   "pages": [
    "skills",
    "models"
   ]
  },
  {
   "date": "2026-09-29",
   "text": "For the coding tasks, a run's tests are the only quality measure. Extra tests with new inputs, written after the runs, reran every coding run whose changes I kept, and all passed. How a run works says so.",
   "pages": [
    "skills"
   ]
  },
  {
   "date": "2026-09-29",
   "text": "Added real app work to the Models page: a bug hunt with 20 planted bugs and a tickets test with 17 hidden tests, in a small TypeScript web app, 3 runs each for Opus 5.5, Sonnet 5.5, GPT-5.6 Sol, GPT-6 Sol and GPT-6.1 Sol, graded by hidden tests, with cost at API prices and time.",
   "pages": [
    "models"
   ]
  },
  {
   "date": "2026-09-29",
   "text": "Correction: caveman's README dropped its 65% figure on 8 September 2026, before my runs. Its GitHub description still says it cuts 65% of tokens, so that is the claim I check, and its card now says the README no longer does. karpathy-skills' card links its CLAUDE.md.",
   "pages": [
    "skills"
   ]
  },
  {
   "date": "2026-09-29",
   "text": "Correction: the effort table left out two settings the runs used, Claude Fable 5.1 on Claude Code 2.1.284 in the timed runs and GPT-5.6 Sol on Codex CLI 0.159.0 in the real app work. Both are listed now, and QA fails if a run's release is ever missing again.",
   "pages": [
    "models"
   ]
  },
  {
   "date": "2026-09-29",
   "text": "Redesigned both pages. The Skills page opens with the count of claimed cuts that held, one chart of each claim against the measured cut, and each skill's claim, finding and one-word verdict. The Models page leads with real app work. Text runs the full page width, long detail sits in folds, charts save as images with CSV, and bar charts show bars only.",
   "pages": [
    "skills",
    "models"
   ]
  },
  {
   "date": "2026-09-30",
   "text": "Correction: some Claude Code runs started with thousands fewer tokens of instructions and tools. They ran before Claude Code had synced the account's claude.ai plugins and skills and loaded its connectors, so they cost less and count fewer total tokens. A number where they make up at least half the runs now carries a star, a new limit explains how I found them, and I removed a remark about the connectors from the published answers. No claim check changes. The count of Claude runs in the 29 September cost entry is corrected from 254 to 194; it had counted Sonnet 5.5's runs twice.",
   "pages": [
    "skills",
    "models"
   ]
  },
  {
   "date": "2026-09-30",
   "text": "One GPT-5.6 Sol tickets run edited an existing test file, which the prompt said not to do. The grader restores the original tests first, so its 17 of 17 stood. Round 13 replaced that run, and no rerun edited a test.",
   "pages": [
    "models"
   ]
  },
  {
   "date": "2026-09-30",
   "text": "Round 13 reran the explain task for every model on the Models page, every Claude model's explain runs, all 30 real app work runs and the timed runs, all with the account extras off. Those numbers no longer carry stars, and Claude's explain-task token counts and costs dropped. caveman now cuts Sonnet 5.5's explanations 28%, where round 9 showed 13%, and Sonnet 5's 72%, still above its 65% claim. All five models still passed all 17 hidden ticket tests, and Opus 5.5 cost more, $0.96 a ticket run.",
   "pages": [
    "skills",
    "models"
   ]
  },
  {
   "date": "2026-09-30",
   "text": "The Skills page now prices GPT runs at OpenAI's published rates, so the GPT models appear in every cost comparison, and the headline counts 29 checks.",
   "pages": [
    "skills"
   ]
  },
  {
   "date": "2026-09-30",
   "text": "Redesigned the findings, tables and colors. Each finding has its own framed chart with horizontal bars and a saved image, the skill cards and real-work tables line up and sort like the Models page table, each vendor keeps one color family, and dark mode is warm.",
   "pages": [
    "skills",
    "models"
   ]
  },
  {
   "date": "2026-09-30",
   "text": "Correction: the Skills page gave Sonnet 5.5's release day as 30 September, when it came out on 28 September. Two table notes still said cost covered Claude only, and the effort table left out the Codex CLI 0.159.0 row for three GPT models. All three are fixed, and QA now reads every run group.",
   "pages": [
    "skills",
    "models"
   ]
  },
  {
   "date": "2026-09-30",
   "text": "Correction: the recheck table labeled Sonnet 5's round 13 runs as rounds 1 to 4, and some chart panels named an older Claude Code release than their runs used. The table now shows all three sessions of caveman on Sonnet 5, at 66%, 62% and 72%, and every release and round label comes from the runs. The run list adds the runs round 13 replaced, marked, with each run's CLI and account extras.",
   "pages": [
    "skills"
   ]
  },
  {
   "date": "2026-09-30",
   "text": "The chart for any measure and task now uses the same framed horizontal bars as the findings, with a saved image. The Models page notes how much round 13 moved its numbers and has its own list of changes, and on a phone its table stays a table.",
   "pages": [
    "skills",
    "models"
   ]
  },
  {
   "date": "2026-09-30",
   "text": "Correction: the page gave three sets of caveman runs on Claude Sonnet 5's explain task and left out round 9's, which cut 64%. All four are shown now, 66%, 62%, 64% and 72%, and two fell under the 65% claim. The claim still counts as held, because the charts' own runs reached it.",
   "pages": [
    "skills"
   ]
  },
  {
   "date": "2026-09-30",
   "text": "The chart for any measure and task is now its own section, Every measure and task. Charts with three skills stack at full width with a labeled scale, bars from failed runs show hollow, and the Models page notes every model the round 13 timing slowed.",
   "pages": [
    "skills",
    "models"
   ]
  },
  {
   "date": "2026-09-30",
   "text": "Correction: 12 Claude coding runs also started light. Until now I had checked only the explain task. They now carry a star in the run list, and no median has half its runs light. The Models table title gives the time spread with and without the slow OpenAI day, and the cost panels say their range adds up each task's cheapest to costliest run.",
   "pages": [
    "skills",
    "models"
   ]
  },
  {
   "date": "2026-09-30",
   "text": "On a phone, bar charts put each model's name above its bar. Saved images keep their panels side by side, the Models token charts are horizontal bars, and the Skills page adds a table of every model's cut under the top chart.",
   "pages": [
    "skills",
    "models"
   ]
  },
  {
   "date": "2026-09-30",
   "pages": [
    "skills"
   ],
   "text": "The set-by-set table now shows every model with more than one clean set of caveman runs. In round 13, with the account extras off, caveman cut more on every Claude model than in any earlier set."
  },
  {
   "date": "2026-09-30",
   "pages": [
    "models"
   ],
   "text": "The GPT models that ran slower in round 13's timing carry a dagger in the explain table and the time chart, and the real app work note says their runs slowed the same day. Costs use one format on every table and chart."
  },
  {
   "date": "2026-09-30",
   "pages": [
    "skills",
    "models"
   ],
   "text": "Bar charts with the name above each bar leave a clear gap between rows, and their lines no longer cross a label. Grid cells that could be chance are outlined like the bars. The cut table names every model and marks the one cut that met its claim."
  },
  {
   "date": "2026-09-30",
   "pages": [
    "skills"
   ],
   "text": "The set-by-set table adds GPT-6 Sol, from rounds 8 and 10, and stars a median that comes from a light run. The Every measure and task title counts only clear cuts, the filled bars."
  },
  {
   "date": "2026-09-30",
   "pages": [
    "skills",
    "models"
   ],
   "text": "Costs use one format everywhere, three significant figures under ten cents and three decimals above. The cost-vs-time title and dots mark the slowed GPT models, and the run list adds round 10."
  },
  {
   "date": "2026-09-30",
   "pages": [
    "skills",
    "models"
   ],
   "text": "The cut table fits tablets and phones, where each claim becomes a block of model and cut pairs. Saved images explain their daggers, small cost bars get their own scale, and tables and charts in folds share one title style."
  },
  {
   "date": "2026-09-30",
   "pages": [
    "skills",
    "models"
   ],
   "text": "Round 8's table now stars its light medians. The real-work title and note use the table's cost figures, and the cost-vs-time title compares cost only. The page says a claim held, in one word, everywhere."
  },
  {
   "date": "2026-09-30",
   "pages": [
    "skills",
    "models"
   ],
   "text": "Output and thinking counts carry the light-run star too. The slowed GPT models are marked \u00a7 instead of a dagger. The cut table says whether a cell was not compared or failed, and dates and model names no longer break across lines on phones."
  },
  {
   "date": "2026-09-30",
   "pages": [
    "skills"
   ],
   "text": "The page reports each claim from the latest round. caveman on Claude Sonnet 5 cut 72% in round 13, and its earlier sets, at 62% to 66%, stay in the set-by-set table."
  },
  {
   "date": "2026-09-30",
   "pages": [
    "skills"
   ],
   "text": "The claimed-against-measured chart uses bars like the other charts: each claim's average cut against a light track that ends at the claimed number, with each model's cut in the table under it."
  },
  {
   "date": "2026-09-30",
   "pages": [
    "skills",
    "models"
   ],
   "text": "The tick strip and the byline are gone: the chart under the intro shows the same checks, and its source line carries the date."
  },
  {
   "date": "2026-09-30",
   "pages": [
    "models"
   ],
   "text": "The real app work chart and the Models board views drop the rings and the scores on each dot, and the rerun note is two sentences."
  },
  {
   "date": "2026-09-30",
   "pages": [
    "models"
   ],
   "text": "The real app work chart says what its tests can't do: every model scored at or near full marks, so they show cost and time on a small job, not which model is best."
  },
  {
   "date": "2026-09-30",
   "pages": [
    "models"
   ],
   "text": "A dot on the real app work chart names its score when it fell short of full marks, as Claude Sonnet 5.5's bug hunt dot does at 19 of 20."
  },
  {
   "date": "2026-10-01",
   "pages": [
    "models"
   ],
   "text": "New page, GPT-6.1 Sol timed three days running: one short Python file explained on launch day, the day after, and on 1 October after OpenAI said it added capacity. The Models board keeps the 30 September runs."
  },
  {
   "date": "2026-10-01",
   "pages": [
    "models"
   ],
   "text": "The GPT-6.1 Sol page says what the test is up front and groups the chart by day. The ChatGPT apps and plugins setting moved to the method, in plain words."
  },
  {
   "date": "2026-10-01",
   "pages": [
    "models"
   ],
   "text": "The GPT-6.1 Sol page adds four runs from the afternoon of 1 October, after @sama said it should be much better: 13.5 to 16.8 s, faster than that morning and still slower than launch day."
  },
  {
   "date": "2026-10-02",
   "pages": [
    "skills"
   ],
   "text": "New page, Karpathy's ASD-STE100 tip, measured: six models with his tip as a one-line instruction on the explain task, against no instruction in the same setting."
  },
  {
   "date": "2026-10-02",
   "pages": [
    "skills"
   ],
   "text": "The ASD-STE100 chart uses one bar color for every model; a change the runs don't clearly show now says \"could be chance\" in its label instead of a pale outline."
  },
  {
   "date": "2026-10-02",
   "pages": [
    "skills"
   ],
   "text": "The ASD-STE100 chart title counts every model whose answers got shorter, then how many of those clearly."
  },
  {
   "date": "2026-10-03",
   "pages": [
    "models"
   ],
   "text": "The GPT-6.1 Sol page adds four runs from the afternoon of 3 October, after a 2 October post on X still called it slow: 15.6 to 26.6 s, slower by median than the afternoon of 1 October. Corrected 5 October: the two afternoons' runs overlap."
  },
  {
   "date": "2026-10-03",
   "pages": [
    "skills"
   ],
   "text": "The ASD-STE100 page adds a chart of ASD-STE100 beside caveman, ponytail, karpathy-skills and my control file on the same explain task, each against its own no-instruction runs."
  },
  {
   "date": "2026-10-03",
   "pages": [
    "skills"
   ],
   "text": "The ASD-STE100 chart's longest labels no longer run off its right edge on tablet and small laptop screens."
  },
  {
   "date": "2026-10-04",
   "pages": [
    "models"
   ],
   "text": "The GPT-6.1 Sol page adds four runs from the afternoon of 4 October: 13.2 to 22.9 s, all slower than launch day."
  },
  {
   "date": "2026-10-05",
   "pages": [
    "models"
   ],
   "text": "The GPT-6.1 Sol page adds four runs from 5 October, before OpenAI's speed post: 14.1 to 18.6 s, all slower than launch day."
  },
  {
   "date": "2026-10-05",
   "pages": [
    "models"
   ],
   "text": "The GPT-6.1 Sol page adds four runs from 5 October, minutes after OpenAI said its speed change would be felt within two hours: 14.3 to 19.5 s."
  },
  {
   "date": "2026-10-05",
   "pages": [
    "models"
   ],
   "text": "Readers were right that the GPT-6.1 Sol page's tokens a second was not TPS. It divides output by the time spent waiting on the model's replies, so the column is now tokens a second of waiting."
  },
  {
   "date": "2026-10-05",
   "pages": [
    "skills",
    "models"
   ],
   "text": "Headlines say a setup cut more, won or held its claim only when its runs sit apart from the other side's. Where runs overlap they now say by median, so three headlines changed, and the all-tasks totals add up each side's full run range."
  },
  {
   "date": "2026-10-05",
   "text": "The GPT-6.1 Sol page adds four runs from a 5 October retest, still inside the two hours OpenAI gave: 11.3 to 20.8 s, three of them inside launch day's range.",
   "pages": [
    "models"
   ]
  },
  {
   "date": "2026-10-05",
   "text": "A TPS column briefly showed a rate from an OpenAI server-side timing field. That field times the engine, not delivery: on launch day it read 3.7 to 3.9 times the rate text reached my machine. It came down. A TPS proxy column now counts the pieces of answer text Codex logged a second, pauses included.",
   "pages": [
    "models"
   ]
  },
  {
   "date": "2026-10-05",
   "text": "Two test runs of the updated timing script on 5 October are listed under How I ran it on the GPT-6.1 Sol page, not charted.",
   "pages": [
    "models"
   ]
  },
  {
   "date": "2026-10-05",
   "text": "The GPT-6.1 Sol and ASD-STE100 pages said each run got a fresh container. Each run gets a fresh copy of the project; runs share the bench's container. The Sol page now says a run passes when its answer contains both \"month\" and \"ValueError\".",
   "pages": [
    "models",
    "skills"
   ]
  },
  {
   "date": "2026-10-05",
   "text": "The GPT-6.1 Sol page's intro said posts on X on 3 October called it slow again. No such post could be found, so it now cites the 2 and 4 October posts that did.",
   "pages": [
    "models"
   ]
  },
  {
   "date": "2026-10-05",
   "text": "The GPT-6.1 Sol page adds four runs from 5 October, after the two hours OpenAI gave: 11.4 to 17.7 s, streaming at 39.2 to 52.9 text pieces a second.",
   "pages": [
    "models"
   ]
  },
  {
   "date": "2026-10-05",
   "text": "Readers asked how launch day was as quick with a slower stream. The GPT-6.1 Sol page now says why: the answer's stream is under half of every run, launch day's answers were shorter, and the runs since that made two model requests, like launch day, finished inside launch day's range.",
   "pages": [
    "models"
   ]
  },
  {
   "date": "2026-10-05",
   "text": "Readers asked to sort the GPT-6.1 Sol run table. Click any column heading: Time puts the fastest first, TPS proxy the highest first, a second click reverses, and Day restores day order.",
   "pages": [
    "models"
   ]
  },
  {
   "date": "2026-10-05",
   "text": "The GPT-6.1 Sol chart took more than a page. It is now one row per sitting in two panels, TPS proxy and seconds, each a median with launch day and the sittings since the 5 October retest picked out; the 5 October sittings are named by OpenAI's post, not the time of day; every run stays in the table. The intro is one line, and the background folds under How I ran it.",
   "pages": [
    "models"
   ]
  },
  {
   "date": "2026-10-05",
   "text": "Readers asked whether the GPT-6.1 Sol page has true TPS. The table now has TPS by OpenAI's own token count for the eight runs since 5 October's retest: the answer's output tokens over the seconds from its first text event to its last, a rate observed at my machine. Earlier rows have no count and stay blank; two script test runs also logged it, at 36.2 and 43.4, and are in the CSV. The text-event rate is now labeled text pieces a second.",
   "pages": [
    "models"
   ]
  },
  {
   "date": "2026-10-05",
   "text": "Readers couldn't find the GPT-6.1 Sol speed page from the main pages. It now has its own tab in the bar on every page (Sol on a phone), and the Models page's time section links it.",
   "pages": [
    "skills",
    "models"
   ]
  },
  {
   "date": "2026-10-05",
   "text": "The Skills page's shorter-answers section now links the ASD-STE100 page, which had no link from the main pages.",
   "pages": [
    "skills"
   ]
  },
  {
   "date": "2026-10-06",
   "text": "The GPT-6.1 Sol page adds four runs from 6 October: 10.3 to 15.7 s, the fastest run yet at 10.3 s, and 43.4 to 51.2 tokens a second by OpenAI's count. Every sitting from 5 October's retest on now counts as since the retest.",
   "pages": [
    "models"
   ]
  },
  {
   "date": "2026-10-06",
   "text": "A reader asked for Claude Opus 5.5 and Claude Sonnet 5.5 beside the GPT-6.1 Sol test. The Sol page now shows them timed the same day and task, 4 runs each, start to finish in each model's own tool: Sonnet 5.5 7.8 to 9.8 s, Sol 10.3 to 15.7 s, Opus 5.5 15.2 to 17.7 s. No TPS for the Claude models: this bench's Claude Code capture does not give a streaming rate yet.",
   "pages": [
    "models"
   ]
  },
  {
   "date": "2026-10-06",
   "text": "Readers loved the speed race videos on X, so the GPT-6.1 Sol page now has the same race, drawn live from every run: one lane per sitting, all leaving together at 5x with real seconds on the clock. It plays once when it scrolls into view and has a Replay button; with reduced motion it waits for Play.",
   "pages": [
    "models"
   ]
  },
  {
   "date": "2026-10-06",
   "text": "Every section of the GPT-6.1 Sol page has a heading with a # link that copies its address, and an On this page row jumps between them: every sitting, the race, same day with other models, and how I ran it.",
   "pages": [
    "models"
   ]
  },
  {
   "date": "2026-10-06",
   "text": "Readers said the race's dots were softer than the videos'. They are now drawn at full screen density with a sharp edge, a glow sized to the dot and the videos' white highlight.",
   "pages": [
    "models"
   ]
  },
  {
   "date": "2026-10-06",
   "text": "The site now counts visits with Cloudflare Web Analytics, which sets no cookies. The footer, which said no analytics, now says so.",
   "pages": [
    "skills",
    "models"
   ]
  },
  {
   "date": "2026-10-06",
   "text": "Each page now ships its text in the HTML, so search engines and AI tools that do not run scripts can read it. The site also has a sitemap, an llms.txt and a footer line linking every page.",
   "pages": [
    "skills",
    "models"
   ]
  },
  {
   "date": "2026-10-06",
   "text": "Readers said the link cards on X did not show the site well. Each card now has the site's bar with its page marked, one finding and one chart, sized to read in a feed; the GPT-6.1 Sol card shows the streaming chart alone.",
   "pages": [
    "skills",
    "models"
   ]
  },
  {
   "date": "2026-10-06",
   "text": "The GPT-6.1 Sol page's method now says the group since 5 October's retest was set after seeing the 5 October runs, and every later sitting joins it whatever it shows. The link cards have larger chart text.",
   "pages": [
    "models"
   ]
  }
 ],
 "criteria": "I tested every skill with more than 100,000 GitHub stars that claims to cut tokens or code and ships as an instruction file an agent reads. I read the star counts from GitHub on 22 September 2026.",
 "excluded": [
  {
   "name": "superpowers",
   "stars": 290109,
   "why": "a development method with no token or code savings claim"
  },
  {
   "name": "ECC",
   "stars": 265217,
   "why": "a full agent setup with its own hooks and memory, not one instruction file"
  },
  {
   "name": "gstack",
   "stars": 133921,
   "why": "a set of role tools with no savings claim"
  },
  {
   "name": "graphify",
   "stars": 120473,
   "why": "a codebase knowledge-graph tool, not an instruction file"
  }
 ],
 "limitations": [
  "Each skill ran three times per task and model. That shows which way a number moved, but three runs are too few to prove it.",
  "The four tasks are small and share one Python project. Only the build task is big enough to show a difference in lines of code, and lines include blanks and comments.",
  "The charts cover six models, one setting each. Claude Sonnet 5, Claude Opus 5.5 and Claude Sonnet 5.5 ran at Claude Code's default effort, which is high for Sonnet 5 and medium for the other two. GPT-5.6 Luna and GPT-6 Luna ran at low effort, and GPT-6.1 Sol at medium on a newer Codex CLI release. Outside the charts, round 8 ran Claude Fable 5.1, GPT-5.6 Sol, GPT-6 Sol and GPT-6 Astra on the explain task and reran both Lunas at medium effort.",
  "Visible output leaves out hidden thinking, a median 51% of output across Sonnet 5's build-task runs. Total tokens and cost still count it.",
  "I ran on subscriptions, with no API keys. For Claude, cost is Claude Code's own estimate from each model's published rates per million tokens, and it includes Claude Code's own system prompt and tools. Claude Sonnet 5 and Claude Sonnet 5.5 each list $2 for input, $4 for one-hour cache writes, $0.20 for cache reads and $10 for output. Claude Opus 5.5 lists $4, $8, $0.20 and $20. Codex CLI reports no cost. A GPT run's cost is its tokens at OpenAI's published rates, cached input at the cache rate, with Codex CLI's hidden first request added.",
  "Time moves with the service's latency. Small time differences mean little.",
  "An explain answer passes if it names ValueError and month. That shows it is on topic, not that it is complete.",
  "My control file is a short set of generic project guidelines I wrote after the first two rounds. It asks for a review pass and is not a neutral placebo. It stays out of the charts and appears in the full tables.",
  "Rounds 1 to 4 did not block runs from editing the tests or keep each run's diff. Round 5 and every round from 6 on did both.",
  "Skills added in later rounds ran after the no-skill runs I compare them with.",
  "I logged each round's plan before running it. The plan itself is private.",
  "My scripts and earlier run records were readable inside the container, except in rounds 6 and 7. No Codex CLI run log shows an agent reading them, and the Claude runs can't be checked.",
  "Whether an explain answer covered the impossible-date case comes from a text search of each answer, which I checked by reading them.",
  "Until round 13, each CLI also loaded the account extras a subscription login brings by default. For Claude Code these are the plugins and skills enabled on the account's claude.ai, synced after a session's first run, and the account's claude.ai connectors. For Codex CLI they are its connected apps and plugins, built-in skills included. Round 13 turned the account extras off: Claude Code with no MCP servers and no claude.ai sync, Codex CLI with its apps and plugins off. No Codex CLI run log shows any of them used. Claude Code's output does not list tool calls, so its runs can't be checked.",
  "GPT-6 Luna's tools worked. In checks outside the benchmark it ran shell commands whenever asked to. It still often answered the explain task without opening the file. Turning off Codex CLI's apps and code-mode host, raising effort to medium, and adding a note that it can run commands did not stop that. I report those runs as failures rather than change its task. In round 8, on a newer Codex CLI release, GPT-6 Luna read the file on every explain run. The failures happened only on the older release.",
  "Codex CLI leaves its hidden first request out of its own token count. I measured it once for each model and CLI release, not in each run, and add it to every Codex CLI run. Its size did not change with the skill file or the task prompt.",
  "Sonnet 5.5's runs went in three parallel streams on one Claude Code release, which can shift its times slightly. The launch-day comparison on the Models page uses its first two explain runs with no skill.",
  "Some Claude Code runs started with thousands fewer tokens of instructions and tools than the same model's other runs. I call them light runs. On the explain task every run takes two or three steps. There I mark a run light when its last request is more than 4,000 tokens smaller than the largest from the same model with the same number of steps. Each light run was a session's first, before Claude Code loaded the account extras. Of the 18 light explain runs across all rounds, 7 ended by saying the account's connectors needed authorization, and no other run said so. I removed that remark from the published answers, but their token counts still include it. Round 13 reran every Claude model's explain runs, the Models page's explain runs and the real app work with the account extras off, and none of those is light. The explain runs still light all ran before round 13. They appear in round 8's table, the Models page's launch-day runs, the set-by-set table's rows from before round 13, and the run list. The coding tasks take many more steps. There I mark a run light when it wrote over 4,000 fewer tokens to the cache than the top run with the same model, task and skill. That marks 12 Claude coding runs, and no median has half its runs light. A number where at least half the runs behind it are light carries a star. The coding tasks had the account extras on both with and without a skill, which keeps the comparison within a model fair."
 ],
 "limit_facts": [
  {
   "n": "51",
   "metric": "thinking_share_all",
   "model": "claude-sonnet-5",
   "task": "T4"
  },
  {
   "n": "2",
   "metric": "rate",
   "model": "claude-sonnet-5",
   "part": "input"
  },
  {
   "n": "4",
   "metric": "rate",
   "model": "claude-sonnet-5",
   "part": "cache_write"
  },
  {
   "n": "0.20",
   "metric": "rate",
   "model": "claude-sonnet-5",
   "part": "cache_read"
  },
  {
   "n": "10",
   "metric": "rate",
   "model": "claude-sonnet-5",
   "part": "output"
  },
  {
   "n": "4",
   "metric": "rate",
   "model": "claude-opus-5-5",
   "part": "input"
  },
  {
   "n": "8",
   "metric": "rate",
   "model": "claude-opus-5-5",
   "part": "cache_write"
  },
  {
   "n": "0.20",
   "metric": "rate",
   "model": "claude-opus-5-5",
   "part": "cache_read"
  },
  {
   "n": "20",
   "metric": "rate",
   "model": "claude-opus-5-5",
   "part": "output"
  },
  {
   "n": "2",
   "metric": "rate",
   "model": "claude-sonnet-5-5",
   "part": "input"
  },
  {
   "n": "4",
   "metric": "rate",
   "model": "claude-sonnet-5-5",
   "part": "cache_write"
  },
  {
   "n": "0.20",
   "metric": "rate",
   "model": "claude-sonnet-5-5",
   "part": "cache_read"
  },
  {
   "n": "10",
   "metric": "rate",
   "model": "claude-sonnet-5-5",
   "part": "output"
  },
  {
   "n": "4,000",
   "metric": "light_gap"
  },
  {
   "n": "18",
   "metric": "light_all"
  },
  {
   "n": "7",
   "metric": "light_remark"
  },
  {
   "n": "6",
   "metric": "round_min",
   "model": "gpt-6-luna"
  },
  {
   "n": "7",
   "metric": "round_max",
   "model": "gpt-6-luna"
  },
  {
   "n": "13",
   "metric": "round_max",
   "model": "claude-sonnet-5-5"
  },
  {
   "n": "5",
   "metric": "round",
   "group": "recheck_runs",
   "model": "claude-sonnet-5"
  },
  {
   "n": "12",
   "metric": "light_coding"
  },
  {
   "n": "1",
   "metric": "round_min",
   "model": "gpt-5.6-luna"
  },
  {
   "n": "8",
   "metric": "round",
   "group": "round8_runs",
   "model": "claude-fable-5-1"
  }
 ],
 "copy": {
  "answer": {
   "text": "The one claim that held was caveman's shorter answers on Claude Sonnet 5. In round 13, the latest, it cut 72% against a claimed 65%, and every one of its answers skipped the impossible-date case, a date like 2024-02-30.",
   "facts": [
    {
     "n": "13",
     "metric": "session_round_max",
     "group": "runs",
     "model": "claude-sonnet-5"
    },
    {
     "n": "72",
     "metric": "session_visible",
     "group": "runs",
     "model": "claude-sonnet-5",
     "skill": "caveman"
    },
    {
     "n": "65",
     "metric": "claim"
    }
   ],
   "checks": [
    {
     "claim": "caveman's shorter answers on Claude Sonnet 5",
     "metric": "claim_met",
     "skill": "caveman",
     "field": "visible",
     "task": "T3",
     "n": 65,
     "want": "claude-sonnet-5"
    },
    {
     "claim": "every one of its answers skipped the impossible-date case",
     "metric": "answer_scan",
     "model": "claude-sonnet-5",
     "skill": "caveman",
     "group": "runs",
     "want": "3 of 3 skipped, 0 of 3 no-skill skipped"
    }
   ]
  },
  "explain": {
   "text": "In round 13, every caveman run on Claude Sonnet 5 came in under the shortest no-skill run. All three of its answers skipped the impossible-date case, which all three no-skill answers covered. A date like 2024-02-30 has the right format, but the code fails when it builds the date.",
   "facts": [
    {
     "n": "13",
     "metric": "session_round_max",
     "group": "runs",
     "model": "claude-sonnet-5"
    }
   ],
   "checks": [
    {
     "claim": "every caveman run on Claude Sonnet 5 came in under the shortest no-skill run",
     "metric": "sessions_side",
     "model": "claude-sonnet-5",
     "skill": "caveman",
     "group": "runs",
     "want": "below",
     "sessions": 1
    },
    {
     "claim": "All three of its answers skipped the impossible-date case, which all three no-skill answers covered",
     "metric": "answer_scan",
     "model": "claude-sonnet-5",
     "skill": "caveman",
     "group": "runs",
     "want": "3 of 3 skipped, 0 of 3 no-skill skipped"
    }
   ]
  },
  "fresh": {
   "text": "caveman made Claude Opus 5.5's explanations 26% shorter, and all 15 of Opus 5.5's explain answers, with and without a skill, still covered the impossible-date case. It cut GPT-6.1 Sol's explanations 36% at medium effort, every run below every no-skill run. GPT-6 Luna failed 9 of 15 explain runs because the model answered without reading the file, often saying it had no file tools. Its panel shows those failures and says nothing about the skills.",
   "facts": [
    {
     "n": "26",
     "metric": "visible",
     "model": "claude-opus-5-5",
     "task": "T3",
     "skill": "caveman"
    },
    {
     "n": "15",
     "metric": "task_runs",
     "model": "claude-opus-5-5",
     "task": "T3"
    },
    {
     "n": "9",
     "metric": "failed_runs",
     "model": "gpt-6-luna",
     "task": "T3"
    },
    {
     "n": "15",
     "metric": "task_runs",
     "model": "gpt-6-luna",
     "task": "T3"
    },
    {
     "n": "36",
     "metric": "visible",
     "model": "gpt-6.1-sol",
     "task": "T3",
     "skill": "caveman"
    }
   ],
   "checks": [
    {
     "claim": "all 15 of Opus 5.5's explain answers, with and without a skill, still covered the impossible-date case",
     "metric": "model_answer_scan",
     "model": "claude-opus-5-5",
     "want": "0 skipped"
    },
    {
     "claim": "the model answered without reading the file",
     "metric": "failed_without_reading",
     "model": "gpt-6-luna",
     "task": "T3",
     "want": "no commands"
    },
    {
     "claim": "every run below every no-skill run",
     "metric": "split",
     "field": "visible",
     "want": "below",
     "cells": [
      [
       "gpt-6.1-sol",
       "T3",
       "caveman"
      ]
     ]
    }
   ]
  },
  "recheck": {
   "text": "Each row is one set of runs, with its rounds, release, effort and runs a side under the model's name. For the Claude models the charts use round 13, run with the account extras off. In round 13 caveman cut more on every Claude model than in any earlier set. It made no clear difference on GPT-5.6 Luna in any of its three sets.",
   "facts": [
    {
     "n": "13",
     "metric": "session_round_max",
     "group": "runs",
     "model": "claude-sonnet-5"
    }
   ],
   "checks": [
    {
     "claim": "In round 13 caveman cut more on every Claude model than in any earlier set",
     "metric": "extras_raised",
     "skill": "caveman",
     "want": "every"
    },
    {
     "claim": "made no clear difference on GPT-5.6 Luna in any of its three sets",
     "metric": "sessions_side",
     "model": "gpt-5.6-luna",
     "skill": "caveman",
     "want": "overlap",
     "sessions": 3
    }
   ]
  },
  "build": {
   "text": "My control file, generic project guidelines that ask for a review pass, cut code only on GPT-6 Luna, 18%, the same as karpathy-skills there. On the other five models it wrote as much code as no skill or more, up to 32% more on GPT-6.1 Sol.",
   "facts": [
    {
     "n": "18",
     "metric": "lines_added",
     "model": "gpt-6-luna",
     "task": "T4",
     "skill": "placebo"
    },
    {
     "n": "18",
     "metric": "lines_added",
     "model": "gpt-6-luna",
     "task": "T4",
     "skill": "karpathy"
    },
    {
     "n": "32",
     "metric": "lines_added",
     "model": "gpt-6.1-sol",
     "task": "T4",
     "skill": "placebo"
    }
   ],
   "checks": [
    {
     "claim": "the same as karpathy-skills there",
     "metric": "equal_change",
     "field": "lines_added",
     "cells": [
      [
       "gpt-6-luna",
       "T4",
       "placebo"
      ],
      [
       "gpt-6-luna",
       "T4",
       "karpathy"
      ]
     ],
     "want": "equal"
    },
    {
     "claim": "up to 32% more on GPT-6.1 Sol",
     "metric": "sign",
     "field": "lines_added",
     "want": "more",
     "cells": [
      [
       "gpt-6.1-sol",
       "T4",
       "placebo"
      ]
     ]
    },
    {
     "claim": "cut code only on GPT-6 Luna",
     "metric": "cut_models",
     "field": "lines_added",
     "task": "T4",
     "skill": "placebo",
     "want": "gpt-6-luna"
    },
    {
     "claim": "up to 32% more on GPT-6.1 Sol",
     "metric": "max_more",
     "field": "lines_added",
     "task": "T4",
     "skill": "placebo",
     "want": "gpt-6.1-sol"
    }
   ]
  },
  "grid": {
   "text": "In 19 of 69 cells, every skill run landed on the same side of every no-skill run. Chance alone would give about 7.",
   "facts": [
    {
     "n": "19",
     "metric": "clean_splits"
    },
    {
     "n": "69",
     "metric": "cells"
    },
    {
     "n": "7",
     "metric": "expected_splits"
    }
   ],
   "checks": []
  },
  "cost": {
   "text": "ponytail claims to be 20% cheaper. The chart adds up all four tasks, but I check that claim on the build task alone. There it ranged from 10% cheaper on GPT-5.6 Luna to 16% pricier on Claude Sonnet 5, and averaged 8% pricier. caveman made Claude Sonnet 5's explain answers 72% shorter but only 2% cheaper. Output was about 15% of that task's cost, and caveman's own file added input to every run.",
   "facts": [
    {
     "n": "72",
     "metric": "visible",
     "model": "claude-sonnet-5",
     "task": "T3",
     "skill": "caveman"
    },
    {
     "n": "2",
     "metric": "cost",
     "model": "claude-sonnet-5",
     "task": "T3",
     "skill": "caveman"
    },
    {
     "n": "15",
     "metric": "output_share_task",
     "model": "claude-sonnet-5",
     "task": "T3",
     "skill": "control"
    },
    {
     "n": "20",
     "metric": "claim"
    },
    {
     "n": "10",
     "metric": "cost",
     "model": "gpt-5.6-luna",
     "task": "T4",
     "skill": "ponytail"
    },
    {
     "n": "16",
     "metric": "cost",
     "model": "claude-sonnet-5",
     "task": "T4",
     "skill": "ponytail"
    },
    {
     "n": "8",
     "metric": "claim_avg",
     "skill": "ponytail",
     "measure": "cost_usd"
    }
   ],
   "checks": [
    {
     "claim": "caveman's own file added input to every run",
     "metric": "more_cache_write",
     "model": "claude-sonnet-5",
     "task": "T3",
     "skill": "caveman",
     "want": "more"
    },
    {
     "claim": "ranged from 10% cheaper on GPT-5.6 Luna to 16% pricier on Claude Sonnet 5",
     "metric": "extremes",
     "field": "cost_usd",
     "task": "T4",
     "skill": "ponytail",
     "want": "gpt-5.6-luna,claude-sonnet-5"
    }
   ]
  },
  "cost_note": {
   "text": "For Claude, these dollar figures are Claude Code's own estimate. It prices the tokens each run sent at Anthropic's API prices, with cache writes at the one-hour rate a subscription login uses. With an API key's default five-minute cache, the same runs price 21% lower on Sonnet 5, 27% lower on Sonnet 5.5 and 29% lower on Opus 5.5. Each skill's gap to no skill moves by up to 3 points. The estimate is not a bill, and a task built directly on the API would send a different prompt and tools.",
   "facts": [
    {
     "n": "21",
     "metric": "cost_5m_drop",
     "model": "claude-sonnet-5"
    },
    {
     "n": "27",
     "metric": "cost_5m_drop",
     "model": "claude-sonnet-5-5"
    },
    {
     "n": "29",
     "metric": "cost_5m_drop",
     "model": "claude-opus-5-5"
    },
    {
     "n": "3",
     "metric": "cost_5m_shift"
    }
   ],
   "checks": [
    {
     "claim": "with cache writes at the one-hour rate a subscription login uses",
     "metric": "all_1h_writes",
     "want": "all one-hour"
    }
   ]
  },
  "round8": {
   "text": "caveman cut visible output 9% to 33% on these models, never near its claimed 65%. Claude Sonnet 5 is still the only model where the claim held. With 2 runs a side, read these as direction.",
   "facts": [
    {
     "n": "9",
     "metric": "r8_visible",
     "model": "gpt-5.6-luna",
     "task": "T3",
     "skill": "caveman"
    },
    {
     "n": "33",
     "metric": "r8_visible",
     "model": "gpt-6-astra",
     "task": "T3",
     "skill": "caveman"
    },
    {
     "n": "65",
     "metric": "claim"
    },
    {
     "n": "2",
     "metric": "r8_side"
    }
   ]
  },
  "heldout": {
   "text": "For the coding tasks, the tests are the only quality measure. After the runs I wrote extra tests with new inputs for the same behavior. I ran them against the kept changes from all 180 coding runs of Claude Opus 5.5, Claude Sonnet 5.5, GPT-6 Luna and GPT-6.1 Sol. The main rounds with Claude Sonnet 5 and GPT-5.6 Luna kept no changes to test. All 180 passed. These tasks can't separate those models on correctness.",
   "facts": [
    {
     "n": "180",
     "metric": "heldout_runs"
    }
   ],
   "checks": [
    {
     "claim": "All 180 passed",
     "metric": "heldout_all_passed",
     "want": "all passed"
    },
    {
     "claim": "Claude Opus 5.5, Claude Sonnet 5.5, GPT-6 Luna and GPT-6.1 Sol",
     "metric": "heldout_model_list",
     "want": "claude-opus-5-5,claude-sonnet-5-5,gpt-6-luna,gpt-6.1-sol"
    }
   ]
  },
  "bughunt_intro": {
   "text": "Each dot is a model's median run. Every model scored at or near full marks, so these tests show cost and time on a small job, not which model is best.",
   "facts": [],
   "checks": []
  },
  "bughunt_note": {
   "text": "Tests written counts the new test cases a run added. Lines changed counts the app code it added or removed. Both are medians of the runs. Neither is a score. More tests or fewer lines isn't better by itself, but both show how much a reviewer has to check.",
   "facts": [],
   "checks": []
  },
  "bughunt_method": {
   "text": "Every model got the same small TypeScript web app, with an API, a React front end and a written spec. Hidden tests the model never saw graded the result. The bug hunt plants 20 bugs across the front end, the API and shared logic. The tickets test asks for weekly repeating bookings, a fix for a double booking reported only by its symptom, and a fix for a security hole, checked by 17 hidden tests. Each model ran each test 3 times with no skill. Claude Opus 5.5 planted half the bugs and GPT-6.1 Sol the other half, and GPT-6.1 Sol reviewed the ticket tests for fairness. The apps and the tests stay private, so no model can train on them. The app is small, so a full pass says little about harder code.",
   "facts": [
    {
     "n": "20",
     "metric": "bughunt_bugs"
    },
    {
     "n": "17",
     "metric": "bughunt_checks"
    },
    {
     "n": "3",
     "metric": "bughunt_runs_per_model"
    }
   ],
   "checks": []
  },
  "code_lead": {
   "text": "Both skills wrote less code on every model.",
   "facts": [],
   "checks": [
    {
     "claim": "Both skills wrote less code on every model",
     "metric": "sign",
     "field": "lines_added",
     "want": "less",
     "cells": [
      [
       "claude-sonnet-5",
       "T4",
       "ponytail"
      ],
      [
       "claude-sonnet-5",
       "T4",
       "karpathy"
      ],
      [
       "claude-opus-5-5",
       "T4",
       "ponytail"
      ],
      [
       "claude-opus-5-5",
       "T4",
       "karpathy"
      ],
      [
       "claude-sonnet-5-5",
       "T4",
       "ponytail"
      ],
      [
       "claude-sonnet-5-5",
       "T4",
       "karpathy"
      ],
      [
       "gpt-5.6-luna",
       "T4",
       "ponytail"
      ],
      [
       "gpt-5.6-luna",
       "T4",
       "karpathy"
      ],
      [
       "gpt-6-luna",
       "T4",
       "ponytail"
      ],
      [
       "gpt-6-luna",
       "T4",
       "karpathy"
      ],
      [
       "gpt-6.1-sol",
       "T4",
       "ponytail"
      ],
      [
       "gpt-6.1-sol",
       "T4",
       "karpathy"
      ]
     ]
    }
   ]
  }
 },
 "round8_models": [
  {
   "id": "claude-fable-5-1",
   "label": "Claude Fable 5.1",
   "short": "Fable 5.1",
   "host": "Claude Code 2.1.283",
   "note": "high effort, Claude Code's default"
  },
  {
   "id": "gpt-5.6-luna",
   "label": "GPT-5.6 Luna",
   "short": "5.6 Luna",
   "host": "Codex CLI 0.157.1",
   "note": "medium effort, set by me"
  },
  {
   "id": "gpt-6-luna",
   "label": "GPT-6 Luna",
   "short": "6 Luna",
   "host": "Codex CLI 0.157.1",
   "note": "medium effort, set by me"
  },
  {
   "id": "gpt-5.6-sol",
   "label": "GPT-5.6 Sol",
   "short": "5.6 Sol",
   "host": "Codex CLI 0.157.1",
   "note": "medium effort, set by me"
  },
  {
   "id": "gpt-6-sol",
   "label": "GPT-6 Sol",
   "short": "6 Sol",
   "host": "Codex CLI 0.157.1",
   "note": "medium effort, set by me"
  },
  {
   "id": "gpt-6-astra",
   "label": "GPT-6 Astra",
   "short": "6 Astra",
   "host": "Codex CLI 0.157.1",
   "note": "medium effort, set by me"
  }
 ],
 "round9_models": [
  {
   "id": "claude-sonnet-5-5",
   "label": "Claude Sonnet 5.5",
   "short": "Sonnet 5.5",
   "host": "Claude Code 2.1.284",
   "note": "medium effort, Claude Code's default"
  },
  {
   "id": "claude-sonnet-5",
   "label": "Claude Sonnet 5",
   "short": "Sonnet 5",
   "host": "Claude Code 2.1.284",
   "note": "high effort, Claude Code's default"
  },
  {
   "id": "claude-opus-5-5",
   "label": "Claude Opus 5.5",
   "short": "Opus 5.5",
   "host": "Claude Code 2.1.284",
   "note": "medium effort, Claude Code's default"
  }
 ],
 "effort_table": [
  {
   "models": [
    "claude-sonnet-5"
   ],
   "cli": "Claude Code 2.1.278, 2.1.283 and 2.1.284",
   "effort": "high",
   "by": "Claude Code's default"
  },
  {
   "models": [
    "claude-opus-5-5"
   ],
   "cli": "Claude Code 2.1.280, 2.1.283 and 2.1.284",
   "effort": "medium",
   "by": "Claude Code's default"
  },
  {
   "models": [
    "claude-sonnet-5-5"
   ],
   "cli": "Claude Code 2.1.284",
   "effort": "medium",
   "by": "Claude Code's default"
  },
  {
   "models": [
    "claude-fable-5-1"
   ],
   "cli": "Claude Code 2.1.283 and 2.1.284",
   "effort": "high",
   "by": "Claude Code's default"
  },
  {
   "models": [
    "gpt-5.6-luna"
   ],
   "cli": "Codex CLI 0.153.3",
   "effort": "low",
   "by": "set by me"
  },
  {
   "models": [
    "gpt-6-luna"
   ],
   "cli": "Codex CLI 0.156.0",
   "effort": "low",
   "by": "set by me"
  },
  {
   "models": [
    "gpt-5.6-luna",
    "gpt-6-luna",
    "gpt-5.6-sol",
    "gpt-6-sol",
    "gpt-6-astra"
   ],
   "cli": "Codex CLI 0.157.1",
   "effort": "medium",
   "by": "set by me"
  },
  {
   "models": [
    "gpt-6.1-sol",
    "gpt-6-sol",
    "gpt-5.6-sol",
    "gpt-5.6-luna",
    "gpt-6-luna",
    "gpt-6-astra"
   ],
   "cli": "Codex CLI 0.159.0",
   "effort": "medium",
   "by": "set by me"
  }
 ],
 "round10_models": [
  {
   "id": "gpt-6.1-sol",
   "label": "GPT-6.1 Sol",
   "short": "6.1 Sol",
   "host": "Codex CLI 0.159.0",
   "note": "medium effort, set by me"
  },
  {
   "id": "gpt-6-sol",
   "label": "GPT-6 Sol",
   "short": "6 Sol",
   "host": "Codex CLI 0.159.0",
   "note": "medium effort, set by me"
  },
  {
   "id": "claude-opus-5-5",
   "label": "Claude Opus 5.5",
   "short": "Opus 5.5",
   "host": "Claude Code 2.1.284",
   "note": "medium effort, Claude Code's default"
  }
 ],
 "api_prices": {
  "checked": "2026-09-29",
  "note": "US dollars per million tokens, standard tier. OpenAI's rates are its short-context rates, which cover every request in these runs. OpenAI lists cache writes only for long context. Claude cache writes at the one-hour rate Claude Code uses, 2x input.",
  "sources": {
   "openai": "https://developers.openai.com/api/docs/pricing",
   "anthropic": "https://claude.com/pricing",
   "anthropic_cache": "https://platform.claude.com/docs/en/build-with-claude/prompt-caching"
  },
  "models": {
   "gpt-6.1-sol": {
    "input": 2.0,
    "cache_read": 0.1,
    "cache_write": null,
    "output": 10.0,
    "source": "openai"
   },
   "gpt-6-sol": {
    "input": 2.0,
    "cache_read": 0.2,
    "cache_write": null,
    "output": 10.0,
    "source": "openai"
   },
   "gpt-6-astra": {
    "input": 10.0,
    "cache_read": 1.0,
    "cache_write": null,
    "output": 50.0,
    "source": "openai"
   },
   "gpt-6-luna": {
    "input": 0.1,
    "cache_read": 0.01,
    "cache_write": null,
    "output": 0.5,
    "source": "openai"
   },
   "gpt-5.6-sol": {
    "input": 4.0,
    "cache_read": 0.4,
    "cache_write": null,
    "output": 20.0,
    "source": "openai"
   },
   "gpt-5.6-luna": {
    "input": 0.2,
    "cache_read": 0.02,
    "cache_write": null,
    "output": 1.2,
    "source": "openai"
   },
   "claude-opus-5-5": {
    "input": 4,
    "cache_read": 0.2,
    "cache_write": 8,
    "output": 20,
    "source": "anthropic"
   },
   "claude-sonnet-5-5": {
    "input": 2,
    "cache_read": 0.2,
    "cache_write": 4,
    "output": 10,
    "source": "anthropic"
   },
   "claude-sonnet-5": {
    "input": 2,
    "cache_read": 0.2,
    "cache_write": 4,
    "output": 10,
    "source": "anthropic"
   },
   "claude-fable-5-1": {
    "input": 10,
    "cache_read": 0.25,
    "cache_write": 20,
    "output": 50,
    "source": "anthropic"
   }
  }
 },
 "generated": "2026-10-06",
 "latest": "2026-09-30",
 "cells": [
  {
   "model": "claude-opus-5-5",
   "task": "T1",
   "skill": "caveman",
   "n": 3,
   "passed": 3,
   "median": 619,
   "min": 570,
   "max": 803,
   "lines_median": 6,
   "lines_min": 6,
   "lines_max": 6,
   "cost_median": 0.19746319999999998,
   "wall_median": 9.3,
   "thinking_share": 0.03231017770597738,
   "vs_control": -5.061349693251538,
   "lines_vs_control": 0.0,
   "cost_vs_control": 12.067904807956431,
   "wall_vs_control": 5.681818181818188
  },
  {
   "model": "claude-opus-5-5",
   "task": "T1",
   "skill": "control",
   "n": 3,
   "passed": 3,
   "median": 652,
   "min": 574,
   "max": 693,
   "lines_median": 6,
   "lines_min": 6,
   "lines_max": 6,
   "cost_median": 0.17619959999999998,
   "wall_median": 8.8,
   "thinking_share": 0.0,
   "vs_control": null,
   "lines_vs_control": null,
   "cost_vs_control": null,
   "wall_vs_control": null
  },
  {
   "model": "claude-opus-5-5",
   "task": "T1",
   "skill": "karpathy",
   "n": 3,
   "passed": 3,
   "median": 772,
   "min": 751,
   "max": 786,
   "lines_median": 6,
   "lines_min": 6,
   "lines_max": 6,
   "cost_median": 0.1888994,
   "wall_median": 11.3,
   "thinking_share": 0.03626943005181347,
   "vs_control": 18.404907975460127,
   "lines_vs_control": 0.0,
   "cost_vs_control": 7.207621356688665,
   "wall_vs_control": 28.409090909090917
  },
  {
   "model": "claude-opus-5-5",
   "task": "T1",
   "skill": "placebo",
   "n": 3,
   "passed": 3,
   "median": 933,
   "min": 896,
   "max": 956,
   "lines_median": 8,
   "lines_min": 6,
   "lines_max": 8,
   "cost_median": 0.2007624,
   "wall_median": 13.1,
   "thinking_share": 0.04072883172561629,
   "vs_control": 43.09815950920246,
   "lines_vs_control": 33.33333333333333,
   "cost_vs_control": 13.940326765781542,
   "wall_vs_control": 48.86363636363635
  },
  {
   "model": "claude-opus-5-5",
   "task": "T1",
   "skill": "ponytail",
   "n": 3,
   "passed": 3,
   "median": 651,
   "min": 616,
   "max": 689,
   "lines_median": 6,
   "lines_min": 6,
   "lines_max": 6,
   "cost_median": 0.18477760000000001,
   "wall_median": 9.1,
   "thinking_share": 0.03533026113671275,
   "vs_control": -0.15337423312883347,
   "lines_vs_control": 0.0,
   "cost_vs_control": 4.868342493399558,
   "wall_vs_control": 3.409090909090895
  },
  {
   "model": "claude-opus-5-5",
   "task": "T2",
   "skill": "caveman",
   "n": 3,
   "passed": 3,
   "median": 994,
   "min": 993,
   "max": 1041,
   "lines_median": 7,
   "lines_min": 7,
   "lines_max": 7,
   "cost_median": 0.23781039999999998,
   "wall_median": 12.5,
   "thinking_share": 0.07243460764587525,
   "vs_control": -12.730465320456542,
   "lines_vs_control": -41.666666666666664,
   "cost_vs_control": 25.102397463131144,
   "wall_vs_control": 0.0
  },
  {
   "model": "claude-opus-5-5",
   "task": "T2",
   "skill": "control",
   "n": 3,
   "passed": 3,
   "median": 1139,
   "min": 1109,
   "max": 1247,
   "lines_median": 12,
   "lines_min": 7,
   "lines_max": 13,
   "cost_median": 0.19009259999999997,
   "wall_median": 12.5,
   "thinking_share": 0.08746618575293057,
   "vs_control": null,
   "lines_vs_control": null,
   "cost_vs_control": null,
   "wall_vs_control": null
  },
  {
   "model": "claude-opus-5-5",
   "task": "T2",
   "skill": "karpathy",
   "n": 3,
   "passed": 3,
   "median": 1045,
   "min": 903,
   "max": 1053,
   "lines_median": 2,
   "lines_min": 2,
   "lines_max": 2,
   "cost_median": 0.1929316,
   "wall_median": 13.0,
   "thinking_share": 0.14174972314507198,
   "vs_control": -8.252853380158031,
   "lines_vs_control": -83.33333333333334,
   "cost_vs_control": 1.4934826500347942,
   "wall_vs_control": 4.0000000000000036
  },
  {
   "model": "claude-opus-5-5",
   "task": "T2",
   "skill": "placebo",
   "n": 3,
   "passed": 3,
   "median": 1333,
   "min": 1252,
   "max": 1630,
   "lines_median": 7,
   "lines_min": 7,
   "lines_max": 17,
   "cost_median": 0.2127214,
   "wall_median": 16.6,
   "thinking_share": 0.17914110429447852,
   "vs_control": 17.032484635645307,
   "lines_vs_control": -41.666666666666664,
   "cost_vs_control": 11.904093057804488,
   "wall_vs_control": 32.800000000000004
  },
  {
   "model": "claude-opus-5-5",
   "task": "T2",
   "skill": "ponytail",
   "n": 3,
   "passed": 3,
   "median": 1241,
   "min": 1126,
   "max": 1417,
   "lines_median": 4,
   "lines_min": 4,
   "lines_max": 4,
   "cost_median": 0.20963240000000002,
   "wall_median": 16.6,
   "thinking_share": 0.2773465067043049,
   "vs_control": 8.955223880597018,
   "lines_vs_control": -66.66666666666667,
   "cost_vs_control": 10.27909555658666,
   "wall_vs_control": 32.800000000000004
  },
  {
   "model": "claude-opus-5-5",
   "task": "T3",
   "skill": "caveman",
   "n": 3,
   "passed": 3,
   "median": 907,
   "min": 880,
   "max": 990,
   "lines_median": 0,
   "lines_min": 0,
   "lines_max": 0,
   "cost_median": 0.0983996,
   "wall_median": 11.9,
   "thinking_share": 0.19735391400220506,
   "vs_control": -22.279348757497864,
   "lines_vs_control": null,
   "cost_vs_control": 20.08239833836527,
   "wall_vs_control": -7.03125
  },
  {
   "model": "claude-opus-5-5",
   "task": "T3",
   "skill": "control",
   "n": 3,
   "passed": 3,
   "median": 1167,
   "min": 1066,
   "max": 1429,
   "lines_median": 0,
   "lines_min": 0,
   "lines_max": 0,
   "cost_median": 0.0819434,
   "wall_median": 12.8,
   "thinking_share": 0.16791744840525327,
   "vs_control": null,
   "lines_vs_control": null,
   "cost_vs_control": null,
   "wall_vs_control": null
  },
  {
   "model": "claude-opus-5-5",
   "task": "T3",
   "skill": "karpathy",
   "n": 3,
   "passed": 3,
   "median": 1223,
   "min": 1176,
   "max": 1420,
   "lines_median": 0,
   "lines_min": 0,
   "lines_max": 0,
   "cost_median": 0.0913698,
   "wall_median": 14.6,
   "thinking_share": 0.195578231292517,
   "vs_control": 4.798628963153395,
   "lines_vs_control": null,
   "cost_vs_control": 11.503550011349306,
   "wall_vs_control": 14.0625
  },
  {
   "model": "claude-opus-5-5",
   "task": "T3",
   "skill": "placebo",
   "n": 3,
   "passed": 3,
   "median": 1210,
   "min": 1066,
   "max": 1227,
   "lines_median": 0,
   "lines_min": 0,
   "lines_max": 0,
   "cost_median": 0.0889534,
   "wall_median": 13.0,
   "thinking_share": 0.17277913610431947,
   "vs_control": 3.6846615252784876,
   "lines_vs_control": null,
   "cost_vs_control": 8.554685307175447,
   "wall_vs_control": 1.5625
  },
  {
   "model": "claude-opus-5-5",
   "task": "T3",
   "skill": "ponytail",
   "n": 3,
   "passed": 3,
   "median": 1272,
   "min": 1010,
   "max": 1294,
   "lines_median": 0,
   "lines_min": 0,
   "lines_max": 0,
   "cost_median": 0.0929616,
   "wall_median": 13.9,
   "thinking_share": 0.19165378670788252,
   "vs_control": 8.997429305912586,
   "lines_vs_control": null,
   "cost_vs_control": 13.446110363982955,
   "wall_vs_control": 8.59375
  },
  {
   "model": "claude-opus-5-5",
   "task": "T4",
   "skill": "caveman",
   "n": 3,
   "passed": 3,
   "median": 1601,
   "min": 1534,
   "max": 1617,
   "lines_median": 56,
   "lines_min": 56,
   "lines_max": 58,
   "cost_median": 0.2635024,
   "wall_median": 16.9,
   "thinking_share": 0.07561929595827901,
   "vs_control": -3.496081977094634,
   "lines_vs_control": 3.703703703703698,
   "cost_vs_control": 23.628904596889,
   "wall_vs_control": -5.056179775280912
  },
  {
   "model": "claude-opus-5-5",
   "task": "T4",
   "skill": "control",
   "n": 3,
   "passed": 3,
   "median": 1659,
   "min": 1496,
   "max": 1681,
   "lines_median": 54,
   "lines_min": 53,
   "lines_max": 56,
   "cost_median": 0.21313980000000002,
   "wall_median": 17.8,
   "thinking_share": 0.06550802139037433,
   "vs_control": null,
   "lines_vs_control": null,
   "cost_vs_control": null,
   "wall_vs_control": null
  },
  {
   "model": "claude-opus-5-5",
   "task": "T4",
   "skill": "karpathy",
   "n": 3,
   "passed": 3,
   "median": 1554,
   "min": 1539,
   "max": 1639,
   "lines_median": 41,
   "lines_min": 40,
   "lines_max": 43,
   "cost_median": 0.22644419999999998,
   "wall_median": 16.0,
   "thinking_share": 0.1402831402831403,
   "vs_control": -6.329113924050633,
   "lines_vs_control": -24.07407407407407,
   "cost_vs_control": 6.242100255325367,
   "wall_vs_control": -10.1123595505618
  },
  {
   "model": "claude-opus-5-5",
   "task": "T4",
   "skill": "placebo",
   "n": 3,
   "passed": 3,
   "median": 2028,
   "min": 1874,
   "max": 2048,
   "lines_median": 66,
   "lines_min": 63,
   "lines_max": 66,
   "cost_median": 0.2276526,
   "wall_median": 20.0,
   "thinking_share": 0.11637080867850098,
   "vs_control": 22.24231464737794,
   "lines_vs_control": 22.222222222222232,
   "cost_vs_control": 6.809052086940115,
   "wall_vs_control": 12.35955056179774
  },
  {
   "model": "claude-opus-5-5",
   "task": "T4",
   "skill": "ponytail",
   "n": 3,
   "passed": 3,
   "median": 1485,
   "min": 1358,
   "max": 1674,
   "lines_median": 40,
   "lines_min": 38,
   "lines_max": 44,
   "cost_median": 0.22532539999999998,
   "wall_median": 15.7,
   "thinking_share": 0.13679808841099164,
   "vs_control": -10.488245931283902,
   "lines_vs_control": -25.92592592592593,
   "cost_vs_control": 5.717186560182541,
   "wall_vs_control": -11.79775280898877
  },
  {
   "model": "claude-sonnet-5",
   "task": "T1",
   "skill": "caveman",
   "n": 3,
   "passed": 3,
   "median": 1026,
   "min": 836,
   "max": 1156,
   "lines_median": 5,
   "lines_min": 5,
   "lines_max": 5,
   "cost_median": 0.16663619999999998,
   "wall_median": 15.1,
   "thinking_share": 0.02729044834307992,
   "vs_control": -10.393013100436676,
   "lines_vs_control": 0.0,
   "cost_vs_control": 8.212632541418351,
   "wall_vs_control": 4.137931034482767
  },
  {
   "model": "claude-sonnet-5",
   "task": "T1",
   "skill": "control",
   "n": 3,
   "passed": 3,
   "median": 1145,
   "min": 1047,
   "max": 1197,
   "lines_median": 5,
   "lines_min": 5,
   "lines_max": 5,
   "cost_median": 0.15398960000000003,
   "wall_median": 14.5,
   "thinking_share": 0.09082969432314411,
   "vs_control": null,
   "lines_vs_control": null,
   "cost_vs_control": null,
   "wall_vs_control": null
  },
  {
   "model": "claude-sonnet-5",
   "task": "T1",
   "skill": "karpathy",
   "n": 3,
   "passed": 3,
   "median": 1197,
   "min": 1117,
   "max": 1279,
   "lines_median": 5,
   "lines_min": 5,
   "lines_max": 5,
   "cost_median": 0.16338799999999998,
   "wall_median": 15.4,
   "thinking_share": 0.03508771929824561,
   "vs_control": 4.541484716157207,
   "lines_vs_control": 0.0,
   "cost_vs_control": 6.1032693116937375,
   "wall_vs_control": 6.20689655172415
  },
  {
   "model": "claude-sonnet-5",
   "task": "T1",
   "skill": "placebo",
   "n": 3,
   "passed": 3,
   "median": 1396,
   "min": 1187,
   "max": 1540,
   "lines_median": 6,
   "lines_min": 5,
   "lines_max": 8,
   "cost_median": 0.1691966,
   "wall_median": 17.5,
   "thinking_share": 0.06948424068767908,
   "vs_control": 21.921397379912655,
   "lines_vs_control": 19.999999999999996,
   "cost_vs_control": 9.875342230903893,
   "wall_vs_control": 20.68965517241379
  },
  {
   "model": "claude-sonnet-5",
   "task": "T1",
   "skill": "ponytail",
   "n": 3,
   "passed": 3,
   "median": 1369,
   "min": 948,
   "max": 1483,
   "lines_median": 7,
   "lines_min": 5,
   "lines_max": 7,
   "cost_median": 0.1710516,
   "wall_median": 15.2,
   "thinking_share": 0.10126582278481013,
   "vs_control": 19.563318777292583,
   "lines_vs_control": 39.99999999999999,
   "cost_vs_control": 11.079969036870008,
   "wall_vs_control": 4.827586206896539
  },
  {
   "model": "claude-sonnet-5",
   "task": "T2",
   "skill": "caveman",
   "n": 3,
   "passed": 3,
   "median": 1256,
   "min": 816,
   "max": 1337,
   "lines_median": 2,
   "lines_min": 2,
   "lines_max": 2,
   "cost_median": 0.1597078,
   "wall_median": 16.2,
   "thinking_share": 0.04538216560509554,
   "vs_control": 2.698282910874905,
   "lines_vs_control": 0.0,
   "cost_vs_control": 0.5473487519390696,
   "wall_vs_control": -26.363636363636367
  },
  {
   "model": "claude-sonnet-5",
   "task": "T2",
   "skill": "control",
   "n": 3,
   "passed": 3,
   "median": 1223,
   "min": 1111,
   "max": 1277,
   "lines_median": 2,
   "lines_min": 2,
   "lines_max": 2,
   "cost_median": 0.15883840000000002,
   "wall_median": 22.0,
   "thinking_share": 0.10801080108010801,
   "vs_control": null,
   "lines_vs_control": null,
   "cost_vs_control": null,
   "wall_vs_control": null
  },
  {
   "model": "claude-sonnet-5",
   "task": "T2",
   "skill": "karpathy",
   "n": 3,
   "passed": 3,
   "median": 1666,
   "min": 1317,
   "max": 1791,
   "lines_median": 2,
   "lines_min": 2,
   "lines_max": 2,
   "cost_median": 0.182276,
   "wall_median": 21.6,
   "thinking_share": 0.29111644657863145,
   "vs_control": 36.222403924775136,
   "lines_vs_control": 0.0,
   "cost_vs_control": 14.75562584362471,
   "wall_vs_control": -1.8181818181818077
  },
  {
   "model": "claude-sonnet-5",
   "task": "T2",
   "skill": "placebo",
   "n": 3,
   "passed": 3,
   "median": 2050,
   "min": 1596,
   "max": 2054,
   "lines_median": 2,
   "lines_min": 2,
   "lines_max": 2,
   "cost_median": 0.1683506,
   "wall_median": 22.1,
   "thinking_share": 0.3578383641674781,
   "vs_control": 67.62060506950122,
   "lines_vs_control": 0.0,
   "cost_vs_control": 5.988602252352049,
   "wall_vs_control": 0.454545454545463
  },
  {
   "model": "claude-sonnet-5",
   "task": "T2",
   "skill": "ponytail",
   "n": 3,
   "passed": 3,
   "median": 1718,
   "min": 1552,
   "max": 1824,
   "lines_median": 4,
   "lines_min": 4,
   "lines_max": 4,
   "cost_median": 0.1778672,
   "wall_median": 20.5,
   "thinking_share": 0.26739690721649484,
   "vs_control": 40.47424366312347,
   "lines_vs_control": 100.0,
   "cost_vs_control": 11.979974615710054,
   "wall_vs_control": -6.818181818181824
  },
  {
   "model": "claude-sonnet-5",
   "task": "T3",
   "skill": "caveman",
   "n": 3,
   "passed": 3,
   "median": 289,
   "min": 239,
   "max": 291,
   "lines_median": 0,
   "lines_min": 0,
   "lines_max": 0,
   "cost_median": 0.0639758,
   "wall_median": 5.9,
   "thinking_share": 0.05154639175257732,
   "vs_control": -70.38934426229508,
   "lines_vs_control": null,
   "cost_vs_control": -2.3723340617484268,
   "wall_vs_control": -45.370370370370374
  },
  {
   "model": "claude-sonnet-5",
   "task": "T3",
   "skill": "control",
   "n": 3,
   "passed": 3,
   "median": 976,
   "min": 744,
   "max": 984,
   "lines_median": 0,
   "lines_min": 0,
   "lines_max": 0,
   "cost_median": 0.06553039999999999,
   "wall_median": 10.8,
   "thinking_share": 0.020491803278688523,
   "vs_control": null,
   "lines_vs_control": null,
   "cost_vs_control": null,
   "wall_vs_control": null
  },
  {
   "model": "claude-sonnet-5",
   "task": "T3",
   "skill": "karpathy",
   "n": 3,
   "passed": 3,
   "median": 785,
   "min": 745,
   "max": 907,
   "lines_median": 0,
   "lines_min": 0,
   "lines_max": 0,
   "cost_median": 0.06211,
   "wall_median": 8.9,
   "thinking_share": 0.025477707006369428,
   "vs_control": -19.56967213114754,
   "lines_vs_control": null,
   "cost_vs_control": -5.2195622184512676,
   "wall_vs_control": -17.59259259259259
  },
  {
   "model": "claude-sonnet-5",
   "task": "T3",
   "skill": "placebo",
   "n": 3,
   "passed": 3,
   "median": 743,
   "min": 724,
   "max": 963,
   "lines_median": 0,
   "lines_min": 0,
   "lines_max": 0,
   "cost_median": 0.0672198,
   "wall_median": 9.8,
   "thinking_share": 0.03314917127071823,
   "vs_control": -23.872950819672134,
   "lines_vs_control": null,
   "cost_vs_control": 2.5780401157325494,
   "wall_vs_control": -9.259259259259256
  },
  {
   "model": "claude-sonnet-5",
   "task": "T3",
   "skill": "ponytail",
   "n": 3,
   "passed": 3,
   "median": 767,
   "min": 749,
   "max": 917,
   "lines_median": 0,
   "lines_min": 0,
   "lines_max": 0,
   "cost_median": 0.0686944,
   "wall_median": 11.8,
   "thinking_share": 0.03489640130861505,
   "vs_control": -21.41393442622951,
   "lines_vs_control": null,
   "cost_vs_control": 4.828293433276798,
   "wall_vs_control": 9.259259259259256
  },
  {
   "model": "claude-sonnet-5",
   "task": "T4",
   "skill": "caveman",
   "n": 3,
   "passed": 3,
   "median": 3191,
   "min": 3124,
   "max": 3258,
   "lines_median": 57,
   "lines_min": 50,
   "lines_max": 59,
   "cost_median": 0.21169880000000005,
   "wall_median": 32.4,
   "thinking_share": 0.49294891883422126,
   "vs_control": -16.509680795395077,
   "lines_vs_control": -8.064516129032262,
   "cost_vs_control": 5.076248960151242,
   "wall_vs_control": -6.62824207492797
  },
  {
   "model": "claude-sonnet-5",
   "task": "T4",
   "skill": "control",
   "n": 3,
   "passed": 3,
   "median": 3822,
   "min": 3161,
   "max": 3891,
   "lines_median": 62,
   "lines_min": 59,
   "lines_max": 65,
   "cost_median": 0.2014716,
   "wall_median": 34.7,
   "thinking_share": 0.49807247494217427,
   "vs_control": null,
   "lines_vs_control": null,
   "cost_vs_control": null,
   "wall_vs_control": null
  },
  {
   "model": "claude-sonnet-5",
   "task": "T4",
   "skill": "karpathy",
   "n": 3,
   "passed": 3,
   "median": 3636,
   "min": 3566,
   "max": 4310,
   "lines_median": 56,
   "lines_min": 55,
   "lines_max": 57,
   "cost_median": 0.19371699999999997,
   "wall_median": 37.3,
   "thinking_share": 0.5420792079207921,
   "vs_control": -4.866562009419151,
   "lines_vs_control": -9.677419354838712,
   "cost_vs_control": -3.8489792109657306,
   "wall_vs_control": 7.492795389048967
  },
  {
   "model": "claude-sonnet-5",
   "task": "T4",
   "skill": "placebo",
   "n": 3,
   "passed": 3,
   "median": 4895,
   "min": 3077,
   "max": 5249,
   "lines_median": 63,
   "lines_min": 63,
   "lines_max": 71,
   "cost_median": 0.2092876,
   "wall_median": 48.4,
   "thinking_share": 0.574585635359116,
   "vs_control": 28.074306645735227,
   "lines_vs_control": 1.6129032258064502,
   "cost_vs_control": 3.879454970328311,
   "wall_vs_control": 39.48126801152736
  },
  {
   "model": "claude-sonnet-5",
   "task": "T4",
   "skill": "ponytail",
   "n": 3,
   "passed": 3,
   "median": 3761,
   "min": 3267,
   "max": 4661,
   "lines_median": 47,
   "lines_min": 41,
   "lines_max": 48,
   "cost_median": 0.23293099999999994,
   "wall_median": 38.4,
   "thinking_share": 0.5139590534432332,
   "vs_control": -1.5960230245944507,
   "lines_vs_control": -24.193548387096776,
   "cost_vs_control": 15.614806255571478,
   "wall_vs_control": 10.662824207492783
  },
  {
   "model": "claude-sonnet-5-5",
   "task": "T1",
   "skill": "caveman",
   "n": 3,
   "passed": 3,
   "median": 530,
   "min": 510,
   "max": 625,
   "lines_median": 8,
   "lines_min": 6,
   "lines_max": 8,
   "cost_median": 0.12258859999999999,
   "wall_median": 7.4,
   "thinking_share": 0.0,
   "vs_control": 4.743083003952564,
   "lines_vs_control": 33.33333333333333,
   "cost_vs_control": 26.246725106691482,
   "wall_vs_control": -13.953488372093014
  },
  {
   "model": "claude-sonnet-5-5",
   "task": "T1",
   "skill": "control",
   "n": 3,
   "passed": 3,
   "median": 506,
   "min": 498,
   "max": 652,
   "lines_median": 6,
   "lines_min": 6,
   "lines_max": 6,
   "cost_median": 0.0971024,
   "wall_median": 8.6,
   "thinking_share": 0.0,
   "vs_control": null,
   "lines_vs_control": null,
   "cost_vs_control": null,
   "wall_vs_control": null
  },
  {
   "model": "claude-sonnet-5-5",
   "task": "T1",
   "skill": "karpathy",
   "n": 3,
   "passed": 3,
   "median": 768,
   "min": 687,
   "max": 792,
   "lines_median": 5,
   "lines_min": 5,
   "lines_max": 6,
   "cost_median": 0.10886100000000001,
   "wall_median": 8.0,
   "thinking_share": 0.0,
   "vs_control": 51.778656126482204,
   "lines_vs_control": -16.666666666666664,
   "cost_vs_control": 12.10948442057045,
   "wall_vs_control": -6.976744186046513
  },
  {
   "model": "claude-sonnet-5-5",
   "task": "T1",
   "skill": "placebo",
   "n": 3,
   "passed": 3,
   "median": 844,
   "min": 830,
   "max": 858,
   "lines_median": 8,
   "lines_min": 8,
   "lines_max": 8,
   "cost_median": 0.107339,
   "wall_median": 9.2,
   "thinking_share": 0.0,
   "vs_control": 66.798418972332,
   "lines_vs_control": 33.33333333333333,
   "cost_vs_control": 10.542066931404381,
   "wall_vs_control": 6.976744186046502
  },
  {
   "model": "claude-sonnet-5-5",
   "task": "T1",
   "skill": "ponytail",
   "n": 3,
   "passed": 3,
   "median": 691,
   "min": 574,
   "max": 784,
   "lines_median": 6,
   "lines_min": 6,
   "lines_max": 6,
   "cost_median": 0.1078068,
   "wall_median": 9.1,
   "thinking_share": 0.0,
   "vs_control": 36.5612648221344,
   "lines_vs_control": 0.0,
   "cost_vs_control": 11.023826393580372,
   "wall_vs_control": 5.813953488372103
  },
  {
   "model": "claude-sonnet-5-5",
   "task": "T2",
   "skill": "caveman",
   "n": 3,
   "passed": 3,
   "median": 616,
   "min": 554,
   "max": 634,
   "lines_median": 2,
   "lines_min": 2,
   "lines_max": 2,
   "cost_median": 0.1133546,
   "wall_median": 6.5,
   "thinking_share": 0.16876971608832808,
   "vs_control": -1.9108280254777066,
   "lines_vs_control": -66.66666666666667,
   "cost_vs_control": 13.721009091292125,
   "wall_vs_control": -18.75
  },
  {
   "model": "claude-sonnet-5-5",
   "task": "T2",
   "skill": "control",
   "n": 3,
   "passed": 3,
   "median": 628,
   "min": 505,
   "max": 771,
   "lines_median": 6,
   "lines_min": 2,
   "lines_max": 6,
   "cost_median": 0.09967780000000001,
   "wall_median": 8.0,
   "thinking_share": 0.0,
   "vs_control": null,
   "lines_vs_control": null,
   "cost_vs_control": null,
   "wall_vs_control": null
  },
  {
   "model": "claude-sonnet-5-5",
   "task": "T2",
   "skill": "karpathy",
   "n": 3,
   "passed": 3,
   "median": 634,
   "min": 500,
   "max": 660,
   "lines_median": 2,
   "lines_min": 2,
   "lines_max": 2,
   "cost_median": 0.0994492,
   "wall_median": 7.6,
   "thinking_share": 0.06060606060606061,
   "vs_control": 0.9554140127388644,
   "lines_vs_control": -66.66666666666667,
   "cost_vs_control": -0.22933893003257433,
   "wall_vs_control": -5.000000000000004
  },
  {
   "model": "claude-sonnet-5-5",
   "task": "T2",
   "skill": "placebo",
   "n": 3,
   "passed": 3,
   "median": 884,
   "min": 772,
   "max": 955,
   "lines_median": 3,
   "lines_min": 3,
   "lines_max": 7,
   "cost_median": 0.10612540000000001,
   "wall_median": 10.4,
   "thinking_share": 0.07126696832579185,
   "vs_control": 40.764331210191074,
   "lines_vs_control": -50.0,
   "cost_vs_control": 6.468441317926366,
   "wall_vs_control": 30.000000000000004
  },
  {
   "model": "claude-sonnet-5-5",
   "task": "T2",
   "skill": "ponytail",
   "n": 3,
   "passed": 3,
   "median": 886,
   "min": 728,
   "max": 1316,
   "lines_median": 4,
   "lines_min": 4,
   "lines_max": 8,
   "cost_median": 0.10491299999999999,
   "wall_median": 10.6,
   "thinking_share": 0.15521978021978022,
   "vs_control": 41.082802547770704,
   "lines_vs_control": -33.333333333333336,
   "cost_vs_control": 5.2521223381735815,
   "wall_vs_control": 32.49999999999999
  },
  {
   "model": "claude-sonnet-5-5",
   "task": "T3",
   "skill": "caveman",
   "n": 3,
   "passed": 3,
   "median": 621,
   "min": 603,
   "max": 694,
   "lines_median": 0,
   "lines_min": 0,
   "lines_max": 0,
   "cost_median": 0.0476228,
   "wall_median": 6.6,
   "thinking_share": 0.0,
   "vs_control": -28.125,
   "lines_vs_control": null,
   "cost_vs_control": 22.403512018588188,
   "wall_vs_control": -15.384615384615385
  },
  {
   "model": "claude-sonnet-5-5",
   "task": "T3",
   "skill": "control",
   "n": 3,
   "passed": 3,
   "median": 864,
   "min": 841,
   "max": 909,
   "lines_median": 0,
   "lines_min": 0,
   "lines_max": 0,
   "cost_median": 0.0389064,
   "wall_median": 7.8,
   "thinking_share": 0.0,
   "vs_control": null,
   "lines_vs_control": null,
   "cost_vs_control": null,
   "wall_vs_control": null
  },
  {
   "model": "claude-sonnet-5-5",
   "task": "T3",
   "skill": "karpathy",
   "n": 3,
   "passed": 3,
   "median": 857,
   "min": 845,
   "max": 900,
   "lines_median": 0,
   "lines_min": 0,
   "lines_max": 0,
   "cost_median": 0.0431284,
   "wall_median": 8.9,
   "thinking_share": 0.0,
   "vs_control": -0.810185185185186,
   "lines_vs_control": null,
   "cost_vs_control": 10.851685069808559,
   "wall_vs_control": 14.10256410256412
  },
  {
   "model": "claude-sonnet-5-5",
   "task": "T3",
   "skill": "placebo",
   "n": 3,
   "passed": 3,
   "median": 788,
   "min": 732,
   "max": 848,
   "lines_median": 0,
   "lines_min": 0,
   "lines_max": 0,
   "cost_median": 0.0413048,
   "wall_median": 8.9,
   "thinking_share": 0.0,
   "vs_control": -8.79629629629629,
   "lines_vs_control": null,
   "cost_vs_control": 6.164538482100634,
   "wall_vs_control": 14.10256410256412
  },
  {
   "model": "claude-sonnet-5-5",
   "task": "T3",
   "skill": "ponytail",
   "n": 3,
   "passed": 3,
   "median": 774,
   "min": 718,
   "max": 842,
   "lines_median": 0,
   "lines_min": 0,
   "lines_max": 0,
   "cost_median": 0.042344599999999996,
   "wall_median": 10.0,
   "thinking_share": 0.0,
   "vs_control": -10.416666666666663,
   "lines_vs_control": null,
   "cost_vs_control": 8.837106491476977,
   "wall_vs_control": 28.205128205128215
  },
  {
   "model": "claude-sonnet-5-5",
   "task": "T4",
   "skill": "caveman",
   "n": 3,
   "passed": 3,
   "median": 1349,
   "min": 1298,
   "max": 1358,
   "lines_median": 55,
   "lines_min": 53,
   "lines_max": 58,
   "cost_median": 0.1316116,
   "wall_median": 11.5,
   "thinking_share": 0.045454545454545456,
   "vs_control": 13.64785172704297,
   "lines_vs_control": 12.244897959183664,
   "cost_vs_control": 28.59275644958943,
   "wall_vs_control": 1.7699115044247815
  },
  {
   "model": "claude-sonnet-5-5",
   "task": "T4",
   "skill": "control",
   "n": 3,
   "passed": 3,
   "median": 1187,
   "min": 1108,
   "max": 1319,
   "lines_median": 49,
   "lines_min": 43,
   "lines_max": 49,
   "cost_median": 0.10234760000000001,
   "wall_median": 11.3,
   "thinking_share": 0.0,
   "vs_control": null,
   "lines_vs_control": null,
   "cost_vs_control": null,
   "wall_vs_control": null
  },
  {
   "model": "claude-sonnet-5-5",
   "task": "T4",
   "skill": "karpathy",
   "n": 3,
   "passed": 3,
   "median": 1349,
   "min": 1275,
   "max": 1405,
   "lines_median": 43,
   "lines_min": 43,
   "lines_max": 44,
   "cost_median": 0.11659180000000002,
   "wall_median": 11.8,
   "thinking_share": 0.07188612099644127,
   "vs_control": 13.64785172704297,
   "lines_vs_control": -12.244897959183676,
   "cost_vs_control": 13.917473394588642,
   "wall_vs_control": 4.424778761061954
  },
  {
   "model": "claude-sonnet-5-5",
   "task": "T4",
   "skill": "placebo",
   "n": 3,
   "passed": 3,
   "median": 1571,
   "min": 1444,
   "max": 1629,
   "lines_median": 59,
   "lines_min": 52,
   "lines_max": 62,
   "cost_median": 0.11419880000000002,
   "wall_median": 13.1,
   "thinking_share": 0.061107574793125397,
   "vs_control": 32.350463352990744,
   "lines_vs_control": 20.408163265306122,
   "cost_vs_control": 11.579362877097266,
   "wall_vs_control": 15.92920353982299
  },
  {
   "model": "claude-sonnet-5-5",
   "task": "T4",
   "skill": "ponytail",
   "n": 3,
   "passed": 3,
   "median": 1159,
   "min": 1101,
   "max": 1248,
   "lines_median": 31,
   "lines_min": 31,
   "lines_max": 34,
   "cost_median": 0.11250380000000001,
   "wall_median": 10.6,
   "thinking_share": 0.11734253666954271,
   "vs_control": -2.3588879528222417,
   "lines_vs_control": -36.73469387755102,
   "cost_vs_control": 9.923241971477603,
   "wall_vs_control": -6.194690265486735
  },
  {
   "model": "gpt-5.6-luna",
   "task": "T1",
   "skill": "caveman",
   "n": 3,
   "passed": 3,
   "median": 958,
   "min": 696,
   "max": 1025,
   "lines_median": 5,
   "lines_min": 5,
   "lines_max": 7,
   "cost_median": 0.006988,
   "wall_median": 35.1,
   "thinking_share": 0.16597077244258873,
   "vs_control": -8.325358851674636,
   "lines_vs_control": 0.0,
   "cost_vs_control": -12.089571015222035,
   "wall_vs_control": -11.809045226130642
  },
  {
   "model": "gpt-5.6-luna",
   "task": "T1",
   "skill": "control",
   "n": 3,
   "passed": 3,
   "median": 1045,
   "min": 1031,
   "max": 1577,
   "lines_median": 5,
   "lines_min": 5,
   "lines_max": 7,
   "cost_median": 0.007949,
   "wall_median": 39.8,
   "thinking_share": 0.1722488038277512,
   "vs_control": null,
   "lines_vs_control": null,
   "cost_vs_control": null,
   "wall_vs_control": null
  },
  {
   "model": "gpt-5.6-luna",
   "task": "T1",
   "skill": "karpathy",
   "n": 3,
   "passed": 3,
   "median": 1199,
   "min": 986,
   "max": 1285,
   "lines_median": 5,
   "lines_min": 5,
   "lines_max": 5,
   "cost_median": 0.007028,
   "wall_median": 32.2,
   "thinking_share": 0.16930775646371976,
   "vs_control": 14.73684210526316,
   "lines_vs_control": 0.0,
   "cost_vs_control": -11.586363064536409,
   "wall_vs_control": -19.095477386934657
  },
  {
   "model": "gpt-5.6-luna",
   "task": "T1",
   "skill": "placebo",
   "n": 3,
   "passed": 3,
   "median": 1087,
   "min": 834,
   "max": 1146,
   "lines_median": 6,
   "lines_min": 6,
   "lines_max": 7,
   "cost_median": 0.00724,
   "wall_median": 39.9,
   "thinking_share": 0.13339466421343146,
   "vs_control": 4.019138755980856,
   "lines_vs_control": 19.999999999999996,
   "cost_vs_control": -8.91936092590262,
   "wall_vs_control": 0.2512562814070307
  },
  {
   "model": "gpt-5.6-luna",
   "task": "T1",
   "skill": "ponytail",
   "n": 3,
   "passed": 3,
   "median": 1009,
   "min": 985,
   "max": 1135,
   "lines_median": 5,
   "lines_min": 5,
   "lines_max": 5,
   "cost_median": 0.007092,
   "wall_median": 34.2,
   "thinking_share": 0.17004405286343613,
   "vs_control": -3.4449760765550286,
   "lines_vs_control": 0.0,
   "cost_vs_control": -10.781230343439418,
   "wall_vs_control": -14.070351758793953
  },
  {
   "model": "gpt-5.6-luna",
   "task": "T2",
   "skill": "caveman",
   "n": 3,
   "passed": 3,
   "median": 870,
   "min": 725,
   "max": 1002,
   "lines_median": 2,
   "lines_min": 1,
   "lines_max": 2,
   "cost_median": 0.007797,
   "wall_median": 31.7,
   "thinking_share": 0.18482758620689654,
   "vs_control": 2.473498233215543,
   "lines_vs_control": 0.0,
   "cost_vs_control": 13.180432573668167,
   "wall_vs_control": -17.66233766233767
  },
  {
   "model": "gpt-5.6-luna",
   "task": "T2",
   "skill": "control",
   "n": 3,
   "passed": 3,
   "median": 849,
   "min": 779,
   "max": 1123,
   "lines_median": 2,
   "lines_min": 2,
   "lines_max": 2,
   "cost_median": 0.006889,
   "wall_median": 38.5,
   "thinking_share": 0.15429917550058891,
   "vs_control": null,
   "lines_vs_control": null,
   "cost_vs_control": null,
   "wall_vs_control": null
  },
  {
   "model": "gpt-5.6-luna",
   "task": "T2",
   "skill": "karpathy",
   "n": 3,
   "passed": 3,
   "median": 1360,
   "min": 958,
   "max": 1391,
   "lines_median": 2,
   "lines_min": 2,
   "lines_max": 2,
   "cost_median": 0.008423,
   "wall_median": 44.9,
   "thinking_share": 0.18404025880661395,
   "vs_control": 60.188457008245,
   "lines_vs_control": 0.0,
   "cost_vs_control": 22.267382784148637,
   "wall_vs_control": 16.62337662337663
  },
  {
   "model": "gpt-5.6-luna",
   "task": "T2",
   "skill": "placebo",
   "n": 3,
   "passed": 3,
   "median": 1265,
   "min": 847,
   "max": 1432,
   "lines_median": 2,
   "lines_min": 2,
   "lines_max": 4,
   "cost_median": 0.007589,
   "wall_median": 38.9,
   "thinking_share": 0.22206703910614525,
   "vs_control": 48.99882214369846,
   "lines_vs_control": 0.0,
   "cost_vs_control": 10.161126433444622,
   "wall_vs_control": 1.0389610389610393
  },
  {
   "model": "gpt-5.6-luna",
   "task": "T2",
   "skill": "ponytail",
   "n": 3,
   "passed": 3,
   "median": 1009,
   "min": 1004,
   "max": 1125,
   "lines_median": 2,
   "lines_min": 2,
   "lines_max": 2,
   "cost_median": 0.007366,
   "wall_median": 40.1,
   "thinking_share": 0.18755555555555556,
   "vs_control": 18.84570082449941,
   "lines_vs_control": 0.0,
   "cost_vs_control": 6.924081869647258,
   "wall_vs_control": 4.155844155844157
  },
  {
   "model": "gpt-5.6-luna",
   "task": "T3",
   "skill": "caveman",
   "n": 3,
   "passed": 3,
   "median": 232,
   "min": 217,
   "max": 246,
   "lines_median": 0,
   "lines_min": 0,
   "lines_max": 0,
   "cost_median": 0.003706,
   "wall_median": 10.1,
   "thinking_share": 0.03879310344827586,
   "vs_control": -3.3333333333333326,
   "lines_vs_control": null,
   "cost_vs_control": 15.415758330738093,
   "wall_vs_control": 10.989010989010994
  },
  {
   "model": "gpt-5.6-luna",
   "task": "T3",
   "skill": "control",
   "n": 3,
   "passed": 3,
   "median": 240,
   "min": 234,
   "max": 247,
   "lines_median": 0,
   "lines_min": 0,
   "lines_max": 0,
   "cost_median": 0.003211,
   "wall_median": 9.1,
   "thinking_share": 0.058333333333333334,
   "vs_control": null,
   "lines_vs_control": null,
   "cost_vs_control": null,
   "wall_vs_control": null
  },
  {
   "model": "gpt-5.6-luna",
   "task": "T3",
   "skill": "karpathy",
   "n": 3,
   "passed": 3,
   "median": 262,
   "min": 250,
   "max": 270,
   "lines_median": 0,
   "lines_min": 0,
   "lines_max": 0,
   "cost_median": 0.003832,
   "wall_median": 13.4,
   "thinking_share": 0.05555555555555555,
   "vs_control": 9.166666666666657,
   "lines_vs_control": null,
   "cost_vs_control": 19.33976954219869,
   "wall_vs_control": 47.25274725274726
  },
  {
   "model": "gpt-5.6-luna",
   "task": "T3",
   "skill": "placebo",
   "n": 3,
   "passed": 3,
   "median": 237,
   "min": 232,
   "max": 252,
   "lines_median": 0,
   "lines_min": 0,
   "lines_max": 0,
   "cost_median": 0.003756,
   "wall_median": 13.3,
   "thinking_share": 0.05952380952380952,
   "vs_control": -1.2499999999999956,
   "lines_vs_control": null,
   "cost_vs_control": 16.972905636873236,
   "wall_vs_control": 46.15384615384617
  },
  {
   "model": "gpt-5.6-luna",
   "task": "T3",
   "skill": "ponytail",
   "n": 3,
   "passed": 3,
   "median": 279,
   "min": 254,
   "max": 290,
   "lines_median": 0,
   "lines_min": 0,
   "lines_max": 0,
   "cost_median": 0.003844,
   "wall_median": 10.3,
   "thinking_share": 0.04482758620689655,
   "vs_control": 16.250000000000007,
   "lines_vs_control": null,
   "cost_vs_control": 19.71348489567115,
   "wall_vs_control": 13.186813186813207
  },
  {
   "model": "gpt-5.6-luna",
   "task": "T4",
   "skill": "caveman",
   "n": 3,
   "passed": 3,
   "median": 1691,
   "min": 1336,
   "max": 2072,
   "lines_median": 50,
   "lines_min": 48,
   "lines_max": 66,
   "cost_median": 0.01217,
   "wall_median": 59.2,
   "thinking_share": 0.15494011976047903,
   "vs_control": -8.988159311087196,
   "lines_vs_control": -1.9607843137254943,
   "cost_vs_control": 11.949222702603258,
   "wall_vs_control": 5.714285714285716
  },
  {
   "model": "gpt-5.6-luna",
   "task": "T4",
   "skill": "control",
   "n": 3,
   "passed": 3,
   "median": 1858,
   "min": 1567,
   "max": 2343,
   "lines_median": 51,
   "lines_min": 50,
   "lines_max": 51,
   "cost_median": 0.010871,
   "wall_median": 56.0,
   "thinking_share": 0.15554359526372444,
   "vs_control": null,
   "lines_vs_control": null,
   "cost_vs_control": null,
   "wall_vs_control": null
  },
  {
   "model": "gpt-5.6-luna",
   "task": "T4",
   "skill": "karpathy",
   "n": 3,
   "passed": 3,
   "median": 1753,
   "min": 1740,
   "max": 2003,
   "lines_median": 45,
   "lines_min": 39,
   "lines_max": 47,
   "cost_median": 0.010477,
   "wall_median": 45.5,
   "thinking_share": 0.2127780946948089,
   "vs_control": -5.651237890204519,
   "lines_vs_control": -11.764705882352944,
   "cost_vs_control": -3.6243215895501835,
   "wall_vs_control": -18.75
  },
  {
   "model": "gpt-5.6-luna",
   "task": "T4",
   "skill": "placebo",
   "n": 3,
   "passed": 3,
   "median": 2031,
   "min": 1569,
   "max": 2060,
   "lines_median": 51,
   "lines_min": 51,
   "lines_max": 72,
   "cost_median": 0.0092,
   "wall_median": 49.5,
   "thinking_share": 0.17135922330097086,
   "vs_control": 9.311087190527445,
   "lines_vs_control": 0.0,
   "cost_vs_control": -15.371171005427286,
   "wall_vs_control": -11.607142857142861
  },
  {
   "model": "gpt-5.6-luna",
   "task": "T4",
   "skill": "ponytail",
   "n": 3,
   "passed": 3,
   "median": 1868,
   "min": 1772,
   "max": 2091,
   "lines_median": 44,
   "lines_min": 43,
   "lines_max": 44,
   "cost_median": 0.009769,
   "wall_median": 71.3,
   "thinking_share": 0.1937901498929336,
   "vs_control": 0.5382131324004336,
   "lines_vs_control": -13.725490196078427,
   "cost_vs_control": -10.137061907828171,
   "wall_vs_control": 27.321428571428562
  },
  {
   "model": "gpt-6-luna",
   "task": "T1",
   "skill": "caveman",
   "n": 3,
   "passed": 3,
   "median": 596,
   "min": 549,
   "max": 653,
   "lines_median": 6,
   "lines_min": 6,
   "lines_max": 8,
   "cost_median": 0.003895,
   "wall_median": 22.2,
   "thinking_share": 0.0,
   "vs_control": -14.735336194563665,
   "lines_vs_control": 19.999999999999996,
   "cost_vs_control": 21.377376129635394,
   "wall_vs_control": -8.264462809917362
  },
  {
   "model": "gpt-6-luna",
   "task": "T1",
   "skill": "control",
   "n": 3,
   "passed": 3,
   "median": 699,
   "min": 528,
   "max": 821,
   "lines_median": 5,
   "lines_min": 5,
   "lines_max": 6,
   "cost_median": 0.003209,
   "wall_median": 24.2,
   "thinking_share": 0.0,
   "vs_control": null,
   "lines_vs_control": null,
   "cost_vs_control": null,
   "wall_vs_control": null
  },
  {
   "model": "gpt-6-luna",
   "task": "T1",
   "skill": "karpathy",
   "n": 3,
   "passed": 3,
   "median": 692,
   "min": 582,
   "max": 773,
   "lines_median": 5,
   "lines_min": 5,
   "lines_max": 7,
   "cost_median": 0.003567,
   "wall_median": 22.6,
   "thinking_share": 0.0,
   "vs_control": -1.0014306151645225,
   "lines_vs_control": 0.0,
   "cost_vs_control": 11.156123402929264,
   "wall_vs_control": -6.611570247933873
  },
  {
   "model": "gpt-6-luna",
   "task": "T1",
   "skill": "placebo",
   "n": 3,
   "passed": 3,
   "median": 799,
   "min": 709,
   "max": 884,
   "lines_median": 5,
   "lines_min": 5,
   "lines_max": 7,
   "cost_median": 0.003751,
   "wall_median": 33.9,
   "thinking_share": 0.0,
   "vs_control": 14.306151645207432,
   "lines_vs_control": 0.0,
   "cost_vs_control": 16.889996883764404,
   "wall_vs_control": 40.08264462809916
  },
  {
   "model": "gpt-6-luna",
   "task": "T1",
   "skill": "ponytail",
   "n": 3,
   "passed": 3,
   "median": 740,
   "min": 695,
   "max": 790,
   "lines_median": 5,
   "lines_min": 5,
   "lines_max": 5,
   "cost_median": 0.003752,
   "wall_median": 32.6,
   "thinking_share": 0.0,
   "vs_control": 5.865522174535043,
   "lines_vs_control": 0.0,
   "cost_vs_control": 16.921159239638527,
   "wall_vs_control": 34.71074380165291
  },
  {
   "model": "gpt-6-luna",
   "task": "T2",
   "skill": "caveman",
   "n": 3,
   "passed": 3,
   "median": 520,
   "min": 482,
   "max": 538,
   "lines_median": 2,
   "lines_min": 2,
   "lines_max": 5,
   "cost_median": 0.003357,
   "wall_median": 19.6,
   "thinking_share": 0.0,
   "vs_control": -23.976608187134506,
   "lines_vs_control": -60.0,
   "cost_vs_control": 1.3281014186538043,
   "wall_vs_control": -16.94915254237288
  },
  {
   "model": "gpt-6-luna",
   "task": "T2",
   "skill": "control",
   "n": 3,
   "passed": 3,
   "median": 684,
   "min": 653,
   "max": 711,
   "lines_median": 5,
   "lines_min": 4,
   "lines_max": 7,
   "cost_median": 0.003313,
   "wall_median": 23.6,
   "thinking_share": 0.0,
   "vs_control": null,
   "lines_vs_control": null,
   "cost_vs_control": null,
   "wall_vs_control": null
  },
  {
   "model": "gpt-6-luna",
   "task": "T2",
   "skill": "karpathy",
   "n": 3,
   "passed": 3,
   "median": 733,
   "min": 646,
   "max": 910,
   "lines_median": 2,
   "lines_min": 2,
   "lines_max": 2,
   "cost_median": 0.003954,
   "wall_median": 22.6,
   "thinking_share": 0.0,
   "vs_control": 7.163742690058483,
   "lines_vs_control": -60.0,
   "cost_vs_control": 19.348022939933607,
   "wall_vs_control": -4.23728813559322
  },
  {
   "model": "gpt-6-luna",
   "task": "T2",
   "skill": "placebo",
   "n": 3,
   "passed": 3,
   "median": 684,
   "min": 662,
   "max": 717,
   "lines_median": 5,
   "lines_min": 4,
   "lines_max": 7,
   "cost_median": 0.003421,
   "wall_median": 26.5,
   "thinking_share": 0.0,
   "vs_control": 0.0,
   "lines_vs_control": 0.0,
   "cost_vs_control": 3.2598853003320327,
   "wall_vs_control": 12.288135593220328
  },
  {
   "model": "gpt-6-luna",
   "task": "T2",
   "skill": "ponytail",
   "n": 3,
   "passed": 3,
   "median": 744,
   "min": 710,
   "max": 744,
   "lines_median": 5,
   "lines_min": 2,
   "lines_max": 5,
   "cost_median": 0.003728,
   "wall_median": 22.8,
   "thinking_share": 0.0,
   "vs_control": 8.771929824561408,
   "lines_vs_control": 0.0,
   "cost_vs_control": 12.526411107757319,
   "wall_vs_control": -3.3898305084745783
  },
  {
   "model": "gpt-6-luna",
   "task": "T3",
   "skill": "caveman",
   "n": 3,
   "passed": 3,
   "median": 119,
   "min": 107,
   "max": 131,
   "lines_median": 0,
   "lines_min": 0,
   "lines_max": 0,
   "cost_median": 0.001974,
   "wall_median": 7.9,
   "thinking_share": 0.0,
   "vs_control": -33.88888888888889,
   "lines_vs_control": null,
   "cost_vs_control": 6.129032258064515,
   "wall_vs_control": -8.139534883720923
  },
  {
   "model": "gpt-6-luna",
   "task": "T3",
   "skill": "control",
   "n": 3,
   "passed": 2,
   "median": 180,
   "min": 47,
   "max": 181,
   "lines_median": 0,
   "lines_min": 0,
   "lines_max": 0,
   "cost_median": 0.00186,
   "wall_median": 8.6,
   "thinking_share": 0.0,
   "vs_control": null,
   "lines_vs_control": null,
   "cost_vs_control": null,
   "wall_vs_control": null
  },
  {
   "model": "gpt-6-luna",
   "task": "T3",
   "skill": "karpathy",
   "n": 3,
   "passed": 1,
   "median": 60,
   "min": 49,
   "max": 177,
   "lines_median": 0,
   "lines_min": 0,
   "lines_max": 0,
   "cost_median": 0.001786,
   "wall_median": 3.9,
   "thinking_share": 0.0,
   "vs_control": -66.66666666666667,
   "lines_vs_control": null,
   "cost_vs_control": -3.978494623655915,
   "wall_vs_control": -54.65116279069767
  },
  {
   "model": "gpt-6-luna",
   "task": "T3",
   "skill": "placebo",
   "n": 3,
   "passed": 0,
   "median": 32,
   "min": 31,
   "max": 54,
   "lines_median": 0,
   "lines_min": 0,
   "lines_max": 0,
   "cost_median": 0.001419,
   "wall_median": 6.3,
   "thinking_share": 0.0,
   "vs_control": -82.22222222222221,
   "lines_vs_control": null,
   "cost_vs_control": -23.709677419354847,
   "wall_vs_control": -26.74418604651163
  },
  {
   "model": "gpt-6-luna",
   "task": "T3",
   "skill": "ponytail",
   "n": 3,
   "passed": 0,
   "median": 48,
   "min": 48,
   "max": 51,
   "lines_median": 0,
   "lines_min": 0,
   "lines_max": 0,
   "cost_median": 0.001442,
   "wall_median": 4.7,
   "thinking_share": 0.0,
   "vs_control": -73.33333333333334,
   "lines_vs_control": null,
   "cost_vs_control": -22.473118279569903,
   "wall_vs_control": -45.34883720930232
  },
  {
   "model": "gpt-6-luna",
   "task": "T4",
   "skill": "caveman",
   "n": 3,
   "passed": 3,
   "median": 929,
   "min": 863,
   "max": 1215,
   "lines_median": 55,
   "lines_min": 50,
   "lines_max": 63,
   "cost_median": 0.004053,
   "wall_median": 29.4,
   "thinking_share": 0.0,
   "vs_control": -26.67719021310182,
   "lines_vs_control": -1.7857142857142905,
   "cost_vs_control": -0.4176904176904084,
   "wall_vs_control": -28.117359413202937
  },
  {
   "model": "gpt-6-luna",
   "task": "T4",
   "skill": "control",
   "n": 3,
   "passed": 3,
   "median": 1267,
   "min": 1115,
   "max": 1357,
   "lines_median": 56,
   "lines_min": 54,
   "lines_max": 59,
   "cost_median": 0.00407,
   "wall_median": 40.9,
   "thinking_share": 0.0,
   "vs_control": null,
   "lines_vs_control": null,
   "cost_vs_control": null,
   "wall_vs_control": null
  },
  {
   "model": "gpt-6-luna",
   "task": "T4",
   "skill": "karpathy",
   "n": 3,
   "passed": 3,
   "median": 1070,
   "min": 1054,
   "max": 1138,
   "lines_median": 46,
   "lines_min": 44,
   "lines_max": 47,
   "cost_median": 0.003605,
   "wall_median": 29.2,
   "thinking_share": 0.0,
   "vs_control": -15.54853985793212,
   "lines_vs_control": -17.85714285714286,
   "cost_vs_control": -11.425061425061422,
   "wall_vs_control": -28.606356968215163
  },
  {
   "model": "gpt-6-luna",
   "task": "T4",
   "skill": "placebo",
   "n": 3,
   "passed": 3,
   "median": 1336,
   "min": 1099,
   "max": 1371,
   "lines_median": 46,
   "lines_min": 44,
   "lines_max": 50,
   "cost_median": 0.004548,
   "wall_median": 37.2,
   "thinking_share": 0.0,
   "vs_control": 5.445935280189418,
   "lines_vs_control": -17.85714285714286,
   "cost_vs_control": 11.744471744471753,
   "wall_vs_control": -9.046454767726154
  },
  {
   "model": "gpt-6-luna",
   "task": "T4",
   "skill": "ponytail",
   "n": 3,
   "passed": 3,
   "median": 1170,
   "min": 1145,
   "max": 1264,
   "lines_median": 50,
   "lines_min": 44,
   "lines_max": 51,
   "cost_median": 0.004676,
   "wall_median": 36.2,
   "thinking_share": 0.0,
   "vs_control": -7.655880031570639,
   "lines_vs_control": -10.71428571428571,
   "cost_vs_control": 14.889434889434883,
   "wall_vs_control": -11.491442542787278
  },
  {
   "model": "gpt-6.1-sol",
   "task": "T1",
   "skill": "caveman",
   "n": 3,
   "passed": 3,
   "median": 542,
   "min": 398,
   "max": 555,
   "lines_median": 6,
   "lines_min": 6,
   "lines_max": 6,
   "cost_median": 0.053026,
   "wall_median": 27.5,
   "thinking_share": 0.0,
   "vs_control": -18.983557548579967,
   "lines_vs_control": -25.0,
   "cost_vs_control": 7.675750314746366,
   "wall_vs_control": -8.637873754152825
  },
  {
   "model": "gpt-6.1-sol",
   "task": "T1",
   "skill": "control",
   "n": 3,
   "passed": 3,
   "median": 669,
   "min": 569,
   "max": 706,
   "lines_median": 8,
   "lines_min": 6,
   "lines_max": 8,
   "cost_median": 0.049246,
   "wall_median": 30.1,
   "thinking_share": 0.019830028328611898,
   "vs_control": null,
   "lines_vs_control": null,
   "cost_vs_control": null,
   "wall_vs_control": null
  },
  {
   "model": "gpt-6.1-sol",
   "task": "T1",
   "skill": "karpathy",
   "n": 3,
   "passed": 3,
   "median": 571,
   "min": 556,
   "max": 620,
   "lines_median": 5,
   "lines_min": 5,
   "lines_max": 5,
   "cost_median": 0.053676,
   "wall_median": 28.2,
   "thinking_share": 0.017741935483870968,
   "vs_control": -14.648729446935727,
   "lines_vs_control": -37.5,
   "cost_vs_control": 8.99565446939854,
   "wall_vs_control": -6.312292358803995
  },
  {
   "model": "gpt-6.1-sol",
   "task": "T1",
   "skill": "placebo",
   "n": 3,
   "passed": 3,
   "median": 561,
   "min": 443,
   "max": 634,
   "lines_median": 6,
   "lines_min": 6,
   "lines_max": 6,
   "cost_median": 0.05108,
   "wall_median": 27.4,
   "thinking_share": 0.0,
   "vs_control": -16.143497757847534,
   "lines_vs_control": -25.0,
   "cost_vs_control": 3.7241603378954657,
   "wall_vs_control": -8.9700996677741
  },
  {
   "model": "gpt-6.1-sol",
   "task": "T1",
   "skill": "ponytail",
   "n": 3,
   "passed": 3,
   "median": 615,
   "min": 536,
   "max": 658,
   "lines_median": 5,
   "lines_min": 5,
   "lines_max": 5,
   "cost_median": 0.052666,
   "wall_median": 28.2,
   "thinking_share": 0.04390243902439024,
   "vs_control": -8.071748878923767,
   "lines_vs_control": -37.5,
   "cost_vs_control": 6.94472647524671,
   "wall_vs_control": -6.312292358803995
  },
  {
   "model": "gpt-6.1-sol",
   "task": "T2",
   "skill": "caveman",
   "n": 3,
   "passed": 3,
   "median": 471,
   "min": 381,
   "max": 486,
   "lines_median": 2,
   "lines_min": 2,
   "lines_max": 2,
   "cost_median": 0.056457,
   "wall_median": 26.0,
   "thinking_share": 0.03909465020576132,
   "vs_control": -12.290502793296088,
   "lines_vs_control": 0.0,
   "cost_vs_control": 11.612597117608669,
   "wall_vs_control": -1.1406844106463865
  },
  {
   "model": "gpt-6.1-sol",
   "task": "T2",
   "skill": "control",
   "n": 3,
   "passed": 3,
   "median": 537,
   "min": 498,
   "max": 652,
   "lines_median": 2,
   "lines_min": 2,
   "lines_max": 5,
   "cost_median": 0.050583,
   "wall_median": 26.3,
   "thinking_share": 0.07262569832402235,
   "vs_control": null,
   "lines_vs_control": null,
   "cost_vs_control": null,
   "wall_vs_control": null
  },
  {
   "model": "gpt-6.1-sol",
   "task": "T2",
   "skill": "karpathy",
   "n": 3,
   "passed": 3,
   "median": 580,
   "min": 561,
   "max": 625,
   "lines_median": 2,
   "lines_min": 2,
   "lines_max": 2,
   "cost_median": 0.052628,
   "wall_median": 27.0,
   "thinking_share": 0.0256,
   "vs_control": 8.007448789571692,
   "lines_vs_control": 0.0,
   "cost_vs_control": 4.042860249490943,
   "wall_vs_control": 2.6615969581748944
  },
  {
   "model": "gpt-6.1-sol",
   "task": "T2",
   "skill": "placebo",
   "n": 3,
   "passed": 3,
   "median": 597,
   "min": 583,
   "max": 630,
   "lines_median": 5,
   "lines_min": 2,
   "lines_max": 5,
   "cost_median": 0.053436,
   "wall_median": 28.4,
   "thinking_share": 0.08253968253968254,
   "vs_control": 11.17318435754191,
   "lines_vs_control": 150.0,
   "cost_vs_control": 5.640234861514726,
   "wall_vs_control": 7.984790874524705
  },
  {
   "model": "gpt-6.1-sol",
   "task": "T2",
   "skill": "ponytail",
   "n": 3,
   "passed": 3,
   "median": 555,
   "min": 542,
   "max": 604,
   "lines_median": 2,
   "lines_min": 2,
   "lines_max": 2,
   "cost_median": 0.051661,
   "wall_median": 27.3,
   "thinking_share": 0.12748344370860928,
   "vs_control": 3.3519553072625774,
   "lines_vs_control": 0.0,
   "cost_vs_control": 2.13115078188324,
   "wall_vs_control": 3.802281368821303
  },
  {
   "model": "gpt-6.1-sol",
   "task": "T3",
   "skill": "caveman",
   "n": 3,
   "passed": 3,
   "median": 184,
   "min": 173,
   "max": 186,
   "lines_median": 0,
   "lines_min": 0,
   "lines_max": 0,
   "cost_median": 0.038405,
   "wall_median": 10.8,
   "thinking_share": 0.0,
   "vs_control": -35.66433566433567,
   "lines_vs_control": null,
   "cost_vs_control": 1.5897788593799644,
   "wall_vs_control": -33.33333333333333
  },
  {
   "model": "gpt-6.1-sol",
   "task": "T3",
   "skill": "control",
   "n": 3,
   "passed": 3,
   "median": 286,
   "min": 232,
   "max": 302,
   "lines_median": 0,
   "lines_min": 0,
   "lines_max": 0,
   "cost_median": 0.037804,
   "wall_median": 16.2,
   "thinking_share": 0.0,
   "vs_control": null,
   "lines_vs_control": null,
   "cost_vs_control": null,
   "wall_vs_control": null
  },
  {
   "model": "gpt-6.1-sol",
   "task": "T3",
   "skill": "karpathy",
   "n": 3,
   "passed": 3,
   "median": 253,
   "min": 237,
   "max": 262,
   "lines_median": 0,
   "lines_min": 0,
   "lines_max": 0,
   "cost_median": 0.036686,
   "wall_median": 12.8,
   "thinking_share": 0.0,
   "vs_control": -11.538461538461542,
   "lines_vs_control": null,
   "cost_vs_control": -2.9573590096285907,
   "wall_vs_control": -20.987654320987648
  },
  {
   "model": "gpt-6.1-sol",
   "task": "T3",
   "skill": "placebo",
   "n": 3,
   "passed": 3,
   "median": 225,
   "min": 216,
   "max": 226,
   "lines_median": 0,
   "lines_min": 0,
   "lines_max": 0,
   "cost_median": 0.036227,
   "wall_median": 12.8,
   "thinking_share": 0.0,
   "vs_control": -21.328671328671334,
   "lines_vs_control": null,
   "cost_vs_control": -4.171516241667539,
   "wall_vs_control": -20.987654320987648
  },
  {
   "model": "gpt-6.1-sol",
   "task": "T3",
   "skill": "ponytail",
   "n": 3,
   "passed": 3,
   "median": 211,
   "min": 211,
   "max": 223,
   "lines_median": 0,
   "lines_min": 0,
   "lines_max": 0,
   "cost_median": 0.036285,
   "wall_median": 11.3,
   "thinking_share": 0.0,
   "vs_control": -26.22377622377622,
   "lines_vs_control": null,
   "cost_vs_control": -4.018093323457839,
   "wall_vs_control": -30.246913580246904
  },
  {
   "model": "gpt-6.1-sol",
   "task": "T4",
   "skill": "caveman",
   "n": 3,
   "passed": 3,
   "median": 936,
   "min": 896,
   "max": 1082,
   "lines_median": 57,
   "lines_min": 51,
   "lines_max": 60,
   "cost_median": 0.06662,
   "wall_median": 39.0,
   "thinking_share": 0.0,
   "vs_control": -8.771929824561408,
   "lines_vs_control": 13.99999999999999,
   "cost_vs_control": 17.340378687802737,
   "wall_vs_control": -4.176904176904184
  },
  {
   "model": "gpt-6.1-sol",
   "task": "T4",
   "skill": "control",
   "n": 3,
   "passed": 3,
   "median": 1026,
   "min": 978,
   "max": 1077,
   "lines_median": 50,
   "lines_min": 50,
   "lines_max": 54,
   "cost_median": 0.056775,
   "wall_median": 40.7,
   "thinking_share": 0.016713091922005572,
   "vs_control": null,
   "lines_vs_control": null,
   "cost_vs_control": null,
   "wall_vs_control": null
  },
  {
   "model": "gpt-6.1-sol",
   "task": "T4",
   "skill": "karpathy",
   "n": 3,
   "passed": 3,
   "median": 1042,
   "min": 1035,
   "max": 1060,
   "lines_median": 46,
   "lines_min": 45,
   "lines_max": 47,
   "cost_median": 0.063276,
   "wall_median": 43.6,
   "thinking_share": 0.08113207547169811,
   "vs_control": 1.5594541910331383,
   "lines_vs_control": -7.9999999999999964,
   "cost_vs_control": 11.450462351387047,
   "wall_vs_control": 7.125307125307123
  },
  {
   "model": "gpt-6.1-sol",
   "task": "T4",
   "skill": "placebo",
   "n": 3,
   "passed": 3,
   "median": 1168,
   "min": 1144,
   "max": 1318,
   "lines_median": 66,
   "lines_min": 66,
   "lines_max": 67,
   "cost_median": 0.064046,
   "wall_median": 44.4,
   "thinking_share": 0.026541095890410957,
   "vs_control": 13.840155945419097,
   "lines_vs_control": 32.00000000000001,
   "cost_vs_control": 12.806693086745934,
   "wall_vs_control": 9.090909090909083
  },
  {
   "model": "gpt-6.1-sol",
   "task": "T4",
   "skill": "ponytail",
   "n": 3,
   "passed": 3,
   "median": 1124,
   "min": 1044,
   "max": 1180,
   "lines_median": 43,
   "lines_min": 41,
   "lines_max": 43,
   "cost_median": 0.062048,
   "wall_median": 44.4,
   "thinking_share": 0.18861209964412812,
   "vs_control": 9.551656920077978,
   "lines_vs_control": -14.000000000000002,
   "cost_vs_control": 9.287538529282259,
   "wall_vs_control": 9.090909090909083
  }
 ],
 "runs": [
  {
   "run": "claude-opus-5-5__caveman__T1__r1__4ea6f0",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T1",
   "skill": "caveman",
   "rep": 1,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 803,
   "input": 8,
   "cache_read": 73550,
   "cache_write": 15201,
   "warmup": null,
   "cache_write_1h": 15201,
   "cost_usd": 0.15241,
   "wall_s": 11.6,
   "turns": 5,
   "prompt_last": 25321,
   "lines_added": 6,
   "lines_deleted": 0,
   "lines": 6,
   "files": 1,
   "thinking": 35,
   "commands": null,
   "heldout": true,
   "api_cost": 0.15241,
   "light": true
  },
  {
   "run": "claude-opus-5-5__caveman__T1__r2__a6eb01",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T1",
   "skill": "caveman",
   "rep": 2,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 619,
   "input": 8,
   "cache_read": 98736,
   "cache_write": 20663,
   "warmup": null,
   "cache_write_1h": 20663,
   "cost_usd": 0.19746319999999998,
   "wall_s": 9.3,
   "turns": 4,
   "prompt_last": 30783,
   "lines_added": 6,
   "lines_deleted": 0,
   "lines": 6,
   "files": 1,
   "thinking": 20,
   "commands": null,
   "heldout": true,
   "api_cost": 0.197463
  },
  {
   "run": "claude-opus-5-5__caveman__T1__r3__7c844d",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T1",
   "skill": "caveman",
   "rep": 3,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 570,
   "input": 8,
   "cache_read": 101380,
   "cache_write": 23294,
   "warmup": null,
   "cache_write_1h": 23294,
   "cost_usd": 0.21805999999999998,
   "wall_s": 9.3,
   "turns": 4,
   "prompt_last": 33414,
   "lines_added": 6,
   "lines_deleted": 0,
   "lines": 6,
   "files": 1,
   "thinking": 0,
   "commands": null,
   "heldout": true,
   "api_cost": 0.21806
  },
  {
   "run": "claude-opus-5-5__control__T1__r1__1b52c4",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T1",
   "skill": "control",
   "rep": 1,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 574,
   "input": 8,
   "cache_read": 90745,
   "cache_write": 17986,
   "warmup": null,
   "cache_write_1h": 17986,
   "cost_usd": 0.17354899999999998,
   "wall_s": 8.7,
   "turns": 4,
   "prompt_last": 28106,
   "lines_added": 6,
   "lines_deleted": 0,
   "lines": 6,
   "files": 1,
   "thinking": 0,
   "commands": null,
   "heldout": true,
   "api_cost": 0.173549
  },
  {
   "run": "claude-opus-5-5__control__T1__r2__557bd8",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T1",
   "skill": "control",
   "rep": 2,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 693,
   "input": 8,
   "cache_read": 90964,
   "cache_write": 18212,
   "warmup": null,
   "cache_write_1h": 18212,
   "cost_usd": 0.17778080000000002,
   "wall_s": 11.1,
   "turns": 4,
   "prompt_last": 28332,
   "lines_added": 6,
   "lines_deleted": 0,
   "lines": 6,
   "files": 1,
   "thinking": 0,
   "commands": null,
   "heldout": true,
   "api_cost": 0.177781
  },
  {
   "run": "claude-opus-5-5__control__T1__r3__a04c0c",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T1",
   "skill": "control",
   "rep": 3,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 652,
   "input": 8,
   "cache_read": 90798,
   "cache_write": 18121,
   "warmup": null,
   "cache_write_1h": 18121,
   "cost_usd": 0.17619959999999998,
   "wall_s": 8.8,
   "turns": 4,
   "prompt_last": 28241,
   "lines_added": 6,
   "lines_deleted": 0,
   "lines": 6,
   "files": 1,
   "thinking": 0,
   "commands": null,
   "heldout": true,
   "api_cost": 0.1762
  },
  {
   "run": "claude-opus-5-5__karpathy__T1__r1__4417de",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T1",
   "skill": "karpathy",
   "rep": 1,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 751,
   "input": 10,
   "cache_read": 123075,
   "cache_write": 19294,
   "warmup": null,
   "cache_write_1h": 19294,
   "cost_usd": 0.194027,
   "wall_s": 13.5,
   "turns": 5,
   "prompt_last": 29414,
   "lines_added": 6,
   "lines_deleted": 0,
   "lines": 6,
   "files": 1,
   "thinking": 51,
   "commands": null,
   "heldout": true,
   "api_cost": 0.194027
  },
  {
   "run": "claude-opus-5-5__karpathy__T1__r2__f35c7f",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T1",
   "skill": "karpathy",
   "rep": 2,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 772,
   "input": 8,
   "cache_read": 93857,
   "cache_write": 19332,
   "warmup": null,
   "cache_write_1h": 19332,
   "cost_usd": 0.1888994,
   "wall_s": 11.3,
   "turns": 5,
   "prompt_last": 29452,
   "lines_added": 6,
   "lines_deleted": 0,
   "lines": 6,
   "files": 1,
   "thinking": 28,
   "commands": null,
   "heldout": true,
   "api_cost": 0.188899
  },
  {
   "run": "claude-opus-5-5__karpathy__T1__r3__662852",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T1",
   "skill": "karpathy",
   "rep": 3,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 786,
   "input": 8,
   "cache_read": 93812,
   "cache_write": 19280,
   "warmup": null,
   "cache_write_1h": 19280,
   "cost_usd": 0.18875440000000002,
   "wall_s": 9.9,
   "turns": 5,
   "prompt_last": 29400,
   "lines_added": 6,
   "lines_deleted": 0,
   "lines": 6,
   "files": 1,
   "thinking": 22,
   "commands": null,
   "heldout": true,
   "api_cost": 0.188754
  },
  {
   "run": "claude-opus-5-5__placebo__T1__r1__c7aa19",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T1",
   "skill": "placebo",
   "rep": 1,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 896,
   "input": 10,
   "cache_read": 123492,
   "cache_write": 19763,
   "warmup": null,
   "cache_write_1h": 19763,
   "cost_usd": 0.2007624,
   "wall_s": 13.1,
   "turns": 5,
   "prompt_last": 29883,
   "lines_added": 6,
   "lines_deleted": 0,
   "lines": 6,
   "files": 1,
   "thinking": 31,
   "commands": null,
   "heldout": true,
   "api_cost": 0.200762
  },
  {
   "run": "claude-opus-5-5__placebo__T1__r2__bde323",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T1",
   "skill": "placebo",
   "rep": 2,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 956,
   "input": 10,
   "cache_read": 123135,
   "cache_write": 20106,
   "warmup": null,
   "cache_write_1h": 20106,
   "cost_usd": 0.204635,
   "wall_s": 13.9,
   "turns": 6,
   "prompt_last": 30226,
   "lines_added": 7,
   "lines_deleted": 1,
   "lines": 8,
   "files": 2,
   "thinking": 39,
   "commands": null,
   "heldout": true,
   "api_cost": 0.204635
  },
  {
   "run": "claude-opus-5-5__placebo__T1__r3__8d388f",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T1",
   "skill": "placebo",
   "rep": 3,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 933,
   "input": 8,
   "cache_read": 94203,
   "cache_write": 20201,
   "warmup": null,
   "cache_write_1h": 20201,
   "cost_usd": 0.19914059999999997,
   "wall_s": 11.2,
   "turns": 5,
   "prompt_last": 30321,
   "lines_added": 7,
   "lines_deleted": 1,
   "lines": 8,
   "files": 2,
   "thinking": 38,
   "commands": null,
   "heldout": true,
   "api_cost": 0.199141
  },
  {
   "run": "claude-opus-5-5__ponytail__T1__r1__cd9241",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T1",
   "skill": "ponytail",
   "rep": 1,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 689,
   "input": 8,
   "cache_read": 93994,
   "cache_write": 19151,
   "warmup": null,
   "cache_write_1h": 19151,
   "cost_usd": 0.1858188,
   "wall_s": 9.1,
   "turns": 4,
   "prompt_last": 29271,
   "lines_added": 6,
   "lines_deleted": 0,
   "lines": 6,
   "files": 1,
   "thinking": 0,
   "commands": null,
   "heldout": true,
   "api_cost": 0.185819
  },
  {
   "run": "claude-opus-5-5__ponytail__T1__r2__c7cef3",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T1",
   "skill": "ponytail",
   "rep": 2,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 616,
   "input": 8,
   "cache_read": 93893,
   "cache_write": 19075,
   "warmup": null,
   "cache_write_1h": 19075,
   "cost_usd": 0.18373060000000002,
   "wall_s": 8.8,
   "turns": 4,
   "prompt_last": 29195,
   "lines_added": 6,
   "lines_deleted": 0,
   "lines": 6,
   "files": 1,
   "thinking": 33,
   "commands": null,
   "heldout": true,
   "api_cost": 0.183731
  },
  {
   "run": "claude-opus-5-5__ponytail__T1__r3__29c563",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T1",
   "skill": "ponytail",
   "rep": 3,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 651,
   "input": 8,
   "cache_read": 93868,
   "cache_write": 19119,
   "warmup": null,
   "cache_write_1h": 19119,
   "cost_usd": 0.18477760000000001,
   "wall_s": 11.1,
   "turns": 4,
   "prompt_last": 29239,
   "lines_added": 6,
   "lines_deleted": 0,
   "lines": 6,
   "files": 1,
   "thinking": 23,
   "commands": null,
   "heldout": true,
   "api_cost": 0.184778
  },
  {
   "run": "claude-opus-5-5__caveman__T2__r1__1175e0",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T2",
   "skill": "caveman",
   "rep": 1,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 994,
   "input": 10,
   "cache_read": 134611,
   "cache_write": 23906,
   "warmup": null,
   "cache_write_1h": 23906,
   "cost_usd": 0.2380902,
   "wall_s": 12.1,
   "turns": 6,
   "prompt_last": 34026,
   "lines_added": 4,
   "lines_deleted": 3,
   "lines": 7,
   "files": 1,
   "thinking": 72,
   "commands": null,
   "heldout": true,
   "api_cost": 0.23809
  },
  {
   "run": "claude-opus-5-5__caveman__T2__r2__c6f272",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T2",
   "skill": "caveman",
   "rep": 2,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1041,
   "input": 10,
   "cache_read": 134372,
   "cache_write": 23667,
   "warmup": null,
   "cache_write_1h": 23667,
   "cost_usd": 0.23707040000000001,
   "wall_s": 12.5,
   "turns": 6,
   "prompt_last": 33787,
   "lines_added": 4,
   "lines_deleted": 3,
   "lines": 7,
   "files": 1,
   "thinking": 106,
   "commands": null,
   "heldout": true,
   "api_cost": 0.23707
  },
  {
   "run": "claude-opus-5-5__caveman__T2__r3__4982e3",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T2",
   "skill": "caveman",
   "rep": 3,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 993,
   "input": 10,
   "cache_read": 134592,
   "cache_write": 23874,
   "warmup": null,
   "cache_write_1h": 23874,
   "cost_usd": 0.23781039999999998,
   "wall_s": 14.5,
   "turns": 6,
   "prompt_last": 33994,
   "lines_added": 4,
   "lines_deleted": 3,
   "lines": 7,
   "files": 1,
   "thinking": 69,
   "commands": null,
   "heldout": true,
   "api_cost": 0.23781
  },
  {
   "run": "claude-opus-5-5__control__T2__r1__ac4a90",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T2",
   "skill": "control",
   "rep": 1,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1109,
   "input": 8,
   "cache_read": 91022,
   "cache_write": 18563,
   "warmup": null,
   "cache_write_1h": 18563,
   "cost_usd": 0.18892040000000002,
   "wall_s": 12.5,
   "turns": 4,
   "prompt_last": 28683,
   "lines_added": 8,
   "lines_deleted": 4,
   "lines": 12,
   "files": 1,
   "thinking": 97,
   "commands": null,
   "heldout": true,
   "api_cost": 0.18892
  },
  {
   "run": "claude-opus-5-5__control__T2__r2__14a28a",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T2",
   "skill": "control",
   "rep": 2,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1139,
   "input": 8,
   "cache_read": 91003,
   "cache_write": 18635,
   "warmup": null,
   "cache_write_1h": 18635,
   "cost_usd": 0.19009259999999997,
   "wall_s": 12.5,
   "turns": 4,
   "prompt_last": 28755,
   "lines_added": 9,
   "lines_deleted": 4,
   "lines": 13,
   "files": 1,
   "thinking": 148,
   "commands": null,
   "heldout": true,
   "api_cost": 0.190093
  },
  {
   "run": "claude-opus-5-5__control__T2__r3__2301b6",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T2",
   "skill": "control",
   "rep": 3,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1247,
   "input": 10,
   "cache_read": 118876,
   "cache_write": 18729,
   "warmup": null,
   "cache_write_1h": 18729,
   "cost_usd": 0.19858720000000002,
   "wall_s": 16.1,
   "turns": 6,
   "prompt_last": 28849,
   "lines_added": 4,
   "lines_deleted": 3,
   "lines": 7,
   "files": 1,
   "thinking": 99,
   "commands": null,
   "heldout": true,
   "api_cost": 0.198587
  },
  {
   "run": "claude-opus-5-5__karpathy__T2__r1__b2b420",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T2",
   "skill": "karpathy",
   "rep": 1,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1053,
   "input": 10,
   "cache_read": 123715,
   "cache_write": 20559,
   "warmup": null,
   "cache_write_1h": 20559,
   "cost_usd": 0.210315,
   "wall_s": 17.8,
   "turns": 6,
   "prompt_last": 30679,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 178,
   "commands": null,
   "heldout": true,
   "api_cost": 0.210315
  },
  {
   "run": "claude-opus-5-5__karpathy__T2__r2__a2f586",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T2",
   "skill": "karpathy",
   "rep": 2,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1045,
   "input": 8,
   "cache_read": 93678,
   "cache_write": 19158,
   "warmup": null,
   "cache_write_1h": 19158,
   "cost_usd": 0.1929316,
   "wall_s": 12.4,
   "turns": 5,
   "prompt_last": 29278,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 126,
   "commands": null,
   "heldout": true,
   "api_cost": 0.192932
  },
  {
   "run": "claude-opus-5-5__karpathy__T2__r3__c631b6",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T2",
   "skill": "karpathy",
   "rep": 3,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 903,
   "input": 8,
   "cache_read": 93599,
   "cache_write": 19171,
   "warmup": null,
   "cache_write_1h": 19171,
   "cost_usd": 0.19017980000000004,
   "wall_s": 13.0,
   "turns": 5,
   "prompt_last": 29291,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 128,
   "commands": null,
   "heldout": true,
   "api_cost": 0.19018
  },
  {
   "run": "claude-opus-5-5__placebo__T2__r1__fcc4ef",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T2",
   "skill": "placebo",
   "rep": 1,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1252,
   "input": 8,
   "cache_read": 95884,
   "cache_write": 21020,
   "warmup": null,
   "cache_write_1h": 21020,
   "cost_usd": 0.2124088,
   "wall_s": 20.4,
   "turns": 4,
   "prompt_last": 31140,
   "lines_added": 4,
   "lines_deleted": 3,
   "lines": 7,
   "files": 1,
   "thinking": 235,
   "commands": null,
   "heldout": true,
   "api_cost": 0.212409
  },
  {
   "run": "claude-opus-5-5__placebo__T2__r2__d1b5f7",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T2",
   "skill": "placebo",
   "rep": 2,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1630,
   "input": 10,
   "cache_read": 123571,
   "cache_write": 20384,
   "warmup": null,
   "cache_write_1h": 20384,
   "cost_usd": 0.22042620000000004,
   "wall_s": 16.6,
   "turns": 5,
   "prompt_last": 30504,
   "lines_added": 13,
   "lines_deleted": 4,
   "lines": 17,
   "files": 1,
   "thinking": 292,
   "commands": null,
   "heldout": true,
   "api_cost": 0.220426
  },
  {
   "run": "claude-opus-5-5__placebo__T2__r3__385a97",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T2",
   "skill": "placebo",
   "rep": 3,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1333,
   "input": 10,
   "cache_read": 123187,
   "cache_write": 20173,
   "warmup": null,
   "cache_write_1h": 20173,
   "cost_usd": 0.2127214,
   "wall_s": 14.8,
   "turns": 6,
   "prompt_last": 30293,
   "lines_added": 4,
   "lines_deleted": 3,
   "lines": 7,
   "files": 1,
   "thinking": 115,
   "commands": null,
   "heldout": true,
   "api_cost": 0.212721
  },
  {
   "run": "claude-opus-5-5__ponytail__T2__r1__110faa",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T2",
   "skill": "ponytail",
   "rep": 1,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1126,
   "input": 8,
   "cache_read": 93705,
   "cache_write": 19287,
   "warmup": null,
   "cache_write_1h": 19287,
   "cost_usd": 0.195589,
   "wall_s": 14.6,
   "turns": 4,
   "prompt_last": 29407,
   "lines_added": 2,
   "lines_deleted": 2,
   "lines": 4,
   "files": 1,
   "thinking": 235,
   "commands": null,
   "heldout": true,
   "api_cost": 0.195589
  },
  {
   "run": "claude-opus-5-5__ponytail__T2__r2__615f5b",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T2",
   "skill": "ponytail",
   "rep": 2,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1241,
   "input": 10,
   "cache_read": 123102,
   "cache_write": 20019,
   "warmup": null,
   "cache_write_1h": 20019,
   "cost_usd": 0.20963240000000002,
   "wall_s": 16.6,
   "turns": 5,
   "prompt_last": 30139,
   "lines_added": 2,
   "lines_deleted": 2,
   "lines": 4,
   "files": 1,
   "thinking": 357,
   "commands": null,
   "heldout": true,
   "api_cost": 0.209632
  },
  {
   "run": "claude-opus-5-5__ponytail__T2__r3__07a8e5",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T2",
   "skill": "ponytail",
   "rep": 3,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1417,
   "input": 10,
   "cache_read": 124138,
   "cache_write": 20865,
   "warmup": null,
   "cache_write_1h": 20865,
   "cost_usd": 0.22012760000000003,
   "wall_s": 17.2,
   "turns": 5,
   "prompt_last": 30985,
   "lines_added": 2,
   "lines_deleted": 2,
   "lines": 4,
   "files": 1,
   "thinking": 393,
   "commands": null,
   "heldout": true,
   "api_cost": 0.220128
  },
  {
   "run": "claude-opus-5-5__caveman__T3__r1__ee998f",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "caveman",
   "rep": 1,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 990,
   "input": 4,
   "cache_read": 29740,
   "cache_write": 9289,
   "warmup": null,
   "cache_write_1h": 9289,
   "cost_usd": 0.100076,
   "wall_s": 12.1,
   "turns": 2,
   "prompt_last": 19918,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 183,
   "commands": null,
   "answer": "**Monthly totals**\n\n`monthly_report(entries, year, month)` at `ledger/core.py:51` does three steps:\n\n1. `entries_in_month` keeps entries where `e.day.year == year` and `e.day.month == month`.\n2. `total` sums `amount` over those entries.\n3. `sum_by_category` builds a dict of category to summed amount.\n\nIt returns `{\"count\", \"total\", \"by_category\"}`.\n\n- **Empty month:** no match gives `count` 0, `total` 0 (int, not float) and `by_category` `{}`.\n- **Float amounts:** amounts are floats with no rounding, so sums can drift (`0.1 + 0.2`).\n- **Categories:** `add_entry` lowercases the category, so \"Food\" and \"food\" merge.\n- **No validation:** `year` and `month` are not checked. Month 13 gives an empty report, not an error.\n\n**Invalid date string**\n\nDates are parsed only in `add_entry`, through `parse_date` (`ledger/core.py:15`). Two failure paths, both `ValueError`:\n\n- **Wrong shape:** not exactly 3 parts split by `-`, or any part not all digits. Raises `ValueError(\"bad date: '<text>'\")`. Examples: `\"2024/01/05\"`, `\"2024-01\"`, `\"2024-Jan-05\"`, `\"\"`.\n- **Right shape, impossible date:** `date(y, m, d)` raises its own `ValueError`, for example `month must be in 1..12`. Examples: `\"2024-13-01\"`, `\"2024-02-30\"`.\n\nIn `add_entry`, the amount check runs first. With a bad amount and a bad date, the error is `amount must be positive`. The date parse runs before `entries.append`, so a bad date leaves the list unchanged. Nothing catches the exception; it propagates to the caller.\n\nTwo gaps:\n\n- **Loose format:** the docstring says YYYY-MM-DD, but digit widths are not checked. `\"2024-1-5\"` parses fine.\n- **Non-string input:** `None` or a `date` object raises `AttributeError` on `.split`, not `ValueError`.\n\nNo files changed.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.100076
  },
  {
   "run": "claude-opus-5-5__caveman__T3__r2__72b00b",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "caveman",
   "rep": 2,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 880,
   "input": 4,
   "cache_read": 29739,
   "cache_write": 9288,
   "warmup": null,
   "cache_write_1h": 9288,
   "cost_usd": 0.09786779999999999,
   "wall_s": 11.9,
   "turns": 2,
   "prompt_last": 19917,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 202,
   "commands": null,
   "answer": "**Monthly totals**\n\n`monthly_report(entries, year, month)` at `ledger/core.py:51` does three steps:\n\n1. `entries_in_month` keeps entries where `e.day.year == year` and `e.day.month == month`.\n2. `total` sums `amount` over those entries with plain `sum`.\n3. `sum_by_category` builds a dict of category to running sum, starting each category at `0.0`.\n\nReturns `{\"count\", \"total\", \"by_category\"}`. Empty month gives count `0`, total `0` (int, not `0.0`), and `{}`.\n\nAmounts are floats, no rounding anywhere. Totals can show float error such as `0.30000000000000004`.\n\nCategories are lowercased in `add_entry`, so \"Food\" and \"food\" merge into one key.\n\n**Invalid date string**\n\n`parse_date` at `ledger/core.py:15` splits on `-`. It raises `ValueError(\"bad date: '<text>'\")` unless there are exactly 3 parts and all are digits.\n\nTwo cases get past that check:\n\n- Out-of-range values such as `2024-13-01` or `2024-02-30`. `date(y, m, d)` raises `ValueError`, but with Python's own message, not `bad date: ...`.\n- Unpadded forms such as `2024-1-5`. These parse without error, though the docstring says YYYY-MM-DD only.\n\nNon-string input (`None`, a `date` object) raises `AttributeError` on `.split`, not `ValueError`.\n\nIn `add_entry`, the amount check runs first. With amount `<= 0` and a bad date, the error is `amount must be positive`. The date parse happens before `entries.append`, so a bad date leaves the list unchanged. Nothing catches the error; it propagates to the caller.\n\nNo files changed.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.097868
  },
  {
   "run": "claude-opus-5-5__caveman__T3__r3__8b785c",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "caveman",
   "rep": 3,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 907,
   "input": 4,
   "cache_read": 29738,
   "cache_write": 9287,
   "warmup": null,
   "cache_write_1h": 9287,
   "cost_usd": 0.0983996,
   "wall_s": 10.1,
   "turns": 2,
   "prompt_last": 19916,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 179,
   "commands": null,
   "answer": "**Monthly totals**\n\n`monthly_report(entries, year, month)` at `ledger/core.py:51` does three steps:\n\n1. `entries_in_month` keeps entries where `e.day.year == year` and `e.day.month == month`.\n2. `total` sums `amount` over those entries.\n3. `sum_by_category` builds a dict of category to summed amount.\n\nIt returns `{\"count\", \"total\", \"by_category\"}`.\n\n- **Category case:** `add_entry` lowercases the category, so \"Food\" and \"food\" merge into one key.\n- **Float amounts:** amounts are floats, so sums can carry rounding error (`0.1 + 0.2`).\n- **Empty month:** count is `0`, total is `0` (int, not `0.0`), and `by_category` is `{}`.\n- **Invalid month:** no check on `year` or `month`. Month 13 returns an empty report, no error.\n\n**Invalid date string**\n\n`parse_date` at `ledger/core.py:15` splits the text on `-`. It needs exactly 3 parts, all digits. Otherwise it raises `ValueError(\"bad date: '<text>'\")`.\n\n- **Out of range:** `2024-13-01` or `2024-02-30` passes the format check. `date(y, m, d)` then raises its own `ValueError`, with a different message (for example `month must be in 1..12`).\n- **No padding check:** `2024-1-5` is accepted, and so is `24-01-05` (year 24).\n- **Non-string input:** `None` or a `date` object raises `AttributeError`, not `ValueError`, because there is no `.split`.\n- **Surrounding whitespace:** `\" 2024-01-05\"` fails `isdigit` and raises the `bad date` error.\n\nIn `add_entry`, the amount check runs first. A bad amount with a bad date reports `amount must be positive`. With a valid amount, the date error propagates before `append`, so the list stays unchanged.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.0984
  },
  {
   "run": "claude-opus-5-5__control__T3__r1__bc721d",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "control",
   "rep": 1,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 1066,
   "input": 4,
   "cache_read": 27094,
   "cache_write": 6641,
   "warmup": null,
   "cache_write_1h": 6641,
   "cost_usd": 0.0798828,
   "wall_s": 12.8,
   "turns": 2,
   "prompt_last": 17270,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 179,
   "commands": null,
   "answer": "**Monthly totals**\n\n`monthly_report(entries, year, month)` in `ledger/core.py:51` does the work in three steps:\n\n1. **Filter:** `entries_in_month` keeps only the entries whose `day.year` and `day.month` both equal the requested values.\n2. **Overall total:** `total` sums the `amount` of the filtered entries.\n3. **Per-category totals:** `sum_by_category` builds a dict keyed by category, adding each amount to a running sum that starts at `0.0`.\n\nThe result is a dict with `count` (number of entries in the month), `total` and `by_category`.\n\n- **Float arithmetic:** amounts are stored as floats in dollars, so totals can show ordinary float rounding artefacts (e.g. `0.1 + 0.2`); nothing is rounded.\n- **Case-insensitive categories:** `add_entry` lowercases the category, so \"Food\" and \"food\" land in the same bucket.\n- **Empty month:** a month with no entries gives `count` 0, `total` as the integer `0` (not `0.0`) and an empty `by_category`.\n\n**Invalid date strings**\n\nDates are parsed by `parse_date` (`ledger/core.py:15`), which `add_entry` calls while building the `Entry`. An invalid date raises `ValueError`, and because that happens before `entries.append` completes, nothing is added to the ledger.\n\n- **Wrong shape:** if the string does not split on `-` into exactly three all-digit parts (e.g. `\"2024/01/05\"`, `\"2024-01\"`, `\"2024-ab-01\"`, `\"\"`), it raises `ValueError(\"bad date: '...'\")`.\n- **Right shape, impossible date:** something like `\"2024-13-01\"` or `\"2024-02-30\"` passes the shape check, and `date(y, m, d)` then raises its own `ValueError` with a different message (e.g. \"month must be in 1..12\").\n- **Non-strings:** passing `None` or an actual `date` object raises `AttributeError` (no `.split`), not `ValueError`, despite what the docstring says.\n- **Looser than the docstring:** digit widths are not checked, so `\"2024-1-5\"` is accepted as 5 January 2024.\n- **Order of checks:** `add_entry` validates the amount first, so a non-positive amount with a bad date reports \"amount must be positive\" and the date is never looked at.\n\nNo files were changed.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.079883
  },
  {
   "run": "claude-opus-5-5__control__T3__r2__7e8f4a",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "control",
   "rep": 2,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 1167,
   "input": 4,
   "cache_read": 27097,
   "cache_write": 6646,
   "warmup": null,
   "cache_write_1h": 6646,
   "cost_usd": 0.0819434,
   "wall_s": 12.3,
   "turns": 2,
   "prompt_last": 17275,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 179,
   "commands": null,
   "answer": "**Monthly totals**\n\n`monthly_report(entries, year, month)` in `ledger/core.py:51` computes the totals in three steps:\n\n1. **Filter:** `entries_in_month` keeps only the entries whose `day.year` and `day.month` both equal the requested values.\n2. **Overall total:** `total` adds up the `amount` of those entries with a plain `sum`.\n3. **Per-category totals:** `sum_by_category` walks the same entries and accumulates amounts in a dict keyed by category. Categories are lowercased in `add_entry`, so \"Food\" and \"food\" land in the same bucket.\n\nThe result is a dict with `count` (number of matching entries), `total` and `by_category`. Nothing is cached or pre-aggregated; every call rescans the full list.\n\nTwo details matter when reading the numbers:\n\n- **Float arithmetic:** Amounts are floats, so sums can show ordinary rounding artefacts (e.g. 0.1 + 0.2), and nothing is rounded to cents.\n- **Empty month:** A month with no entries returns `count` 0, `total` as the integer `0` (not `0.0`) and an empty `by_category`.\n\n**Invalid date strings**\n\nDates are only parsed in `add_entry`, via `parse_date` (`ledger/core.py:15`). A bad date fails in one of three ways:\n\n- **Wrong shape:** If the string does not split on `-` into exactly three all-digit parts, `parse_date` raises `ValueError(\"bad date: '<text>'\")`. This covers `\"2024/01/05\"`, `\"2024-01\"`, `\"2024-Jan-05\"`, an empty string, and anything with a sign or whitespace.\n- **Right shape, impossible date:** Strings such as `\"2024-13-01\"` or `\"2024-02-30\"` pass the shape check and then fail inside `date(y, m, d)`. This is still a `ValueError`, but with Python's own message (e.g. \"month must be in 1..12\") rather than the \"bad date\" one.\n- **Not a string:** Passing `None` or a `date` object raises `AttributeError` on `.split`, not `ValueError`, despite what the docstring says.\n\nIn every case the exception propagates out of `add_entry` uncaught. The `Entry` is never built, so nothing is appended and the list is unchanged.\n\n`add_entry` checks the amount first, so a call with both a non-positive amount and a bad date reports \"amount must be positive\" and never reaches the date.\n\nThe shape check is looser than the docstring's \"YYYY-MM-DD\": it does not enforce digit widths, so `\"2024-1-5\"` and `\"24-01-05\"` (year 24 AD) are accepted as valid.\n\nNo files were changed.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.081943
  },
  {
   "run": "claude-opus-5-5__control__T3__r3__1a047a",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "control",
   "rep": 3,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 1429,
   "input": 4,
   "cache_read": 27094,
   "cache_write": 6641,
   "warmup": null,
   "cache_write_1h": 6641,
   "cost_usd": 0.08714279999999999,
   "wall_s": 15.1,
   "turns": 2,
   "prompt_last": 17270,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 313,
   "commands": null,
   "answer": "I read `ledger/core.py` and changed nothing.\n\n## How monthly totals are computed\n\n`monthly_report(entries, year, month)` (`ledger/core.py:51`) does three things:\n\n1. **Filters to the month.** `entries_in_month` (line 40) keeps only entries whose `day.year` and `day.month` both equal the arguments.\n2. **Sums the amounts.** `total` (line 31) adds up `e.amount` over the filtered entries with the built-in `sum`.\n3. **Breaks the total down by category.** `sum_by_category` (line 44) builds a dict mapping each category to the running sum of its amounts.\n\nThe result is a dict with `count` (number of entries in the month), `total` and `by_category`.\n\nA few behaviours follow from this:\n\n- **Empty month:** the report is `{\"count\": 0, \"total\": 0, \"by_category\": {}}`. The total is the integer `0`, not `0.0`, because `sum` of nothing returns its integer start value.\n- **Float arithmetic:** amounts are stored as floats in dollars, so totals can show rounding artefacts such as `0.1 + 0.2` giving `0.30000000000000004`. Nothing is rounded.\n- **Category case:** categories are lower-cased when added through `add_entry`, so \"Food\" and \"food\" merge into one bucket. An `Entry` constructed directly skips that normalisation.\n- **Unchecked arguments:** `year` and `month` are not validated, so `month=13` just returns an empty report.\n\n## What happens when a date string is invalid\n\nDates are only parsed in `add_entry` (line 24), via `parse_date` (line 15). An invalid date raises `ValueError`, and nothing is appended to the list because parsing happens before the append.\n\nThere are two ways the error arises:\n\n- **Wrong shape:** if the string does not split on `-` into exactly three all-digit parts, `parse_date` raises `ValueError(\"bad date: '...'\")`. This covers `\"2024/01/05\"`, `\"2024-01\"`, `\"2024-Jan-05\"`, `\"\"`, and strings with surrounding whitespace.\n- **Right shape, impossible date:** strings like `\"2024-13-01\"` or `\"2023-02-30\"` pass the shape check and fail inside `date(y, m, d)`. That is still a `ValueError`, but with Python's own message (e.g. \"month must be in 1..12\") rather than the `bad date:` one.\n\nSome edge cases are worth knowing:\n\n- **Looser than the docstring says:** it claims YYYY-MM-DD only, but digit widths are not checked. `\"2024-1-5\"` is accepted, and `\"24-01-05\"` is accepted as year 24 AD.\n- **Amount is checked first:** `add_entry` validates the amount before the date, so a call with both a non-positive amount and a bad date reports only `\"amount must be positive\"`.\n- **Non-string input:** passing `None` or a `date` object raises `AttributeError` (no `.split`), not `ValueError`.\n- **No handling downstream:** nothing in this module catches these errors; they propagate to the caller.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.087143
  },
  {
   "run": "claude-opus-5-5__karpathy__T3__r1__46efd2",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "karpathy",
   "rep": 1,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 1176,
   "input": 4,
   "cache_read": 28112,
   "cache_write": 7662,
   "warmup": null,
   "cache_write_1h": 7662,
   "cost_usd": 0.0904544,
   "wall_s": 14.6,
   "turns": 2,
   "prompt_last": 18291,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 230,
   "commands": null,
   "answer": "**Monthly totals**\n\n`monthly_report(entries, year, month)` at `ledger/core.py:51` computes everything on demand from the entry list, in three steps:\n\n1. **Filter:** `entries_in_month` keeps the entries whose `day.year` and `day.month` both equal the requested values.\n2. **Sum:** `total` adds up `amount` across those entries.\n3. **Break down:** `sum_by_category` builds a dict mapping each category to the sum of its amounts.\n\nThe result is `{\"count\": ..., \"total\": ..., \"by_category\": ...}`.\n\nA few behaviours follow from the code:\n\n- **Empty month:** a month with no entries gives `count` 0, `total` 0 (the integer, since `sum()` of nothing is `0`) and an empty `by_category`.\n- **Float arithmetic:** amounts are plain floats, so totals can show rounding artefacts (e.g. 0.1 + 0.2), and nothing rounds to cents.\n- **Category case:** categories are lowercased in `add_entry`, so \"Food\" and \"food\" merge in the breakdown. An `Entry` constructed directly bypasses that.\n\n**Invalid date strings**\n\nDates are parsed by `parse_date` at `ledger/core.py:15`, which `add_entry` calls. It splits the text on `-` and requires exactly three parts, each made only of digits.\n\n- **Wrong shape:** inputs such as `\"2024/01/05\"`, `\"2024-01\"`, `\"2024-01-xx\"`, `\"\"`, or anything with surrounding whitespace raise `ValueError(\"bad date: '<text>'\")`.\n- **Right shape, impossible date:** inputs such as `\"2024-13-01\"` or `\"2024-02-30\"` pass the shape check and then fail inside `date(y, m, d)`. This is still a `ValueError`, but with Python's own message (e.g. \"month must be in 1..12\"), not the \"bad date\" one.\n- **Not a string:** `None` or an existing `date` object raises `AttributeError` on `.split`, not `ValueError`, despite what the docstring says.\n- **Looser than the docstring:** digit widths are not checked, so `\"2024-1-5\"` and `\"24-01-05\"` (year 24 AD) are accepted.\n\nIn `add_entry`, the `amount <= 0` check runs first, so a bad amount combined with a bad date reports the amount error. The date is parsed before the `Entry` is appended, so an invalid date leaves the list unchanged. Nothing in this file catches the exception; it propagates to the caller.\n\nNo files were changed.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.090454
  },
  {
   "run": "claude-opus-5-5__karpathy__T3__r2__a5462b",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "karpathy",
   "rep": 2,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 1420,
   "input": 4,
   "cache_read": 28111,
   "cache_write": 7661,
   "warmup": null,
   "cache_write_1h": 7661,
   "cost_usd": 0.0953262,
   "wall_s": 14.7,
   "turns": 2,
   "prompt_last": 18290,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 374,
   "commands": null,
   "answer": "I read `ledger/core.py` and changed nothing.\n\n## How monthly totals are computed\n\n`monthly_report(entries, year, month)` (line 51) does it in two steps:\n\n1. **Filter.** `entries_in_month` keeps only the entries whose `day.year` and `day.month` both equal the requested values.\n2. **Aggregate.** From that filtered list it builds a dict with three keys:\n   - `count` is the number of entries in the month.\n   - `total` is the plain sum of every entry's `amount`, via `total()`.\n   - `by_category` comes from `sum_by_category`, which keeps a running sum per category name in a dict.\n\nA few behaviours follow from this:\n\n- **No rounding.** Amounts are floats, so totals can show floating-point artefacts such as `0.1 + 0.2` giving `0.30000000000000004`.\n- **Empty months.** A month with no entries gives `count` 0, `by_category` `{}`, and `total` as the integer `0` rather than `0.0`, because `sum()` of nothing returns int 0.\n- **Category case.** Categories merge case-insensitively only because `add_entry` lowercases them on the way in. `sum_by_category` does not normalise, so an `Entry` constructed directly with `\"Food\"` would land in a separate bucket from `\"food\"`.\n\n## What happens with an invalid date string\n\nDates are only parsed in `parse_date` (line 15), which `add_entry` calls. It splits the text on `-` and handles bad input in two ways:\n\n- **Wrong shape.** If there are not exactly three parts, or any part is not all digits, it raises `ValueError(\"bad date: '<text>'\")`. This covers `\"2024/01/05\"`, `\"2024-01\"`, `\"2024-Jan-05\"`, an empty string, and anything with surrounding whitespace.\n- **Right shape, impossible date.** Something like `\"2024-02-30\"` or `\"2024-13-01\"` passes the shape check and fails inside `date(y, m, d)`. It is still a `ValueError`, but with Python's own message (e.g. \"day is out of range for month\") rather than the `bad date:` one.\n\nIn both cases the error propagates out of `add_entry` uncaught. The exception fires before `entries.append` runs, so the ledger is left unmodified.\n\nSome edge cases are worth knowing:\n\n- **Amount is checked first.** In `add_entry`, a non-positive amount combined with a bad date reports \"amount must be positive\", not the date error.\n- **Validation is looser than the docstring says.** It does not enforce digit widths, so `\"2024-1-5\"` is accepted, and `\"24-01-05\"` is accepted as year 24 AD.\n- **Non-string input is not a `ValueError`.** Passing `None` or a `date` object raises `AttributeError` (no `.split`), contrary to the docstring's \"Raises ValueError on anything else\".",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.095326
  },
  {
   "run": "claude-opus-5-5__karpathy__T3__r3__e9e031",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "karpathy",
   "rep": 3,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 1223,
   "input": 4,
   "cache_read": 28109,
   "cache_write": 7659,
   "warmup": null,
   "cache_write_1h": 7659,
   "cost_usd": 0.0913698,
   "wall_s": 12.7,
   "turns": 2,
   "prompt_last": 18288,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 109,
   "commands": null,
   "answer": "I read `ledger/core.py` and changed nothing.\n\n## Monthly totals\n\n`monthly_report(entries, year, month)` (`ledger/core.py:51`) computes the totals in three steps:\n\n1. **Filter.** `entries_in_month` (line 40) keeps only the entries whose `day.year` and `day.month` both equal the arguments.\n2. **Grand total.** `total` (line 31) sums `amount` over the filtered entries.\n3. **Per-category totals.** `sum_by_category` (line 44) builds a dict mapping each category to the running sum of its amounts.\n\nThe result is a dict with `count` (number of entries in the month), `total` and `by_category`.\n\n- **Empty months:** a month with no entries gives `count` 0, `by_category` `{}` and `total` as the integer `0` rather than `0.0`, because `sum()` of nothing returns `0`.\n- **Category case:** categories are lowercased when an entry is added (line 27), so \"Food\" and \"food\" land in the same bucket.\n- **Rounding:** amounts are plain floats and nothing is rounded, so totals can show float artefacts such as `0.30000000000000004`.\n- **Out-of-range month:** `year` and `month` are not validated, so month 13 simply matches nothing and returns the empty report.\n\n## Invalid date strings\n\nDates are parsed only in `parse_date` (line 15), which `add_entry` calls. An invalid date always ends in a `ValueError`, but by one of two routes:\n\n- **Wrong shape:** if splitting on `-` does not give exactly three parts, or any part is not all digits, the function raises `ValueError(\"bad date: '<text>'\")`. This covers `\"2024/01/05\"`, `\"2024-01\"`, `\"2024-01-xx\"`, `\"\"` and anything with whitespace or a sign.\n- **Right shape, impossible date:** strings like `\"2024-02-30\"` or `\"2024-13-01\"` pass the shape check and fail inside `date(y, m, d)`. That is still a `ValueError`, but with Python's own message (e.g. \"day is out of range for month\"), not the \"bad date\" one.\n\nRelated behaviours:\n\n- **Looser than the docstring:** field widths are not checked, so `\"2024-1-5\"` is accepted, and so is `\"5-1-2024\"` (read as year 5, month 1, day 2024, which then fails in `date()`). `str.isdigit()` is also true for some non-ASCII characters; superscript digits, for example, pass the check and then fail in `int()`, again as a `ValueError`.\n- **Non-string input:** passing `None` or a `date` object raises `AttributeError` on `.split`, not `ValueError`.\n- **No partial writes:** in `add_entry`, the date is parsed before the `Entry` is appended, so a bad date propagates to the caller and leaves the list unchanged.\n- **Check order:** the amount check (line 25) runs first, so with both a non-positive amount and a bad date you get \"amount must be positive\".\n- **No handling in this file:** nothing here catches the exception; it is up to the caller.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.09137
  },
  {
   "run": "claude-opus-5-5__placebo__T3__r1__29557c",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "placebo",
   "rep": 1,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 1066,
   "input": 4,
   "cache_read": 27845,
   "cache_write": 7393,
   "warmup": null,
   "cache_write_1h": 7393,
   "cost_usd": 0.08604899999999999,
   "wall_s": 13.0,
   "turns": 2,
   "prompt_last": 18022,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 157,
   "commands": null,
   "answer": "**Monthly totals**\n\n`monthly_report(entries, year, month)` in `ledger/core.py:51` computes the totals in three steps:\n\n1. **Filter:** `entries_in_month` keeps only entries whose `day.year` and `day.month` both equal the requested values.\n2. **Sum:** `total` adds up the `amount` of those entries using plain float addition, with no rounding.\n3. **Break down:** `sum_by_category` builds a dict that maps each category to the sum of its amounts.\n\nThe result is a dict with `count` (number of matching entries), `total` and `by_category`. A month with no entries gives `{\"count\": 0, \"total\": 0, \"by_category\": {}}`; nothing is raised. The `year` and `month` arguments are not validated, so a month of 13 also gives that empty report.\n\nCategories are lowercased in `add_entry`, so \"Food\" and \"food\" land in the same bucket.\n\n**Invalid date strings**\n\nDates are parsed only when an entry is added: `add_entry` calls `parse_date` (`ledger/core.py:15`), and an invalid date surfaces as an exception that nothing in this module catches. There are two paths to failure:\n\n- **Wrong shape:** if the string does not split on `-` into exactly three all-digit parts, `parse_date` raises `ValueError(\"bad date: '<text>'\")`. This covers inputs such as `\"2024/01/05\"`, `\"2024-01\"`, `\"2024-Jan-05\"` and `\"\"`.\n- **Right shape, impossible date:** inputs such as `\"2024-02-30\"` or `\"2024-13-01\"` pass the shape check and fail inside `date(y, m, d)`. This is still a `ValueError`, but with Python's own message (for example \"day is out of range for month\") instead of the \"bad date\" one.\n\nIn both cases the exception is raised before `entries.append` runs, so the list is left unchanged and no partial entry is stored.\n\nThree related behaviours are worth knowing:\n\n- **Amount is checked first:** `add_entry` checks `amount <= 0` before parsing the date, so a call with both a bad amount and a bad date reports \"amount must be positive\" only.\n- **Looser than the docstring:** the shape check does not enforce digit widths, so `\"2024-1-5\"` is accepted even though the docstring says YYYY-MM-DD.\n- **Non-string input:** passing `None` or a `date` object raises `AttributeError` from `text.split`, not `ValueError`.\n\nNo files were changed.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.086049
  },
  {
   "run": "claude-opus-5-5__placebo__T3__r2__c561f7",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "placebo",
   "rep": 2,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 1210,
   "input": 4,
   "cache_read": 27847,
   "cache_write": 7396,
   "warmup": null,
   "cache_write_1h": 7396,
   "cost_usd": 0.0889534,
   "wall_s": 12.7,
   "turns": 2,
   "prompt_last": 18025,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 230,
   "commands": null,
   "answer": "**How monthly totals are computed**\n\n`monthly_report(entries, year, month)` in `ledger/core.py:51` builds the report in three steps:\n\n1. **Filter:** `entries_in_month` keeps only the entries whose `day.year` and `day.month` both equal the requested values.\n2. **Total:** `total` sums the `amount` of the filtered entries.\n3. **Break down:** `sum_by_category` walks the same filtered entries and accumulates amounts into a dict keyed by category.\n\nThe result is a dict with `count` (number of entries in that month), `total`, and `by_category`.\n\nA few behaviours follow from this:\n\n- **Case-insensitive categories:** `add_entry` lowercases the category on the way in, so \"Food\" and \"food\" land in the same bucket.\n- **No rounding:** amounts are plain floats, so totals can show ordinary floating-point artefacts such as `0.1 + 0.2`.\n- **Empty months:** a month with no entries gives `count` 0, `total` as the integer `0` (not `0.0`), and an empty `by_category`.\n- **No month validation:** asking for month 13 raises nothing and returns the empty report.\n\n**What happens when a date string is invalid**\n\nDates are parsed only when an entry is added: `add_entry` calls `parse_date` (`ledger/core.py:15`). An invalid date raises `ValueError`, nothing is appended, and the error propagates to the caller uncaught. There are two distinct failure paths:\n\n- **Wrong shape:** if splitting on `-` does not give exactly three all-digit parts (for example `\"2024/01/05\"`, `\"2024-01\"`, `\"2024-Jan-05\"`, an empty string, or anything with spaces or a sign), it raises `ValueError(\"bad date: '<text>'\")`.\n- **Right shape, impossible date:** strings like `\"2024-02-30\"` or `\"2024-13-01\"` pass the shape check and fail inside `date(y, m, d)`. This is still a `ValueError`, but with Python's own message (such as \"day is out of range for month\") and without the \"bad date\" prefix or the offending text.\n\nSome edge cases:\n\n- **Looser than the docstring:** field widths are not checked, so `\"2024-1-5\"` and `\"24-01-05\"` (year 24 AD) are accepted despite the \"YYYY-MM-DD\" docstring.\n- **Non-string input:** passing `None` or a `date` object raises `AttributeError` from `.split`, not `ValueError`.\n- **Amount checked first:** `add_entry` validates the amount before parsing the date, so a call with both a non-positive amount and a bad date reports only \"amount must be positive\".\n\nNo files were changed.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.088953
  },
  {
   "run": "claude-opus-5-5__placebo__T3__r3__932f64",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "placebo",
   "rep": 3,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 1227,
   "input": 4,
   "cache_read": 27843,
   "cache_write": 7391,
   "warmup": null,
   "cache_write_1h": 7391,
   "cost_usd": 0.0892526,
   "wall_s": 13.3,
   "turns": 2,
   "prompt_last": 18020,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 212,
   "commands": null,
   "answer": "**Monthly totals**\n\n`monthly_report(entries, year, month)` (`ledger/core.py:51`) computes the totals in two steps:\n\n1. **Filter:** `entries_in_month` keeps only the entries whose `day.year` and `day.month` both equal the requested values.\n2. **Aggregate:** the filtered list is summarised into a dict with three keys:\n   - `count` is the number of entries in that month.\n   - `total` is the plain `sum()` of their amounts, from `total()`.\n   - `by_category` is a dict mapping each category to the sum of its amounts, from `sum_by_category()`.\n\nA few behaviours follow from that:\n\n- **Empty month:** a month with no entries gives `{\"count\": 0, \"total\": 0, \"by_category\": {}}`. The total here is the integer `0`, not `0.0`, because `sum()` of an empty sequence returns `0`.\n- **No rounding:** amounts are stored as floats and summed as floats, so the usual floating-point artefacts can appear (for example, 0.1 + 0.2 gives 0.30000000000000004).\n- **Category case:** categories are lowercased when an entry is added through `add_entry`, so \"Food\" and \"food\" land in the same bucket.\n- **No month validation:** `year` and `month` are not checked, so asking for month 13 returns an empty report, not an error.\n\n**Invalid date strings**\n\nDates are parsed by `parse_date` (`ledger/core.py:15`), which `add_entry` calls. It splits the text on `-` and requires exactly three parts, each consisting only of digits. What happens next depends on how the string is wrong:\n\n- **Wrong shape** (such as `\"2024/01/05\"`, `\"2024-01\"`, `\"2024-Jan-05\"`, `\"\"`, or a negative component): it raises `ValueError(\"bad date: '<text>'\")`.\n- **Right shape but impossible date** (such as `\"2024-13-01\"` or `\"2023-02-30\"`): the shape check passes, and the `date(y, m, d)` constructor raises its own `ValueError`. The message is Python's (for example, \"month must be in 1..12\"), not the \"bad date\" one.\n- **Not a string** (such as `None` or a `date` object): it raises `AttributeError` from `.split`, not `ValueError`.\n\nIn every case the exception propagates to the caller of `add_entry`; nothing catches it. The `Entry` is built before `append` runs, so a bad date leaves the entries list unchanged. The amount check comes first, so a non-positive amount reports \"amount must be positive\" even if the date is also bad.\n\nThe parser is looser than its \"YYYY-MM-DD\" docstring suggests: it does not enforce digit widths, so `\"2024-1-5\"` is accepted as 5 January 2024.\n\nNo files were changed.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.089253
  },
  {
   "run": "claude-opus-5-5__ponytail__T3__r1__e0e9f9",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "ponytail",
   "rep": 1,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 1294,
   "input": 4,
   "cache_read": 28128,
   "cache_write": 7680,
   "warmup": null,
   "cache_write_1h": 7680,
   "cost_usd": 0.0929616,
   "wall_s": 13.9,
   "turns": 2,
   "prompt_last": 18309,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 248,
   "commands": null,
   "answer": "**Monthly totals**\n\n`monthly_report(entries, year, month)` at `ledger/core.py:51` does three things:\n\n1. **Filters to the month.** `entries_in_month` keeps every entry whose `day.year` and `day.month` both equal the arguments. It is a linear scan of the whole list on every call, with no pre-grouping or caching.\n2. **Sums the filtered list.** `total` is a plain `sum()` of `e.amount`, and `sum_by_category` accumulates amounts into a dict keyed by category.\n3. **Returns a dict** of the form `{\"count\": ..., \"total\": ..., \"by_category\": {...}}`.\n\nA few behaviours follow from that:\n\n- **Float arithmetic:** amounts are stored as floats (`add_entry` calls `float(amount)`), so totals can show the usual drift (e.g. `0.1 + 0.2`), and nothing is rounded.\n- **Empty months:** a month with no entries gives `count` 0, `by_category` `{}` and `total` as the integer `0` rather than `0.0`, because `sum()` of an empty sequence returns its int start value.\n- **Category case:** categories are lowercased when added, so \"Food\" and \"food\" land in the same bucket.\n- **Unvalidated month:** `year` and `month` are not checked, so `month=13` quietly returns an empty report.\n\n**Invalid date strings**\n\nDates are parsed only in `add_entry`, via `parse_date` at `ledger/core.py:15`. It splits on `-` and requires exactly three parts that are all digits; otherwise it raises `ValueError(\"bad date: '<text>'\")`. That covers inputs like `\"2024/01/05\"`, `\"2024-01\"`, `\"abc\"`, `\"\"`, or anything with whitespace or a sign.\n\nIf the shape is right but the date does not exist (`\"2024-02-30\"`, `\"2024-13-01\"`), the `date(y, m, d)` constructor raises its own `ValueError`, with a message like \"day is out of range for month\" instead of \"bad date\".\n\nIn both cases the exception propagates out of `add_entry` uncaught. Parsing happens before `entries.append` runs, so the ledger is left unchanged and nothing is partially written. The amount check runs first, so a non-positive amount with a bad date reports the amount error.\n\nEdge cases in the parser:\n\n- **Loose format:** it is looser than the documented `YYYY-MM-DD`. No digit widths are enforced, so `\"2024-1-5\"` and `\"24-01-05\"` (year 24 AD) are both accepted.\n- **Non-string input:** `None` or another non-string raises `AttributeError` on `.split`, not `ValueError`.\n- **Unicode digits:** characters like `\"\u00b2\"` pass `isdigit()` but fail in `int()`. The result is still a `ValueError`, just with a different message.\n\nNo files were changed.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.092962
  },
  {
   "run": "claude-opus-5-5__ponytail__T3__r2__1ade79",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "ponytail",
   "rep": 2,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 1272,
   "input": 4,
   "cache_read": 28121,
   "cache_write": 7671,
   "warmup": null,
   "cache_write_1h": 7671,
   "cost_usd": 0.09244820000000001,
   "wall_s": 15.7,
   "turns": 2,
   "prompt_last": 18300,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 305,
   "commands": null,
   "answer": "**Monthly totals**\n\n`monthly_report(entries, year, month)` at `ledger/core.py:51` does three things:\n\n1. **Filters**: `entries_in_month` keeps only entries whose `day.year` and `day.month` both equal the arguments. It is a linear scan of the full list on every call.\n2. **Totals**: `total` sums `amount` over the filtered entries.\n3. **Breaks down by category**: `sum_by_category` builds a dict of category to running sum. Categories are lowercased on the way in by `add_entry`, so \"Food\" and \"food\" land in the same bucket.\n\nThe result is `{\"count\": ..., \"total\": ..., \"by_category\": ...}`.\n\nA few behaviours to be aware of:\n\n- **Float money**: amounts are floats in dollars, so totals can carry binary rounding error (e.g. `0.1 + 0.2`), and nothing is rounded.\n- **Empty months**: a month with no entries returns `count: 0`, `total: 0` (the int `0`, not `0.0`) and an empty `by_category`.\n- **Unvalidated arguments**: `year` and `month` are not checked, so `month=13` quietly returns the same empty report instead of raising.\n\n**Invalid date strings**\n\nDates are parsed only in `add_entry`, through `parse_date` (`ledger/core.py:15`). It splits on `-` and needs exactly three parts that are all digits; otherwise it raises `ValueError(\"bad date: '...'\")`. That covers inputs like `\"2024/01/05\"`, `\"2024-01\"`, `\"Jan-5-2024\"`, the empty string, and anything with whitespace or a sign.\n\nA string that has the right shape but is not a real date, such as `\"2024-02-30\"` or `\"2024-13-01\"`, passes that check and then fails in `date(y, m, d)`. That is still a `ValueError`, but with the standard library's message (e.g. \"day is out of range for month\") rather than \"bad date\".\n\nIn both cases the exception propagates out of `add_entry` before the append happens, so the ledger is left unchanged. Nothing is caught or skipped silently. The amount check runs first, so a non-positive amount together with a bad date reports the amount error.\n\nTwo gaps in the parser:\n\n- **Looser than the docstring**: it says YYYY-MM-DD, but no widths are enforced. `\"2024-1-5\"` is accepted, and `\"24-01-05\"` is accepted as year 24 AD, which would then never show up in a 2024 report.\n- **Non-strings**: passing `None` or a `date` object raises `AttributeError` (no `.split`), not `ValueError`.\n\nNo files were changed.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.092448
  },
  {
   "run": "claude-opus-5-5__ponytail__T3__r3__86f3c0",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "ponytail",
   "rep": 3,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 1010,
   "input": 4,
   "cache_read": 20236,
   "cache_write": 13951,
   "warmup": null,
   "cache_write_1h": 13951,
   "cost_usd": 0.1358712,
   "wall_s": 13.5,
   "turns": 2,
   "prompt_last": 17498,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 100,
   "commands": null,
   "answer": "No files changed; this is what `ledger/core.py` does.\n\n## Monthly totals\n\n`monthly_report(entries, year, month)` (`ledger/core.py:51`) does the work in three steps:\n\n1. **Filter.** `entries_in_month` keeps only the entries whose `day.year` and `day.month` both equal the requested values.\n2. **Sum.** `total` adds up the `amount` of every surviving entry with a plain `sum()`. An empty month gives an integer `0`, not `0.0`.\n3. **Break down.** `sum_by_category` accumulates the amounts into a dict keyed by category. Categories are lowercased when added through `add_entry`, so \"Food\" and \"food\" merge.\n\nThe result is a dict with `count` (number of entries in the month), `total` and `by_category`.\n\nAmounts are floats in dollars, so totals can show ordinary floating-point drift (for example `0.1 + 0.2`). Nothing is rounded.\n\n## Invalid date strings\n\nDates are parsed by `parse_date` (`ledger/core.py:15`), which `add_entry` calls. An invalid date always ends in an exception, but which one depends on how it is wrong:\n\n- **Wrong shape:** if the string does not split on `-` into exactly three all-digit parts (such as `\"2024/01/05\"`, `\"2024-01\"`, `\"2024-1-x\"` or `\"\"`), it raises `ValueError(\"bad date: '...'\")`.\n- **Right shape, impossible date:** `\"2024-02-30\"` or `\"2024-13-01\"` pass the shape check and then fail inside `date(y, m, d)`. That is still a `ValueError`, but with the standard library's message (such as \"day is out of range for month\"), not the \"bad date\" one.\n- **Not a string:** `None` or a `date` object raises `AttributeError` on `.split`, not `ValueError`, despite what the docstring says.\n\nIn every case nothing is added to the ledger, because the exception fires before `entries.append` runs. `add_entry` checks the amount first, so a bad amount together with a bad date reports \"amount must be positive\" and the date error is never reached.\n\n## Looser than the docstring\n\nThe shape check only tests for digits, not their width, so `\"2024-1-5\"` and `\"02024-01-05\"` are accepted as valid dates. `str.isdigit()` is also true for some characters `int()` cannot parse, such as superscript `\"\u00b2\"`. Those still raise a `ValueError`, but with `int()`'s message.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.135871
  },
  {
   "run": "claude-opus-5-5__caveman__T4__r1__22d189",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T4",
   "skill": "caveman",
   "rep": 1,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1617,
   "input": 10,
   "cache_read": 138121,
   "cache_write": 25525,
   "warmup": null,
   "cache_write_1h": 25525,
   "cost_usd": 0.2642042,
   "wall_s": 16.4,
   "turns": 5,
   "prompt_last": 35645,
   "lines_added": 56,
   "lines_deleted": 0,
   "lines": 56,
   "files": 1,
   "thinking": 167,
   "commands": null,
   "heldout": true,
   "api_cost": 0.264204
  },
  {
   "run": "claude-opus-5-5__caveman__T4__r2__a27ddd",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T4",
   "skill": "caveman",
   "rep": 2,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1601,
   "input": 10,
   "cache_read": 138052,
   "cache_write": 25479,
   "warmup": null,
   "cache_write_1h": 25479,
   "cost_usd": 0.2635024,
   "wall_s": 18.0,
   "turns": 5,
   "prompt_last": 35599,
   "lines_added": 56,
   "lines_deleted": 0,
   "lines": 56,
   "files": 1,
   "thinking": 101,
   "commands": null,
   "heldout": true,
   "api_cost": 0.263502
  },
  {
   "run": "claude-opus-5-5__caveman__T4__r3__b3b938",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T4",
   "skill": "caveman",
   "rep": 3,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1534,
   "input": 8,
   "cache_read": 102601,
   "cache_write": 25473,
   "warmup": null,
   "cache_write_1h": 25473,
   "cost_usd": 0.2550162,
   "wall_s": 16.9,
   "turns": 5,
   "prompt_last": 35593,
   "lines_added": 58,
   "lines_deleted": 0,
   "lines": 58,
   "files": 1,
   "thinking": 116,
   "commands": null,
   "heldout": true,
   "api_cost": 0.255016
  },
  {
   "run": "claude-opus-5-5__control__T4__r1__cac76c",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T4",
   "skill": "control",
   "rep": 1,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1681,
   "input": 8,
   "cache_read": 92040,
   "cache_write": 20245,
   "warmup": null,
   "cache_write_1h": 20245,
   "cost_usd": 0.21402,
   "wall_s": 18.1,
   "turns": 5,
   "prompt_last": 30365,
   "lines_added": 54,
   "lines_deleted": 0,
   "lines": 54,
   "files": 1,
   "thinking": 234,
   "commands": null,
   "heldout": true,
   "api_cost": 0.21402
  },
  {
   "run": "claude-opus-5-5__control__T4__r2__202e20",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T4",
   "skill": "control",
   "rep": 2,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1496,
   "input": 8,
   "cache_read": 92043,
   "cache_write": 20067,
   "warmup": null,
   "cache_write_1h": 20067,
   "cost_usd": 0.2088966,
   "wall_s": 14.9,
   "turns": 4,
   "prompt_last": 30187,
   "lines_added": 53,
   "lines_deleted": 0,
   "lines": 53,
   "files": 2,
   "thinking": 98,
   "commands": null,
   "heldout": true,
   "api_cost": 0.208897
  },
  {
   "run": "claude-opus-5-5__control__T4__r3__a5e8d5",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T4",
   "skill": "control",
   "rep": 3,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1659,
   "input": 10,
   "cache_read": 121519,
   "cache_write": 19452,
   "warmup": null,
   "cache_write_1h": 19452,
   "cost_usd": 0.21313980000000002,
   "wall_s": 17.8,
   "turns": 6,
   "prompt_last": 29572,
   "lines_added": 56,
   "lines_deleted": 0,
   "lines": 56,
   "files": 1,
   "thinking": 83,
   "commands": null,
   "heldout": true,
   "api_cost": 0.21314
  },
  {
   "run": "claude-opus-5-5__karpathy__T4__r1__875414",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T4",
   "skill": "karpathy",
   "rep": 1,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1539,
   "input": 8,
   "cache_read": 95960,
   "cache_write": 21977,
   "warmup": null,
   "cache_write_1h": 21977,
   "cost_usd": 0.22581999999999997,
   "wall_s": 15.4,
   "turns": 5,
   "prompt_last": 32097,
   "lines_added": 41,
   "lines_deleted": 0,
   "lines": 41,
   "files": 1,
   "thinking": 169,
   "commands": null,
   "heldout": true,
   "api_cost": 0.22582
  },
  {
   "run": "claude-opus-5-5__karpathy__T4__r2__e30d3e",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T4",
   "skill": "karpathy",
   "rep": 2,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1554,
   "input": 8,
   "cache_read": 96021,
   "cache_write": 22016,
   "warmup": null,
   "cache_write_1h": 22016,
   "cost_usd": 0.22644419999999998,
   "wall_s": 16.0,
   "turns": 5,
   "prompt_last": 32136,
   "lines_added": 43,
   "lines_deleted": 0,
   "lines": 43,
   "files": 1,
   "thinking": 218,
   "commands": null,
   "heldout": true,
   "api_cost": 0.226444
  },
  {
   "run": "claude-opus-5-5__karpathy__T4__r3__98f75b",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T4",
   "skill": "karpathy",
   "rep": 3,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1639,
   "input": 10,
   "cache_read": 127916,
   "cache_write": 21989,
   "warmup": null,
   "cache_write_1h": 21989,
   "cost_usd": 0.23431520000000003,
   "wall_s": 18.8,
   "turns": 5,
   "prompt_last": 32109,
   "lines_added": 40,
   "lines_deleted": 0,
   "lines": 40,
   "files": 1,
   "thinking": 242,
   "commands": null,
   "heldout": true,
   "api_cost": 0.234315
  },
  {
   "run": "claude-opus-5-5__placebo__T4__r1__2fb6fb",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T4",
   "skill": "placebo",
   "rep": 1,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 2048,
   "input": 10,
   "cache_read": 125076,
   "cache_write": 21081,
   "warmup": null,
   "cache_write_1h": 21081,
   "cost_usd": 0.23466320000000002,
   "wall_s": 20.0,
   "turns": 5,
   "prompt_last": 31201,
   "lines_added": 66,
   "lines_deleted": 0,
   "lines": 66,
   "files": 2,
   "thinking": 194,
   "commands": null,
   "heldout": true,
   "api_cost": 0.234663
  },
  {
   "run": "claude-opus-5-5__placebo__T4__r2__fe2fff",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T4",
   "skill": "placebo",
   "rep": 2,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 2028,
   "input": 8,
   "cache_read": 94103,
   "cache_write": 21030,
   "warmup": null,
   "cache_write_1h": 21030,
   "cost_usd": 0.2276526,
   "wall_s": 20.8,
   "turns": 4,
   "prompt_last": 31150,
   "lines_added": 66,
   "lines_deleted": 0,
   "lines": 66,
   "files": 2,
   "thinking": 236,
   "commands": null,
   "heldout": true,
   "api_cost": 0.227653
  },
  {
   "run": "claude-opus-5-5__placebo__T4__r3__e49aff",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T4",
   "skill": "placebo",
   "rep": 3,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1874,
   "input": 8,
   "cache_read": 94094,
   "cache_write": 20941,
   "warmup": null,
   "cache_write_1h": 20941,
   "cost_usd": 0.22385880000000002,
   "wall_s": 19.6,
   "turns": 4,
   "prompt_last": 31061,
   "lines_added": 63,
   "lines_deleted": 0,
   "lines": 63,
   "files": 2,
   "thinking": 260,
   "commands": null,
   "heldout": true,
   "api_cost": 0.223859
  },
  {
   "run": "claude-opus-5-5__ponytail__T4__r1__81e73d",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T4",
   "skill": "ponytail",
   "rep": 1,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1674,
   "input": 10,
   "cache_read": 127441,
   "cache_write": 21384,
   "warmup": null,
   "cache_write_1h": 21384,
   "cost_usd": 0.23008019999999998,
   "wall_s": 19.6,
   "turns": 6,
   "prompt_last": 31504,
   "lines_added": 44,
   "lines_deleted": 0,
   "lines": 44,
   "files": 1,
   "thinking": 229,
   "commands": null,
   "heldout": true,
   "api_cost": 0.23008
  },
  {
   "run": "claude-opus-5-5__ponytail__T4__r2__1c2547",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T4",
   "skill": "ponytail",
   "rep": 2,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1485,
   "input": 10,
   "cache_read": 127327,
   "cache_write": 21265,
   "warmup": null,
   "cache_write_1h": 21265,
   "cost_usd": 0.22532539999999998,
   "wall_s": 15.7,
   "turns": 6,
   "prompt_last": 31385,
   "lines_added": 40,
   "lines_deleted": 0,
   "lines": 40,
   "files": 1,
   "thinking": 201,
   "commands": null,
   "heldout": true,
   "api_cost": 0.225325
  },
  {
   "run": "claude-opus-5-5__ponytail__T4__r3__cb4bc7",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T4",
   "skill": "ponytail",
   "rep": 3,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1358,
   "input": 8,
   "cache_read": 72199,
   "cache_write": 14421,
   "warmup": null,
   "cache_write_1h": 14421,
   "cost_usd": 0.15699979999999997,
   "wall_s": 12.8,
   "turns": 5,
   "prompt_last": 24541,
   "lines_added": 38,
   "lines_deleted": 0,
   "lines": 38,
   "files": 1,
   "thinking": 188,
   "commands": null,
   "heldout": true,
   "api_cost": 0.157,
   "light": true
  },
  {
   "run": "claude-sonnet-5__caveman__T1__r1__a9cca4",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T1",
   "skill": "caveman",
   "rep": 1,
   "round": 1,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 836,
   "input": 12,
   "cache_read": 227140,
   "cache_write": 24526,
   "warmup": null,
   "cache_write_1h": 24526,
   "cost_usd": 0.15191599999999997,
   "wall_s": 12.1,
   "turns": 7,
   "prompt_last": 43167,
   "lines_added": 5,
   "lines_deleted": 0,
   "lines": 5,
   "files": null,
   "thinking": 20,
   "commands": null,
   "api_cost": 0.151916
  },
  {
   "run": "claude-sonnet-5__caveman__T1__r2__2f28f1",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T1",
   "skill": "caveman",
   "rep": 2,
   "round": 2,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1156,
   "input": 14,
   "cache_read": 271861,
   "cache_write": 25169,
   "warmup": null,
   "cache_write_1h": 25169,
   "cost_usd": 0.16663619999999998,
   "wall_s": 15.1,
   "turns": 10,
   "prompt_last": 43810,
   "lines_added": 5,
   "lines_deleted": 0,
   "lines": 5,
   "files": null,
   "thinking": 44,
   "commands": null,
   "api_cost": 0.166636
  },
  {
   "run": "claude-sonnet-5__caveman__T1__r3__885f5d",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T1",
   "skill": "caveman",
   "rep": 3,
   "round": 2,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1026,
   "input": 16,
   "cache_read": 323210,
   "cache_write": 27447,
   "warmup": null,
   "cache_write_1h": 27447,
   "cost_usd": 0.184722,
   "wall_s": 17.1,
   "turns": 9,
   "prompt_last": 46088,
   "lines_added": 5,
   "lines_deleted": 0,
   "lines": 5,
   "files": null,
   "thinking": 28,
   "commands": null,
   "api_cost": 0.184722
  },
  {
   "run": "claude-sonnet-5__control__T1__r1__82c6d2",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T1",
   "skill": "control",
   "rep": 1,
   "round": 1,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1145,
   "input": 14,
   "cache_read": 255498,
   "cache_write": 22853,
   "warmup": null,
   "cache_write_1h": 22853,
   "cost_usd": 0.15398960000000003,
   "wall_s": 13.4,
   "turns": 10,
   "prompt_last": 41494,
   "lines_added": 5,
   "lines_deleted": 0,
   "lines": 5,
   "files": null,
   "thinking": 104,
   "commands": null,
   "api_cost": 0.15399
  },
  {
   "run": "claude-sonnet-5__control__T1__r2__de2440",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T1",
   "skill": "control",
   "rep": 2,
   "round": 2,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1047,
   "input": 14,
   "cache_read": 256134,
   "cache_write": 22241,
   "warmup": null,
   "cache_write_1h": 22241,
   "cost_usd": 0.1506888,
   "wall_s": 14.5,
   "turns": 9,
   "prompt_last": 40882,
   "lines_added": 5,
   "lines_deleted": 0,
   "lines": 5,
   "files": null,
   "thinking": 83,
   "commands": null,
   "api_cost": 0.150689
  },
  {
   "run": "claude-sonnet-5__control__T1__r3__893f9a",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T1",
   "skill": "control",
   "rep": 3,
   "round": 2,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1197,
   "input": 16,
   "cache_read": 299504,
   "cache_write": 23585,
   "warmup": null,
   "cache_write_1h": 23585,
   "cost_usd": 0.1662428,
   "wall_s": 24.4,
   "turns": 8,
   "prompt_last": 42226,
   "lines_added": 5,
   "lines_deleted": 0,
   "lines": 5,
   "files": null,
   "thinking": 301,
   "commands": null,
   "api_cost": 0.166243
  },
  {
   "run": "claude-sonnet-5__karpathy__T1__r1__0f8cb2",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T1",
   "skill": "karpathy",
   "rep": 1,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1117,
   "input": 14,
   "cache_read": 218950,
   "cache_write": 17535,
   "warmup": null,
   "cache_write_1h": 17535,
   "cost_usd": 0.12512800000000002,
   "wall_s": 13.7,
   "turns": 8,
   "prompt_last": 36176,
   "lines_added": 5,
   "lines_deleted": 0,
   "lines": 5,
   "files": 1,
   "thinking": 30,
   "commands": null,
   "api_cost": 0.125128,
   "light": true
  },
  {
   "run": "claude-sonnet-5__karpathy__T1__r2__656146",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T1",
   "skill": "karpathy",
   "rep": 2,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1279,
   "input": 16,
   "cache_read": 311651,
   "cache_write": 26005,
   "warmup": null,
   "cache_write_1h": 26005,
   "cost_usd": 0.17917220000000003,
   "wall_s": 17.9,
   "turns": 10,
   "prompt_last": 44646,
   "lines_added": 5,
   "lines_deleted": 0,
   "lines": 5,
   "files": 1,
   "thinking": 103,
   "commands": null,
   "api_cost": 0.179172
  },
  {
   "run": "claude-sonnet-5__karpathy__T1__r3__e91b8a",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T1",
   "skill": "karpathy",
   "rep": 3,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1197,
   "input": 14,
   "cache_read": 267110,
   "cache_write": 24492,
   "warmup": null,
   "cache_write_1h": 24492,
   "cost_usd": 0.16338799999999998,
   "wall_s": 15.4,
   "turns": 10,
   "prompt_last": 43133,
   "lines_added": 5,
   "lines_deleted": 0,
   "lines": 5,
   "files": 1,
   "thinking": 42,
   "commands": null,
   "api_cost": 0.163388
  },
  {
   "run": "claude-sonnet-5__placebo__T1__r1__dbfc68",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T1",
   "skill": "placebo",
   "rep": 1,
   "round": 4,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1396,
   "input": 14,
   "cache_read": 268183,
   "cache_write": 25393,
   "warmup": null,
   "cache_write_1h": 25393,
   "cost_usd": 0.1691966,
   "wall_s": 16.6,
   "turns": 10,
   "prompt_last": 44034,
   "lines_added": 6,
   "lines_deleted": 0,
   "lines": 6,
   "files": 1,
   "thinking": 97,
   "commands": null,
   "api_cost": 0.169197
  },
  {
   "run": "claude-sonnet-5__placebo__T1__r2__1d379f",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T1",
   "skill": "placebo",
   "rep": 2,
   "round": 4,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1540,
   "input": 18,
   "cache_read": 347439,
   "cache_write": 24560,
   "warmup": null,
   "cache_write_1h": 24560,
   "cost_usd": 0.1831638,
   "wall_s": 18.6,
   "turns": 10,
   "prompt_last": 43201,
   "lines_added": 7,
   "lines_deleted": 1,
   "lines": 8,
   "files": 2,
   "thinking": 121,
   "commands": null,
   "api_cost": 0.183164
  },
  {
   "run": "claude-sonnet-5__placebo__T1__r3__2c1124",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T1",
   "skill": "placebo",
   "rep": 3,
   "round": 4,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1187,
   "input": 14,
   "cache_read": 264296,
   "cache_write": 24229,
   "warmup": null,
   "cache_write_1h": 24229,
   "cost_usd": 0.1616732,
   "wall_s": 17.5,
   "turns": 10,
   "prompt_last": 42870,
   "lines_added": 5,
   "lines_deleted": 0,
   "lines": 5,
   "files": 1,
   "thinking": 12,
   "commands": null,
   "api_cost": 0.161673
  },
  {
   "run": "claude-sonnet-5__ponytail__T1__r1__1e7770",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T1",
   "skill": "ponytail",
   "rep": 1,
   "round": 1,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1369,
   "input": 16,
   "cache_read": 303708,
   "cache_write": 24147,
   "warmup": null,
   "cache_write_1h": 24147,
   "cost_usd": 0.1710516,
   "wall_s": 15.2,
   "turns": 9,
   "prompt_last": 42788,
   "lines_added": 6,
   "lines_deleted": 1,
   "lines": 7,
   "files": null,
   "thinking": 76,
   "commands": null,
   "api_cost": 0.171052
  },
  {
   "run": "claude-sonnet-5__ponytail__T1__r2__15616a",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T1",
   "skill": "ponytail",
   "rep": 2,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1483,
   "input": 18,
   "cache_read": 343739,
   "cache_write": 24277,
   "warmup": null,
   "cache_write_1h": 24277,
   "cost_usd": 0.18072180000000002,
   "wall_s": 16.6,
   "turns": 10,
   "prompt_last": 42918,
   "lines_added": 6,
   "lines_deleted": 1,
   "lines": 7,
   "files": 2,
   "thinking": 169,
   "commands": null,
   "api_cost": 0.180722
  },
  {
   "run": "claude-sonnet-5__ponytail__T1__r3__4b8b4a",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T1",
   "skill": "ponytail",
   "rep": 3,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 948,
   "input": 12,
   "cache_read": 222153,
   "cache_write": 24050,
   "warmup": null,
   "cache_write_1h": 24050,
   "cost_usd": 0.15013459999999998,
   "wall_s": 15.1,
   "turns": 7,
   "prompt_last": 42691,
   "lines_added": 5,
   "lines_deleted": 0,
   "lines": 5,
   "files": 1,
   "thinking": 96,
   "commands": null,
   "api_cost": 0.150135
  },
  {
   "run": "claude-sonnet-5__caveman__T2__r1__0c7f37",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T2",
   "skill": "caveman",
   "rep": 1,
   "round": 1,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1256,
   "input": 18,
   "cache_read": 296583,
   "cache_write": 18459,
   "warmup": null,
   "cache_write_1h": 18459,
   "cost_usd": 0.1457486,
   "wall_s": 16.2,
   "turns": 9,
   "prompt_last": 37100,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": null,
   "thinking": 57,
   "commands": null,
   "api_cost": 0.145749,
   "light": true
  },
  {
   "run": "claude-sonnet-5__caveman__T2__r2__54e7b3",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T2",
   "skill": "caveman",
   "rep": 2,
   "round": 2,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1337,
   "input": 14,
   "cache_read": 284612,
   "cache_write": 28234,
   "warmup": null,
   "cache_write_1h": 28234,
   "cost_usd": 0.1832564,
   "wall_s": 25.5,
   "turns": 10,
   "prompt_last": 46875,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": null,
   "thinking": 242,
   "commands": null,
   "api_cost": 0.183256
  },
  {
   "run": "claude-sonnet-5__caveman__T2__r3__75dfbd",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T2",
   "skill": "caveman",
   "rep": 3,
   "round": 2,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 816,
   "input": 14,
   "cache_read": 269459,
   "cache_write": 24407,
   "warmup": null,
   "cache_write_1h": 24407,
   "cost_usd": 0.1597078,
   "wall_s": 14.4,
   "turns": 7,
   "prompt_last": 43048,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": null,
   "thinking": 34,
   "commands": null,
   "api_cost": 0.159708
  },
  {
   "run": "claude-sonnet-5__control__T2__r1__d9a2f4",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T2",
   "skill": "control",
   "rep": 1,
   "round": 1,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1223,
   "input": 12,
   "cache_read": 216164,
   "cache_write": 22553,
   "warmup": null,
   "cache_write_1h": 22553,
   "cost_usd": 0.1456988,
   "wall_s": 22.0,
   "turns": 8,
   "prompt_last": 41194,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": null,
   "thinking": 332,
   "commands": null,
   "api_cost": 0.145699
  },
  {
   "run": "claude-sonnet-5__control__T2__r2__2f4e5a",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T2",
   "skill": "control",
   "rep": 2,
   "round": 2,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1111,
   "input": 16,
   "cache_read": 292902,
   "cache_write": 22279,
   "warmup": null,
   "cache_write_1h": 22279,
   "cost_usd": 0.15883840000000002,
   "wall_s": 14.3,
   "turns": 8,
   "prompt_last": 40920,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": null,
   "thinking": 120,
   "commands": null,
   "api_cost": 0.158838
  },
  {
   "run": "claude-sonnet-5__control__T2__r3__4daafd",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T2",
   "skill": "control",
   "rep": 3,
   "round": 2,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1277,
   "input": 16,
   "cache_read": 299967,
   "cache_write": 23166,
   "warmup": null,
   "cache_write_1h": 23166,
   "cost_usd": 0.1654594,
   "wall_s": 24.1,
   "turns": 10,
   "prompt_last": 41807,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": null,
   "thinking": 136,
   "commands": null,
   "api_cost": 0.165459
  },
  {
   "run": "claude-sonnet-5__karpathy__T2__r1__7d6ec8",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T2",
   "skill": "karpathy",
   "rep": 1,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1317,
   "input": 18,
   "cache_read": 346230,
   "cache_write": 24956,
   "warmup": null,
   "cache_write_1h": 24956,
   "cost_usd": 0.182276,
   "wall_s": 26.4,
   "turns": 9,
   "prompt_last": 43597,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 269,
   "commands": null,
   "api_cost": 0.182276
  },
  {
   "run": "claude-sonnet-5__karpathy__T2__r2__acf876",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T2",
   "skill": "karpathy",
   "rep": 2,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1791,
   "input": 16,
   "cache_read": 309438,
   "cache_write": 25938,
   "warmup": null,
   "cache_write_1h": 25938,
   "cost_usd": 0.18358159999999998,
   "wall_s": 21.6,
   "turns": 11,
   "prompt_last": 44579,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 587,
   "commands": null,
   "api_cost": 0.183582
  },
  {
   "run": "claude-sonnet-5__karpathy__T2__r3__1f2408",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T2",
   "skill": "karpathy",
   "rep": 3,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1666,
   "input": 14,
   "cache_read": 265256,
   "cache_write": 25219,
   "warmup": null,
   "cache_write_1h": 25219,
   "cost_usd": 0.17061520000000002,
   "wall_s": 17.5,
   "turns": 11,
   "prompt_last": 43860,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 485,
   "commands": null,
   "api_cost": 0.170615
  },
  {
   "run": "claude-sonnet-5__placebo__T2__r1__eec975",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T2",
   "skill": "placebo",
   "rep": 1,
   "round": 4,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 2050,
   "input": 12,
   "cache_read": 221012,
   "cache_write": 24100,
   "warmup": null,
   "cache_write_1h": 24100,
   "cost_usd": 0.1611264,
   "wall_s": 22.1,
   "turns": 8,
   "prompt_last": 42741,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 967,
   "commands": null,
   "api_cost": 0.161126
  },
  {
   "run": "claude-sonnet-5__placebo__T2__r2__ead22d",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T2",
   "skill": "placebo",
   "rep": 2,
   "round": 4,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1596,
   "input": 14,
   "cache_read": 262873,
   "cache_write": 24947,
   "warmup": null,
   "cache_write_1h": 24947,
   "cost_usd": 0.1683506,
   "wall_s": 17.4,
   "turns": 9,
   "prompt_last": 43588,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 527,
   "commands": null,
   "api_cost": 0.168351
  },
  {
   "run": "claude-sonnet-5__placebo__T2__r3__f52293",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T2",
   "skill": "placebo",
   "rep": 3,
   "round": 4,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 2054,
   "input": 14,
   "cache_read": 263772,
   "cache_write": 25437,
   "warmup": null,
   "cache_write_1h": 25437,
   "cost_usd": 0.17507040000000001,
   "wall_s": 22.9,
   "turns": 9,
   "prompt_last": 44078,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 735,
   "commands": null,
   "api_cost": 0.17507
  },
  {
   "run": "claude-sonnet-5__ponytail__T2__r1__3aa0a3",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T2",
   "skill": "ponytail",
   "rep": 1,
   "round": 1,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1824,
   "input": 16,
   "cache_read": 306856,
   "cache_write": 24556,
   "warmup": null,
   "cache_write_1h": 24556,
   "cost_usd": 0.1778672,
   "wall_s": 29.9,
   "turns": 10,
   "prompt_last": 43197,
   "lines_added": 2,
   "lines_deleted": 2,
   "lines": 4,
   "files": null,
   "thinking": 503,
   "commands": null,
   "api_cost": 0.177867
  },
  {
   "run": "claude-sonnet-5__ponytail__T2__r2__ba62f0",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T2",
   "skill": "ponytail",
   "rep": 2,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1718,
   "input": 16,
   "cache_read": 308564,
   "cache_write": 24956,
   "warmup": null,
   "cache_write_1h": 24956,
   "cost_usd": 0.17874879999999999,
   "wall_s": 20.5,
   "turns": 10,
   "prompt_last": 43597,
   "lines_added": 2,
   "lines_deleted": 2,
   "lines": 4,
   "files": 1,
   "thinking": 446,
   "commands": null,
   "api_cost": 0.178749
  },
  {
   "run": "claude-sonnet-5__ponytail__T2__r3__6059be",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T2",
   "skill": "ponytail",
   "rep": 3,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1552,
   "input": 18,
   "cache_read": 286548,
   "cache_write": 17695,
   "warmup": null,
   "cache_write_1h": 17695,
   "cost_usd": 0.1436456,
   "wall_s": 19.3,
   "turns": 9,
   "prompt_last": 36336,
   "lines_added": 2,
   "lines_deleted": 2,
   "lines": 4,
   "files": 1,
   "thinking": 415,
   "commands": null,
   "api_cost": 0.143646,
   "light": true
  },
  {
   "run": "claude-sonnet-5__caveman__T3__r1__12290b",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "caveman",
   "rep": 1,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 289,
   "input": 4,
   "cache_read": 49249,
   "cache_write": 12807,
   "warmup": null,
   "cache_write_1h": 12807,
   "cost_usd": 0.0639758,
   "wall_s": 5.9,
   "turns": 2,
   "prompt_last": 31448,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 17,
   "commands": null,
   "answer": "Monthly totals: `monthly_report` call `entries_in_month`, filter entries by `year` and `month` match on `e.day`. Then `total()` sum `amount` field over filtered entries. `sum_by_category` also run on same filtered set, group amount by category.\n\nInvalid date string: `parse_date` split on \"-\", check 3 parts, all digit. Fail either check, raise `ValueError(f\"bad date: {text!r}\")`. Bad date never reach `Entry` creation, so `add_entry` propagate exception, entries list unchanged.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.063976
  },
  {
   "run": "claude-sonnet-5__caveman__T3__r2__1173e0",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "caveman",
   "rep": 2,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 291,
   "input": 4,
   "cache_read": 49251,
   "cache_write": 12808,
   "warmup": null,
   "cache_write_1h": 12808,
   "cost_usd": 0.0640002,
   "wall_s": 6.3,
   "turns": 2,
   "prompt_last": 31449,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 15,
   "commands": null,
   "answer": "Monthly total: `monthly_report` filter entries by year+month via `entries_in_month`, then `total()` sum `amount` field over filtered list. Also count entries and `sum_by_category` breakdown same filtered set.\n\nInvalid date: `parse_date` split on `-`, need exactly 3 parts, all digit. Fail either check \u2192 raise `ValueError(f\"bad date: {text!r}\")`. Called from `add_entry`, so bad date raise at entry add time, before entry ever reach ledger. No catch anywhere \u2014 error propagate to caller.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.064
  },
  {
   "run": "claude-sonnet-5__caveman__T3__r3__017b6f",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "caveman",
   "rep": 3,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 239,
   "input": 4,
   "cache_read": 49250,
   "cache_write": 12791,
   "warmup": null,
   "cache_write_1h": 12791,
   "cost_usd": 0.063412,
   "wall_s": 4.9,
   "turns": 2,
   "prompt_last": 31432,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "Monthly total: `monthly_report` call `entries_in_month` filter entries by year and month, then `total` sum their `amount` field.\n\nInvalid date: `parse_date` split string on `-`. Need exact 3 parts, each digit-only, else raise `ValueError(f\"bad date: {text!r}\")`. `add_entry` call `parse_date` before append, so bad date abort whole entry, entry never enter ledger, error propagate up to caller.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.063412
  },
  {
   "run": "claude-sonnet-5__control__T3__r1__01c6bd",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "control",
   "rep": 1,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 984,
   "input": 6,
   "cache_read": 74452,
   "cache_write": 10197,
   "warmup": null,
   "cache_write_1h": 10197,
   "cost_usd": 0.06553039999999999,
   "wall_s": 12.8,
   "turns": 3,
   "prompt_last": 28838,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 27,
   "commands": null,
   "answer": "Here's how it works:\n\n**Monthly totals** \u2014 `monthly_report(entries, year, month)` (ledger/core.py:51-53) first calls `entries_in_month`, which filters the full entry list down to those whose `day.year` and `day.month` match the requested year/month (line 40-41). It then computes three things from that filtered subset: `count` (number of entries), `total` (via `total()`, which just sums `e.amount` across the filtered entries, line 31-32), and `by_category` (via `sum_by_category()`, which builds a dict accumulating amounts per lowercased category, line 44-48). So the \"monthly total\" is simply the sum of `amount` over every entry whose stored `date` falls in that year/month \u2014 there's no separate date-parsing step at this stage since entries already store real `date` objects.\n\n**Invalid date strings** \u2014 date parsing only happens earlier, when an entry is created via `add_entry`, which calls `parse_date(day)` (line 27). `parse_date` (lines 15-21) splits the input on `-` and requires exactly 3 parts that are all purely digit strings (`str.isdigit()`); if not, it raises `ValueError(f\"bad date: {text!r}\")` immediately. If the split parts look numeric but form an impossible date (e.g., month 13, or Feb 30), the digit check passes but `date(y, m, d)` itself raises `ValueError` (Python's built-in validation) when constructing the `date` object. Either way, `add_entry` doesn't catch this \u2014 the `ValueError` propagates up to the caller, and the entry is never appended to the list. So invalid dates never make it into `entries` at all, and therefore can never silently corrupt a monthly total; they fail loudly at insertion time instead.\n\nOne edge case worth noting: `isdigit()` accepts unicode digit characters and doesn't allow a leading `-` for negative years, and it doesn't restrict length (e.g., `\"2024-1-1\"` or `\"02024-001-01\"` would pass the digit check and just get coerced by `int()`), so `parse_date` is stricter about *format shape* (three dash-separated numeric groups) than about zero-padding or magnitude \u2014 validity of the actual calendar date is left entirely to `date()`'s constructor.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.06553
  },
  {
   "run": "claude-sonnet-5__control__T3__r2__bfb834",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "control",
   "rep": 2,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 744,
   "input": 6,
   "cache_read": 74580,
   "cache_write": 10235,
   "warmup": null,
   "cache_write_1h": 10235,
   "cost_usd": 0.063308,
   "wall_s": 9.9,
   "turns": 3,
   "prompt_last": 28876,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 11,
   "commands": null,
   "answer": "Here's how it works:\n\n**Monthly totals:** `monthly_report(entries, year, month)` first calls `entries_in_month`, which filters the full entry list down to those whose `day.year` and `day.month` match the requested year/month. That filtered subset is then passed to `total`, which simply sums the `amount` field across those entries. The report also computes a per-category breakdown (`sum_by_category`) over the same filtered subset and a count of matching entries. So \"monthly total\" is just a plain sum of dollar amounts for entries whose date falls in that calendar month \u2014 no weighting, rounding, or currency conversion involved.\n\n**Invalid date strings:** Dates only enter the system through `add_entry`, which calls `parse_date(day)`. `parse_date` splits the string on `-` and requires exactly 3 parts, each consisting solely of digits (via `str.isdigit()`). If that check fails \u2014 e.g., wrong number of parts, non-numeric characters, or a format like `MM/DD/YYYY` \u2014 it raises `ValueError(f\"bad date: {text!r}\")` immediately, and no `Entry` is created. If the string passes that shape check but represents an impossible calendar date (e.g., `\"2024-02-30\"` or `\"2024-13-01\"`), the `date(y, m, d)` constructor itself raises a `ValueError` (Python's built-in date validation), which propagates uncaught out of `parse_date` and `add_entry`. Either way, there's no try/except anywhere in this module \u2014 an invalid date string always results in an unhandled `ValueError` bubbling up to the caller, and the entry is never added to the ledger.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.063308
  },
  {
   "run": "claude-sonnet-5__control__T3__r3__8bb30a",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "control",
   "rep": 3,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 976,
   "input": 6,
   "cache_read": 74786,
   "cache_write": 10359,
   "warmup": null,
   "cache_write_1h": 10359,
   "cost_usd": 0.0661652,
   "wall_s": 10.8,
   "turns": 3,
   "prompt_last": 29000,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 20,
   "commands": null,
   "answer": "Here's how it works:\n\n**Monthly totals** (`monthly_report`, line 51-53): It first calls `entries_in_month(entries, year, month)`, which filters the entry list to those whose `day.year` and `day.month` match the requested year/month (line 40-41). Then it builds a report dict with:\n- `count`: number of matching entries\n- `total`: the sum of `.amount` across those filtered entries, via `total()` (line 31-32), which just does `sum(e.amount for e in entries)`\n- `by_category`: a dict mapping category \u2192 summed amount for that month, computed by `sum_by_category()` (line 44-48), which iterates the filtered entries and accumulates into a dict with `out.get(e.category, 0.0) + e.amount`\n\nSo the monthly total is a plain sum of the `amount` field over whatever entries already exist in the ledger for that year/month \u2014 there's no re-parsing or validation happening at report time.\n\n**Invalid date strings**: Validation only happens once, at entry-creation time, not at report time. `add_entry()` (line 24-28) calls `parse_date(day)` before constructing the `Entry`. `parse_date()` (line 15-21) splits the string on `-` and requires exactly 3 parts that are all digit strings (`str.isdigit()`); if that check fails, it raises `ValueError(f\"bad date: {text!r}\")` immediately (line 19). If the format passes but the actual numbers are out of range (e.g., month 13, day 32, or a genuinely invalid combination like Feb 30), the `date(y, m, d)` constructor call on line 21 raises its own `ValueError` (e.g. \"day is out of range for month\").\n\nPractically, this means `add_entry` will raise and the entry is never appended to the list \u2014 so an invalid date string never makes it into `entries` at all, and therefore never affects `monthly_report`/`total`. There's no try/except anywhere in this file, so the `ValueError` simply propagates up to whatever code called `add_entry`.\n\nOne edge case worth noting: `isdigit()` doesn't accept a leading `-` sign, so negative years/components aren't silently parsed \u2014 they just fail the `isdigit` check and raise the \"bad date\" error rather than reaching `date(...)`.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.066165
  },
  {
   "run": "claude-sonnet-5__karpathy__T3__r1__cec673",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "karpathy",
   "rep": 1,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 745,
   "input": 4,
   "cache_read": 47617,
   "cache_write": 11166,
   "warmup": null,
   "cache_write_1h": 11166,
   "cost_usd": 0.061645399999999996,
   "wall_s": 8.0,
   "turns": 2,
   "prompt_last": 29807,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 7,
   "commands": null,
   "answer": "## How monthly totals are computed\n\n`monthly_report(entries, year, month)` (core.py:51-53) does this in three steps:\n\n1. **Filter to the month** \u2014 `entries_in_month` (core.py:40-41) keeps only entries whose `day.year` and `day.month` match the requested year/month.\n2. **Sum the total** \u2014 `total()` (core.py:31-32) sums `amount` across those filtered entries with a plain `sum(...)`.\n3. **Break down by category** \u2014 `sum_by_category()` (core.py:44-48) builds a dict accumulating `amount` per (lowercased) `category` for the same filtered entries.\n\nThe report returns `{\"count\": ..., \"total\": ..., \"by_category\": ...}` \u2014 the count and category breakdown are computed from the same filtered list, so they stay consistent with the total.\n\n## What happens with an invalid date string\n\nDate parsing happens earlier, in `add_entry` \u2192 `parse_date` (core.py:15-21), not at report time \u2014 by the time entries exist in the list, their `day` is already a valid `date` object.\n\n`parse_date` splits the string on `-` and requires exactly 3 parts that are all digit strings (`isdigit()`). If that check fails \u2014 wrong number of parts, non-numeric characters, empty string, etc. \u2014 it raises `ValueError(f\"bad date: {text!r}\")` immediately, before ever constructing a `date`.\n\nNote this means `add_entry` will raise for any malformed date (e.g. `\"2024/01/15\"`, `\"15-01-2024\"`, `\"not-a-date\"`), so no invalid dates ever make it into `entries` \u2014 `monthly_report` and its helpers never have to handle a bad date themselves. One subtlety: `isdigit()` doesn't reject negative numbers or catch semantically invalid dates (e.g. `\"2024-13-45\"` passes the digit check but then raises a `ValueError` from the underlying `date(y, m, d)` constructor instead, with a different error message).",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.061645
  },
  {
   "run": "claude-sonnet-5__karpathy__T3__r2__e71590",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "karpathy",
   "rep": 2,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 785,
   "input": 4,
   "cache_read": 47620,
   "cache_write": 11182,
   "warmup": null,
   "cache_write_1h": 11182,
   "cost_usd": 0.06211,
   "wall_s": 8.9,
   "turns": 2,
   "prompt_last": 29823,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 20,
   "commands": null,
   "answer": "## Monthly totals\n\n`monthly_report(entries, year, month)` (`ledger/core.py:51-53`) first calls `entries_in_month`, which filters `entries` down to those whose `day.year` and `day.month` match the given values. It then builds a dict with:\n\n- `count` \u2014 number of matching entries\n- `total` \u2014 `total(month_entries)`, i.e. the sum of `.amount` across those filtered entries\n- `by_category` \u2014 `sum_by_category(month_entries)`, a dict mapping each category to the sum of amounts in that category, built by iterating entries and accumulating into `out.get(e.category, 0.0) + e.amount`\n\nSo the \"monthly total\" is just a plain sum of `amount` over entries already stored with valid `date` objects for that year/month \u2014 no filtering or validation happens at this stage.\n\n## Invalid date strings\n\nValidation happens earlier, at entry-creation time in `parse_date` (`ledger/core.py:15-21`), which `add_entry` calls before appending an `Entry`:\n\n- The string is split on `-`. If it doesn't split into exactly 3 parts, or any part isn't all digits (`str.isdigit()`), it raises `ValueError(f\"bad date: {text!r}\")` immediately.\n- If it passes that check, the three parts are converted to ints and passed to `date(y, m, d)`. If they're digit-only but not a valid calendar date (e.g. month 13, day 31 in February, or a `date()`-range issue), the underlying `date()` constructor raises its own `ValueError` (e.g. \"month must be in 1..12\"), which propagates up uncaught.\n\nEither way, an invalid date string causes `add_entry` to raise `ValueError` and the entry is never appended to `entries`. There's no try/except anywhere in this file \u2014 the exception just propagates to the caller. Since invalid entries never make it into the list, `monthly_report`/`entries_in_month` never has to deal with bad dates; by the time totals are computed, every `Entry.day` is guaranteed to be a valid `date` object.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.06211
  },
  {
   "run": "claude-sonnet-5__karpathy__T3__r3__31fe49",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "karpathy",
   "rep": 3,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 907,
   "input": 6,
   "cache_read": 76835,
   "cache_write": 11394,
   "warmup": null,
   "cache_write_1h": 11394,
   "cost_usd": 0.070025,
   "wall_s": 11.6,
   "turns": 3,
   "prompt_last": 30035,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 36,
   "commands": null,
   "answer": "Here's how it works:\n\n**Monthly totals** \u2014 `monthly_report(entries, year, month)` (core.py:51-53) first calls `entries_in_month`, which filters the full entry list down to those whose `day.year` and `day.month` match the given values (core.py:40-41). It then builds a report dict with:\n- `count`: number of matching entries\n- `total`: `total(month_entries)` (core.py:31-32), which is just `sum(e.amount for e in entries)` \u2014 a plain sum of the `amount` field across the filtered entries, no rounding or currency handling\n- `by_category`: `sum_by_category(month_entries)` (core.py:44-48), which buckets those same filtered entries into a dict keyed by `category`, accumulating amounts per key\n\nSo the monthly total is simply the sum of `amount` over every `Entry` whose date falls in that year/month \u2014 it relies entirely on `Entry.day` already being a proper `date` object.\n\n**Invalid date strings** \u2014 dates only enter the system via `add_entry`, which calls `parse_date(day)` (core.py:27) to convert the string into a `date` before constructing the `Entry`. `parse_date` (core.py:15-21) splits the string on `-` and requires exactly 3 parts that are all digit-only. If that check fails \u2014 wrong number of segments, non-numeric characters, empty string, etc. \u2014 it raises `ValueError(f\"bad date: {text!r}\")` immediately (core.py:19), and no `Entry` is created (`add_entry` never appends). Note this is a fairly loose check: it doesn't validate the day/month ranges itself, so something like `\"2024-13-40\"` passes the `isdigit` check and is only caught when `date(y, m, d)` itself raises `ValueError` for an out-of-range month/day. Either way, the error propagates up out of `add_entry` uncaught \u2014 there's no try/except anywhere in this module, so invalid dates simply prevent the entry from being added and the exception surfaces to the caller.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.070025
  },
  {
   "run": "claude-sonnet-5__placebo__T3__r1__80a541",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "placebo",
   "rep": 1,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 963,
   "input": 6,
   "cache_read": 76298,
   "cache_write": 11121,
   "warmup": null,
   "cache_write_1h": 11121,
   "cost_usd": 0.06938559999999999,
   "wall_s": 12.1,
   "turns": 3,
   "prompt_last": 29762,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 35,
   "commands": null,
   "answer": "Here's how it works, based on `ledger/core.py`:\n\n**Monthly totals**\n\n`monthly_report(entries, year, month)` (line 51) is the entry point:\n1. It calls `entries_in_month(entries, year, month)`, which filters the full list of `Entry` objects down to those whose `day.year` and `day.month` match the given `year`/`month` (line 40-41).\n2. It computes `total()` over just that filtered subset \u2014 a plain `sum()` of each entry's `amount` (line 31-32).\n3. It also computes `sum_by_category()` on the same subset, which builds a dict mapping each entry's `category` to the running sum of `amount` for that category (line 44-48).\n4. The result is a dict: `{\"count\": ..., \"total\": ..., \"by_category\": ...}`.\n\nSo the \"monthly total\" is just the sum of `amount` across every entry whose stored `date` falls in that year/month \u2014 there's no rounding or currency handling beyond plain float addition.\n\n**Invalid date strings**\n\nDates only get parsed when an entry is created, via `add_entry` \u2192 `parse_date` (line 15-21). `parse_date` splits the text on `-` and requires exactly 3 numeric parts (`len(parts) != 3 or not all(p.isdigit() for p in parts)`). If that check fails, it raises `ValueError(f\"bad date: {text!r}\")` immediately \u2014 it never reaches `date(y, m, d)`.\n\nIf the string *looks* numeric and has 3 parts (e.g. `\"2024-13-40\"`) but represents an impossible calendar date, `parse_date` passes it straight to Python's `date(y, m, d)` constructor, which raises its own `ValueError` (e.g. `\"month must be in 1..12\"`) instead of the custom \"bad date\" message.\n\nEither way, invalid input raises rather than being silently accepted, and since `add_entry` calls `parse_date` before appending to `entries`, a bad date string means the entry is never added \u2014 so it can't corrupt `total()`, `sum_by_category()`, or `monthly_report()` downstream. There's no `try/except` anywhere in this module, so the exception propagates straight to the caller.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.069386
  },
  {
   "run": "claude-sonnet-5__placebo__T3__r2__b90dd2",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "placebo",
   "rep": 2,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 743,
   "input": 6,
   "cache_read": 76309,
   "cache_write": 11129,
   "warmup": null,
   "cache_write_1h": 11129,
   "cost_usd": 0.0672198,
   "wall_s": 8.5,
   "turns": 3,
   "prompt_last": 29770,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 11,
   "commands": null,
   "answer": "**Monthly totals:** `monthly_report(entries, year, month)` first calls `entries_in_month`, which filters `entries` down to those whose `day.year` and `day.month` match the given `year`/`month`. That filtered list is then passed to `total`, which simply sums the `amount` field across those entries. The same filtered list also goes through `sum_by_category` to build a per-category breakdown. The report returns a dict with `count` (number of matching entries), `total` (the summed amount), and `by_category` (a dict mapping category \u2192 summed amount).\n\nEntries only end up with a `day` in the first place via `add_entry`, which calls `parse_date` on the incoming date string.\n\n**Invalid date strings:** `parse_date` splits the text on `-` and requires exactly 3 parts that are all purely numeric digit strings (`str.isdigit()`). If that check fails \u2014 wrong number of parts, non-numeric characters, empty strings, etc. \u2014 it raises `ValueError(f\"bad date: {text!r}\")` immediately, before ever constructing a `date`. If the split passes that check but the numbers are out of range (e.g., month 13, day 31 in February), the subsequent `date(y, m, d)` call itself raises a `ValueError` (from the standard library), just with `datetime`'s own message rather than the custom one. Either way, `add_entry` (and thus anything building entries from raw strings) propagates a `ValueError` uncaught \u2014 there's no try/except anywhere in this module, so invalid dates simply abort entry creation rather than being silently skipped or coerced.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.06722
  },
  {
   "run": "claude-sonnet-5__placebo__T3__r3__db08cf",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "placebo",
   "rep": 3,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 724,
   "input": 6,
   "cache_read": 76305,
   "cache_write": 11126,
   "warmup": null,
   "cache_write_1h": 11126,
   "cost_usd": 0.06701700000000001,
   "wall_s": 9.8,
   "turns": 3,
   "prompt_last": 29767,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 24,
   "commands": null,
   "answer": "Here's how it works:\n\n**Monthly totals.** `monthly_report(entries, year, month)` first calls `entries_in_month`, which filters the full entry list down to those whose `day.year` and `day.month` match the requested period. That filtered list is then run through `total()` (a plain `sum()` of the `amount` fields) to get the month's total, and through `sum_by_category()` to get a dict of per-category subtotals. The report returned is `{\"count\", \"total\", \"by_category\"}` \u2014 the count and both totals are all derived from that same filtered subset, so there's no double-counting or cross-month leakage.\n\n**Invalid date strings.** Dates only enter the system through `add_entry`, which calls `parse_date(day)` before constructing the `Entry`. `parse_date` splits the string on `-` and requires exactly 3 parts that are all digit-only; if that check fails, it raises `ValueError(f\"bad date: {text!r}\")` immediately \u2014 no entry is appended. If the split passes but the numbers aren't a real calendar date (e.g. month 13, or Feb 30), the `date(y, m, d)` constructor itself raises `ValueError` (Python's built-in message, e.g. `\"month must be in 1..12\"`), which propagates up uncaught. Either way, an invalid date string means `add_entry` raises before the entry is ever added to the ledger, so it can't silently corrupt a later monthly total \u2014 the failure happens at insertion time, not at reporting time.\n\nNo files were changed.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.067017
  },
  {
   "run": "claude-sonnet-5__ponytail__T3__r1__573a92",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "ponytail",
   "rep": 1,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 749,
   "input": 6,
   "cache_read": 76851,
   "cache_write": 11400,
   "warmup": null,
   "cache_write_1h": 11400,
   "cost_usd": 0.0684722,
   "wall_s": 11.8,
   "turns": 3,
   "prompt_last": 30041,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 16,
   "commands": null,
   "answer": "Monthly totals: `monthly_report(entries, year, month)` first calls `entries_in_month`, which filters `entries` to those whose `Entry.day` matches the given `year` and `month` (day-of-month is irrelevant, only year/month must match). It then builds a dict with `count` (number of matching entries), `total` (the sum of `.amount` over just those filtered entries, via `total()`), and `by_category` (a dict mapping category \u2192 summed amount, via `sum_by_category()`, again over only the filtered entries).\n\nNote that `day` on each `Entry` is already a `date` object by the time these functions run \u2014 the actual string parsing happens earlier, in `parse_date`, when an entry is created via `add_entry`.\n\nInvalid date strings: `parse_date` is where validation happens. It splits the string on `-` and requires exactly 3 parts, each consisting only of digits (`str.isdigit()`); if that check fails, it raises `ValueError(f\"bad date: {text!r}\")` immediately. If the format looks right but the numbers are out of range (e.g. month 13, day 32, or Feb 30), the `date(y, m, d)` constructor itself raises `ValueError` (with its own message, not the \"bad date\" one). Either way, `add_entry` doesn't catch this \u2014 the exception propagates straight up out of `add_entry`, so the entry is never appended and no partial/invalid `Entry` ends up in the ledger. There's no validation error handling inside `monthly_report`/`entries_in_month` themselves since they only ever operate on already-parsed `date` objects.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.068472
  },
  {
   "run": "claude-sonnet-5__ponytail__T3__r2__8cc741",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "ponytail",
   "rep": 2,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 767,
   "input": 6,
   "cache_read": 76862,
   "cache_write": 11410,
   "warmup": null,
   "cache_write_1h": 11410,
   "cost_usd": 0.0686944,
   "wall_s": 9.3,
   "turns": 3,
   "prompt_last": 30051,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 42,
   "commands": null,
   "answer": "**Monthly totals:** `monthly_report(entries, year, month)` (core.py:51-53) calls `entries_in_month` to filter entries where `e.day.year == year and e.day.month == month` (core.py:40-41), then passes that filtered list to `total()`, which just sums `e.amount` across the entries (core.py:31-32). So the monthly total is a plain sum of amounts for entries whose stored `date` matches the given year/month \u2014 no averaging, no category weighting. `monthly_report` also returns the count and a per-category breakdown (`sum_by_category`) of that same filtered list.\n\n**Invalid date strings:** dates are parsed once, at entry-creation time, by `parse_date` (core.py:15-21), called from `add_entry` (core.py:27). It splits on `-`, requires exactly 3 parts that are all digit strings, and raises `ValueError(f\"bad date: {text!r}\")` if not. If the split passes but the numbers are out of range (e.g. month 13, day 31 in April), the underlying `date(y, m, d)` constructor call raises its own `ValueError` (e.g. \"month must be in 1..12\"), which propagates uncaught. Either way, `add_entry` raises before the entry is ever appended to the list \u2014 so an invalid date string means no `Entry` is created, nothing enters `entries`, and it therefore can't skew any later `total`/`monthly_report` call. There's no catching/logging of bad dates anywhere in this file; the caller of `add_entry` is responsible for handling the exception.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.068694
  },
  {
   "run": "claude-sonnet-5__ponytail__T3__r3__69adef",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "ponytail",
   "rep": 3,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 917,
   "input": 6,
   "cache_read": 76871,
   "cache_write": 11420,
   "warmup": null,
   "cache_write_1h": 11420,
   "cost_usd": 0.0702362,
   "wall_s": 11.8,
   "turns": 3,
   "prompt_last": 30061,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 32,
   "commands": null,
   "answer": "## Monthly totals\n\n`monthly_report(entries, year, month)` computes the total in two steps:\n\n1. `entries_in_month` filters the full entry list down to only those `Entry` objects whose `day.year` and `day.month` match the requested year/month (day-of-month is irrelevant).\n2. `total()` sums the `.amount` field over that filtered list (plain `sum()`, so an empty month yields `0.0`).\n\nSeparately, `sum_by_category` walks the same filtered list and accumulates a `{category: amount}` dict, so `monthly_report` returns `count`, `total`, and `by_category` all derived from the same month-filtered subset.\n\nNote: `total(entries)` sums *all* entries passed to it \u2014 it has no month awareness itself. It only reflects a single month because `monthly_report` first narrows `entries` via `entries_in_month`.\n\n## Invalid date strings\n\nDates only enter the ledger through `add_entry`, which calls `parse_date(day)` before constructing the `Entry`. `parse_date` is strict:\n\n- It splits on `-` and requires exactly 3 parts, each all-digit (`str.isdigit()`).\n- If that check fails (wrong number of parts, non-numeric parts, empty string, extra whitespace, etc.), it raises `ValueError(f\"bad date: {text!r}\")` immediately \u2014 the entry is never appended, and `entries` is left unchanged.\n- If the split/digit check passes but the numbers aren't a real calendar date (e.g. month 13, day 31 in February, or a `day.isdigit()` string like `\"-1\"` won't even pass since `\"-1\".isdigit()` is `False`... but something like `2024-02-30` will), `date(y, m, d)` itself raises `ValueError` (e.g. `\"day is out of range for month\"`), which propagates uncaught out of `parse_date` and `add_entry`.\n\nSo in both cases the caller gets an uncaught `ValueError` and no partial entry is added \u2014 there's no silent failure or defaulting.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.070236
  },
  {
   "run": "claude-sonnet-5__caveman__T4__r1__0aab4e",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T4",
   "skill": "caveman",
   "rep": 1,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 3191,
   "input": 14,
   "cache_read": 290224,
   "cache_write": 30429,
   "warmup": null,
   "cache_write_1h": 30429,
   "cost_usd": 0.21169880000000005,
   "wall_s": 41.6,
   "turns": 10,
   "prompt_last": 49070,
   "lines_added": 59,
   "lines_deleted": 0,
   "lines": 59,
   "files": 1,
   "thinking": 1573,
   "commands": null,
   "api_cost": 0.211699
  },
  {
   "run": "claude-sonnet-5__caveman__T4__r2__e0c2e8",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T4",
   "skill": "caveman",
   "rep": 2,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 3258,
   "input": 14,
   "cache_read": 290465,
   "cache_write": 30529,
   "warmup": null,
   "cache_write_1h": 30529,
   "cost_usd": 0.212817,
   "wall_s": 32.4,
   "turns": 10,
   "prompt_last": 49170,
   "lines_added": 50,
   "lines_deleted": 0,
   "lines": 50,
   "files": 1,
   "thinking": 1734,
   "commands": null,
   "api_cost": 0.212817
  },
  {
   "run": "claude-sonnet-5__caveman__T4__r3__d92321",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T4",
   "skill": "caveman",
   "rep": 3,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 3124,
   "input": 18,
   "cache_read": 315519,
   "cache_write": 22550,
   "warmup": null,
   "cache_write_1h": 22550,
   "cost_usd": 0.1845798,
   "wall_s": 28.4,
   "turns": 11,
   "prompt_last": 41191,
   "lines_added": 57,
   "lines_deleted": 0,
   "lines": 57,
   "files": 1,
   "thinking": 1379,
   "commands": null,
   "api_cost": 0.18458,
   "light": true
  },
  {
   "run": "claude-sonnet-5__control__T4__r1__826f80",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T4",
   "skill": "control",
   "rep": 1,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 3822,
   "input": 14,
   "cache_read": 270318,
   "cache_write": 27290,
   "warmup": null,
   "cache_write_1h": 27290,
   "cost_usd": 0.2014716,
   "wall_s": 34.7,
   "turns": 10,
   "prompt_last": 45931,
   "lines_added": 65,
   "lines_deleted": 0,
   "lines": 65,
   "files": 1,
   "thinking": 2066,
   "commands": null,
   "api_cost": 0.201472
  },
  {
   "run": "claude-sonnet-5__control__T4__r2__27c293",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T4",
   "skill": "control",
   "rep": 2,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 3891,
   "input": 16,
   "cache_read": 314305,
   "cache_write": 27975,
   "warmup": null,
   "cache_write_1h": 27975,
   "cost_usd": 0.21370299999999998,
   "wall_s": 35.3,
   "turns": 11,
   "prompt_last": 46616,
   "lines_added": 59,
   "lines_deleted": 0,
   "lines": 59,
   "files": 1,
   "thinking": 1938,
   "commands": null,
   "api_cost": 0.213703
  },
  {
   "run": "claude-sonnet-5__control__T4__r3__f933b1",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T4",
   "skill": "control",
   "rep": 3,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 3161,
   "input": 12,
   "cache_read": 221888,
   "cache_write": 24810,
   "warmup": null,
   "cache_write_1h": 24810,
   "cost_usd": 0.17525159999999998,
   "wall_s": 28.4,
   "turns": 9,
   "prompt_last": 43451,
   "lines_added": 62,
   "lines_deleted": 0,
   "lines": 62,
   "files": 1,
   "thinking": 1560,
   "commands": null,
   "api_cost": 0.175252
  },
  {
   "run": "claude-sonnet-5__karpathy__T4__r1__9b7129",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T4",
   "skill": "karpathy",
   "rep": 1,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 3636,
   "input": 12,
   "cache_read": 231205,
   "cache_write": 27773,
   "warmup": null,
   "cache_write_1h": 27773,
   "cost_usd": 0.19371699999999997,
   "wall_s": 39.5,
   "turns": 10,
   "prompt_last": 46414,
   "lines_added": 56,
   "lines_deleted": 0,
   "lines": 56,
   "files": 1,
   "thinking": 1971,
   "commands": null,
   "api_cost": 0.193717
  },
  {
   "run": "claude-sonnet-5__karpathy__T4__r2__ee640a",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T4",
   "skill": "karpathy",
   "rep": 2,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 4310,
   "input": 14,
   "cache_read": 224947,
   "cache_write": 20316,
   "warmup": null,
   "cache_write_1h": 20316,
   "cost_usd": 0.1693814,
   "wall_s": 37.3,
   "turns": 11,
   "prompt_last": 38957,
   "lines_added": 57,
   "lines_deleted": 0,
   "lines": 57,
   "files": 1,
   "thinking": 2510,
   "commands": null,
   "api_cost": 0.169381,
   "light": true
  },
  {
   "run": "claude-sonnet-5__karpathy__T4__r3__f137b6",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T4",
   "skill": "karpathy",
   "rep": 3,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 3566,
   "input": 18,
   "cache_read": 368655,
   "cache_write": 29444,
   "warmup": null,
   "cache_write_1h": 29444,
   "cost_usd": 0.227203,
   "wall_s": 32.9,
   "turns": 12,
   "prompt_last": 48085,
   "lines_added": 55,
   "lines_deleted": 0,
   "lines": 55,
   "files": 1,
   "thinking": 1632,
   "commands": null,
   "api_cost": 0.227203
  },
  {
   "run": "claude-sonnet-5__placebo__T4__r1__7b98ae",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T4",
   "skill": "placebo",
   "rep": 1,
   "round": 4,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 3077,
   "input": 18,
   "cache_read": 309811,
   "cache_write": 21960,
   "warmup": null,
   "cache_write_1h": 21960,
   "cost_usd": 0.18060819999999997,
   "wall_s": 30.3,
   "turns": 13,
   "prompt_last": 40601,
   "lines_added": 71,
   "lines_deleted": 0,
   "lines": 71,
   "files": 2,
   "thinking": 671,
   "commands": null,
   "api_cost": 0.180608,
   "light": true
  },
  {
   "run": "claude-sonnet-5__placebo__T4__r2__944f08",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T4",
   "skill": "placebo",
   "rep": 2,
   "round": 4,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 5249,
   "input": 18,
   "cache_read": 374415,
   "cache_write": 30955,
   "warmup": null,
   "cache_write_1h": 30955,
   "cost_usd": 0.251229,
   "wall_s": 48.4,
   "turns": 12,
   "prompt_last": 49596,
   "lines_added": 63,
   "lines_deleted": 0,
   "lines": 63,
   "files": 1,
   "thinking": 3016,
   "commands": null,
   "api_cost": 0.251229
  },
  {
   "run": "claude-sonnet-5__placebo__T4__r3__d020a6",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T4",
   "skill": "placebo",
   "rep": 3,
   "round": 4,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 4895,
   "input": 12,
   "cache_read": 231568,
   "cache_write": 28500,
   "warmup": null,
   "cache_write_1h": 28500,
   "cost_usd": 0.2092876,
   "wall_s": 54.2,
   "turns": 10,
   "prompt_last": 47141,
   "lines_added": 63,
   "lines_deleted": 0,
   "lines": 63,
   "files": 1,
   "thinking": 3068,
   "commands": null,
   "api_cost": 0.209288
  },
  {
   "run": "claude-sonnet-5__ponytail__T4__r1__cd5c7f",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T4",
   "skill": "ponytail",
   "rep": 1,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 3761,
   "input": 16,
   "cache_read": 326607,
   "cache_write": 29940,
   "warmup": null,
   "cache_write_1h": 29940,
   "cost_usd": 0.2227234,
   "wall_s": 38.4,
   "turns": 11,
   "prompt_last": 48581,
   "lines_added": 48,
   "lines_deleted": 0,
   "lines": 48,
   "files": 1,
   "thinking": 1933,
   "commands": null,
   "api_cost": 0.222723
  },
  {
   "run": "claude-sonnet-5__ponytail__T4__r2__fb3371",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T4",
   "skill": "ponytail",
   "rep": 2,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 3267,
   "input": 20,
   "cache_read": 413165,
   "cache_write": 29397,
   "warmup": null,
   "cache_write_1h": 29397,
   "cost_usd": 0.23293099999999994,
   "wall_s": 34.9,
   "turns": 13,
   "prompt_last": 48038,
   "lines_added": 47,
   "lines_deleted": 0,
   "lines": 47,
   "files": 1,
   "thinking": 1097,
   "commands": null,
   "api_cost": 0.232931
  },
  {
   "run": "claude-sonnet-5__ponytail__T4__r3__6d398c",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T4",
   "skill": "ponytail",
   "rep": 3,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 4661,
   "input": 18,
   "cache_read": 371077,
   "cache_write": 29653,
   "warmup": null,
   "cache_write_1h": 29653,
   "cost_usd": 0.2394734,
   "wall_s": 45.2,
   "turns": 12,
   "prompt_last": 48294,
   "lines_added": 41,
   "lines_deleted": 0,
   "lines": 41,
   "files": 1,
   "thinking": 2799,
   "commands": null,
   "api_cost": 0.239473
  },
  {
   "run": "claude-sonnet-5-5__caveman__T1__r1__591169",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T1",
   "skill": "caveman",
   "rep": 1,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 530,
   "input": 10,
   "cache_read": 131401,
   "cache_write": 23310,
   "warmup": null,
   "cache_write_1h": 23310,
   "cost_usd": 0.1248402,
   "wall_s": 8.5,
   "turns": 5,
   "prompt_last": 33546,
   "lines_added": 7,
   "lines_deleted": 1,
   "lines": 8,
   "files": 2,
   "thinking": 0,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.12484
  },
  {
   "run": "claude-sonnet-5-5__caveman__T1__r2__9befd4",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T1",
   "skill": "caveman",
   "rep": 2,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 510,
   "input": 10,
   "cache_read": 131057,
   "cache_write": 22383,
   "warmup": null,
   "cache_write_1h": 22383,
   "cost_usd": 0.12086340000000001,
   "wall_s": 6.7,
   "turns": 5,
   "prompt_last": 32619,
   "lines_added": 7,
   "lines_deleted": 1,
   "lines": 8,
   "files": 2,
   "thinking": 0,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.120863
  },
  {
   "run": "claude-sonnet-5-5__caveman__T1__r3__a2fcec",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T1",
   "skill": "caveman",
   "rep": 3,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 625,
   "input": 10,
   "cache_read": 131213,
   "cache_write": 22519,
   "warmup": null,
   "cache_write_1h": 22519,
   "cost_usd": 0.12258859999999999,
   "wall_s": 7.4,
   "turns": 5,
   "prompt_last": 32755,
   "lines_added": 6,
   "lines_deleted": 0,
   "lines": 6,
   "files": 1,
   "thinking": 0,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.122589
  },
  {
   "run": "claude-sonnet-5-5__control__T1__r1__9e2317",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T1",
   "skill": "control",
   "rep": 1,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 652,
   "input": 10,
   "cache_read": 114904,
   "cache_write": 17368,
   "warmup": null,
   "cache_write_1h": 17368,
   "cost_usd": 0.0989928,
   "wall_s": 8.6,
   "turns": 5,
   "prompt_last": 27604,
   "lines_added": 6,
   "lines_deleted": 0,
   "lines": 6,
   "files": 1,
   "thinking": 0,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.098993
  },
  {
   "run": "claude-sonnet-5-5__control__T1__r2__5d10cb",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T1",
   "skill": "control",
   "rep": 2,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 498,
   "input": 10,
   "cache_read": 114744,
   "cache_write": 17275,
   "warmup": null,
   "cache_write_1h": 17275,
   "cost_usd": 0.09704880000000002,
   "wall_s": 6.9,
   "turns": 5,
   "prompt_last": 27511,
   "lines_added": 6,
   "lines_deleted": 0,
   "lines": 6,
   "files": 1,
   "thinking": 0,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.097049
  },
  {
   "run": "claude-sonnet-5-5__control__T1__r3__64fa05",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T1",
   "skill": "control",
   "rep": 3,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 506,
   "input": 10,
   "cache_read": 114732,
   "cache_write": 17269,
   "warmup": null,
   "cache_write_1h": 17269,
   "cost_usd": 0.0971024,
   "wall_s": 9.8,
   "turns": 5,
   "prompt_last": 27505,
   "lines_added": 6,
   "lines_deleted": 0,
   "lines": 6,
   "files": 1,
   "thinking": 0,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.097102
  },
  {
   "run": "claude-sonnet-5-5__karpathy__T1__r1__c1d152",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T1",
   "skill": "karpathy",
   "rep": 1,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 792,
   "input": 10,
   "cache_read": 121315,
   "cache_write": 19195,
   "warmup": null,
   "cache_write_1h": 19195,
   "cost_usd": 0.10898300000000001,
   "wall_s": 8.0,
   "turns": 5,
   "prompt_last": 29431,
   "lines_added": 5,
   "lines_deleted": 0,
   "lines": 5,
   "files": 1,
   "thinking": 0,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.108983
  },
  {
   "run": "claude-sonnet-5-5__karpathy__T1__r2__143825",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T1",
   "skill": "karpathy",
   "rep": 2,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 687,
   "input": 10,
   "cache_read": 121267,
   "cache_write": 19125,
   "warmup": null,
   "cache_write_1h": 19125,
   "cost_usd": 0.1076434,
   "wall_s": 7.4,
   "turns": 5,
   "prompt_last": 29361,
   "lines_added": 5,
   "lines_deleted": 0,
   "lines": 5,
   "files": 1,
   "thinking": 0,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.107643
  },
  {
   "run": "claude-sonnet-5-5__karpathy__T1__r3__1d856a",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T1",
   "skill": "karpathy",
   "rep": 3,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 768,
   "input": 10,
   "cache_read": 121325,
   "cache_write": 19224,
   "warmup": null,
   "cache_write_1h": 19224,
   "cost_usd": 0.10886100000000001,
   "wall_s": 8.4,
   "turns": 5,
   "prompt_last": 29460,
   "lines_added": 6,
   "lines_deleted": 0,
   "lines": 6,
   "files": 1,
   "thinking": 27,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.108861
  },
  {
   "run": "claude-sonnet-5-5__placebo__T1__r1__249611",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T1",
   "skill": "placebo",
   "rep": 1,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 844,
   "input": 10,
   "cache_read": 119695,
   "cache_write": 18735,
   "warmup": null,
   "cache_write_1h": 18735,
   "cost_usd": 0.107339,
   "wall_s": 10.1,
   "turns": 5,
   "prompt_last": 28971,
   "lines_added": 7,
   "lines_deleted": 1,
   "lines": 8,
   "files": 2,
   "thinking": 0,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.107339
  },
  {
   "run": "claude-sonnet-5-5__placebo__T1__r2__0ea028",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T1",
   "skill": "placebo",
   "rep": 2,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 858,
   "input": 10,
   "cache_read": 119074,
   "cache_write": 18764,
   "warmup": null,
   "cache_write_1h": 18764,
   "cost_usd": 0.1074708,
   "wall_s": 9.2,
   "turns": 5,
   "prompt_last": 29000,
   "lines_added": 7,
   "lines_deleted": 1,
   "lines": 8,
   "files": 2,
   "thinking": 0,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.107471
  },
  {
   "run": "claude-sonnet-5-5__placebo__T1__r3__3df074",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T1",
   "skill": "placebo",
   "rep": 3,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 830,
   "input": 10,
   "cache_read": 119705,
   "cache_write": 18740,
   "warmup": null,
   "cache_write_1h": 18740,
   "cost_usd": 0.10722100000000001,
   "wall_s": 8.2,
   "turns": 5,
   "prompt_last": 28976,
   "lines_added": 7,
   "lines_deleted": 1,
   "lines": 8,
   "files": 2,
   "thinking": 0,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.107221
  },
  {
   "run": "claude-sonnet-5-5__ponytail__T1__r1__7a18ea",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T1",
   "skill": "ponytail",
   "rep": 1,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 574,
   "input": 10,
   "cache_read": 120741,
   "cache_write": 19251,
   "warmup": null,
   "cache_write_1h": 19251,
   "cost_usd": 0.1069122,
   "wall_s": 7.0,
   "turns": 5,
   "prompt_last": 29487,
   "lines_added": 6,
   "lines_deleted": 0,
   "lines": 6,
   "files": 1,
   "thinking": 0,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.106912
  },
  {
   "run": "claude-sonnet-5-5__ponytail__T1__r2__d74d20",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T1",
   "skill": "ponytail",
   "rep": 2,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 691,
   "input": 10,
   "cache_read": 121304,
   "cache_write": 19154,
   "warmup": null,
   "cache_write_1h": 19154,
   "cost_usd": 0.1078068,
   "wall_s": 9.1,
   "turns": 5,
   "prompt_last": 29390,
   "lines_added": 6,
   "lines_deleted": 0,
   "lines": 6,
   "files": 1,
   "thinking": 0,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.107807
  },
  {
   "run": "claude-sonnet-5-5__ponytail__T1__r3__5a3dc2",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T1",
   "skill": "ponytail",
   "rep": 3,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 784,
   "input": 10,
   "cache_read": 121755,
   "cache_write": 19621,
   "warmup": null,
   "cache_write_1h": 19621,
   "cost_usd": 0.110695,
   "wall_s": 10.8,
   "turns": 5,
   "prompt_last": 29857,
   "lines_added": 6,
   "lines_deleted": 0,
   "lines": 6,
   "files": 1,
   "thinking": 0,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.110695
  },
  {
   "run": "claude-sonnet-5-5__caveman__T2__r1__8b7509",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T2",
   "skill": "caveman",
   "rep": 1,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 616,
   "input": 8,
   "cache_read": 71163,
   "cache_write": 14120,
   "warmup": null,
   "cache_write_1h": 14120,
   "cost_usd": 0.0768886,
   "wall_s": 6.0,
   "turns": 4,
   "prompt_last": 24356,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 106,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.076889,
   "light": true
  },
  {
   "run": "claude-sonnet-5-5__caveman__T2__r2__63d293",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T2",
   "skill": "caveman",
   "rep": 2,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 554,
   "input": 8,
   "cache_read": 98653,
   "cache_write": 22017,
   "warmup": null,
   "cache_write_1h": 22017,
   "cost_usd": 0.1133546,
   "wall_s": 6.5,
   "turns": 4,
   "prompt_last": 32253,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 0,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.113355
  },
  {
   "run": "claude-sonnet-5-5__caveman__T2__r3__ceaea3",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T2",
   "skill": "caveman",
   "rep": 3,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 634,
   "input": 8,
   "cache_read": 98659,
   "cache_write": 22113,
   "warmup": null,
   "cache_write_1h": 22113,
   "cost_usd": 0.1145398,
   "wall_s": 7.2,
   "turns": 4,
   "prompt_last": 32349,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 107,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.11454
  },
  {
   "run": "claude-sonnet-5-5__control__T2__r1__df7130",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T2",
   "skill": "control",
   "rep": 1,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 771,
   "input": 10,
   "cache_read": 114539,
   "cache_write": 17260,
   "warmup": null,
   "cache_write_1h": 17260,
   "cost_usd": 0.09967780000000001,
   "wall_s": 8.0,
   "turns": 5,
   "prompt_last": 27496,
   "lines_added": 3,
   "lines_deleted": 3,
   "lines": 6,
   "files": 1,
   "thinking": 0,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.099678
  },
  {
   "run": "claude-sonnet-5-5__control__T2__r2__6bb3d2",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T2",
   "skill": "control",
   "rep": 2,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 628,
   "input": 10,
   "cache_read": 115035,
   "cache_write": 17734,
   "warmup": null,
   "cache_write_1h": 17734,
   "cost_usd": 0.10024300000000001,
   "wall_s": 11.4,
   "turns": 5,
   "prompt_last": 27970,
   "lines_added": 3,
   "lines_deleted": 3,
   "lines": 6,
   "files": 1,
   "thinking": 0,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.100243
  },
  {
   "run": "claude-sonnet-5-5__control__T2__r3__d48363",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T2",
   "skill": "control",
   "rep": 3,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 505,
   "input": 10,
   "cache_read": 114869,
   "cache_write": 17350,
   "warmup": null,
   "cache_write_1h": 17350,
   "cost_usd": 0.09744380000000001,
   "wall_s": 6.9,
   "turns": 5,
   "prompt_last": 27586,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 0,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.097444
  },
  {
   "run": "claude-sonnet-5-5__karpathy__T2__r1__a7283b",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T2",
   "skill": "karpathy",
   "rep": 1,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 660,
   "input": 8,
   "cache_read": 92028,
   "cache_write": 18666,
   "warmup": null,
   "cache_write_1h": 18666,
   "cost_usd": 0.09968560000000001,
   "wall_s": 7.6,
   "turns": 4,
   "prompt_last": 28902,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 40,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.099686
  },
  {
   "run": "claude-sonnet-5-5__karpathy__T2__r2__edc7ab",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T2",
   "skill": "karpathy",
   "rep": 2,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 634,
   "input": 8,
   "cache_read": 92046,
   "cache_write": 18671,
   "warmup": null,
   "cache_write_1h": 18671,
   "cost_usd": 0.0994492,
   "wall_s": 9.2,
   "turns": 4,
   "prompt_last": 28907,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 37,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.099449
  },
  {
   "run": "claude-sonnet-5-5__karpathy__T2__r3__3f44a9",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T2",
   "skill": "karpathy",
   "rep": 3,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 500,
   "input": 8,
   "cache_read": 69201,
   "cache_write": 12351,
   "warmup": null,
   "cache_write_1h": 12351,
   "cost_usd": 0.0682602,
   "wall_s": 6.2,
   "turns": 4,
   "prompt_last": 22587,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 49,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.06826,
   "light": true
  },
  {
   "run": "claude-sonnet-5-5__placebo__T2__r1__63117c",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T2",
   "skill": "placebo",
   "rep": 1,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 884,
   "input": 10,
   "cache_read": 118907,
   "cache_write": 18371,
   "warmup": null,
   "cache_write_1h": 18371,
   "cost_usd": 0.10612540000000001,
   "wall_s": 8.7,
   "turns": 5,
   "prompt_last": 28607,
   "lines_added": 2,
   "lines_deleted": 1,
   "lines": 3,
   "files": 1,
   "thinking": 63,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.106125
  },
  {
   "run": "claude-sonnet-5-5__placebo__T2__r2__e25345",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T2",
   "skill": "placebo",
   "rep": 2,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 955,
   "input": 10,
   "cache_read": 118833,
   "cache_write": 19194,
   "warmup": null,
   "cache_write_1h": 19194,
   "cost_usd": 0.11011259999999999,
   "wall_s": 16.5,
   "turns": 5,
   "prompt_last": 29430,
   "lines_added": 4,
   "lines_deleted": 3,
   "lines": 7,
   "files": 1,
   "thinking": 98,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.110113
  },
  {
   "run": "claude-sonnet-5-5__placebo__T2__r3__8c4e11",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T2",
   "skill": "placebo",
   "rep": 3,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 772,
   "input": 8,
   "cache_read": 90974,
   "cache_write": 18184,
   "warmup": null,
   "cache_write_1h": 18184,
   "cost_usd": 0.0986668,
   "wall_s": 10.4,
   "turns": 4,
   "prompt_last": 28420,
   "lines_added": 2,
   "lines_deleted": 1,
   "lines": 3,
   "files": 1,
   "thinking": 35,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.098667
  },
  {
   "run": "claude-sonnet-5-5__ponytail__T2__r1__c6cb9f",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T2",
   "skill": "ponytail",
   "rep": 1,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 1316,
   "input": 12,
   "cache_read": 151183,
   "cache_write": 19994,
   "warmup": null,
   "cache_write_1h": 19994,
   "cost_usd": 0.12339660000000001,
   "wall_s": 12.1,
   "turns": 6,
   "prompt_last": 30230,
   "lines_added": 3,
   "lines_deleted": 5,
   "lines": 8,
   "files": 1,
   "thinking": 196,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.123397
  },
  {
   "run": "claude-sonnet-5-5__ponytail__T2__r2__4305a6",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T2",
   "skill": "ponytail",
   "rep": 2,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 886,
   "input": 8,
   "cache_read": 92585,
   "cache_write": 19380,
   "warmup": null,
   "cache_write_1h": 19380,
   "cost_usd": 0.10491299999999999,
   "wall_s": 10.6,
   "turns": 4,
   "prompt_last": 29616,
   "lines_added": 2,
   "lines_deleted": 2,
   "lines": 4,
   "files": 1,
   "thinking": 183,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.104913
  },
  {
   "run": "claude-sonnet-5-5__ponytail__T2__r3__e6f3f7",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T2",
   "skill": "ponytail",
   "rep": 3,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 728,
   "input": 8,
   "cache_read": 92107,
   "cache_write": 18822,
   "warmup": null,
   "cache_write_1h": 18822,
   "cost_usd": 0.10100540000000001,
   "wall_s": 10.3,
   "turns": 4,
   "prompt_last": 29058,
   "lines_added": 2,
   "lines_deleted": 2,
   "lines": 4,
   "files": 1,
   "thinking": 113,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.101005
  },
  {
   "run": "claude-sonnet-5-5__caveman__T3__r1__99f08c",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "caveman",
   "rep": 1,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 694,
   "input": 4,
   "cache_read": 28582,
   "cache_write": 8919,
   "warmup": null,
   "cache_write_1h": 8919,
   "cost_usd": 0.0483404,
   "wall_s": 6.6,
   "turns": 2,
   "prompt_last": 19155,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` first calls `entries_in_month`. That keeps only entries whose `day.year` and `day.month` both match the arguments. Then it builds a dict with three values:\n- `count`: the number of matching entries.\n- `total`: the sum of their `amount` values, from `total`.\n- `by_category`: a dict from `sum_by_category`, which adds up amounts per category. Categories are lowercased when an entry is added, so `Food` and `food` merge.\n\nNothing is rounded. Amounts are floats, so sums can carry small floating-point error. A month with no entries gives `count` 0, `total` 0 and an empty `by_category`.\n\n**Invalid date strings.** `parse_date` handles the strings. It splits the text on `-`. It raises `ValueError(\"bad date: ...\")` if there are not exactly three parts, or if any part is not all digits. A negative year or a sign character fails the digit check. It then converts the parts to ints and calls `date(y, m, d)`. Out-of-range values, such as month 13 or Feb 30, make `date` raise its own `ValueError`, with a different message. Both errors are the same type, so callers can catch `ValueError`.\n\n**Effect on totals.** `add_entry` calls `parse_date` while building the `Entry`. A bad date therefore raises before `append`, and the entry is not added. `add_entry` checks `amount <= 0` first, so a bad amount raises before the date is parsed. The report functions never see date strings, only parsed `date` objects. An invalid date cannot reach a monthly total.\n\n**Edge case.** `isdigit()` also accepts some non-ASCII digit characters, such as `\u00b2`. `int()` then raises a `ValueError` that has no \"bad date\" message. It is still a `ValueError`.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.04834
  },
  {
   "run": "claude-sonnet-5-5__caveman__T3__r2__7a306a",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "caveman",
   "rep": 2,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 603,
   "input": 4,
   "cache_read": 28579,
   "cache_write": 8916,
   "warmup": null,
   "cache_write_1h": 8916,
   "cost_usd": 0.047417799999999996,
   "wall_s": 7.3,
   "turns": 2,
   "prompt_last": 19152,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` computes them. It calls `entries_in_month`, which keeps entries whose `day.year` and `day.month` both match the arguments. It then builds a dict with three values:\n- `count`: the number of matching entries.\n- `total`: the sum of their amounts, from `total`.\n- `by_category`: a dict of per-category sums, from `sum_by_category`.\n\nAn empty month gives `count` 0, `total` 0 and an empty `by_category`. Amounts are plain floats, so rounding error can build up.\n\n**Invalid date.** Entries store `date` objects, not strings. `add_entry` converts the string with `parse_date`.\n\n`parse_date` splits the text on `-`. It raises `ValueError(f\"bad date: {text!r}\")` in two cases:\n- The text does not have exactly 3 parts.\n- Any part is not all digits. A negative year, a sign, or spaces fail this check.\n\nWell-formed text with impossible values, such as `2024-02-30` or month 13, passes that check. The `date(y, m, d)` constructor then raises its own `ValueError`, with a different message. Non-string input such as `None` fails at `text.split` with `AttributeError`, not `ValueError`.\n\n`add_entry` checks `amount <= 0` before it parses the date. A bad amount raises first, even if the date is also bad. On any error, nothing is appended, because `entries.append` runs only after `parse_date` succeeds. The caller sees the exception. Nothing in this file catches it.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.047418
  },
  {
   "run": "claude-sonnet-5-5__caveman__T3__r3__81b4b3",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "caveman",
   "rep": 3,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 621,
   "input": 4,
   "cache_read": 28584,
   "cache_write": 8922,
   "warmup": null,
   "cache_write_1h": 8922,
   "cost_usd": 0.0476228,
   "wall_s": 6.3,
   "turns": 2,
   "prompt_last": 19158,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` calls `entries_in_month`. That keeps only entries whose `day.year` and `day.month` both match the arguments. It then builds a dict with three values:\n- `count`: number of matching entries.\n- `total`: sum of their `amount` values, from `total`.\n- `by_category`: per-category sums, from `sum_by_category`. Categories are lowercased when entries are added, so `Food` and `food` merge.\n\nIf no entries match, `count` is 0, `total` is 0 and `by_category` is `{}`. Amounts are plain floats, so sums can show float rounding error.\n\n**Invalid date strings.** Dates are parsed only in `add_entry`, through `parse_date`. `parse_date` splits the text on `-`. It raises `ValueError(f\"bad date: {text!r}\")` in two cases:\n- The text does not have exactly 3 parts.\n- Any part is not all digits.\n\nTwo things follow from this:\n- Text that has three digit parts but is not a real date, such as `2024-13-45`, passes that check. `date(y, m, d)` then raises its own `ValueError`, such as \"month must be in 1..12\". The message differs from \"bad date\".\n- A non-string input, such as `None`, fails at `text.split` with `AttributeError`, not `ValueError`.\n\n`add_entry` checks `amount <= 0` before it parses the date. A bad amount therefore raises first. When the date is invalid, `append` never runs, so the entry is not added and the list stays unchanged. The error propagates to the caller, and nothing catches it.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.047623
  },
  {
   "run": "claude-sonnet-5-5__control__T3__r1__863973",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "control",
   "rep": 1,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 909,
   "input": 4,
   "cache_read": 25845,
   "cache_write": 6178,
   "warmup": null,
   "cache_write_1h": 6178,
   "cost_usd": 0.038979,
   "wall_s": 8.7,
   "turns": 2,
   "prompt_last": 16414,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**How monthly totals are computed**\n\n`monthly_report(entries, year, month)` in `ledger/core.py:51` does the work in three steps:\n\n1. `entries_in_month` (line 40) keeps only the entries whose `day.year` and `day.month` both match the arguments.\n2. `total` (line 31) adds up the `amount` of those entries.\n3. `sum_by_category` (line 44) builds a dict that maps each category to its summed amount for that month.\n\n`monthly_report` returns a dict with three keys:\n- `count`: the number of entries in that month.\n- `total`: the sum of their amounts.\n- `by_category`: the per-category sums.\n\nSome details:\n- Amounts are floats, so the sums are ordinary floating-point sums with no rounding.\n- Categories are lowercased when an entry is added (`add_entry`, line 27), so \"Food\" and \"food\" fall into the same bucket.\n- A month with no entries gives `count` 0, `total` 0 (an int, since `sum` of an empty sequence is 0) and an empty `by_category` dict.\n\n**What happens with an invalid date string**\n\nDates are parsed only when an entry is added. `add_entry` calls `parse_date` (line 15), which splits the string on `-`. It raises `ValueError(f\"bad date: ...\")` if the result isn't exactly three parts or if any part isn't all digits. Strings like `\"2024/01/05\"`, `\"2024-1\"`, `\"2024-01-xx\"` and `\"-2024-01-05\"` fail this way.\n\nA string that passes that check can still fail inside `date(y, m, d)`, which raises its own `ValueError`. `\"2024-13-40\"` does this, with a message like \"month must be in 1..12\". So callers get a `ValueError` either way, but the messages differ.\n\nSome consequences:\n- `add_entry` checks `amount <= 0` first, and only then parses the date. A call with both a bad amount and a bad date reports the amount error.\n- Nothing is appended to `entries` when the date is invalid, because the `Entry` is built and appended in a single expression. The list is left unchanged.\n- `parse_date` is lenient about digit count. `\"24-1-5\"` parses as year 24, January 5. `str.isdigit()` also accepts some non-ASCII digit characters, which `int()` then converts.\n- The monthly functions never see date strings, only parsed `date` objects. An invalid date can't reach the report, because it is rejected when the entry is added.\n\nI didn't change any files.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.038979
  },
  {
   "run": "claude-sonnet-5-5__control__T3__r2__66f5d7",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "control",
   "rep": 2,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 864,
   "input": 4,
   "cache_read": 25932,
   "cache_write": 6268,
   "warmup": null,
   "cache_write_1h": 6268,
   "cost_usd": 0.0389064,
   "wall_s": 7.5,
   "turns": 2,
   "prompt_last": 16504,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` does the work in three steps:\n\n1. It calls `entries_in_month`. That keeps only the entries whose `day.year` and `day.month` both match the arguments.\n2. It counts those entries and passes them to `total`. `total` adds up every entry's `amount`, so the result is a plain sum of floats.\n3. It passes the same filtered list to `sum_by_category`. That builds a dict mapping each category to its summed amount, starting from `0.0` for categories it hasn't seen yet.\n\nThe report comes back as a dict with `count`, `total` and `by_category`. If no entries fall in the month, you get `count` 0, `total` 0 (an int, because `sum` of an empty sequence is `0`) and an empty `by_category`. Categories are already lowercased in `add_entry`, so `\"Food\"` and `\"food\"` end up in the same bucket.\n\n**Invalid dates.** Dates are parsed only when an entry is added, by `parse_date` inside `add_entry`. `monthly_report` never sees a date string, because entries store real `date` objects.\n\n`parse_date` splits the string on `-`. It raises `ValueError(f\"bad date: ...\")` if there aren't exactly three parts or if any part isn't all digits. That check catches things like `\"2024/01/05\"`, `\"2024-1\"`, `\"2024-01-xx\"` and negative numbers.\n\nA string that passes that check but isn't a real calendar date, such as `\"2024-13-40\"`, reaches `date(y, m, d)`. That raises its own `ValueError` (for example \"month must be in 1..12\"), and the message is different from the \"bad date\" one. Either way the caller gets a `ValueError`.\n\nIn `add_entry`, the amount check runs first, then the date is parsed while the `Entry` is being built. If the date is invalid, the exception propagates before `entries.append` runs, so the list is left unchanged. Nothing is caught or swallowed.\n\nA few edge cases:\n- `\"2024-01-05 \"` (trailing space) fails the digit check.\n- Non-ASCII digits, such as Arabic-Indic ones, pass `isdigit()` and are converted by `int()`, so they may parse as valid.\n- A non-string input like `None` raises `AttributeError` on `.split`, not `ValueError`.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.038906
  },
  {
   "run": "claude-sonnet-5-5__control__T3__r3__de04a9",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "control",
   "rep": 3,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 841,
   "input": 4,
   "cache_read": 25931,
   "cache_write": 6266,
   "warmup": null,
   "cache_write_1h": 6266,
   "cost_usd": 0.0386682,
   "wall_s": 7.8,
   "turns": 2,
   "prompt_last": 16502,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` does the work. It first calls `entries_in_month`, which keeps only entries whose `day.year` and `day.month` both match the arguments. It then returns a dict with three values:\n\n- `count`: the number of entries in that month.\n- `total`: the sum of their `amount` values, from `total()`.\n- `by_category`: a dict mapping each category to its summed amount, from `sum_by_category()`.\n\nCategories are lowercased when an entry is added, so \"Food\" and \"food\" land in the same bucket. Amounts are plain floats, so the sums are ordinary floating-point sums with no rounding. If no entries match, you get a count of 0, a total of 0 and an empty `by_category` dict.\n\n**Invalid date strings.** Dates are only parsed in `add_entry`, through `parse_date`, and never during reporting. `parse_date` splits the string on `-`. It raises `ValueError(f\"bad date: ...\")` if there aren't exactly three parts or if any part contains a non-digit character. That check catches things like `\"2024/01/05\"`, `\"2024-1-x\"` and negative numbers. Empty parts are also rejected, because `\"\".isdigit()` is False.\n\nA string that passes the shape check but isn't a real date, such as `\"2024-02-30\"` or `\"2024-13-01\"`, is not caught by that check. It goes on to `date(y, m, d)`, which raises its own `ValueError` (for example \"day is out of range for month\"). Callers therefore always see a `ValueError`, but the message differs between the two cases.\n\nA few other behaviours follow from this:\n\n- `add_entry` checks `amount <= 0` before it parses the date. A bad amount raises its own `ValueError` first, and a bad date is only reported if the amount is valid.\n- The date is parsed inside the `Entry(...)` call, before `entries.append`. An invalid date therefore leaves the list unchanged.\n- A non-string `day`, such as `None`, would raise `AttributeError` from `text.split` rather than `ValueError`.\n- The check accepts unusual but \"digit-only\" strings such as `\"2024-01-05\"` with extra padding (`\"02024-001-005\"`). These parse fine, because `int()` handles leading zeros.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.038668
  },
  {
   "run": "claude-sonnet-5-5__karpathy__T3__r1__ded2de",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "karpathy",
   "rep": 1,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 900,
   "input": 4,
   "cache_read": 26951,
   "cache_write": 7289,
   "warmup": null,
   "cache_write_1h": 7289,
   "cost_usd": 0.0435542,
   "wall_s": 11.0,
   "turns": 2,
   "prompt_last": 17525,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` in `ledger/core.py` does the work in three steps:\n\n1. `entries_in_month` keeps only the entries whose `day.year` and `day.month` both match the arguments.\n2. `total` adds up the `amount` of those entries, which gives the month's overall total.\n3. `sum_by_category` builds a dict that maps each category to the sum of its amounts. Categories are lowercased when an entry is added, so `\"Food\"` and `\"food\"` land in the same bucket.\n\nThe function returns a dict with three keys: `count` (the number of entries in the month), `total`, and `by_category`. A month with no entries gives a count of 0, a total of `0` (the integer that `sum` returns for an empty sequence, not `0.0`), and an empty `by_category` dict. The amounts are plain floats, so the totals can pick up ordinary floating-point rounding error.\n\n**Invalid date strings.** Dates are only parsed when an entry is added. `add_entry` calls `parse_date(day)`, which does the following:\n\n- It splits the text on `-`. If the result isn't exactly three parts, or any part isn't made only of digits, it raises `ValueError(f\"bad date: {text!r}\")`.\n- Otherwise it converts the parts to ints and passes them to `datetime.date(y, m, d)`.\n- If the parts are all digits but don't form a real date, such as `\"2024-13-01\"` or `\"2024-02-30\"`, the `date` constructor raises its own `ValueError`. The message is different, for example \"month must be in 1..12\".\n\nBoth failures are `ValueError`, so a caller can catch them together. There are a few side effects and edge cases:\n\n- **Order of checks:** `add_entry` checks `amount <= 0` first, then parses the date. A bad date is never reported if the amount is already invalid.\n- **Nothing is appended on failure:** `entries.append(...)` runs only after `parse_date` succeeds, so a bad entry doesn't leave the list partly modified.\n- **Not strict ISO:** `\"2024-1-5\"` is accepted, because only the digits are checked and not the field widths.\n- **Non-string input:** a non-string such as `None` fails with `AttributeError` on `.split`, not `ValueError`.\n- **Reports never parse dates:** `monthly_report` and the functions it calls work with `date` objects that are already stored, so they never see an invalid date string.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.043554
  },
  {
   "run": "claude-sonnet-5-5__karpathy__T3__r2__c28a43",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "karpathy",
   "rep": 2,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 845,
   "input": 4,
   "cache_read": 26951,
   "cache_write": 7289,
   "warmup": null,
   "cache_write_1h": 7289,
   "cost_usd": 0.0430042,
   "wall_s": 8.9,
   "turns": 2,
   "prompt_last": 17525,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` does the work in three steps:\n\n1. It calls `entries_in_month`. That keeps only the entries whose `day.year` and `day.month` both match the arguments.\n2. It passes that filtered list to `total`, which adds up each entry's `amount`.\n3. It passes the same list to `sum_by_category`, which builds a dict mapping each category to its summed amount.\n\nIt returns a dict with three keys: `count` (the number of entries in that month), `total` (the overall sum), and `by_category` (the per-category sums). If no entries match the month, `count` is 0, `total` is 0 and `by_category` is `{}`. The amounts are plain floats, so the sums can carry ordinary floating-point rounding error.\n\n**Invalid date strings.** Dates are parsed only when an entry is added. `add_entry` calls `parse_date(day)`, and `parse_date` splits the string on `-`. It raises `ValueError(f\"bad date: ...\")` if the result isn't exactly three parts or if any part isn't made only of digits. Examples are `\"2024/01/05\"`, `\"2024-1\"`, `\"abc\"` and `\"2024-01--5\"`. Negative-looking or signed parts fail the digit check too.\n\nIf the shape is right but the values are impossible, such as `\"2024-13-01\"` or `\"2024-02-30\"`, the `date(y, m, d)` constructor raises its own `ValueError` with a different message. Either way the caller gets a `ValueError`.\n\n**Ordering in `add_entry`.** The amount check (`amount <= 0` raises `ValueError`) runs before the date is parsed. An entry with both problems therefore reports the amount error first. The parse happens inside the `Entry(...)` construction, before `entries.append`, so a bad date means nothing is appended and the list is left unchanged.\n\n**Two things the code doesn't do.**\n- **Non-string input:** `parse_date` calls `text.split`, so a non-string such as `None` raises `AttributeError` instead of `ValueError`.\n- **Malformed dates in reports:** `monthly_report` never sees a bad date string. Entries hold parsed `date` objects, so a malformed date can't reach the report.\n\nI didn't change any files.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.043004
  },
  {
   "run": "claude-sonnet-5-5__karpathy__T3__r3__4e7993",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "karpathy",
   "rep": 3,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 857,
   "input": 4,
   "cache_read": 26952,
   "cache_write": 7290,
   "warmup": null,
   "cache_write_1h": 7290,
   "cost_usd": 0.0431284,
   "wall_s": 7.4,
   "turns": 2,
   "prompt_last": 17526,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` is the entry point. It calls `entries_in_month`, which keeps only entries whose `day.year` and `day.month` both match the arguments. That filtered list is then used three ways:\n\n- `count` is the number of matching entries.\n- `total` is `total()`, a plain `sum` of every matching entry's `amount`.\n- `by_category` is `sum_by_category()`, which builds a dict from category to the running sum of amounts. Categories are already lowercased in `add_entry`, so `\"Food\"` and `\"food\"` land in the same bucket.\n\nAmounts are plain floats, so there is no rounding and no currency handling. Sums can pick up ordinary floating-point error. A month with no entries gives `{\"count\": 0, \"total\": 0, \"by_category\": {}}`. The `total` there is the integer `0`, because `sum` of an empty sequence returns `0`, not `0.0`.\n\n**Invalid date strings.** Dates are only parsed when an entry is added, in `add_entry`, which calls `parse_date`. `parse_date` splits the string on `-`. It raises `ValueError(f\"bad date: {text!r}\")` if there aren't exactly three parts or if any part isn't all digits. Examples are `\"2024/01/05\"`, `\"2024-1\"`, `\"2024-01-xx\"` and negative-looking parts.\n\nIf the shape passes, it builds `date(y, m, d)`. Values that are numeric but impossible, such as `\"2024-13-01\"` or `\"2024-02-30\"`, make `datetime.date` raise its own `ValueError` (for example \"month must be in 1..12\"). So callers see a `ValueError` either way, but the message differs.\n\nThe parse happens as an argument to `Entry(...)`, before `entries.append`. A bad date therefore raises before anything is added, and the list is left unchanged. The amount check runs even earlier, so a non-positive amount fails first.\n\nTwo smaller points:\n\n- `parse_date` doesn't strip whitespace, so `\" 2024-01-05\"` is rejected.\n- `str.isdigit()` accepts some non-ASCII digit characters. Those pass the check, and `int()` then converts them.\n\nBecause `monthly_report` only receives `Entry` objects, it never sees an invalid date string.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.043128
  },
  {
   "run": "claude-sonnet-5-5__placebo__T3__r1__92b526",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "placebo",
   "rep": 1,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 788,
   "input": 4,
   "cache_read": 26684,
   "cache_write": 7020,
   "warmup": null,
   "cache_write_1h": 7020,
   "cost_usd": 0.0413048,
   "wall_s": 8.9,
   "turns": 2,
   "prompt_last": 17256,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` is the entry point. It calls `entries_in_month`, which keeps only the entries whose `day.year` and `day.month` match the arguments. Then it builds a dictionary from that filtered list:\n- `count` is the number of entries in the month.\n- `total` comes from `total()`, which adds up the `amount` of each entry.\n- `by_category` comes from `sum_by_category()`, which keeps a running sum per category. Category names were lowercased when the entries were added, so \"Food\" and \"food\" land in the same bucket.\n\nIf nothing matches the month, you get `count` 0, `total` 0 and an empty `by_category`. No error is raised. `year` and `month` are compared directly, so an out-of-range month such as 13 just matches nothing.\n\n**Invalid date strings.** Dates are only parsed in `add_entry`, through `parse_date`. The report functions work with `date` objects that are already stored on each `Entry`.\n\n`parse_date` splits the text on `-`. It raises `ValueError(f\"bad date: ...\")` if the result isn't exactly three parts, or if any part isn't all digits. That catches things like `\"2024/01/05\"`, `\"2024-1\"`, `\"abc\"` and negative-looking parts. Parts that are digits but not a real date, such as `\"2024-13-40\"`, get past that check. The `date(y, m, d)` constructor then raises its own `ValueError` (for example \"month must be in 1..12\").\n\nNothing catches either error, so it goes up to the caller of `add_entry`. Two things follow from the order of the code in `add_entry`:\n- The `amount <= 0` check runs first, so a call with a bad amount and a bad date reports the amount error.\n- The date is parsed before `entries.append`, so a bad date never adds a partial entry and the list is left unchanged.\n\nTwo side cases. A non-string `text`, such as `None`, fails on `.split` with an `AttributeError` instead of a `ValueError`. And `isdigit()` accepts some non-ASCII digit characters, which `int()` will usually still convert.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.041305
  },
  {
   "run": "claude-sonnet-5-5__placebo__T3__r2__bded23",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "placebo",
   "rep": 2,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 848,
   "input": 4,
   "cache_read": 26685,
   "cache_write": 7021,
   "warmup": null,
   "cache_write_1h": 7021,
   "cost_usd": 0.041909,
   "wall_s": 9.7,
   "turns": 2,
   "prompt_last": 17257,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` in `ledger/core.py:51` does the work.\n\n1. It calls `entries_in_month`, which keeps only the entries whose `day.year` and `day.month` both match the arguments.\n2. On that filtered list it computes three things:\n   - `count`, the number of entries.\n   - `total`, the sum of every `amount`, from `total()`.\n   - `by_category`, a dict of per-category sums built by `sum_by_category`. Categories were lowercased when the entries were added, so \"Food\" and \"food\" land in the same bucket.\n3. It returns these as a dict with the keys `count`, `total` and `by_category`.\n\nIf no entries match the month, the result is `count` 0, `total` 0 (the integer, since `sum` of an empty sequence is 0) and an empty `by_category`. It doesn't raise an error.\n\n**Invalid date strings.** Dates are only parsed when an entry is added. `add_entry` calls `parse_date(day)`, and the monthly functions work on `date` objects that are already parsed.\n\n`parse_date` expects `YYYY-MM-DD`. It splits the text on `-`, and it raises `ValueError(f\"bad date: {text!r}\")` in either of these cases:\n- The split doesn't give exactly three parts.\n- Any part contains something other than digits.\n\nOtherwise it converts the parts to ints and builds a `date(y, m, d)`. If the digits are well-formed but not a real date, such as `2024-13-45`, the `date` constructor raises its own `ValueError` (for example \"month must be in 1..12\"). That message is less specific than the \"bad date\" one.\n\nTwo side effects of how this is written:\n- Nothing catches the error, so it propagates to the caller of `add_entry`.\n- `add_entry` checks `amount <= 0` before it parses the date, so an entry with both a bad amount and a bad date reports the amount error first. In every failure case the entry is never appended, so the ledger stays unchanged.\n\nTwo edge cases in the validation:\n- Non-string input such as `None` fails with an `AttributeError` on `.split`, not a `ValueError`.\n- `str.isdigit()` accepts some non-ASCII digit characters, and `int()` would convert them.\n\nI didn't change any files.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.041909
  },
  {
   "run": "claude-sonnet-5-5__placebo__T3__r3__800c1c",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "placebo",
   "rep": 3,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 732,
   "input": 4,
   "cache_read": 26688,
   "cache_write": 7025,
   "warmup": null,
   "cache_write_1h": 7025,
   "cost_usd": 0.0407656,
   "wall_s": 7.3,
   "turns": 2,
   "prompt_last": 17261,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` is the entry point. It first calls `entries_in_month`, which keeps only the entries whose `day.year` and `day.month` both match the arguments. On that filtered list it builds a dict with three values:\n- `count`: the number of matching entries.\n- `total`: the sum of their `amount` values, computed by `total`.\n- `by_category`: a dict from category to summed amount, computed by `sum_by_category`. Categories are lowercased when an entry is added, so \"Food\" and \"food\" land in the same bucket.\n\nIf no entries match the month, the report has a count of 0, a total of 0 and an empty `by_category`. The function does no rounding, so the amounts are plain float sums.\n\n**Invalid date strings.** Dates are only parsed when an entry is added, through `add_entry`, which calls `parse_date`. `parse_date` splits the text on `-`. It raises `ValueError(\"bad date: ...\")` if there aren't exactly three parts, or if any part isn't made up of digits. That check rejects things like `\"2024/01/05\"`, `\"2024-1\"`, `\"abc\"` and negative numbers. It also rejects an empty string, because `\"\".split(\"-\")` gives one part.\n\nStrings that pass the shape check but aren't real dates, such as `\"2024-13-01\"` or `\"2024-02-30\"`, make `date(y, m, d)` raise its own `ValueError`. Its message differs, for example \"month must be in 1..12\". A non-string such as `None` would fail with an `AttributeError` on `.split`, not a `ValueError`.\n\n`add_entry` checks the amount before it parses the date, so a non-positive amount raises first. Because `parse_date` runs while the `Entry` is being built, before `append`, an invalid date leaves the `entries` list unchanged. Nothing catches the error, so it reaches the caller. Invalid dates therefore never reach the monthly totals.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.040766
  },
  {
   "run": "claude-sonnet-5-5__ponytail__T3__r1__3bcccf",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "ponytail",
   "rep": 1,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 842,
   "input": 4,
   "cache_read": 26967,
   "cache_write": 7306,
   "warmup": null,
   "cache_write_1h": 7306,
   "cost_usd": 0.0430454,
   "wall_s": 8.4,
   "turns": 2,
   "prompt_last": 17542,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` is the entry point. It calls `entries_in_month`, which keeps only the entries whose `day.year` and `day.month` both match the arguments. It then builds a dict from that filtered list:\n- `count` is the number of matching entries.\n- `total` is `total()`, a plain `sum` of the `amount` floats.\n- `by_category` is `sum_by_category()`. It walks the entries and adds each amount into a dict keyed by the entry's category, which `add_entry` lowercased when the entry was created.\n\nThe amounts are floats and are summed as-is, with no rounding. Totals can therefore pick up small floating-point errors, such as 0.1 + 0.2 giving 0.30000000000000004. A month with no entries gives `count` 0, `total` 0 (an int, because `sum` of nothing is 0), and an empty `by_category`.\n\n**Invalid date strings.** Dates are only parsed in `add_entry`, through `parse_date`. `parse_date` splits the string on `-`. It raises `ValueError(\"bad date: ...\")` if there aren't exactly three parts or if any part isn't all digits. That check rejects `\"2024/01/05\"`, `\"2024-1\"`, `\"2024-01-xx\"` and negative-looking strings. Otherwise it converts the parts to ints and calls `date(y, m, d)`. That call raises its own `ValueError` for out-of-range values such as month 13 or Feb 30, with a different message (\"month must be in 1..12\" and similar).\n\nThe parts aren't length-checked, so `\"2024-1-5\"` is accepted even though the docstring says YYYY-MM-DD. Nothing catches the error. Because `parse_date` runs while the `Entry` is being built, the exception propagates out of `add_entry` before `entries.append` runs. The list is left unchanged, and the caller has to handle the error. The `amount <= 0` check runs first, so a bad amount is reported ahead of a bad date.\n\n`monthly_report` never sees invalid dates, since every stored `Entry.day` is already a real `date`. It doesn't validate its own `year` or `month` arguments, though. An out-of-range month such as 13 just matches nothing and returns an empty report.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.043045
  },
  {
   "run": "claude-sonnet-5-5__ponytail__T3__r2__aa7247",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "ponytail",
   "rep": 2,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 774,
   "input": 4,
   "cache_read": 26963,
   "cache_write": 7301,
   "warmup": null,
   "cache_write_1h": 7301,
   "cost_usd": 0.042344599999999996,
   "wall_s": 10.0,
   "turns": 2,
   "prompt_last": 17537,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` is the entry point. It calls `entries_in_month`, which keeps the entries whose `day.year` and `day.month` both match. It then builds a dict from that filtered list with three keys:\n- `count` is the number of entries in the month.\n- `total` is the sum of their `amount` values, computed by `total()`.\n- `by_category` is a dict of per-category sums, computed by `sum_by_category()`. Categories were lowercased when the entry was added, so `Food` and `food` land in the same bucket.\n\nThe amounts are plain floats, so the sums can pick up ordinary floating-point rounding error. Nothing rounds them. If the month has no entries, you get `count` 0, `total` 0 and an empty `by_category`.\n\n**Invalid dates.** Dates are only parsed in `add_entry`, through `parse_date`. `monthly_report` never sees a date string, because `Entry.day` is already a `date` object by then.\n\n`parse_date` splits the text on `-`. It raises `ValueError(\"bad date: ...\")` if the result isn't exactly three parts or any part contains a non-digit character. That covers `\"2024/01/05\"`, `\"2024-1\"`, `\"abc\"` and negative-looking pieces. If the text passes that check, it calls `date(y, m, d)`. That call raises its own `ValueError` for impossible values such as `\"2024-13-01\"` or `\"2024-02-30\"`, with a message like \"month must be in 1..12\". So every kind of bad date ends in a `ValueError`, but the message differs depending on which check failed.\n\nTwo details matter for `add_entry`:\n- **Order of checks:** it validates `amount <= 0` first and parses the date second. A bad date therefore never gets as far as `entries.append`.\n- **Failure behavior:** the exception propagates to the caller, and the ledger list is left unchanged. There is no catching or fallback.\n\n`parse_date` would also raise `AttributeError` if `text` weren't a string, for example `None`. That isn't handled either.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.042345
  },
  {
   "run": "claude-sonnet-5-5__ponytail__T3__r3__88e026",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "ponytail",
   "rep": 3,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 718,
   "input": 4,
   "cache_read": 26960,
   "cache_write": 7297,
   "warmup": null,
   "cache_write_1h": 7297,
   "cost_usd": 0.041768,
   "wall_s": 11.1,
   "turns": 2,
   "prompt_last": 17533,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` is the entry point.\n\n1. It calls `entries_in_month`, which keeps only the entries whose `day.year` and `day.month` both match the arguments.\n2. It returns a dict with three values:\n   - `count`: the number of entries in that month.\n   - `total`: the sum of their `amount` values, from `total()`.\n   - `by_category`: a dict of per-category sums, from `sum_by_category()`. It starts each category at 0.0 and adds amounts as it goes.\n\nCategories are lowercased when an entry is added, so \"Food\" and \"food\" end up in the same bucket. A month with no entries gives `count` 0, `total` 0 (an int, because `sum` of an empty sequence is 0) and an empty `by_category`.\n\n**Invalid date strings.** Dates are parsed only when an entry is added. `add_entry` calls `parse_date(day)`, which splits the string on `-`. It raises `ValueError(f\"bad date: ...\")` if either of these is true:\n- the string doesn't have exactly three parts\n- any part isn't all digits\n\nTwo other cases are worth knowing about:\n- **Out-of-range values:** a string like `2024-13-45` passes those checks. `date(y, m, d)` then raises its own `ValueError` (\"month must be in 1..12\" or similar). The caller gets a `ValueError` either way, but the message differs.\n- **Non-string input:** `None` or another non-string fails on `.split` with an `AttributeError`, not a `ValueError`.\n\nBecause `parse_date` is called inside the `Entry(...)` construction, which is inside `entries.append(...)`, a bad date raises before anything is appended. The ledger is never left with a partial entry.\n\nThe `amount <= 0` check also runs first, so an entry with both a bad amount and a bad date reports the amount error.\n\nI didn't change any files.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.041768
  },
  {
   "run": "claude-sonnet-5-5__caveman__T4__r1__0f1d14",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T4",
   "skill": "caveman",
   "rep": 1,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 1349,
   "input": 8,
   "cache_read": 100268,
   "cache_write": 24513,
   "warmup": null,
   "cache_write_1h": 24513,
   "cost_usd": 0.1316116,
   "wall_s": 17.2,
   "turns": 5,
   "prompt_last": 34749,
   "lines_added": 55,
   "lines_deleted": 0,
   "lines": 55,
   "files": 1,
   "thinking": 69,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.131612
  },
  {
   "run": "claude-sonnet-5-5__caveman__T4__r2__8af7a6",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T4",
   "skill": "caveman",
   "rep": 2,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 1358,
   "input": 8,
   "cache_read": 100277,
   "cache_write": 24498,
   "warmup": null,
   "cache_write_1h": 24498,
   "cost_usd": 0.1316434,
   "wall_s": 11.5,
   "turns": 5,
   "prompt_last": 34734,
   "lines_added": 53,
   "lines_deleted": 0,
   "lines": 53,
   "files": 1,
   "thinking": 43,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.131643
  },
  {
   "run": "claude-sonnet-5-5__caveman__T4__r3__36a065",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T4",
   "skill": "caveman",
   "rep": 3,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 1298,
   "input": 8,
   "cache_read": 99909,
   "cache_write": 23929,
   "warmup": null,
   "cache_write_1h": 23929,
   "cost_usd": 0.12869380000000002,
   "wall_s": 10.9,
   "turns": 4,
   "prompt_last": 34165,
   "lines_added": 58,
   "lines_deleted": 0,
   "lines": 58,
   "files": 1,
   "thinking": 59,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.128694
  },
  {
   "run": "claude-sonnet-5-5__control__T4__r1__80fe70",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T4",
   "skill": "control",
   "rep": 1,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 1187,
   "input": 8,
   "cache_read": 65454,
   "cache_write": 11154,
   "warmup": null,
   "cache_write_1h": 11154,
   "cost_usd": 0.0695928,
   "wall_s": 11.3,
   "turns": 4,
   "prompt_last": 21390,
   "lines_added": 43,
   "lines_deleted": 0,
   "lines": 43,
   "files": 1,
   "thinking": 0,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.069593,
   "light": true
  },
  {
   "run": "claude-sonnet-5-5__control__T4__r2__a8a77a",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T4",
   "skill": "control",
   "rep": 2,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 1319,
   "input": 8,
   "cache_read": 89668,
   "cache_write": 18955,
   "warmup": null,
   "cache_write_1h": 18955,
   "cost_usd": 0.10695959999999999,
   "wall_s": 14.5,
   "turns": 4,
   "prompt_last": 29191,
   "lines_added": 49,
   "lines_deleted": 0,
   "lines": 49,
   "files": 1,
   "thinking": 80,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.10696
  },
  {
   "run": "claude-sonnet-5-5__control__T4__r3__381a37",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T4",
   "skill": "control",
   "rep": 3,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 1108,
   "input": 8,
   "cache_read": 89158,
   "cache_write": 18355,
   "warmup": null,
   "cache_write_1h": 18355,
   "cost_usd": 0.10234760000000001,
   "wall_s": 11.3,
   "turns": 4,
   "prompt_last": 28591,
   "lines_added": 49,
   "lines_deleted": 0,
   "lines": 49,
   "files": 1,
   "thinking": 0,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.102348
  },
  {
   "run": "claude-sonnet-5-5__karpathy__T4__r1__5ea177",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T4",
   "skill": "karpathy",
   "rep": 1,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 1405,
   "input": 8,
   "cache_read": 93609,
   "cache_write": 20951,
   "warmup": null,
   "cache_write_1h": 20951,
   "cost_usd": 0.11659180000000002,
   "wall_s": 11.8,
   "turns": 5,
   "prompt_last": 31187,
   "lines_added": 44,
   "lines_deleted": 0,
   "lines": 44,
   "files": 1,
   "thinking": 101,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.116592
  },
  {
   "run": "claude-sonnet-5-5__karpathy__T4__r2__c15ddc",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T4",
   "skill": "karpathy",
   "rep": 2,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 1349,
   "input": 8,
   "cache_read": 94637,
   "cache_write": 21737,
   "warmup": null,
   "cache_write_1h": 21737,
   "cost_usd": 0.11938140000000001,
   "wall_s": 15.4,
   "turns": 4,
   "prompt_last": 31973,
   "lines_added": 43,
   "lines_deleted": 0,
   "lines": 43,
   "files": 1,
   "thinking": 176,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.119381
  },
  {
   "run": "claude-sonnet-5-5__karpathy__T4__r3__e6e5ff",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T4",
   "skill": "karpathy",
   "rep": 3,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 1275,
   "input": 8,
   "cache_read": 65282,
   "cache_write": 11920,
   "warmup": null,
   "cache_write_1h": 11920,
   "cost_usd": 0.0735024,
   "wall_s": 9.8,
   "turns": 4,
   "prompt_last": 22156,
   "lines_added": 43,
   "lines_deleted": 0,
   "lines": 43,
   "files": 1,
   "thinking": 76,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.073502,
   "light": true
  },
  {
   "run": "claude-sonnet-5-5__placebo__T4__r1__864871",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T4",
   "skill": "placebo",
   "rep": 1,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 1444,
   "input": 8,
   "cache_read": 92222,
   "cache_write": 20079,
   "warmup": null,
   "cache_write_1h": 20079,
   "cost_usd": 0.1132164,
   "wall_s": 12.9,
   "turns": 4,
   "prompt_last": 30315,
   "lines_added": 52,
   "lines_deleted": 0,
   "lines": 52,
   "files": 2,
   "thinking": 124,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.113216
  },
  {
   "run": "claude-sonnet-5-5__placebo__T4__r2__e6b0df",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T4",
   "skill": "placebo",
   "rep": 2,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 1629,
   "input": 8,
   "cache_read": 91744,
   "cache_write": 19886,
   "warmup": null,
   "cache_write_1h": 19886,
   "cost_usd": 0.11419880000000002,
   "wall_s": 13.1,
   "turns": 5,
   "prompt_last": 30122,
   "lines_added": 59,
   "lines_deleted": 0,
   "lines": 59,
   "files": 1,
   "thinking": 79,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.114199
  },
  {
   "run": "claude-sonnet-5-5__placebo__T4__r3__bc333d",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T4",
   "skill": "placebo",
   "rep": 3,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 1571,
   "input": 8,
   "cache_read": 92581,
   "cache_write": 20571,
   "warmup": null,
   "cache_write_1h": 20571,
   "cost_usd": 0.1165262,
   "wall_s": 13.6,
   "turns": 4,
   "prompt_last": 30807,
   "lines_added": 62,
   "lines_deleted": 0,
   "lines": 62,
   "files": 2,
   "thinking": 96,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.116526
  },
  {
   "run": "claude-sonnet-5-5__ponytail__T4__r1__551c43",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T4",
   "skill": "ponytail",
   "rep": 1,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 1101,
   "input": 8,
   "cache_read": 93669,
   "cache_write": 20686,
   "warmup": null,
   "cache_write_1h": 20686,
   "cost_usd": 0.11250380000000001,
   "wall_s": 10.3,
   "turns": 4,
   "prompt_last": 30922,
   "lines_added": 31,
   "lines_deleted": 0,
   "lines": 31,
   "files": 1,
   "thinking": 126,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.112504
  },
  {
   "run": "claude-sonnet-5-5__ponytail__T4__r2__026843",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T4",
   "skill": "ponytail",
   "rep": 2,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 1159,
   "input": 8,
   "cache_read": 93326,
   "cache_write": 20320,
   "warmup": null,
   "cache_write_1h": 20320,
   "cost_usd": 0.11155119999999999,
   "wall_s": 10.6,
   "turns": 4,
   "prompt_last": 30556,
   "lines_added": 31,
   "lines_deleted": 0,
   "lines": 31,
   "files": 1,
   "thinking": 136,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.111551
  },
  {
   "run": "claude-sonnet-5-5__ponytail__T4__r3__70536e",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T4",
   "skill": "ponytail",
   "rep": 3,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 1248,
   "input": 8,
   "cache_read": 93594,
   "cache_write": 20671,
   "warmup": null,
   "cache_write_1h": 20671,
   "cost_usd": 0.1138988,
   "wall_s": 13.0,
   "turns": 4,
   "prompt_last": 30907,
   "lines_added": 34,
   "lines_deleted": 0,
   "lines": 34,
   "files": 1,
   "thinking": 205,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.113899
  },
  {
   "run": "gpt-5.6-luna__caveman__T1__r1__cbf1b1",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T1",
   "skill": "caveman",
   "rep": 1,
   "round": 1,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 696,
   "input": 75746,
   "cache_read": 65280,
   "cache_write": 20217,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.006184,
   "wall_s": 22.8,
   "turns": null,
   "prompt_last": null,
   "lines_added": 5,
   "lines_deleted": 0,
   "lines": 5,
   "files": null,
   "thinking": 115,
   "commands": 4,
   "api_cost": 0.006184
  },
  {
   "run": "gpt-5.6-luna__caveman__T1__r2__c6040a",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T1",
   "skill": "caveman",
   "rep": 2,
   "round": 2,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 958,
   "input": 109962,
   "cache_read": 90368,
   "cache_write": 29345,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.008826,
   "wall_s": 37.1,
   "turns": null,
   "prompt_last": null,
   "lines_added": 6,
   "lines_deleted": 1,
   "lines": 7,
   "files": null,
   "thinking": 159,
   "commands": 5,
   "api_cost": 0.008826
  },
  {
   "run": "gpt-5.6-luna__caveman__T1__r3__7f30b6",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T1",
   "skill": "caveman",
   "rep": 3,
   "round": 2,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1025,
   "input": 101294,
   "cache_read": 91392,
   "cache_write": 19653,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.006988,
   "wall_s": 35.1,
   "turns": null,
   "prompt_last": null,
   "lines_added": 5,
   "lines_deleted": 0,
   "lines": 5,
   "files": null,
   "thinking": 197,
   "commands": 6,
   "api_cost": 0.006988
  },
  {
   "run": "gpt-5.6-luna__control__T1__r1__b30dfb",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T1",
   "skill": "control",
   "rep": 1,
   "round": 1,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1045,
   "input": 76715,
   "cache_read": 58880,
   "cache_write": 27586,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.007949,
   "wall_s": 39.8,
   "turns": null,
   "prompt_last": null,
   "lines_added": 5,
   "lines_deleted": 0,
   "lines": 5,
   "files": null,
   "thinking": 180,
   "commands": 5,
   "api_cost": 0.007949
  },
  {
   "run": "gpt-5.6-luna__control__T1__r2__ceee3f",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T1",
   "skill": "control",
   "rep": 2,
   "round": 2,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1031,
   "input": 89451,
   "cache_read": 76032,
   "cache_write": 23170,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.007392,
   "wall_s": 29.3,
   "turns": null,
   "prompt_last": null,
   "lines_added": 5,
   "lines_deleted": 0,
   "lines": 5,
   "files": null,
   "thinking": 136,
   "commands": 5,
   "api_cost": 0.007392
  },
  {
   "run": "gpt-5.6-luna__control__T1__r3__729b1f",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T1",
   "skill": "control",
   "rep": 3,
   "round": 2,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1577,
   "input": 139670,
   "cache_read": 125696,
   "cache_write": 23725,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.009151,
   "wall_s": 40.4,
   "turns": null,
   "prompt_last": null,
   "lines_added": 6,
   "lines_deleted": 1,
   "lines": 7,
   "files": null,
   "thinking": 295,
   "commands": 6,
   "api_cost": 0.009151
  },
  {
   "run": "gpt-5.6-luna__karpathy__T1__r1__3d726e",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T1",
   "skill": "karpathy",
   "rep": 1,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1285,
   "input": 95503,
   "cache_read": 82176,
   "cache_write": 23078,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.007801,
   "wall_s": 32.6,
   "turns": null,
   "prompt_last": null,
   "lines_added": 5,
   "lines_deleted": 0,
   "lines": 5,
   "files": 1,
   "thinking": 225,
   "commands": 6,
   "api_cost": 0.007801
  },
  {
   "run": "gpt-5.6-luna__karpathy__T1__r2__d3b108",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T1",
   "skill": "karpathy",
   "rep": 2,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1199,
   "input": 81090,
   "cache_read": 70144,
   "cache_write": 20697,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.006981,
   "wall_s": 32.2,
   "turns": null,
   "prompt_last": null,
   "lines_added": 5,
   "lines_deleted": 0,
   "lines": 5,
   "files": 1,
   "thinking": 203,
   "commands": 5,
   "api_cost": 0.006981
  },
  {
   "run": "gpt-5.6-luna__karpathy__T1__r3__72a9f4",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T1",
   "skill": "karpathy",
   "rep": 3,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 986,
   "input": 80758,
   "cache_read": 68096,
   "cache_write": 22413,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.007028,
   "wall_s": 28.0,
   "turns": null,
   "prompt_last": null,
   "lines_added": 5,
   "lines_deleted": 0,
   "lines": 5,
   "files": 1,
   "thinking": 115,
   "commands": 4,
   "api_cost": 0.007028
  },
  {
   "run": "gpt-5.6-luna__placebo__T1__r1__600110",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T1",
   "skill": "placebo",
   "rep": 1,
   "round": 4,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1087,
   "input": 78593,
   "cache_read": 65024,
   "cache_write": 23320,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.007269,
   "wall_s": 39.9,
   "turns": null,
   "prompt_last": null,
   "lines_added": 6,
   "lines_deleted": 1,
   "lines": 7,
   "files": 2,
   "thinking": 145,
   "commands": 4,
   "api_cost": 0.007269
  },
  {
   "run": "gpt-5.6-luna__placebo__T1__r2__26d39a",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T1",
   "skill": "placebo",
   "rep": 2,
   "round": 4,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1146,
   "input": 79938,
   "cache_read": 67072,
   "cache_write": 22617,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.00724,
   "wall_s": 40.4,
   "turns": null,
   "prompt_last": null,
   "lines_added": 6,
   "lines_deleted": 0,
   "lines": 6,
   "files": 1,
   "thinking": 180,
   "commands": 5,
   "api_cost": 0.00724
  },
  {
   "run": "gpt-5.6-luna__placebo__T1__r3__24034d",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T1",
   "skill": "placebo",
   "rep": 3,
   "round": 4,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 834,
   "input": 65936,
   "cache_read": 55040,
   "cache_write": 20647,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.006231,
   "wall_s": 25.1,
   "turns": null,
   "prompt_last": null,
   "lines_added": 6,
   "lines_deleted": 0,
   "lines": 6,
   "files": 1,
   "thinking": 108,
   "commands": 4,
   "api_cost": 0.006231
  },
  {
   "run": "gpt-5.6-luna__ponytail__T1__r1__c3d28f",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T1",
   "skill": "ponytail",
   "rep": 1,
   "round": 1,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 985,
   "input": 65647,
   "cache_read": 50944,
   "cache_write": 24454,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.007092,
   "wall_s": 28.7,
   "turns": null,
   "prompt_last": null,
   "lines_added": 5,
   "lines_deleted": 0,
   "lines": 5,
   "files": null,
   "thinking": 170,
   "commands": 4,
   "api_cost": 0.007092
  },
  {
   "run": "gpt-5.6-luna__ponytail__T1__r2__f7405d",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T1",
   "skill": "ponytail",
   "rep": 2,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1135,
   "input": 80556,
   "cache_read": 72192,
   "cache_write": 18115,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.006429,
   "wall_s": 34.2,
   "turns": null,
   "prompt_last": null,
   "lines_added": 5,
   "lines_deleted": 0,
   "lines": 5,
   "files": 1,
   "thinking": 193,
   "commands": 5,
   "api_cost": 0.006429
  },
  {
   "run": "gpt-5.6-luna__ponytail__T1__r3__50bcd4",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T1",
   "skill": "ponytail",
   "rep": 3,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1009,
   "input": 109190,
   "cache_read": 92160,
   "cache_write": 26781,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.00841,
   "wall_s": 36.2,
   "turns": null,
   "prompt_last": null,
   "lines_added": 5,
   "lines_deleted": 0,
   "lines": 5,
   "files": 1,
   "thinking": 159,
   "commands": 6,
   "api_cost": 0.00841
  },
  {
   "run": "gpt-5.6-luna__caveman__T2__r1__a0b8e4",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T2",
   "skill": "caveman",
   "rep": 1,
   "round": 1,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1002,
   "input": 126066,
   "cache_read": 102400,
   "cache_write": 33417,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.009934,
   "wall_s": 44.9,
   "turns": null,
   "prompt_last": null,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": null,
   "thinking": 216,
   "commands": 6,
   "api_cost": 0.009934
  },
  {
   "run": "gpt-5.6-luna__caveman__T2__r2__6eb546",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T2",
   "skill": "caveman",
   "rep": 2,
   "round": 2,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 870,
   "input": 122625,
   "cache_read": 109568,
   "cache_write": 22808,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.007797,
   "wall_s": 29.0,
   "turns": null,
   "prompt_last": null,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": null,
   "thinking": 84,
   "commands": 6,
   "api_cost": 0.007797
  },
  {
   "run": "gpt-5.6-luna__caveman__T2__r3__9ad627",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T2",
   "skill": "caveman",
   "rep": 3,
   "round": 2,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 725,
   "input": 85277,
   "cache_read": 76288,
   "cache_write": 18740,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.006144,
   "wall_s": 31.7,
   "turns": null,
   "prompt_last": null,
   "lines_added": 1,
   "lines_deleted": 0,
   "lines": 1,
   "files": null,
   "thinking": 134,
   "commands": 5,
   "api_cost": 0.006144
  },
  {
   "run": "gpt-5.6-luna__control__T2__r1__384e52",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T2",
   "skill": "control",
   "rep": 1,
   "round": 1,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 849,
   "input": 75516,
   "cache_read": 62976,
   "cache_write": 22291,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.006737,
   "wall_s": 39.7,
   "turns": null,
   "prompt_last": null,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": null,
   "thinking": 131,
   "commands": 5,
   "api_cost": 0.006737
  },
  {
   "run": "gpt-5.6-luna__control__T2__r2__358342",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T2",
   "skill": "control",
   "rep": 2,
   "round": 2,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1123,
   "input": 104628,
   "cache_read": 89088,
   "cache_write": 25291,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.008188,
   "wall_s": 31.4,
   "turns": null,
   "prompt_last": null,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": null,
   "thinking": 158,
   "commands": 6,
   "api_cost": 0.008188
  },
  {
   "run": "gpt-5.6-luna__control__T2__r3__11d3a0",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T2",
   "skill": "control",
   "rep": 3,
   "round": 2,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 779,
   "input": 75776,
   "cache_read": 61952,
   "cache_write": 23575,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.006889,
   "wall_s": 38.5,
   "turns": null,
   "prompt_last": null,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": null,
   "thinking": 148,
   "commands": 4,
   "api_cost": 0.006889
  },
  {
   "run": "gpt-5.6-luna__karpathy__T2__r1__a90cf1",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T2",
   "skill": "karpathy",
   "rep": 1,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1391,
   "input": 123847,
   "cache_read": 109312,
   "cache_write": 24286,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.008713,
   "wall_s": 44.9,
   "turns": null,
   "prompt_last": null,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 256,
   "commands": 6,
   "api_cost": 0.008713
  },
  {
   "run": "gpt-5.6-luna__karpathy__T2__r2__1e3aa4",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T2",
   "skill": "karpathy",
   "rep": 2,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1360,
   "input": 94474,
   "cache_read": 78080,
   "cache_write": 26145,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.008423,
   "wall_s": 49.8,
   "turns": null,
   "prompt_last": null,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 186,
   "commands": 5,
   "api_cost": 0.008423
  },
  {
   "run": "gpt-5.6-luna__karpathy__T2__r3__e29bef",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T2",
   "skill": "karpathy",
   "rep": 3,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 958,
   "input": 65423,
   "cache_read": 58112,
   "cache_write": 17062,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.005724,
   "wall_s": 27.1,
   "turns": null,
   "prompt_last": null,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 191,
   "commands": 4,
   "api_cost": 0.005724
  },
  {
   "run": "gpt-5.6-luna__placebo__T2__r1__d412ef",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T2",
   "skill": "placebo",
   "rep": 1,
   "round": 4,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1432,
   "input": 122374,
   "cache_read": 103168,
   "cache_write": 28957,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.009573,
   "wall_s": 53.7,
   "turns": null,
   "prompt_last": null,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 318,
   "commands": 6,
   "api_cost": 0.009573
  },
  {
   "run": "gpt-5.6-luna__placebo__T2__r2__9a9db2",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T2",
   "skill": "placebo",
   "rep": 2,
   "round": 4,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 847,
   "input": 78528,
   "cache_read": 61952,
   "cache_write": 26327,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.007521,
   "wall_s": 27.4,
   "turns": null,
   "prompt_last": null,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 139,
   "commands": 4,
   "api_cost": 0.007521
  },
  {
   "run": "gpt-5.6-luna__placebo__T2__r3__c9a6b2",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T2",
   "skill": "placebo",
   "rep": 3,
   "round": 4,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1265,
   "input": 109080,
   "cache_read": 98304,
   "cache_write": 20527,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.007589,
   "wall_s": 38.9,
   "turns": null,
   "prompt_last": null,
   "lines_added": 3,
   "lines_deleted": 1,
   "lines": 4,
   "files": 1,
   "thinking": 300,
   "commands": 6,
   "api_cost": 0.007589
  },
  {
   "run": "gpt-5.6-luna__ponytail__T2__r1__1cb95e",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T2",
   "skill": "ponytail",
   "rep": 1,
   "round": 1,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1125,
   "input": 82556,
   "cache_read": 68096,
   "cache_write": 24211,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.007554,
   "wall_s": 30.9,
   "turns": null,
   "prompt_last": null,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": null,
   "thinking": 211,
   "commands": 4,
   "api_cost": 0.007554
  },
  {
   "run": "gpt-5.6-luna__ponytail__T2__r2__5b182b",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T2",
   "skill": "ponytail",
   "rep": 2,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1009,
   "input": 79580,
   "cache_read": 67072,
   "cache_write": 22259,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.007004,
   "wall_s": 40.1,
   "turns": null,
   "prompt_last": null,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 252,
   "commands": 5,
   "api_cost": 0.007004
  },
  {
   "run": "gpt-5.6-luna__ponytail__T2__r3__be095c",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T2",
   "skill": "ponytail",
   "rep": 3,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1004,
   "input": 79575,
   "cache_read": 65024,
   "cache_write": 24302,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.007366,
   "wall_s": 94.0,
   "turns": null,
   "prompt_last": null,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 157,
   "commands": 5,
   "api_cost": 0.007366
  },
  {
   "run": "gpt-5.6-luna__caveman__T3__r1__001a4a",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "caveman",
   "rep": 1,
   "round": 1,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 232,
   "input": 26198,
   "cache_read": 20992,
   "cache_write": 14957,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.00369,
   "wall_s": 8.2,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": null,
   "thinking": 9,
   "commands": 1,
   "answer": "Monthly totals:\n\n- `entries_in_month` keeps entries whose `day.year` and `day.month` match requested year and month.\n- `total` sums their `amount` values.\n- `monthly_report` returns entry count, total amount, and category subtotals.\n\nInvalid date strings:\n\n- `parse_date` requires three numeric parts in `YYYY-MM-DD` format.\n- Wrong structure or non-numeric parts raise `ValueError(\"bad date: ...\")`.\n- Valid-looking but impossible dates, like `2024-02-30`, also raise `ValueError` from `date`.\n- `add_entry` does not add entry when parsing fails.",
   "account_remark": false,
   "api_cost": 0.00369
  },
  {
   "run": "gpt-5.6-luna__caveman__T3__r2__a25870",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "caveman",
   "rep": 2,
   "round": 2,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 217,
   "input": 25999,
   "cache_read": 17920,
   "cache_write": 17830,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.004185,
   "wall_s": 11.8,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": null,
   "thinking": 20,
   "commands": 1,
   "answer": "Monthly report:\n\n- Filters entries matching requested `year` and `month`.\n- Counts matching entries.\n- Sums their `amount` values.\n- Groups sums by lowercase `category`.\n\nInvalid date string:\n\n- `parse_date` requires three numeric `YYYY-MM-DD` parts.\n- Wrong format raises `ValueError(\"bad date: ...\")`.\n- Valid-looking but impossible dates, such as `2024-02-30`, also raise `ValueError` from `date`.\n- `add_entry` does not add entry when parsing fails.",
   "account_remark": false,
   "api_cost": 0.004185
  },
  {
   "run": "gpt-5.6-luna__caveman__T3__r3__b02ac3",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "caveman",
   "rep": 3,
   "round": 2,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 246,
   "input": 26198,
   "cache_read": 20992,
   "cache_write": 14957,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.003706,
   "wall_s": 10.1,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": null,
   "thinking": 8,
   "commands": 1,
   "answer": "Monthly totals:\n\n- `entries_in_month` keeps entries whose `day.year` and `day.month` match requested year and month.\n- `monthly_report` sums matching `amount` values with `total`.\n- It returns entry count, total amount, and category totals in `by_category`.\n- Category names come from stored lowercase values.\n\nInvalid date strings:\n\n- `parse_date` expects `YYYY-MM-DD`.\n- Wrong format or non-numeric parts raises `ValueError` with `bad date: ...`.\n- Valid-looking but impossible dates also raise `ValueError` from `date`, such as `2024-02-30`.\n- `add_entry` does not add entry when parsing fails, because parsing happens before `append`.",
   "account_remark": false,
   "api_cost": 0.003706
  },
  {
   "run": "gpt-5.6-luna__control__T3__r1__f6e8fd",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "control",
   "rep": 1,
   "round": 1,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 247,
   "input": 22845,
   "cache_read": 19968,
   "cache_write": 12628,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.003221,
   "wall_s": 9.1,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": null,
   "thinking": 18,
   "commands": 1,
   "answer": "Monthly totals are computed by:\n\n1. Filtering entries whose `day.year` and `day.month` match the requested year and month.\n2. Summing their `amount` values with `sum`.\n3. Returning the result in `monthly_report` under `\"total\"`, along with the entry count and category breakdown.\n\nDate strings are parsed as `YYYY-MM-DD`. Invalid formats, non-numeric components, or impossible dates (such as month 13 or February 30) raise `ValueError`. In `add_entry`, that error prevents the entry from being added.",
   "account_remark": false,
   "api_cost": 0.003221
  },
  {
   "run": "gpt-5.6-luna__control__T3__r2__bef3fd",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "control",
   "rep": 2,
   "round": 2,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 234,
   "input": 22827,
   "cache_read": 19968,
   "cache_write": 12610,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.003202,
   "wall_s": 8.8,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": null,
   "thinking": 13,
   "commands": 1,
   "answer": "Monthly totals are computed by `monthly_report(entries, year, month)`. It first filters entries whose date has the requested year and month, then returns:\n\n- `count`: number of matching entries\n- `total`: sum of their amounts\n- `by_category`: separate totals for each category\n\nIf a date string is invalid, `parse_date` raises `ValueError`. This happens for incorrectly formatted strings or impossible dates, such as `2024-02-30`. Because `add_entry` calls `parse_date`, the entry is not added.",
   "account_remark": false,
   "api_cost": 0.003202
  },
  {
   "run": "gpt-5.6-luna__control__T3__r3__1721c5",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "control",
   "rep": 3,
   "round": 2,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 240,
   "input": 22833,
   "cache_read": 19968,
   "cache_write": 12616,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.003211,
   "wall_s": 9.5,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": null,
   "thinking": 14,
   "commands": 1,
   "answer": "Monthly totals are computed by:\n\n1. Filtering entries whose `day.year` and `day.month` match the requested year and month.\n2. Counting those entries.\n3. Summing their `amount` values with `sum`.\n4. Grouping their amounts by category.\n\n`monthly_report` returns all three results: `count`, `total`, and `by_category`.\n\nDate strings are parsed as `YYYY-MM-DD`. Malformed strings or impossible dates cause `parse_date` to raise `ValueError`; consequently, `add_entry` fails and does not append an entry.",
   "account_remark": false,
   "api_cost": 0.003211
  },
  {
   "run": "gpt-5.6-luna__karpathy__T3__r1__841b6b",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "karpathy",
   "rep": 1,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 262,
   "input": 23964,
   "cache_read": 17920,
   "cache_write": 15795,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.003832,
   "wall_s": 13.4,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 14,
   "commands": 1,
   "answer": "Monthly totals are computed by `monthly_report(entries, year, month)`:\n\n1. It filters entries whose `day.year` and `day.month` match the requested year and month.\n2. It counts those entries.\n3. It sums their `amount` values.\n4. It also groups the amounts by category.\n\nThe returned dictionary contains `count`, `total`, and `by_category`.\n\nDate strings are parsed as `YYYY-MM-DD`. Invalid formats, non-numeric components, or impossible calendar dates cause `parse_date` to raise `ValueError`. Since `add_entry` does not catch that exception, the entry is not added and the error propagates to the caller.",
   "account_remark": false,
   "api_cost": 0.003832
  },
  {
   "run": "gpt-5.6-luna__karpathy__T3__r2__6368e2",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "karpathy",
   "rep": 2,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 250,
   "input": 23966,
   "cache_read": 17920,
   "cache_write": 15797,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.003818,
   "wall_s": 13.8,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 14,
   "commands": 1,
   "answer": "Monthly totals are computed by:\n\n1. Filtering entries whose `day.year` and `day.month` match the requested year and month.\n2. Summing their `amount` values with Python\u2019s `sum()`.\n3. Returning the result in `monthly_report` under `\"total\"`, along with the entry count and category breakdown.\n\nAn invalid date string causes `parse_date` to raise `ValueError`. This happens either when the string is not three numeric hyphen-separated parts or when the resulting values do not form a valid calendar date. Consequently, `add_entry` fails and does not append an entry.",
   "account_remark": false,
   "api_cost": 0.003818
  },
  {
   "run": "gpt-5.6-luna__karpathy__T3__r3__a6ffb4",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "karpathy",
   "rep": 3,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 270,
   "input": 23974,
   "cache_read": 17920,
   "cache_write": 15805,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.003843,
   "wall_s": 13.1,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 15,
   "commands": 1,
   "answer": "Monthly totals are computed by:\n\n1. Selecting entries whose `day.year` and `day.month` match the requested year and month.\n2. Counting those entries.\n3. Summing their `amount` values with `sum`.\n4. Grouping the amounts by category and summing each category separately.\n\n`monthly_report` returns all three values: `count`, `total`, and `by_category`.\n\nFor an invalid date string, `parse_date` raises `ValueError`. This happens if the string is not three numeric hyphen-separated parts, or if the resulting year, month, or day is not a valid calendar date. Consequently, `add_entry` fails and does not append an entry.",
   "account_remark": false,
   "api_cost": 0.003843
  },
  {
   "run": "gpt-5.6-luna__placebo__T3__r1__f46c27",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "placebo",
   "rep": 1,
   "round": 4,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 252,
   "input": 23775,
   "cache_read": 17920,
   "cache_write": 15606,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.003782,
   "wall_s": 11.3,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 15,
   "commands": 1,
   "answer": "Monthly totals are computed by `monthly_report(entries, year, month)`. It first filters entries whose `day.year` and `day.month` match the requested year and month. It then returns:\n\n- `count`: number of matching entries\n- `total`: sum of their amounts\n- `by_category`: category-wise sums\n\nAn invalid date string is rejected by `parse_date` with a `ValueError`. This covers malformed formats as well as impossible dates (for example, month 13 or February 30). Since `add_entry` calls `parse_date`, the entry is not added when parsing fails.",
   "account_remark": false,
   "api_cost": 0.003782
  },
  {
   "run": "gpt-5.6-luna__placebo__T3__r2__8ead7b",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "placebo",
   "rep": 2,
   "round": 4,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 237,
   "input": 23776,
   "cache_read": 19968,
   "cache_write": 13559,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.003396,
   "wall_s": 13.3,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 17,
   "commands": 1,
   "answer": "Monthly totals are calculated by first selecting entries whose `day.year` and `day.month` match the requested year and month. The report then sums their `amount` values, counts the matching entries, and also provides category-level subtotals.\n\nDate strings are parsed as `YYYY-MM-DD`. If the format is wrong or any component is not a valid calendar value, `parse_date` raises `ValueError`. Since `add_entry` does not catch that exception, the entry is not added and the error propagates to the caller.",
   "account_remark": false,
   "api_cost": 0.003396
  },
  {
   "run": "gpt-5.6-luna__placebo__T3__r3__b33892",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "placebo",
   "rep": 3,
   "round": 4,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 232,
   "input": 23764,
   "cache_read": 17920,
   "cache_write": 15595,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.003756,
   "wall_s": 16.5,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 13,
   "commands": 1,
   "answer": "Monthly totals are computed by `monthly_report(entries, year, month)`. It first filters entries whose `day.year` and `day.month` match the requested year and month, then sums their amounts with `total()`. The report also includes the matching entry count and totals grouped by lowercased category.\n\nDate strings are parsed as `YYYY-MM-DD`. Invalid formatting or non-numeric components raise `ValueError`; invalid calendar dates (such as month 13 or February 30) also raise `ValueError` from `datetime.date`.",
   "account_remark": false,
   "api_cost": 0.003756
  },
  {
   "run": "gpt-5.6-luna__ponytail__T3__r1__766e17",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "ponytail",
   "rep": 1,
   "round": 1,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 279,
   "input": 24067,
   "cache_read": 17920,
   "cache_write": 15898,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.003873,
   "wall_s": 10.3,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": null,
   "thinking": 14,
   "commands": 1,
   "answer": "Monthly totals are produced by `monthly_report(entries, year, month)`:\n\n1. `entries_in_month` keeps entries whose `day.year` and `day.month` match the requested year and month.\n2. The report\u2019s `\"total\"` is the sum of those entries\u2019 amounts.\n3. It also returns the matching entry count and a per-category breakdown.\n\nDate strings are parsed by `parse_date`, which expects three numeric parts in `YYYY-MM-DD` form. Invalid formats raise `ValueError(\"bad date: ...\")`; valid-looking but impossible dates (such as month 13) also raise `ValueError` from `datetime.date`. Through `add_entry`, the entry is not appended if parsing fails.",
   "account_remark": false,
   "api_cost": 0.003873
  },
  {
   "run": "gpt-5.6-luna__ponytail__T3__r2__ee5c0a",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "ponytail",
   "rep": 2,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 254,
   "input": 24075,
   "cache_read": 17920,
   "cache_write": 15906,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.003844,
   "wall_s": 8.4,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 8,
   "commands": 1,
   "answer": "Monthly totals are computed by `monthly_report(entries, year, month)`. It first keeps entries whose `day.year` and `day.month` match the requested year and month. It then returns:\n\n- `count`: number of matching entries\n- `total`: sum of their amounts\n- `by_category`: category-specific sums\n\nIf a date string is invalid, `parse_date` raises `ValueError`. This happens for malformed formats, non-numeric components, or impossible dates such as month 13 or February 30. Since `add_entry` does not catch that exception, the entry is not added.",
   "account_remark": false,
   "api_cost": 0.003844
  },
  {
   "run": "gpt-5.6-luna__ponytail__T3__r3__12a542",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "ponytail",
   "rep": 3,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 290,
   "input": 24068,
   "cache_read": 19968,
   "cache_write": 13851,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.003518,
   "wall_s": 10.8,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 13,
   "commands": 1,
   "answer": "Monthly totals are produced by `monthly_report(entries, year, month)`:\n\n- It first selects entries whose `day.year` and `day.month` match the requested year and month.\n- It returns the number of matching entries, their summed `amount`, and category-specific subtotals.\n- Amounts are summed directly as floats; entries from other months are ignored.\n\nDate strings are parsed by `parse_date`, expecting three hyphen-separated numeric parts in `YYYY-MM-DD` form. Invalid formatting raises `ValueError`; invalid calendar dates (such as month 13 or February 30) also raise `ValueError` from `datetime.date`. Since `add_entry` does not catch that exception, the entry is not added and the error propagates to the caller.",
   "account_remark": false,
   "api_cost": 0.003518
  },
  {
   "run": "gpt-5.6-luna__caveman__T4__r1__0cfad4",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T4",
   "skill": "caveman",
   "rep": 1,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1691,
   "input": 159607,
   "cache_read": 131840,
   "cache_write": 37518,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.01217,
   "wall_s": 59.2,
   "turns": null,
   "prompt_last": null,
   "lines_added": 48,
   "lines_deleted": 0,
   "lines": 48,
   "files": 1,
   "thinking": 220,
   "commands": 6,
   "api_cost": 0.01217
  },
  {
   "run": "gpt-5.6-luna__caveman__T4__r2__a1a51c",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T4",
   "skill": "caveman",
   "rep": 2,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 2072,
   "input": 159733,
   "cache_read": 130816,
   "cache_write": 38668,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.012836,
   "wall_s": 66.4,
   "turns": null,
   "prompt_last": null,
   "lines_added": 66,
   "lines_deleted": 0,
   "lines": 66,
   "files": 1,
   "thinking": 345,
   "commands": 7,
   "api_cost": 0.012836
  },
  {
   "run": "gpt-5.6-luna__caveman__T4__r3__701ba2",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T4",
   "skill": "caveman",
   "rep": 3,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1336,
   "input": 88958,
   "cache_read": 72192,
   "cache_write": 26517,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.00835,
   "wall_s": 35.7,
   "turns": null,
   "prompt_last": null,
   "lines_added": 50,
   "lines_deleted": 0,
   "lines": 50,
   "files": 1,
   "thinking": 207,
   "commands": 4,
   "api_cost": 0.00835
  },
  {
   "run": "gpt-5.6-luna__control__T4__r1__0beb4e",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T4",
   "skill": "control",
   "rep": 1,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1567,
   "input": 94912,
   "cache_read": 75008,
   "cache_write": 29655,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.009312,
   "wall_s": 48.7,
   "turns": null,
   "prompt_last": null,
   "lines_added": 51,
   "lines_deleted": 0,
   "lines": 51,
   "files": 1,
   "thinking": 169,
   "commands": 5,
   "api_cost": 0.009312
  },
  {
   "run": "gpt-5.6-luna__control__T4__r2__94a6de",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T4",
   "skill": "control",
   "rep": 2,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1858,
   "input": 124462,
   "cache_read": 101120,
   "cache_write": 33093,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.010871,
   "wall_s": 58.7,
   "turns": null,
   "prompt_last": null,
   "lines_added": 50,
   "lines_deleted": 0,
   "lines": 50,
   "files": 1,
   "thinking": 289,
   "commands": 7,
   "api_cost": 0.010871
  },
  {
   "run": "gpt-5.6-luna__control__T4__r3__3714df",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T4",
   "skill": "control",
   "rep": 3,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 2343,
   "input": 142695,
   "cache_read": 115456,
   "cache_write": 36990,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.012519,
   "wall_s": 56.0,
   "turns": null,
   "prompt_last": null,
   "lines_added": 51,
   "lines_deleted": 0,
   "lines": 51,
   "files": 1,
   "thinking": 403,
   "commands": 7,
   "api_cost": 0.012519
  },
  {
   "run": "gpt-5.6-luna__karpathy__T4__r1__43a63b",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T4",
   "skill": "karpathy",
   "rep": 1,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 2003,
   "input": 143284,
   "cache_read": 122368,
   "cache_write": 30667,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.010984,
   "wall_s": 65.7,
   "turns": null,
   "prompt_last": null,
   "lines_added": 39,
   "lines_deleted": 0,
   "lines": 39,
   "files": 1,
   "thinking": 373,
   "commands": 8,
   "api_cost": 0.010984
  },
  {
   "run": "gpt-5.6-luna__karpathy__T4__r2__4d2622",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T4",
   "skill": "karpathy",
   "rep": 2,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1740,
   "input": 113296,
   "cache_read": 90112,
   "cache_write": 32935,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.010477,
   "wall_s": 43.2,
   "turns": null,
   "prompt_last": null,
   "lines_added": 47,
   "lines_deleted": 0,
   "lines": 47,
   "files": 1,
   "thinking": 372,
   "commands": 5,
   "api_cost": 0.010477
  },
  {
   "run": "gpt-5.6-luna__karpathy__T4__r3__fb7a5f",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T4",
   "skill": "karpathy",
   "rep": 3,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1753,
   "input": 110160,
   "cache_read": 98304,
   "cache_write": 21607,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.008391,
   "wall_s": 45.5,
   "turns": null,
   "prompt_last": null,
   "lines_added": 45,
   "lines_deleted": 0,
   "lines": 45,
   "files": 1,
   "thinking": 373,
   "commands": 5,
   "api_cost": 0.008391
  },
  {
   "run": "gpt-5.6-luna__placebo__T4__r1__001c01",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T4",
   "skill": "placebo",
   "rep": 1,
   "round": 4,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1569,
   "input": 97626,
   "cache_read": 85248,
   "cache_write": 22129,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.008014,
   "wall_s": 39.1,
   "turns": null,
   "prompt_last": null,
   "lines_added": 51,
   "lines_deleted": 0,
   "lines": 51,
   "files": 1,
   "thinking": 226,
   "commands": 5,
   "api_cost": 0.008014
  },
  {
   "run": "gpt-5.6-luna__placebo__T4__r2__a37550",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T4",
   "skill": "placebo",
   "rep": 2,
   "round": 4,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 2031,
   "input": 125209,
   "cache_read": 112384,
   "cache_write": 22576,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.0092,
   "wall_s": 49.5,
   "turns": null,
   "prompt_last": null,
   "lines_added": 72,
   "lines_deleted": 0,
   "lines": 72,
   "files": 1,
   "thinking": 389,
   "commands": 6,
   "api_cost": 0.0092
  },
  {
   "run": "gpt-5.6-luna__placebo__T4__r3__e32185",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T4",
   "skill": "placebo",
   "rep": 3,
   "round": 4,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 2060,
   "input": 126743,
   "cache_read": 108288,
   "cache_write": 28206,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.010279,
   "wall_s": 72.8,
   "turns": null,
   "prompt_last": null,
   "lines_added": 51,
   "lines_deleted": 0,
   "lines": 51,
   "files": 1,
   "thinking": 353,
   "commands": 6,
   "api_cost": 0.010279
  },
  {
   "run": "gpt-5.6-luna__ponytail__T4__r1__dd28b3",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T4",
   "skill": "ponytail",
   "rep": 1,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1772,
   "input": 132370,
   "cache_read": 115456,
   "cache_write": 26665,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.009769,
   "wall_s": 224.9,
   "turns": null,
   "prompt_last": null,
   "lines_added": 44,
   "lines_deleted": 0,
   "lines": 44,
   "files": 1,
   "thinking": 298,
   "commands": 6,
   "api_cost": 0.009769
  },
  {
   "run": "gpt-5.6-luna__ponytail__T4__r2__f1fda7",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T4",
   "skill": "ponytail",
   "rep": 2,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1868,
   "input": 114333,
   "cache_read": 98304,
   "cache_write": 25780,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.009364,
   "wall_s": 63.5,
   "turns": null,
   "prompt_last": null,
   "lines_added": 44,
   "lines_deleted": 0,
   "lines": 44,
   "files": 1,
   "thinking": 362,
   "commands": 5,
   "api_cost": 0.009364
  },
  {
   "run": "gpt-5.6-luna__ponytail__T4__r3__16291e",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T4",
   "skill": "ponytail",
   "rep": 3,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 2091,
   "input": 131514,
   "cache_read": 111360,
   "cache_write": 29905,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.010717,
   "wall_s": 71.3,
   "turns": null,
   "prompt_last": null,
   "lines_added": 43,
   "lines_deleted": 0,
   "lines": 43,
   "files": 1,
   "thinking": 423,
   "commands": 6,
   "api_cost": 0.010717
  },
  {
   "run": "gpt-6-luna__caveman__T1__r1__224025",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T1",
   "skill": "caveman",
   "rep": 1,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 549,
   "input": 142213,
   "cache_read": 125952,
   "cache_write": 28004,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.004334,
   "wall_s": 20.9,
   "turns": null,
   "prompt_last": null,
   "lines_added": 6,
   "lines_deleted": 0,
   "lines": 6,
   "files": 1,
   "thinking": 0,
   "commands": 7,
   "heldout": true,
   "api_cost": 0.004334
  },
  {
   "run": "gpt-6-luna__caveman__T1__r2__df46e4",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T1",
   "skill": "caveman",
   "rep": 2,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 596,
   "input": 130733,
   "cache_read": 120064,
   "cache_write": 22412,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.00374,
   "wall_s": 23.1,
   "turns": null,
   "prompt_last": null,
   "lines_added": 6,
   "lines_deleted": 0,
   "lines": 6,
   "files": 1,
   "thinking": 0,
   "commands": 5,
   "heldout": true,
   "api_cost": 0.00374
  },
  {
   "run": "gpt-6-luna__caveman__T1__r3__54b8f3",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T1",
   "skill": "caveman",
   "rep": 3,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 653,
   "input": 127394,
   "cache_read": 114944,
   "cache_write": 24193,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.003895,
   "wall_s": 22.2,
   "turns": null,
   "prompt_last": null,
   "lines_added": 7,
   "lines_deleted": 1,
   "lines": 8,
   "files": 2,
   "thinking": 0,
   "commands": 6,
   "heldout": true,
   "api_cost": 0.003895
  },
  {
   "run": "gpt-6-luna__control__T1__r1__979d49",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T1",
   "skill": "control",
   "rep": 1,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 699,
   "input": 102793,
   "cache_read": 95488,
   "cache_write": 19048,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.003209,
   "wall_s": 29.4,
   "turns": null,
   "prompt_last": null,
   "lines_added": 5,
   "lines_deleted": 0,
   "lines": 5,
   "files": 1,
   "thinking": 0,
   "commands": 5,
   "heldout": true,
   "api_cost": 0.003209
  },
  {
   "run": "gpt-6-luna__control__T1__r2__ebfb95",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T1",
   "skill": "control",
   "rep": 2,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 528,
   "input": 87436,
   "cache_read": 80384,
   "cache_write": 18795,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.002947,
   "wall_s": 18.7,
   "turns": null,
   "prompt_last": null,
   "lines_added": 6,
   "lines_deleted": 0,
   "lines": 6,
   "files": 1,
   "thinking": 0,
   "commands": 4,
   "heldout": true,
   "api_cost": 0.002947
  },
  {
   "run": "gpt-6-luna__control__T1__r3__14d82f",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T1",
   "skill": "control",
   "rep": 3,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 821,
   "input": 116226,
   "cache_read": 109568,
   "cache_write": 18401,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.003346,
   "wall_s": 24.2,
   "turns": null,
   "prompt_last": null,
   "lines_added": 5,
   "lines_deleted": 0,
   "lines": 5,
   "files": 1,
   "thinking": 0,
   "commands": 8,
   "heldout": true,
   "api_cost": 0.003346
  },
  {
   "run": "gpt-6-luna__karpathy__T1__r1__429fc7",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T1",
   "skill": "karpathy",
   "rep": 1,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 692,
   "input": 127376,
   "cache_read": 118784,
   "cache_write": 20335,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.003567,
   "wall_s": 22.5,
   "turns": null,
   "prompt_last": null,
   "lines_added": 5,
   "lines_deleted": 0,
   "lines": 5,
   "files": 1,
   "thinking": 0,
   "commands": 6,
   "heldout": true,
   "api_cost": 0.003567
  },
  {
   "run": "gpt-6-luna__karpathy__T1__r2__fde07d",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T1",
   "skill": "karpathy",
   "rep": 2,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 773,
   "input": 125130,
   "cache_read": 113664,
   "cache_write": 23209,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.003844,
   "wall_s": 24.8,
   "turns": null,
   "prompt_last": null,
   "lines_added": 6,
   "lines_deleted": 1,
   "lines": 7,
   "files": 2,
   "thinking": 0,
   "commands": 5,
   "heldout": true,
   "api_cost": 0.003844
  },
  {
   "run": "gpt-6-luna__karpathy__T1__r3__2fa4cc",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T1",
   "skill": "karpathy",
   "rep": 3,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 582,
   "input": 93038,
   "cache_read": 86528,
   "cache_write": 18253,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.002982,
   "wall_s": 22.6,
   "turns": null,
   "prompt_last": null,
   "lines_added": 5,
   "lines_deleted": 0,
   "lines": 5,
   "files": 1,
   "thinking": 0,
   "commands": 4,
   "heldout": true,
   "api_cost": 0.002982
  },
  {
   "run": "gpt-6-luna__placebo__T1__r1__b75b85",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T1",
   "skill": "placebo",
   "rep": 1,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 799,
   "input": 138930,
   "cache_read": 126720,
   "cache_write": 23953,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.004062,
   "wall_s": 33.9,
   "turns": null,
   "prompt_last": null,
   "lines_added": 5,
   "lines_deleted": 0,
   "lines": 5,
   "files": 1,
   "thinking": 0,
   "commands": 7,
   "heldout": true,
   "api_cost": 0.004062
  },
  {
   "run": "gpt-6-luna__placebo__T1__r2__48d641",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T1",
   "skill": "placebo",
   "rep": 2,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 884,
   "input": 109131,
   "cache_read": 97536,
   "cache_write": 23338,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.003751,
   "wall_s": 50.4,
   "turns": null,
   "prompt_last": null,
   "lines_added": 6,
   "lines_deleted": 1,
   "lines": 7,
   "files": 2,
   "thinking": 0,
   "commands": 5,
   "heldout": true,
   "api_cost": 0.003751
  },
  {
   "run": "gpt-6-luna__placebo__T1__r3__689997",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T1",
   "skill": "placebo",
   "rep": 3,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 709,
   "input": 124502,
   "cache_read": 116736,
   "cache_write": 19509,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.003473,
   "wall_s": 24.3,
   "turns": null,
   "prompt_last": null,
   "lines_added": 5,
   "lines_deleted": 0,
   "lines": 5,
   "files": 1,
   "thinking": 0,
   "commands": 7,
   "heldout": true,
   "api_cost": 0.003473
  },
  {
   "run": "gpt-6-luna__ponytail__T1__r1__44f616",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T1",
   "skill": "ponytail",
   "rep": 1,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 790,
   "input": 140485,
   "cache_read": 131840,
   "cache_write": 20388,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.003752,
   "wall_s": 32.6,
   "turns": null,
   "prompt_last": null,
   "lines_added": 5,
   "lines_deleted": 0,
   "lines": 5,
   "files": 1,
   "thinking": 0,
   "commands": 7,
   "heldout": true,
   "api_cost": 0.003752
  },
  {
   "run": "gpt-6-luna__ponytail__T1__r2__7bb6f6",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T1",
   "skill": "ponytail",
   "rep": 2,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 695,
   "input": 125072,
   "cache_read": 118784,
   "cache_write": 18031,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.003338,
   "wall_s": 28.7,
   "turns": null,
   "prompt_last": null,
   "lines_added": 5,
   "lines_deleted": 0,
   "lines": 5,
   "files": 1,
   "thinking": 0,
   "commands": 6,
   "heldout": true,
   "api_cost": 0.003338
  },
  {
   "run": "gpt-6-luna__ponytail__T1__r3__ebbc24",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T1",
   "skill": "ponytail",
   "rep": 3,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 740,
   "input": 146619,
   "cache_read": 133888,
   "cache_write": 24474,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.004156,
   "wall_s": 34.2,
   "turns": null,
   "prompt_last": null,
   "lines_added": 5,
   "lines_deleted": 0,
   "lines": 5,
   "files": 1,
   "thinking": 0,
   "commands": 8,
   "heldout": true,
   "api_cost": 0.004156
  },
  {
   "run": "gpt-6-luna__caveman__T2__r1__567d19",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T2",
   "skill": "caveman",
   "rep": 1,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 482,
   "input": 155917,
   "cache_read": 147200,
   "cache_write": 20460,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.003759,
   "wall_s": 19.8,
   "turns": null,
   "prompt_last": null,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 0,
   "commands": 7,
   "heldout": true,
   "api_cost": 0.003759
  },
  {
   "run": "gpt-6-luna__caveman__T2__r2__f1df5c",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T2",
   "skill": "caveman",
   "rep": 2,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 520,
   "input": 102490,
   "cache_read": 94720,
   "cache_write": 19513,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.003158,
   "wall_s": 18.8,
   "turns": null,
   "prompt_last": null,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 0,
   "commands": 5,
   "heldout": true,
   "api_cost": 0.003158
  },
  {
   "run": "gpt-6-luna__caveman__T2__r3__ecf0c8",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T2",
   "skill": "caveman",
   "rep": 3,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 538,
   "input": 121664,
   "cache_read": 113920,
   "cache_write": 19487,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.003357,
   "wall_s": 19.6,
   "turns": null,
   "prompt_last": null,
   "lines_added": 3,
   "lines_deleted": 2,
   "lines": 5,
   "files": 1,
   "thinking": 0,
   "commands": 5,
   "heldout": true,
   "api_cost": 0.003357
  },
  {
   "run": "gpt-6-luna__control__T2__r1__fc18e8",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T2",
   "skill": "control",
   "rep": 1,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 684,
   "input": 115169,
   "cache_read": 108544,
   "cache_write": 18368,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.003264,
   "wall_s": 25.6,
   "turns": null,
   "prompt_last": null,
   "lines_added": 3,
   "lines_deleted": 2,
   "lines": 5,
   "files": 1,
   "thinking": 0,
   "commands": 6,
   "heldout": true,
   "api_cost": 0.003264
  },
  {
   "run": "gpt-6-luna__control__T2__r2__65626b",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T2",
   "skill": "control",
   "rep": 2,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 653,
   "input": 115813,
   "cache_read": 108544,
   "cache_write": 19012,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.003313,
   "wall_s": 22.5,
   "turns": null,
   "prompt_last": null,
   "lines_added": 2,
   "lines_deleted": 2,
   "lines": 4,
   "files": 1,
   "thinking": 0,
   "commands": 6,
   "heldout": true,
   "api_cost": 0.003313
  },
  {
   "run": "gpt-6-luna__control__T2__r3__0e3d58",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T2",
   "skill": "control",
   "rep": 3,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 711,
   "input": 117747,
   "cache_read": 110592,
   "cache_write": 18898,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.003351,
   "wall_s": 23.6,
   "turns": null,
   "prompt_last": null,
   "lines_added": 4,
   "lines_deleted": 3,
   "lines": 7,
   "files": 1,
   "thinking": 0,
   "commands": 7,
   "heldout": true,
   "api_cost": 0.003351
  },
  {
   "run": "gpt-6-luna__karpathy__T2__r1__910bf3",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T2",
   "skill": "karpathy",
   "rep": 1,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 910,
   "input": 128308,
   "cache_read": 116736,
   "cache_write": 23315,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.003954,
   "wall_s": 26.3,
   "turns": null,
   "prompt_last": null,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 0,
   "commands": 7,
   "heldout": true,
   "api_cost": 0.003954
  },
  {
   "run": "gpt-6-luna__karpathy__T2__r2__a23d52",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T2",
   "skill": "karpathy",
   "rep": 2,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 646,
   "input": 112637,
   "cache_read": 101632,
   "cache_write": 22748,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.003614,
   "wall_s": 22.6,
   "turns": null,
   "prompt_last": null,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 0,
   "commands": 5,
   "heldout": true,
   "api_cost": 0.003614
  },
  {
   "run": "gpt-6-luna__karpathy__T2__r3__490fc9",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T2",
   "skill": "karpathy",
   "rep": 3,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 733,
   "input": 145633,
   "cache_read": 133888,
   "cache_write": 23488,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.004054,
   "wall_s": 22.4,
   "turns": null,
   "prompt_last": null,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 0,
   "commands": 7,
   "heldout": true,
   "api_cost": 0.004054
  },
  {
   "run": "gpt-6-luna__placebo__T2__r1__463cb8",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T2",
   "skill": "placebo",
   "rep": 1,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 717,
   "input": 110346,
   "cache_read": 101632,
   "cache_write": 20457,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.003421,
   "wall_s": 26.5,
   "turns": null,
   "prompt_last": null,
   "lines_added": 3,
   "lines_deleted": 1,
   "lines": 4,
   "files": 1,
   "thinking": 0,
   "commands": 5,
   "heldout": true,
   "api_cost": 0.003421
  },
  {
   "run": "gpt-6-luna__placebo__T2__r2__4f2553",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T2",
   "skill": "placebo",
   "rep": 2,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 662,
   "input": 124886,
   "cache_read": 116736,
   "cache_write": 19893,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.003488,
   "wall_s": 28.2,
   "turns": null,
   "prompt_last": null,
   "lines_added": 3,
   "lines_deleted": 2,
   "lines": 5,
   "files": 1,
   "thinking": 0,
   "commands": 6,
   "heldout": true,
   "api_cost": 0.003488
  },
  {
   "run": "gpt-6-luna__placebo__T2__r3__59c6a0",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T2",
   "skill": "placebo",
   "rep": 3,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 684,
   "input": 108798,
   "cache_read": 100608,
   "cache_write": 19933,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.003341,
   "wall_s": 21.8,
   "turns": null,
   "prompt_last": null,
   "lines_added": 4,
   "lines_deleted": 3,
   "lines": 7,
   "files": 1,
   "thinking": 0,
   "commands": 6,
   "heldout": true,
   "api_cost": 0.003341
  },
  {
   "run": "gpt-6-luna__ponytail__T2__r1__cd2868",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T2",
   "skill": "ponytail",
   "rep": 1,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 744,
   "input": 125035,
   "cache_read": 114688,
   "cache_write": 22090,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.003728,
   "wall_s": 28.9,
   "turns": null,
   "prompt_last": null,
   "lines_added": 3,
   "lines_deleted": 2,
   "lines": 5,
   "files": 1,
   "thinking": 0,
   "commands": 6,
   "heldout": true,
   "api_cost": 0.003728
  },
  {
   "run": "gpt-6-luna__ponytail__T2__r2__dda70a",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T2",
   "skill": "ponytail",
   "rep": 2,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 744,
   "input": 121673,
   "cache_read": 107520,
   "cache_write": 25896,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.004037,
   "wall_s": 22.6,
   "turns": null,
   "prompt_last": null,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 0,
   "commands": 7,
   "heldout": true,
   "api_cost": 0.004037
  },
  {
   "run": "gpt-6-luna__ponytail__T2__r3__3d5ad7",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T2",
   "skill": "ponytail",
   "rep": 3,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 710,
   "input": 94258,
   "cache_read": 88576,
   "cache_write": 17425,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.002983,
   "wall_s": 22.8,
   "turns": null,
   "prompt_last": null,
   "lines_added": 3,
   "lines_deleted": 2,
   "lines": 5,
   "files": 1,
   "thinking": 0,
   "commands": 5,
   "heldout": true,
   "api_cost": 0.002983
  },
  {
   "run": "gpt-6-luna__caveman__T3__r1__115eaa",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "caveman",
   "rep": 1,
   "round": 6,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 131,
   "input": 30997,
   "cache_read": 26112,
   "cache_write": 16628,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.001989,
   "wall_s": 10.8,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "`monthly_report` filters entries to those whose date has the requested year and month. It returns their count, summed amount, and totals grouped by category.\n\n`parse_date` splits the string into three numeric parts, then builds a calendar date. If the format or calendar date is invalid, it raises `ValueError`. When `add_entry` receives an invalid date, that error stops the entry from being added.",
   "account_remark": false,
   "api_cost": 0.001989
  },
  {
   "run": "gpt-6-luna__caveman__T3__r2__782bfd",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "caveman",
   "rep": 2,
   "round": 6,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 107,
   "input": 30959,
   "cache_read": 26112,
   "cache_write": 16590,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.001974,
   "wall_s": 7.9,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "Monthly totals come from entries whose date year and month match the requested year and month. The report returns their count, summed amount, and totals grouped by category.\n\nAn invalid date string raises `ValueError`. `add_entry` parses the date before appending, so that entry is not added.",
   "account_remark": false,
   "api_cost": 0.001974
  },
  {
   "run": "gpt-6-luna__caveman__T3__r3__6e8b0d",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "caveman",
   "rep": 3,
   "round": 6,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 119,
   "input": 30983,
   "cache_read": 28160,
   "cache_write": 14566,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.001798,
   "wall_s": 6.8,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "`monthly_report` selects entries whose date\u2019s year and month match the requested year and month. It returns their count, summed amount, and totals grouped by category.\n\n`parse_date` raises `ValueError` for malformed or impossible dates. Since `add_entry` parses the date before appending, an invalid date raises an error and adds no entry.",
   "account_remark": false,
   "api_cost": 0.001798
  },
  {
   "run": "gpt-6-luna__control__T3__r1__06d3f3",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "control",
   "rep": 1,
   "round": 6,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 180,
   "input": 27617,
   "cache_read": 24064,
   "cache_write": 15296,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.00186,
   "wall_s": 8.6,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "`monthly_report` first filters entries to those whose `day.year` and `day.month` match the requested year and month. It returns the number of matching entries, their summed amounts, and a per-category sum of those amounts.\n\nDates are parsed as `YYYY-MM-DD` when an entry is added. If the string has the wrong format or represents an invalid calendar date, `parse_date` raises `ValueError`. Since `add_entry` doesn\u2019t catch that error, the entry isn\u2019t added.",
   "account_remark": false,
   "api_cost": 0.00186
  },
  {
   "run": "gpt-6-luna__control__T3__r2__0687a1",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "control",
   "rep": 2,
   "round": 6,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 181,
   "input": 27615,
   "cache_read": 24064,
   "cache_write": 15294,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.001861,
   "wall_s": 9.2,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "`monthly_report(entries, year, month)` first keeps entries whose date has the requested year and month. It then returns their count, the sum of their amounts, and a per-category sum of amounts.\n\nDates are parsed as `YYYY-MM-DD`. If the string has the wrong format, contains non-digits in its components, or describes an invalid calendar date, `parse_date` raises `ValueError`. When called through `add_entry`, that error propagates, so the entry is not appended.",
   "account_remark": false,
   "api_cost": 0.001861
  },
  {
   "run": "gpt-6-luna__control__T3__r3__57601f",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "control",
   "rep": 3,
   "round": 6,
   "date": "2026-09-22",
   "exit": 0,
   "passed": false,
   "output": 47,
   "input": 13572,
   "cache_read": 11008,
   "cache_write": 14307,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.001564,
   "wall_s": 3.7,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 0,
   "answer": "I can\u2019t read `ledger/core.py` because no file-reading or shell tool is available in this session. If you paste the file contents, I can explain the monthly totals and invalid-date behavior without changing anything.",
   "account_remark": false,
   "api_cost": 0.001564
  },
  {
   "run": "gpt-6-luna__karpathy__T3__r1__aa0a9e",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "karpathy",
   "rep": 1,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": false,
   "output": 49,
   "input": 14140,
   "cache_read": 13056,
   "cache_write": 12827,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.001438,
   "wall_s": 3.8,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 0,
   "answer": "Monthly totals are grouped by each transaction\u2019s month, derived from its date, and the amounts are summed within each month. An invalid date string raises a parsing error; it isn\u2019t silently skipped or converted into a default date.",
   "account_remark": false,
   "api_cost": 0.001438
  },
  {
   "run": "gpt-6-luna__karpathy__T3__r2__51835d",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "karpathy",
   "rep": 2,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": false,
   "output": 60,
   "input": 14134,
   "cache_read": 0,
   "cache_write": 25877,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.002618,
   "wall_s": 3.9,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 0,
   "answer": "Monthly totals are computed by parsing each entry\u2019s date, grouping the entry by its year and month, and adding its amount to that month\u2019s running total. If a date string is invalid, parsing raises an error; the function does not skip the entry or return a partial total.",
   "account_remark": false,
   "api_cost": 0.002618
  },
  {
   "run": "gpt-6-luna__karpathy__T3__r3__afd055",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "karpathy",
   "rep": 3,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 177,
   "input": 28729,
   "cache_read": 26112,
   "cache_write": 14360,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.001786,
   "wall_s": 8.1,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "`monthly_report` selects entries whose `day` has the requested year and month, then returns their count, summed amount, and amount totals grouped by category.\n\n`parse_date` splits a string on hyphens and requires exactly three all-digit parts. It then constructs a calendar date, which also rejects impossible dates such as an out-of-range month or day. An invalid string raises `ValueError`; when passed to `add_entry`, that error occurs before an entry is appended.",
   "account_remark": false,
   "api_cost": 0.001786
  },
  {
   "run": "gpt-6-luna__placebo__T3__r1__eeccc9",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "placebo",
   "rep": 1,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": false,
   "output": 31,
   "input": 14038,
   "cache_read": 13056,
   "cache_write": 12725,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.001419,
   "wall_s": 6.8,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 0,
   "answer": "I can\u2019t read `ledger/core.py` because no file-reading tool is available in this session. I haven\u2019t changed any files.",
   "account_remark": false,
   "api_cost": 0.001419
  },
  {
   "run": "gpt-6-luna__placebo__T3__r2__dfa9a5",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "placebo",
   "rep": 2,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": false,
   "output": 54,
   "input": 14041,
   "cache_read": 13056,
   "cache_write": 12728,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.00143,
   "wall_s": 6.3,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 0,
   "answer": "Monthly totals are computed by parsing each entry\u2019s date, grouping its amount by the resulting month, then summing the amounts for each month.\n\nAn invalid date string is caught during parsing, and that entry is skipped rather than included in a monthly total.",
   "account_remark": false,
   "api_cost": 0.00143
  },
  {
   "run": "gpt-6-luna__placebo__T3__r3__d64330",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "placebo",
   "rep": 3,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": false,
   "output": 32,
   "input": 14038,
   "cache_read": 13056,
   "cache_write": 12725,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.001419,
   "wall_s": 3.9,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 0,
   "answer": "I can\u2019t read `ledger/core.py` because I don\u2019t have a file-reading tool available in this turn. No files were changed.",
   "account_remark": false,
   "api_cost": 0.001419
  },
  {
   "run": "gpt-6-luna__ponytail__T3__r1__e85102",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "ponytail",
   "rep": 1,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": false,
   "output": 48,
   "input": 14184,
   "cache_read": 13056,
   "cache_write": 12871,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.001442,
   "wall_s": 4.7,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 0,
   "answer": "Monthly totals are computed by grouping entries by the `YYYY-MM` portion of each date string, then summing their amounts within each month. An invalid date string is skipped, so it contributes nothing to any monthly total.",
   "account_remark": false,
   "api_cost": 0.001442
  },
  {
   "run": "gpt-6-luna__ponytail__T3__r2__523185",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "ponytail",
   "rep": 2,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": false,
   "output": 51,
   "input": 14181,
   "cache_read": 13056,
   "cache_write": 12868,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.001443,
   "wall_s": 7.7,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 0,
   "answer": "I can\u2019t read `ledger/core.py` because I don\u2019t have access to the workspace file tools in this turn. If you paste the file, I\u2019ll explain how it computes monthly totals and handles invalid date strings without changing anything.",
   "account_remark": false,
   "api_cost": 0.001443
  },
  {
   "run": "gpt-6-luna__ponytail__T3__r3__d8a58e",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "ponytail",
   "rep": 3,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": false,
   "output": 48,
   "input": 14190,
   "cache_read": 13056,
   "cache_write": 12877,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.001442,
   "wall_s": 3.5,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 0,
   "answer": "I can\u2019t see the workspace files in this chat, so I can\u2019t accurately describe `ledger/core.py` yet. Paste the file contents here and I\u2019ll explain the monthly totals and invalid-date behavior without changing anything.",
   "account_remark": false,
   "api_cost": 0.001442
  },
  {
   "run": "gpt-6-luna__caveman__T4__r1__6c217a",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T4",
   "skill": "caveman",
   "rep": 1,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1215,
   "input": 188853,
   "cache_read": 175616,
   "cache_write": 24980,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.004862,
   "wall_s": 37.7,
   "turns": null,
   "prompt_last": null,
   "lines_added": 63,
   "lines_deleted": 0,
   "lines": 63,
   "files": 1,
   "thinking": 0,
   "commands": 7,
   "heldout": true,
   "api_cost": 0.004862
  },
  {
   "run": "gpt-6-luna__caveman__T4__r2__37af56",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T4",
   "skill": "caveman",
   "rep": 2,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 863,
   "input": 104086,
   "cache_read": 93696,
   "cache_write": 22133,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.003582,
   "wall_s": 27.1,
   "turns": null,
   "prompt_last": null,
   "lines_added": 55,
   "lines_deleted": 0,
   "lines": 55,
   "files": 1,
   "thinking": 0,
   "commands": 5,
   "heldout": true,
   "api_cost": 0.003582
  },
  {
   "run": "gpt-6-luna__caveman__T4__r3__3171eb",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T4",
   "skill": "caveman",
   "rep": 3,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 929,
   "input": 129437,
   "cache_read": 116992,
   "cache_write": 24188,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.004053,
   "wall_s": 29.4,
   "turns": null,
   "prompt_last": null,
   "lines_added": 50,
   "lines_deleted": 0,
   "lines": 50,
   "files": 1,
   "thinking": 0,
   "commands": 5,
   "heldout": true,
   "api_cost": 0.004053
  },
  {
   "run": "gpt-6-luna__control__T4__r1__e3ea2b",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T4",
   "skill": "control",
   "rep": 1,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1267,
   "input": 137593,
   "cache_read": 127744,
   "cache_write": 21592,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.00407,
   "wall_s": 55.9,
   "turns": null,
   "prompt_last": null,
   "lines_added": 54,
   "lines_deleted": 0,
   "lines": 54,
   "files": 1,
   "thinking": 0,
   "commands": 9,
   "heldout": true,
   "api_cost": 0.00407
  },
  {
   "run": "gpt-6-luna__control__T4__r2__565cde",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T4",
   "skill": "control",
   "rep": 2,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1357,
   "input": 152780,
   "cache_read": 140800,
   "cache_write": 23723,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.004459,
   "wall_s": 40.9,
   "turns": null,
   "prompt_last": null,
   "lines_added": 59,
   "lines_deleted": 0,
   "lines": 59,
   "files": 1,
   "thinking": 0,
   "commands": 8,
   "heldout": true,
   "api_cost": 0.004459
  },
  {
   "run": "gpt-6-luna__control__T4__r3__84dc85",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T4",
   "skill": "control",
   "rep": 3,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1115,
   "input": 106817,
   "cache_read": 98560,
   "cache_write": 20000,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.003543,
   "wall_s": 30.6,
   "turns": null,
   "prompt_last": null,
   "lines_added": 56,
   "lines_deleted": 0,
   "lines": 56,
   "files": 1,
   "thinking": 0,
   "commands": 5,
   "heldout": true,
   "api_cost": 0.003543
  },
  {
   "run": "gpt-6-luna__karpathy__T4__r1__509f17",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T4",
   "skill": "karpathy",
   "rep": 1,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1138,
   "input": 96115,
   "cache_read": 84480,
   "cache_write": 23378,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.003752,
   "wall_s": 29.2,
   "turns": null,
   "prompt_last": null,
   "lines_added": 44,
   "lines_deleted": 0,
   "lines": 44,
   "files": 1,
   "thinking": 0,
   "commands": 9,
   "heldout": true,
   "api_cost": 0.003752
  },
  {
   "run": "gpt-6-luna__karpathy__T4__r2__7fa976",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T4",
   "skill": "karpathy",
   "rep": 2,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1054,
   "input": 110422,
   "cache_read": 101632,
   "cache_write": 20533,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.003597,
   "wall_s": 29.2,
   "turns": null,
   "prompt_last": null,
   "lines_added": 46,
   "lines_deleted": 0,
   "lines": 46,
   "files": 1,
   "thinking": 0,
   "commands": 5,
   "heldout": true,
   "api_cost": 0.003597
  },
  {
   "run": "gpt-6-luna__karpathy__T4__r3__35deea",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T4",
   "skill": "karpathy",
   "rep": 3,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1070,
   "input": 114117,
   "cache_read": 105728,
   "cache_write": 20132,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.003605,
   "wall_s": 31.3,
   "turns": null,
   "prompt_last": null,
   "lines_added": 47,
   "lines_deleted": 0,
   "lines": 47,
   "files": 1,
   "thinking": 0,
   "commands": 5,
   "heldout": true,
   "api_cost": 0.003605
  },
  {
   "run": "gpt-6-luna__placebo__T4__r1__e94bc8",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T4",
   "skill": "placebo",
   "rep": 1,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1099,
   "input": 128597,
   "cache_read": 119808,
   "cache_write": 20532,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.003801,
   "wall_s": 30.4,
   "turns": null,
   "prompt_last": null,
   "lines_added": 44,
   "lines_deleted": 0,
   "lines": 44,
   "files": 1,
   "thinking": 0,
   "commands": 6,
   "heldout": true,
   "api_cost": 0.003801
  },
  {
   "run": "gpt-6-luna__placebo__T4__r2__341e74",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T4",
   "skill": "placebo",
   "rep": 2,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1336,
   "input": 168567,
   "cache_read": 155136,
   "cache_write": 25174,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.004737,
   "wall_s": 37.4,
   "turns": null,
   "prompt_last": null,
   "lines_added": 46,
   "lines_deleted": 0,
   "lines": 46,
   "files": 1,
   "thinking": 0,
   "commands": 7,
   "heldout": true,
   "api_cost": 0.004737
  },
  {
   "run": "gpt-6-luna__placebo__T4__r3__68b1bf",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T4",
   "skill": "placebo",
   "rep": 3,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1371,
   "input": 169270,
   "cache_read": 158208,
   "cache_write": 22805,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.004548,
   "wall_s": 37.2,
   "turns": null,
   "prompt_last": null,
   "lines_added": 50,
   "lines_deleted": 0,
   "lines": 50,
   "files": 1,
   "thinking": 0,
   "commands": 7,
   "heldout": true,
   "api_cost": 0.004548
  },
  {
   "run": "gpt-6-luna__ponytail__T4__r1__d2b120",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T4",
   "skill": "ponytail",
   "rep": 1,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1170,
   "input": 173064,
   "cache_read": 157184,
   "cache_write": 27623,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.004919,
   "wall_s": 38.4,
   "turns": null,
   "prompt_last": null,
   "lines_added": 44,
   "lines_deleted": 0,
   "lines": 44,
   "files": 1,
   "thinking": 0,
   "commands": 7,
   "heldout": true,
   "api_cost": 0.004919
  },
  {
   "run": "gpt-6-luna__ponytail__T4__r2__120519",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T4",
   "skill": "ponytail",
   "rep": 2,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1264,
   "input": 172006,
   "cache_read": 159232,
   "cache_write": 24517,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.004676,
   "wall_s": 36.2,
   "turns": null,
   "prompt_last": null,
   "lines_added": 50,
   "lines_deleted": 0,
   "lines": 50,
   "files": 1,
   "thinking": 0,
   "commands": 7,
   "heldout": true,
   "api_cost": 0.004676
  },
  {
   "run": "gpt-6-luna__ponytail__T4__r3__5507a2",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T4",
   "skill": "ponytail",
   "rep": 3,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1145,
   "input": 167661,
   "cache_read": 155136,
   "cache_write": 24268,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.004551,
   "wall_s": 33.4,
   "turns": null,
   "prompt_last": null,
   "lines_added": 51,
   "lines_deleted": 0,
   "lines": 51,
   "files": 1,
   "thinking": 0,
   "commands": 8,
   "heldout": true,
   "api_cost": 0.004551
  },
  {
   "run": "gpt-6.1-sol__caveman__T1__r1__079fc4",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T1",
   "skill": "caveman",
   "rep": 1,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 555,
   "input": 94913,
   "cache_read": 88576,
   "cache_write": 18834,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.052076,
   "wall_s": 30.3,
   "turns": null,
   "prompt_last": null,
   "lines_added": 6,
   "lines_deleted": 0,
   "lines": 6,
   "files": 1,
   "thinking": 0,
   "commands": 9,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.052076
  },
  {
   "run": "gpt-6.1-sol__caveman__T1__r2__6cbc67",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T1",
   "skill": "caveman",
   "rep": 2,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 542,
   "input": 109616,
   "cache_read": 101120,
   "cache_write": 20993,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.057518,
   "wall_s": 27.5,
   "turns": null,
   "prompt_last": null,
   "lines_added": 6,
   "lines_deleted": 0,
   "lines": 6,
   "files": 1,
   "thinking": 0,
   "commands": 6,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.057518
  },
  {
   "run": "gpt-6.1-sol__caveman__T1__r3__8ed234",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T1",
   "skill": "caveman",
   "rep": 3,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 398,
   "input": 89607,
   "cache_read": 81664,
   "cache_write": 20440,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.053026,
   "wall_s": 21.0,
   "turns": null,
   "prompt_last": null,
   "lines_added": 6,
   "lines_deleted": 0,
   "lines": 6,
   "files": 1,
   "thinking": 0,
   "commands": 4,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.053026
  },
  {
   "run": "gpt-6.1-sol__control__T1__r1__ff72f9",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T1",
   "skill": "control",
   "rep": 1,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 569,
   "input": 93102,
   "cache_read": 87680,
   "cache_write": 17919,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.050296,
   "wall_s": 28.2,
   "turns": null,
   "prompt_last": null,
   "lines_added": 6,
   "lines_deleted": 0,
   "lines": 6,
   "files": 1,
   "thinking": 12,
   "commands": 5,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.050296
  },
  {
   "run": "gpt-6.1-sol__control__T1__r2__c7f0a2",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T1",
   "skill": "control",
   "rep": 2,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 706,
   "input": 76449,
   "cache_read": 71424,
   "cache_write": 17522,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.049246,
   "wall_s": 30.2,
   "turns": null,
   "prompt_last": null,
   "lines_added": 7,
   "lines_deleted": 1,
   "lines": 8,
   "files": 2,
   "thinking": 14,
   "commands": 8,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.049246
  },
  {
   "run": "gpt-6.1-sol__control__T1__r3__c3ff5e",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T1",
   "skill": "control",
   "rep": 3,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 669,
   "input": 76192,
   "cache_read": 71296,
   "cache_write": 17393,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.048606,
   "wall_s": 30.1,
   "turns": null,
   "prompt_last": null,
   "lines_added": 7,
   "lines_deleted": 1,
   "lines": 8,
   "files": 2,
   "thinking": 10,
   "commands": 6,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.048606
  },
  {
   "run": "gpt-6.1-sol__karpathy__T1__r1__6a23c8",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T1",
   "skill": "karpathy",
   "rep": 1,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 620,
   "input": 83395,
   "cache_read": 76416,
   "cache_write": 19476,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.052794,
   "wall_s": 28.2,
   "turns": null,
   "prompt_last": null,
   "lines_added": 5,
   "lines_deleted": 0,
   "lines": 5,
   "files": 1,
   "thinking": 11,
   "commands": 8,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.052794
  },
  {
   "run": "gpt-6.1-sol__karpathy__T1__r2__333f3f",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T1",
   "skill": "karpathy",
   "rep": 2,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 571,
   "input": 101059,
   "cache_read": 93696,
   "cache_write": 19860,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.0548,
   "wall_s": 27.9,
   "turns": null,
   "prompt_last": null,
   "lines_added": 5,
   "lines_deleted": 0,
   "lines": 5,
   "files": 1,
   "thinking": 11,
   "commands": 5,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.0548
  },
  {
   "run": "gpt-6.1-sol__karpathy__T1__r3__10aeb2",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T1",
   "skill": "karpathy",
   "rep": 3,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 556,
   "input": 100086,
   "cache_read": 93184,
   "cache_write": 19399,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.053676,
   "wall_s": 30.2,
   "turns": null,
   "prompt_last": null,
   "lines_added": 5,
   "lines_deleted": 0,
   "lines": 5,
   "files": 1,
   "thinking": 0,
   "commands": 6,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.053676
  },
  {
   "run": "gpt-6.1-sol__placebo__T1__r1__e3bbfa",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T1",
   "skill": "placebo",
   "rep": 1,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 634,
   "input": 96782,
   "cache_read": 90368,
   "cache_write": 18911,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.053199,
   "wall_s": 30.8,
   "turns": null,
   "prompt_last": null,
   "lines_added": 6,
   "lines_deleted": 0,
   "lines": 6,
   "files": 1,
   "thinking": 28,
   "commands": 8,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.053199
  },
  {
   "run": "gpt-6.1-sol__placebo__T1__r2__300555",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T1",
   "skill": "placebo",
   "rep": 2,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 561,
   "input": 80766,
   "cache_read": 74240,
   "cache_write": 19023,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.05108,
   "wall_s": 27.4,
   "turns": null,
   "prompt_last": null,
   "lines_added": 6,
   "lines_deleted": 0,
   "lines": 6,
   "files": 1,
   "thinking": 0,
   "commands": 11,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.05108
  },
  {
   "run": "gpt-6.1-sol__placebo__T1__r3__7ece92",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T1",
   "skill": "placebo",
   "rep": 3,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 443,
   "input": 81559,
   "cache_read": 75392,
   "cache_write": 18664,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.049297,
   "wall_s": 22.4,
   "turns": null,
   "prompt_last": null,
   "lines_added": 6,
   "lines_deleted": 0,
   "lines": 6,
   "files": 1,
   "thinking": 0,
   "commands": 5,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.049297
  },
  {
   "run": "gpt-6.1-sol__ponytail__T1__r1__51132d",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T1",
   "skill": "ponytail",
   "rep": 1,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 536,
   "input": 82036,
   "cache_read": 75776,
   "cache_write": 18757,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.050452,
   "wall_s": 24.7,
   "turns": null,
   "prompt_last": null,
   "lines_added": 5,
   "lines_deleted": 0,
   "lines": 5,
   "files": 1,
   "thinking": 22,
   "commands": 5,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.050452
  },
  {
   "run": "gpt-6.1-sol__ponytail__T1__r2__86eaa6",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T1",
   "skill": "ponytail",
   "rep": 2,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 658,
   "input": 81925,
   "cache_read": 75136,
   "cache_write": 19286,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.052666,
   "wall_s": 30.5,
   "turns": null,
   "prompt_last": null,
   "lines_added": 5,
   "lines_deleted": 0,
   "lines": 5,
   "files": 1,
   "thinking": 58,
   "commands": 9,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.052666
  },
  {
   "run": "gpt-6.1-sol__ponytail__T1__r3__46b96f",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T1",
   "skill": "ponytail",
   "rep": 3,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 615,
   "input": 99444,
   "cache_read": 92800,
   "cache_write": 19141,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.053712,
   "wall_s": 28.2,
   "turns": null,
   "prompt_last": null,
   "lines_added": 5,
   "lines_deleted": 0,
   "lines": 5,
   "files": 1,
   "thinking": 27,
   "commands": 6,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.053712
  },
  {
   "run": "gpt-6.1-sol__caveman__T2__r1__d3a0db",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T2",
   "skill": "caveman",
   "rep": 1,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 471,
   "input": 108588,
   "cache_read": 100352,
   "cache_write": 20733,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.056211,
   "wall_s": 29.3,
   "turns": null,
   "prompt_last": null,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 28,
   "commands": 6,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.056211
  },
  {
   "run": "gpt-6.1-sol__caveman__T2__r2__8f2007",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T2",
   "skill": "caveman",
   "rep": 2,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 486,
   "input": 128852,
   "cache_read": 120320,
   "cache_write": 21029,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.05895,
   "wall_s": 26.0,
   "turns": null,
   "prompt_last": null,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 19,
   "commands": 6,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.05895
  },
  {
   "run": "gpt-6.1-sol__caveman__T2__r3__c8917e",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T2",
   "skill": "caveman",
   "rep": 3,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 381,
   "input": 94812,
   "cache_read": 85248,
   "cache_write": 22061,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.056457,
   "wall_s": 20.2,
   "turns": null,
   "prompt_last": null,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 13,
   "commands": 5,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.056457
  },
  {
   "run": "gpt-6.1-sol__control__T2__r1__4e1056",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T2",
   "skill": "control",
   "rep": 1,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 498,
   "input": 91091,
   "cache_read": 83456,
   "cache_write": 20132,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.05359,
   "wall_s": 25.0,
   "turns": null,
   "prompt_last": null,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 21,
   "commands": 5,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.05359
  },
  {
   "run": "gpt-6.1-sol__control__T2__r2__aac3b9",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T2",
   "skill": "control",
   "rep": 2,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 537,
   "input": 75828,
   "cache_read": 71168,
   "cache_write": 17157,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.046801,
   "wall_s": 26.3,
   "turns": null,
   "prompt_last": null,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 39,
   "commands": 6,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.046801
  },
  {
   "run": "gpt-6.1-sol__control__T2__r3__75efec",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T2",
   "skill": "control",
   "rep": 3,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 652,
   "input": 92101,
   "cache_read": 86912,
   "cache_write": 17686,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.050583,
   "wall_s": 40.6,
   "turns": null,
   "prompt_last": null,
   "lines_added": 3,
   "lines_deleted": 2,
   "lines": 5,
   "files": 1,
   "thinking": 49,
   "commands": 6,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.050583
  },
  {
   "run": "gpt-6.1-sol__karpathy__T2__r1__fd6eb9",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T2",
   "skill": "karpathy",
   "rep": 1,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 580,
   "input": 100468,
   "cache_read": 93440,
   "cache_write": 19525,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.054194,
   "wall_s": 27.0,
   "turns": null,
   "prompt_last": null,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 13,
   "commands": 6,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.054194
  },
  {
   "run": "gpt-6.1-sol__karpathy__T2__r2__5d778d",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T2",
   "skill": "karpathy",
   "rep": 2,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 625,
   "input": 83044,
   "cache_read": 76160,
   "cache_write": 19381,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.052628,
   "wall_s": 30.1,
   "turns": null,
   "prompt_last": null,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 16,
   "commands": 9,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.052628
  },
  {
   "run": "gpt-6.1-sol__karpathy__T2__r3__f3cf22",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T2",
   "skill": "karpathy",
   "rep": 3,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 561,
   "input": 82717,
   "cache_read": 76032,
   "cache_write": 19182,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.051577,
   "wall_s": 25.0,
   "turns": null,
   "prompt_last": null,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 28,
   "commands": 6,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.051577
  },
  {
   "run": "gpt-6.1-sol__placebo__T2__r1__34f6e8",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T2",
   "skill": "placebo",
   "rep": 1,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 583,
   "input": 96100,
   "cache_read": 89984,
   "cache_write": 18613,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.052054,
   "wall_s": 28.4,
   "turns": null,
   "prompt_last": null,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 34,
   "commands": 8,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.052054
  },
  {
   "run": "gpt-6.1-sol__placebo__T2__r2__048216",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T2",
   "skill": "placebo",
   "rep": 2,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 630,
   "input": 96434,
   "cache_read": 89856,
   "cache_write": 19075,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.053436,
   "wall_s": 28.5,
   "turns": null,
   "prompt_last": null,
   "lines_added": 3,
   "lines_deleted": 2,
   "lines": 5,
   "files": 1,
   "thinking": 52,
   "commands": 7,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.053436
  },
  {
   "run": "gpt-6.1-sol__placebo__T2__r3__b5e85f",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T2",
   "skill": "placebo",
   "rep": 3,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 597,
   "input": 97489,
   "cache_read": 90752,
   "cache_write": 19234,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.053513,
   "wall_s": 27.5,
   "turns": null,
   "prompt_last": null,
   "lines_added": 3,
   "lines_deleted": 2,
   "lines": 5,
   "files": 1,
   "thinking": 59,
   "commands": 5,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.053513
  },
  {
   "run": "gpt-6.1-sol__ponytail__T2__r1__46a1b5",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T2",
   "skill": "ponytail",
   "rep": 1,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 542,
   "input": 82066,
   "cache_read": 75648,
   "cache_write": 18915,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.050815,
   "wall_s": 27.3,
   "turns": null,
   "prompt_last": null,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 77,
   "commands": 5,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.050815
  },
  {
   "run": "gpt-6.1-sol__ponytail__T2__r2__ed72b8",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T2",
   "skill": "ponytail",
   "rep": 2,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 555,
   "input": 99029,
   "cache_read": 92416,
   "cache_write": 19110,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.053012,
   "wall_s": 27.6,
   "turns": null,
   "prompt_last": null,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 59,
   "commands": 6,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.053012
  },
  {
   "run": "gpt-6.1-sol__ponytail__T2__r3__cc9edb",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T2",
   "skill": "ponytail",
   "rep": 3,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 604,
   "input": 82544,
   "cache_read": 76032,
   "cache_write": 19009,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.051661,
   "wall_s": 26.5,
   "turns": null,
   "prompt_last": null,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 77,
   "commands": 7,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.051661
  },
  {
   "run": "gpt-6.1-sol__caveman__T3__r1__3dc0c3",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T3",
   "skill": "caveman",
   "rep": 1,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 173,
   "input": 32394,
   "cache_read": 15744,
   "cache_write": 29147,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.061598,
   "wall_s": 10.8,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "`monthly_report` selects entries matching requested year and month. It returns entry count, sum of amounts, and totals grouped by category. Empty month returns count `0`, total `0`, and `{}` for categories.\n\n`parse_date` splits text on `-` and requires three digit-only parts. Malformed strings raise `ValueError` with `bad date: <repr>`. Invalid calendar dates raise `ValueError` from `datetime.date`. Digit widths are not enforced.\n\nIn `add_entry`, date errors propagate; no entry gets appended. No files changed.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "api_cost": 0.061598
  },
  {
   "run": "gpt-6.1-sol__caveman__T3__r2__af9b46",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T3",
   "skill": "caveman",
   "rep": 2,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 186,
   "input": 32398,
   "cache_read": 28032,
   "cache_write": 16863,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.038389,
   "wall_s": 10.1,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "`monthly_report` selects entries matching requested year and month. It returns entry count, sum of amounts, and totals grouped by category. Empty month returns count `0`, total `0`, and empty category dictionary.\n\n`parse_date` requires three hyphen-separated digit-only parts, then constructs Python `date`. Wrong structure raises `ValueError(\"bad date: ...\")`. Impossible dates, such as February 30, raise `ValueError` from `date`. Fixed digit widths are not enforced.\n\n`add_entry` propagates date errors without appending an entry. No files changed.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "api_cost": 0.038389
  },
  {
   "run": "gpt-6.1-sol__caveman__T3__r3__09ad94",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T3",
   "skill": "caveman",
   "rep": 3,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 184,
   "input": 32416,
   "cache_read": 28032,
   "cache_write": 16881,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.038405,
   "wall_s": 11.3,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "`monthly_report` selects entries matching requested year and month. It returns entry count, sum of amounts, and totals grouped by category. Empty month returns count `0`, total `0`, and empty category dictionary.\n\n`parse_date` requires three digit-only parts separated by hyphens, then constructs Python `date`. Malformed strings raise `ValueError` with `bad date: ...`. Impossible dates, such as February 30, raise `ValueError` from `date`. Fixed digit widths aren\u2019t enforced.\n\nIn `add_entry`, invalid date stops operation before append. Entries remain unchanged. No files changed.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "api_cost": 0.038405
  },
  {
   "run": "gpt-6.1-sol__control__T3__r1__3e3546",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T3",
   "skill": "control",
   "rep": 1,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 232,
   "input": 29032,
   "cache_read": 26368,
   "cache_write": 15161,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.035279,
   "wall_s": 13.5,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "In `ledger/core.py`, `monthly_report` selects entries whose date matches the requested year and month. It returns their count, the sum of their amounts, and totals grouped by category. If there are no matching entries, it returns a count and total of zero and an empty category dictionary.\n\n`parse_date` splits a date string on hyphens and requires exactly three digit-only parts. Malformed strings raise `ValueError` with a `bad date` message. Those parts are converted to integers and passed to Python\u2019s `date` constructor, which also raises `ValueError` for impossible dates, such as February 30. Exact digit widths are not enforced, so `2026-9-2` is accepted.\n\nWhen `add_entry` encounters an invalid date, the error propagates and no entry is appended. No files were changed.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "api_cost": 0.035279
  },
  {
   "run": "gpt-6.1-sol__control__T3__r2__7a1b3d",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T3",
   "skill": "control",
   "rep": 2,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 302,
   "input": 43545,
   "cache_read": 40576,
   "cache_write": 15466,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.03801,
   "wall_s": 17.0,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 2,
   "answer": "In [ledger/core.py](/tmp/bench-work/gpt-6.1-sol__control__T3__r2__7a1b3d/ledger/core.py), `monthly_report` selects entries whose date matches the requested year and month. It returns their count, the sum of their amounts, and totals grouped by category. With no matching entries, these are `0`, `0`, and `{}`.\n\n`parse_date` splits a date string on hyphens and requires three digit-only parts. Malformed strings raise `ValueError` with a `bad date` message; impossible dates, such as February 30, raise `ValueError` from Python\u2019s `date` constructor. Despite the docstring\u2019s `YYYY-MM-DD` wording, fixed digit widths are not enforced.\n\nWhen `add_entry` encounters an invalid date, the exception propagates and no entry is appended. No files were changed.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "api_cost": 0.03801
  },
  {
   "run": "gpt-6.1-sol__control__T3__r3__995a6d",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T3",
   "skill": "control",
   "rep": 3,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 286,
   "input": 43522,
   "cache_read": 40576,
   "cache_write": 15443,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.037804,
   "wall_s": 16.2,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 2,
   "answer": "In [ledger/core.py](/tmp/bench-work/gpt-6.1-sol__control__T3__r3__995a6d/ledger/core.py), `monthly_report` selects entries whose dates match the requested year and month. It returns their count, the sum of their amounts, and totals grouped by category. With no matching entries, these are `0`, `0`, and `{}`.\n\n`parse_date` requires three hyphen-separated numeric parts, then passes them to `datetime.date`. Malformed strings raise `ValueError` with a `bad date` message; impossible dates, such as February 30, raise `ValueError` from `datetime.date`. It does not enforce exact `YYYY-MM-DD` digit widths. When called through `add_entry`, an invalid date prevents the entry from being appended.\n\nNo files were changed.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "api_cost": 0.037804
  },
  {
   "run": "gpt-6.1-sol__karpathy__T3__r1__17c810",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T3",
   "skill": "karpathy",
   "rep": 1,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 237,
   "input": 30124,
   "cache_read": 26880,
   "cache_write": 15741,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.03654,
   "wall_s": 12.3,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "In `ledger/core.py`, `monthly_report` selects entries whose dates match the requested year and month. It returns their count, the sum of their amounts, and a dictionary of totals grouped by category. If no entries match, these are `0`, `0`, and `{}`.\n\n`parse_date` splits a date string on hyphens and requires three digit-only parts. A malformed string raises `ValueError` with a message such as `bad date: 'invalid'`. Numeric parts are converted to integers and passed to Python\u2019s `date` constructor, which raises `ValueError` for impossible dates, such as February 30. Despite the docstring\u2019s `YYYY-MM-DD` wording, fixed digit widths aren\u2019t enforced.\n\nWhen `add_entry` encounters an invalid date, the exception propagates and no entry is appended. No files were changed.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "api_cost": 0.03654
  },
  {
   "run": "gpt-6.1-sol__karpathy__T3__r2__e5a294",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T3",
   "skill": "karpathy",
   "rep": 2,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 262,
   "input": 30129,
   "cache_read": 26880,
   "cache_write": 15746,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.0368,
   "wall_s": 12.8,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "In [ledger/core.py](/tmp/bench-work/gpt-6.1-sol__karpathy__T3__r2__e5a294/ledger/core.py), `monthly_report` selects entries whose dates match the requested year and month. It returns their count, the sum of their amounts, and totals grouped by category. With no matching entries, these are `0`, `0`, and an empty dictionary.\n\n`parse_date` splits the string on hyphens and requires exactly three digit-only parts. Otherwise, it raises `ValueError` with a message such as `bad date: 'invalid'`. It converts those parts to integers and constructs a Python `date`, which also raises `ValueError` for impossible dates, such as February 30. Component widths aren\u2019t enforced, so `2026-2-3` is accepted.\n\nWhen `add_entry` encounters an invalid date, the exception propagates and no entry is appended. No files were changed.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "api_cost": 0.0368
  },
  {
   "run": "gpt-6.1-sol__karpathy__T3__r3__018578",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T3",
   "skill": "karpathy",
   "rep": 3,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 253,
   "input": 30117,
   "cache_read": 26880,
   "cache_write": 15734,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.036686,
   "wall_s": 13.4,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "In [ledger/core.py](/tmp/bench-work/gpt-6.1-sol__karpathy__T3__r3__018578/ledger/core.py), `monthly_report` selects entries whose date matches the requested year and month. It returns their count, the sum of their amounts, and totals grouped by category. An empty month produces a count and total of zero and an empty category dictionary.\n\n`parse_date` splits a string on hyphens and requires exactly three digit-only parts. Malformed strings raise `ValueError` with a `bad date: ...` message. Numeric parts are passed to Python\u2019s `date` constructor, which raises `ValueError` for impossible dates, such as February 30. Although the docstring specifies `YYYY-MM-DD`, fixed digit widths are not enforced.\n\nWhen `add_entry` encounters an invalid date, the exception propagates and no entry is appended. No files were changed.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "api_cost": 0.036686
  },
  {
   "run": "gpt-6.1-sol__placebo__T3__r1__6d9f9d",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T3",
   "skill": "placebo",
   "rep": 1,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 216,
   "input": 29951,
   "cache_read": 26752,
   "cache_write": 15696,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.036227,
   "wall_s": 12.8,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "`monthly_report` selects entries whose dates match the requested year and month. It returns their count, the sum of their amounts, and totals grouped by category. An empty month returns a count and total of zero and an empty category dictionary.\n\n`parse_date` splits a string on hyphens and requires three digit-only parts. If that check fails, it raises `ValueError` with `bad date: <repr of input>`. Otherwise, it converts the parts to integers and constructs a `datetime.date`, which raises `ValueError` for impossible dates or out-of-range values. It does not enforce exact `YYYY-MM-DD` widths. When parsing fails in `add_entry`, the exception propagates and no entry is appended.\n\nNo files were changed.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "api_cost": 0.036227
  },
  {
   "run": "gpt-6.1-sol__placebo__T3__r2__c5386c",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T3",
   "skill": "placebo",
   "rep": 2,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 225,
   "input": 29970,
   "cache_read": 26880,
   "cache_write": 15587,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.036112,
   "wall_s": 13.8,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "`monthly_report` selects entries whose dates match the requested year and month. It returns their count, the sum of their amounts, and totals grouped by category. If no entries match, the count and total are zero, and the category totals are an empty dictionary.\n\n`parse_date` requires three hyphen-separated digit strings, converts them to integers, and constructs a Python `date`. Malformed strings raise `ValueError` with a message such as `bad date: 'invalid'`. Impossible dates, such as February 30, raise `ValueError` from the `date` constructor. Despite the docstring\u2019s `YYYY-MM-DD` wording, fixed digit widths are not enforced.\n\nWhen `add_entry` encounters an invalid date, the exception propagates and no entry is appended. No files were changed.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "api_cost": 0.036112
  },
  {
   "run": "gpt-6.1-sol__placebo__T3__r3__0c9f5a",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T3",
   "skill": "placebo",
   "rep": 3,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 226,
   "input": 29948,
   "cache_read": 26752,
   "cache_write": 15693,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.036321,
   "wall_s": 11.6,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "`monthly_report` selects entries whose date matches the requested year and month. It returns their count, the sum of their amounts, and a dictionary of totals grouped by category. If no entries match, the count and total are zero and the category dictionary is empty.\n\n`parse_date` splits the string on hyphens and requires three digit-only parts. Otherwise, it raises `ValueError` with `bad date: <repr of input>`. It converts those parts to integers and calls Python\u2019s `date` constructor, which raises `ValueError` for impossible dates or out-of-range values. Despite the docstring\u2019s `YYYY-MM-DD` wording, fixed digit widths are not enforced.\n\nWhen `add_entry` encounters an invalid date, the exception propagates and no entry is appended. No files were changed.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "api_cost": 0.036321
  },
  {
   "run": "gpt-6.1-sol__ponytail__T3__r1__1039a7",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T3",
   "skill": "ponytail",
   "rep": 1,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 223,
   "input": 30228,
   "cache_read": 27008,
   "cache_write": 15717,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.036365,
   "wall_s": 12.5,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "In `ledger/core.py`, `monthly_report` selects entries whose date matches the requested year and month. It returns their count, the sum of their amounts, and totals grouped by category. An empty month produces a count and total of zero and an empty category dictionary.\n\n`parse_date` splits the date string on hyphens and requires three digit-only parts. Malformed input raises `ValueError` with a `bad date` message. The parts then become integers passed to Python\u2019s `date` constructor, which raises `ValueError` for impossible dates, such as February 30. Despite the docstring\u2019s `YYYY-MM-DD` wording, fixed-width components are not enforced.\n\nWhen `add_entry` encounters an invalid date, the exception propagates and no entry is appended. No files were changed.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "api_cost": 0.036365
  },
  {
   "run": "gpt-6.1-sol__ponytail__T3__r2__d05495",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T3",
   "skill": "ponytail",
   "rep": 2,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 211,
   "input": 30222,
   "cache_read": 27008,
   "cache_write": 15711,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.036233,
   "wall_s": 11.3,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "`monthly_report` selects entries whose date matches the requested year and month. It returns their count, the sum of their amounts, and totals grouped by category. An empty month returns a count and total of `0`, with an empty category dictionary.\n\n`parse_date` requires three hyphen-separated digit-only parts, then passes their integer values to Python\u2019s `date` constructor. Malformed strings raise `ValueError` with a `bad date` message; impossible dates, such as February 30, raise `ValueError` from the constructor. Despite the docstring, exact `YYYY-MM-DD` widths aren\u2019t enforced.\n\nWhen `add_entry` encounters an invalid date, the exception propagates and no entry is appended. No files were changed.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "api_cost": 0.036233
  },
  {
   "run": "gpt-6.1-sol__ponytail__T3__r3__ceb680",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T3",
   "skill": "ponytail",
   "rep": 3,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 211,
   "input": 30248,
   "cache_read": 27008,
   "cache_write": 15737,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.036285,
   "wall_s": 11.1,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "`monthly_report` selects entries whose date matches the requested year and month. It returns their count, the sum of their amounts, and totals grouped by category. With no matching entries, it returns a count of `0`, a total of `0`, and an empty category dictionary.\n\n`parse_date` splits the string on hyphens and requires three digit-only parts. Malformed strings raise `ValueError` with a `bad date` message; impossible calendar dates raise `ValueError` from Python\u2019s `date` constructor. Despite the docstring\u2019s `YYYY-MM-DD` wording, fixed-width parts aren\u2019t enforced.\n\nWhen `add_entry` encounters an invalid date, the exception propagates and no entry is appended. No files were changed.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "api_cost": 0.036285
  },
  {
   "run": "gpt-6.1-sol__caveman__T4__r1__b2ac80",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T4",
   "skill": "caveman",
   "rep": 1,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 896,
   "input": 98337,
   "cache_read": 86912,
   "cache_write": 23922,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.065495,
   "wall_s": 35.9,
   "turns": null,
   "prompt_last": null,
   "lines_added": 51,
   "lines_deleted": 0,
   "lines": 51,
   "files": 1,
   "thinking": 0,
   "commands": 5,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.065495
  },
  {
   "run": "gpt-6.1-sol__caveman__T4__r2__a9a43a",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T4",
   "skill": "caveman",
   "rep": 2,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 1082,
   "input": 112440,
   "cache_read": 102144,
   "cache_write": 22793,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.06662,
   "wall_s": 44.3,
   "turns": null,
   "prompt_last": null,
   "lines_added": 60,
   "lines_deleted": 0,
   "lines": 60,
   "files": 1,
   "thinking": 44,
   "commands": 7,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.06662
  },
  {
   "run": "gpt-6.1-sol__caveman__T4__r3__5b07af",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T4",
   "skill": "caveman",
   "rep": 3,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 936,
   "input": 119456,
   "cache_read": 107904,
   "cache_write": 24049,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.068248,
   "wall_s": 39.0,
   "turns": null,
   "prompt_last": null,
   "lines_added": 57,
   "lines_deleted": 0,
   "lines": 57,
   "files": 1,
   "thinking": 0,
   "commands": 7,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.068248
  },
  {
   "run": "gpt-6.1-sol__control__T4__r1__2afbad",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T4",
   "skill": "control",
   "rep": 1,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 1026,
   "input": 94856,
   "cache_read": 88192,
   "cache_write": 19161,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.057401,
   "wall_s": 40.7,
   "turns": null,
   "prompt_last": null,
   "lines_added": 50,
   "lines_deleted": 0,
   "lines": 50,
   "files": 1,
   "thinking": 19,
   "commands": 7,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.057401
  },
  {
   "run": "gpt-6.1-sol__control__T4__r2__3e0313",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T4",
   "skill": "control",
   "rep": 2,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 1077,
   "input": 95504,
   "cache_read": 89472,
   "cache_write": 18529,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.056775,
   "wall_s": 42.1,
   "turns": null,
   "prompt_last": null,
   "lines_added": 54,
   "lines_deleted": 0,
   "lines": 54,
   "files": 1,
   "thinking": 18,
   "commands": 5,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.056775
  },
  {
   "run": "gpt-6.1-sol__control__T4__r3__d879d9",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T4",
   "skill": "control",
   "rep": 3,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 978,
   "input": 79729,
   "cache_read": 73344,
   "cache_write": 18882,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.054878,
   "wall_s": 40.2,
   "turns": null,
   "prompt_last": null,
   "lines_added": 50,
   "lines_deleted": 0,
   "lines": 50,
   "files": 1,
   "thinking": 12,
   "commands": 6,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.054878
  },
  {
   "run": "gpt-6.1-sol__karpathy__T4__r1__51edef",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T4",
   "skill": "karpathy",
   "rep": 1,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 1035,
   "input": 84859,
   "cache_read": 74624,
   "cache_write": 22732,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.063276,
   "wall_s": 43.6,
   "turns": null,
   "prompt_last": null,
   "lines_added": 45,
   "lines_deleted": 0,
   "lines": 45,
   "files": 1,
   "thinking": 131,
   "commands": 6,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.063276
  },
  {
   "run": "gpt-6.1-sol__karpathy__T4__r2__340bbf",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T4",
   "skill": "karpathy",
   "rep": 2,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 1060,
   "input": 85048,
   "cache_read": 74624,
   "cache_write": 22921,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.063904,
   "wall_s": 45.0,
   "turns": null,
   "prompt_last": null,
   "lines_added": 47,
   "lines_deleted": 0,
   "lines": 47,
   "files": 1,
   "thinking": 86,
   "commands": 7,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.063904
  },
  {
   "run": "gpt-6.1-sol__karpathy__T4__r3__42329c",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T4",
   "skill": "karpathy",
   "rep": 3,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 1042,
   "input": 85639,
   "cache_read": 77312,
   "cache_write": 20824,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.059799,
   "wall_s": 42.2,
   "turns": null,
   "prompt_last": null,
   "lines_added": 46,
   "lines_deleted": 0,
   "lines": 46,
   "files": 1,
   "thinking": 62,
   "commands": 8,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.059799
  },
  {
   "run": "gpt-6.1-sol__placebo__T4__r1__93a359",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T4",
   "skill": "placebo",
   "rep": 1,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 1168,
   "input": 102503,
   "cache_read": 94208,
   "cache_write": 20792,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.062685,
   "wall_s": 44.4,
   "turns": null,
   "prompt_last": null,
   "lines_added": 66,
   "lines_deleted": 0,
   "lines": 66,
   "files": 1,
   "thinking": 31,
   "commands": 6,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.062685
  },
  {
   "run": "gpt-6.1-sol__placebo__T4__r2__e29288",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T4",
   "skill": "placebo",
   "rep": 2,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 1144,
   "input": 101878,
   "cache_read": 87936,
   "cache_write": 26439,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.073112,
   "wall_s": 44.2,
   "turns": null,
   "prompt_last": null,
   "lines_added": 66,
   "lines_deleted": 0,
   "lines": 66,
   "files": 1,
   "thinking": 12,
   "commands": 6,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.073112
  },
  {
   "run": "gpt-6.1-sol__placebo__T4__r3__c06e6f",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T4",
   "skill": "placebo",
   "rep": 3,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 1318,
   "input": 101461,
   "cache_read": 93184,
   "cache_write": 20774,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.064046,
   "wall_s": 51.6,
   "turns": null,
   "prompt_last": null,
   "lines_added": 67,
   "lines_deleted": 0,
   "lines": 67,
   "files": 1,
   "thinking": 36,
   "commands": 6,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.064046
  },
  {
   "run": "gpt-6.1-sol__ponytail__T4__r1__338139",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T4",
   "skill": "ponytail",
   "rep": 1,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 1044,
   "input": 84678,
   "cache_read": 76416,
   "cache_write": 20759,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.0596,
   "wall_s": 40.1,
   "turns": null,
   "prompt_last": null,
   "lines_added": 43,
   "lines_deleted": 0,
   "lines": 43,
   "files": 1,
   "thinking": 109,
   "commands": 8,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.0596
  },
  {
   "run": "gpt-6.1-sol__ponytail__T4__r2__45ab06",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T4",
   "skill": "ponytail",
   "rep": 2,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 1124,
   "input": 102625,
   "cache_read": 92032,
   "cache_write": 23090,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.066623,
   "wall_s": 44.4,
   "turns": null,
   "prompt_last": null,
   "lines_added": 41,
   "lines_deleted": 0,
   "lines": 41,
   "files": 1,
   "thinking": 212,
   "commands": 6,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.066623
  },
  {
   "run": "gpt-6.1-sol__ponytail__T4__r3__e2f54c",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T4",
   "skill": "ponytail",
   "rep": 3,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 1180,
   "input": 86560,
   "cache_read": 77824,
   "cache_write": 21233,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.062048,
   "wall_s": 46.4,
   "turns": null,
   "prompt_last": null,
   "lines_added": 43,
   "lines_deleted": 0,
   "lines": 43,
   "files": 1,
   "thinking": 265,
   "commands": 7,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.062048
  }
 ],
 "recheck_runs": [
  {
   "run": "claude-sonnet-5__caveman__T3__r4__f175a5",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "caveman",
   "rep": 4,
   "round": 5,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 292,
   "input": 4,
   "cache_read": 59211,
   "cache_write": 22752,
   "warmup": null,
   "cache_write_1h": 22752,
   "cost_usd": 0.1057782,
   "wall_s": 5.5,
   "turns": 2,
   "prompt_last": 41393,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "Monthly total: `monthly_report` call `entries_in_month`, filter entries where `day.year == year` and `day.month == month`. Then `total()` sum `amount` field over those entries. Report also add count and per-category sum via `sum_by_category`.\n\nInvalid date: `parse_date` split text on `-`. If not exact 3 parts, or any part not digit, raise `ValueError(f\"bad date: {text!r}\")`. Called from `add_entry` when entry created, entry never add to list, `ValueError` propagate to caller. No catch anywhere in file \u2014 error not silent, must handle upstream or crash.",
   "account_remark": false,
   "api_cost": 0.105778
  },
  {
   "run": "claude-sonnet-5__caveman__T3__r5__c23bd8",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "caveman",
   "rep": 5,
   "round": 5,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 291,
   "input": 4,
   "cache_read": 49087,
   "cache_write": 12981,
   "warmup": null,
   "cache_write_1h": 12981,
   "cost_usd": 0.0646594,
   "wall_s": 4.5,
   "turns": 2,
   "prompt_last": 31622,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 12,
   "commands": null,
   "answer": "`monthly_report` filter entries by year+month (`entries_in_month`), sum amount field (`total`), group sums by category (`sum_by_category`). Return count, total, by_category dict.\n\nInvalid date string: `parse_date` split on \"-\", check 3 parts, all digit. Fail either check, raise `ValueError(f\"bad date: {text!r}\")`. Called from `add_entry`, so bad date reject entry before add, not during monthly total. Once entry stored, date always valid date object, so `entries_in_month`/`monthly_report` never see invalid string.",
   "account_remark": false,
   "api_cost": 0.064659,
   "light": true
  },
  {
   "run": "claude-sonnet-5__caveman__T3__r6__355400",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "caveman",
   "rep": 6,
   "round": 5,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 324,
   "input": 4,
   "cache_read": 59205,
   "cache_write": 22744,
   "warmup": null,
   "cache_write_1h": 22744,
   "cost_usd": 0.10606499999999999,
   "wall_s": 7.6,
   "turns": 2,
   "prompt_last": 41385,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "Monthly total flow: `monthly_report` call `entries_in_month` filter entries by year+month, then `total` sum their `amount` field (plain float sum, no rounding). `by_category` breakdown come from `sum_by_category` on same filtered set.\n\nInvalid date string: `parse_date` split on `-`, need exactly 3 parts, all digit-only. Fail either check \u2192 raise `ValueError(f\"bad date: {text!r}\")`. Called from `add_entry`, so bad date reject entry at add time, before it ever reach monthly totals. Note: digit-only check miss out-of-range values like month 13 or day 32 \u2014 those pass split check but raise a different `ValueError` from `date()` constructor itself.",
   "account_remark": false,
   "api_cost": 0.106065
  },
  {
   "run": "claude-sonnet-5__control__T3__r4__c1ae1e",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "control",
   "rep": 4,
   "round": 5,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 858,
   "input": 6,
   "cache_read": 94681,
   "cache_write": 20294,
   "warmup": null,
   "cache_write_1h": 20294,
   "cost_usd": 0.1087042,
   "wall_s": 10.2,
   "turns": 3,
   "prompt_last": 38935,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 25,
   "commands": null,
   "answer": "## Monthly totals\n\n`monthly_report(entries, year, month)` (ledger/core.py:51-53) drives this:\n\n1. `entries_in_month` filters `entries` down to those whose `day.year` and `day.month` match the requested year/month.\n2. `total` sums the `amount` field across that filtered list to produce `\"total\"`.\n3. `sum_by_category` separately walks the same filtered list and accumulates amounts into a dict keyed by (lowercased) category, for `\"by_category\"`.\n4. The result also includes `\"count\"`, the number of matching entries.\n\nSo the monthly total is just a plain sum of `amount` over entries whose `day` falls in that year/month \u2014 no weighting, rounding, or currency conversion involved.\n\n## Invalid date strings\n\nDates only enter the ledger via `add_entry`, which calls `parse_date(day)` (ledger/core.py:15-21) before constructing an `Entry`. `parse_date`:\n\n- Splits the string on `\"-\"`.\n- Requires exactly 3 parts, each all-digit (`isdigit()`), otherwise raises `ValueError(f\"bad date: {text!r}\")`.\n- Otherwise converts the three parts to ints and passes them to `date(y, m, d)`.\n\nSo malformed strings like `\"2024/01/01\"` or `\"abc\"` fail the format check and raise `ValueError` immediately. But strings that pass the format check yet are semantically invalid \u2014 e.g. `\"2024-13-40\"` \u2014 skip the custom check and hit `date(2024, 13, 40)` directly, which raises its own `ValueError` (e.g. \"month must be in 1..12\") from the stdlib `datetime.date` constructor. Either way, no invalid entry is ever added: `add_entry` doesn't catch the exception, so it propagates to the caller and the entry never makes it into `entries`, meaning it can never affect a later `monthly_report`/`total` computation.",
   "account_remark": false,
   "api_cost": 0.108704
  },
  {
   "run": "claude-sonnet-5__control__T3__r5__fe7a02",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "control",
   "rep": 5,
   "round": 5,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 793,
   "input": 6,
   "cache_read": 94719,
   "cache_write": 20331,
   "warmup": null,
   "cache_write_1h": 20331,
   "cost_usd": 0.1082098,
   "wall_s": 10.2,
   "turns": 3,
   "prompt_last": 38972,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 30,
   "commands": null,
   "answer": "Here's how it works:\n\n**Monthly totals.** `monthly_report(entries, year, month)` first calls `entries_in_month`, which filters the full entry list down to those whose `day.year` and `day.month` match the requested year/month. That filtered list is then run through `total()` (a plain `sum()` of `e.amount` across the matched entries) to get the `\"total\"` field, and through `sum_by_category()` to build a per-category breakdown dict. The report also includes `\"count\"`, the number of matching entries. So the \"monthly total\" is simply the sum of amounts for every entry whose date falls in that year/month \u2014 there's no separate month-aware accumulation logic; it's filter-then-sum.\n\n**Invalid date strings.** Dates enter the system via `add_entry`, which calls `parse_date(day)` before constructing an `Entry`. `parse_date` splits the string on `-` and requires exactly 3 parts, all of which must be purely digit strings (`str.isdigit()`); if that check fails, it raises `ValueError(f\"bad date: {text!r}\")` immediately. If the format check passes but the numeric values are out of range (e.g., month 13, day 31 in February), the `date(y, m, d)` constructor itself raises a `ValueError` (uncaught, so it propagates with Python's own message rather than the custom one). Either way, an invalid date string causes `add_entry` to raise before the entry is ever appended \u2014 it's never added to the ledger, so it can't silently corrupt a monthly total. There's no `try/except` anywhere in this file, so callers of `add_entry` must handle the `ValueError` themselves if they want to recover instead of crashing.",
   "account_remark": false,
   "api_cost": 0.10821
  },
  {
   "run": "claude-sonnet-5__control__T3__r6__a8ddbc",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "control",
   "rep": 6,
   "round": 5,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 793,
   "input": 6,
   "cache_read": 94676,
   "cache_write": 20288,
   "warmup": null,
   "cache_write_1h": 20288,
   "cost_usd": 0.10802920000000002,
   "wall_s": 11.7,
   "turns": 3,
   "prompt_last": 38929,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 37,
   "commands": null,
   "answer": "Here's how it works:\n\n**Monthly totals.** `monthly_report(entries, year, month)` (line 51-53) first calls `entries_in_month`, which filters the full entry list down to just those whose `day.year` and `day.month` match the requested values (line 40-41). It then builds a dict with the count of matching entries, the sum of their `amount` fields via `total()` (a plain `sum()` over the filtered list, line 31-32), and a per-category breakdown via `sum_by_category()`, which accumulates amounts into a dict keyed by lowercased category (line 44-48).\n\nNote that this filtering happens on `Entry.day`, a `date` object \u2014 so the \"monthly total\" depends entirely on entries already having been constructed with valid dates.\n\n**Invalid date strings.** Dates only enter the system through `add_entry`, which calls `parse_date(day)` (line 27) before constructing the `Entry`. `parse_date` (line 15-21) splits the input on `-` and requires exactly 3 parts that are all digit strings; if not, it raises `ValueError(f\"bad date: {text!r}\")` immediately (line 18-19). If the format looks right but the numbers are out of range (e.g. month 13, day 31 of February), the underlying `date(y, m, d)` constructor call (line 21) raises its own `ValueError` (e.g. \"month must be in 1..12\").\n\nEither way, there's no catching or swallowing of the exception anywhere in this file \u2014 an invalid date string propagates as an uncaught `ValueError` all the way up through `add_entry`, so the entry is never added to the list and the caller must handle the exception themselves. There's no partial-entry or default-date fallback behavior.",
   "account_remark": false,
   "api_cost": 0.108029
  },
  {
   "run": "gpt-5.6-luna__caveman__T3__r4__c5b334",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "caveman",
   "rep": 4,
   "round": 5,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 218,
   "input": 26187,
   "cache_read": 20992,
   "cache_write": 14946,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.003671,
   "wall_s": 10.7,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "Monthly totals:\n\n- `monthly_report(entries, year, month)` filters entries matching given year and month.\n- It returns:\n  - `count`: number of matching entries.\n  - `total`: sum of their amounts.\n  - `by_category`: amounts summed separately by category.\n\nInvalid date strings:\n\n- `parse_date` requires `YYYY-MM-DD`, with three numeric parts.\n- Wrong format raises `ValueError(\"bad date: ...\")`.\n- Valid-looking but impossible dates, such as `2024-02-30`, also raise `ValueError` from `datetime.date`.\n- `add_entry` does not add entry when parsing fails.",
   "account_remark": false,
   "api_cost": 0.003671
  },
  {
   "run": "gpt-5.6-luna__caveman__T3__r5__1b869a",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "caveman",
   "rep": 5,
   "round": 5,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 230,
   "input": 26211,
   "cache_read": 20992,
   "cache_write": 14970,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.00369,
   "wall_s": 11.8,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 22,
   "commands": 1,
   "answer": "Monthly totals:\n\n- `entries_in_month` keeps entries whose `day.year` and `day.month` match requested year and month.\n- `monthly_report` counts those entries.\n- `total` sums their `amount` values.\n- `sum_by_category` groups amounts by lowercase category.\n- Report returns `count`, `total`, and `by_category`.\n\nInvalid date strings cause `parse_date` to raise `ValueError`. This happens when format is not `YYYY-MM-DD`, parts are not numeric, or date values are invalid. `add_entry` stops; no entry is added.",
   "account_remark": false,
   "api_cost": 0.00369
  },
  {
   "run": "gpt-5.6-luna__caveman__T3__r6__a9ba72",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "caveman",
   "rep": 6,
   "round": 5,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 231,
   "input": 26198,
   "cache_read": 20992,
   "cache_write": 14957,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.003688,
   "wall_s": 28.2,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 9,
   "commands": 1,
   "answer": "Monthly totals:\n\n1. `entries_in_month` keeps entries whose `day.year` and `day.month` match requested year and month.\n2. `total` sums their `amount` values.\n3. `monthly_report` returns entry count, total amount, and category totals.\n\nInvalid date string:\n\n- `parse_date` requires three numeric parts in `YYYY-MM-DD` form.\n- Wrong format raises `ValueError(\"bad date: ...\")`.\n- Invalid calendar dates, such as month `13` or day `31` in February, also raise `ValueError` from `date`.\n- `add_entry` does not append entry when parsing fails.",
   "account_remark": false,
   "api_cost": 0.003688
  },
  {
   "run": "gpt-5.6-luna__control__T3__r4__8b0909",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "control",
   "rep": 4,
   "round": 5,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 253,
   "input": 22662,
   "cache_read": 18944,
   "cache_write": 13469,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.003376,
   "wall_s": 12.6,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 14,
   "commands": 1,
   "answer": "Monthly totals are computed by first selecting entries whose `day.year` and `day.month` match the requested year and month. The report then sums their `amount` values directly, returning the result under `\"total\"`; it also includes the entry count and category subtotals.\n\nDate strings must be in `YYYY-MM-DD` form with three numeric components. Malformed strings raise `ValueError` with a `bad date: ...` message. Valid-looking but impossible dates\u2014such as `2024-02-30`\u2014also raise `ValueError` from Python\u2019s `date` constructor. No entry is added when parsing fails.",
   "account_remark": false,
   "api_cost": 0.003376
  },
  {
   "run": "gpt-5.6-luna__control__T3__r5__0f162e",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "control",
   "rep": 5,
   "round": 5,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 217,
   "input": 22822,
   "cache_read": 17920,
   "cache_write": 14653,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.003549,
   "wall_s": 14.2,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 9,
   "commands": 1,
   "answer": "Monthly totals are computed by `monthly_report(entries, year, month)`. It first selects entries whose `day.year` and `day.month` match the requested year and month. It then returns the number of matching entries, their summed `amount`, and category-level subtotals.\n\nDate strings are parsed as `YYYY-MM-DD`. Invalid formats, non-numeric components, or impossible calendar dates cause `parse_date` to raise `ValueError`; consequently, `add_entry` fails and does not append an entry.",
   "account_remark": false,
   "api_cost": 0.003549
  },
  {
   "run": "gpt-5.6-luna__control__T3__r6__c8af3e",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "control",
   "rep": 6,
   "round": 5,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 232,
   "input": 22840,
   "cache_read": 17920,
   "cache_write": 14671,
   "warmup": 9751,
   "cache_write_1h": null,
   "cost_usd": 0.003571,
   "wall_s": 8.1,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 16,
   "commands": 1,
   "answer": "Monthly totals are computed by `monthly_report(entries, year, month)`. It first filters entries whose `day.year` and `day.month` match the requested year and month. It then returns the number of matching entries, their summed amount, and category-level subtotals.\n\nAn invalid date string causes `parse_date` to raise `ValueError`. This happens for malformed formats, non-numeric components, or impossible dates such as month 13. Consequently, `add_entry` fails and does not append the entry.",
   "account_remark": false,
   "api_cost": 0.003571
  }
 ],
 "round8_runs": [
  {
   "run": "claude-fable-5-1__caveman__T3__r1__c33e3f",
   "model": "claude-fable-5-1",
   "host": "claude",
   "task": "T3",
   "skill": "caveman",
   "rep": 1,
   "round": 8,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 1725,
   "input": 4,
   "cache_read": 28653,
   "cache_write": 13163,
   "warmup": null,
   "cache_write_1h": 13163,
   "cost_usd": 0.35671325,
   "wall_s": 23.5,
   "turns": 2,
   "prompt_last": 23283,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 722,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` works in two steps. First, `entries_in_month` keeps entries whose `day.year` and `day.month` match the given numbers exactly. Second, the function builds a dict from that subset with three keys:\n\n- `count` is `len` of the subset.\n- `total` comes from `total()`, which is plain `sum()` over `amount`.\n- `by_category` comes from `sum_by_category()`, which loops over entries and adds each amount into a dict keyed by category, starting each key at `0.0`.\n\nA few details worth knowing:\n\n- `add_entry` lowercases categories, so `\"Food\"` and `\"food\"` land in the same bucket. `filter_by_category` also lowercases its argument.\n- Amounts are stored as `float` and never rounded, so totals can show normal float drift (for example `0.1 + 0.2` gives `0.30000000000000004`).\n- For an empty month, `total` is the int `0`, not `0.0`, because `sum()` starts from `0`. `by_category` is `{}`.\n- `year` and `month` are not checked. `month=13` gives no error, only an empty report.\n\n**Invalid date strings.** `parse_date` splits the text on `\"-\"`. If it gets anything other than exactly 3 parts, or any part fails `str.isdigit()`, it raises `ValueError(\"bad date: '...'\")`. That check rejects empty parts (`\"2024--01\"`), leading or trailing whitespace, a trailing newline, and slash formats. Strings that pass that check still go to `datetime.date(y, m, d)`, which raises its own `ValueError` for impossible dates:\n\n- `\"2024-02-30\"` gives `day is out of range for month`.\n- `\"2024-13-01\"` gives `month must be in 1..12`.\n- `\"0000-01-01\"` gives a year-out-of-range error.\n\nSo every bad string raises `ValueError`, but the message format depends on which check failed. Some edge cases:\n\n- Unpadded forms like `\"2024-1-5\"` are accepted.\n- Unicode digits such as `\"\u00b2\"` pass `isdigit()` and then fail inside `int()`, with a different message again.\n- Non-string input such as `None` or a `date` object raises `AttributeError` on `.split`, not `ValueError`.\n\n**Effect on `add_entry`.** Nothing catches the exception, so it goes straight up to the caller. The amount check runs first, so a bad amount is reported before a bad date. `parse_date` runs while the `Entry` is being built, before `append`, so a failed call leaves the list unchanged. Because of this, invalid dates never reach the monthly totals. The only way around it is to build an `Entry` by hand without `add_entry`.\n\nI did not change any files.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.283 (Claude Code)",
   "api_cost": 0.356713,
   "light": true
  },
  {
   "run": "claude-fable-5-1__caveman__T3__r2__4ce4b0",
   "model": "claude-fable-5-1",
   "host": "claude",
   "task": "T3",
   "skill": "caveman",
   "rep": 2,
   "round": 8,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 1327,
   "input": 4,
   "cache_read": 20236,
   "cache_write": 22683,
   "warmup": null,
   "cache_write_1h": 22683,
   "cost_usd": 0.525109,
   "wall_s": 19.4,
   "turns": 2,
   "prompt_last": 22887,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 593,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report` in `ledger/core.py` builds the report in two steps. First, `entries_in_month` keeps only entries whose date has the requested year and month. Then the report counts those entries, sums their amounts with `total`, and groups them per category with `sum_by_category`. That grouping loops over the entries and adds each amount to a running float for its category, starting from 0.0. Categories are already lowercase because `add_entry` lowercases them on insert. Amounts are plain floats, so the sums can show normal floating-point rounding. Nothing checks the year and month arguments. A month like 13 simply matches nothing. For an empty month, count is 0, the total is the integer 0 rather than 0.0, and the category map is empty.\n\n**Invalid date strings.** `parse_date` splits the text on hyphens. It needs exactly three parts, and every part must pass `str.isdigit`. If not, it raises a ValueError with a message like `bad date: '2024/01/05'`. Strings that pass this check still go through Python's `date` constructor. That constructor raises its own ValueError for impossible dates such as month 13, February 30, or year 0, with messages like `month must be in 1..12`. The check is looser than the docstring's YYYY-MM-DD claims. Unpadded input like `2024-1-5` is accepted. Unicode digits such as superscripts pass `isdigit` but make `int` fail, which is still a ValueError. A non-string argument raises AttributeError instead of ValueError.\n\nNothing in the module catches these errors, so they reach the caller of `add_entry`. `add_entry` checks the amount before it parses the date. A bad amount therefore gets reported first, and the date is never checked. The date is parsed before the append happens, so a bad date leaves the entries list unchanged.\n\nI changed no files.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.283 (Claude Code)",
   "api_cost": 0.525109,
   "light": true
  },
  {
   "run": "claude-fable-5-1__control__T3__r1__03e5c3",
   "model": "claude-fable-5-1",
   "host": "claude",
   "task": "T3",
   "skill": "control",
   "rep": 1,
   "round": 8,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 1859,
   "input": 4,
   "cache_read": 36679,
   "cache_write": 18148,
   "warmup": null,
   "cache_write_1h": 18148,
   "cost_usd": 0.46511975,
   "wall_s": 29.3,
   "turns": 2,
   "prompt_last": 28268,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 567,
   "commands": null,
   "answer": "## How monthly totals are computed\n\n`monthly_report(entries, year, month)` (`ledger/core.py:51`) works in two steps. First it filters the entries, then it summarizes the ones that are left.\n\n**Filtering.** `entries_in_month` keeps an entry only if its `day` has the same year and month as the arguments you pass. The comparison uses plain `==` and nothing is checked first. So a month like `13`, or a year passed as the string `\"2024\"` instead of the integer `2024`, doesn't raise an error. It just matches nothing, and you get an empty report back.\n\n**Summarizing.** The report is a dictionary with three keys:\n- `count` is the number of matching entries.\n- `total` comes from `total()`, which adds up the `amount` fields. Amounts are stored as floats, so the usual floating-point rounding can show up (for example, `0.1 + 0.2` gives `0.30000000000000004`). For a month with no entries, `sum()` of an empty sequence returns the integer `0`, not `0.0`.\n- `by_category` comes from `sum_by_category()`, which builds a dictionary from each category to its running total, starting each one at `0.0`.\n\nCategories are grouped by exact string match. `add_entry` lowercases the category when it stores an entry, so \"Food\" and \"food\" end up in the same group. But if you create `Entry` objects directly instead of going through `add_entry`, nothing is lowercased, and those would count as two different categories.\n\n## What happens with an invalid date string\n\nDates are only parsed in `add_entry`, which calls `parse_date` (`ledger/core.py:15`). A bad date raises a `ValueError` in one of two places:\n\n1. **Format check.** The string is split on `-`. If that doesn't give exactly three parts, or any part isn't made of digits, `parse_date` raises `ValueError(\"bad date: '...'\")`. This catches things like `\"2024/01/05\"`, `\"2024-01\"`, `\"2024-01-\"` (an empty part fails `isdigit()`), `\"2024-Jan-05\"`, and strings with extra spaces around them.\n2. **Calendar check.** If the format check passes, the three numbers go to `datetime.date(y, m, d)`. That raises its own `ValueError` for dates that don't exist, like `\"2024-02-30\"`, `\"2024-13-01\"`, or year `0`. Its message is Python's (e.g. \"day is out of range for month\"), not the module's \"bad date\" message.\n\nSome other behaviors to know about:\n- **Loose format.** The check only requires digits, not the exact `YYYY-MM-DD` widths. So `\"2024-1-5\"` is accepted, and `\"24-01-05\"` is read as the year 24 AD.\n- **Odd digit characters.** `str.isdigit()` returns true for some Unicode characters, like superscript `\"\u00b2\"`, that `int()` can't convert. Those fail at the `int()` call, which also raises `ValueError`.\n- **Non-string input.** Passing something that isn't a string, such as a `date` object, raises `AttributeError` (there's no `.split`), not `ValueError`.\n- **Check order.** `add_entry` checks the amount first. If both the amount and the date are bad, you get the \"amount must be positive\" error.\n- **The list stays unchanged.** The date is parsed before `append` runs, so a failed parse leaves `entries` exactly as it was. The error isn't caught anywhere in the module, so it goes straight to whoever called `add_entry`.\n\nOnce an entry is in the list, `day` is always a real `date`, so the monthly functions never have to deal with bad dates.\n\nI didn't change any files.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.283 (Claude Code)",
   "api_cost": 0.46512
  },
  {
   "run": "claude-fable-5-1__control__T3__r2__6b0c61",
   "model": "claude-fable-5-1",
   "host": "claude",
   "task": "T3",
   "skill": "control",
   "rep": 2,
   "round": 8,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 1503,
   "input": 4,
   "cache_read": 20236,
   "cache_write": 17390,
   "warmup": null,
   "cache_write_1h": 17390,
   "cost_usd": 0.428049,
   "wall_s": 21.6,
   "turns": 2,
   "prompt_last": 20245,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 527,
   "commands": null,
   "answer": "The monthly report works in two steps: it first selects the entries for one month, then summarizes only that subset. Invalid dates are rejected when an entry is added, so they never reach the monthly calculation. I did not change any files.\n\n**How monthly totals are computed.** The report function starts by filtering the full list of entries. It keeps an entry only if both its year and its month match the requested year and month. It then builds a result with three pieces of information about that subset. The first is how many entries there are. The second is the grand total, which adds up every entry's amount. The third is a breakdown by category. That breakdown walks through the entries and keeps a running sum for each category name, starting each new category at zero.\n\nA few details affect the results:\n\n- **Categories are case-folded.** Names are lowercased when an entry is added, so \"Food\" and \"food\" end up in the same bucket.\n- **Empty months are handled quietly.** A month with no entries reports a count of zero, a total of zero and an empty breakdown.\n- **Year and month are not validated.** Asking for month 13 raises no error. It simply matches nothing and returns an empty report.\n- **Amounts are floats.** Totals can show small rounding artifacts, such as a sum that comes out a hair off from the expected cents.\n\n**What happens with an invalid date string.** Date strings are parsed only when an entry is added, and the function that does it is `parse_date`. It splits the text on hyphens and requires exactly three pieces, each made only of digits. If either check fails, it raises a ValueError with a message saying the date is bad and quoting the input. If the pieces pass, they become integers and go to Python's date constructor. That constructor raises its own ValueError for impossible values like month 13, February 30 or year zero.\n\nThe error is not caught anywhere, so it passes straight through to whoever called the add function. The entry is never appended, because the date is parsed while the entry is being built, before the append happens. The list stays unchanged. There is one ordering detail. The amount is checked before the date, so a non-positive amount combined with a bad date reports the amount error.\n\nThe parser is looser than its docstring suggests:\n\n- **Short parts are accepted.** Inputs like \"2024-1-5\" or \"24-01-05\" pass, and the second one becomes the year 24.\n- **Some Unicode characters slip past the digit check.** Characters like superscript \"\u00b2\" count as digits, but converting them to integers fails. The result is still a ValueError, just with Python's generic message instead of the \"bad date\" one.\n- **Non-string input fails differently.** Passing something like None raises an AttributeError rather than a ValueError.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.283 (Claude Code)",
   "api_cost": 0.428049,
   "light": true
  },
  {
   "run": "claude-opus-5-5__control__T3__r1__60fe1a",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "control",
   "rep": 1,
   "round": 8,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 1679,
   "input": 4,
   "cache_read": 25776,
   "cache_write": 10284,
   "warmup": null,
   "cache_write_1h": 10284,
   "cost_usd": 0.1210232,
   "wall_s": 17.7,
   "turns": 2,
   "prompt_last": 20404,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 510,
   "commands": null,
   "answer": "I didn't change any files. Everything below comes from reading `ledger/core.py`.\n\n## How monthly totals are computed\n\n`monthly_report(entries, year, month)` (`ledger/core.py:51`) works in three steps:\n\n1. **Pick the month's entries.** `entries_in_month` keeps every entry whose date has the requested year and month. Entries from other months are left out, and so are entries from the same month in a different year. The `year` and `month` values aren't checked, so asking for month 13 doesn't raise an error. It just finds nothing and returns an empty report.\n\n2. **Add up the amounts.** `total` sums the `amount` of every entry that was kept. Amounts were already converted to floats when they were added, so this is plain float addition. That means small rounding errors can show up (for example, 0.1 + 0.2 comes out as 0.30000000000000004). There's one small quirk: if no entries match, `sum` of an empty list returns the integer `0`, not `0.0`.\n\n3. **Break it down by category.** `sum_by_category` goes through the same entries and keeps a running total for each category, starting each one at `0.0`. Categories were lowercased when the entries were added, so \"Food\" and \"food\" end up in the same bucket.\n\nThe report is a dictionary with three keys: `count` (how many entries matched), `total`, and `by_category`. The `total` and `by_category` numbers are calculated separately from the same entries, so they should agree apart from float rounding.\n\n## What happens when a date string is invalid\n\nDates are checked in `parse_date` (`ledger/core.py:15`), and both kinds of check raise `ValueError`:\n\n- **Wrong format.** The string is split on `-`. If it doesn't give exactly three pieces, or any piece isn't all digits, you get `ValueError(\"bad date: '...'\")`. This catches text like `\"2024/01/05\"`, `\"2024-01\"`, `\"2024--05\"` (empty piece), `\" 2024-01-05\"` (leading space) and `\"-2024-01-05\"` (four pieces).\n- **Correct format but not a real date.** Something like `\"2024-02-30\"` or `\"2024-13-01\"` passes the format check. It then fails when Python's `date(y, m, d)` is built, which raises its own `ValueError` (e.g. \"day is out of range for month\"). So the error message depends on which check failed.\n\nSome details that go beyond the docstring:\n\n- **Zero-padding isn't required.** `\"2024-1-5\"` is accepted, even though the docstring says `YYYY-MM-DD`.\n- **Non-strings fail differently.** Passing something that isn't a string, such as `None`, raises `AttributeError` (from `.split`) rather than `ValueError`.\n\n**Effect on the ledger:** in `add_entry` (`ledger/core.py:24`), the amount is checked first, so a bad amount is reported before the date is even looked at. `parse_date` then runs while the new `Entry` is being built, before anything is appended. If the date is invalid, the error goes straight back to the caller and the `entries` list is left untouched. A bad entry is never partly added.",
   "account_remark": true,
   "effort": null,
   "cli": "2.1.283 (Claude Code)",
   "api_cost": 0.121023,
   "light": true
  },
  {
   "run": "claude-sonnet-5__control__T3__r1__d03dc0",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "control",
   "rep": 1,
   "round": 8,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 981,
   "input": 6,
   "cache_read": 75606,
   "cache_write": 11357,
   "warmup": null,
   "cache_write_1h": 11357,
   "cost_usd": 0.0703712,
   "wall_s": 14.8,
   "turns": 3,
   "prompt_last": 29998,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 34,
   "commands": null,
   "answer": "Here's how it works, based on `ledger/core.py`:\n\n**Monthly totals**\n\n`monthly_report(entries, year, month)` (line 51) is the entry point:\n\n1. It calls `entries_in_month` (line 40), which filters the full entry list down to those whose `day.year` and `day.month` match the requested `year`/`month`.\n2. It calls `total()` (line 31) on that filtered list, which just sums the `amount` field across the matching entries.\n3. It also calls `sum_by_category()` (line 44), which builds a dict mapping each entry's `category` to the running sum of `amount` for that category, using `out.get(e.category, 0.0)` to default missing keys to 0.\n4. The result is a dict: `{\"count\": <num entries>, \"total\": <sum>, \"by_category\": {...}}`.\n\nSo the monthly total is simply the sum of `amount` over all entries whose stored `date` falls in that year/month \u2014 there's no separate month-tracking field; filtering happens against the real `date` object each time.\n\n**Invalid date strings**\n\nDates only enter the ledger through `parse_date()` (line 15), which is called from `add_entry()` (line 24) when constructing an `Entry`. `parse_date` splits the input on `-` and requires exactly 3 parts, all of which must be purely digit strings (`str.isdigit()`). If that check fails \u2014 e.g. wrong number of segments, non-numeric parts, or a format like `\"09/28/2026\"` \u2014 it raises `ValueError(f\"bad date: {text!r}\")` immediately, before ever constructing a `date` object.\n\nIf the check passes but the values are out of range (e.g. `\"2026-13-01\"` or `\"2026-02-30\"`), `parse_date` doesn't catch that itself \u2014 it lets `date(y, m, d)` on line 21 raise its own `ValueError` (from the standard library) instead.\n\nEither way, `add_entry` doesn't catch the exception \u2014 it propagates up to the caller, so an invalid date string means the entry is never added to the ledger and the exception must be handled (or will crash the program) at the call site. Since entries only get into the list via `add_entry`, this also means every entry that reaches `monthly_report`/`entries_in_month` is guaranteed to have a valid `date`.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.283 (Claude Code)",
   "api_cost": 0.070371,
   "light": true
  },
  {
   "run": "gpt-5.6-luna__caveman__T3__r1__a1131c",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "caveman",
   "rep": 1,
   "round": 8,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 264,
   "input": 27504,
   "cache_read": 19968,
   "cache_write": 17931,
   "warmup": 10395,
   "cache_write_1h": null,
   "cost_usd": 0.004302,
   "wall_s": 13.2,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 26,
   "commands": 1,
   "answer": "Monthly totals:\n\n- `entries_in_month` keeps entries whose `day.year == year` and `day.month == month`.\n- `monthly_report` counts those entries.\n- `total` sums their `amount` values.\n- `sum_by_category` adds amounts separately for each category.\n- Report returns `count`, `total`, and `by_category`.\n\nInvalid date strings:\n\n- `parse_date` requires three numeric parts separated by `-`, such as `YYYY-MM-DD`.\n- Wrong format raises `ValueError(\"bad date: ...\")`.\n- Invalid calendar dates, such as month `13`, also raise `ValueError` from `date`.\n- `add_entry` does not add entry when parsing fails; exception propagates.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.157.1",
   "api_cost": 0.004302
  },
  {
   "run": "gpt-5.6-luna__caveman__T3__r2__4e6e32",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "caveman",
   "rep": 2,
   "round": 8,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 268,
   "input": 27509,
   "cache_read": 23040,
   "cache_write": 14864,
   "warmup": 10395,
   "cache_write_1h": null,
   "cost_usd": 0.003755,
   "wall_s": 9.7,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 87,
   "commands": 1,
   "answer": "Monthly totals:\n\n- `entries_in_month` keeps entries matching given year and month.\n- `total` sums their `amount` values.\n- `monthly_report` returns entry count, total amount, and category subtotals.\n\nInvalid date strings cause `parse_date` to raise `ValueError`. This happens for wrong format, non-numeric parts, or invalid calendar dates. `add_entry` propagates error and does not add entry.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.157.1",
   "api_cost": 0.003755
  },
  {
   "run": "gpt-5.6-luna__control__T3__r1__3c2d48",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "control",
   "rep": 1,
   "round": 8,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 287,
   "input": 24126,
   "cache_read": 19968,
   "cache_write": 14553,
   "warmup": 10395,
   "cache_write_1h": null,
   "cost_usd": 0.003654,
   "wall_s": 11.2,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 75,
   "commands": 1,
   "answer": "Monthly totals are computed by selecting entries whose `day.year` and `day.month` match the requested year and month, then summing their `amount` values. `monthly_report` also returns the matching entry count and category subtotals.\n\nA date string must have three numeric parts in `YYYY-MM-DD` form. Invalid formatting or impossible calendar dates cause `parse_date` to raise `ValueError`. Through `add_entry`, that error propagates and the invalid entry is not added.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.157.1",
   "api_cost": 0.003654
  },
  {
   "run": "gpt-5.6-luna__control__T3__r2__2c7c4d",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "control",
   "rep": 2,
   "round": 8,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 352,
   "input": 24131,
   "cache_read": 19968,
   "cache_write": 14558,
   "warmup": 10395,
   "cache_write_1h": null,
   "cost_usd": 0.003733,
   "wall_s": 12.6,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 104,
   "commands": 1,
   "answer": "Monthly totals are computed by:\n\n1. Filtering entries whose `day.year` and `day.month` match the requested year and month.\n2. Counting those entries.\n3. Summing their `amount` values.\n4. Grouping the same entries by lowercase category and summing each category\u2019s amounts.\n\n`monthly_report` returns all three values: `count`, `total`, and `by_category`.\n\nInvalid date strings cause `parse_date` to raise `ValueError`. This includes incorrectly formatted strings, non-numeric components, and impossible calendar dates such as `2024-02-30`. When used through `add_entry`, the invalid entry is not appended.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.157.1",
   "api_cost": 0.003733
  },
  {
   "run": "gpt-5.6-sol__caveman__T3__r1__72ab4d",
   "model": "gpt-5.6-sol",
   "host": "codex",
   "task": "T3",
   "skill": "caveman",
   "rep": 1,
   "round": 8,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 183,
   "input": 30539,
   "cache_read": 25984,
   "cache_write": 15927,
   "warmup": 11372,
   "cache_write_1h": null,
   "cost_usd": 0.077762,
   "wall_s": 10.2,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "Monthly totals use entries whose `day.year` and `day.month` match requested values. Report returns:\n\n- `count`: matching entry count\n- `total`: sum of matching amounts\n- `by_category`: amount sums grouped by category\n\nInvalid date strings raise `ValueError`. Wrong format or non-digits produce `bad date: '...'`. Impossible dates, such as `2026-02-30`, produce Python `date()` validation error. Entry is not appended.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.157.1",
   "api_cost": 0.077762
  },
  {
   "run": "gpt-5.6-sol__caveman__T3__r2__bf0e51",
   "model": "gpt-5.6-sol",
   "host": "codex",
   "task": "T3",
   "skill": "caveman",
   "rep": 2,
   "round": 8,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 202,
   "input": 30539,
   "cache_read": 25984,
   "cache_write": 15927,
   "warmup": 11372,
   "cache_write_1h": null,
   "cost_usd": 0.078142,
   "wall_s": 8.7,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "`monthly_report` selects entries whose `day.year` and `day.month` match inputs. It returns:\n\n- `count`: matching entry count\n- `total`: sum of matching amounts\n- `by_category`: amount totals grouped by category\n\n`parse_date` expects `YYYY-MM-DD`. Wrong structure or non-digits raise `ValueError(\"bad date: ...\")`. Impossible dates, such as `2026-02-30`, raise `ValueError` from `datetime.date`. `add_entry` propagates either error, so invalid entry is not appended.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.157.1",
   "api_cost": 0.078142
  },
  {
   "run": "gpt-5.6-sol__control__T3__r1__c80627",
   "model": "gpt-5.6-sol",
   "host": "codex",
   "task": "T3",
   "skill": "control",
   "rep": 1,
   "round": 8,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 374,
   "input": 27001,
   "cache_read": 24064,
   "cache_write": 14309,
   "warmup": 11372,
   "cache_write_1h": null,
   "cost_usd": 0.074342,
   "wall_s": 12.2,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 135,
   "commands": 1,
   "answer": "`monthly_report(entries, year, month)`:\n\n- Filters entries whose `day.year` and `day.month` exactly match.\n- Returns their count.\n- Sums their `amount` values for the monthly total.\n- Also sums amounts separately by normalized category.\n\nFor invalid date strings, `parse_date()` raises `ValueError`. Incorrect structure or non-numeric components produce `ValueError(\"bad date: ...\")`; impossible calendar dates such as `2026-02-30` are rejected by Python\u2019s `date()` constructor. In `add_entry()`, the exception propagates and no entry is appended.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.157.1",
   "api_cost": 0.074342
  },
  {
   "run": "gpt-5.6-sol__control__T3__r2__d5bb20",
   "model": "gpt-5.6-sol",
   "host": "codex",
   "task": "T3",
   "skill": "control",
   "rep": 2,
   "round": 8,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 394,
   "input": 27173,
   "cache_read": 24320,
   "cache_write": 14225,
   "warmup": 11372,
   "cache_write_1h": null,
   "cost_usd": 0.074508,
   "wall_s": 12.6,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 131,
   "commands": 1,
   "answer": "`monthly_report(entries, year, month)`:\n\n1. Filters entries whose `day.year` and `day.month` match.\n2. Counts those entries.\n3. Sums their `amount` values.\n4. Separately totals amounts by category.\n\nIt returns:\n\n```python\n{\"count\": ..., \"total\": ..., \"by_category\": {...}}\n```\n\nFor date strings, `parse_date` expects three hyphen-separated numeric parts. Malformed strings raise `ValueError(\"bad date: ...\")`. Numeric but impossible dates\u2014such as `2026-02-30`\u2014raise the standard `datetime.date` `ValueError`. In `add_entry`, parsing happens before appending, so an invalid date leaves the entries list unchanged.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.157.1",
   "api_cost": 0.074508
  },
  {
   "run": "gpt-6-astra__caveman__T3__r1__04d147",
   "model": "gpt-6-astra",
   "host": "codex",
   "task": "T3",
   "skill": "caveman",
   "rep": 1,
   "round": 8,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 188,
   "input": 32323,
   "cache_read": 27904,
   "cache_write": 16790,
   "warmup": 12371,
   "cache_write_1h": null,
   "cost_usd": 0.205204,
   "wall_s": 14.1,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "`monthly_report` selects entries matching requested year and month. Returns entry count, sum of amounts in dollars, and totals per category. Empty month returns count `0`, total `0`, and `{}` for categories.\n\n`parse_date` requires three hyphen-separated digit groups, converts them to integers, then calls `datetime.date`. Wrong structure raises `ValueError(\"bad date: ...\")`; impossible dates raise `ValueError` from `datetime.date`. Fixed digit widths are not enforced, so `2026-2-3` is accepted.\n\nInvalid dates passed to `add_entry` propagate that error. No entry is appended. Files unchanged.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.157.1",
   "api_cost": 0.205204
  },
  {
   "run": "gpt-6-astra__caveman__T3__r2__5ed1a4",
   "model": "gpt-6-astra",
   "host": "codex",
   "task": "T3",
   "skill": "caveman",
   "rep": 2,
   "round": 8,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 181,
   "input": 32334,
   "cache_read": 27904,
   "cache_write": 16801,
   "warmup": 12371,
   "cache_write_1h": null,
   "cost_usd": 0.204964,
   "wall_s": 16.2,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "`monthly_report` selects entries matching both year and month. Returns entry count, sum of dollar amounts, and sums grouped by category. Empty month returns count `0`, total `0`, and `{}` for categories.\n\n`parse_date` requires three hyphen-separated numeric parts, then constructs Python `date`. Malformed strings raise `ValueError(\"bad date: ...\")`; impossible dates raise `ValueError` from `date`. Component widths aren\u2019t enforced: `2026-2-3` works.\n\nInvalid date passed to `add_entry` propagates error; no entry gets appended. No files changed.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.157.1",
   "api_cost": 0.204964
  },
  {
   "run": "gpt-6-astra__control__T3__r1__7221c0",
   "model": "gpt-6-astra",
   "host": "codex",
   "task": "T3",
   "skill": "control",
   "rep": 1,
   "round": 8,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 250,
   "input": 28942,
   "cache_read": 24320,
   "cache_write": 16993,
   "warmup": 12371,
   "cache_write_1h": null,
   "cost_usd": 0.20675,
   "wall_s": 13.7,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "In [ledger/core.py](/tmp/bench-work/gpt-6-astra__control__T3__r1__7221c0/ledger/core.py), `monthly_report` selects entries whose date matches the requested year and month. It returns their count, the sum of their amounts, and totals grouped by category. If no entries match, these are `0`, `0`, and `{}`.\n\n`parse_date` splits a string on hyphens and requires three all-digit parts. Otherwise, it raises `ValueError` with `bad date: <string representation>`. It converts those parts to year, month, and day, then calls `datetime.date`, which also raises `ValueError` for impossible dates. Despite the documented `YYYY-MM-DD` format, fixed digit widths are not enforced.\n\nWhen `add_entry` encounters an invalid date, the exception propagates and no entry is appended. No files were changed.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.157.1",
   "api_cost": 0.20675
  },
  {
   "run": "gpt-6-astra__control__T3__r2__dfac44",
   "model": "gpt-6-astra",
   "host": "codex",
   "task": "T3",
   "skill": "control",
   "rep": 2,
   "round": 8,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 302,
   "input": 43404,
   "cache_read": 40448,
   "cache_write": 15327,
   "warmup": 12371,
   "cache_write_1h": null,
   "cost_usd": 0.208818,
   "wall_s": 15.2,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 2,
   "answer": "In [ledger/core.py](/tmp/bench-work/gpt-6-astra__control__T3__r2__dfac44/ledger/core.py), `monthly_report` selects entries whose dates match the requested year and month. It returns their count, the sum of their amounts, and totals grouped by category. If no entries match, it returns a count and total of `0` and an empty category dictionary.\n\n`parse_date` splits a date string on hyphens and requires three digit-only parts. Malformed strings raise `ValueError` with a `bad date` message. Otherwise, it converts the parts to integers and calls Python\u2019s `date` constructor, which raises `ValueError` for impossible dates, such as February 30. It does not enforce exact `YYYY-MM-DD` widths, so `2026-9-2` is accepted. When an invalid date reaches `add_entry`, the exception propagates and no entry is appended.\n\nNo files were changed.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.157.1",
   "api_cost": 0.208818
  },
  {
   "run": "gpt-6-luna__caveman__T3__r1__34effd",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "caveman",
   "rep": 1,
   "round": 8,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 159,
   "input": 31016,
   "cache_read": 26112,
   "cache_write": 16647,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.002005,
   "wall_s": 7.8,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "`monthly_report` keeps entries whose `day` has the requested year and month. It returns their count, summed amount, and totals grouped by category.\n\n`parse_date` splits the date string into three numeric parts, then builds a calendar date. If the format or date is invalid, it raises `ValueError`. When `add_entry` receives that string, the error stops the entry from being added.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.157.1",
   "api_cost": 0.002005
  },
  {
   "run": "gpt-6-luna__caveman__T3__r2__5bb112",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "caveman",
   "rep": 2,
   "round": 8,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 129,
   "input": 30955,
   "cache_read": 26112,
   "cache_write": 16586,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.001984,
   "wall_s": 7.4,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "`monthly_report` filters entries to those whose date year and month match the requested values. It returns their count, summed amount, and totals grouped by category.\n\n`parse_date` raises `ValueError` if the string is not three numeric parts or does not form a valid calendar date. Since `add_entry` parses the date before appending, an invalid date raises an error and adds no entry.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.157.1",
   "api_cost": 0.001984
  },
  {
   "run": "gpt-6-luna__control__T3__r1__e2e798",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "control",
   "rep": 1,
   "round": 8,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 197,
   "input": 27598,
   "cache_read": 24064,
   "cache_write": 15277,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.001867,
   "wall_s": 9.6,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "`monthly_report(entries, year, month)` first selects entries whose date has the requested year and month. It returns their count, the sum of their amounts, and a per-category sum. Each category total is accumulated by adding each matching entry\u2019s amount to that category\u2019s running total.\n\nWhen adding an entry, `parse_date` expects three hyphen-separated numeric parts in `YYYY-MM-DD` form, then uses Python\u2019s `date` constructor to validate the calendar date. If the format or date is invalid, it raises `ValueError`; `add_entry` does not catch it, so no entry is appended.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.157.1",
   "api_cost": 0.001867
  },
  {
   "run": "gpt-6-luna__control__T3__r2__3eb741",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "control",
   "rep": 2,
   "round": 8,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 164,
   "input": 27594,
   "cache_read": 24064,
   "cache_write": 15273,
   "warmup": 11743,
   "cache_write_1h": null,
   "cost_usd": 0.00185,
   "wall_s": 9.0,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "`monthly_report(entries, year, month)` first keeps entries whose date has the requested year and month. It then returns their count, the sum of their amounts, and a per-category sum of those amounts.\n\n`add_entry` parses a date string with `parse_date`. A malformed string or an impossible calendar date raises `ValueError`; the entry is not appended. The parser expects three hyphen-separated numeric parts in `YYYY-MM-DD` form.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.157.1",
   "api_cost": 0.00185
  },
  {
   "run": "gpt-6-sol__caveman__T3__r1__00d16a",
   "model": "gpt-6-sol",
   "host": "codex",
   "task": "T3",
   "skill": "caveman",
   "rep": 1,
   "round": 8,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 168,
   "input": 31336,
   "cache_read": 27008,
   "cache_write": 16250,
   "warmup": 11922,
   "cache_write_1h": null,
   "cost_usd": 0.039582,
   "wall_s": 10.3,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "`monthly_report` selects entries whose dates match given year and month. It returns entry count, sum of their amounts, and totals grouped by category. No matches produce count `0`, total `0`, and empty category totals.\n\n`parse_date` requires three numeric parts separated by hyphens. Invalid format raises `ValueError(\"bad date: ...\")`; invalid calendar dates also raise `ValueError`. `add_entry` propagates error, so it does not append entry.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.157.1",
   "api_cost": 0.039582
  },
  {
   "run": "gpt-6-sol__caveman__T3__r2__4aadd8",
   "model": "gpt-6-sol",
   "host": "codex",
   "task": "T3",
   "skill": "caveman",
   "rep": 2,
   "round": 8,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 144,
   "input": 31334,
   "cache_read": 27008,
   "cache_write": 16248,
   "warmup": 11922,
   "cache_write_1h": null,
   "cost_usd": 0.039338,
   "wall_s": 10.9,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "`monthly_report` selects entries whose dates match given year and month. It returns entry count, sum of their dollar amounts, and totals grouped by category. No matching entries yield count `0`, total `0`, and empty category totals.\n\n`add_entry` passes date string to `parse_date`. Invalid format or impossible date raises `ValueError`; entry is not added.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.157.1",
   "api_cost": 0.039338
  },
  {
   "run": "gpt-6-sol__control__T3__r1__d18a7a",
   "model": "gpt-6-sol",
   "host": "codex",
   "task": "T3",
   "skill": "control",
   "rep": 1,
   "round": 8,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 392,
   "input": 27998,
   "cache_read": 25344,
   "cache_write": 14576,
   "warmup": 11922,
   "cache_write_1h": null,
   "cost_usd": 0.038141,
   "wall_s": 15.3,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 118,
   "commands": 1,
   "answer": "`monthly_report(entries, year, month)` selects entries whose date has the requested year and month. It returns their count, a total formed by summing their amounts, and separate sums for each category. The code does not round the sums.\n\nFor an invalid date string, `parse_date` raises `ValueError`: it rejects strings without exactly three numeric parts, and Python\u2019s `date` constructor rejects impossible dates such as February 30. When called through `add_entry`, that error propagates and no entry is appended. The parser\u2019s docstring says `YYYY-MM-DD`, but the code does not enforce four-digit years or two-digit months and days. See [ledger/core.py](/tmp/bench-work/gpt-6-sol__control__T3__r1__d18a7a/ledger/core.py).",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.157.1",
   "api_cost": 0.038141
  },
  {
   "run": "gpt-6-sol__control__T3__r2__0ddfb4",
   "model": "gpt-6-sol",
   "host": "codex",
   "task": "T3",
   "skill": "control",
   "rep": 2,
   "round": 8,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 249,
   "input": 27977,
   "cache_read": 25344,
   "cache_write": 14555,
   "warmup": 11922,
   "cache_write_1h": null,
   "cost_usd": 0.036669,
   "wall_s": 13.1,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 67,
   "commands": 1,
   "answer": "`monthly_report` selects entries whose stored date has the requested year and month. It returns their count, the sum of their amounts, and a breakdown of amounts by category. If there are no matching entries, the total is `0` and the category breakdown is empty.\n\nWhen `add_entry` receives a date string, `parse_date` splits it on hyphens and converts the three numeric parts into a date. An invalid string raises `ValueError`, either from the format check or from the date constructor for an impossible date. The entry is not appended.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.157.1",
   "api_cost": 0.036669
  }
 ],
 "round9_runs": [
  {
   "run": "claude-opus-5-5__control__T3__r1__f40cec",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "control",
   "rep": 1,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 1628,
   "input": 4,
   "cache_read": 25687,
   "cache_write": 8211,
   "warmup": null,
   "cache_write_1h": 8211,
   "cost_usd": 0.1034014,
   "wall_s": 16.4,
   "turns": 2,
   "prompt_last": 18331,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 478,
   "commands": null,
   "answer": "I only read `ledger/core.py` and didn't change anything.\n\n## How monthly totals are computed\n\n`monthly_report(entries, year, month)` (`ledger/core.py:51`) does it in three steps:\n\n1. **Filter to the month.** `entries_in_month` keeps entries whose `day.year` and `day.month` match the `year` and `month` you pass in. This works because every entry's `day` is a real `datetime.date`, created when the entry was added. The `year` and `month` arguments themselves aren't checked, so `month=13` just returns an empty list instead of raising an error.\n\n2. **Compute the total.** `total` adds up the `amount` of each matching entry with Python's `sum`. Amounts were converted to `float` when added, so the total is a float. Because `sum` starts at the integer `0`, a month with no entries gives `0` (an int), not `0.0`. Normal float rounding also applies, so something like `0.1 + 0.2` gives `0.30000000000000004`. Nothing rounds to cents.\n\n3. **Break it down by category.** `sum_by_category` builds a dictionary from each category to its summed amounts, starting each at `0.0`. Categories were lowercased when entries were added, so `\"Food\"` and `\"food\"` count as one category.\n\nThe function returns a dict with `count` (number of entries that month), `total`, and `by_category`.\n\n## What happens when a date string is invalid\n\nDate strings are only parsed in `add_entry`, which calls `parse_date` (`ledger/core.py:15`). An invalid date raises a `ValueError` in one of two places:\n\n- **The format check in `parse_date`.** The string is split on `\"-\"`. If there aren't exactly three parts, or any part contains a non-digit, it raises `ValueError(\"bad date: '...'\")`. This catches things like `\"2024/01/05\"`, `\"Jan 5 2024\"`, `\" 2024-01-05\"` (leading space) and `\"\"`.\n- **The `datetime.date` constructor.** Strings that pass the format check but aren't real dates, like `\"2024-13-01\"`, `\"2023-02-29\"` or `\"0-1-1\"`, get past the check. Then `date(y, m, d)` raises its own `ValueError` (for example, \"month must be in 1..12\"). This message is different and doesn't include the original string.\n\nSome other details:\n\n- **No partial writes.** In `add_entry`, the date is parsed while the `Entry` is being built, before `append`. If parsing fails, the list is unchanged. So a monthly report never sees a bad date.\n- **The amount is checked first.** If both the amount and the date are invalid, you get the `\"amount must be positive\"` error, not the date error.\n- **The format check is looser than its docstring.** The docstring says `YYYY-MM-DD`, but zero-padding isn't enforced, so `\"2024-1-5\"` is accepted as 5 January 2024.\n- **Non-string input raises a different error.** Passing `None` or an existing `date` object raises `AttributeError` (there's no `.split`), not `ValueError`.",
   "account_remark": true,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "api_cost": 0.103401,
   "light": true
  },
  {
   "run": "claude-opus-5-5__control__T3__r2__cd6101",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "control",
   "rep": 2,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 1335,
   "input": 4,
   "cache_read": 25777,
   "cache_write": 10284,
   "warmup": null,
   "cache_write_1h": 10284,
   "cost_usd": 0.1141434,
   "wall_s": 17.7,
   "turns": 2,
   "prompt_last": 20404,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 370,
   "commands": null,
   "answer": "**Monthly totals**\n\n`monthly_report(entries, year, month)` (`ledger/core.py:51`) works in two steps. First, `entries_in_month` keeps only the entries whose `day.year` and `day.month` match the requested year and month. Then it builds a dictionary from that filtered list:\n\n- **`count`**: how many entries fell in that month.\n- **`total`**: the sum of their `amount` values, from `total()`. It's a plain `sum()` over floats, so ordinary floating-point rounding applies (e.g. `0.1 + 0.2` gives `0.30000000000000004`).\n- **`by_category`**: a dict mapping each category to the sum of its amounts, from `sum_by_category`. Categories are already lowercase because `add_entry` lowercases them, so \"Food\" and \"food\" are added together.\n\nSome edge cases:\n- A month with no entries returns `{\"count\": 0, \"total\": 0, \"by_category\": {}}`. Here `total` is the integer `0`, not `0.0`, because `sum()` of an empty list starts at `0`.\n- `year` and `month` are never checked, so asking for month 13 just returns an empty report instead of an error.\n- The report doesn't store anything or reuse earlier results. It rescans the full list on every call.\n\n**Invalid date strings**\n\nDates are checked when an entry is added, not when a report runs. `add_entry` calls `parse_date` (`ledger/core.py:15`), which splits the string on `-` and requires exactly three parts made only of digits. If that check fails, it raises `ValueError(\"bad date: '...'\")`. Strings like `\"2024/01/05\"`, `\"2024-01\"`, `\"2024--05\"` or `\" 2024-01-05\"` are rejected this way.\n\nIf the format check passes, the three numbers go to `datetime.date`. That can still raise a `ValueError`, but with Python's own message instead of \"bad date\". This happens for dates that don't exist, like `\"2024-02-30\"`, `\"2024-13-01\"` or year `0`.\n\nThe format check is looser than the docstring suggests. It doesn't require zero-padding or a four-digit year, so `\"2024-1-5\"` is accepted as January 5, 2024.\n\n`add_entry` checks the amount before it parses the date. So if both the amount and the date are bad, you get the \"amount must be positive\" error. Either way, the exception is raised before `entries.append`, so the list isn't changed. The error goes straight to the caller, because nothing in the module catches it.\n\nPassing something that isn't a string, such as `None`, fails differently: `.split` raises an `AttributeError` instead of a `ValueError`.\n\nI didn't change any files.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "api_cost": 0.114143,
   "light": true
  },
  {
   "run": "claude-sonnet-5__caveman__T3__r1__e2e4d0",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "caveman",
   "rep": 1,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 318,
   "input": 4,
   "cache_read": 59051,
   "cache_write": 22612,
   "warmup": null,
   "cache_write_1h": 22612,
   "cost_usd": 0.1054462,
   "wall_s": 5.3,
   "turns": 2,
   "prompt_last": 41253,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 17,
   "commands": null,
   "answer": "Monthly total: `monthly_report` filter entries to given year/month via `entries_in_month`, then `total()` sum `amount` field over filtered set. Also return count and per-category breakdown (`sum_by_category`).\n\nInvalid date: `parse_date` split text on `-`. If not exactly 3 parts, or any part not digit, raise `ValueError(f\"bad date: {text!r}\")`. Bad calendar value (e.g. month 13) pass digit check but `date(y, m, d)` itself raise `ValueError` from stdlib. Either way exception propagate up through `add_entry`, uncaught, no entry added.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "api_cost": 0.105446
  },
  {
   "run": "claude-sonnet-5__caveman__T3__r2__bc08b8",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "caveman",
   "rep": 2,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 302,
   "input": 4,
   "cache_read": 49172,
   "cache_write": 15933,
   "warmup": null,
   "cache_write_1h": 15933,
   "cost_usd": 0.0765944,
   "wall_s": 4.3,
   "turns": 2,
   "prompt_last": 34574,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 12,
   "commands": null,
   "answer": "Monthly total flow: `monthly_report` call `entries_in_month` filter entries by year+month match, then `total` sum amount field over those entries. `sum_by_category` also split same filtered set into per-category dict.\n\nInvalid date: `parse_date` split text on \"-\". Need exactly 3 parts, all digit. Fail either check raise `ValueError(f\"bad date: {text!r}\")`. Called from `add_entry` at insert time, not at report time. So bad date reject at entry creation, never stored. `monthly_report` itself never validate date, work only on already-parsed `date` objects in entries.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "api_cost": 0.076594,
   "light": true
  },
  {
   "run": "claude-sonnet-5__control__T3__r1__23eede",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "control",
   "rep": 1,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 807,
   "input": 6,
   "cache_read": 94399,
   "cache_write": 21069,
   "warmup": null,
   "cache_write_1h": 21069,
   "cost_usd": 0.1112378,
   "wall_s": 14.2,
   "turns": 3,
   "prompt_last": 39710,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 26,
   "commands": null,
   "answer": "Here's how it works:\n\n**Monthly totals** (`monthly_report`, line 51): For a given `year`/`month`, it first filters entries down to that month via `entries_in_month` (line 40), which keeps only entries whose `day.year` and `day.month` match. It then builds a report dict with three pieces:\n- `count` \u2014 how many entries fell in that month\n- `total` \u2014 the sum of `amount` across those entries, via `total()` (line 31), a plain `sum()` over the filtered list\n- `by_category` \u2014 a breakdown from `sum_by_category()` (line 44), which accumulates amounts into a dict keyed by category\n\nSo the \"monthly total\" is just the sum of `amount` for every entry whose date falls in that year/month \u2014 no date parsing happens at this stage, since entries already store real `date` objects by the time they're aggregated.\n\n**Invalid date strings**: Parsing happens earlier, in `parse_date` (line 15), which is only invoked when an entry is created via `add_entry` (line 27) \u2014 not during reporting. `parse_date` splits the string on `-` and requires exactly 3 numeric parts; if that check fails, it raises `ValueError(f\"bad date: {text!r}\")` immediately. If the parts are numeric but form an impossible date (e.g., month 13, or day 31 in a 30-day month), `date(y, m, d)` itself raises a `ValueError` (Python's built-in one, with its own message, not the custom \"bad date\" message). Either way, an invalid date string causes `add_entry` to raise `ValueError` and the entry is never appended to the list \u2014 so it can never reach `monthly_report`/`total_by_month`. There's no try/except anywhere in this file, so the exception propagates straight up to whatever code called `add_entry`.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "api_cost": 0.111238
  },
  {
   "run": "claude-sonnet-5__control__T3__r2__42134c",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "control",
   "rep": 2,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 876,
   "input": 6,
   "cache_read": 94370,
   "cache_write": 20151,
   "warmup": null,
   "cache_write_1h": 20151,
   "cost_usd": 0.10825,
   "wall_s": 10.3,
   "turns": 3,
   "prompt_last": 38792,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 27,
   "commands": null,
   "answer": "Here's how it works:\n\n**Monthly totals** \u2014 `monthly_report(entries, year, month)` (line 51) first calls `entries_in_month`, which filters the full entry list down to those whose `day.year` and `day.month` match the requested year/month (line 40-41). It then builds a report dict with three pieces, all derived from that filtered subset:\n- `count`: number of matching entries\n- `total`: the plain dollar sum of `e.amount` across those entries, via `total()` (line 31-32), which is just `sum(...)`\n- `by_category`: a per-category breakdown built by `sum_by_category` (line 44-48), which accumulates amounts into a dict keyed by lowercased category name\n\nSo a \"monthly total\" is simply the sum of every entry's `amount` whose `day` falls in that year/month \u2014 there's no rounding, currency handling, or sign logic beyond what's already enforced at entry time.\n\n**Invalid date strings** \u2014 Dates only enter the ledger through `add_entry`, which calls `parse_date(day)` (line 27). `parse_date` (line 15-21) splits the string on `-` and requires exactly 3 parts that are all digit strings; if that check fails, it raises `ValueError(f\"bad date: {text!r}\")` immediately (line 19). If the split parts look numeric but form an impossible date (e.g. `\"2024-02-30\"` or `\"2024-13-01\"`), the `date(y, m, d)` constructor itself raises `ValueError` (this one uncustomized, e.g. \"day is out of range for month\").\n\nEither way, an invalid date string means the entry is **never added** \u2014 `add_entry` raises before `entries.append(...)` runs, so nothing partial or malformed reaches the ledger. There's no try/except anywhere in this file, so the exception propagates straight up to the caller; `monthly_report` and friends never see bad dates because they only ever operate on entries that already passed `parse_date` successfully.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "api_cost": 0.10825
  },
  {
   "run": "claude-sonnet-5-5__caveman__T1__r1__591169",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T1",
   "skill": "caveman",
   "rep": 1,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 530,
   "input": 10,
   "cache_read": 131401,
   "cache_write": 23310,
   "warmup": null,
   "cache_write_1h": 23310,
   "cost_usd": 0.1248402,
   "wall_s": 8.5,
   "turns": 5,
   "prompt_last": 33546,
   "lines_added": 7,
   "lines_deleted": 1,
   "lines": 8,
   "files": 2,
   "thinking": 0,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.12484
  },
  {
   "run": "claude-sonnet-5-5__caveman__T1__r2__9befd4",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T1",
   "skill": "caveman",
   "rep": 2,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 510,
   "input": 10,
   "cache_read": 131057,
   "cache_write": 22383,
   "warmup": null,
   "cache_write_1h": 22383,
   "cost_usd": 0.12086340000000001,
   "wall_s": 6.7,
   "turns": 5,
   "prompt_last": 32619,
   "lines_added": 7,
   "lines_deleted": 1,
   "lines": 8,
   "files": 2,
   "thinking": 0,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.120863
  },
  {
   "run": "claude-sonnet-5-5__caveman__T1__r3__a2fcec",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T1",
   "skill": "caveman",
   "rep": 3,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 625,
   "input": 10,
   "cache_read": 131213,
   "cache_write": 22519,
   "warmup": null,
   "cache_write_1h": 22519,
   "cost_usd": 0.12258859999999999,
   "wall_s": 7.4,
   "turns": 5,
   "prompt_last": 32755,
   "lines_added": 6,
   "lines_deleted": 0,
   "lines": 6,
   "files": 1,
   "thinking": 0,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.122589
  },
  {
   "run": "claude-sonnet-5-5__control__T1__r1__9e2317",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T1",
   "skill": "control",
   "rep": 1,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 652,
   "input": 10,
   "cache_read": 114904,
   "cache_write": 17368,
   "warmup": null,
   "cache_write_1h": 17368,
   "cost_usd": 0.0989928,
   "wall_s": 8.6,
   "turns": 5,
   "prompt_last": 27604,
   "lines_added": 6,
   "lines_deleted": 0,
   "lines": 6,
   "files": 1,
   "thinking": 0,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.098993
  },
  {
   "run": "claude-sonnet-5-5__control__T1__r2__5d10cb",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T1",
   "skill": "control",
   "rep": 2,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 498,
   "input": 10,
   "cache_read": 114744,
   "cache_write": 17275,
   "warmup": null,
   "cache_write_1h": 17275,
   "cost_usd": 0.09704880000000002,
   "wall_s": 6.9,
   "turns": 5,
   "prompt_last": 27511,
   "lines_added": 6,
   "lines_deleted": 0,
   "lines": 6,
   "files": 1,
   "thinking": 0,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.097049
  },
  {
   "run": "claude-sonnet-5-5__control__T1__r3__64fa05",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T1",
   "skill": "control",
   "rep": 3,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 506,
   "input": 10,
   "cache_read": 114732,
   "cache_write": 17269,
   "warmup": null,
   "cache_write_1h": 17269,
   "cost_usd": 0.0971024,
   "wall_s": 9.8,
   "turns": 5,
   "prompt_last": 27505,
   "lines_added": 6,
   "lines_deleted": 0,
   "lines": 6,
   "files": 1,
   "thinking": 0,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.097102
  },
  {
   "run": "claude-sonnet-5-5__karpathy__T1__r1__c1d152",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T1",
   "skill": "karpathy",
   "rep": 1,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 792,
   "input": 10,
   "cache_read": 121315,
   "cache_write": 19195,
   "warmup": null,
   "cache_write_1h": 19195,
   "cost_usd": 0.10898300000000001,
   "wall_s": 8.0,
   "turns": 5,
   "prompt_last": 29431,
   "lines_added": 5,
   "lines_deleted": 0,
   "lines": 5,
   "files": 1,
   "thinking": 0,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.108983
  },
  {
   "run": "claude-sonnet-5-5__karpathy__T1__r2__143825",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T1",
   "skill": "karpathy",
   "rep": 2,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 687,
   "input": 10,
   "cache_read": 121267,
   "cache_write": 19125,
   "warmup": null,
   "cache_write_1h": 19125,
   "cost_usd": 0.1076434,
   "wall_s": 7.4,
   "turns": 5,
   "prompt_last": 29361,
   "lines_added": 5,
   "lines_deleted": 0,
   "lines": 5,
   "files": 1,
   "thinking": 0,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.107643
  },
  {
   "run": "claude-sonnet-5-5__karpathy__T1__r3__1d856a",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T1",
   "skill": "karpathy",
   "rep": 3,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 768,
   "input": 10,
   "cache_read": 121325,
   "cache_write": 19224,
   "warmup": null,
   "cache_write_1h": 19224,
   "cost_usd": 0.10886100000000001,
   "wall_s": 8.4,
   "turns": 5,
   "prompt_last": 29460,
   "lines_added": 6,
   "lines_deleted": 0,
   "lines": 6,
   "files": 1,
   "thinking": 27,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.108861
  },
  {
   "run": "claude-sonnet-5-5__placebo__T1__r1__249611",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T1",
   "skill": "placebo",
   "rep": 1,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 844,
   "input": 10,
   "cache_read": 119695,
   "cache_write": 18735,
   "warmup": null,
   "cache_write_1h": 18735,
   "cost_usd": 0.107339,
   "wall_s": 10.1,
   "turns": 5,
   "prompt_last": 28971,
   "lines_added": 7,
   "lines_deleted": 1,
   "lines": 8,
   "files": 2,
   "thinking": 0,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.107339
  },
  {
   "run": "claude-sonnet-5-5__placebo__T1__r2__0ea028",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T1",
   "skill": "placebo",
   "rep": 2,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 858,
   "input": 10,
   "cache_read": 119074,
   "cache_write": 18764,
   "warmup": null,
   "cache_write_1h": 18764,
   "cost_usd": 0.1074708,
   "wall_s": 9.2,
   "turns": 5,
   "prompt_last": 29000,
   "lines_added": 7,
   "lines_deleted": 1,
   "lines": 8,
   "files": 2,
   "thinking": 0,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.107471
  },
  {
   "run": "claude-sonnet-5-5__placebo__T1__r3__3df074",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T1",
   "skill": "placebo",
   "rep": 3,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 830,
   "input": 10,
   "cache_read": 119705,
   "cache_write": 18740,
   "warmup": null,
   "cache_write_1h": 18740,
   "cost_usd": 0.10722100000000001,
   "wall_s": 8.2,
   "turns": 5,
   "prompt_last": 28976,
   "lines_added": 7,
   "lines_deleted": 1,
   "lines": 8,
   "files": 2,
   "thinking": 0,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.107221
  },
  {
   "run": "claude-sonnet-5-5__ponytail__T1__r1__7a18ea",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T1",
   "skill": "ponytail",
   "rep": 1,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 574,
   "input": 10,
   "cache_read": 120741,
   "cache_write": 19251,
   "warmup": null,
   "cache_write_1h": 19251,
   "cost_usd": 0.1069122,
   "wall_s": 7.0,
   "turns": 5,
   "prompt_last": 29487,
   "lines_added": 6,
   "lines_deleted": 0,
   "lines": 6,
   "files": 1,
   "thinking": 0,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.106912
  },
  {
   "run": "claude-sonnet-5-5__ponytail__T1__r2__d74d20",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T1",
   "skill": "ponytail",
   "rep": 2,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 691,
   "input": 10,
   "cache_read": 121304,
   "cache_write": 19154,
   "warmup": null,
   "cache_write_1h": 19154,
   "cost_usd": 0.1078068,
   "wall_s": 9.1,
   "turns": 5,
   "prompt_last": 29390,
   "lines_added": 6,
   "lines_deleted": 0,
   "lines": 6,
   "files": 1,
   "thinking": 0,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.107807
  },
  {
   "run": "claude-sonnet-5-5__ponytail__T1__r3__5a3dc2",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T1",
   "skill": "ponytail",
   "rep": 3,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 784,
   "input": 10,
   "cache_read": 121755,
   "cache_write": 19621,
   "warmup": null,
   "cache_write_1h": 19621,
   "cost_usd": 0.110695,
   "wall_s": 10.8,
   "turns": 5,
   "prompt_last": 29857,
   "lines_added": 6,
   "lines_deleted": 0,
   "lines": 6,
   "files": 1,
   "thinking": 0,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.110695
  },
  {
   "run": "claude-sonnet-5-5__caveman__T2__r1__8b7509",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T2",
   "skill": "caveman",
   "rep": 1,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 616,
   "input": 8,
   "cache_read": 71163,
   "cache_write": 14120,
   "warmup": null,
   "cache_write_1h": 14120,
   "cost_usd": 0.0768886,
   "wall_s": 6.0,
   "turns": 4,
   "prompt_last": 24356,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 106,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.076889,
   "light": true
  },
  {
   "run": "claude-sonnet-5-5__caveman__T2__r2__63d293",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T2",
   "skill": "caveman",
   "rep": 2,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 554,
   "input": 8,
   "cache_read": 98653,
   "cache_write": 22017,
   "warmup": null,
   "cache_write_1h": 22017,
   "cost_usd": 0.1133546,
   "wall_s": 6.5,
   "turns": 4,
   "prompt_last": 32253,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 0,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.113355
  },
  {
   "run": "claude-sonnet-5-5__caveman__T2__r3__ceaea3",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T2",
   "skill": "caveman",
   "rep": 3,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 634,
   "input": 8,
   "cache_read": 98659,
   "cache_write": 22113,
   "warmup": null,
   "cache_write_1h": 22113,
   "cost_usd": 0.1145398,
   "wall_s": 7.2,
   "turns": 4,
   "prompt_last": 32349,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 107,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.11454
  },
  {
   "run": "claude-sonnet-5-5__control__T2__r1__df7130",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T2",
   "skill": "control",
   "rep": 1,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 771,
   "input": 10,
   "cache_read": 114539,
   "cache_write": 17260,
   "warmup": null,
   "cache_write_1h": 17260,
   "cost_usd": 0.09967780000000001,
   "wall_s": 8.0,
   "turns": 5,
   "prompt_last": 27496,
   "lines_added": 3,
   "lines_deleted": 3,
   "lines": 6,
   "files": 1,
   "thinking": 0,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.099678
  },
  {
   "run": "claude-sonnet-5-5__control__T2__r2__6bb3d2",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T2",
   "skill": "control",
   "rep": 2,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 628,
   "input": 10,
   "cache_read": 115035,
   "cache_write": 17734,
   "warmup": null,
   "cache_write_1h": 17734,
   "cost_usd": 0.10024300000000001,
   "wall_s": 11.4,
   "turns": 5,
   "prompt_last": 27970,
   "lines_added": 3,
   "lines_deleted": 3,
   "lines": 6,
   "files": 1,
   "thinking": 0,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.100243
  },
  {
   "run": "claude-sonnet-5-5__control__T2__r3__d48363",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T2",
   "skill": "control",
   "rep": 3,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 505,
   "input": 10,
   "cache_read": 114869,
   "cache_write": 17350,
   "warmup": null,
   "cache_write_1h": 17350,
   "cost_usd": 0.09744380000000001,
   "wall_s": 6.9,
   "turns": 5,
   "prompt_last": 27586,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 0,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.097444
  },
  {
   "run": "claude-sonnet-5-5__karpathy__T2__r1__a7283b",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T2",
   "skill": "karpathy",
   "rep": 1,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 660,
   "input": 8,
   "cache_read": 92028,
   "cache_write": 18666,
   "warmup": null,
   "cache_write_1h": 18666,
   "cost_usd": 0.09968560000000001,
   "wall_s": 7.6,
   "turns": 4,
   "prompt_last": 28902,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 40,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.099686
  },
  {
   "run": "claude-sonnet-5-5__karpathy__T2__r2__edc7ab",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T2",
   "skill": "karpathy",
   "rep": 2,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 634,
   "input": 8,
   "cache_read": 92046,
   "cache_write": 18671,
   "warmup": null,
   "cache_write_1h": 18671,
   "cost_usd": 0.0994492,
   "wall_s": 9.2,
   "turns": 4,
   "prompt_last": 28907,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 37,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.099449
  },
  {
   "run": "claude-sonnet-5-5__karpathy__T2__r3__3f44a9",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T2",
   "skill": "karpathy",
   "rep": 3,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 500,
   "input": 8,
   "cache_read": 69201,
   "cache_write": 12351,
   "warmup": null,
   "cache_write_1h": 12351,
   "cost_usd": 0.0682602,
   "wall_s": 6.2,
   "turns": 4,
   "prompt_last": 22587,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 49,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.06826,
   "light": true
  },
  {
   "run": "claude-sonnet-5-5__placebo__T2__r1__63117c",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T2",
   "skill": "placebo",
   "rep": 1,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 884,
   "input": 10,
   "cache_read": 118907,
   "cache_write": 18371,
   "warmup": null,
   "cache_write_1h": 18371,
   "cost_usd": 0.10612540000000001,
   "wall_s": 8.7,
   "turns": 5,
   "prompt_last": 28607,
   "lines_added": 2,
   "lines_deleted": 1,
   "lines": 3,
   "files": 1,
   "thinking": 63,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.106125
  },
  {
   "run": "claude-sonnet-5-5__placebo__T2__r2__e25345",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T2",
   "skill": "placebo",
   "rep": 2,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 955,
   "input": 10,
   "cache_read": 118833,
   "cache_write": 19194,
   "warmup": null,
   "cache_write_1h": 19194,
   "cost_usd": 0.11011259999999999,
   "wall_s": 16.5,
   "turns": 5,
   "prompt_last": 29430,
   "lines_added": 4,
   "lines_deleted": 3,
   "lines": 7,
   "files": 1,
   "thinking": 98,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.110113
  },
  {
   "run": "claude-sonnet-5-5__placebo__T2__r3__8c4e11",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T2",
   "skill": "placebo",
   "rep": 3,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 772,
   "input": 8,
   "cache_read": 90974,
   "cache_write": 18184,
   "warmup": null,
   "cache_write_1h": 18184,
   "cost_usd": 0.0986668,
   "wall_s": 10.4,
   "turns": 4,
   "prompt_last": 28420,
   "lines_added": 2,
   "lines_deleted": 1,
   "lines": 3,
   "files": 1,
   "thinking": 35,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.098667
  },
  {
   "run": "claude-sonnet-5-5__ponytail__T2__r1__c6cb9f",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T2",
   "skill": "ponytail",
   "rep": 1,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 1316,
   "input": 12,
   "cache_read": 151183,
   "cache_write": 19994,
   "warmup": null,
   "cache_write_1h": 19994,
   "cost_usd": 0.12339660000000001,
   "wall_s": 12.1,
   "turns": 6,
   "prompt_last": 30230,
   "lines_added": 3,
   "lines_deleted": 5,
   "lines": 8,
   "files": 1,
   "thinking": 196,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.123397
  },
  {
   "run": "claude-sonnet-5-5__ponytail__T2__r2__4305a6",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T2",
   "skill": "ponytail",
   "rep": 2,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 886,
   "input": 8,
   "cache_read": 92585,
   "cache_write": 19380,
   "warmup": null,
   "cache_write_1h": 19380,
   "cost_usd": 0.10491299999999999,
   "wall_s": 10.6,
   "turns": 4,
   "prompt_last": 29616,
   "lines_added": 2,
   "lines_deleted": 2,
   "lines": 4,
   "files": 1,
   "thinking": 183,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.104913
  },
  {
   "run": "claude-sonnet-5-5__ponytail__T2__r3__e6f3f7",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T2",
   "skill": "ponytail",
   "rep": 3,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 728,
   "input": 8,
   "cache_read": 92107,
   "cache_write": 18822,
   "warmup": null,
   "cache_write_1h": 18822,
   "cost_usd": 0.10100540000000001,
   "wall_s": 10.3,
   "turns": 4,
   "prompt_last": 29058,
   "lines_added": 2,
   "lines_deleted": 2,
   "lines": 4,
   "files": 1,
   "thinking": 113,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.101005
  },
  {
   "run": "claude-sonnet-5-5__caveman__T3__r1__d38e6a",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "caveman",
   "rep": 1,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 653,
   "input": 4,
   "cache_read": 38360,
   "cache_write": 18698,
   "warmup": null,
   "cache_write_1h": 18698,
   "cost_usd": 0.089002,
   "wall_s": 6.8,
   "turns": 2,
   "prompt_last": 28934,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` calls `entries_in_month`. That function keeps only entries whose `day.year` and `day.month` both match. The report is a dict with three keys:\n- `count`: number of matching entries.\n- `total`: `total()` sums `amount` over the matching entries.\n- `by_category`: `sum_by_category()` groups matching amounts by category. Category is already lowercased at insert time.\n\nAn empty month gives `count` 0, `total` 0 (int, not float, since `sum` of empty starts at 0), and `by_category` `{}`.\n\n**Invalid date.** Dates are parsed only in `add_entry`, through `parse_date`. `parse_date` splits the text on `-`. It raises `ValueError(f\"bad date: {text!r}\")` when either of these holds:\n- The text does not have exactly 3 parts.\n- Any part is not all digits. This also catches empty parts and negative signs.\n\nValues that pass the format check but are impossible, like `2024-13-45`, make `date(y, m, d)` raise its own `ValueError`. The message differs, for example \"month must be in 1..12\".\n\nOrder in `add_entry` matters. The amount check runs first. Then `parse_date` runs while the `Entry` is built, before `append`. So an invalid date raises before any change to `entries`, and the list stays unmodified. Nothing catches the error. It goes to the caller.\n\nNon-string input, like `None`, raises `AttributeError` from `.split`, not `ValueError`.\n\nNote: the `isdigit()` check accepts non-ASCII digits, such as Arabic-Indic numerals. `int()` also accepts them, so those dates parse.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "api_cost": 0.089002
  },
  {
   "run": "claude-sonnet-5-5__caveman__T3__r2__0b5098",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "caveman",
   "rep": 2,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 746,
   "input": 4,
   "cache_read": 38357,
   "cache_write": 18694,
   "warmup": null,
   "cache_write_1h": 18694,
   "cost_usd": 0.08991539999999999,
   "wall_s": 9.2,
   "turns": 2,
   "prompt_last": 28930,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` computes them. It first calls `entries_in_month`. That function keeps each entry whose `day.year` equals `year` and whose `day.month` equals `month`. The report then returns a dict with three keys:\n\n- `count`: number of entries in that month.\n- `total`: sum of their `amount` values, from `total`.\n- `by_category`: dict from `sum_by_category`, mapping each category to its summed amount.\n\nCategories are already lowercase, because `add_entry` lowercases them when it stores an entry. So \"Food\" and \"food\" land in the same bucket. An empty month gives `count` 0, `total` 0 and an empty `by_category`. Amounts are plain `float`, so sums can carry normal floating-point rounding error.\n\n**Invalid date strings.** `monthly_report` never sees date strings. Dates are parsed earlier, in `add_entry`, through `parse_date`. `parse_date` splits the text on `-`. It raises `ValueError(f\"bad date: {text!r}\")` if either of these holds:\n\n- The text does not split into exactly 3 parts.\n- Any part is not all digits. This also rejects negative numbers and empty parts.\n\nIf the format passes, it calls `date(y, m, d)`. That call raises its own `ValueError` for out-of-range values, such as month 13 or Feb 30. The message is different, for example \"month must be in 1..12\".\n\n`parse_date` runs inside the `Entry(...)` constructor call in `add_entry`. A bad date therefore raises before `entries.append`, and the list stays unchanged. The `amount <= 0` check runs first. A bad amount is reported before a bad date.\n\nThe code does not catch the error. The caller must handle it.\n\nSome inputs that look wrong still pass the format check. `\"2024-1-5\"` is accepted because `isdigit` does not require zero padding. A non-string `text` fails with `AttributeError` on `.split`, not `ValueError`.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "api_cost": 0.089915
  },
  {
   "run": "claude-sonnet-5-5__caveman__T3__r3__4afd58",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "caveman",
   "rep": 3,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 747,
   "input": 4,
   "cache_read": 28407,
   "cache_write": 9067,
   "warmup": null,
   "cache_write_1h": 9067,
   "cost_usd": 0.049427399999999996,
   "wall_s": 8.9,
   "turns": 2,
   "prompt_last": 19303,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` computes them. It calls `entries_in_month`, which keeps only entries whose `day.year` and `day.month` match the arguments. It returns a dict with three values:\n- `count`: number of matching entries.\n- `total`: `total()` over the matching entries, a plain float sum of `amount`.\n- `by_category`: `sum_by_category()` over the matching entries, a dict of category to summed amount. Categories are already lowercased by `add_entry`.\n\nEntries from other months or years never enter the sums. An empty month gives count 0, total 0 (the int from `sum` of an empty sequence), and an empty dict.\n\n**Invalid date strings.** `monthly_report` never parses strings. `Entry.day` is already a `date` object. Parsing happens earlier, in `add_entry`, which calls `parse_date(day)`.\n\n`parse_date` splits the text on `-`. It raises `ValueError(f\"bad date: {text!r}\")` if there are not exactly three parts, or if any part is not all digits. That check rejects `\"2024/01/05\"`, `\"2024-1\"`, `\"-2024-01-05\"` and `\"2024-01-x5\"`.\n\nIf the shape is valid but the values are impossible, such as `\"2024-13-01\"` or `\"2024-02-30\"`, the `date(y, m, d)` constructor raises its own `ValueError`. The message is different, for example \"month must be in 1..12\".\n\nEither way `ValueError` propagates out of `add_entry`. The check happens in the `Entry(...)` argument list, before `append`, so the bad entry is not added. The amount check runs first: a non-positive amount raises a different `ValueError` before the date is parsed.\n\nTwo edge cases:\n- `parse_date` accepts non-padded parts such as `\"2024-1-5\"`.\n- `parse_date` passes `None` or other non-strings through to `.split`, which raises `AttributeError`, not `ValueError`.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "api_cost": 0.049427,
   "light": true
  },
  {
   "run": "claude-sonnet-5-5__control__T3__r1__a75a74",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "control",
   "rep": 1,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 964,
   "input": 4,
   "cache_read": 25759,
   "cache_write": 6417,
   "warmup": null,
   "cache_write_1h": 6417,
   "cost_usd": 0.0404678,
   "wall_s": 10.5,
   "turns": 2,
   "prompt_last": 16653,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 73,
   "commands": null,
   "answer": "**How monthly totals are computed**\n\n`monthly_report(entries, year, month)` produces the monthly figures in two steps:\n\n1. It calls `entries_in_month`. That function keeps only the entries whose `day.year` and `day.month` both match the arguments. It compares the parsed `date` object's fields directly and does no string handling.\n2. It builds a dict from that filtered list:\n   - `count` is the number of entries.\n   - `total` is the sum of their `amount` values, from `total()`.\n   - `by_category` maps each category to its summed amount, from `sum_by_category()`.\n\nA month with no entries gives a count of 0, a total of 0 and an empty `by_category` dict. Categories are lowercased in `add_entry`, so \"Food\" and \"food\" land in the same bucket.\n\n**What happens with an invalid date string**\n\nDate strings are only handled when an entry is added. `add_entry` calls `parse_date(day)`, which:\n\n- splits the string on `-`;\n- raises `ValueError(f\"bad date: ...\")` if there aren't exactly three parts or any part isn't all digits.\n\nOtherwise it passes the integers to `date(y, m, d)`. That constructor raises its own `ValueError` for out-of-range values such as month 13 or Feb 30.\n\nSince `add_entry` evaluates `parse_date` while constructing the `Entry`, a bad date raises before anything is appended. The `entries` list is left unchanged, and the error propagates to the caller. The check for `amount <= 0` runs first, so a call with both a bad amount and a bad date reports the amount error.\n\nSome inputs slip through `parse_date`'s digit check and fail later or parse oddly:\n\n- `\"2024-1-5\"` is accepted, because the parts aren't required to be zero-padded.\n- A non-string input such as `None` raises `AttributeError` on `.split`, not `ValueError`.\n- A negative-looking part like `\"2024--1-5\"` produces an empty part, which fails `isdigit`, so it is rejected.\n\n`monthly_report` never sees date strings. Once entries are stored their dates are valid `date` objects, so it can't fail on a bad date.",
   "account_remark": true,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "api_cost": 0.040468,
   "light": true
  },
  {
   "run": "claude-sonnet-5-5__control__T3__r2__c4fa7b",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "control",
   "rep": 2,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 858,
   "input": 4,
   "cache_read": 35712,
   "cache_write": 16049,
   "warmup": null,
   "cache_write_1h": 16049,
   "cost_usd": 0.07992640000000001,
   "wall_s": 8.3,
   "turns": 2,
   "prompt_last": 26285,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` in `ledger/core.py:51` does the work. It first calls `entries_in_month`, which keeps only entries whose `day.year` and `day.month` both match. It then builds a dict from that filtered list:\n\n- `count` is the number of matching entries.\n- `total` is the sum of their `amount` values, computed by `total()`.\n- `by_category` is a dict from `sum_by_category()`. It maps each category to the sum of its amounts, starting each category at `0.0`.\n\nCategories are lowercased when an entry is added, so \"Food\" and \"food\" land in the same bucket. A month with no entries gives `count` 0, `total` 0 (the integer, since `sum` of an empty sequence returns 0) and an empty `by_category`.\n\n**Invalid date strings.** Dates are parsed once, when an entry is added. `add_entry` calls `parse_date(day)` at `ledger/core.py:27`. `parse_date` splits the text on `-` and raises `ValueError(f\"bad date: ...\")` if there are not exactly three parts or if any part is not all digits. Examples are `\"2024/01/05\"`, `\"2024-1\"`, `\"abc\"` and `\"2024-01--5\"`. Because the parse happens inside the `Entry(...)` construction, the error is raised before `entries.append` runs, so the bad entry is never added and the list is unchanged.\n\nTwo related behaviours:\n\n- **Format-valid but impossible dates.** Something like `\"2024-13-45\"` passes the digit check. It then fails inside `date(y, m, d)`, which raises its own `ValueError` (\"month must be in 1..12\" and similar). Callers see a `ValueError` either way, but with a different message.\n- **Loose formatting.** Because the check only requires digits, `\"2024-1-5\"` is accepted, as are unpadded parts. Negative numbers are rejected, since `-` is the separator and leaves an empty part.\n\nIn `add_entry`, the `amount <= 0` check runs before date parsing. An entry with both a bad amount and a bad date therefore reports the amount error first.\n\nThe monthly functions never see raw strings, because `Entry.day` is already a `date`. An invalid date can't reach the monthly totals. It fails at insertion.\n\nI didn't change any files.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "api_cost": 0.079926
  },
  {
   "run": "claude-sonnet-5-5__control__T3__r3__70757d",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "control",
   "rep": 3,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 854,
   "input": 4,
   "cache_read": 25850,
   "cache_write": 9342,
   "warmup": null,
   "cache_write_1h": 9342,
   "cost_usd": 0.051086,
   "wall_s": 7.5,
   "turns": 2,
   "prompt_last": 19578,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals**\n\n`monthly_report(entries, year, month)` (line 51) builds the report in three steps:\n\n1. `entries_in_month` filters the list, keeping entries whose `day.year` and `day.month` both match the arguments. Each `Entry.day` is already a `date` object.\n2. It returns a dict with three keys:\n   - `count` is the number of entries in that month.\n   - `total` comes from `total()`, which sums `amount` over the filtered entries.\n   - `by_category` comes from `sum_by_category()`. It walks the entries and adds each amount into a dict keyed by category, starting from 0.0.\n3. Categories are lowercased in `add_entry`, so \"Food\" and \"food\" land in the same bucket.\n\nThe sums are plain `float` addition, with no rounding. Amounts like 0.1 and 0.2 can therefore give results with floating-point noise.\n\n**Invalid date strings**\n\nDate parsing happens only when an entry is added, in `add_entry` (line 27), which calls `parse_date`. The reporting functions never see strings.\n\n`parse_date` splits the text on `-`. It raises `ValueError(f\"bad date: {text!r}\")` if either of these holds:\n- There aren't exactly three parts.\n- Any part isn't all digits.\n\nThat check catches `\"2024/01/05\"`, `\"2024-1\"`, `\"abc\"`, and negative-looking or blank parts.\n\nA string that passes the check but isn't a real calendar date, such as `\"2024-13-01\"` or `\"2024-02-30\"`, gets past the format check. It then fails inside `date(y, m, d)`, which raises its own `ValueError` (e.g. \"month must be in 1..12\"). The message differs from the \"bad date\" one, but the exception type is the same.\n\nNon-string input, such as `None`, fails earlier with an `AttributeError` on `.split`, not a `ValueError`.\n\nNothing catches the error, so it propagates to the caller of `add_entry`. Two details about that call:\n- `amount` is validated before the date, so a non-positive amount raises its own `ValueError` first.\n- `parse_date` runs while the `Entry(...)` arguments are being evaluated, before `entries.append` executes. A bad date therefore never adds a partial entry, and the list is left unchanged.\n\nI didn't change any files.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "api_cost": 0.051086,
   "light": true
  },
  {
   "run": "claude-sonnet-5-5__karpathy__T3__r1__97de84",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "karpathy",
   "rep": 1,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 785,
   "input": 4,
   "cache_read": 36727,
   "cache_write": 17064,
   "warmup": null,
   "cache_write_1h": 17064,
   "cost_usd": 0.08345939999999999,
   "wall_s": 8.0,
   "turns": 2,
   "prompt_last": 27300,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` in `ledger/core.py:51` does the work. It first calls `entries_in_month`, which keeps only the entries whose `day.year` and `day.month` both match the arguments. It then builds a dict from that filtered list with three values:\n- `count` is the number of matching entries.\n- `total` is the sum of their `amount` values, from `total()`.\n- `by_category` is a dict of per-category sums, from `sum_by_category()`. Categories are lowercased when an entry is added, so \"Food\" and \"food\" land in the same bucket.\n\nA month with no entries gives a count of 0, a total of 0 (the integer, since `sum` of an empty sequence is 0) and an empty `by_category`.\n\n**Invalid date strings.** Date parsing happens only in `add_entry` (`ledger/core.py:24`), which calls `parse_date` on the incoming string. `parse_date` splits the text on `-`. It raises `ValueError(\"bad date: ...\")` if there aren't exactly three parts or if any part isn't all digits. That means it rejects things like `\"2024/01/05\"`, `\"2024-1\"` and `\"2024-01-xx\"`, and also negative numbers, because the minus sign becomes a separator.\n\nIf the string has three digit groups but isn't a real date, such as `\"2024-13-40\"`, the `date(y, m, d)` constructor raises its own `ValueError`. The message is different, for example \"month must be in 1..12\".\n\nIn `add_entry` the amount check runs first. A non-positive amount raises before the date is parsed. If parsing fails, the exception propagates and nothing is appended. The `Entry` is built as an argument to `append`, so the list is left unchanged.\n\nOnce an entry exists, its `day` is always a valid `date`. The monthly report never has to handle bad dates, and an invalid date can only fail at insertion time.\n\n`parse_date` accepts non-padded parts like `\"2024-1-5\"`, since it only checks that each part is digits. The docstring says \"YYYY-MM-DD\", so this is looser than it claims.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "api_cost": 0.083459
  },
  {
   "run": "claude-sonnet-5-5__karpathy__T3__r2__559943",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "karpathy",
   "rep": 2,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 868,
   "input": 4,
   "cache_read": 36725,
   "cache_write": 17061,
   "warmup": null,
   "cache_write_1h": 17061,
   "cost_usd": 0.084277,
   "wall_s": 9.1,
   "turns": 2,
   "prompt_last": 27297,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` in `ledger/core.py:51` does the work in three steps:\n\n1. It calls `entries_in_month`, which keeps only entries whose `day.year` and `day.month` both match the arguments.\n2. It passes that filtered list to `total`, which adds up every `amount`. This gives the month's overall figure.\n3. It passes the same list to `sum_by_category`, which builds a dict of category to running sum.\n\nIt returns a dict with three keys: `count` (the number of entries that month), `total`, and `by_category`. If no entries match, `count` is 0, `total` is `0` (the integer that `sum` returns for an empty sequence, not `0.0`), and `by_category` is `{}`.\n\nNothing is stored between calls. Each call re-filters and re-sums the whole list. Amounts are plain floats, so the sums can pick up ordinary floating-point rounding error.\n\n**Invalid date strings.** Dates are only parsed when an entry is added, in `add_entry` (line 27), which calls `parse_date`. `parse_date` splits the text on `-`. It raises `ValueError(f\"bad date: {text!r}\")` if there aren't exactly three parts or if any part contains a non-digit character. A negative year or a string like `\"2024-1-5\"` gets past the digit check. Only the part count and the digits matter here, so single-digit months and days are accepted.\n\nThere are two more failure modes:\n- If the parts are all digits but don't form a real date, such as `2024-02-30` or month `13`, the `date(y, m, d)` constructor raises its own `ValueError`, with a different message such as \"day is out of range for month\".\n- If `text` isn't a string (for example `None`), `text.split` raises `AttributeError`, not `ValueError`, despite the docstring.\n\nNeither `add_entry` nor `parse_date` catches any of these, so the exception goes straight to the caller. `add_entry` parses the date while building the `Entry`, before `entries.append` runs, so a bad date never adds a partial entry to the list. It does check `amount <= 0` first, so a call with both a bad amount and a bad date reports the amount error.\n\nBecause dates are validated on the way in, `monthly_report` and the other functions can assume every `Entry.day` is a real `date`. They have no invalid-date handling of their own.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "api_cost": 0.084277
  },
  {
   "run": "claude-sonnet-5-5__karpathy__T3__r3__920bca",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "karpathy",
   "rep": 3,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 840,
   "input": 4,
   "cache_read": 36726,
   "cache_write": 17063,
   "warmup": null,
   "cache_write_1h": 17063,
   "cost_usd": 0.0840052,
   "wall_s": 8.3,
   "turns": 2,
   "prompt_last": 27299,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` in `ledger/core.py:51` first calls `entries_in_month`. That function keeps only the entries whose `day.year` and `day.month` both match the arguments. The report is a dict with three values:\n- `count` is the number of entries left after filtering.\n- `total` is `total()` applied to those entries, a plain sum of their `amount` floats.\n- `by_category` comes from `sum_by_category()`. It builds a dict mapping each category to its running sum, starting from 0.0.\n\nCategory names are lowercased when an entry is added (`add_entry`), so \"Food\" and \"food\" land in the same bucket. The sums are plain float addition with no rounding. If no entries match the month, you get `count` 0, `total` 0 (an int, since `sum` of an empty sequence is 0), and an empty `by_category` dict.\n\n**Invalid date strings.** Dates are only parsed in `add_entry`, through `parse_date` (`core.py:15`). The monthly functions work on already-parsed `date` objects and never see strings. `parse_date` splits the text on `-`. It raises `ValueError(\"bad date: ...\")` if there aren't exactly three parts, or if any part isn't all digits. It then converts the parts to ints and calls `date(y, m, d)`. If the format is right but the values are impossible, such as `2024-13-01` or `2024-02-30`, `date()` raises its own `ValueError`, with a different message.\n\n`add_entry` validates the amount before it parses the date. A bad date therefore propagates as an uncaught `ValueError` and nothing is appended to the list. Nothing catches or logs the error, so callers must handle it.\n\nSome edge cases follow from the digit check:\n- `\"2024-1-5\"` is accepted, because the parts aren't required to be zero-padded.\n- Negative parts like `\"-2024-01-01\"` are rejected, since the split gives an empty first part.\n- A non-string input, such as `None`, raises `AttributeError` at `.split` instead of `ValueError`.\n- `str.isdigit()` also accepts some non-ASCII digit characters, for example superscripts like `\"\u00b2\"`. `int()` can't convert those, so they raise a `ValueError` with a different message.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "api_cost": 0.084005
  },
  {
   "run": "claude-sonnet-5-5__placebo__T3__r1__87ea61",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "placebo",
   "rep": 1,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 889,
   "input": 4,
   "cache_read": 36497,
   "cache_write": 16834,
   "warmup": null,
   "cache_write_1h": 16834,
   "cost_usd": 0.08353340000000001,
   "wall_s": 10.6,
   "turns": 2,
   "prompt_last": 27070,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` in `ledger/core.py:51` builds the monthly figures in three steps.\n\n1. `entries_in_month` (line 40) keeps only the entries whose `day.year` and `day.month` both match the arguments.\n2. `total` (line 31) sums the `amount` of those entries. For an empty month, `sum` of nothing returns `0`.\n3. `sum_by_category` (line 44) builds a dict from category to running sum, starting each category at `0.0`.\n\nThe function returns a dict with three keys:\n- `count`: the number of entries in the month.\n- `total`: the sum of their amounts.\n- `by_category`: the per-category sums.\n\nCategories are lowercased when an entry is added, so `\"Food\"` and `\"food\"` land in the same bucket. `monthly_report` doesn't parse any dates itself. It compares against the `date` objects already stored on each `Entry`.\n\n**Invalid date strings.** Date strings are only handled when an entry is added, through `add_entry` (line 24), which calls `parse_date` (line 15).\n\n- `parse_date` splits the text on `-`. It raises `ValueError(f\"bad date: {text!r}\")` if there aren't exactly three parts or if any part isn't made up only of digits. That rejects things like `\"2024/01/05\"`, `\"2024-1\"`, `\"2024-01-xx\"` and negative numbers.\n- If the format passes, it calls `date(y, m, d)`. A well-formed but impossible date such as `\"2024-13-40\"` or `\"2023-02-29\"` makes `datetime.date` raise its own `ValueError`, for example \"month must be in 1..12\". That message is less specific than the \"bad date\" one.\n- `parse_date` also assumes `text` is a string. `None` or another non-string type would raise `AttributeError` at `.split`, not `ValueError`.\n- Nothing catches these errors, so they propagate to the caller of `add_entry`. Because `parse_date` runs while the `Entry` is being constructed, the `append` never happens. A bad date therefore never adds a partial entry, and the list is left unchanged. The `amount <= 0` check runs first, so an entry with a non-positive amount and a bad date reports the amount error.\n\nInvalid dates can't reach the monthly totals. They are rejected at entry time, so `monthly_report` only ever sees valid `date` objects.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "api_cost": 0.083533
  },
  {
   "run": "claude-sonnet-5-5__placebo__T3__r2__744d38",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "placebo",
   "rep": 2,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 910,
   "input": 4,
   "cache_read": 26604,
   "cache_write": 10098,
   "warmup": null,
   "cache_write_1h": 10098,
   "cost_usd": 0.0548208,
   "wall_s": 9.4,
   "turns": 2,
   "prompt_last": 20334,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals**\n\n`monthly_report(entries, year, month)` in `ledger/core.py` builds the monthly figures in three steps:\n\n1. `entries_in_month` keeps only the entries whose `day.year` and `day.month` both match the arguments. Each `Entry` stores `day` as a real `date` object, so this is a plain attribute comparison.\n2. `total` adds up the `amount` of those entries to give the month's overall spend.\n3. `sum_by_category` walks the same entries and builds a dict from category to running sum. Categories were lowercased when the entry was added, so `Food` and `food` end up in the same bucket.\n\nThe function returns a dict with `count` (the number of entries in the month), `total`, and `by_category`. A month with no entries gives a count of 0, a total of `0` (the int that `sum` returns for an empty sequence) and an empty category dict. Nothing is raised in that case.\n\n**Invalid date strings**\n\n`monthly_report` never sees date strings. Parsing happens earlier, in `add_entry`, which calls `parse_date(day)`.\n\n`parse_date` expects `YYYY-MM-DD`. It splits the text on `-` and raises `ValueError(f\"bad date: {text!r}\")` if either of these is true:\n- there aren't exactly three parts\n- any part contains a non-digit character (this includes empty parts, and negative years, because the sign splits into an extra part)\n\nIf the shape is right, it converts the parts to ints and calls `date(y, m, d)`. Values that are numeric but impossible, such as `2024-13-01` or `2023-02-30`, fail inside `date()`. That raises a `ValueError` too, with the standard library's message (for example \"month must be in 1..12\") instead of the \"bad date\" one. The check also doesn't enforce digit counts, so `2024-1-5` is accepted.\n\nIn `add_entry`, `parse_date` is evaluated while the `Entry` is being constructed, which happens before `entries.append`. A bad date therefore propagates the `ValueError` to the caller and the entry is not added. The list is left unchanged. The amount check (`amount <= 0` raises `ValueError`) also runs first, so an entry with both a bad amount and a bad date reports the amount error. `parse_date` is not wrapped in a try/except, so callers have to handle the exception themselves.\n\nA non-string `day` (such as `None`) would fail earlier with an `AttributeError` on `.split`, not a `ValueError`.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "api_cost": 0.054821,
   "light": true
  },
  {
   "run": "claude-sonnet-5-5__placebo__T3__r3__39a254",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "placebo",
   "rep": 3,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 930,
   "input": 4,
   "cache_read": 36461,
   "cache_write": 16797,
   "warmup": null,
   "cache_write_1h": 16797,
   "cost_usd": 0.08378820000000001,
   "wall_s": 8.7,
   "turns": 2,
   "prompt_last": 27033,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` in `ledger/core.py:51` builds the monthly figures in three steps:\n\n1. It calls `entries_in_month`, which keeps only the entries whose `day.year` and `day.month` both match the arguments.\n2. It passes that filtered list to `total`, which adds up the `amount` of each entry. That gives the month's overall total.\n3. It also passes the list to `sum_by_category`, which builds a dict mapping each category name to the sum of its amounts.\n\nIt returns a dict with three keys:\n- `count`: the number of entries in the month.\n- `total`: the month's overall total.\n- `by_category`: the per-category sums.\n\nA month with no entries gives `count` 0, `total` 0 and an empty `by_category`. The `total` is the integer 0 in that case, because `sum` over an empty sequence returns `0`. Amounts are plain floats, so sums can carry ordinary floating-point rounding noise. Nothing rounds them.\n\n**Invalid date strings.** Dates are only parsed when an entry is added. `add_entry` calls `parse_date`, which splits the text on `-`. It raises `ValueError(f\"bad date: ...\")` if there aren't exactly three parts or if any part isn't all digits. That check catches things like `\"2024/01/05\"`, `\"2024-1\"`, `\"\"` and `\"2024-01--5\"`.\n\nOnce the format passes, the values go to `date(y, m, d)`. If they are out of range, for example `\"2024-13-01\"` or `\"2024-02-30\"`, the `datetime.date` constructor raises its own `ValueError`, such as \"month must be in 1..12\". That message doesn't include the original string.\n\n`add_entry` doesn't catch either error, so the caller sees the `ValueError`. Because `parse_date` runs inside the `Entry(...)` construction, which happens before `entries.append`, a bad date never adds a partial entry to the list.\n\nThe amount check runs first. A non-positive amount raises `ValueError(\"amount must be positive\")` before the date is parsed. So an entry with both problems reports the amount error.\n\nSome inputs fail differently. If `text` isn't a string, for example `None`, `text.split` raises `AttributeError` instead of `ValueError`. Also, `str.isdigit()` accepts some non-ASCII digit characters, and `int()` converts them, so a few odd inputs get through the format check.\n\nSince `monthly_report` works on `Entry` objects that already hold `date` values, it never sees invalid dates.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "api_cost": 0.083788
  },
  {
   "run": "claude-sonnet-5-5__ponytail__T3__r1__7b8509",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "ponytail",
   "rep": 1,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 818,
   "input": 4,
   "cache_read": 36741,
   "cache_write": 17079,
   "warmup": null,
   "cache_write_1h": 17079,
   "cost_usd": 0.08385219999999999,
   "wall_s": 9.9,
   "turns": 2,
   "prompt_last": 27315,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` in `ledger/core.py:51` does three things:\n\n1. `entries_in_month` keeps only the entries whose `day.year` and `day.month` both match the arguments.\n2. `total` adds up the `amount` of those entries with `sum`.\n3. `sum_by_category` builds a dict that maps each category to its running total. It uses `out.get(cat, 0.0) + amount`.\n\nIt returns `{\"count\": ..., \"total\": ..., \"by_category\": {...}}`. A month with no entries gives count 0, total 0 and an empty dict.\n\nAmounts are plain floats, so the totals can carry small rounding error. Categories are lowercased when an entry is added (`add_entry`), so \"Food\" and \"food\" land in the same bucket.\n\n**Invalid date strings.** Dates are parsed only in `add_entry`, through `parse_date` (`ledger/core.py:15`). That function splits the text on `-` and raises `ValueError(\"bad date: ...\")` in two cases:\n\n- The text doesn't have exactly three parts.\n- Any part contains something other than digits.\n\nOtherwise it converts the parts to ints and calls `date(y, m, d)`. That call raises its own `ValueError` for out-of-range values such as month 13 or Feb 30. Its message is different from the \"bad date\" one.\n\n`add_entry` calls `parse_date` inside the `Entry(...)` construction. An invalid date therefore raises before `entries.append` runs, and the list is left unchanged. Nothing catches the error, so the caller has to handle it.\n\n`add_entry` checks `amount <= 0` before it parses the date. If both the amount and the date are bad, you get the amount error.\n\nA few inputs behave in ways you might not expect:\n\n- `\"2024-1-5\"` is accepted, because there is no width check.\n- `\" 2024-01-05\"` is rejected, because the space makes the first part fail `isdigit`.\n- Unicode digits such as `\"\u00b2\"` pass `isdigit` but then make `int()` raise a `ValueError` with yet another message.\n\n`monthly_report` never sees date strings, only `date` objects, so bad dates can't reach the totals.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "api_cost": 0.083852
  },
  {
   "run": "claude-sonnet-5-5__ponytail__T3__r2__ca430d",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "ponytail",
   "rep": 2,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 724,
   "input": 4,
   "cache_read": 36740,
   "cache_write": 17078,
   "warmup": null,
   "cache_write_1h": 17078,
   "cost_usd": 0.08290799999999998,
   "wall_s": 7.1,
   "turns": 2,
   "prompt_last": 27314,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` in `ledger/core.py:51` does the work in three steps:\n\n1. It calls `entries_in_month`, which keeps only the entries whose `day.year` and `day.month` both match the arguments.\n2. It passes that filtered list to `total`, which adds up each entry's `amount` with `sum`.\n3. It passes the same list to `sum_by_category`, which builds a dict of category to running sum.\n\nIt returns `{\"count\", \"total\", \"by_category\"}`. A month with no entries gives count 0, total 0 and an empty dict. Nothing is stored or cached. The report is recomputed from the raw entry list on every call.\n\n**Invalid date strings.** Dates are parsed only when an entry is added, in `add_entry` (`core.py:27`), which calls `parse_date`. `parse_date` splits the text on `-`. It raises `ValueError(\"bad date: ...\")` if the result isn't exactly three parts or any part isn't all digits.\n\nOtherwise it passes the integers to `datetime.date(y, m, d)`. That raises its own `ValueError` for out-of-range values such as month 13 or Feb 30. So malformed strings and impossible dates both raise `ValueError`, with different messages.\n\nThe error is not caught anywhere in this file. In `add_entry`, the amount check runs first. Then `parse_date` runs while the `Entry` is being built, before `append`. A bad date therefore leaves the list unchanged, and the caller has to handle the exception.\n\n`monthly_report` never sees a bad date. By the time an entry is in the list, its `day` is already a valid `date` object.\n\nTwo quirks in `parse_date`:\n- `str.isdigit()` accepts some non-ASCII digits, and `date()` would still take those.\n- Inputs like `\"2024-1-5\"` are accepted, because the parts aren't checked for a fixed width.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "api_cost": 0.082908
  },
  {
   "run": "claude-sonnet-5-5__ponytail__T3__r3__72b8dc",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "ponytail",
   "rep": 3,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 785,
   "input": 4,
   "cache_read": 36743,
   "cache_write": 17082,
   "warmup": null,
   "cache_write_1h": 17082,
   "cost_usd": 0.08353460000000001,
   "wall_s": 10.0,
   "turns": 2,
   "prompt_last": 27318,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` in `ledger/core.py:51` builds the totals in three steps:\n\n1. `entries_in_month` (line 40) keeps only the entries whose `day.year` and `day.month` match the arguments.\n2. `total` (line 31) adds up the `amount` of those entries. `sum_by_category` (line 44) does the same, grouped by each entry's lowercased category.\n3. It returns a dict with `count`, `total` and `by_category`.\n\nAn empty month gives `count` 0, `total` 0 and an empty `by_category`. The function reads only `Entry.day`, which is already a `date`, so it never sees date strings.\n\n**Invalid date strings.** Strings are only parsed when an entry is added. `add_entry` (line 24) calls `parse_date` (line 15), which splits the text on `-`. It raises `ValueError(f\"bad date: ...\")` if there aren't exactly three parts or any part isn't all digits. That rejects `\"2024/01/05\"`, `\"2024-1\"`, `\"2024-01-xx\"` and negative numbers.\n\nIf the format passes, the parts become ints and go to `date(y, m, d)`. That constructor raises its own `ValueError` for impossible values, such as month 13 or `2024-02-30`. Its message is different, for example \"month must be in 1..12\".\n\n`add_entry` doesn't catch either error. The exception goes straight to the caller. It also fires before `entries.append`, so a bad date never adds a partial entry. The one earlier check is `amount <= 0`, which raises its own `ValueError` first.\n\nTwo things I noticed:\n- `parse_date` calls `text.split`, so `None` or another non-string raises `AttributeError` rather than `ValueError`.\n- `isdigit()` accepts some non-ASCII digit characters. `int()` usually handles them, but it is looser than the docstring's \"YYYY-MM-DD\" suggests. It also doesn't require four-digit years or two-digit months, so `\"2024-1-5\"` is accepted.\n\nI didn't change any files.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "api_cost": 0.083535
  },
  {
   "run": "claude-sonnet-5-5__caveman__T4__r1__0f1d14",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T4",
   "skill": "caveman",
   "rep": 1,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 1349,
   "input": 8,
   "cache_read": 100268,
   "cache_write": 24513,
   "warmup": null,
   "cache_write_1h": 24513,
   "cost_usd": 0.1316116,
   "wall_s": 17.2,
   "turns": 5,
   "prompt_last": 34749,
   "lines_added": 55,
   "lines_deleted": 0,
   "lines": 55,
   "files": 1,
   "thinking": 69,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.131612
  },
  {
   "run": "claude-sonnet-5-5__caveman__T4__r2__8af7a6",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T4",
   "skill": "caveman",
   "rep": 2,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 1358,
   "input": 8,
   "cache_read": 100277,
   "cache_write": 24498,
   "warmup": null,
   "cache_write_1h": 24498,
   "cost_usd": 0.1316434,
   "wall_s": 11.5,
   "turns": 5,
   "prompt_last": 34734,
   "lines_added": 53,
   "lines_deleted": 0,
   "lines": 53,
   "files": 1,
   "thinking": 43,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.131643
  },
  {
   "run": "claude-sonnet-5-5__caveman__T4__r3__36a065",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T4",
   "skill": "caveman",
   "rep": 3,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 1298,
   "input": 8,
   "cache_read": 99909,
   "cache_write": 23929,
   "warmup": null,
   "cache_write_1h": 23929,
   "cost_usd": 0.12869380000000002,
   "wall_s": 10.9,
   "turns": 4,
   "prompt_last": 34165,
   "lines_added": 58,
   "lines_deleted": 0,
   "lines": 58,
   "files": 1,
   "thinking": 59,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.128694
  },
  {
   "run": "claude-sonnet-5-5__control__T4__r1__80fe70",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T4",
   "skill": "control",
   "rep": 1,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 1187,
   "input": 8,
   "cache_read": 65454,
   "cache_write": 11154,
   "warmup": null,
   "cache_write_1h": 11154,
   "cost_usd": 0.0695928,
   "wall_s": 11.3,
   "turns": 4,
   "prompt_last": 21390,
   "lines_added": 43,
   "lines_deleted": 0,
   "lines": 43,
   "files": 1,
   "thinking": 0,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.069593,
   "light": true
  },
  {
   "run": "claude-sonnet-5-5__control__T4__r2__a8a77a",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T4",
   "skill": "control",
   "rep": 2,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 1319,
   "input": 8,
   "cache_read": 89668,
   "cache_write": 18955,
   "warmup": null,
   "cache_write_1h": 18955,
   "cost_usd": 0.10695959999999999,
   "wall_s": 14.5,
   "turns": 4,
   "prompt_last": 29191,
   "lines_added": 49,
   "lines_deleted": 0,
   "lines": 49,
   "files": 1,
   "thinking": 80,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.10696
  },
  {
   "run": "claude-sonnet-5-5__control__T4__r3__381a37",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T4",
   "skill": "control",
   "rep": 3,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 1108,
   "input": 8,
   "cache_read": 89158,
   "cache_write": 18355,
   "warmup": null,
   "cache_write_1h": 18355,
   "cost_usd": 0.10234760000000001,
   "wall_s": 11.3,
   "turns": 4,
   "prompt_last": 28591,
   "lines_added": 49,
   "lines_deleted": 0,
   "lines": 49,
   "files": 1,
   "thinking": 0,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.102348
  },
  {
   "run": "claude-sonnet-5-5__karpathy__T4__r1__5ea177",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T4",
   "skill": "karpathy",
   "rep": 1,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 1405,
   "input": 8,
   "cache_read": 93609,
   "cache_write": 20951,
   "warmup": null,
   "cache_write_1h": 20951,
   "cost_usd": 0.11659180000000002,
   "wall_s": 11.8,
   "turns": 5,
   "prompt_last": 31187,
   "lines_added": 44,
   "lines_deleted": 0,
   "lines": 44,
   "files": 1,
   "thinking": 101,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.116592
  },
  {
   "run": "claude-sonnet-5-5__karpathy__T4__r2__c15ddc",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T4",
   "skill": "karpathy",
   "rep": 2,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 1349,
   "input": 8,
   "cache_read": 94637,
   "cache_write": 21737,
   "warmup": null,
   "cache_write_1h": 21737,
   "cost_usd": 0.11938140000000001,
   "wall_s": 15.4,
   "turns": 4,
   "prompt_last": 31973,
   "lines_added": 43,
   "lines_deleted": 0,
   "lines": 43,
   "files": 1,
   "thinking": 176,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.119381
  },
  {
   "run": "claude-sonnet-5-5__karpathy__T4__r3__e6e5ff",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T4",
   "skill": "karpathy",
   "rep": 3,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 1275,
   "input": 8,
   "cache_read": 65282,
   "cache_write": 11920,
   "warmup": null,
   "cache_write_1h": 11920,
   "cost_usd": 0.0735024,
   "wall_s": 9.8,
   "turns": 4,
   "prompt_last": 22156,
   "lines_added": 43,
   "lines_deleted": 0,
   "lines": 43,
   "files": 1,
   "thinking": 76,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.073502,
   "light": true
  },
  {
   "run": "claude-sonnet-5-5__placebo__T4__r1__864871",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T4",
   "skill": "placebo",
   "rep": 1,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 1444,
   "input": 8,
   "cache_read": 92222,
   "cache_write": 20079,
   "warmup": null,
   "cache_write_1h": 20079,
   "cost_usd": 0.1132164,
   "wall_s": 12.9,
   "turns": 4,
   "prompt_last": 30315,
   "lines_added": 52,
   "lines_deleted": 0,
   "lines": 52,
   "files": 2,
   "thinking": 124,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.113216
  },
  {
   "run": "claude-sonnet-5-5__placebo__T4__r2__e6b0df",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T4",
   "skill": "placebo",
   "rep": 2,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 1629,
   "input": 8,
   "cache_read": 91744,
   "cache_write": 19886,
   "warmup": null,
   "cache_write_1h": 19886,
   "cost_usd": 0.11419880000000002,
   "wall_s": 13.1,
   "turns": 5,
   "prompt_last": 30122,
   "lines_added": 59,
   "lines_deleted": 0,
   "lines": 59,
   "files": 1,
   "thinking": 79,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.114199
  },
  {
   "run": "claude-sonnet-5-5__placebo__T4__r3__bc333d",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T4",
   "skill": "placebo",
   "rep": 3,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 1571,
   "input": 8,
   "cache_read": 92581,
   "cache_write": 20571,
   "warmup": null,
   "cache_write_1h": 20571,
   "cost_usd": 0.1165262,
   "wall_s": 13.6,
   "turns": 4,
   "prompt_last": 30807,
   "lines_added": 62,
   "lines_deleted": 0,
   "lines": 62,
   "files": 2,
   "thinking": 96,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.116526
  },
  {
   "run": "claude-sonnet-5-5__ponytail__T4__r1__551c43",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T4",
   "skill": "ponytail",
   "rep": 1,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 1101,
   "input": 8,
   "cache_read": 93669,
   "cache_write": 20686,
   "warmup": null,
   "cache_write_1h": 20686,
   "cost_usd": 0.11250380000000001,
   "wall_s": 10.3,
   "turns": 4,
   "prompt_last": 30922,
   "lines_added": 31,
   "lines_deleted": 0,
   "lines": 31,
   "files": 1,
   "thinking": 126,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.112504
  },
  {
   "run": "claude-sonnet-5-5__ponytail__T4__r2__026843",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T4",
   "skill": "ponytail",
   "rep": 2,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 1159,
   "input": 8,
   "cache_read": 93326,
   "cache_write": 20320,
   "warmup": null,
   "cache_write_1h": 20320,
   "cost_usd": 0.11155119999999999,
   "wall_s": 10.6,
   "turns": 4,
   "prompt_last": 30556,
   "lines_added": 31,
   "lines_deleted": 0,
   "lines": 31,
   "files": 1,
   "thinking": 136,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.111551
  },
  {
   "run": "claude-sonnet-5-5__ponytail__T4__r3__70536e",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T4",
   "skill": "ponytail",
   "rep": 3,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 1248,
   "input": 8,
   "cache_read": 93594,
   "cache_write": 20671,
   "warmup": null,
   "cache_write_1h": 20671,
   "cost_usd": 0.1138988,
   "wall_s": 13.0,
   "turns": 4,
   "prompt_last": 30907,
   "lines_added": 34,
   "lines_deleted": 0,
   "lines": 34,
   "files": 1,
   "thinking": 205,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "heldout": true,
   "api_cost": 0.113899
  }
 ],
 "round10_runs": [
  {
   "run": "claude-opus-5-5__control__T3__r1__e556c8",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "control",
   "rep": 1,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 1728,
   "input": 4,
   "cache_read": 36452,
   "cache_write": 17017,
   "warmup": null,
   "cache_write_1h": 17017,
   "cost_usd": 0.1780024,
   "wall_s": 21.4,
   "turns": 2,
   "prompt_last": 27137,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 446,
   "commands": null,
   "answer": "I read `ledger/core.py` and changed nothing.\n\n## How monthly totals are computed\n\n`monthly_report(entries, year, month)` (`ledger/core.py:51`) works in three steps:\n\n1. **Filter by month.** `entries_in_month` (`ledger/core.py:40`) keeps only the entries whose `day` has the given year and month. It compares integers on the stored `date` objects, so no string handling happens at this point.\n2. **Sum the amounts.** `total` (`ledger/core.py:31`) adds up the `amount` of every entry that passed the filter.\n3. **Break it down by category.** `sum_by_category` (`ledger/core.py:44`) builds a dictionary that maps each category to its running total. It starts each category at `0.0`.\n\nThe result is a dictionary with three keys: `count` (number of entries that month), `total`, and `by_category`.\n\nSome details worth knowing:\n\n- **Nothing is saved between calls.** Every call scans the whole list again, so the report always reflects the list as it is right now.\n- **Categories are lowercased when you add an entry.** \"Food\" and \"food\" end up in the same bucket.\n- **Year and month are not checked.** Asking for month 13 or month 0 doesn't raise an error. It just returns an empty report.\n- **An empty month returns the integer `0` as its total, not `0.0`.** That's because `sum()` over nothing gives `0`. It compares equal to `0.0`, but its type is `int`.\n- **Amounts are floats.** Totals can show small rounding errors, such as `0.1 + 0.2` giving `0.30000000000000004`.\n\n## What happens when a date string is invalid\n\nDates are only parsed in `add_entry`, which calls `parse_date` (`ledger/core.py:15`). An invalid date is caught in two ways, and both raise a `ValueError`:\n\n1. **Format check.** The string is split on `-`. It must give exactly three parts, and each part must consist only of digits. If not, the function raises `ValueError(\"bad date: '...'\")`. This rejects things like `\"2024/01/05\"`, `\"2024-01\"`, `\" 2024-01-05\"` (with a leading space) and `\"2024-01-05T00:00\"`.\n2. **Calendar check.** Strings that pass the format check are turned into integers and passed to `datetime.date`. That raises its own `ValueError` for dates that don't exist, such as `\"2024-02-30\"`, `\"2024-13-01\"` or `\"0000-01-01\"`. The message is Python's default (for example, \"day is out of range for month\"), not the custom \"bad date\" text.\n\nBecause the error happens while the `Entry` is being built, the ledger list is never changed. A failed `add_entry` doesn't leave a half-added entry behind.\n\nSome edge cases:\n\n- **Order of checks.** The amount is checked before the date. If both are bad, you get the \"amount must be positive\" error.\n- **Loose formatting is accepted.** The format check doesn't require zero-padding or exact widths, so `\"2024-1-5\"` works fine.\n- **Unusual digit characters.** Some Unicode characters, such as `\"\u00b2\"`, count as digits for Python's `isdigit()` but can't be converted by `int()`. These still raise a `ValueError`, just from `int()` instead of the \"bad date\" check.\n- **Non-string input.** Passing something that isn't a string (for example, `None` or a `date` object) raises an `AttributeError` from `.split`, not a `ValueError`.\n- **Bypassing the checks.** If you create an `Entry` directly instead of using `add_entry`, none of this checking happens.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "api_cost": 0.178002
  },
  {
   "run": "claude-opus-5-5__control__T3__r2__38091e",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "control",
   "rep": 2,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 1771,
   "input": 4,
   "cache_read": 25775,
   "cache_write": 10281,
   "warmup": null,
   "cache_write_1h": 10281,
   "cost_usd": 0.12283899999999999,
   "wall_s": 19.7,
   "turns": 2,
   "prompt_last": 20401,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 558,
   "commands": null,
   "answer": "## How monthly totals are computed\n\n`monthly_report(entries, year, month)` (`ledger/core.py:51`) works in two steps: it filters the entries, then summarizes them.\n\n1. **Filtering.** `entries_in_month` (`core.py:40`) keeps each entry whose `day.year` equals `year` and whose `day.month` equals `month`. These are plain integer comparisons on the stored `date` objects. Nothing checks the arguments, so `month=13` or `month=0` doesn't raise an error. It just matches no entries.\n\n2. **Summarizing.** The function returns a dict with three keys:\n   - **`count`**: how many entries are in that month.\n   - **`total`**: from `total()` (`core.py:31`), a plain `sum()` of the entries' `amount` values. Amounts are converted to `float` when added, so ordinary floating-point rounding can show up (for example, `0.1 + 0.2` gives `0.30000000000000004`). If no entries match, `sum` of an empty sequence returns the integer `0`, not `0.0`.\n   - **`by_category`**: from `sum_by_category` (`core.py:44`), which adds up amounts per category name in a dict, starting each category at `0.0`. `add_entry` lowercases categories when storing them, so \"Food\" and \"food\" land in the same bucket. An empty month gives `{}`.\n\nNotes and tags don't affect the report.\n\n## What happens with an invalid date string\n\nDates are only parsed in `add_entry`, which calls `parse_date` (`core.py:15`). An invalid date can fail in two ways, and both raise `ValueError`:\n\n1. **Wrong shape.** The string is split on `-`. If there aren't exactly three parts, or any part isn't all digits, it raises `ValueError(\"bad date: '...'\")`. That covers `\"2024/01/05\"`, `\"2024-01\"`, `\"2024-01-05T10:00\"`, a string with spaces around it, and a part that's empty (as in `\"2024--05\"`).\n\n2. **Right shape, impossible date.** If the string passes the shape check, the three parts are converted to integers and passed to `datetime.date(y, m, d)`. That raises its own `ValueError` for out-of-range values, such as `\"2024-02-30\"`, `\"2024-13-01\"` or year `0000`. The message comes from the standard library (for example, \"day is out of range for month\"), not the \"bad date\" message.\n\nWhat this means in practice:\n\n- **The ledger is left unchanged.** `parse_date` runs while the new `Entry` is being built, before `entries.append` is called. A bad date raises before anything is appended, so no half-built entry gets in.\n- **Nothing catches the error.** It goes straight back to whoever called `add_entry`.\n- **The amount is checked first.** If the amount is zero or negative and the date is also bad, you get the \"amount must be positive\" error instead.\n- **The format check is loose.** Parts only need to be digits, not a fixed width, so `\"2024-1-5\"` is accepted as January 5, 2024.\n- **A non-string date gives a different error.** Passing a number, for example, fails on `.split` with an `AttributeError`, not a `ValueError`.\n\nI didn't change any files.",
   "account_remark": true,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "api_cost": 0.122839,
   "light": true
  },
  {
   "run": "gpt-6-sol__caveman__T3__r1__e3ce20",
   "model": "gpt-6-sol",
   "host": "codex",
   "task": "T3",
   "skill": "caveman",
   "rep": 1,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 126,
   "input": 31252,
   "cache_read": 18944,
   "cache_write": 24292,
   "warmup": 11984,
   "cache_write_1h": null,
   "cost_usd": 0.053633,
   "wall_s": 16.0,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "`monthly_report` selects entries whose dates match given year and month. It returns entry count, sum of amounts, and totals grouped by category.\n\n`parse_date` expects numeric `YYYY-MM-DD` parts. Invalid text or impossible dates raise `ValueError`. `add_entry` then stops before appending entry.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "api_cost": 0.053633
  },
  {
   "run": "gpt-6-sol__caveman__T3__r2__87723d",
   "model": "gpt-6-sol",
   "host": "codex",
   "task": "T3",
   "skill": "caveman",
   "rep": 2,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 161,
   "input": 31251,
   "cache_read": 27008,
   "cache_write": 16227,
   "warmup": 11984,
   "cache_write_1h": null,
   "cost_usd": 0.039466,
   "wall_s": 10.4,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "`monthly_report` selects entries whose dates match given year and month. It returns entry count, sum of their amounts, and totals grouped by category. Empty month returns count `0`, total `0`, and empty category map.\n\n`parse_date` expects three numeric parts separated by hyphens. Invalid text raises `ValueError`; impossible dates also raise `ValueError` from `date()`. In `add_entry`, invalid date stops entry from being appended.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "api_cost": 0.039466
  },
  {
   "run": "gpt-6-sol__control__T3__r1__dc522d",
   "model": "gpt-6-sol",
   "host": "codex",
   "task": "T3",
   "skill": "control",
   "rep": 1,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 277,
   "input": 27910,
   "cache_read": 25344,
   "cache_write": 14550,
   "warmup": 11984,
   "cache_write_1h": null,
   "cost_usd": 0.036939,
   "wall_s": 13.6,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 79,
   "commands": 1,
   "answer": "`monthly_report(entries, year, month)` selects entries whose date has that year and month. It returns their count, the sum of their amounts, and a breakdown of amounts by category. If there are no matching entries, the count and total are zero and the category breakdown is empty.\n\nWhen `add_entry` receives an invalid date string, `parse_date` raises `ValueError`\u2014either for malformed text or for an impossible calendar date. The entry is not added.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "api_cost": 0.036939
  },
  {
   "run": "gpt-6-sol__control__T3__r2__db7400",
   "model": "gpt-6-sol",
   "host": "codex",
   "task": "T3",
   "skill": "control",
   "rep": 2,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 281,
   "input": 27913,
   "cache_read": 18944,
   "cache_write": 20953,
   "warmup": 11984,
   "cache_write_1h": null,
   "cost_usd": 0.048505,
   "wall_s": 11.1,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 81,
   "commands": 1,
   "answer": "`monthly_report(entries, year, month)` selects entries whose stored date has that year and month. It returns their count, the sum of their amounts, and totals grouped by category.\n\nWhen `add_entry` receives an invalid date string, `parse_date` raises `ValueError`, and the entry is not added. It rejects strings that do not have three numeric parts separated by hyphens; Python\u2019s `date` constructor rejects impossible dates such as February 30.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "api_cost": 0.048505
  },
  {
   "run": "gpt-6.1-sol__caveman__T1__r1__079fc4",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T1",
   "skill": "caveman",
   "rep": 1,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 555,
   "input": 94913,
   "cache_read": 88576,
   "cache_write": 18834,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.052076,
   "wall_s": 30.3,
   "turns": null,
   "prompt_last": null,
   "lines_added": 6,
   "lines_deleted": 0,
   "lines": 6,
   "files": 1,
   "thinking": 0,
   "commands": 9,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.052076
  },
  {
   "run": "gpt-6.1-sol__caveman__T1__r2__6cbc67",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T1",
   "skill": "caveman",
   "rep": 2,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 542,
   "input": 109616,
   "cache_read": 101120,
   "cache_write": 20993,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.057518,
   "wall_s": 27.5,
   "turns": null,
   "prompt_last": null,
   "lines_added": 6,
   "lines_deleted": 0,
   "lines": 6,
   "files": 1,
   "thinking": 0,
   "commands": 6,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.057518
  },
  {
   "run": "gpt-6.1-sol__caveman__T1__r3__8ed234",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T1",
   "skill": "caveman",
   "rep": 3,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 398,
   "input": 89607,
   "cache_read": 81664,
   "cache_write": 20440,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.053026,
   "wall_s": 21.0,
   "turns": null,
   "prompt_last": null,
   "lines_added": 6,
   "lines_deleted": 0,
   "lines": 6,
   "files": 1,
   "thinking": 0,
   "commands": 4,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.053026
  },
  {
   "run": "gpt-6.1-sol__control__T1__r1__ff72f9",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T1",
   "skill": "control",
   "rep": 1,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 569,
   "input": 93102,
   "cache_read": 87680,
   "cache_write": 17919,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.050296,
   "wall_s": 28.2,
   "turns": null,
   "prompt_last": null,
   "lines_added": 6,
   "lines_deleted": 0,
   "lines": 6,
   "files": 1,
   "thinking": 12,
   "commands": 5,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.050296
  },
  {
   "run": "gpt-6.1-sol__control__T1__r2__c7f0a2",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T1",
   "skill": "control",
   "rep": 2,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 706,
   "input": 76449,
   "cache_read": 71424,
   "cache_write": 17522,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.049246,
   "wall_s": 30.2,
   "turns": null,
   "prompt_last": null,
   "lines_added": 7,
   "lines_deleted": 1,
   "lines": 8,
   "files": 2,
   "thinking": 14,
   "commands": 8,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.049246
  },
  {
   "run": "gpt-6.1-sol__control__T1__r3__c3ff5e",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T1",
   "skill": "control",
   "rep": 3,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 669,
   "input": 76192,
   "cache_read": 71296,
   "cache_write": 17393,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.048606,
   "wall_s": 30.1,
   "turns": null,
   "prompt_last": null,
   "lines_added": 7,
   "lines_deleted": 1,
   "lines": 8,
   "files": 2,
   "thinking": 10,
   "commands": 6,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.048606
  },
  {
   "run": "gpt-6.1-sol__karpathy__T1__r1__6a23c8",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T1",
   "skill": "karpathy",
   "rep": 1,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 620,
   "input": 83395,
   "cache_read": 76416,
   "cache_write": 19476,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.052794,
   "wall_s": 28.2,
   "turns": null,
   "prompt_last": null,
   "lines_added": 5,
   "lines_deleted": 0,
   "lines": 5,
   "files": 1,
   "thinking": 11,
   "commands": 8,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.052794
  },
  {
   "run": "gpt-6.1-sol__karpathy__T1__r2__333f3f",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T1",
   "skill": "karpathy",
   "rep": 2,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 571,
   "input": 101059,
   "cache_read": 93696,
   "cache_write": 19860,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.0548,
   "wall_s": 27.9,
   "turns": null,
   "prompt_last": null,
   "lines_added": 5,
   "lines_deleted": 0,
   "lines": 5,
   "files": 1,
   "thinking": 11,
   "commands": 5,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.0548
  },
  {
   "run": "gpt-6.1-sol__karpathy__T1__r3__10aeb2",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T1",
   "skill": "karpathy",
   "rep": 3,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 556,
   "input": 100086,
   "cache_read": 93184,
   "cache_write": 19399,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.053676,
   "wall_s": 30.2,
   "turns": null,
   "prompt_last": null,
   "lines_added": 5,
   "lines_deleted": 0,
   "lines": 5,
   "files": 1,
   "thinking": 0,
   "commands": 6,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.053676
  },
  {
   "run": "gpt-6.1-sol__placebo__T1__r1__e3bbfa",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T1",
   "skill": "placebo",
   "rep": 1,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 634,
   "input": 96782,
   "cache_read": 90368,
   "cache_write": 18911,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.053199,
   "wall_s": 30.8,
   "turns": null,
   "prompt_last": null,
   "lines_added": 6,
   "lines_deleted": 0,
   "lines": 6,
   "files": 1,
   "thinking": 28,
   "commands": 8,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.053199
  },
  {
   "run": "gpt-6.1-sol__placebo__T1__r2__300555",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T1",
   "skill": "placebo",
   "rep": 2,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 561,
   "input": 80766,
   "cache_read": 74240,
   "cache_write": 19023,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.05108,
   "wall_s": 27.4,
   "turns": null,
   "prompt_last": null,
   "lines_added": 6,
   "lines_deleted": 0,
   "lines": 6,
   "files": 1,
   "thinking": 0,
   "commands": 11,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.05108
  },
  {
   "run": "gpt-6.1-sol__placebo__T1__r3__7ece92",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T1",
   "skill": "placebo",
   "rep": 3,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 443,
   "input": 81559,
   "cache_read": 75392,
   "cache_write": 18664,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.049297,
   "wall_s": 22.4,
   "turns": null,
   "prompt_last": null,
   "lines_added": 6,
   "lines_deleted": 0,
   "lines": 6,
   "files": 1,
   "thinking": 0,
   "commands": 5,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.049297
  },
  {
   "run": "gpt-6.1-sol__ponytail__T1__r1__51132d",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T1",
   "skill": "ponytail",
   "rep": 1,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 536,
   "input": 82036,
   "cache_read": 75776,
   "cache_write": 18757,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.050452,
   "wall_s": 24.7,
   "turns": null,
   "prompt_last": null,
   "lines_added": 5,
   "lines_deleted": 0,
   "lines": 5,
   "files": 1,
   "thinking": 22,
   "commands": 5,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.050452
  },
  {
   "run": "gpt-6.1-sol__ponytail__T1__r2__86eaa6",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T1",
   "skill": "ponytail",
   "rep": 2,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 658,
   "input": 81925,
   "cache_read": 75136,
   "cache_write": 19286,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.052666,
   "wall_s": 30.5,
   "turns": null,
   "prompt_last": null,
   "lines_added": 5,
   "lines_deleted": 0,
   "lines": 5,
   "files": 1,
   "thinking": 58,
   "commands": 9,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.052666
  },
  {
   "run": "gpt-6.1-sol__ponytail__T1__r3__46b96f",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T1",
   "skill": "ponytail",
   "rep": 3,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 615,
   "input": 99444,
   "cache_read": 92800,
   "cache_write": 19141,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.053712,
   "wall_s": 28.2,
   "turns": null,
   "prompt_last": null,
   "lines_added": 5,
   "lines_deleted": 0,
   "lines": 5,
   "files": 1,
   "thinking": 27,
   "commands": 6,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.053712
  },
  {
   "run": "gpt-6.1-sol__caveman__T2__r1__d3a0db",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T2",
   "skill": "caveman",
   "rep": 1,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 471,
   "input": 108588,
   "cache_read": 100352,
   "cache_write": 20733,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.056211,
   "wall_s": 29.3,
   "turns": null,
   "prompt_last": null,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 28,
   "commands": 6,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.056211
  },
  {
   "run": "gpt-6.1-sol__caveman__T2__r2__8f2007",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T2",
   "skill": "caveman",
   "rep": 2,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 486,
   "input": 128852,
   "cache_read": 120320,
   "cache_write": 21029,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.05895,
   "wall_s": 26.0,
   "turns": null,
   "prompt_last": null,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 19,
   "commands": 6,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.05895
  },
  {
   "run": "gpt-6.1-sol__caveman__T2__r3__c8917e",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T2",
   "skill": "caveman",
   "rep": 3,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 381,
   "input": 94812,
   "cache_read": 85248,
   "cache_write": 22061,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.056457,
   "wall_s": 20.2,
   "turns": null,
   "prompt_last": null,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 13,
   "commands": 5,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.056457
  },
  {
   "run": "gpt-6.1-sol__control__T2__r1__4e1056",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T2",
   "skill": "control",
   "rep": 1,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 498,
   "input": 91091,
   "cache_read": 83456,
   "cache_write": 20132,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.05359,
   "wall_s": 25.0,
   "turns": null,
   "prompt_last": null,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 21,
   "commands": 5,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.05359
  },
  {
   "run": "gpt-6.1-sol__control__T2__r2__aac3b9",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T2",
   "skill": "control",
   "rep": 2,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 537,
   "input": 75828,
   "cache_read": 71168,
   "cache_write": 17157,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.046801,
   "wall_s": 26.3,
   "turns": null,
   "prompt_last": null,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 39,
   "commands": 6,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.046801
  },
  {
   "run": "gpt-6.1-sol__control__T2__r3__75efec",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T2",
   "skill": "control",
   "rep": 3,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 652,
   "input": 92101,
   "cache_read": 86912,
   "cache_write": 17686,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.050583,
   "wall_s": 40.6,
   "turns": null,
   "prompt_last": null,
   "lines_added": 3,
   "lines_deleted": 2,
   "lines": 5,
   "files": 1,
   "thinking": 49,
   "commands": 6,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.050583
  },
  {
   "run": "gpt-6.1-sol__karpathy__T2__r1__fd6eb9",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T2",
   "skill": "karpathy",
   "rep": 1,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 580,
   "input": 100468,
   "cache_read": 93440,
   "cache_write": 19525,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.054194,
   "wall_s": 27.0,
   "turns": null,
   "prompt_last": null,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 13,
   "commands": 6,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.054194
  },
  {
   "run": "gpt-6.1-sol__karpathy__T2__r2__5d778d",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T2",
   "skill": "karpathy",
   "rep": 2,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 625,
   "input": 83044,
   "cache_read": 76160,
   "cache_write": 19381,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.052628,
   "wall_s": 30.1,
   "turns": null,
   "prompt_last": null,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 16,
   "commands": 9,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.052628
  },
  {
   "run": "gpt-6.1-sol__karpathy__T2__r3__f3cf22",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T2",
   "skill": "karpathy",
   "rep": 3,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 561,
   "input": 82717,
   "cache_read": 76032,
   "cache_write": 19182,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.051577,
   "wall_s": 25.0,
   "turns": null,
   "prompt_last": null,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 28,
   "commands": 6,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.051577
  },
  {
   "run": "gpt-6.1-sol__placebo__T2__r1__34f6e8",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T2",
   "skill": "placebo",
   "rep": 1,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 583,
   "input": 96100,
   "cache_read": 89984,
   "cache_write": 18613,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.052054,
   "wall_s": 28.4,
   "turns": null,
   "prompt_last": null,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 34,
   "commands": 8,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.052054
  },
  {
   "run": "gpt-6.1-sol__placebo__T2__r2__048216",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T2",
   "skill": "placebo",
   "rep": 2,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 630,
   "input": 96434,
   "cache_read": 89856,
   "cache_write": 19075,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.053436,
   "wall_s": 28.5,
   "turns": null,
   "prompt_last": null,
   "lines_added": 3,
   "lines_deleted": 2,
   "lines": 5,
   "files": 1,
   "thinking": 52,
   "commands": 7,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.053436
  },
  {
   "run": "gpt-6.1-sol__placebo__T2__r3__b5e85f",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T2",
   "skill": "placebo",
   "rep": 3,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 597,
   "input": 97489,
   "cache_read": 90752,
   "cache_write": 19234,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.053513,
   "wall_s": 27.5,
   "turns": null,
   "prompt_last": null,
   "lines_added": 3,
   "lines_deleted": 2,
   "lines": 5,
   "files": 1,
   "thinking": 59,
   "commands": 5,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.053513
  },
  {
   "run": "gpt-6.1-sol__ponytail__T2__r1__46a1b5",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T2",
   "skill": "ponytail",
   "rep": 1,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 542,
   "input": 82066,
   "cache_read": 75648,
   "cache_write": 18915,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.050815,
   "wall_s": 27.3,
   "turns": null,
   "prompt_last": null,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 77,
   "commands": 5,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.050815
  },
  {
   "run": "gpt-6.1-sol__ponytail__T2__r2__ed72b8",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T2",
   "skill": "ponytail",
   "rep": 2,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 555,
   "input": 99029,
   "cache_read": 92416,
   "cache_write": 19110,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.053012,
   "wall_s": 27.6,
   "turns": null,
   "prompt_last": null,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 59,
   "commands": 6,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.053012
  },
  {
   "run": "gpt-6.1-sol__ponytail__T2__r3__cc9edb",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T2",
   "skill": "ponytail",
   "rep": 3,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 604,
   "input": 82544,
   "cache_read": 76032,
   "cache_write": 19009,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.051661,
   "wall_s": 26.5,
   "turns": null,
   "prompt_last": null,
   "lines_added": 1,
   "lines_deleted": 1,
   "lines": 2,
   "files": 1,
   "thinking": 77,
   "commands": 7,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.051661
  },
  {
   "run": "gpt-6.1-sol__caveman__T3__r1__3dc0c3",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T3",
   "skill": "caveman",
   "rep": 1,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 173,
   "input": 32394,
   "cache_read": 15744,
   "cache_write": 29147,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.061598,
   "wall_s": 10.8,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "`monthly_report` selects entries matching requested year and month. It returns entry count, sum of amounts, and totals grouped by category. Empty month returns count `0`, total `0`, and `{}` for categories.\n\n`parse_date` splits text on `-` and requires three digit-only parts. Malformed strings raise `ValueError` with `bad date: <repr>`. Invalid calendar dates raise `ValueError` from `datetime.date`. Digit widths are not enforced.\n\nIn `add_entry`, date errors propagate; no entry gets appended. No files changed.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "api_cost": 0.061598
  },
  {
   "run": "gpt-6.1-sol__caveman__T3__r2__af9b46",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T3",
   "skill": "caveman",
   "rep": 2,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 186,
   "input": 32398,
   "cache_read": 28032,
   "cache_write": 16863,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.038389,
   "wall_s": 10.1,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "`monthly_report` selects entries matching requested year and month. It returns entry count, sum of amounts, and totals grouped by category. Empty month returns count `0`, total `0`, and empty category dictionary.\n\n`parse_date` requires three hyphen-separated digit-only parts, then constructs Python `date`. Wrong structure raises `ValueError(\"bad date: ...\")`. Impossible dates, such as February 30, raise `ValueError` from `date`. Fixed digit widths are not enforced.\n\n`add_entry` propagates date errors without appending an entry. No files changed.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "api_cost": 0.038389
  },
  {
   "run": "gpt-6.1-sol__caveman__T3__r3__09ad94",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T3",
   "skill": "caveman",
   "rep": 3,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 184,
   "input": 32416,
   "cache_read": 28032,
   "cache_write": 16881,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.038405,
   "wall_s": 11.3,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "`monthly_report` selects entries matching requested year and month. It returns entry count, sum of amounts, and totals grouped by category. Empty month returns count `0`, total `0`, and empty category dictionary.\n\n`parse_date` requires three digit-only parts separated by hyphens, then constructs Python `date`. Malformed strings raise `ValueError` with `bad date: ...`. Impossible dates, such as February 30, raise `ValueError` from `date`. Fixed digit widths aren\u2019t enforced.\n\nIn `add_entry`, invalid date stops operation before append. Entries remain unchanged. No files changed.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "api_cost": 0.038405
  },
  {
   "run": "gpt-6.1-sol__control__T3__r1__3e3546",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T3",
   "skill": "control",
   "rep": 1,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 232,
   "input": 29032,
   "cache_read": 26368,
   "cache_write": 15161,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.035279,
   "wall_s": 13.5,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "In `ledger/core.py`, `monthly_report` selects entries whose date matches the requested year and month. It returns their count, the sum of their amounts, and totals grouped by category. If there are no matching entries, it returns a count and total of zero and an empty category dictionary.\n\n`parse_date` splits a date string on hyphens and requires exactly three digit-only parts. Malformed strings raise `ValueError` with a `bad date` message. Those parts are converted to integers and passed to Python\u2019s `date` constructor, which also raises `ValueError` for impossible dates, such as February 30. Exact digit widths are not enforced, so `2026-9-2` is accepted.\n\nWhen `add_entry` encounters an invalid date, the error propagates and no entry is appended. No files were changed.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "api_cost": 0.035279
  },
  {
   "run": "gpt-6.1-sol__control__T3__r2__7a1b3d",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T3",
   "skill": "control",
   "rep": 2,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 302,
   "input": 43545,
   "cache_read": 40576,
   "cache_write": 15466,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.03801,
   "wall_s": 17.0,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 2,
   "answer": "In [ledger/core.py](/tmp/bench-work/gpt-6.1-sol__control__T3__r2__7a1b3d/ledger/core.py), `monthly_report` selects entries whose date matches the requested year and month. It returns their count, the sum of their amounts, and totals grouped by category. With no matching entries, these are `0`, `0`, and `{}`.\n\n`parse_date` splits a date string on hyphens and requires three digit-only parts. Malformed strings raise `ValueError` with a `bad date` message; impossible dates, such as February 30, raise `ValueError` from Python\u2019s `date` constructor. Despite the docstring\u2019s `YYYY-MM-DD` wording, fixed digit widths are not enforced.\n\nWhen `add_entry` encounters an invalid date, the exception propagates and no entry is appended. No files were changed.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "api_cost": 0.03801
  },
  {
   "run": "gpt-6.1-sol__control__T3__r3__995a6d",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T3",
   "skill": "control",
   "rep": 3,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 286,
   "input": 43522,
   "cache_read": 40576,
   "cache_write": 15443,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.037804,
   "wall_s": 16.2,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 2,
   "answer": "In [ledger/core.py](/tmp/bench-work/gpt-6.1-sol__control__T3__r3__995a6d/ledger/core.py), `monthly_report` selects entries whose dates match the requested year and month. It returns their count, the sum of their amounts, and totals grouped by category. With no matching entries, these are `0`, `0`, and `{}`.\n\n`parse_date` requires three hyphen-separated numeric parts, then passes them to `datetime.date`. Malformed strings raise `ValueError` with a `bad date` message; impossible dates, such as February 30, raise `ValueError` from `datetime.date`. It does not enforce exact `YYYY-MM-DD` digit widths. When called through `add_entry`, an invalid date prevents the entry from being appended.\n\nNo files were changed.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "api_cost": 0.037804
  },
  {
   "run": "gpt-6.1-sol__karpathy__T3__r1__17c810",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T3",
   "skill": "karpathy",
   "rep": 1,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 237,
   "input": 30124,
   "cache_read": 26880,
   "cache_write": 15741,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.03654,
   "wall_s": 12.3,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "In `ledger/core.py`, `monthly_report` selects entries whose dates match the requested year and month. It returns their count, the sum of their amounts, and a dictionary of totals grouped by category. If no entries match, these are `0`, `0`, and `{}`.\n\n`parse_date` splits a date string on hyphens and requires three digit-only parts. A malformed string raises `ValueError` with a message such as `bad date: 'invalid'`. Numeric parts are converted to integers and passed to Python\u2019s `date` constructor, which raises `ValueError` for impossible dates, such as February 30. Despite the docstring\u2019s `YYYY-MM-DD` wording, fixed digit widths aren\u2019t enforced.\n\nWhen `add_entry` encounters an invalid date, the exception propagates and no entry is appended. No files were changed.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "api_cost": 0.03654
  },
  {
   "run": "gpt-6.1-sol__karpathy__T3__r2__e5a294",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T3",
   "skill": "karpathy",
   "rep": 2,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 262,
   "input": 30129,
   "cache_read": 26880,
   "cache_write": 15746,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.0368,
   "wall_s": 12.8,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "In [ledger/core.py](/tmp/bench-work/gpt-6.1-sol__karpathy__T3__r2__e5a294/ledger/core.py), `monthly_report` selects entries whose dates match the requested year and month. It returns their count, the sum of their amounts, and totals grouped by category. With no matching entries, these are `0`, `0`, and an empty dictionary.\n\n`parse_date` splits the string on hyphens and requires exactly three digit-only parts. Otherwise, it raises `ValueError` with a message such as `bad date: 'invalid'`. It converts those parts to integers and constructs a Python `date`, which also raises `ValueError` for impossible dates, such as February 30. Component widths aren\u2019t enforced, so `2026-2-3` is accepted.\n\nWhen `add_entry` encounters an invalid date, the exception propagates and no entry is appended. No files were changed.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "api_cost": 0.0368
  },
  {
   "run": "gpt-6.1-sol__karpathy__T3__r3__018578",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T3",
   "skill": "karpathy",
   "rep": 3,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 253,
   "input": 30117,
   "cache_read": 26880,
   "cache_write": 15734,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.036686,
   "wall_s": 13.4,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "In [ledger/core.py](/tmp/bench-work/gpt-6.1-sol__karpathy__T3__r3__018578/ledger/core.py), `monthly_report` selects entries whose date matches the requested year and month. It returns their count, the sum of their amounts, and totals grouped by category. An empty month produces a count and total of zero and an empty category dictionary.\n\n`parse_date` splits a string on hyphens and requires exactly three digit-only parts. Malformed strings raise `ValueError` with a `bad date: ...` message. Numeric parts are passed to Python\u2019s `date` constructor, which raises `ValueError` for impossible dates, such as February 30. Although the docstring specifies `YYYY-MM-DD`, fixed digit widths are not enforced.\n\nWhen `add_entry` encounters an invalid date, the exception propagates and no entry is appended. No files were changed.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "api_cost": 0.036686
  },
  {
   "run": "gpt-6.1-sol__placebo__T3__r1__6d9f9d",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T3",
   "skill": "placebo",
   "rep": 1,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 216,
   "input": 29951,
   "cache_read": 26752,
   "cache_write": 15696,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.036227,
   "wall_s": 12.8,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "`monthly_report` selects entries whose dates match the requested year and month. It returns their count, the sum of their amounts, and totals grouped by category. An empty month returns a count and total of zero and an empty category dictionary.\n\n`parse_date` splits a string on hyphens and requires three digit-only parts. If that check fails, it raises `ValueError` with `bad date: <repr of input>`. Otherwise, it converts the parts to integers and constructs a `datetime.date`, which raises `ValueError` for impossible dates or out-of-range values. It does not enforce exact `YYYY-MM-DD` widths. When parsing fails in `add_entry`, the exception propagates and no entry is appended.\n\nNo files were changed.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "api_cost": 0.036227
  },
  {
   "run": "gpt-6.1-sol__placebo__T3__r2__c5386c",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T3",
   "skill": "placebo",
   "rep": 2,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 225,
   "input": 29970,
   "cache_read": 26880,
   "cache_write": 15587,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.036112,
   "wall_s": 13.8,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "`monthly_report` selects entries whose dates match the requested year and month. It returns their count, the sum of their amounts, and totals grouped by category. If no entries match, the count and total are zero, and the category totals are an empty dictionary.\n\n`parse_date` requires three hyphen-separated digit strings, converts them to integers, and constructs a Python `date`. Malformed strings raise `ValueError` with a message such as `bad date: 'invalid'`. Impossible dates, such as February 30, raise `ValueError` from the `date` constructor. Despite the docstring\u2019s `YYYY-MM-DD` wording, fixed digit widths are not enforced.\n\nWhen `add_entry` encounters an invalid date, the exception propagates and no entry is appended. No files were changed.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "api_cost": 0.036112
  },
  {
   "run": "gpt-6.1-sol__placebo__T3__r3__0c9f5a",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T3",
   "skill": "placebo",
   "rep": 3,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 226,
   "input": 29948,
   "cache_read": 26752,
   "cache_write": 15693,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.036321,
   "wall_s": 11.6,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "`monthly_report` selects entries whose date matches the requested year and month. It returns their count, the sum of their amounts, and a dictionary of totals grouped by category. If no entries match, the count and total are zero and the category dictionary is empty.\n\n`parse_date` splits the string on hyphens and requires three digit-only parts. Otherwise, it raises `ValueError` with `bad date: <repr of input>`. It converts those parts to integers and calls Python\u2019s `date` constructor, which raises `ValueError` for impossible dates or out-of-range values. Despite the docstring\u2019s `YYYY-MM-DD` wording, fixed digit widths are not enforced.\n\nWhen `add_entry` encounters an invalid date, the exception propagates and no entry is appended. No files were changed.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "api_cost": 0.036321
  },
  {
   "run": "gpt-6.1-sol__ponytail__T3__r1__1039a7",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T3",
   "skill": "ponytail",
   "rep": 1,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 223,
   "input": 30228,
   "cache_read": 27008,
   "cache_write": 15717,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.036365,
   "wall_s": 12.5,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "In `ledger/core.py`, `monthly_report` selects entries whose date matches the requested year and month. It returns their count, the sum of their amounts, and totals grouped by category. An empty month produces a count and total of zero and an empty category dictionary.\n\n`parse_date` splits the date string on hyphens and requires three digit-only parts. Malformed input raises `ValueError` with a `bad date` message. The parts then become integers passed to Python\u2019s `date` constructor, which raises `ValueError` for impossible dates, such as February 30. Despite the docstring\u2019s `YYYY-MM-DD` wording, fixed-width components are not enforced.\n\nWhen `add_entry` encounters an invalid date, the exception propagates and no entry is appended. No files were changed.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "api_cost": 0.036365
  },
  {
   "run": "gpt-6.1-sol__ponytail__T3__r2__d05495",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T3",
   "skill": "ponytail",
   "rep": 2,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 211,
   "input": 30222,
   "cache_read": 27008,
   "cache_write": 15711,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.036233,
   "wall_s": 11.3,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "`monthly_report` selects entries whose date matches the requested year and month. It returns their count, the sum of their amounts, and totals grouped by category. An empty month returns a count and total of `0`, with an empty category dictionary.\n\n`parse_date` requires three hyphen-separated digit-only parts, then passes their integer values to Python\u2019s `date` constructor. Malformed strings raise `ValueError` with a `bad date` message; impossible dates, such as February 30, raise `ValueError` from the constructor. Despite the docstring, exact `YYYY-MM-DD` widths aren\u2019t enforced.\n\nWhen `add_entry` encounters an invalid date, the exception propagates and no entry is appended. No files were changed.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "api_cost": 0.036233
  },
  {
   "run": "gpt-6.1-sol__ponytail__T3__r3__ceb680",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T3",
   "skill": "ponytail",
   "rep": 3,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 211,
   "input": 30248,
   "cache_read": 27008,
   "cache_write": 15737,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.036285,
   "wall_s": 11.1,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "`monthly_report` selects entries whose date matches the requested year and month. It returns their count, the sum of their amounts, and totals grouped by category. With no matching entries, it returns a count of `0`, a total of `0`, and an empty category dictionary.\n\n`parse_date` splits the string on hyphens and requires three digit-only parts. Malformed strings raise `ValueError` with a `bad date` message; impossible calendar dates raise `ValueError` from Python\u2019s `date` constructor. Despite the docstring\u2019s `YYYY-MM-DD` wording, fixed-width parts aren\u2019t enforced.\n\nWhen `add_entry` encounters an invalid date, the exception propagates and no entry is appended. No files were changed.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "api_cost": 0.036285
  },
  {
   "run": "gpt-6.1-sol__caveman__T4__r1__b2ac80",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T4",
   "skill": "caveman",
   "rep": 1,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 896,
   "input": 98337,
   "cache_read": 86912,
   "cache_write": 23922,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.065495,
   "wall_s": 35.9,
   "turns": null,
   "prompt_last": null,
   "lines_added": 51,
   "lines_deleted": 0,
   "lines": 51,
   "files": 1,
   "thinking": 0,
   "commands": 5,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.065495
  },
  {
   "run": "gpt-6.1-sol__caveman__T4__r2__a9a43a",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T4",
   "skill": "caveman",
   "rep": 2,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 1082,
   "input": 112440,
   "cache_read": 102144,
   "cache_write": 22793,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.06662,
   "wall_s": 44.3,
   "turns": null,
   "prompt_last": null,
   "lines_added": 60,
   "lines_deleted": 0,
   "lines": 60,
   "files": 1,
   "thinking": 44,
   "commands": 7,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.06662
  },
  {
   "run": "gpt-6.1-sol__caveman__T4__r3__5b07af",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T4",
   "skill": "caveman",
   "rep": 3,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 936,
   "input": 119456,
   "cache_read": 107904,
   "cache_write": 24049,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.068248,
   "wall_s": 39.0,
   "turns": null,
   "prompt_last": null,
   "lines_added": 57,
   "lines_deleted": 0,
   "lines": 57,
   "files": 1,
   "thinking": 0,
   "commands": 7,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.068248
  },
  {
   "run": "gpt-6.1-sol__control__T4__r1__2afbad",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T4",
   "skill": "control",
   "rep": 1,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 1026,
   "input": 94856,
   "cache_read": 88192,
   "cache_write": 19161,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.057401,
   "wall_s": 40.7,
   "turns": null,
   "prompt_last": null,
   "lines_added": 50,
   "lines_deleted": 0,
   "lines": 50,
   "files": 1,
   "thinking": 19,
   "commands": 7,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.057401
  },
  {
   "run": "gpt-6.1-sol__control__T4__r2__3e0313",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T4",
   "skill": "control",
   "rep": 2,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 1077,
   "input": 95504,
   "cache_read": 89472,
   "cache_write": 18529,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.056775,
   "wall_s": 42.1,
   "turns": null,
   "prompt_last": null,
   "lines_added": 54,
   "lines_deleted": 0,
   "lines": 54,
   "files": 1,
   "thinking": 18,
   "commands": 5,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.056775
  },
  {
   "run": "gpt-6.1-sol__control__T4__r3__d879d9",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T4",
   "skill": "control",
   "rep": 3,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 978,
   "input": 79729,
   "cache_read": 73344,
   "cache_write": 18882,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.054878,
   "wall_s": 40.2,
   "turns": null,
   "prompt_last": null,
   "lines_added": 50,
   "lines_deleted": 0,
   "lines": 50,
   "files": 1,
   "thinking": 12,
   "commands": 6,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.054878
  },
  {
   "run": "gpt-6.1-sol__karpathy__T4__r1__51edef",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T4",
   "skill": "karpathy",
   "rep": 1,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 1035,
   "input": 84859,
   "cache_read": 74624,
   "cache_write": 22732,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.063276,
   "wall_s": 43.6,
   "turns": null,
   "prompt_last": null,
   "lines_added": 45,
   "lines_deleted": 0,
   "lines": 45,
   "files": 1,
   "thinking": 131,
   "commands": 6,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.063276
  },
  {
   "run": "gpt-6.1-sol__karpathy__T4__r2__340bbf",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T4",
   "skill": "karpathy",
   "rep": 2,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 1060,
   "input": 85048,
   "cache_read": 74624,
   "cache_write": 22921,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.063904,
   "wall_s": 45.0,
   "turns": null,
   "prompt_last": null,
   "lines_added": 47,
   "lines_deleted": 0,
   "lines": 47,
   "files": 1,
   "thinking": 86,
   "commands": 7,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.063904
  },
  {
   "run": "gpt-6.1-sol__karpathy__T4__r3__42329c",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T4",
   "skill": "karpathy",
   "rep": 3,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 1042,
   "input": 85639,
   "cache_read": 77312,
   "cache_write": 20824,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.059799,
   "wall_s": 42.2,
   "turns": null,
   "prompt_last": null,
   "lines_added": 46,
   "lines_deleted": 0,
   "lines": 46,
   "files": 1,
   "thinking": 62,
   "commands": 8,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.059799
  },
  {
   "run": "gpt-6.1-sol__placebo__T4__r1__93a359",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T4",
   "skill": "placebo",
   "rep": 1,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 1168,
   "input": 102503,
   "cache_read": 94208,
   "cache_write": 20792,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.062685,
   "wall_s": 44.4,
   "turns": null,
   "prompt_last": null,
   "lines_added": 66,
   "lines_deleted": 0,
   "lines": 66,
   "files": 1,
   "thinking": 31,
   "commands": 6,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.062685
  },
  {
   "run": "gpt-6.1-sol__placebo__T4__r2__e29288",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T4",
   "skill": "placebo",
   "rep": 2,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 1144,
   "input": 101878,
   "cache_read": 87936,
   "cache_write": 26439,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.073112,
   "wall_s": 44.2,
   "turns": null,
   "prompt_last": null,
   "lines_added": 66,
   "lines_deleted": 0,
   "lines": 66,
   "files": 1,
   "thinking": 12,
   "commands": 6,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.073112
  },
  {
   "run": "gpt-6.1-sol__placebo__T4__r3__c06e6f",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T4",
   "skill": "placebo",
   "rep": 3,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 1318,
   "input": 101461,
   "cache_read": 93184,
   "cache_write": 20774,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.064046,
   "wall_s": 51.6,
   "turns": null,
   "prompt_last": null,
   "lines_added": 67,
   "lines_deleted": 0,
   "lines": 67,
   "files": 1,
   "thinking": 36,
   "commands": 6,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.064046
  },
  {
   "run": "gpt-6.1-sol__ponytail__T4__r1__338139",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T4",
   "skill": "ponytail",
   "rep": 1,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 1044,
   "input": 84678,
   "cache_read": 76416,
   "cache_write": 20759,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.0596,
   "wall_s": 40.1,
   "turns": null,
   "prompt_last": null,
   "lines_added": 43,
   "lines_deleted": 0,
   "lines": 43,
   "files": 1,
   "thinking": 109,
   "commands": 8,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.0596
  },
  {
   "run": "gpt-6.1-sol__ponytail__T4__r2__45ab06",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T4",
   "skill": "ponytail",
   "rep": 2,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 1124,
   "input": 102625,
   "cache_read": 92032,
   "cache_write": 23090,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.066623,
   "wall_s": 44.4,
   "turns": null,
   "prompt_last": null,
   "lines_added": 41,
   "lines_deleted": 0,
   "lines": 41,
   "files": 1,
   "thinking": 212,
   "commands": 6,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.066623
  },
  {
   "run": "gpt-6.1-sol__ponytail__T4__r3__e2f54c",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T4",
   "skill": "ponytail",
   "rep": 3,
   "round": 10,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 1180,
   "input": 86560,
   "cache_read": 77824,
   "cache_write": 21233,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.062048,
   "wall_s": 46.4,
   "turns": null,
   "prompt_last": null,
   "lines_added": 43,
   "lines_deleted": 0,
   "lines": 43,
   "files": 1,
   "thinking": 265,
   "commands": 7,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "heldout": true,
   "api_cost": 0.062048
  }
 ],
 "round13_runs": [
  {
   "run": "claude-fable-5-1__control__T3__r1__99d099",
   "model": "claude-fable-5-1",
   "host": "claude",
   "task": "T3",
   "skill": "control",
   "rep": 1,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 1155,
   "input": 4,
   "cache_read": 27827,
   "cache_write": 7373,
   "warmup": null,
   "cache_write_1h": 7373,
   "cost_usd": 0.21220675,
   "wall_s": 12.0,
   "turns": 2,
   "prompt_last": 18002,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 234,
   "commands": null,
   "answer": "**Monthly totals**\n\n`monthly_report(entries, year, month)` at `ledger/core.py:51` computes them in two steps:\n\n1. **Filter:** `entries_in_month` keeps only entries whose `day.year` and `day.month` both equal the requested year and month.\n2. **Aggregate:** the report is a dict built from that filtered list, with three keys:\n   - `count` is the number of matching entries.\n   - `total` is a plain `sum` of their `amount` values.\n   - `by_category` is a dict of per-category running sums, starting each category at `0.0`.\n\nCategories are lowercased when entries are added, so \"Food\" and \"food\" land in the same bucket. Amounts are floats in dollars, so totals can carry ordinary floating-point rounding error (e.g. `0.1 + 0.2`). A month with no entries gives `{\"count\": 0, \"total\": 0, \"by_category\": {}}`, where the total is the integer `0` rather than `0.0`.\n\n**Invalid date strings**\n\n`parse_date` at `ledger/core.py:15` raises `ValueError` for an invalid date string, by one of two paths:\n\n- **Wrong shape:** if splitting on `-` does not give exactly three parts, or any part is not all digits, it raises `ValueError(\"bad date: '<text>'\")`. This covers inputs like `\"2024/01/05\"`, `\"2024-01\"`, `\"abc\"`, `\"\"`, or a leading minus sign.\n- **Right shape, impossible date:** inputs like `\"2024-02-30\"` or `\"2024-13-01\"` pass the shape check, and the `date(y, m, d)` constructor then raises its own `ValueError` (e.g. \"day is out of range for month\"), not the \"bad date\" message.\n\nIn `add_entry`, the date is parsed before the `Entry` is built and appended, so a bad date propagates the `ValueError` to the caller and leaves the list unchanged. The amount check runs first, so a non-positive amount combined with a bad date reports \"amount must be positive\".\n\nThree edge cases fall outside what the docstring promises:\n\n- **Loose format:** the check does not enforce digit widths, so `\"2024-1-5\"` is accepted as 5 January 2024.\n- **Non-strings:** passing `None` or another non-string raises `AttributeError` or `TypeError`, not `ValueError`.\n- **Unicode digits:** characters like `\"\u00b2\"` pass `isdigit()` but fail in `int()`, which still raises `ValueError`, though with a different message.\n\nNo files were changed.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": true,
   "api_cost": 0.212207
  },
  {
   "run": "claude-fable-5-1__control__T3__r2__cb59e9",
   "model": "claude-fable-5-1",
   "host": "claude",
   "task": "T3",
   "skill": "control",
   "rep": 2,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 1142,
   "input": 4,
   "cache_read": 27832,
   "cache_write": 7380,
   "warmup": null,
   "cache_write_1h": 7380,
   "cost_usd": 0.21169799999999997,
   "wall_s": 13.3,
   "turns": 2,
   "prompt_last": 18009,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 247,
   "commands": null,
   "answer": "**Monthly totals**\n\n`monthly_report(entries, year, month)` in `ledger/core.py:51` computes them in three steps:\n\n1. **Filter:** `entries_in_month` keeps only the entries whose `day.year` and `day.month` both match the requested year and month.\n2. **Sum:** `total` adds up the `amount` of those entries with a plain `sum()`.\n3. **Break down:** `sum_by_category` builds a dict of per-category sums for the same entries.\n\nThe result is a dict with `count` (number of matching entries), `total` and `by_category`.\n\n- **Empty month:** a month with no entries gives `count` 0, `total` 0 (the integer, since `sum()` of nothing is `0`) and an empty `by_category`.\n- **Float arithmetic:** amounts are stored as floats, so totals can carry the usual rounding artefacts (e.g. 0.1 + 0.2); nothing is rounded.\n- **Category case:** categories are lowercased when an entry is added, so \"Food\" and \"food\" land in the same bucket.\n\n**Invalid date strings**\n\nDates are parsed only when an entry is added: `add_entry` calls `parse_date` (`ledger/core.py:15`), which splits the string on `-`.\n\n- **Wrong shape:** if there are not exactly three parts, or any part is not all digits, it raises `ValueError(\"bad date: '<text>'\")`. This covers `\"2024/01/05\"`, `\"2024-01\"`, `\"abc\"` and an empty string.\n- **Right shape, impossible date:** something like `\"2024-13-01\"` or `\"2024-02-30\"` passes the shape check, but `date(y, m, d)` then raises its own `ValueError` (e.g. \"month must be in 1..12\"). The exception type is the same; the message is not the \"bad date\" one.\n- **Loose formats accepted:** digit widths are not checked, so `\"2024-1-5\"` parses as 5 January 2024 despite the docstring saying YYYY-MM-DD.\n- **Non-strings:** passing `None` or a `date` object raises `AttributeError` (no `.split`), not `ValueError`.\n\nIn every failure case the exception propagates out of `add_entry` uncaught. The `Entry` is constructed before the append, so nothing is added and the list is left unchanged. The amount check runs first, so a non-positive amount raises \"amount must be positive\" before the date is ever looked at.\n\nNo files were changed.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": true,
   "api_cost": 0.211698
  },
  {
   "run": "claude-opus-5-5__caveman__T3__r1__ee998f",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "caveman",
   "rep": 1,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 990,
   "input": 4,
   "cache_read": 29740,
   "cache_write": 9289,
   "warmup": null,
   "cache_write_1h": 9289,
   "cost_usd": 0.100076,
   "wall_s": 12.1,
   "turns": 2,
   "prompt_last": 19918,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 183,
   "commands": null,
   "answer": "**Monthly totals**\n\n`monthly_report(entries, year, month)` at `ledger/core.py:51` does three steps:\n\n1. `entries_in_month` keeps entries where `e.day.year == year` and `e.day.month == month`.\n2. `total` sums `amount` over those entries.\n3. `sum_by_category` builds a dict of category to summed amount.\n\nIt returns `{\"count\", \"total\", \"by_category\"}`.\n\n- **Empty month:** no match gives `count` 0, `total` 0 (int, not float) and `by_category` `{}`.\n- **Float amounts:** amounts are floats with no rounding, so sums can drift (`0.1 + 0.2`).\n- **Categories:** `add_entry` lowercases the category, so \"Food\" and \"food\" merge.\n- **No validation:** `year` and `month` are not checked. Month 13 gives an empty report, not an error.\n\n**Invalid date string**\n\nDates are parsed only in `add_entry`, through `parse_date` (`ledger/core.py:15`). Two failure paths, both `ValueError`:\n\n- **Wrong shape:** not exactly 3 parts split by `-`, or any part not all digits. Raises `ValueError(\"bad date: '<text>'\")`. Examples: `\"2024/01/05\"`, `\"2024-01\"`, `\"2024-Jan-05\"`, `\"\"`.\n- **Right shape, impossible date:** `date(y, m, d)` raises its own `ValueError`, for example `month must be in 1..12`. Examples: `\"2024-13-01\"`, `\"2024-02-30\"`.\n\nIn `add_entry`, the amount check runs first. With a bad amount and a bad date, the error is `amount must be positive`. The date parse runs before `entries.append`, so a bad date leaves the list unchanged. Nothing catches the exception; it propagates to the caller.\n\nTwo gaps:\n\n- **Loose format:** the docstring says YYYY-MM-DD, but digit widths are not checked. `\"2024-1-5\"` parses fine.\n- **Non-string input:** `None` or a `date` object raises `AttributeError` on `.split`, not `ValueError`.\n\nNo files changed.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.100076
  },
  {
   "run": "claude-opus-5-5__caveman__T3__r2__72b00b",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "caveman",
   "rep": 2,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 880,
   "input": 4,
   "cache_read": 29739,
   "cache_write": 9288,
   "warmup": null,
   "cache_write_1h": 9288,
   "cost_usd": 0.09786779999999999,
   "wall_s": 11.9,
   "turns": 2,
   "prompt_last": 19917,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 202,
   "commands": null,
   "answer": "**Monthly totals**\n\n`monthly_report(entries, year, month)` at `ledger/core.py:51` does three steps:\n\n1. `entries_in_month` keeps entries where `e.day.year == year` and `e.day.month == month`.\n2. `total` sums `amount` over those entries with plain `sum`.\n3. `sum_by_category` builds a dict of category to running sum, starting each category at `0.0`.\n\nReturns `{\"count\", \"total\", \"by_category\"}`. Empty month gives count `0`, total `0` (int, not `0.0`), and `{}`.\n\nAmounts are floats, no rounding anywhere. Totals can show float error such as `0.30000000000000004`.\n\nCategories are lowercased in `add_entry`, so \"Food\" and \"food\" merge into one key.\n\n**Invalid date string**\n\n`parse_date` at `ledger/core.py:15` splits on `-`. It raises `ValueError(\"bad date: '<text>'\")` unless there are exactly 3 parts and all are digits.\n\nTwo cases get past that check:\n\n- Out-of-range values such as `2024-13-01` or `2024-02-30`. `date(y, m, d)` raises `ValueError`, but with Python's own message, not `bad date: ...`.\n- Unpadded forms such as `2024-1-5`. These parse without error, though the docstring says YYYY-MM-DD only.\n\nNon-string input (`None`, a `date` object) raises `AttributeError` on `.split`, not `ValueError`.\n\nIn `add_entry`, the amount check runs first. With amount `<= 0` and a bad date, the error is `amount must be positive`. The date parse happens before `entries.append`, so a bad date leaves the list unchanged. Nothing catches the error; it propagates to the caller.\n\nNo files changed.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.097868
  },
  {
   "run": "claude-opus-5-5__caveman__T3__r3__8b785c",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "caveman",
   "rep": 3,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 907,
   "input": 4,
   "cache_read": 29738,
   "cache_write": 9287,
   "warmup": null,
   "cache_write_1h": 9287,
   "cost_usd": 0.0983996,
   "wall_s": 10.1,
   "turns": 2,
   "prompt_last": 19916,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 179,
   "commands": null,
   "answer": "**Monthly totals**\n\n`monthly_report(entries, year, month)` at `ledger/core.py:51` does three steps:\n\n1. `entries_in_month` keeps entries where `e.day.year == year` and `e.day.month == month`.\n2. `total` sums `amount` over those entries.\n3. `sum_by_category` builds a dict of category to summed amount.\n\nIt returns `{\"count\", \"total\", \"by_category\"}`.\n\n- **Category case:** `add_entry` lowercases the category, so \"Food\" and \"food\" merge into one key.\n- **Float amounts:** amounts are floats, so sums can carry rounding error (`0.1 + 0.2`).\n- **Empty month:** count is `0`, total is `0` (int, not `0.0`), and `by_category` is `{}`.\n- **Invalid month:** no check on `year` or `month`. Month 13 returns an empty report, no error.\n\n**Invalid date string**\n\n`parse_date` at `ledger/core.py:15` splits the text on `-`. It needs exactly 3 parts, all digits. Otherwise it raises `ValueError(\"bad date: '<text>'\")`.\n\n- **Out of range:** `2024-13-01` or `2024-02-30` passes the format check. `date(y, m, d)` then raises its own `ValueError`, with a different message (for example `month must be in 1..12`).\n- **No padding check:** `2024-1-5` is accepted, and so is `24-01-05` (year 24).\n- **Non-string input:** `None` or a `date` object raises `AttributeError`, not `ValueError`, because there is no `.split`.\n- **Surrounding whitespace:** `\" 2024-01-05\"` fails `isdigit` and raises the `bad date` error.\n\nIn `add_entry`, the amount check runs first. A bad amount with a bad date reports `amount must be positive`. With a valid amount, the date error propagates before `append`, so the list stays unchanged.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.0984
  },
  {
   "run": "claude-opus-5-5__control__T3__r1__e5c415",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "control",
   "rep": 1,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 1176,
   "input": 4,
   "cache_read": 27094,
   "cache_write": 6641,
   "warmup": null,
   "cache_write_1h": 6641,
   "cost_usd": 0.08208279999999998,
   "wall_s": 14.0,
   "turns": 2,
   "prompt_last": 17270,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 220,
   "commands": null,
   "answer": "**Monthly totals**\n\n`monthly_report(entries, year, month)` (`ledger/core.py:51`) computes everything on demand from the entries list; nothing is cached or pre-aggregated.\n\n- **Filtering:** `entries_in_month` keeps only entries whose `day.year` and `day.month` both equal the requested values.\n- **Aggregating:** the filtered list feeds three results, returned as a dict:\n  - `count` is the number of matching entries.\n  - `total` is a plain `sum` of their `amount` values.\n  - `by_category` is a dict built by `sum_by_category`, accumulating amounts per category name.\n- **Empty month:** a month with no entries gives a count of 0, a total of integer `0` (not `0.0`), and an empty `by_category` dict.\n- **Float arithmetic:** amounts are dollar floats, so totals can show ordinary rounding artefacts (e.g. `0.1 + 0.2`), and nothing rounds them.\n- **Category case:** categories are lowercased in `add_entry`, so \"Food\" and \"food\" land in the same bucket.\n\n**Invalid date strings**\n\nDates are parsed only in `add_entry`, via `parse_date` (`ledger/core.py:15`), and an invalid one raises `ValueError`. There are two ways this happens:\n\n- **Wrong shape:** if the string does not split on `-` into exactly three all-digit parts, it raises `ValueError(\"bad date: '<text>'\")`. This covers things like `\"2024/01/05\"`, `\"2024-01\"`, `\"2024-xx-01\"` and the empty string.\n- **Right shape, impossible date:** something like `\"2024-02-30\"` or `\"2024-13-01\"` passes the shape check but fails inside `date(y, m, d)`. It is still a `ValueError`, but with Python's own message (e.g. \"day is out of range for month\") rather than the \"bad date\" one.\n\nIn both cases the exception propagates out of `add_entry` uncaught. The `Entry` is never constructed or appended, so the list is left unchanged; there is no skip-and-continue or fallback.\n\nSome related behaviours are easy to miss:\n\n- **Amount is checked first:** a non-positive amount together with a bad date reports \"amount must be positive\", not the date error.\n- **The format check is looser than the docstring:** it does not enforce digit widths, so `\"2024-1-5\"` is accepted, as is `\"24-01-05\"` (parsed as year 24).\n- **Non-strings are not a `ValueError`:** passing `None` or a `date` object fails with `AttributeError` on `.split`.\n\nNo files were changed.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": true,
   "api_cost": 0.082083
  },
  {
   "run": "claude-opus-5-5__control__T3__r1__bc721d",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "control",
   "rep": 1,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 1066,
   "input": 4,
   "cache_read": 27094,
   "cache_write": 6641,
   "warmup": null,
   "cache_write_1h": 6641,
   "cost_usd": 0.0798828,
   "wall_s": 12.8,
   "turns": 2,
   "prompt_last": 17270,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 179,
   "commands": null,
   "answer": "**Monthly totals**\n\n`monthly_report(entries, year, month)` in `ledger/core.py:51` does the work in three steps:\n\n1. **Filter:** `entries_in_month` keeps only the entries whose `day.year` and `day.month` both equal the requested values.\n2. **Overall total:** `total` sums the `amount` of the filtered entries.\n3. **Per-category totals:** `sum_by_category` builds a dict keyed by category, adding each amount to a running sum that starts at `0.0`.\n\nThe result is a dict with `count` (number of entries in the month), `total` and `by_category`.\n\n- **Float arithmetic:** amounts are stored as floats in dollars, so totals can show ordinary float rounding artefacts (e.g. `0.1 + 0.2`); nothing is rounded.\n- **Case-insensitive categories:** `add_entry` lowercases the category, so \"Food\" and \"food\" land in the same bucket.\n- **Empty month:** a month with no entries gives `count` 0, `total` as the integer `0` (not `0.0`) and an empty `by_category`.\n\n**Invalid date strings**\n\nDates are parsed by `parse_date` (`ledger/core.py:15`), which `add_entry` calls while building the `Entry`. An invalid date raises `ValueError`, and because that happens before `entries.append` completes, nothing is added to the ledger.\n\n- **Wrong shape:** if the string does not split on `-` into exactly three all-digit parts (e.g. `\"2024/01/05\"`, `\"2024-01\"`, `\"2024-ab-01\"`, `\"\"`), it raises `ValueError(\"bad date: '...'\")`.\n- **Right shape, impossible date:** something like `\"2024-13-01\"` or `\"2024-02-30\"` passes the shape check, and `date(y, m, d)` then raises its own `ValueError` with a different message (e.g. \"month must be in 1..12\").\n- **Non-strings:** passing `None` or an actual `date` object raises `AttributeError` (no `.split`), not `ValueError`, despite what the docstring says.\n- **Looser than the docstring:** digit widths are not checked, so `\"2024-1-5\"` is accepted as 5 January 2024.\n- **Order of checks:** `add_entry` validates the amount first, so a non-positive amount with a bad date reports \"amount must be positive\" and the date is never looked at.\n\nNo files were changed.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.079883
  },
  {
   "run": "claude-opus-5-5__control__T3__r2__7e6eb9",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "control",
   "rep": 2,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 1243,
   "input": 4,
   "cache_read": 20236,
   "cache_write": 11891,
   "warmup": null,
   "cache_write_1h": 11891,
   "cost_usd": 0.1240512,
   "wall_s": 15.5,
   "turns": 2,
   "prompt_last": 16467,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 210,
   "commands": null,
   "answer": "**Monthly totals**\n\n`monthly_report(entries, year, month)` in `ledger/core.py:51` computes the totals in three steps:\n\n1. **Filter:** `entries_in_month` keeps only the entries whose `day.year` and `day.month` both match the requested year and month.\n2. **Overall total:** `total` sums the `amount` of the filtered entries.\n3. **Per-category totals:** `sum_by_category` builds a dict from category to the running sum of amounts in that category.\n\nThe report is a dict with `count` (number of matching entries), `total` and `by_category`.\n\nA few behaviours follow from how this is written:\n\n- **Case-insensitive categories:** `add_entry` lowercases the category on the way in, so \"Food\" and \"food\" land in the same bucket.\n- **Float arithmetic:** amounts are stored as floats in dollars, so totals can carry small rounding artefacts (such as `0.1 + 0.2`), and nothing rounds them.\n- **Empty months:** a month with no entries gives `count: 0`, `total: 0` (the integer, not `0.0`) and an empty `by_category`.\n- **No month validation:** asking for month 13 simply returns the empty report rather than an error.\n\n**Invalid date strings**\n\nDates are parsed by `parse_date` (`ledger/core.py:15`), which `add_entry` calls. It splits the text on `-` and handles bad input in two ways:\n\n- **Wrong shape:** if there are not exactly three parts, or any part is not all digits, it raises `ValueError(\"bad date: '<text>'\")`. This covers `\"2024/01/05\"`, `\"2024-01\"`, `\"2024-Jan-05\"`, an empty string, and anything with a minus sign or whitespace.\n- **Right shape, impossible date:** input such as `\"2024-02-30\"` or `\"2024-13-01\"` passes the shape check but fails inside the `date(y, m, d)` constructor. This is still a `ValueError`, but with Python's own message (e.g. \"day is out of range for month\") rather than the \"bad date\" one.\n\nIn both cases the exception propagates out of `add_entry` uncaught. Parsing happens before the `append`, so the entries list is left unchanged and no partial entry is added.\n\nThree edge cases are worth knowing:\n\n- **Looser than the docstring:** the check only requires digits, not field widths, so `\"2024-1-5\"` or `\"24-01-05\"` (year 24 AD) are accepted even though the docstring says YYYY-MM-DD.\n- **Amount checked first:** `add_entry` validates the amount before the date, so a call with both a non-positive amount and a bad date reports \"amount must be positive\" and never reaches the date error.\n- **Non-string input:** passing `None` or a `date` object raises `AttributeError` (no `.split`), not `ValueError`.\n\nNo files were changed.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": true,
   "api_cost": 0.124051
  },
  {
   "run": "claude-opus-5-5__control__T3__r2__7e8f4a",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "control",
   "rep": 2,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 1167,
   "input": 4,
   "cache_read": 27097,
   "cache_write": 6646,
   "warmup": null,
   "cache_write_1h": 6646,
   "cost_usd": 0.0819434,
   "wall_s": 12.3,
   "turns": 2,
   "prompt_last": 17275,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 179,
   "commands": null,
   "answer": "**Monthly totals**\n\n`monthly_report(entries, year, month)` in `ledger/core.py:51` computes the totals in three steps:\n\n1. **Filter:** `entries_in_month` keeps only the entries whose `day.year` and `day.month` both equal the requested values.\n2. **Overall total:** `total` adds up the `amount` of those entries with a plain `sum`.\n3. **Per-category totals:** `sum_by_category` walks the same entries and accumulates amounts in a dict keyed by category. Categories are lowercased in `add_entry`, so \"Food\" and \"food\" land in the same bucket.\n\nThe result is a dict with `count` (number of matching entries), `total` and `by_category`. Nothing is cached or pre-aggregated; every call rescans the full list.\n\nTwo details matter when reading the numbers:\n\n- **Float arithmetic:** Amounts are floats, so sums can show ordinary rounding artefacts (e.g. 0.1 + 0.2), and nothing is rounded to cents.\n- **Empty month:** A month with no entries returns `count` 0, `total` as the integer `0` (not `0.0`) and an empty `by_category`.\n\n**Invalid date strings**\n\nDates are only parsed in `add_entry`, via `parse_date` (`ledger/core.py:15`). A bad date fails in one of three ways:\n\n- **Wrong shape:** If the string does not split on `-` into exactly three all-digit parts, `parse_date` raises `ValueError(\"bad date: '<text>'\")`. This covers `\"2024/01/05\"`, `\"2024-01\"`, `\"2024-Jan-05\"`, an empty string, and anything with a sign or whitespace.\n- **Right shape, impossible date:** Strings such as `\"2024-13-01\"` or `\"2024-02-30\"` pass the shape check and then fail inside `date(y, m, d)`. This is still a `ValueError`, but with Python's own message (e.g. \"month must be in 1..12\") rather than the \"bad date\" one.\n- **Not a string:** Passing `None` or a `date` object raises `AttributeError` on `.split`, not `ValueError`, despite what the docstring says.\n\nIn every case the exception propagates out of `add_entry` uncaught. The `Entry` is never built, so nothing is appended and the list is unchanged.\n\n`add_entry` checks the amount first, so a call with both a non-positive amount and a bad date reports \"amount must be positive\" and never reaches the date.\n\nThe shape check is looser than the docstring's \"YYYY-MM-DD\": it does not enforce digit widths, so `\"2024-1-5\"` and `\"24-01-05\"` (year 24 AD) are accepted as valid.\n\nNo files were changed.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.081943
  },
  {
   "run": "claude-opus-5-5__control__T3__r3__1a047a",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "control",
   "rep": 3,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 1429,
   "input": 4,
   "cache_read": 27094,
   "cache_write": 6641,
   "warmup": null,
   "cache_write_1h": 6641,
   "cost_usd": 0.08714279999999999,
   "wall_s": 15.1,
   "turns": 2,
   "prompt_last": 17270,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 313,
   "commands": null,
   "answer": "I read `ledger/core.py` and changed nothing.\n\n## How monthly totals are computed\n\n`monthly_report(entries, year, month)` (`ledger/core.py:51`) does three things:\n\n1. **Filters to the month.** `entries_in_month` (line 40) keeps only entries whose `day.year` and `day.month` both equal the arguments.\n2. **Sums the amounts.** `total` (line 31) adds up `e.amount` over the filtered entries with the built-in `sum`.\n3. **Breaks the total down by category.** `sum_by_category` (line 44) builds a dict mapping each category to the running sum of its amounts.\n\nThe result is a dict with `count` (number of entries in the month), `total` and `by_category`.\n\nA few behaviours follow from this:\n\n- **Empty month:** the report is `{\"count\": 0, \"total\": 0, \"by_category\": {}}`. The total is the integer `0`, not `0.0`, because `sum` of nothing returns its integer start value.\n- **Float arithmetic:** amounts are stored as floats in dollars, so totals can show rounding artefacts such as `0.1 + 0.2` giving `0.30000000000000004`. Nothing is rounded.\n- **Category case:** categories are lower-cased when added through `add_entry`, so \"Food\" and \"food\" merge into one bucket. An `Entry` constructed directly skips that normalisation.\n- **Unchecked arguments:** `year` and `month` are not validated, so `month=13` just returns an empty report.\n\n## What happens when a date string is invalid\n\nDates are only parsed in `add_entry` (line 24), via `parse_date` (line 15). An invalid date raises `ValueError`, and nothing is appended to the list because parsing happens before the append.\n\nThere are two ways the error arises:\n\n- **Wrong shape:** if the string does not split on `-` into exactly three all-digit parts, `parse_date` raises `ValueError(\"bad date: '...'\")`. This covers `\"2024/01/05\"`, `\"2024-01\"`, `\"2024-Jan-05\"`, `\"\"`, and strings with surrounding whitespace.\n- **Right shape, impossible date:** strings like `\"2024-13-01\"` or `\"2023-02-30\"` pass the shape check and fail inside `date(y, m, d)`. That is still a `ValueError`, but with Python's own message (e.g. \"month must be in 1..12\") rather than the `bad date:` one.\n\nSome edge cases are worth knowing:\n\n- **Looser than the docstring says:** it claims YYYY-MM-DD only, but digit widths are not checked. `\"2024-1-5\"` is accepted, and `\"24-01-05\"` is accepted as year 24 AD.\n- **Amount is checked first:** `add_entry` validates the amount before the date, so a call with both a non-positive amount and a bad date reports only `\"amount must be positive\"`.\n- **Non-string input:** passing `None` or a `date` object raises `AttributeError` (no `.split`), not `ValueError`.\n- **No handling downstream:** nothing in this module catches these errors; they propagate to the caller.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.087143
  },
  {
   "run": "claude-opus-5-5__karpathy__T3__r1__46efd2",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "karpathy",
   "rep": 1,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 1176,
   "input": 4,
   "cache_read": 28112,
   "cache_write": 7662,
   "warmup": null,
   "cache_write_1h": 7662,
   "cost_usd": 0.0904544,
   "wall_s": 14.6,
   "turns": 2,
   "prompt_last": 18291,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 230,
   "commands": null,
   "answer": "**Monthly totals**\n\n`monthly_report(entries, year, month)` at `ledger/core.py:51` computes everything on demand from the entry list, in three steps:\n\n1. **Filter:** `entries_in_month` keeps the entries whose `day.year` and `day.month` both equal the requested values.\n2. **Sum:** `total` adds up `amount` across those entries.\n3. **Break down:** `sum_by_category` builds a dict mapping each category to the sum of its amounts.\n\nThe result is `{\"count\": ..., \"total\": ..., \"by_category\": ...}`.\n\nA few behaviours follow from the code:\n\n- **Empty month:** a month with no entries gives `count` 0, `total` 0 (the integer, since `sum()` of nothing is `0`) and an empty `by_category`.\n- **Float arithmetic:** amounts are plain floats, so totals can show rounding artefacts (e.g. 0.1 + 0.2), and nothing rounds to cents.\n- **Category case:** categories are lowercased in `add_entry`, so \"Food\" and \"food\" merge in the breakdown. An `Entry` constructed directly bypasses that.\n\n**Invalid date strings**\n\nDates are parsed by `parse_date` at `ledger/core.py:15`, which `add_entry` calls. It splits the text on `-` and requires exactly three parts, each made only of digits.\n\n- **Wrong shape:** inputs such as `\"2024/01/05\"`, `\"2024-01\"`, `\"2024-01-xx\"`, `\"\"`, or anything with surrounding whitespace raise `ValueError(\"bad date: '<text>'\")`.\n- **Right shape, impossible date:** inputs such as `\"2024-13-01\"` or `\"2024-02-30\"` pass the shape check and then fail inside `date(y, m, d)`. This is still a `ValueError`, but with Python's own message (e.g. \"month must be in 1..12\"), not the \"bad date\" one.\n- **Not a string:** `None` or an existing `date` object raises `AttributeError` on `.split`, not `ValueError`, despite what the docstring says.\n- **Looser than the docstring:** digit widths are not checked, so `\"2024-1-5\"` and `\"24-01-05\"` (year 24 AD) are accepted.\n\nIn `add_entry`, the `amount <= 0` check runs first, so a bad amount combined with a bad date reports the amount error. The date is parsed before the `Entry` is appended, so an invalid date leaves the list unchanged. Nothing in this file catches the exception; it propagates to the caller.\n\nNo files were changed.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.090454
  },
  {
   "run": "claude-opus-5-5__karpathy__T3__r2__a5462b",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "karpathy",
   "rep": 2,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 1420,
   "input": 4,
   "cache_read": 28111,
   "cache_write": 7661,
   "warmup": null,
   "cache_write_1h": 7661,
   "cost_usd": 0.0953262,
   "wall_s": 14.7,
   "turns": 2,
   "prompt_last": 18290,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 374,
   "commands": null,
   "answer": "I read `ledger/core.py` and changed nothing.\n\n## How monthly totals are computed\n\n`monthly_report(entries, year, month)` (line 51) does it in two steps:\n\n1. **Filter.** `entries_in_month` keeps only the entries whose `day.year` and `day.month` both equal the requested values.\n2. **Aggregate.** From that filtered list it builds a dict with three keys:\n   - `count` is the number of entries in the month.\n   - `total` is the plain sum of every entry's `amount`, via `total()`.\n   - `by_category` comes from `sum_by_category`, which keeps a running sum per category name in a dict.\n\nA few behaviours follow from this:\n\n- **No rounding.** Amounts are floats, so totals can show floating-point artefacts such as `0.1 + 0.2` giving `0.30000000000000004`.\n- **Empty months.** A month with no entries gives `count` 0, `by_category` `{}`, and `total` as the integer `0` rather than `0.0`, because `sum()` of nothing returns int 0.\n- **Category case.** Categories merge case-insensitively only because `add_entry` lowercases them on the way in. `sum_by_category` does not normalise, so an `Entry` constructed directly with `\"Food\"` would land in a separate bucket from `\"food\"`.\n\n## What happens with an invalid date string\n\nDates are only parsed in `parse_date` (line 15), which `add_entry` calls. It splits the text on `-` and handles bad input in two ways:\n\n- **Wrong shape.** If there are not exactly three parts, or any part is not all digits, it raises `ValueError(\"bad date: '<text>'\")`. This covers `\"2024/01/05\"`, `\"2024-01\"`, `\"2024-Jan-05\"`, an empty string, and anything with surrounding whitespace.\n- **Right shape, impossible date.** Something like `\"2024-02-30\"` or `\"2024-13-01\"` passes the shape check and fails inside `date(y, m, d)`. It is still a `ValueError`, but with Python's own message (e.g. \"day is out of range for month\") rather than the `bad date:` one.\n\nIn both cases the error propagates out of `add_entry` uncaught. The exception fires before `entries.append` runs, so the ledger is left unmodified.\n\nSome edge cases are worth knowing:\n\n- **Amount is checked first.** In `add_entry`, a non-positive amount combined with a bad date reports \"amount must be positive\", not the date error.\n- **Validation is looser than the docstring says.** It does not enforce digit widths, so `\"2024-1-5\"` is accepted, and `\"24-01-05\"` is accepted as year 24 AD.\n- **Non-string input is not a `ValueError`.** Passing `None` or a `date` object raises `AttributeError` (no `.split`), contrary to the docstring's \"Raises ValueError on anything else\".",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.095326
  },
  {
   "run": "claude-opus-5-5__karpathy__T3__r3__e9e031",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "karpathy",
   "rep": 3,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 1223,
   "input": 4,
   "cache_read": 28109,
   "cache_write": 7659,
   "warmup": null,
   "cache_write_1h": 7659,
   "cost_usd": 0.0913698,
   "wall_s": 12.7,
   "turns": 2,
   "prompt_last": 18288,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 109,
   "commands": null,
   "answer": "I read `ledger/core.py` and changed nothing.\n\n## Monthly totals\n\n`monthly_report(entries, year, month)` (`ledger/core.py:51`) computes the totals in three steps:\n\n1. **Filter.** `entries_in_month` (line 40) keeps only the entries whose `day.year` and `day.month` both equal the arguments.\n2. **Grand total.** `total` (line 31) sums `amount` over the filtered entries.\n3. **Per-category totals.** `sum_by_category` (line 44) builds a dict mapping each category to the running sum of its amounts.\n\nThe result is a dict with `count` (number of entries in the month), `total` and `by_category`.\n\n- **Empty months:** a month with no entries gives `count` 0, `by_category` `{}` and `total` as the integer `0` rather than `0.0`, because `sum()` of nothing returns `0`.\n- **Category case:** categories are lowercased when an entry is added (line 27), so \"Food\" and \"food\" land in the same bucket.\n- **Rounding:** amounts are plain floats and nothing is rounded, so totals can show float artefacts such as `0.30000000000000004`.\n- **Out-of-range month:** `year` and `month` are not validated, so month 13 simply matches nothing and returns the empty report.\n\n## Invalid date strings\n\nDates are parsed only in `parse_date` (line 15), which `add_entry` calls. An invalid date always ends in a `ValueError`, but by one of two routes:\n\n- **Wrong shape:** if splitting on `-` does not give exactly three parts, or any part is not all digits, the function raises `ValueError(\"bad date: '<text>'\")`. This covers `\"2024/01/05\"`, `\"2024-01\"`, `\"2024-01-xx\"`, `\"\"` and anything with whitespace or a sign.\n- **Right shape, impossible date:** strings like `\"2024-02-30\"` or `\"2024-13-01\"` pass the shape check and fail inside `date(y, m, d)`. That is still a `ValueError`, but with Python's own message (e.g. \"day is out of range for month\"), not the \"bad date\" one.\n\nRelated behaviours:\n\n- **Looser than the docstring:** field widths are not checked, so `\"2024-1-5\"` is accepted, and so is `\"5-1-2024\"` (read as year 5, month 1, day 2024, which then fails in `date()`). `str.isdigit()` is also true for some non-ASCII characters; superscript digits, for example, pass the check and then fail in `int()`, again as a `ValueError`.\n- **Non-string input:** passing `None` or a `date` object raises `AttributeError` on `.split`, not `ValueError`.\n- **No partial writes:** in `add_entry`, the date is parsed before the `Entry` is appended, so a bad date propagates to the caller and leaves the list unchanged.\n- **Check order:** the amount check (line 25) runs first, so with both a non-positive amount and a bad date you get \"amount must be positive\".\n- **No handling in this file:** nothing here catches the exception; it is up to the caller.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.09137
  },
  {
   "run": "claude-opus-5-5__placebo__T3__r1__29557c",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "placebo",
   "rep": 1,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 1066,
   "input": 4,
   "cache_read": 27845,
   "cache_write": 7393,
   "warmup": null,
   "cache_write_1h": 7393,
   "cost_usd": 0.08604899999999999,
   "wall_s": 13.0,
   "turns": 2,
   "prompt_last": 18022,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 157,
   "commands": null,
   "answer": "**Monthly totals**\n\n`monthly_report(entries, year, month)` in `ledger/core.py:51` computes the totals in three steps:\n\n1. **Filter:** `entries_in_month` keeps only entries whose `day.year` and `day.month` both equal the requested values.\n2. **Sum:** `total` adds up the `amount` of those entries using plain float addition, with no rounding.\n3. **Break down:** `sum_by_category` builds a dict that maps each category to the sum of its amounts.\n\nThe result is a dict with `count` (number of matching entries), `total` and `by_category`. A month with no entries gives `{\"count\": 0, \"total\": 0, \"by_category\": {}}`; nothing is raised. The `year` and `month` arguments are not validated, so a month of 13 also gives that empty report.\n\nCategories are lowercased in `add_entry`, so \"Food\" and \"food\" land in the same bucket.\n\n**Invalid date strings**\n\nDates are parsed only when an entry is added: `add_entry` calls `parse_date` (`ledger/core.py:15`), and an invalid date surfaces as an exception that nothing in this module catches. There are two paths to failure:\n\n- **Wrong shape:** if the string does not split on `-` into exactly three all-digit parts, `parse_date` raises `ValueError(\"bad date: '<text>'\")`. This covers inputs such as `\"2024/01/05\"`, `\"2024-01\"`, `\"2024-Jan-05\"` and `\"\"`.\n- **Right shape, impossible date:** inputs such as `\"2024-02-30\"` or `\"2024-13-01\"` pass the shape check and fail inside `date(y, m, d)`. This is still a `ValueError`, but with Python's own message (for example \"day is out of range for month\") instead of the \"bad date\" one.\n\nIn both cases the exception is raised before `entries.append` runs, so the list is left unchanged and no partial entry is stored.\n\nThree related behaviours are worth knowing:\n\n- **Amount is checked first:** `add_entry` checks `amount <= 0` before parsing the date, so a call with both a bad amount and a bad date reports \"amount must be positive\" only.\n- **Looser than the docstring:** the shape check does not enforce digit widths, so `\"2024-1-5\"` is accepted even though the docstring says YYYY-MM-DD.\n- **Non-string input:** passing `None` or a `date` object raises `AttributeError` from `text.split`, not `ValueError`.\n\nNo files were changed.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.086049
  },
  {
   "run": "claude-opus-5-5__placebo__T3__r2__c561f7",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "placebo",
   "rep": 2,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 1210,
   "input": 4,
   "cache_read": 27847,
   "cache_write": 7396,
   "warmup": null,
   "cache_write_1h": 7396,
   "cost_usd": 0.0889534,
   "wall_s": 12.7,
   "turns": 2,
   "prompt_last": 18025,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 230,
   "commands": null,
   "answer": "**How monthly totals are computed**\n\n`monthly_report(entries, year, month)` in `ledger/core.py:51` builds the report in three steps:\n\n1. **Filter:** `entries_in_month` keeps only the entries whose `day.year` and `day.month` both equal the requested values.\n2. **Total:** `total` sums the `amount` of the filtered entries.\n3. **Break down:** `sum_by_category` walks the same filtered entries and accumulates amounts into a dict keyed by category.\n\nThe result is a dict with `count` (number of entries in that month), `total`, and `by_category`.\n\nA few behaviours follow from this:\n\n- **Case-insensitive categories:** `add_entry` lowercases the category on the way in, so \"Food\" and \"food\" land in the same bucket.\n- **No rounding:** amounts are plain floats, so totals can show ordinary floating-point artefacts such as `0.1 + 0.2`.\n- **Empty months:** a month with no entries gives `count` 0, `total` as the integer `0` (not `0.0`), and an empty `by_category`.\n- **No month validation:** asking for month 13 raises nothing and returns the empty report.\n\n**What happens when a date string is invalid**\n\nDates are parsed only when an entry is added: `add_entry` calls `parse_date` (`ledger/core.py:15`). An invalid date raises `ValueError`, nothing is appended, and the error propagates to the caller uncaught. There are two distinct failure paths:\n\n- **Wrong shape:** if splitting on `-` does not give exactly three all-digit parts (for example `\"2024/01/05\"`, `\"2024-01\"`, `\"2024-Jan-05\"`, an empty string, or anything with spaces or a sign), it raises `ValueError(\"bad date: '<text>'\")`.\n- **Right shape, impossible date:** strings like `\"2024-02-30\"` or `\"2024-13-01\"` pass the shape check and fail inside `date(y, m, d)`. This is still a `ValueError`, but with Python's own message (such as \"day is out of range for month\") and without the \"bad date\" prefix or the offending text.\n\nSome edge cases:\n\n- **Looser than the docstring:** field widths are not checked, so `\"2024-1-5\"` and `\"24-01-05\"` (year 24 AD) are accepted despite the \"YYYY-MM-DD\" docstring.\n- **Non-string input:** passing `None` or a `date` object raises `AttributeError` from `.split`, not `ValueError`.\n- **Amount checked first:** `add_entry` validates the amount before parsing the date, so a call with both a non-positive amount and a bad date reports only \"amount must be positive\".\n\nNo files were changed.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.088953
  },
  {
   "run": "claude-opus-5-5__placebo__T3__r3__932f64",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "placebo",
   "rep": 3,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 1227,
   "input": 4,
   "cache_read": 27843,
   "cache_write": 7391,
   "warmup": null,
   "cache_write_1h": 7391,
   "cost_usd": 0.0892526,
   "wall_s": 13.3,
   "turns": 2,
   "prompt_last": 18020,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 212,
   "commands": null,
   "answer": "**Monthly totals**\n\n`monthly_report(entries, year, month)` (`ledger/core.py:51`) computes the totals in two steps:\n\n1. **Filter:** `entries_in_month` keeps only the entries whose `day.year` and `day.month` both equal the requested values.\n2. **Aggregate:** the filtered list is summarised into a dict with three keys:\n   - `count` is the number of entries in that month.\n   - `total` is the plain `sum()` of their amounts, from `total()`.\n   - `by_category` is a dict mapping each category to the sum of its amounts, from `sum_by_category()`.\n\nA few behaviours follow from that:\n\n- **Empty month:** a month with no entries gives `{\"count\": 0, \"total\": 0, \"by_category\": {}}`. The total here is the integer `0`, not `0.0`, because `sum()` of an empty sequence returns `0`.\n- **No rounding:** amounts are stored as floats and summed as floats, so the usual floating-point artefacts can appear (for example, 0.1 + 0.2 gives 0.30000000000000004).\n- **Category case:** categories are lowercased when an entry is added through `add_entry`, so \"Food\" and \"food\" land in the same bucket.\n- **No month validation:** `year` and `month` are not checked, so asking for month 13 returns an empty report, not an error.\n\n**Invalid date strings**\n\nDates are parsed by `parse_date` (`ledger/core.py:15`), which `add_entry` calls. It splits the text on `-` and requires exactly three parts, each consisting only of digits. What happens next depends on how the string is wrong:\n\n- **Wrong shape** (such as `\"2024/01/05\"`, `\"2024-01\"`, `\"2024-Jan-05\"`, `\"\"`, or a negative component): it raises `ValueError(\"bad date: '<text>'\")`.\n- **Right shape but impossible date** (such as `\"2024-13-01\"` or `\"2023-02-30\"`): the shape check passes, and the `date(y, m, d)` constructor raises its own `ValueError`. The message is Python's (for example, \"month must be in 1..12\"), not the \"bad date\" one.\n- **Not a string** (such as `None` or a `date` object): it raises `AttributeError` from `.split`, not `ValueError`.\n\nIn every case the exception propagates to the caller of `add_entry`; nothing catches it. The `Entry` is built before `append` runs, so a bad date leaves the entries list unchanged. The amount check comes first, so a non-positive amount reports \"amount must be positive\" even if the date is also bad.\n\nThe parser is looser than its \"YYYY-MM-DD\" docstring suggests: it does not enforce digit widths, so `\"2024-1-5\"` is accepted as 5 January 2024.\n\nNo files were changed.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.089253
  },
  {
   "run": "claude-opus-5-5__ponytail__T3__r1__e0e9f9",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "ponytail",
   "rep": 1,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 1294,
   "input": 4,
   "cache_read": 28128,
   "cache_write": 7680,
   "warmup": null,
   "cache_write_1h": 7680,
   "cost_usd": 0.0929616,
   "wall_s": 13.9,
   "turns": 2,
   "prompt_last": 18309,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 248,
   "commands": null,
   "answer": "**Monthly totals**\n\n`monthly_report(entries, year, month)` at `ledger/core.py:51` does three things:\n\n1. **Filters to the month.** `entries_in_month` keeps every entry whose `day.year` and `day.month` both equal the arguments. It is a linear scan of the whole list on every call, with no pre-grouping or caching.\n2. **Sums the filtered list.** `total` is a plain `sum()` of `e.amount`, and `sum_by_category` accumulates amounts into a dict keyed by category.\n3. **Returns a dict** of the form `{\"count\": ..., \"total\": ..., \"by_category\": {...}}`.\n\nA few behaviours follow from that:\n\n- **Float arithmetic:** amounts are stored as floats (`add_entry` calls `float(amount)`), so totals can show the usual drift (e.g. `0.1 + 0.2`), and nothing is rounded.\n- **Empty months:** a month with no entries gives `count` 0, `by_category` `{}` and `total` as the integer `0` rather than `0.0`, because `sum()` of an empty sequence returns its int start value.\n- **Category case:** categories are lowercased when added, so \"Food\" and \"food\" land in the same bucket.\n- **Unvalidated month:** `year` and `month` are not checked, so `month=13` quietly returns an empty report.\n\n**Invalid date strings**\n\nDates are parsed only in `add_entry`, via `parse_date` at `ledger/core.py:15`. It splits on `-` and requires exactly three parts that are all digits; otherwise it raises `ValueError(\"bad date: '<text>'\")`. That covers inputs like `\"2024/01/05\"`, `\"2024-01\"`, `\"abc\"`, `\"\"`, or anything with whitespace or a sign.\n\nIf the shape is right but the date does not exist (`\"2024-02-30\"`, `\"2024-13-01\"`), the `date(y, m, d)` constructor raises its own `ValueError`, with a message like \"day is out of range for month\" instead of \"bad date\".\n\nIn both cases the exception propagates out of `add_entry` uncaught. Parsing happens before `entries.append` runs, so the ledger is left unchanged and nothing is partially written. The amount check runs first, so a non-positive amount with a bad date reports the amount error.\n\nEdge cases in the parser:\n\n- **Loose format:** it is looser than the documented `YYYY-MM-DD`. No digit widths are enforced, so `\"2024-1-5\"` and `\"24-01-05\"` (year 24 AD) are both accepted.\n- **Non-string input:** `None` or another non-string raises `AttributeError` on `.split`, not `ValueError`.\n- **Unicode digits:** characters like `\"\u00b2\"` pass `isdigit()` but fail in `int()`. The result is still a `ValueError`, just with a different message.\n\nNo files were changed.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.092962
  },
  {
   "run": "claude-opus-5-5__ponytail__T3__r2__1ade79",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "ponytail",
   "rep": 2,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 1272,
   "input": 4,
   "cache_read": 28121,
   "cache_write": 7671,
   "warmup": null,
   "cache_write_1h": 7671,
   "cost_usd": 0.09244820000000001,
   "wall_s": 15.7,
   "turns": 2,
   "prompt_last": 18300,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 305,
   "commands": null,
   "answer": "**Monthly totals**\n\n`monthly_report(entries, year, month)` at `ledger/core.py:51` does three things:\n\n1. **Filters**: `entries_in_month` keeps only entries whose `day.year` and `day.month` both equal the arguments. It is a linear scan of the full list on every call.\n2. **Totals**: `total` sums `amount` over the filtered entries.\n3. **Breaks down by category**: `sum_by_category` builds a dict of category to running sum. Categories are lowercased on the way in by `add_entry`, so \"Food\" and \"food\" land in the same bucket.\n\nThe result is `{\"count\": ..., \"total\": ..., \"by_category\": ...}`.\n\nA few behaviours to be aware of:\n\n- **Float money**: amounts are floats in dollars, so totals can carry binary rounding error (e.g. `0.1 + 0.2`), and nothing is rounded.\n- **Empty months**: a month with no entries returns `count: 0`, `total: 0` (the int `0`, not `0.0`) and an empty `by_category`.\n- **Unvalidated arguments**: `year` and `month` are not checked, so `month=13` quietly returns the same empty report instead of raising.\n\n**Invalid date strings**\n\nDates are parsed only in `add_entry`, through `parse_date` (`ledger/core.py:15`). It splits on `-` and needs exactly three parts that are all digits; otherwise it raises `ValueError(\"bad date: '...'\")`. That covers inputs like `\"2024/01/05\"`, `\"2024-01\"`, `\"Jan-5-2024\"`, the empty string, and anything with whitespace or a sign.\n\nA string that has the right shape but is not a real date, such as `\"2024-02-30\"` or `\"2024-13-01\"`, passes that check and then fails in `date(y, m, d)`. That is still a `ValueError`, but with the standard library's message (e.g. \"day is out of range for month\") rather than \"bad date\".\n\nIn both cases the exception propagates out of `add_entry` before the append happens, so the ledger is left unchanged. Nothing is caught or skipped silently. The amount check runs first, so a non-positive amount together with a bad date reports the amount error.\n\nTwo gaps in the parser:\n\n- **Looser than the docstring**: it says YYYY-MM-DD, but no widths are enforced. `\"2024-1-5\"` is accepted, and `\"24-01-05\"` is accepted as year 24 AD, which would then never show up in a 2024 report.\n- **Non-strings**: passing `None` or a `date` object raises `AttributeError` (no `.split`), not `ValueError`.\n\nNo files were changed.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.092448
  },
  {
   "run": "claude-opus-5-5__ponytail__T3__r3__86f3c0",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "ponytail",
   "rep": 3,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 1010,
   "input": 4,
   "cache_read": 20236,
   "cache_write": 13951,
   "warmup": null,
   "cache_write_1h": 13951,
   "cost_usd": 0.1358712,
   "wall_s": 13.5,
   "turns": 2,
   "prompt_last": 17498,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 100,
   "commands": null,
   "answer": "No files changed; this is what `ledger/core.py` does.\n\n## Monthly totals\n\n`monthly_report(entries, year, month)` (`ledger/core.py:51`) does the work in three steps:\n\n1. **Filter.** `entries_in_month` keeps only the entries whose `day.year` and `day.month` both equal the requested values.\n2. **Sum.** `total` adds up the `amount` of every surviving entry with a plain `sum()`. An empty month gives an integer `0`, not `0.0`.\n3. **Break down.** `sum_by_category` accumulates the amounts into a dict keyed by category. Categories are lowercased when added through `add_entry`, so \"Food\" and \"food\" merge.\n\nThe result is a dict with `count` (number of entries in the month), `total` and `by_category`.\n\nAmounts are floats in dollars, so totals can show ordinary floating-point drift (for example `0.1 + 0.2`). Nothing is rounded.\n\n## Invalid date strings\n\nDates are parsed by `parse_date` (`ledger/core.py:15`), which `add_entry` calls. An invalid date always ends in an exception, but which one depends on how it is wrong:\n\n- **Wrong shape:** if the string does not split on `-` into exactly three all-digit parts (such as `\"2024/01/05\"`, `\"2024-01\"`, `\"2024-1-x\"` or `\"\"`), it raises `ValueError(\"bad date: '...'\")`.\n- **Right shape, impossible date:** `\"2024-02-30\"` or `\"2024-13-01\"` pass the shape check and then fail inside `date(y, m, d)`. That is still a `ValueError`, but with the standard library's message (such as \"day is out of range for month\"), not the \"bad date\" one.\n- **Not a string:** `None` or a `date` object raises `AttributeError` on `.split`, not `ValueError`, despite what the docstring says.\n\nIn every case nothing is added to the ledger, because the exception fires before `entries.append` runs. `add_entry` checks the amount first, so a bad amount together with a bad date reports \"amount must be positive\" and the date error is never reached.\n\n## Looser than the docstring\n\nThe shape check only tests for digits, not their width, so `\"2024-1-5\"` and `\"02024-01-05\"` are accepted as valid dates. `str.isdigit()` is also true for some characters `int()` cannot parse, such as superscript `\"\u00b2\"`. Those still raise a `ValueError`, but with `int()`'s message.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.135871
  },
  {
   "run": "claude-sonnet-5__caveman__T3__r1__12290b",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "caveman",
   "rep": 1,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 289,
   "input": 4,
   "cache_read": 49249,
   "cache_write": 12807,
   "warmup": null,
   "cache_write_1h": 12807,
   "cost_usd": 0.0639758,
   "wall_s": 5.9,
   "turns": 2,
   "prompt_last": 31448,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 17,
   "commands": null,
   "answer": "Monthly totals: `monthly_report` call `entries_in_month`, filter entries by `year` and `month` match on `e.day`. Then `total()` sum `amount` field over filtered entries. `sum_by_category` also run on same filtered set, group amount by category.\n\nInvalid date string: `parse_date` split on \"-\", check 3 parts, all digit. Fail either check, raise `ValueError(f\"bad date: {text!r}\")`. Bad date never reach `Entry` creation, so `add_entry` propagate exception, entries list unchanged.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.063976
  },
  {
   "run": "claude-sonnet-5__caveman__T3__r2__1173e0",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "caveman",
   "rep": 2,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 291,
   "input": 4,
   "cache_read": 49251,
   "cache_write": 12808,
   "warmup": null,
   "cache_write_1h": 12808,
   "cost_usd": 0.0640002,
   "wall_s": 6.3,
   "turns": 2,
   "prompt_last": 31449,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 15,
   "commands": null,
   "answer": "Monthly total: `monthly_report` filter entries by year+month via `entries_in_month`, then `total()` sum `amount` field over filtered list. Also count entries and `sum_by_category` breakdown same filtered set.\n\nInvalid date: `parse_date` split on `-`, need exactly 3 parts, all digit. Fail either check \u2192 raise `ValueError(f\"bad date: {text!r}\")`. Called from `add_entry`, so bad date raise at entry add time, before entry ever reach ledger. No catch anywhere \u2014 error propagate to caller.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.064
  },
  {
   "run": "claude-sonnet-5__caveman__T3__r3__017b6f",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "caveman",
   "rep": 3,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 239,
   "input": 4,
   "cache_read": 49250,
   "cache_write": 12791,
   "warmup": null,
   "cache_write_1h": 12791,
   "cost_usd": 0.063412,
   "wall_s": 4.9,
   "turns": 2,
   "prompt_last": 31432,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "Monthly total: `monthly_report` call `entries_in_month` filter entries by year and month, then `total` sum their `amount` field.\n\nInvalid date: `parse_date` split string on `-`. Need exact 3 parts, each digit-only, else raise `ValueError(f\"bad date: {text!r}\")`. `add_entry` call `parse_date` before append, so bad date abort whole entry, entry never enter ledger, error propagate up to caller.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.063412
  },
  {
   "run": "claude-sonnet-5__control__T3__r1__d024d9",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "control",
   "rep": 1,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 975,
   "input": 6,
   "cache_read": 74748,
   "cache_write": 10322,
   "warmup": null,
   "cache_write_1h": 10322,
   "cost_usd": 0.06599959999999999,
   "wall_s": 12.7,
   "turns": 3,
   "prompt_last": 28963,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 15,
   "commands": null,
   "answer": "## How monthly totals are computed\n\n`monthly_report(entries, year, month)` (core.py:51-53) does this in three steps:\n\n1. **Filter** \u2014 `entries_in_month` (core.py:40-41) keeps only entries whose `day.year` and `day.month` match the requested year/month.\n2. **Total** \u2014 `total()` (core.py:31-32) sums the `amount` field across that filtered subset.\n3. **Category breakdown** \u2014 `sum_by_category()` (core.py:44-48) accumulates amounts per `category` for the same filtered subset.\n\nThe result is a dict: `{\"count\": <number of entries that month>, \"total\": <sum of their amounts>, \"by_category\": <per-category sums>}`.\n\nNote that entries only get their `day` set via `add_entry`, which calls `parse_date` on the input string \u2014 so by the time an entry exists, its date has already been validated (see below).\n\n## What happens when a date string is invalid\n\n`parse_date` (core.py:15-21) is strict:\n\n- It splits the string on `\"-\"`.\n- If it doesn't split into exactly 3 parts, or any part isn't all-digit (`str.isdigit()`), it raises `ValueError(f\"bad date: {text!r}\")` immediately.\n- If it passes that check, it converts each part to `int` and constructs `date(y, m, d)` \u2014 which itself can raise `ValueError` if the numbers aren't a real calendar date (e.g. month 13, day 31 of February) or `OverflowError`/`ValueError` for out-of-range years.\n\nThis is called from `add_entry` (core.py:24-28), so an invalid date string causes `add_entry` to raise before the entry is ever appended to the ledger. There's no try/except anywhere in this file \u2014 the exception propagates straight to the caller. Practical implications:\n\n- A malformed date (wrong format, non-digit characters, wrong number of `-`-separated parts) never makes it into `entries`, so it can never skew a monthly total.\n- `isdigit()` accepts things like leading zeros or unicode digit characters but not negative numbers (`-` is treated as a separator, not a sign), so negative years/months/days fall into the \"bad date\" ValueError rather than reaching `date(...)`.\n- Empty string, `None`, or missing separators all fail the `len(parts) != 3` check first.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": true,
   "api_cost": 0.066
  },
  {
   "run": "claude-sonnet-5__control__T3__r1__01c6bd",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "control",
   "rep": 1,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 984,
   "input": 6,
   "cache_read": 74452,
   "cache_write": 10197,
   "warmup": null,
   "cache_write_1h": 10197,
   "cost_usd": 0.06553039999999999,
   "wall_s": 12.8,
   "turns": 3,
   "prompt_last": 28838,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 27,
   "commands": null,
   "answer": "Here's how it works:\n\n**Monthly totals** \u2014 `monthly_report(entries, year, month)` (ledger/core.py:51-53) first calls `entries_in_month`, which filters the full entry list down to those whose `day.year` and `day.month` match the requested year/month (line 40-41). It then computes three things from that filtered subset: `count` (number of entries), `total` (via `total()`, which just sums `e.amount` across the filtered entries, line 31-32), and `by_category` (via `sum_by_category()`, which builds a dict accumulating amounts per lowercased category, line 44-48). So the \"monthly total\" is simply the sum of `amount` over every entry whose stored `date` falls in that year/month \u2014 there's no separate date-parsing step at this stage since entries already store real `date` objects.\n\n**Invalid date strings** \u2014 date parsing only happens earlier, when an entry is created via `add_entry`, which calls `parse_date(day)` (line 27). `parse_date` (lines 15-21) splits the input on `-` and requires exactly 3 parts that are all purely digit strings (`str.isdigit()`); if not, it raises `ValueError(f\"bad date: {text!r}\")` immediately. If the split parts look numeric but form an impossible date (e.g., month 13, or Feb 30), the digit check passes but `date(y, m, d)` itself raises `ValueError` (Python's built-in validation) when constructing the `date` object. Either way, `add_entry` doesn't catch this \u2014 the `ValueError` propagates up to the caller, and the entry is never appended to the list. So invalid dates never make it into `entries` at all, and therefore can never silently corrupt a monthly total; they fail loudly at insertion time instead.\n\nOne edge case worth noting: `isdigit()` accepts unicode digit characters and doesn't allow a leading `-` for negative years, and it doesn't restrict length (e.g., `\"2024-1-1\"` or `\"02024-001-01\"` would pass the digit check and just get coerced by `int()`), so `parse_date` is stricter about *format shape* (three dash-separated numeric groups) than about zero-padding or magnitude \u2014 validity of the actual calendar date is left entirely to `date()`'s constructor.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.06553
  },
  {
   "run": "claude-sonnet-5__control__T3__r2__9cc707",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "control",
   "rep": 2,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 719,
   "input": 4,
   "cache_read": 46606,
   "cache_write": 10155,
   "warmup": null,
   "cache_write_1h": 10155,
   "cost_usd": 0.0571392,
   "wall_s": 11.0,
   "turns": 2,
   "prompt_last": 28796,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 19,
   "commands": null,
   "answer": "Here's how it works in `ledger/core.py`:\n\n**Monthly totals.** `monthly_report(entries, year, month)` (line 51) first calls `entries_in_month`, which filters the full entry list down to those whose `day.year` and `day.month` match the requested year/month (line 40-41). It then builds a dict with `count` (number of matching entries), `total` (the sum of their `amount` fields, via `total()` at line 31-32), and `by_category` (a per-category breakdown from `sum_by_category`, line 44-48, which accumulates amounts into a dict keyed by lowercased category). So the \"monthly total\" is just a plain sum of the `amount` values on every `Entry` whose date falls in that year/month \u2014 there's no weighting, rounding, or currency conversion involved.\n\n**Invalid date strings.** Dates only enter the system through `add_entry`, which calls `parse_date(day)` (line 27) before constructing an `Entry`. `parse_date` (line 15-21) splits the input on `-` and requires exactly 3 parts that are all digit strings; if not, it raises `ValueError(f\"bad date: {text!r}\")` immediately (line 19). If the split does pass that check, it converts each part to `int` and passes them to `date(y, m, d)` \u2014 so a string like `\"2024-13-40\"` passes the digit/format check but then raises `ValueError` from the standard library's `date` constructor itself (invalid month/day), not from the custom message.\n\nEither way, the error propagates up as an uncaught `ValueError` \u2014 `add_entry` has no try/except around `parse_date`, so the entry is never appended and the exception surfaces to the caller. There's no validation of date strings elsewhere in the file (e.g., in `entries_in_month` or `monthly_report`), since by the time entries exist in the list, their `day` is already a valid `date` object.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": true,
   "api_cost": 0.057139
  },
  {
   "run": "claude-sonnet-5__control__T3__r2__bfb834",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "control",
   "rep": 2,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 744,
   "input": 6,
   "cache_read": 74580,
   "cache_write": 10235,
   "warmup": null,
   "cache_write_1h": 10235,
   "cost_usd": 0.063308,
   "wall_s": 9.9,
   "turns": 3,
   "prompt_last": 28876,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 11,
   "commands": null,
   "answer": "Here's how it works:\n\n**Monthly totals:** `monthly_report(entries, year, month)` first calls `entries_in_month`, which filters the full entry list down to those whose `day.year` and `day.month` match the requested year/month. That filtered subset is then passed to `total`, which simply sums the `amount` field across those entries. The report also computes a per-category breakdown (`sum_by_category`) over the same filtered subset and a count of matching entries. So \"monthly total\" is just a plain sum of dollar amounts for entries whose date falls in that calendar month \u2014 no weighting, rounding, or currency conversion involved.\n\n**Invalid date strings:** Dates only enter the system through `add_entry`, which calls `parse_date(day)`. `parse_date` splits the string on `-` and requires exactly 3 parts, each consisting solely of digits (via `str.isdigit()`). If that check fails \u2014 e.g., wrong number of parts, non-numeric characters, or a format like `MM/DD/YYYY` \u2014 it raises `ValueError(f\"bad date: {text!r}\")` immediately, and no `Entry` is created. If the string passes that shape check but represents an impossible calendar date (e.g., `\"2024-02-30\"` or `\"2024-13-01\"`), the `date(y, m, d)` constructor itself raises a `ValueError` (Python's built-in date validation), which propagates uncaught out of `parse_date` and `add_entry`. Either way, there's no try/except anywhere in this module \u2014 an invalid date string always results in an unhandled `ValueError` bubbling up to the caller, and the entry is never added to the ledger.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.063308
  },
  {
   "run": "claude-sonnet-5__control__T3__r3__8bb30a",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "control",
   "rep": 3,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 976,
   "input": 6,
   "cache_read": 74786,
   "cache_write": 10359,
   "warmup": null,
   "cache_write_1h": 10359,
   "cost_usd": 0.0661652,
   "wall_s": 10.8,
   "turns": 3,
   "prompt_last": 29000,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 20,
   "commands": null,
   "answer": "Here's how it works:\n\n**Monthly totals** (`monthly_report`, line 51-53): It first calls `entries_in_month(entries, year, month)`, which filters the entry list to those whose `day.year` and `day.month` match the requested year/month (line 40-41). Then it builds a report dict with:\n- `count`: number of matching entries\n- `total`: the sum of `.amount` across those filtered entries, via `total()` (line 31-32), which just does `sum(e.amount for e in entries)`\n- `by_category`: a dict mapping category \u2192 summed amount for that month, computed by `sum_by_category()` (line 44-48), which iterates the filtered entries and accumulates into a dict with `out.get(e.category, 0.0) + e.amount`\n\nSo the monthly total is a plain sum of the `amount` field over whatever entries already exist in the ledger for that year/month \u2014 there's no re-parsing or validation happening at report time.\n\n**Invalid date strings**: Validation only happens once, at entry-creation time, not at report time. `add_entry()` (line 24-28) calls `parse_date(day)` before constructing the `Entry`. `parse_date()` (line 15-21) splits the string on `-` and requires exactly 3 parts that are all digit strings (`str.isdigit()`); if that check fails, it raises `ValueError(f\"bad date: {text!r}\")` immediately (line 19). If the format passes but the actual numbers are out of range (e.g., month 13, day 32, or a genuinely invalid combination like Feb 30), the `date(y, m, d)` constructor call on line 21 raises its own `ValueError` (e.g. \"day is out of range for month\").\n\nPractically, this means `add_entry` will raise and the entry is never appended to the list \u2014 so an invalid date string never makes it into `entries` at all, and therefore never affects `monthly_report`/`total`. There's no try/except anywhere in this file, so the `ValueError` simply propagates up to whatever code called `add_entry`.\n\nOne edge case worth noting: `isdigit()` doesn't accept a leading `-` sign, so negative years/components aren't silently parsed \u2014 they just fail the `isdigit` check and raise the \"bad date\" error rather than reaching `date(...)`.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.066165
  },
  {
   "run": "claude-sonnet-5__karpathy__T3__r1__cec673",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "karpathy",
   "rep": 1,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 745,
   "input": 4,
   "cache_read": 47617,
   "cache_write": 11166,
   "warmup": null,
   "cache_write_1h": 11166,
   "cost_usd": 0.061645399999999996,
   "wall_s": 8.0,
   "turns": 2,
   "prompt_last": 29807,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 7,
   "commands": null,
   "answer": "## How monthly totals are computed\n\n`monthly_report(entries, year, month)` (core.py:51-53) does this in three steps:\n\n1. **Filter to the month** \u2014 `entries_in_month` (core.py:40-41) keeps only entries whose `day.year` and `day.month` match the requested year/month.\n2. **Sum the total** \u2014 `total()` (core.py:31-32) sums `amount` across those filtered entries with a plain `sum(...)`.\n3. **Break down by category** \u2014 `sum_by_category()` (core.py:44-48) builds a dict accumulating `amount` per (lowercased) `category` for the same filtered entries.\n\nThe report returns `{\"count\": ..., \"total\": ..., \"by_category\": ...}` \u2014 the count and category breakdown are computed from the same filtered list, so they stay consistent with the total.\n\n## What happens with an invalid date string\n\nDate parsing happens earlier, in `add_entry` \u2192 `parse_date` (core.py:15-21), not at report time \u2014 by the time entries exist in the list, their `day` is already a valid `date` object.\n\n`parse_date` splits the string on `-` and requires exactly 3 parts that are all digit strings (`isdigit()`). If that check fails \u2014 wrong number of parts, non-numeric characters, empty string, etc. \u2014 it raises `ValueError(f\"bad date: {text!r}\")` immediately, before ever constructing a `date`.\n\nNote this means `add_entry` will raise for any malformed date (e.g. `\"2024/01/15\"`, `\"15-01-2024\"`, `\"not-a-date\"`), so no invalid dates ever make it into `entries` \u2014 `monthly_report` and its helpers never have to handle a bad date themselves. One subtlety: `isdigit()` doesn't reject negative numbers or catch semantically invalid dates (e.g. `\"2024-13-45\"` passes the digit check but then raises a `ValueError` from the underlying `date(y, m, d)` constructor instead, with a different error message).",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.061645
  },
  {
   "run": "claude-sonnet-5__karpathy__T3__r2__e71590",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "karpathy",
   "rep": 2,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 785,
   "input": 4,
   "cache_read": 47620,
   "cache_write": 11182,
   "warmup": null,
   "cache_write_1h": 11182,
   "cost_usd": 0.06211,
   "wall_s": 8.9,
   "turns": 2,
   "prompt_last": 29823,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 20,
   "commands": null,
   "answer": "## Monthly totals\n\n`monthly_report(entries, year, month)` (`ledger/core.py:51-53`) first calls `entries_in_month`, which filters `entries` down to those whose `day.year` and `day.month` match the given values. It then builds a dict with:\n\n- `count` \u2014 number of matching entries\n- `total` \u2014 `total(month_entries)`, i.e. the sum of `.amount` across those filtered entries\n- `by_category` \u2014 `sum_by_category(month_entries)`, a dict mapping each category to the sum of amounts in that category, built by iterating entries and accumulating into `out.get(e.category, 0.0) + e.amount`\n\nSo the \"monthly total\" is just a plain sum of `amount` over entries already stored with valid `date` objects for that year/month \u2014 no filtering or validation happens at this stage.\n\n## Invalid date strings\n\nValidation happens earlier, at entry-creation time in `parse_date` (`ledger/core.py:15-21`), which `add_entry` calls before appending an `Entry`:\n\n- The string is split on `-`. If it doesn't split into exactly 3 parts, or any part isn't all digits (`str.isdigit()`), it raises `ValueError(f\"bad date: {text!r}\")` immediately.\n- If it passes that check, the three parts are converted to ints and passed to `date(y, m, d)`. If they're digit-only but not a valid calendar date (e.g. month 13, day 31 in February, or a `date()`-range issue), the underlying `date()` constructor raises its own `ValueError` (e.g. \"month must be in 1..12\"), which propagates up uncaught.\n\nEither way, an invalid date string causes `add_entry` to raise `ValueError` and the entry is never appended to `entries`. There's no try/except anywhere in this file \u2014 the exception just propagates to the caller. Since invalid entries never make it into the list, `monthly_report`/`entries_in_month` never has to deal with bad dates; by the time totals are computed, every `Entry.day` is guaranteed to be a valid `date` object.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.06211
  },
  {
   "run": "claude-sonnet-5__karpathy__T3__r3__31fe49",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "karpathy",
   "rep": 3,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 907,
   "input": 6,
   "cache_read": 76835,
   "cache_write": 11394,
   "warmup": null,
   "cache_write_1h": 11394,
   "cost_usd": 0.070025,
   "wall_s": 11.6,
   "turns": 3,
   "prompt_last": 30035,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 36,
   "commands": null,
   "answer": "Here's how it works:\n\n**Monthly totals** \u2014 `monthly_report(entries, year, month)` (core.py:51-53) first calls `entries_in_month`, which filters the full entry list down to those whose `day.year` and `day.month` match the given values (core.py:40-41). It then builds a report dict with:\n- `count`: number of matching entries\n- `total`: `total(month_entries)` (core.py:31-32), which is just `sum(e.amount for e in entries)` \u2014 a plain sum of the `amount` field across the filtered entries, no rounding or currency handling\n- `by_category`: `sum_by_category(month_entries)` (core.py:44-48), which buckets those same filtered entries into a dict keyed by `category`, accumulating amounts per key\n\nSo the monthly total is simply the sum of `amount` over every `Entry` whose date falls in that year/month \u2014 it relies entirely on `Entry.day` already being a proper `date` object.\n\n**Invalid date strings** \u2014 dates only enter the system via `add_entry`, which calls `parse_date(day)` (core.py:27) to convert the string into a `date` before constructing the `Entry`. `parse_date` (core.py:15-21) splits the string on `-` and requires exactly 3 parts that are all digit-only. If that check fails \u2014 wrong number of segments, non-numeric characters, empty string, etc. \u2014 it raises `ValueError(f\"bad date: {text!r}\")` immediately (core.py:19), and no `Entry` is created (`add_entry` never appends). Note this is a fairly loose check: it doesn't validate the day/month ranges itself, so something like `\"2024-13-40\"` passes the `isdigit` check and is only caught when `date(y, m, d)` itself raises `ValueError` for an out-of-range month/day. Either way, the error propagates up out of `add_entry` uncaught \u2014 there's no try/except anywhere in this module, so invalid dates simply prevent the entry from being added and the exception surfaces to the caller.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.070025
  },
  {
   "run": "claude-sonnet-5__placebo__T3__r1__80a541",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "placebo",
   "rep": 1,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 963,
   "input": 6,
   "cache_read": 76298,
   "cache_write": 11121,
   "warmup": null,
   "cache_write_1h": 11121,
   "cost_usd": 0.06938559999999999,
   "wall_s": 12.1,
   "turns": 3,
   "prompt_last": 29762,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 35,
   "commands": null,
   "answer": "Here's how it works, based on `ledger/core.py`:\n\n**Monthly totals**\n\n`monthly_report(entries, year, month)` (line 51) is the entry point:\n1. It calls `entries_in_month(entries, year, month)`, which filters the full list of `Entry` objects down to those whose `day.year` and `day.month` match the given `year`/`month` (line 40-41).\n2. It computes `total()` over just that filtered subset \u2014 a plain `sum()` of each entry's `amount` (line 31-32).\n3. It also computes `sum_by_category()` on the same subset, which builds a dict mapping each entry's `category` to the running sum of `amount` for that category (line 44-48).\n4. The result is a dict: `{\"count\": ..., \"total\": ..., \"by_category\": ...}`.\n\nSo the \"monthly total\" is just the sum of `amount` across every entry whose stored `date` falls in that year/month \u2014 there's no rounding or currency handling beyond plain float addition.\n\n**Invalid date strings**\n\nDates only get parsed when an entry is created, via `add_entry` \u2192 `parse_date` (line 15-21). `parse_date` splits the text on `-` and requires exactly 3 numeric parts (`len(parts) != 3 or not all(p.isdigit() for p in parts)`). If that check fails, it raises `ValueError(f\"bad date: {text!r}\")` immediately \u2014 it never reaches `date(y, m, d)`.\n\nIf the string *looks* numeric and has 3 parts (e.g. `\"2024-13-40\"`) but represents an impossible calendar date, `parse_date` passes it straight to Python's `date(y, m, d)` constructor, which raises its own `ValueError` (e.g. `\"month must be in 1..12\"`) instead of the custom \"bad date\" message.\n\nEither way, invalid input raises rather than being silently accepted, and since `add_entry` calls `parse_date` before appending to `entries`, a bad date string means the entry is never added \u2014 so it can't corrupt `total()`, `sum_by_category()`, or `monthly_report()` downstream. There's no `try/except` anywhere in this module, so the exception propagates straight to the caller.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.069386
  },
  {
   "run": "claude-sonnet-5__placebo__T3__r2__b90dd2",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "placebo",
   "rep": 2,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 743,
   "input": 6,
   "cache_read": 76309,
   "cache_write": 11129,
   "warmup": null,
   "cache_write_1h": 11129,
   "cost_usd": 0.0672198,
   "wall_s": 8.5,
   "turns": 3,
   "prompt_last": 29770,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 11,
   "commands": null,
   "answer": "**Monthly totals:** `monthly_report(entries, year, month)` first calls `entries_in_month`, which filters `entries` down to those whose `day.year` and `day.month` match the given `year`/`month`. That filtered list is then passed to `total`, which simply sums the `amount` field across those entries. The same filtered list also goes through `sum_by_category` to build a per-category breakdown. The report returns a dict with `count` (number of matching entries), `total` (the summed amount), and `by_category` (a dict mapping category \u2192 summed amount).\n\nEntries only end up with a `day` in the first place via `add_entry`, which calls `parse_date` on the incoming date string.\n\n**Invalid date strings:** `parse_date` splits the text on `-` and requires exactly 3 parts that are all purely numeric digit strings (`str.isdigit()`). If that check fails \u2014 wrong number of parts, non-numeric characters, empty strings, etc. \u2014 it raises `ValueError(f\"bad date: {text!r}\")` immediately, before ever constructing a `date`. If the split passes that check but the numbers are out of range (e.g., month 13, day 31 in February), the subsequent `date(y, m, d)` call itself raises a `ValueError` (from the standard library), just with `datetime`'s own message rather than the custom one. Either way, `add_entry` (and thus anything building entries from raw strings) propagates a `ValueError` uncaught \u2014 there's no try/except anywhere in this module, so invalid dates simply abort entry creation rather than being silently skipped or coerced.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.06722
  },
  {
   "run": "claude-sonnet-5__placebo__T3__r3__db08cf",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "placebo",
   "rep": 3,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 724,
   "input": 6,
   "cache_read": 76305,
   "cache_write": 11126,
   "warmup": null,
   "cache_write_1h": 11126,
   "cost_usd": 0.06701700000000001,
   "wall_s": 9.8,
   "turns": 3,
   "prompt_last": 29767,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 24,
   "commands": null,
   "answer": "Here's how it works:\n\n**Monthly totals.** `monthly_report(entries, year, month)` first calls `entries_in_month`, which filters the full entry list down to those whose `day.year` and `day.month` match the requested period. That filtered list is then run through `total()` (a plain `sum()` of the `amount` fields) to get the month's total, and through `sum_by_category()` to get a dict of per-category subtotals. The report returned is `{\"count\", \"total\", \"by_category\"}` \u2014 the count and both totals are all derived from that same filtered subset, so there's no double-counting or cross-month leakage.\n\n**Invalid date strings.** Dates only enter the system through `add_entry`, which calls `parse_date(day)` before constructing the `Entry`. `parse_date` splits the string on `-` and requires exactly 3 parts that are all digit-only; if that check fails, it raises `ValueError(f\"bad date: {text!r}\")` immediately \u2014 no entry is appended. If the split passes but the numbers aren't a real calendar date (e.g. month 13, or Feb 30), the `date(y, m, d)` constructor itself raises `ValueError` (Python's built-in message, e.g. `\"month must be in 1..12\"`), which propagates up uncaught. Either way, an invalid date string means `add_entry` raises before the entry is ever added to the ledger, so it can't silently corrupt a later monthly total \u2014 the failure happens at insertion time, not at reporting time.\n\nNo files were changed.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.067017
  },
  {
   "run": "claude-sonnet-5__ponytail__T3__r1__573a92",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "ponytail",
   "rep": 1,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 749,
   "input": 6,
   "cache_read": 76851,
   "cache_write": 11400,
   "warmup": null,
   "cache_write_1h": 11400,
   "cost_usd": 0.0684722,
   "wall_s": 11.8,
   "turns": 3,
   "prompt_last": 30041,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 16,
   "commands": null,
   "answer": "Monthly totals: `monthly_report(entries, year, month)` first calls `entries_in_month`, which filters `entries` to those whose `Entry.day` matches the given `year` and `month` (day-of-month is irrelevant, only year/month must match). It then builds a dict with `count` (number of matching entries), `total` (the sum of `.amount` over just those filtered entries, via `total()`), and `by_category` (a dict mapping category \u2192 summed amount, via `sum_by_category()`, again over only the filtered entries).\n\nNote that `day` on each `Entry` is already a `date` object by the time these functions run \u2014 the actual string parsing happens earlier, in `parse_date`, when an entry is created via `add_entry`.\n\nInvalid date strings: `parse_date` is where validation happens. It splits the string on `-` and requires exactly 3 parts, each consisting only of digits (`str.isdigit()`); if that check fails, it raises `ValueError(f\"bad date: {text!r}\")` immediately. If the format looks right but the numbers are out of range (e.g. month 13, day 32, or Feb 30), the `date(y, m, d)` constructor itself raises `ValueError` (with its own message, not the \"bad date\" one). Either way, `add_entry` doesn't catch this \u2014 the exception propagates straight up out of `add_entry`, so the entry is never appended and no partial/invalid `Entry` ends up in the ledger. There's no validation error handling inside `monthly_report`/`entries_in_month` themselves since they only ever operate on already-parsed `date` objects.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.068472
  },
  {
   "run": "claude-sonnet-5__ponytail__T3__r2__8cc741",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "ponytail",
   "rep": 2,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 767,
   "input": 6,
   "cache_read": 76862,
   "cache_write": 11410,
   "warmup": null,
   "cache_write_1h": 11410,
   "cost_usd": 0.0686944,
   "wall_s": 9.3,
   "turns": 3,
   "prompt_last": 30051,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 42,
   "commands": null,
   "answer": "**Monthly totals:** `monthly_report(entries, year, month)` (core.py:51-53) calls `entries_in_month` to filter entries where `e.day.year == year and e.day.month == month` (core.py:40-41), then passes that filtered list to `total()`, which just sums `e.amount` across the entries (core.py:31-32). So the monthly total is a plain sum of amounts for entries whose stored `date` matches the given year/month \u2014 no averaging, no category weighting. `monthly_report` also returns the count and a per-category breakdown (`sum_by_category`) of that same filtered list.\n\n**Invalid date strings:** dates are parsed once, at entry-creation time, by `parse_date` (core.py:15-21), called from `add_entry` (core.py:27). It splits on `-`, requires exactly 3 parts that are all digit strings, and raises `ValueError(f\"bad date: {text!r}\")` if not. If the split passes but the numbers are out of range (e.g. month 13, day 31 in April), the underlying `date(y, m, d)` constructor call raises its own `ValueError` (e.g. \"month must be in 1..12\"), which propagates uncaught. Either way, `add_entry` raises before the entry is ever appended to the list \u2014 so an invalid date string means no `Entry` is created, nothing enters `entries`, and it therefore can't skew any later `total`/`monthly_report` call. There's no catching/logging of bad dates anywhere in this file; the caller of `add_entry` is responsible for handling the exception.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.068694
  },
  {
   "run": "claude-sonnet-5__ponytail__T3__r3__69adef",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "ponytail",
   "rep": 3,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 917,
   "input": 6,
   "cache_read": 76871,
   "cache_write": 11420,
   "warmup": null,
   "cache_write_1h": 11420,
   "cost_usd": 0.0702362,
   "wall_s": 11.8,
   "turns": 3,
   "prompt_last": 30061,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 32,
   "commands": null,
   "answer": "## Monthly totals\n\n`monthly_report(entries, year, month)` computes the total in two steps:\n\n1. `entries_in_month` filters the full entry list down to only those `Entry` objects whose `day.year` and `day.month` match the requested year/month (day-of-month is irrelevant).\n2. `total()` sums the `.amount` field over that filtered list (plain `sum()`, so an empty month yields `0.0`).\n\nSeparately, `sum_by_category` walks the same filtered list and accumulates a `{category: amount}` dict, so `monthly_report` returns `count`, `total`, and `by_category` all derived from the same month-filtered subset.\n\nNote: `total(entries)` sums *all* entries passed to it \u2014 it has no month awareness itself. It only reflects a single month because `monthly_report` first narrows `entries` via `entries_in_month`.\n\n## Invalid date strings\n\nDates only enter the ledger through `add_entry`, which calls `parse_date(day)` before constructing the `Entry`. `parse_date` is strict:\n\n- It splits on `-` and requires exactly 3 parts, each all-digit (`str.isdigit()`).\n- If that check fails (wrong number of parts, non-numeric parts, empty string, extra whitespace, etc.), it raises `ValueError(f\"bad date: {text!r}\")` immediately \u2014 the entry is never appended, and `entries` is left unchanged.\n- If the split/digit check passes but the numbers aren't a real calendar date (e.g. month 13, day 31 in February, or a `day.isdigit()` string like `\"-1\"` won't even pass since `\"-1\".isdigit()` is `False`... but something like `2024-02-30` will), `date(y, m, d)` itself raises `ValueError` (e.g. `\"day is out of range for month\"`), which propagates uncaught out of `parse_date` and `add_entry`.\n\nSo in both cases the caller gets an uncaught `ValueError` and no partial entry is added \u2014 there's no silent failure or defaulting.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.070236
  },
  {
   "run": "claude-sonnet-5-5__caveman__T3__r1__99f08c",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "caveman",
   "rep": 1,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 694,
   "input": 4,
   "cache_read": 28582,
   "cache_write": 8919,
   "warmup": null,
   "cache_write_1h": 8919,
   "cost_usd": 0.0483404,
   "wall_s": 6.6,
   "turns": 2,
   "prompt_last": 19155,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` first calls `entries_in_month`. That keeps only entries whose `day.year` and `day.month` both match the arguments. Then it builds a dict with three values:\n- `count`: the number of matching entries.\n- `total`: the sum of their `amount` values, from `total`.\n- `by_category`: a dict from `sum_by_category`, which adds up amounts per category. Categories are lowercased when an entry is added, so `Food` and `food` merge.\n\nNothing is rounded. Amounts are floats, so sums can carry small floating-point error. A month with no entries gives `count` 0, `total` 0 and an empty `by_category`.\n\n**Invalid date strings.** `parse_date` handles the strings. It splits the text on `-`. It raises `ValueError(\"bad date: ...\")` if there are not exactly three parts, or if any part is not all digits. A negative year or a sign character fails the digit check. It then converts the parts to ints and calls `date(y, m, d)`. Out-of-range values, such as month 13 or Feb 30, make `date` raise its own `ValueError`, with a different message. Both errors are the same type, so callers can catch `ValueError`.\n\n**Effect on totals.** `add_entry` calls `parse_date` while building the `Entry`. A bad date therefore raises before `append`, and the entry is not added. `add_entry` checks `amount <= 0` first, so a bad amount raises before the date is parsed. The report functions never see date strings, only parsed `date` objects. An invalid date cannot reach a monthly total.\n\n**Edge case.** `isdigit()` also accepts some non-ASCII digit characters, such as `\u00b2`. `int()` then raises a `ValueError` that has no \"bad date\" message. It is still a `ValueError`.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.04834
  },
  {
   "run": "claude-sonnet-5-5__caveman__T3__r2__7a306a",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "caveman",
   "rep": 2,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 603,
   "input": 4,
   "cache_read": 28579,
   "cache_write": 8916,
   "warmup": null,
   "cache_write_1h": 8916,
   "cost_usd": 0.047417799999999996,
   "wall_s": 7.3,
   "turns": 2,
   "prompt_last": 19152,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` computes them. It calls `entries_in_month`, which keeps entries whose `day.year` and `day.month` both match the arguments. It then builds a dict with three values:\n- `count`: the number of matching entries.\n- `total`: the sum of their amounts, from `total`.\n- `by_category`: a dict of per-category sums, from `sum_by_category`.\n\nAn empty month gives `count` 0, `total` 0 and an empty `by_category`. Amounts are plain floats, so rounding error can build up.\n\n**Invalid date.** Entries store `date` objects, not strings. `add_entry` converts the string with `parse_date`.\n\n`parse_date` splits the text on `-`. It raises `ValueError(f\"bad date: {text!r}\")` in two cases:\n- The text does not have exactly 3 parts.\n- Any part is not all digits. A negative year, a sign, or spaces fail this check.\n\nWell-formed text with impossible values, such as `2024-02-30` or month 13, passes that check. The `date(y, m, d)` constructor then raises its own `ValueError`, with a different message. Non-string input such as `None` fails at `text.split` with `AttributeError`, not `ValueError`.\n\n`add_entry` checks `amount <= 0` before it parses the date. A bad amount raises first, even if the date is also bad. On any error, nothing is appended, because `entries.append` runs only after `parse_date` succeeds. The caller sees the exception. Nothing in this file catches it.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.047418
  },
  {
   "run": "claude-sonnet-5-5__caveman__T3__r3__81b4b3",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "caveman",
   "rep": 3,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 621,
   "input": 4,
   "cache_read": 28584,
   "cache_write": 8922,
   "warmup": null,
   "cache_write_1h": 8922,
   "cost_usd": 0.0476228,
   "wall_s": 6.3,
   "turns": 2,
   "prompt_last": 19158,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` calls `entries_in_month`. That keeps only entries whose `day.year` and `day.month` both match the arguments. It then builds a dict with three values:\n- `count`: number of matching entries.\n- `total`: sum of their `amount` values, from `total`.\n- `by_category`: per-category sums, from `sum_by_category`. Categories are lowercased when entries are added, so `Food` and `food` merge.\n\nIf no entries match, `count` is 0, `total` is 0 and `by_category` is `{}`. Amounts are plain floats, so sums can show float rounding error.\n\n**Invalid date strings.** Dates are parsed only in `add_entry`, through `parse_date`. `parse_date` splits the text on `-`. It raises `ValueError(f\"bad date: {text!r}\")` in two cases:\n- The text does not have exactly 3 parts.\n- Any part is not all digits.\n\nTwo things follow from this:\n- Text that has three digit parts but is not a real date, such as `2024-13-45`, passes that check. `date(y, m, d)` then raises its own `ValueError`, such as \"month must be in 1..12\". The message differs from \"bad date\".\n- A non-string input, such as `None`, fails at `text.split` with `AttributeError`, not `ValueError`.\n\n`add_entry` checks `amount <= 0` before it parses the date. A bad amount therefore raises first. When the date is invalid, `append` never runs, so the entry is not added and the list stays unchanged. The error propagates to the caller, and nothing catches it.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.047623
  },
  {
   "run": "claude-sonnet-5-5__control__T3__r1__863973",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "control",
   "rep": 1,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 909,
   "input": 4,
   "cache_read": 25845,
   "cache_write": 6178,
   "warmup": null,
   "cache_write_1h": 6178,
   "cost_usd": 0.038979,
   "wall_s": 8.7,
   "turns": 2,
   "prompt_last": 16414,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**How monthly totals are computed**\n\n`monthly_report(entries, year, month)` in `ledger/core.py:51` does the work in three steps:\n\n1. `entries_in_month` (line 40) keeps only the entries whose `day.year` and `day.month` both match the arguments.\n2. `total` (line 31) adds up the `amount` of those entries.\n3. `sum_by_category` (line 44) builds a dict that maps each category to its summed amount for that month.\n\n`monthly_report` returns a dict with three keys:\n- `count`: the number of entries in that month.\n- `total`: the sum of their amounts.\n- `by_category`: the per-category sums.\n\nSome details:\n- Amounts are floats, so the sums are ordinary floating-point sums with no rounding.\n- Categories are lowercased when an entry is added (`add_entry`, line 27), so \"Food\" and \"food\" fall into the same bucket.\n- A month with no entries gives `count` 0, `total` 0 (an int, since `sum` of an empty sequence is 0) and an empty `by_category` dict.\n\n**What happens with an invalid date string**\n\nDates are parsed only when an entry is added. `add_entry` calls `parse_date` (line 15), which splits the string on `-`. It raises `ValueError(f\"bad date: ...\")` if the result isn't exactly three parts or if any part isn't all digits. Strings like `\"2024/01/05\"`, `\"2024-1\"`, `\"2024-01-xx\"` and `\"-2024-01-05\"` fail this way.\n\nA string that passes that check can still fail inside `date(y, m, d)`, which raises its own `ValueError`. `\"2024-13-40\"` does this, with a message like \"month must be in 1..12\". So callers get a `ValueError` either way, but the messages differ.\n\nSome consequences:\n- `add_entry` checks `amount <= 0` first, and only then parses the date. A call with both a bad amount and a bad date reports the amount error.\n- Nothing is appended to `entries` when the date is invalid, because the `Entry` is built and appended in a single expression. The list is left unchanged.\n- `parse_date` is lenient about digit count. `\"24-1-5\"` parses as year 24, January 5. `str.isdigit()` also accepts some non-ASCII digit characters, which `int()` then converts.\n- The monthly functions never see date strings, only parsed `date` objects. An invalid date can't reach the report, because it is rejected when the entry is added.\n\nI didn't change any files.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.038979
  },
  {
   "run": "claude-sonnet-5-5__control__T3__r2__66f5d7",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "control",
   "rep": 2,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 864,
   "input": 4,
   "cache_read": 25932,
   "cache_write": 6268,
   "warmup": null,
   "cache_write_1h": 6268,
   "cost_usd": 0.0389064,
   "wall_s": 7.5,
   "turns": 2,
   "prompt_last": 16504,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` does the work in three steps:\n\n1. It calls `entries_in_month`. That keeps only the entries whose `day.year` and `day.month` both match the arguments.\n2. It counts those entries and passes them to `total`. `total` adds up every entry's `amount`, so the result is a plain sum of floats.\n3. It passes the same filtered list to `sum_by_category`. That builds a dict mapping each category to its summed amount, starting from `0.0` for categories it hasn't seen yet.\n\nThe report comes back as a dict with `count`, `total` and `by_category`. If no entries fall in the month, you get `count` 0, `total` 0 (an int, because `sum` of an empty sequence is `0`) and an empty `by_category`. Categories are already lowercased in `add_entry`, so `\"Food\"` and `\"food\"` end up in the same bucket.\n\n**Invalid dates.** Dates are parsed only when an entry is added, by `parse_date` inside `add_entry`. `monthly_report` never sees a date string, because entries store real `date` objects.\n\n`parse_date` splits the string on `-`. It raises `ValueError(f\"bad date: ...\")` if there aren't exactly three parts or if any part isn't all digits. That check catches things like `\"2024/01/05\"`, `\"2024-1\"`, `\"2024-01-xx\"` and negative numbers.\n\nA string that passes that check but isn't a real calendar date, such as `\"2024-13-40\"`, reaches `date(y, m, d)`. That raises its own `ValueError` (for example \"month must be in 1..12\"), and the message is different from the \"bad date\" one. Either way the caller gets a `ValueError`.\n\nIn `add_entry`, the amount check runs first, then the date is parsed while the `Entry` is being built. If the date is invalid, the exception propagates before `entries.append` runs, so the list is left unchanged. Nothing is caught or swallowed.\n\nA few edge cases:\n- `\"2024-01-05 \"` (trailing space) fails the digit check.\n- Non-ASCII digits, such as Arabic-Indic ones, pass `isdigit()` and are converted by `int()`, so they may parse as valid.\n- A non-string input like `None` raises `AttributeError` on `.split`, not `ValueError`.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.038906
  },
  {
   "run": "claude-sonnet-5-5__control__T3__r3__de04a9",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "control",
   "rep": 3,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 841,
   "input": 4,
   "cache_read": 25931,
   "cache_write": 6266,
   "warmup": null,
   "cache_write_1h": 6266,
   "cost_usd": 0.0386682,
   "wall_s": 7.8,
   "turns": 2,
   "prompt_last": 16502,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` does the work. It first calls `entries_in_month`, which keeps only entries whose `day.year` and `day.month` both match the arguments. It then returns a dict with three values:\n\n- `count`: the number of entries in that month.\n- `total`: the sum of their `amount` values, from `total()`.\n- `by_category`: a dict mapping each category to its summed amount, from `sum_by_category()`.\n\nCategories are lowercased when an entry is added, so \"Food\" and \"food\" land in the same bucket. Amounts are plain floats, so the sums are ordinary floating-point sums with no rounding. If no entries match, you get a count of 0, a total of 0 and an empty `by_category` dict.\n\n**Invalid date strings.** Dates are only parsed in `add_entry`, through `parse_date`, and never during reporting. `parse_date` splits the string on `-`. It raises `ValueError(f\"bad date: ...\")` if there aren't exactly three parts or if any part contains a non-digit character. That check catches things like `\"2024/01/05\"`, `\"2024-1-x\"` and negative numbers. Empty parts are also rejected, because `\"\".isdigit()` is False.\n\nA string that passes the shape check but isn't a real date, such as `\"2024-02-30\"` or `\"2024-13-01\"`, is not caught by that check. It goes on to `date(y, m, d)`, which raises its own `ValueError` (for example \"day is out of range for month\"). Callers therefore always see a `ValueError`, but the message differs between the two cases.\n\nA few other behaviours follow from this:\n\n- `add_entry` checks `amount <= 0` before it parses the date. A bad amount raises its own `ValueError` first, and a bad date is only reported if the amount is valid.\n- The date is parsed inside the `Entry(...)` call, before `entries.append`. An invalid date therefore leaves the list unchanged.\n- A non-string `day`, such as `None`, would raise `AttributeError` from `text.split` rather than `ValueError`.\n- The check accepts unusual but \"digit-only\" strings such as `\"2024-01-05\"` with extra padding (`\"02024-001-005\"`). These parse fine, because `int()` handles leading zeros.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.038668
  },
  {
   "run": "claude-sonnet-5-5__karpathy__T3__r1__ded2de",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "karpathy",
   "rep": 1,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 900,
   "input": 4,
   "cache_read": 26951,
   "cache_write": 7289,
   "warmup": null,
   "cache_write_1h": 7289,
   "cost_usd": 0.0435542,
   "wall_s": 11.0,
   "turns": 2,
   "prompt_last": 17525,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` in `ledger/core.py` does the work in three steps:\n\n1. `entries_in_month` keeps only the entries whose `day.year` and `day.month` both match the arguments.\n2. `total` adds up the `amount` of those entries, which gives the month's overall total.\n3. `sum_by_category` builds a dict that maps each category to the sum of its amounts. Categories are lowercased when an entry is added, so `\"Food\"` and `\"food\"` land in the same bucket.\n\nThe function returns a dict with three keys: `count` (the number of entries in the month), `total`, and `by_category`. A month with no entries gives a count of 0, a total of `0` (the integer that `sum` returns for an empty sequence, not `0.0`), and an empty `by_category` dict. The amounts are plain floats, so the totals can pick up ordinary floating-point rounding error.\n\n**Invalid date strings.** Dates are only parsed when an entry is added. `add_entry` calls `parse_date(day)`, which does the following:\n\n- It splits the text on `-`. If the result isn't exactly three parts, or any part isn't made only of digits, it raises `ValueError(f\"bad date: {text!r}\")`.\n- Otherwise it converts the parts to ints and passes them to `datetime.date(y, m, d)`.\n- If the parts are all digits but don't form a real date, such as `\"2024-13-01\"` or `\"2024-02-30\"`, the `date` constructor raises its own `ValueError`. The message is different, for example \"month must be in 1..12\".\n\nBoth failures are `ValueError`, so a caller can catch them together. There are a few side effects and edge cases:\n\n- **Order of checks:** `add_entry` checks `amount <= 0` first, then parses the date. A bad date is never reported if the amount is already invalid.\n- **Nothing is appended on failure:** `entries.append(...)` runs only after `parse_date` succeeds, so a bad entry doesn't leave the list partly modified.\n- **Not strict ISO:** `\"2024-1-5\"` is accepted, because only the digits are checked and not the field widths.\n- **Non-string input:** a non-string such as `None` fails with `AttributeError` on `.split`, not `ValueError`.\n- **Reports never parse dates:** `monthly_report` and the functions it calls work with `date` objects that are already stored, so they never see an invalid date string.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.043554
  },
  {
   "run": "claude-sonnet-5-5__karpathy__T3__r2__c28a43",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "karpathy",
   "rep": 2,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 845,
   "input": 4,
   "cache_read": 26951,
   "cache_write": 7289,
   "warmup": null,
   "cache_write_1h": 7289,
   "cost_usd": 0.0430042,
   "wall_s": 8.9,
   "turns": 2,
   "prompt_last": 17525,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` does the work in three steps:\n\n1. It calls `entries_in_month`. That keeps only the entries whose `day.year` and `day.month` both match the arguments.\n2. It passes that filtered list to `total`, which adds up each entry's `amount`.\n3. It passes the same list to `sum_by_category`, which builds a dict mapping each category to its summed amount.\n\nIt returns a dict with three keys: `count` (the number of entries in that month), `total` (the overall sum), and `by_category` (the per-category sums). If no entries match the month, `count` is 0, `total` is 0 and `by_category` is `{}`. The amounts are plain floats, so the sums can carry ordinary floating-point rounding error.\n\n**Invalid date strings.** Dates are parsed only when an entry is added. `add_entry` calls `parse_date(day)`, and `parse_date` splits the string on `-`. It raises `ValueError(f\"bad date: ...\")` if the result isn't exactly three parts or if any part isn't made only of digits. Examples are `\"2024/01/05\"`, `\"2024-1\"`, `\"abc\"` and `\"2024-01--5\"`. Negative-looking or signed parts fail the digit check too.\n\nIf the shape is right but the values are impossible, such as `\"2024-13-01\"` or `\"2024-02-30\"`, the `date(y, m, d)` constructor raises its own `ValueError` with a different message. Either way the caller gets a `ValueError`.\n\n**Ordering in `add_entry`.** The amount check (`amount <= 0` raises `ValueError`) runs before the date is parsed. An entry with both problems therefore reports the amount error first. The parse happens inside the `Entry(...)` construction, before `entries.append`, so a bad date means nothing is appended and the list is left unchanged.\n\n**Two things the code doesn't do.**\n- **Non-string input:** `parse_date` calls `text.split`, so a non-string such as `None` raises `AttributeError` instead of `ValueError`.\n- **Malformed dates in reports:** `monthly_report` never sees a bad date string. Entries hold parsed `date` objects, so a malformed date can't reach the report.\n\nI didn't change any files.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.043004
  },
  {
   "run": "claude-sonnet-5-5__karpathy__T3__r3__4e7993",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "karpathy",
   "rep": 3,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 857,
   "input": 4,
   "cache_read": 26952,
   "cache_write": 7290,
   "warmup": null,
   "cache_write_1h": 7290,
   "cost_usd": 0.0431284,
   "wall_s": 7.4,
   "turns": 2,
   "prompt_last": 17526,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` is the entry point. It calls `entries_in_month`, which keeps only entries whose `day.year` and `day.month` both match the arguments. That filtered list is then used three ways:\n\n- `count` is the number of matching entries.\n- `total` is `total()`, a plain `sum` of every matching entry's `amount`.\n- `by_category` is `sum_by_category()`, which builds a dict from category to the running sum of amounts. Categories are already lowercased in `add_entry`, so `\"Food\"` and `\"food\"` land in the same bucket.\n\nAmounts are plain floats, so there is no rounding and no currency handling. Sums can pick up ordinary floating-point error. A month with no entries gives `{\"count\": 0, \"total\": 0, \"by_category\": {}}`. The `total` there is the integer `0`, because `sum` of an empty sequence returns `0`, not `0.0`.\n\n**Invalid date strings.** Dates are only parsed when an entry is added, in `add_entry`, which calls `parse_date`. `parse_date` splits the string on `-`. It raises `ValueError(f\"bad date: {text!r}\")` if there aren't exactly three parts or if any part isn't all digits. Examples are `\"2024/01/05\"`, `\"2024-1\"`, `\"2024-01-xx\"` and negative-looking parts.\n\nIf the shape passes, it builds `date(y, m, d)`. Values that are numeric but impossible, such as `\"2024-13-01\"` or `\"2024-02-30\"`, make `datetime.date` raise its own `ValueError` (for example \"month must be in 1..12\"). So callers see a `ValueError` either way, but the message differs.\n\nThe parse happens as an argument to `Entry(...)`, before `entries.append`. A bad date therefore raises before anything is added, and the list is left unchanged. The amount check runs even earlier, so a non-positive amount fails first.\n\nTwo smaller points:\n\n- `parse_date` doesn't strip whitespace, so `\" 2024-01-05\"` is rejected.\n- `str.isdigit()` accepts some non-ASCII digit characters. Those pass the check, and `int()` then converts them.\n\nBecause `monthly_report` only receives `Entry` objects, it never sees an invalid date string.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.043128
  },
  {
   "run": "claude-sonnet-5-5__placebo__T3__r1__92b526",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "placebo",
   "rep": 1,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 788,
   "input": 4,
   "cache_read": 26684,
   "cache_write": 7020,
   "warmup": null,
   "cache_write_1h": 7020,
   "cost_usd": 0.0413048,
   "wall_s": 8.9,
   "turns": 2,
   "prompt_last": 17256,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` is the entry point. It calls `entries_in_month`, which keeps only the entries whose `day.year` and `day.month` match the arguments. Then it builds a dictionary from that filtered list:\n- `count` is the number of entries in the month.\n- `total` comes from `total()`, which adds up the `amount` of each entry.\n- `by_category` comes from `sum_by_category()`, which keeps a running sum per category. Category names were lowercased when the entries were added, so \"Food\" and \"food\" land in the same bucket.\n\nIf nothing matches the month, you get `count` 0, `total` 0 and an empty `by_category`. No error is raised. `year` and `month` are compared directly, so an out-of-range month such as 13 just matches nothing.\n\n**Invalid date strings.** Dates are only parsed in `add_entry`, through `parse_date`. The report functions work with `date` objects that are already stored on each `Entry`.\n\n`parse_date` splits the text on `-`. It raises `ValueError(f\"bad date: ...\")` if the result isn't exactly three parts, or if any part isn't all digits. That catches things like `\"2024/01/05\"`, `\"2024-1\"`, `\"abc\"` and negative-looking parts. Parts that are digits but not a real date, such as `\"2024-13-40\"`, get past that check. The `date(y, m, d)` constructor then raises its own `ValueError` (for example \"month must be in 1..12\").\n\nNothing catches either error, so it goes up to the caller of `add_entry`. Two things follow from the order of the code in `add_entry`:\n- The `amount <= 0` check runs first, so a call with a bad amount and a bad date reports the amount error.\n- The date is parsed before `entries.append`, so a bad date never adds a partial entry and the list is left unchanged.\n\nTwo side cases. A non-string `text`, such as `None`, fails on `.split` with an `AttributeError` instead of a `ValueError`. And `isdigit()` accepts some non-ASCII digit characters, which `int()` will usually still convert.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.041305
  },
  {
   "run": "claude-sonnet-5-5__placebo__T3__r2__bded23",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "placebo",
   "rep": 2,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 848,
   "input": 4,
   "cache_read": 26685,
   "cache_write": 7021,
   "warmup": null,
   "cache_write_1h": 7021,
   "cost_usd": 0.041909,
   "wall_s": 9.7,
   "turns": 2,
   "prompt_last": 17257,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` in `ledger/core.py:51` does the work.\n\n1. It calls `entries_in_month`, which keeps only the entries whose `day.year` and `day.month` both match the arguments.\n2. On that filtered list it computes three things:\n   - `count`, the number of entries.\n   - `total`, the sum of every `amount`, from `total()`.\n   - `by_category`, a dict of per-category sums built by `sum_by_category`. Categories were lowercased when the entries were added, so \"Food\" and \"food\" land in the same bucket.\n3. It returns these as a dict with the keys `count`, `total` and `by_category`.\n\nIf no entries match the month, the result is `count` 0, `total` 0 (the integer, since `sum` of an empty sequence is 0) and an empty `by_category`. It doesn't raise an error.\n\n**Invalid date strings.** Dates are only parsed when an entry is added. `add_entry` calls `parse_date(day)`, and the monthly functions work on `date` objects that are already parsed.\n\n`parse_date` expects `YYYY-MM-DD`. It splits the text on `-`, and it raises `ValueError(f\"bad date: {text!r}\")` in either of these cases:\n- The split doesn't give exactly three parts.\n- Any part contains something other than digits.\n\nOtherwise it converts the parts to ints and builds a `date(y, m, d)`. If the digits are well-formed but not a real date, such as `2024-13-45`, the `date` constructor raises its own `ValueError` (for example \"month must be in 1..12\"). That message is less specific than the \"bad date\" one.\n\nTwo side effects of how this is written:\n- Nothing catches the error, so it propagates to the caller of `add_entry`.\n- `add_entry` checks `amount <= 0` before it parses the date, so an entry with both a bad amount and a bad date reports the amount error first. In every failure case the entry is never appended, so the ledger stays unchanged.\n\nTwo edge cases in the validation:\n- Non-string input such as `None` fails with an `AttributeError` on `.split`, not a `ValueError`.\n- `str.isdigit()` accepts some non-ASCII digit characters, and `int()` would convert them.\n\nI didn't change any files.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.041909
  },
  {
   "run": "claude-sonnet-5-5__placebo__T3__r3__800c1c",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "placebo",
   "rep": 3,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 732,
   "input": 4,
   "cache_read": 26688,
   "cache_write": 7025,
   "warmup": null,
   "cache_write_1h": 7025,
   "cost_usd": 0.0407656,
   "wall_s": 7.3,
   "turns": 2,
   "prompt_last": 17261,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` is the entry point. It first calls `entries_in_month`, which keeps only the entries whose `day.year` and `day.month` both match the arguments. On that filtered list it builds a dict with three values:\n- `count`: the number of matching entries.\n- `total`: the sum of their `amount` values, computed by `total`.\n- `by_category`: a dict from category to summed amount, computed by `sum_by_category`. Categories are lowercased when an entry is added, so \"Food\" and \"food\" land in the same bucket.\n\nIf no entries match the month, the report has a count of 0, a total of 0 and an empty `by_category`. The function does no rounding, so the amounts are plain float sums.\n\n**Invalid date strings.** Dates are only parsed when an entry is added, through `add_entry`, which calls `parse_date`. `parse_date` splits the text on `-`. It raises `ValueError(\"bad date: ...\")` if there aren't exactly three parts, or if any part isn't made up of digits. That check rejects things like `\"2024/01/05\"`, `\"2024-1\"`, `\"abc\"` and negative numbers. It also rejects an empty string, because `\"\".split(\"-\")` gives one part.\n\nStrings that pass the shape check but aren't real dates, such as `\"2024-13-01\"` or `\"2024-02-30\"`, make `date(y, m, d)` raise its own `ValueError`. Its message differs, for example \"month must be in 1..12\". A non-string such as `None` would fail with an `AttributeError` on `.split`, not a `ValueError`.\n\n`add_entry` checks the amount before it parses the date, so a non-positive amount raises first. Because `parse_date` runs while the `Entry` is being built, before `append`, an invalid date leaves the `entries` list unchanged. Nothing catches the error, so it reaches the caller. Invalid dates therefore never reach the monthly totals.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.040766
  },
  {
   "run": "claude-sonnet-5-5__ponytail__T3__r1__3bcccf",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "ponytail",
   "rep": 1,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 842,
   "input": 4,
   "cache_read": 26967,
   "cache_write": 7306,
   "warmup": null,
   "cache_write_1h": 7306,
   "cost_usd": 0.0430454,
   "wall_s": 8.4,
   "turns": 2,
   "prompt_last": 17542,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` is the entry point. It calls `entries_in_month`, which keeps only the entries whose `day.year` and `day.month` both match the arguments. It then builds a dict from that filtered list:\n- `count` is the number of matching entries.\n- `total` is `total()`, a plain `sum` of the `amount` floats.\n- `by_category` is `sum_by_category()`. It walks the entries and adds each amount into a dict keyed by the entry's category, which `add_entry` lowercased when the entry was created.\n\nThe amounts are floats and are summed as-is, with no rounding. Totals can therefore pick up small floating-point errors, such as 0.1 + 0.2 giving 0.30000000000000004. A month with no entries gives `count` 0, `total` 0 (an int, because `sum` of nothing is 0), and an empty `by_category`.\n\n**Invalid date strings.** Dates are only parsed in `add_entry`, through `parse_date`. `parse_date` splits the string on `-`. It raises `ValueError(\"bad date: ...\")` if there aren't exactly three parts or if any part isn't all digits. That check rejects `\"2024/01/05\"`, `\"2024-1\"`, `\"2024-01-xx\"` and negative-looking strings. Otherwise it converts the parts to ints and calls `date(y, m, d)`. That call raises its own `ValueError` for out-of-range values such as month 13 or Feb 30, with a different message (\"month must be in 1..12\" and similar).\n\nThe parts aren't length-checked, so `\"2024-1-5\"` is accepted even though the docstring says YYYY-MM-DD. Nothing catches the error. Because `parse_date` runs while the `Entry` is being built, the exception propagates out of `add_entry` before `entries.append` runs. The list is left unchanged, and the caller has to handle the error. The `amount <= 0` check runs first, so a bad amount is reported ahead of a bad date.\n\n`monthly_report` never sees invalid dates, since every stored `Entry.day` is already a real `date`. It doesn't validate its own `year` or `month` arguments, though. An out-of-range month such as 13 just matches nothing and returns an empty report.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.043045
  },
  {
   "run": "claude-sonnet-5-5__ponytail__T3__r2__aa7247",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "ponytail",
   "rep": 2,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 774,
   "input": 4,
   "cache_read": 26963,
   "cache_write": 7301,
   "warmup": null,
   "cache_write_1h": 7301,
   "cost_usd": 0.042344599999999996,
   "wall_s": 10.0,
   "turns": 2,
   "prompt_last": 17537,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` is the entry point. It calls `entries_in_month`, which keeps the entries whose `day.year` and `day.month` both match. It then builds a dict from that filtered list with three keys:\n- `count` is the number of entries in the month.\n- `total` is the sum of their `amount` values, computed by `total()`.\n- `by_category` is a dict of per-category sums, computed by `sum_by_category()`. Categories were lowercased when the entry was added, so `Food` and `food` land in the same bucket.\n\nThe amounts are plain floats, so the sums can pick up ordinary floating-point rounding error. Nothing rounds them. If the month has no entries, you get `count` 0, `total` 0 and an empty `by_category`.\n\n**Invalid dates.** Dates are only parsed in `add_entry`, through `parse_date`. `monthly_report` never sees a date string, because `Entry.day` is already a `date` object by then.\n\n`parse_date` splits the text on `-`. It raises `ValueError(\"bad date: ...\")` if the result isn't exactly three parts or any part contains a non-digit character. That covers `\"2024/01/05\"`, `\"2024-1\"`, `\"abc\"` and negative-looking pieces. If the text passes that check, it calls `date(y, m, d)`. That call raises its own `ValueError` for impossible values such as `\"2024-13-01\"` or `\"2024-02-30\"`, with a message like \"month must be in 1..12\". So every kind of bad date ends in a `ValueError`, but the message differs depending on which check failed.\n\nTwo details matter for `add_entry`:\n- **Order of checks:** it validates `amount <= 0` first and parses the date second. A bad date therefore never gets as far as `entries.append`.\n- **Failure behavior:** the exception propagates to the caller, and the ledger list is left unchanged. There is no catching or fallback.\n\n`parse_date` would also raise `AttributeError` if `text` weren't a string, for example `None`. That isn't handled either.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.042345
  },
  {
   "run": "claude-sonnet-5-5__ponytail__T3__r3__88e026",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "ponytail",
   "rep": 3,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 718,
   "input": 4,
   "cache_read": 26960,
   "cache_write": 7297,
   "warmup": null,
   "cache_write_1h": 7297,
   "cost_usd": 0.041768,
   "wall_s": 11.1,
   "turns": 2,
   "prompt_last": 17533,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` is the entry point.\n\n1. It calls `entries_in_month`, which keeps only the entries whose `day.year` and `day.month` both match the arguments.\n2. It returns a dict with three values:\n   - `count`: the number of entries in that month.\n   - `total`: the sum of their `amount` values, from `total()`.\n   - `by_category`: a dict of per-category sums, from `sum_by_category()`. It starts each category at 0.0 and adds amounts as it goes.\n\nCategories are lowercased when an entry is added, so \"Food\" and \"food\" end up in the same bucket. A month with no entries gives `count` 0, `total` 0 (an int, because `sum` of an empty sequence is 0) and an empty `by_category`.\n\n**Invalid date strings.** Dates are parsed only when an entry is added. `add_entry` calls `parse_date(day)`, which splits the string on `-`. It raises `ValueError(f\"bad date: ...\")` if either of these is true:\n- the string doesn't have exactly three parts\n- any part isn't all digits\n\nTwo other cases are worth knowing about:\n- **Out-of-range values:** a string like `2024-13-45` passes those checks. `date(y, m, d)` then raises its own `ValueError` (\"month must be in 1..12\" or similar). The caller gets a `ValueError` either way, but the message differs.\n- **Non-string input:** `None` or another non-string fails on `.split` with an `AttributeError`, not a `ValueError`.\n\nBecause `parse_date` is called inside the `Entry(...)` construction, which is inside `entries.append(...)`, a bad date raises before anything is appended. The ledger is never left with a partial entry.\n\nThe `amount <= 0` check also runs first, so an entry with both a bad amount and a bad date reports the amount error.\n\nI didn't change any files.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "field": false,
   "api_cost": 0.041768
  },
  {
   "run": "gpt-5.6-luna__control__T3__r1__45aa28",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "control",
   "rep": 1,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 357,
   "input": 20964,
   "cache_read": 17920,
   "cache_write": 12572,
   "warmup": 9528,
   "cache_write_1h": null,
   "cost_usd": 0.003301,
   "wall_s": 13.3,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 115,
   "commands": 1,
   "answer": "Monthly totals are computed by `monthly_report`: it first keeps entries whose `day.year` and `day.month` match the requested year and month. It then sums their `amount` values, reports the matching entry count, and provides category subtotals.\n\nDate strings are parsed by splitting on `-`. The function raises `ValueError` if there are not exactly three numeric parts or if the resulting date is invalid (for example, month 13). When used through `add_entry`, that error propagates and the entry is not added. Zero-padding is not actually required, so strings like `2026-9-3` are accepted.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "extras_off": true,
   "field": true,
   "api_cost": 0.003301
  },
  {
   "run": "gpt-5.6-luna__control__T3__r2__483133",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "control",
   "rep": 2,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 302,
   "input": 20955,
   "cache_read": 11776,
   "cache_write": 18707,
   "warmup": 9528,
   "cache_write_1h": null,
   "cost_usd": 0.004339,
   "wall_s": 14.1,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 80,
   "commands": 1,
   "answer": "Monthly totals are computed by:\n\n- Filtering entries whose `day.year` and `day.month` match the requested year and month.\n- Counting those entries.\n- Summing their `amount` values.\n- Grouping amounts by category in `by_category`.\n\nThe report returns `{\"count\": ..., \"total\": ..., \"by_category\": ...}`.\n\nAn invalid date string raises `ValueError`. This includes malformed formats, non-numeric components, or impossible calendar dates. When used through `add_entry`, the exception propagates and the entry is not appended.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "extras_off": true,
   "field": true,
   "api_cost": 0.004339
  },
  {
   "run": "gpt-5.6-sol__control__T3__r1__594251",
   "model": "gpt-5.6-sol",
   "host": "codex",
   "task": "T3",
   "skill": "control",
   "rep": 1,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 358,
   "input": 23902,
   "cache_read": 11520,
   "cache_write": 22835,
   "warmup": 10453,
   "cache_write_1h": null,
   "cost_usd": 0.103108,
   "wall_s": 22.5,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 127,
   "commands": 1,
   "answer": "`monthly_report(entries, year, month)`:\n\n- Filters entries whose `day.year` and `day.month` match.\n- Returns the matching entry count.\n- Sums their amounts for `total`.\n- Groups and sums amounts by category for `by_category`.\n\nAn invalid date causes `parse_date()` to raise `ValueError`. Malformed strings produce `bad date: <value>`; impossible calendar dates (such as `2025-02-30`) raise `ValueError` from Python\u2019s `date` constructor. The exception propagates, and `add_entry()` does not append an entry.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "extras_off": true,
   "field": true,
   "api_cost": 0.103108
  },
  {
   "run": "gpt-5.6-sol__control__T3__r2__92c332",
   "model": "gpt-5.6-sol",
   "host": "codex",
   "task": "T3",
   "skill": "control",
   "rep": 2,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 570,
   "input": 36911,
   "cache_read": 23552,
   "cache_write": 23812,
   "warmup": 10453,
   "cache_write_1h": null,
   "cost_usd": 0.116069,
   "wall_s": 36.6,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 148,
   "commands": 2,
   "answer": "`monthly_report` first filters entries whose `day.year` and `day.month` exactly match the requested year and month. It then returns:\n\n- `count`: number of matching entries\n- `total`: sum of their `amount` values; `0` if none match\n- `by_category`: subtotals grouped by the entries\u2019 normalized category names\n\nSee [ledger/core.py](/tmp/bench-work/gpt-5.6-sol__control__T3__r2__92c332/ledger/core.py:40).\n\nDates are split on `-` and must contain exactly three numeric parts. Bad structure or nonnumeric parts raise `ValueError(\"bad date: ...\")`. Numeric but impossible dates\u2014such as `2026-02-30` or month `13`\u2014raise the standard `ValueError` from `datetime.date`. When this occurs through `add_entry`, the exception propagates and no entry is appended. The code does not strictly enforce field widths, so a string such as `2026-2-3` is accepted despite the `YYYY-MM-DD` docstring.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "extras_off": true,
   "field": true,
   "api_cost": 0.116069
  },
  {
   "run": "gpt-6-astra__control__T3__r1__ff434e",
   "model": "gpt-6-astra",
   "host": "codex",
   "task": "T3",
   "skill": "control",
   "rep": 1,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 260,
   "input": 26386,
   "cache_read": 19968,
   "cache_write": 17870,
   "warmup": 11452,
   "cache_write_1h": null,
   "cost_usd": 0.211668,
   "wall_s": 16.8,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "In [ledger/core.py](/tmp/bench-work/gpt-6-astra__control__T3__r1__ff434e/ledger/core.py), `monthly_report` selects entries whose date matches the requested year and month. It returns their count, the sum of their amounts, and totals grouped by category. If none match, it returns a count of `0`, a total of `0`, and an empty category dictionary.\n\n`parse_date` splits the string on hyphens and requires three digit-only parts; otherwise, it raises `ValueError` with a `bad date` message. It converts those parts to year, month, and day, then uses Python\u2019s `date` constructor, which also raises `ValueError` for impossible dates. It does not strictly enforce field widths, so `2026-9-3` is accepted. When called through `add_entry`, an invalid date propagates the exception and no entry is appended.\n\nNo files were changed.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "extras_off": true,
   "field": true,
   "api_cost": 0.211668
  },
  {
   "run": "gpt-6-astra__control__T3__r2__c60535",
   "model": "gpt-6-astra",
   "host": "codex",
   "task": "T3",
   "skill": "control",
   "rep": 2,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 314,
   "input": 39809,
   "cache_read": 36992,
   "cache_write": 14269,
   "warmup": 11452,
   "cache_write_1h": null,
   "cost_usd": 0.195382,
   "wall_s": 16.4,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 2,
   "answer": "In [ledger/core.py](/tmp/bench-work/gpt-6-astra__control__T3__r2__c60535/ledger/core.py:51), `monthly_report` selects entries whose date matches both the requested year and month. It returns their count, the sum of their dollar amounts, and a dictionary summing amounts by category. With no matching entries, these are `0`, `0`, and `{}`.\n\n`parse_date` splits the date string on hyphens and requires three digit-only parts. Otherwise, it raises `ValueError` with `bad date: <repr of input>`. It converts the parts to year, month, and day, then calls Python\u2019s `date` constructor, which also raises `ValueError` for impossible dates. It does not enforce fixed widths, so `2026-9-3` is accepted. When parsing fails during `add_entry`, the exception propagates and no entry is appended.\n\nNo files were changed.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "extras_off": true,
   "field": true,
   "api_cost": 0.195382
  },
  {
   "run": "gpt-6-luna__control__T3__r1__2106c9",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "control",
   "rep": 1,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 158,
   "input": 25051,
   "cache_read": 22016,
   "cache_write": 13859,
   "warmup": 10824,
   "cache_write_1h": null,
   "cost_usd": 0.001685,
   "wall_s": 7.4,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "`monthly_report` first filters entries to those whose year and month match the requested values. It returns the number of matching entries, their summed amounts, and the summed amount for each category.\n\nWhen an invalid date string is passed to `add_entry`, `parse_date` raises `ValueError`\u2014for malformed strings or dates that are not valid calendar dates. The entry is not appended.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "extras_off": true,
   "field": true,
   "api_cost": 0.001685
  },
  {
   "run": "gpt-6-luna__control__T3__r2__b1bef4",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "control",
   "rep": 2,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 158,
   "input": 25044,
   "cache_read": 18944,
   "cache_write": 16924,
   "warmup": 10824,
   "cache_write_1h": null,
   "cost_usd": 0.001961,
   "wall_s": 11.7,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "`monthly_report` first selects entries whose date\u2019s year and month match the requested year and month. It returns their count, the sum of their amounts, and totals grouped by category. The total is a straightforward sum of the selected entries\u2019 amounts.\n\nAn invalid date string passed to `parse_date` raises `ValueError`. `add_entry` calls `parse_date` before appending, so that entry is not added.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "extras_off": true,
   "field": true,
   "api_cost": 0.001961
  },
  {
   "run": "gpt-6-sol__control__T3__r1__92552b",
   "model": "gpt-6-sol",
   "host": "codex",
   "task": "T3",
   "skill": "control",
   "rep": 1,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 307,
   "input": 25434,
   "cache_read": 19456,
   "cache_write": 16981,
   "warmup": 11003,
   "cache_write_1h": null,
   "cost_usd": 0.040923,
   "wall_s": 20.7,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 105,
   "commands": 1,
   "answer": "`monthly_report` selects entries whose date has the requested year and month. It returns their count, the sum of their amounts, and a separate sum for each category. If there are no matching entries, the count and total are zero and the category breakdown is empty.\n\nWhen `add_entry` receives an invalid date string, `parse_date` raises a `ValueError`, either for malformed text or for an impossible calendar date. The error is not caught, so no entry is appended.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "extras_off": true,
   "field": true,
   "api_cost": 0.040923
  },
  {
   "run": "gpt-6-sol__control__T3__r2__affff7",
   "model": "gpt-6-sol",
   "host": "codex",
   "task": "T3",
   "skill": "control",
   "rep": 2,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 340,
   "input": 25441,
   "cache_read": 19456,
   "cache_write": 16988,
   "warmup": 11003,
   "cache_write_1h": null,
   "cost_usd": 0.041267,
   "wall_s": 23.5,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 95,
   "commands": 1,
   "answer": "`monthly_report(entries, year, month)` selects entries whose date has that year and month. It returns their count, a total computed by summing their amounts, and separate sums by category. See [ledger/core.py](/tmp/bench-work/gpt-6-sol__control__T3__r2__affff7/ledger/core.py:44).\n\nWhen `add_entry` receives an invalid date string, `parse_date` raises `ValueError`. It rejects strings that do not have three numeric parts separated by hyphens, and Python\u2019s `date` constructor rejects invalid calendar dates. The error propagates, so the entry is not appended.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "extras_off": true,
   "field": true,
   "api_cost": 0.041267
  },
  {
   "run": "gpt-6.1-sol__control__T3__r1__cc2c12",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T3",
   "skill": "control",
   "rep": 1,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 286,
   "input": 39789,
   "cache_read": 35584,
   "cache_write": 15721,
   "warmup": 11516,
   "cache_write_1h": null,
   "cost_usd": 0.03786,
   "wall_s": 46.5,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 2,
   "answer": "In [ledger/core.py](/tmp/bench-work/gpt-6.1-sol__control__T3__r1__cc2c12/ledger/core.py), `monthly_report` selects entries whose dates match the requested year and month. It returns their count, the sum of their amounts, and totals grouped by category. With no matching entries, these are `0`, `0`, and an empty dictionary.\n\n`parse_date` requires three hyphen-separated digit strings, then passes their integer values to Python\u2019s `date` constructor. Malformed strings raise `ValueError` with a `bad date` message; impossible dates, such as February 30, raise `ValueError` from the constructor. Despite the docstring\u2019s `YYYY-MM-DD` wording, digit widths are not enforced. When called through `add_entry`, an invalid date prevents the entry from being appended.\n\nNo files were changed.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "extras_off": true,
   "field": true,
   "api_cost": 0.03786
  },
  {
   "run": "gpt-6.1-sol__control__T3__r2__423070",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T3",
   "skill": "control",
   "rep": 2,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 293,
   "input": 39773,
   "cache_read": 37120,
   "cache_write": 14169,
   "warmup": 11516,
   "cache_write_1h": null,
   "cost_usd": 0.03498,
   "wall_s": 23.2,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 2,
   "answer": "In [ledger/core.py](/tmp/bench-work/gpt-6.1-sol__control__T3__r2__423070/ledger/core.py), `monthly_report` selects entries whose date matches the requested year and month. It returns their count, the sum of their amounts, and totals grouped by category. With no matching entries, these are `0`, `0`, and `{}`.\n\n`parse_date` splits the string on hyphens and requires three digit-only parts. If that check fails, it raises `ValueError` with `bad date: <repr of input>`. Otherwise, it converts the parts to integers and calls `datetime.date`, which raises `ValueError` for impossible dates, such as February 30. It does not enforce fixed-width `YYYY-MM-DD` formatting.\n\nIf date parsing fails in `add_entry`, the exception propagates and no entry is appended. No files were changed.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "extras_off": true,
   "field": true,
   "api_cost": 0.03498
  }
 ],
 "ste_runs": [
  {
   "run": "claude-opus-5-5__ste__T3__r1__674370",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "ste",
   "rep": 1,
   "round": 16,
   "date": "2026-10-02",
   "exit": 0,
   "passed": true,
   "output": 1568,
   "input": 4,
   "cache_read": 26041,
   "cache_write": 6597,
   "warmup": null,
   "cache_write_1h": 6597,
   "cost_usd": 0.0893602,
   "wall_s": 16.9,
   "turns": 2,
   "prompt_last": 16721,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 450,
   "commands": null,
   "answer": "## How the code calculates monthly totals\n\nThe `monthly_report` function (`ledger/core.py:51`) makes the monthly report. It uses three steps:\n\n1. **Select the entries for the month.** `entries_in_month` (`ledger/core.py:40`) examines each entry. It keeps an entry only when the year and the month of its date are the same as the given `year` and `month`. The day of the month has no effect.\n2. **Calculate the grand total.** `total` (`ledger/core.py:31`) adds the `amount` values of the selected entries.\n3. **Calculate a total for each category.** `sum_by_category` (`ledger/core.py:44`) makes a dictionary. Each key is a category name. Each value is the sum of the amounts in that category. The first amount in a category starts at `0.0`.\n\nThe function returns a dictionary with three items:\n\n- `\"count\"`: the number of entries in the month.\n- `\"total\"`: the sum of all amounts in the month.\n- `\"by_category\"`: the dictionary of totals for each category.\n\nSome details:\n\n- `add_entry` changes each category name to lowercase before it keeps the entry. Thus, \"Food\" and \"food\" go into the same category total.\n- If no entries are in the month, `\"count\"` is `0`, `\"by_category\"` is `{}`, and `\"total\"` is the integer `0`, not `0.0`. This is because `sum` of an empty sequence gives `0`.\n- The amounts are floats. Thus, a total can have small rounding errors, for example `0.30000000000000004`.\n\n## What happens when a date string is not valid\n\n`parse_date` (`ledger/core.py:15`) does the check in two stages.\n\n**Stage 1: the format check.** The function divides the text at each hyphen (`-`). It raises `ValueError(\"bad date: ...\")` if one of these conditions is true:\n\n- The result does not have exactly three parts. Examples: `\"2024/01/05\"` or `\"2024-01\"`.\n- A part is not all digits. Examples: `\"2024-Jan-05\"`, `\" 2024-01-05\"`, or `\"2024--05\"`. An empty part is not all digits.\n\n**Stage 2: the calendar check.** The function changes the three parts to integers. Then it gives them to `datetime.date`. If the date does not exist, `date` raises its own `ValueError`. Examples: `\"2024-02-30\"`, `\"2024-13-01\"`, or `\"0000-01-01\"`. This message is different from the stage 1 message. For example, the message can be \"day is out of range for month\".\n\nSome details:\n\n- The function does not require two digits for the month and the day. Thus, it accepts `\"2024-1-5\"`.\n- If `text` is not a string (for example, `None`), the function raises `AttributeError`, not `ValueError`.\n\n**The effect on `add_entry`:** `add_entry` (`ledger/core.py:24`) calls `parse_date` before it adds the entry to the list. If the date is not valid, the error goes to the caller, and the list does not change. But `add_entry` checks the amount first. Thus, if the amount and the date are both bad, you get the \"amount must be positive\" error.\n\nI did not change any files.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "api_cost": 0.08936
  },
  {
   "run": "claude-opus-5-5__ste__T3__r2__f3a784",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "ste",
   "rep": 2,
   "round": 16,
   "date": "2026-10-02",
   "exit": 0,
   "passed": true,
   "output": 1571,
   "input": 4,
   "cache_read": 26049,
   "cache_write": 6607,
   "warmup": null,
   "cache_write_1h": 6607,
   "cost_usd": 0.08950179999999999,
   "wall_s": 17.9,
   "turns": 2,
   "prompt_last": 16731,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 469,
   "commands": null,
   "answer": "## How the code calculates monthly totals\n\nThe function `monthly_report(entries, year, month)` (`ledger/core.py:51`) makes the report for one month. It does these steps:\n\n1. It calls `entries_in_month` (`ledger/core.py:40`). This function keeps each entry whose `day.year` is equal to `year` and whose `day.month` is equal to `month`. It does not change the list that you give it. It makes a new list.\n2. It counts the entries in the new list. This value is `count`.\n3. It calls `total` (`ledger/core.py:31`). This function adds the `amount` of all entries in the new list. This value is `total`.\n4. It calls `sum_by_category` (`ledger/core.py:44`). This function makes a dictionary. Each key is a category. Each value is the sum of the amounts for that category. This value is `by_category`.\n5. It returns a dictionary with the three values: `count`, `total`, and `by_category`.\n\nSome special conditions:\n\n- If no entries are in the month, `count` is 0, `total` is the integer `0` (not `0.0`), and `by_category` is an empty dictionary.\n- The amounts are floats. Thus, some sums can have small rounding errors. For example, `0.1 + 0.2` gives `0.30000000000000004`.\n- `add_entry` changes each category to lowercase. Thus, \"Food\" and \"food\" go into the same category.\n- The month filter uses only the `date` object in each entry. It does not read date strings. The code parses date strings only when you add an entry.\n\n## What occurs when a date string is not valid\n\nThe function `parse_date` (`ledger/core.py:15`) parses the date string. It does these checks:\n\n1. It divides the string at each \"-\" character.\n2. If the result does not have exactly three parts, it raises `ValueError` with the message `bad date: '<text>'`.\n3. If a part has a character that is not a digit, it raises the same `ValueError`. An empty part also causes this error, for example `\"2024--05\"`. Spaces also cause this error, for example `\" 2024-01-05\"`.\n4. It changes the three parts to integers. Then it calls `date(y, m, d)`.\n5. If the numbers do not make a real date, `date()` raises `ValueError`. Examples are `\"2024-13-01\"` and `\"2023-02-29\"`. This error has a different message, for example `month must be in 1..12`. The code does not change this message to the \"bad date\" message.\n\nThe function `add_entry` calls `parse_date` before it appends the entry. Thus, if the date is not valid, the code does not add the entry, and the list does not change. The module does not catch the `ValueError`. The error goes to the code that called `add_entry`.\n\nSome limits of the check:\n\n- The amount check comes first in `add_entry`. If both the amount and the date are bad, you get only the amount error.\n- The code does not check the number of digits in each part. Thus, it accepts `\"24-1-5\"` as the year 24, January 5.\n- A date is checked only when it goes through `add_entry`. If you make an `Entry` directly, no check occurs.\n\nI did not change any files.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "api_cost": 0.089502
  },
  {
   "run": "claude-opus-5-5__ste__T3__r3__3f7ac6",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "ste",
   "rep": 3,
   "round": 16,
   "date": "2026-10-02",
   "exit": 0,
   "passed": true,
   "output": 1579,
   "input": 4,
   "cache_read": 26052,
   "cache_write": 6612,
   "warmup": null,
   "cache_write_1h": 6612,
   "cost_usd": 0.0897024,
   "wall_s": 17.6,
   "turns": 2,
   "prompt_last": 16736,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 478,
   "commands": null,
   "answer": "## How the code calculates monthly totals\n\nThe function `monthly_report(entries, year, month)` at `ledger/core.py:51` makes the monthly report. It does these steps:\n\n1. It calls `entries_in_month` (`ledger/core.py:40`). This function keeps only the entries whose `day` has the same year and the same month as the arguments. It does not change the original list.\n2. It counts the entries that stay. This value goes into the `\"count\"` key.\n3. It calls `total` (`ledger/core.py:31`). This function adds the `amount` of each entry that stays. The result goes into the `\"total\"` key.\n4. It calls `sum_by_category` (`ledger/core.py:44`). This function makes a dictionary. Each key is a category name. Each value is the sum of the amounts for that category. The result goes into the `\"by_category\"` key.\n\nThe function then returns a dictionary with these three keys.\n\nSome notes on this behavior:\n\n- `add_entry` changes each category name to lowercase. Thus, \"Food\" and \"food\" go into the same category.\n- If no entries are in the month, `\"total\"` is the integer `0`, not the float `0.0`. This occurs because `sum()` starts at `0`.\n- The code does not examine the `year` and `month` arguments. If you give a month of 13, or a year as a string such as `\"2024\"`, no entry agrees. The function then returns an empty report and does not show an error.\n- The amounts are floats. Thus, the totals can have small rounding errors, for example `0.30000000000000004`.\n\n## What occurs when a date string is not valid\n\nThe function `parse_date(text)` at `ledger/core.py:15` reads the date. It does these checks:\n\n1. It divides the text at each `-` character.\n2. If the result does not have exactly three parts, it raises `ValueError(\"bad date: '...'\")`.\n3. If a part is empty or has a character that is not a digit, it raises the same `ValueError`. For example, `\"2024/01/05\"`, `\"2024-1a-05\"`, and `\"2024--05\"` all cause this error.\n4. It changes the three parts to integers. It then gives them to `date(y, m, d)`.\n5. If the numbers do not make a real date, `date()` raises its own `ValueError`. For example, `\"2024-13-01\"` and `\"2023-02-30\"` cause this error. This message is different from the `\"bad date\"` message.\n\nThe code does not catch these errors. Thus, the `ValueError` goes up to the code that called `add_entry`. If the date is not valid, `add_entry` does not add an entry, and the list stays the same. This is because Python calculates `parse_date(day)` before it calls `append`.\n\nSome limits of the check:\n\n- The check does not make sure that the text has exactly the `YYYY-MM-DD` format. For example, the code accepts `\"24-1-5\"` as January 5 of the year 24.\n- `add_entry` examines the amount before it examines the date. If both values are bad, you get the amount error.\n- If `day` is not a string, for example `None`, the code raises `AttributeError`, not `ValueError`.\n\nI did not change any files.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "api_cost": 0.089702
  },
  {
   "run": "claude-sonnet-5__ste__T3__r1__97d545",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "ste",
   "rep": 1,
   "round": 16,
   "date": "2026-10-02",
   "exit": 0,
   "passed": true,
   "output": 1122,
   "input": 6,
   "cache_read": 75018,
   "cache_write": 10493,
   "warmup": null,
   "cache_write_1h": 10493,
   "cost_usd": 0.0682076,
   "wall_s": 15.1,
   "turns": 3,
   "prompt_last": 29134,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 447,
   "commands": null,
   "answer": "## How Monthly Totals Are Computed\n\nThe function `monthly_report` makes the monthly total. It does these steps:\n\n1. The function `entries_in_month` selects entries. It selects only entries where the year and the month match the given values.\n2. The function `total` adds the amounts of the selected entries. The result is the sum amount for the month.\n3. The function `sum_by_category` groups the selected entries by category. It adds the amounts in each category.\n\n`monthly_report` sends back a result. The result has three parts: the count of entries, the total amount, and the amounts by category.\n\n## What Happens When a Date String Is Invalid\n\nThe function `parse_date` checks the date string. The function does these checks:\n\n1. The function splits the string at each dash.\n2. The function checks that there are exactly three parts.\n3. The function checks that each part has only digits.\n\nIf a check fails, the function raises a `ValueError`. The error message shows the bad date string.\n\nIf all checks pass, the function converts the three parts to numbers. The function sends the numbers to the `date` constructor. If the numbers do not make a real date \u2014 for example, month 13, or day 32 \u2014 the `date` constructor also raises a `ValueError`.\n\nA caller usually sees this error through `add_entry`. This function calls `parse_date` first. If `parse_date` raises a `ValueError`, `add_entry` does not catch the error. The error goes up to the caller. The entry is not added to the list.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "api_cost": 0.068208
  },
  {
   "run": "claude-sonnet-5__ste__T3__r2__129c77",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "ste",
   "rep": 2,
   "round": 16,
   "date": "2026-10-02",
   "exit": 0,
   "passed": true,
   "output": 965,
   "input": 6,
   "cache_read": 74828,
   "cache_write": 10396,
   "warmup": null,
   "cache_write_1h": 10396,
   "cost_usd": 0.06621160000000001,
   "wall_s": 12.4,
   "turns": 3,
   "prompt_last": 29037,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 344,
   "commands": null,
   "answer": "## How the Code Works\n\n**Monthly totals.**\nThe function `monthly_report` makes the monthly total.\nThe function does three steps.\n\n1. The function `entries_in_month` finds all entries for the given year and month. The function compares the year and the month of each entry date with the input values.\n2. The function `total` adds the `amount` value of each found entry. The result is the monthly total.\n3. The function `sum_by_category` adds the amounts again, but it groups the amounts by category. The function makes a dictionary. Each key is a category name. Each value is the sum of the amounts for that category.\n\nThe function `monthly_report` gives back a dictionary. The dictionary has the entry count, the total, and the sums by category.\n\n**Invalid date strings.**\nThe function `parse_date` checks the date text before it makes a date.\n\nThe function does two checks:\n- The text must split into exactly three parts at each \"-\" character.\n- Each part must have only digit characters.\n\nIf a check fails, the function raises a `ValueError`. The error message shows the bad text.\n\nThe function `add_entry` calls `parse_date` to convert the day argument. If `parse_date` raises a `ValueError`, `add_entry` does not catch the error. The error goes up to the caller. The new entry is not added to the list.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "api_cost": 0.066212
  },
  {
   "run": "claude-sonnet-5__ste__T3__r3__5fd808",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "ste",
   "rep": 3,
   "round": 16,
   "date": "2026-10-02",
   "exit": 0,
   "passed": true,
   "output": 1079,
   "input": 6,
   "cache_read": 75200,
   "cache_write": 10582,
   "warmup": null,
   "cache_write_1h": 10582,
   "cost_usd": 0.06817000000000001,
   "wall_s": 11.3,
   "turns": 3,
   "prompt_last": 29227,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 427,
   "commands": null,
   "answer": "## How Monthly Totals Work\n\nThe function `monthly_report` calculates the monthly total. It does these steps:\n\n1. It calls `entries_in_month`. This function selects the entries that match the year and the month.\n2. It calls `total`. This function adds the `amount` value of each selected entry.\n3. The result is the monthly total.\n\n`monthly_report` also calls `sum_by_category`. This function adds the amounts for each category. It gives a total amount for each category.\n\n## What Happens When a Date String Is Invalid\n\nThe function `parse_date` checks the date text. The date text must have this format: YYYY-MM-DD.\n\n`parse_date` splits the text at each \"-\" character. The split must give three parts. Each part must have only digits.\n\nIf the text does not meet these conditions, `parse_date` raises a `ValueError`. The error message shows the bad text.\n\nThe function `add_entry` calls `parse_date` when you add a new entry. If you give an invalid date string, `add_entry` raises the error at that time. The invalid entry is not added to the ledger.\n\n**Note:** `monthly_report` does not parse date strings. It only reads entries that are already in the ledger. These entries already have valid dates, because `add_entry` checked them before. So an invalid date cannot cause an error during the monthly total calculation. The error can only occur when you add the entry.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "api_cost": 0.06817
  },
  {
   "run": "claude-sonnet-5-5__ste__T3__r1__d65598",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "ste",
   "rep": 1,
   "round": 16,
   "date": "2026-10-02",
   "exit": 0,
   "passed": true,
   "output": 726,
   "input": 4,
   "cache_read": 26120,
   "cache_write": 6447,
   "warmup": null,
   "cache_write_1h": 6447,
   "cost_usd": 0.038279999999999995,
   "wall_s": 7.3,
   "turns": 2,
   "prompt_last": 16687,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**How monthly totals are computed**\n\nThe function `monthly_report(entries, year, month)` does the work. It uses these steps:\n\n1. It calls `entries_in_month`. This function keeps only the entries where `day.year` equals `year` and `day.month` equals `month`.\n2. It counts the kept entries.\n3. It calls `total` on the kept entries. `total` adds up the `amount` of each entry.\n4. It calls `sum_by_category` on the kept entries. This function makes a dictionary. Each key is a category. Each value is the sum of the amounts for that category.\n\nThe function returns a dictionary with three keys: `count`, `total`, and `by_category`. If no entry is in the month, the count is 0, the total is 0, and `by_category` is empty. The function does not change the list of entries.\n\n**What happens when a date string is invalid**\n\nDates enter the ledger through `add_entry`. This function calls `parse_date` on the `day` string.\n\n`parse_date` splits the string at each \"-\". It raises `ValueError` with the message `bad date: '<text>'` in these cases:\n\n- The string does not have exactly three parts.\n- Any part has a character that is not a digit. For example, \"2024-1-x\" fails, and a negative number fails.\n\nIf the format is correct but the date does not exist, the `date(y, m, d)` call raises its own `ValueError`. For example, \"2024-02-30\" or \"2024-13-01\" gives this error. The message is different from the one in `parse_date`.\n\nTwo other points are important:\n\n- `parse_date` does not catch these errors. They go up to the caller of `add_entry`.\n- `add_entry` checks the amount first, and only then parses the date. A bad amount raises an error before the date is checked. Because the parse happens inside the `append` call, a bad date adds no entry to the list.\n\nAlso, `parse_date` has no handling for non-string input. If `text` is `None`, the call fails with `AttributeError`, not `ValueError`.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "api_cost": 0.03828
  },
  {
   "run": "claude-sonnet-5-5__ste__T3__r2__7f6829",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "ste",
   "rep": 2,
   "round": 16,
   "date": "2026-10-02",
   "exit": 0,
   "passed": true,
   "output": 779,
   "input": 4,
   "cache_read": 26122,
   "cache_write": 6450,
   "warmup": null,
   "cache_write_1h": 6450,
   "cost_usd": 0.0388224,
   "wall_s": 10.4,
   "turns": 2,
   "prompt_last": 16690,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**How monthly totals are computed**\n\nThe function `monthly_report(entries, year, month)` does the work. It takes three steps:\n\n1. It calls `entries_in_month`. This function keeps only the entries where `day.year` and `day.month` match the values you give.\n2. It calls `total` on the kept entries. `total` adds the `amount` of each entry.\n3. It calls `sum_by_category` on the kept entries. This function makes a dictionary. Each key is a category. Each value is the sum of the amounts for that category.\n\nThe function returns a dictionary with three keys:\n- `count`: the number of entries in the month.\n- `total`: the sum of all amounts in the month.\n- `by_category`: the per-category sums.\n\nIf no entry matches the month, `count` is 0, `total` is 0, and `by_category` is an empty dictionary.\n\n**What happens when a date string is invalid**\n\nThe function `parse_date` reads the date text. It does these checks:\n\n- It splits the text at each `-`. It must get exactly three parts.\n- Each part must contain only digits.\n\nIf a check fails, `parse_date` raises `ValueError` with the message `bad date: <text>`.\n\nA second kind of failure can occur. The text can have the correct form but a date that does not exist, for example `2024-02-30` or `2024-13-01`. In this case the `date(y, m, d)` call raises its own `ValueError`. Its message is different from the `bad date` message.\n\nA negative number also fails the digit check, because the `-` sign splits the text into more than three parts. Also, `parse_date` does not check the length of each part. For example, `2024-1-5` is accepted.\n\n`add_entry` calls `parse_date`. It does this when it builds the new `Entry`. If the date is invalid, the `ValueError` goes up to the caller. The function does not catch it. The entry is not added to the list. The amount check happens first, so a non-positive amount raises its own `ValueError` before the date is read.\n\nMonthly reports are not affected by bad dates. Every `Entry` already holds a valid `date` object, because `add_entry` rejects bad dates when the entry is created.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "api_cost": 0.038822
  },
  {
   "run": "claude-sonnet-5-5__ste__T3__r3__87eb7c",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "ste",
   "rep": 3,
   "round": 16,
   "date": "2026-10-02",
   "exit": 0,
   "passed": true,
   "output": 733,
   "input": 4,
   "cache_read": 26126,
   "cache_write": 6455,
   "warmup": null,
   "cache_write_1h": 6455,
   "cost_usd": 0.0383832,
   "wall_s": 8.3,
   "turns": 2,
   "prompt_last": 16695,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**How monthly totals are computed**\n\nThe function `monthly_report` takes a list of entries, a year, and a month. It does these steps:\n\n1. It calls `entries_in_month`. This function keeps only the entries where the year and the month of `e.day` match the values you give.\n2. It counts the kept entries. This is the `count` value.\n3. It calls `total` on the kept entries. `total` adds the `amount` of each entry. This is the `total` value.\n4. It calls `sum_by_category` on the kept entries. This function makes a dictionary. Each key is a category name. Each value is the sum of the amounts for that category. This is the `by_category` value.\n\nThe function returns a dictionary with the keys `count`, `total`, and `by_category`. If no entry matches the month, `count` is 0 and `total` is 0. `by_category` is an empty dictionary.\n\n**What happens when a date string is invalid**\n\n`monthly_report` does not read date strings. It reads `date` objects that are already in each entry. The date string is checked earlier, when you add an entry.\n\n`add_entry` calls `parse_date`. `parse_date` splits the text at each \"-\". It raises `ValueError` with the message `bad date: ...` if one of these is true:\n\n- The text does not have exactly three parts.\n- Any part has characters that are not digits.\n\nIf the text has three digit parts but the values are not a real date (for example, month 13 or day 32), the `date(y, m, d)` call raises its own `ValueError`. The message for this case is different.\n\nIn all these cases, the error occurs before `entries.append` runs. The entry is not added, and the list does not change. `add_entry` does not catch the error. The caller must handle it.\n\nNote these two points:\n\n- `add_entry` checks the amount first. If the amount is zero or less, it raises `ValueError` before it checks the date.\n- `parse_date` accepts parts of any length, such as `2024-1-5`. It does not force the format to be exactly YYYY-MM-DD.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "extras_off": true,
   "api_cost": 0.038383
  },
  {
   "run": "gpt-5.6-luna__control__T3__r1__f7c8e1",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "control",
   "rep": 1,
   "round": 16,
   "date": "2026-10-02",
   "exit": 0,
   "passed": true,
   "output": 245,
   "input": 20970,
   "cache_read": 18944,
   "cache_write": 11554,
   "warmup": 9528,
   "cache_write_1h": null,
   "cost_usd": 0.002984,
   "wall_s": 9.9,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 9,
   "commands": 1,
   "answer": "Monthly totals are computed by `monthly_report(entries, year, month)`. It first keeps entries whose date\u2019s year and month match the requested values, then returns:\n\n- `count`: number of matching entries\n- `total`: sum of their amounts\n- `by_category`: category-by-category sums\n\nAn invalid date string causes `parse_date` to raise `ValueError`. This happens either when the string is not three numeric, hyphen-separated parts (`YYYY-MM-DD`) or when the resulting date is invalid, such as month `13`. Consequently, `add_entry` fails and does not append the entry.",
   "account_remark": false,
   "effort": "low",
   "cli": "codex-cli 0.159.0",
   "extras_off": true,
   "api_cost": 0.002984
  },
  {
   "run": "gpt-5.6-luna__control__T3__r2__1cee51",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "control",
   "rep": 2,
   "round": 16,
   "date": "2026-10-02",
   "exit": 0,
   "passed": true,
   "output": 243,
   "input": 20960,
   "cache_read": 18944,
   "cache_write": 11544,
   "warmup": 9528,
   "cache_write_1h": null,
   "cost_usd": 0.002979,
   "wall_s": 8.7,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 14,
   "commands": 1,
   "answer": "`monthly_report(entries, year, month)` first filters entries whose `day.year` and `day.month` match the requested year and month. It then returns:\n\n- `count`: number of matching entries\n- `total`: sum of their amounts\n- `by_category`: category-by-category sums\n\nAn invalid date string causes `parse_date` to raise `ValueError`. This happens either when the format is not exactly three numeric hyphen-separated parts, or when the resulting year/month/day is not a valid calendar date. Consequently, `add_entry` fails and does not append the entry.",
   "account_remark": false,
   "effort": "low",
   "cli": "codex-cli 0.159.0",
   "extras_off": true,
   "api_cost": 0.002979
  },
  {
   "run": "gpt-5.6-luna__control__T3__r3__b2e8f5",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "control",
   "rep": 3,
   "round": 16,
   "date": "2026-10-02",
   "exit": 0,
   "passed": true,
   "output": 249,
   "input": 20972,
   "cache_read": 18944,
   "cache_write": 11556,
   "warmup": 9528,
   "cache_write_1h": null,
   "cost_usd": 0.002989,
   "wall_s": 11.2,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 11,
   "commands": 1,
   "answer": "Monthly totals are computed by:\n\n1. Selecting entries whose `day.year` and `day.month` match the requested year and month.\n2. Counting those entries.\n3. Summing their `amount` values.\n4. Grouping the amounts by category.\n\n`monthly_report` returns all three results: `count`, `total`, and `by_category`.\n\nDate strings are parsed as `YYYY-MM-DD`. If the format is malformed or the date is invalid (for example, month 13), `parse_date` raises `ValueError`. Consequently, `add_entry` does not append an entry when its date is invalid.",
   "account_remark": false,
   "effort": "low",
   "cli": "codex-cli 0.159.0",
   "extras_off": true,
   "api_cost": 0.002989
  },
  {
   "run": "gpt-5.6-luna__ste__T3__r1__dba02c",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "ste",
   "rep": 1,
   "round": 16,
   "date": "2026-10-02",
   "exit": 0,
   "passed": true,
   "output": 292,
   "input": 21100,
   "cache_read": 17920,
   "cache_write": 12708,
   "warmup": 9528,
   "cache_write_1h": null,
   "cost_usd": 0.00325,
   "wall_s": 11.1,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 69,
   "commands": 1,
   "answer": "Monthly totals are computed by:\n\n1. Selecting entries with the requested year and month.\n2. Counting these entries.\n3. Adding their `amount` values.\n4. Adding category subtotals.\n\nThe report returns `count`, `total`, and `by_category`.\n\nAn invalid date string causes `parse_date()` to raise `ValueError`. This occurs when the format is not `YYYY-MM-DD` or when the date values are not valid. In `add_entry()`, the entry is not added when this error occurs.",
   "account_remark": false,
   "effort": "low",
   "cli": "codex-cli 0.159.0",
   "extras_off": true,
   "api_cost": 0.00325
  },
  {
   "run": "gpt-5.6-luna__ste__T3__r2__fb907f",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "ste",
   "rep": 2,
   "round": 16,
   "date": "2026-10-02",
   "exit": 0,
   "passed": true,
   "output": 354,
   "input": 21098,
   "cache_read": 18944,
   "cache_write": 11682,
   "warmup": 9528,
   "cache_write_1h": null,
   "cost_usd": 0.00314,
   "wall_s": 12.1,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 92,
   "commands": 1,
   "answer": "Monthly totals are computed by `monthly_report()`:\n\n- It selects entries with the requested year and month.\n- It counts these entries.\n- It adds their `amount` values.\n- It groups the amounts by category.\n\nIt returns `count`, `total`, and `by_category`.\n\n`parse_date()` accepts only a date in `YYYY-MM-DD` form. It raises `ValueError` when:\n\n- The text does not have three hyphen-separated parts.\n- Any part contains a non-digit character.\n- The date values are not valid, such as month `13` or day `32`.\n\nWhen `add_entry()` receives an invalid date, it raises the error and does not add the entry.",
   "account_remark": false,
   "effort": "low",
   "cli": "codex-cli 0.159.0",
   "extras_off": true,
   "api_cost": 0.00314
  },
  {
   "run": "gpt-5.6-luna__ste__T3__r3__add0d1",
   "model": "gpt-5.6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "ste",
   "rep": 3,
   "round": 16,
   "date": "2026-10-02",
   "exit": 0,
   "passed": true,
   "output": 316,
   "input": 21106,
   "cache_read": 14848,
   "cache_write": 15786,
   "warmup": 9528,
   "cache_write_1h": null,
   "cost_usd": 0.003833,
   "wall_s": 10.8,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 66,
   "commands": 1,
   "answer": "Monthly totals are computed as follows:\n\n- The code selects entries with the requested year and month.\n- It counts these entries.\n- It adds their `amount` values.\n- It also groups the amounts by category.\n- `monthly_report` returns the count, total, and category totals.\n\nIf a date string is invalid, `parse_date` raises `ValueError`. This occurs when the string does not have three numeric parts separated by hyphens, or when the date is not valid in the calendar. For example, `2025-02-30` is invalid. When this occurs in `add_entry`, the entry is not added.",
   "account_remark": false,
   "effort": "low",
   "cli": "codex-cli 0.159.0",
   "extras_off": true,
   "api_cost": 0.003833
  },
  {
   "run": "gpt-6-luna__control__T3__r1__257e8a",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "control",
   "rep": 1,
   "round": 16,
   "date": "2026-10-02",
   "exit": 0,
   "passed": true,
   "output": 195,
   "input": 25050,
   "cache_read": 22016,
   "cache_write": 13858,
   "warmup": 10824,
   "cache_write_1h": null,
   "cost_usd": 0.001703,
   "wall_s": 7.5,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "`monthly_report` first filters entries to those whose `day.year` and `day.month` match the requested year and month. It then returns the number of matching entries, their summed amounts, and a per-category sum of those amounts.\n\nWhen a date string is added, `parse_date` requires three hyphen-separated, all-digit parts in `YYYY-MM-DD` form, then constructs a `date`. If the format is malformed or the date is impossible, it raises `ValueError`. `add_entry` doesn\u2019t catch that error, so the entry isn\u2019t appended.",
   "account_remark": false,
   "effort": "low",
   "cli": "codex-cli 0.159.0",
   "extras_off": true,
   "api_cost": 0.001703
  },
  {
   "run": "gpt-6-luna__control__T3__r2__34c0e7",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "control",
   "rep": 2,
   "round": 16,
   "date": "2026-10-02",
   "exit": 0,
   "passed": true,
   "output": 179,
   "input": 25048,
   "cache_read": 22016,
   "cache_write": 13856,
   "warmup": 10824,
   "cache_write_1h": null,
   "cost_usd": 0.001695,
   "wall_s": 7.1,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "`monthly_report` first selects entries whose date\u2019s year and month match the requested values. It returns the number of matching entries, their summed amounts, and the amounts summed by category.\n\nWhen a date string is parsed, it must have three all-numeric parts in `YYYY-MM-DD` form, and the resulting date must be valid. Otherwise, `parse_date` raises `ValueError`. `add_entry` calls `parse_date` before appending, so an invalid date raises an error and that entry is not added.",
   "account_remark": false,
   "effort": "low",
   "cli": "codex-cli 0.159.0",
   "extras_off": true,
   "api_cost": 0.001695
  },
  {
   "run": "gpt-6-luna__control__T3__r3__85b82e",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "control",
   "rep": 3,
   "round": 16,
   "date": "2026-10-02",
   "exit": 0,
   "passed": true,
   "output": 186,
   "input": 25046,
   "cache_read": 22016,
   "cache_write": 13854,
   "warmup": 10824,
   "cache_write_1h": null,
   "cost_usd": 0.001699,
   "wall_s": 8.4,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "`monthly_report` selects entries whose date has the requested year and month, then returns their count, summed amount, and totals grouped by category.\n\nWhen a date string is used to add an entry, `parse_date` expects three hyphen-separated numeric parts in `YYYY-MM-DD` format. A malformed string raises `ValueError` with a `bad date` message; an impossible calendar date, such as an invalid month or day, also raises `ValueError` from Python\u2019s date constructor. No entry is added because parsing happens before the append.",
   "account_remark": false,
   "effort": "low",
   "cli": "codex-cli 0.159.0",
   "extras_off": true,
   "api_cost": 0.001699
  },
  {
   "run": "gpt-6-luna__ste__T3__r1__f8d64f",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "ste",
   "rep": 1,
   "round": 16,
   "date": "2026-10-02",
   "exit": 0,
   "passed": false,
   "output": 52,
   "input": 12356,
   "cache_read": 12032,
   "cache_write": 11148,
   "warmup": 10824,
   "cache_write_1h": null,
   "cost_usd": 0.001261,
   "wall_s": 3.8,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 0,
   "answer": "Monthly totals are computed from the records for each month. The code adds each record\u2019s amount to that month\u2019s total.\n\nIf a date string is invalid, the date parse fails. The code does not add that record to a monthly total.",
   "account_remark": false,
   "effort": "low",
   "cli": "codex-cli 0.159.0",
   "extras_off": true,
   "api_cost": 0.001261
  },
  {
   "run": "gpt-6-luna__ste__T3__r2__7b6bde",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "ste",
   "rep": 2,
   "round": 16,
   "date": "2026-10-02",
   "exit": 0,
   "passed": false,
   "output": 57,
   "input": 12356,
   "cache_read": 9984,
   "cache_write": 13196,
   "warmup": 10824,
   "cache_write_1h": null,
   "cost_usd": 0.001448,
   "wall_s": 5.5,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 0,
   "answer": "Monthly totals are computed by adding the transaction amounts for each month. The code uses the year and month from each transaction date to select the month.\n\nIf a date string is invalid, the code raises an error. It does not add that transaction to a monthly total.",
   "account_remark": false,
   "effort": "low",
   "cli": "codex-cli 0.159.0",
   "extras_off": true,
   "api_cost": 0.001448
  },
  {
   "run": "gpt-6-luna__ste__T3__r3__3824f8",
   "model": "gpt-6-luna",
   "host": "codex",
   "task": "T3",
   "skill": "ste",
   "rep": 3,
   "round": 16,
   "date": "2026-10-02",
   "exit": 0,
   "passed": false,
   "output": 48,
   "input": 12353,
   "cache_read": 9984,
   "cache_write": 13193,
   "warmup": 10824,
   "cache_write_1h": null,
   "cost_usd": 0.001443,
   "wall_s": 3.6,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 0,
   "answer": "I cannot explain `ledger/core.py` because I do not have access to the workspace files in this conversation. Please provide the file contents, and I will explain how it computes monthly totals and handles an invalid date string.",
   "account_remark": false,
   "effort": "low",
   "cli": "codex-cli 0.159.0",
   "extras_off": true,
   "api_cost": 0.001443
  },
  {
   "run": "gpt-6.1-sol__control__T3__r1__a00ce5",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T3",
   "skill": "control",
   "rep": 1,
   "round": 16,
   "date": "2026-10-02",
   "exit": 0,
   "passed": true,
   "output": 309,
   "input": 39803,
   "cache_read": 35584,
   "cache_write": 15735,
   "warmup": 11516,
   "cache_write_1h": null,
   "cost_usd": 0.038118,
   "wall_s": 19.5,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 2,
   "answer": "In [ledger/core.py](/tmp/bench-work/gpt-6.1-sol__control__T3__r1__a00ce5/ledger/core.py), `monthly_report` selects entries whose date matches the requested year and month. It returns the number of matching entries, the sum of their amounts, and totals grouped by category. An empty month produces a count and total of zero and an empty category dictionary.\n\n`parse_date` splits a date string on hyphens and requires three digit-only parts. If that check fails, it raises `ValueError` with a `bad date` message. Otherwise, it converts the parts to integers and calls `datetime.date`, which raises `ValueError` for impossible dates or out-of-range values. Despite the docstring\u2019s `YYYY-MM-DD` wording, the parser does not enforce fixed digit widths.\n\nWhen `add_entry` encounters an invalid date, the exception propagates and no entry is appended. No files were changed.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "extras_off": true,
   "api_cost": 0.038118
  },
  {
   "run": "gpt-6.1-sol__control__T3__r2__0bf079",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T3",
   "skill": "control",
   "rep": 2,
   "round": 16,
   "date": "2026-10-02",
   "exit": 0,
   "passed": true,
   "output": 236,
   "input": 26517,
   "cache_read": 22784,
   "cache_write": 15249,
   "warmup": 11516,
   "cache_write_1h": null,
   "cost_usd": 0.035136,
   "wall_s": 14.0,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "In `ledger/core.py`, `monthly_report` selects entries whose dates match the requested year and month. It returns the number of matching entries, the sum of their amounts, and totals grouped by category. If none match, the count and total are zero and the category totals are an empty dictionary.\n\n`parse_date` splits a date string on hyphens and requires three digit-only parts. Malformed strings raise `ValueError` with a message such as `bad date: 'abc'`. The parts are converted to integers and passed to Python\u2019s `date` constructor, which also raises `ValueError` for impossible dates, such as February 30. Fixed-width formatting is not enforced, so `2026-2-3` is accepted.\n\nWhen `add_entry` encounters an invalid date, the exception propagates and no entry is appended. No files were changed.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "extras_off": true,
   "api_cost": 0.035136
  },
  {
   "run": "gpt-6.1-sol__control__T3__r3__cacf06",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T3",
   "skill": "control",
   "rep": 3,
   "round": 16,
   "date": "2026-10-02",
   "exit": 0,
   "passed": true,
   "output": 301,
   "input": 39807,
   "cache_read": 35584,
   "cache_write": 15739,
   "warmup": 11516,
   "cache_write_1h": null,
   "cost_usd": 0.038046,
   "wall_s": 18.2,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 2,
   "answer": "In [ledger/core.py](/tmp/bench-work/gpt-6.1-sol__control__T3__r3__cacf06/ledger/core.py), `monthly_report` selects entries whose dates match the requested year and month. It returns their count, the sum of their amounts, and totals grouped by category. With no matching entries, these are `0`, `0`, and `{}`.\n\n`parse_date` requires three hyphen-separated numeric parts, then passes them as year, month, and day to Python\u2019s `date` constructor. Malformed strings raise `ValueError` with a `bad date: ...` message; impossible dates also raise `ValueError` from the constructor. Despite the docstring\u2019s `YYYY-MM-DD` wording, fixed digit widths aren\u2019t enforced.\n\nWhen `add_entry` encounters an invalid date, the exception propagates and no entry is appended. No files were changed.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "extras_off": true,
   "api_cost": 0.038046
  },
  {
   "run": "gpt-6.1-sol__ste__T3__r1__f6cb57",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T3",
   "skill": "ste",
   "rep": 1,
   "round": 16,
   "date": "2026-10-02",
   "exit": 0,
   "passed": true,
   "output": 264,
   "input": 26649,
   "cache_read": 22784,
   "cache_write": 15381,
   "warmup": 11516,
   "cache_write_1h": null,
   "cost_usd": 0.03568,
   "wall_s": 16.7,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "In [ledger/core.py](/tmp/bench-work/gpt-6.1-sol__ste__T3__r1__f6cb57/ledger/core.py), `monthly_report` selects entries with the specified year and month. It adds their amounts to calculate the monthly total in dollars. It also returns the number of selected entries and a total for each category. If no entries match, the count and total are zero, and the category dictionary is empty.\n\n`parse_date` splits the date string at each hyphen. It requires three parts that contain only digits. If this check fails, it raises `ValueError` with a `bad date` message. Otherwise, it converts the parts to integers and calls `date(year, month, day)`. An invalid calendar date also causes `ValueError`. The code does not check the length of each part.\n\nIf date conversion fails in `add_entry`, the error passes to the caller. The function does not add an entry.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "extras_off": true,
   "api_cost": 0.03568
  },
  {
   "run": "gpt-6.1-sol__ste__T3__r2__fc3ecc",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T3",
   "skill": "ste",
   "rep": 2,
   "round": 16,
   "date": "2026-10-02",
   "exit": 0,
   "passed": true,
   "output": 247,
   "input": 26646,
   "cache_read": 22784,
   "cache_write": 15378,
   "warmup": 11516,
   "cache_write_1h": null,
   "cost_usd": 0.035504,
   "wall_s": 15.1,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "In [ledger/core.py](/tmp/bench-work/gpt-6.1-sol__ste__T3__r2__fc3ecc/ledger/core.py), `monthly_report` selects entries with the specified year and month. It counts these entries, adds their amounts, and adds the amounts for each category. If there are no matching entries, the count and total are zero. The category results are empty.\n\n`parse_date` splits the date string at each hyphen. It requires three parts that contain only digits. If this check fails, it raises `ValueError` with a `bad date` message. It then converts the parts to integers and calls `date(year, month, day)`. If these values do not form a valid date, this call raises `ValueError`. When `add_entry` gets either error, it does not add an entry.\n\nNo files were changed.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "extras_off": true,
   "api_cost": 0.035504
  },
  {
   "run": "gpt-6.1-sol__ste__T3__r3__9f5ed0",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "T3",
   "skill": "ste",
   "rep": 3,
   "round": 16,
   "date": "2026-10-02",
   "exit": 0,
   "passed": true,
   "output": 222,
   "input": 26654,
   "cache_read": 24320,
   "cache_write": 13850,
   "warmup": 11516,
   "cache_write_1h": null,
   "cost_usd": 0.032352,
   "wall_s": 14.2,
   "turns": null,
   "prompt_last": null,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": 1,
   "answer": "In `ledger/core.py`, `monthly_report` first selects entries with the specified year and month. It then returns the number of selected entries, the sum of their amounts, and the sum for each category. If no entries match, the count and total are zero, and the category dictionary is empty.\n\n`parse_date` splits the date string at each hyphen. It requires three parts that contain only digits. If this check fails, it raises `ValueError` with a `bad date` message. It then converts the parts to integers and creates a date. An impossible date, such as February 30, also causes `ValueError`. The code does not require fixed digit counts.\n\nIf date parsing fails in `add_entry`, the error passes to the caller. No entry is added.",
   "account_remark": false,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "extras_off": true,
   "api_cost": 0.032352
  }
 ],
 "timing_runs": [
  {
   "host": "claude",
   "rep": 1,
   "cli": "2.1.284 (Claude Code)",
   "effort": "medium",
   "passed": true,
   "date": "2026-09-28",
   "total": 10.799,
   "ready": 0.444,
   "tools": 0.418,
   "first_token": 1.077,
   "answer_start": 7.486,
   "answer_span": 0.766,
   "answer_pieces": 176,
   "out": 887,
   "out_per_s": 89.262,
   "requests": 2,
   "waiting": 9.937,
   "model": "claude-sonnet-5-5",
   "extras_off": false,
   "replaced_by": 13
  },
  {
   "host": "claude",
   "rep": 1,
   "cli": "2.1.284 (Claude Code)",
   "effort": "high",
   "passed": true,
   "date": "2026-09-28",
   "total": 13.826,
   "ready": 2.537,
   "tools": 0.584,
   "first_token": 1.017,
   "answer_start": 0.846,
   "answer_span": 7.385,
   "answer_pieces": 166,
   "out": 990,
   "out_per_s": 92.48,
   "requests": 3,
   "waiting": 10.705,
   "model": "claude-sonnet-5",
   "extras_off": false,
   "replaced_by": 13
  },
  {
   "host": "claude",
   "rep": 1,
   "cli": "2.1.284 (Claude Code)",
   "effort": "medium",
   "passed": true,
   "date": "2026-09-28",
   "total": 15.635,
   "ready": 1.165,
   "tools": 0.694,
   "first_token": 0.949,
   "answer_start": 9.359,
   "answer_span": 2.435,
   "answer_pieces": 196,
   "out": 1096,
   "out_per_s": 79.559,
   "requests": 2,
   "waiting": 13.776,
   "model": "claude-opus-5-5",
   "extras_off": false,
   "replaced_by": 13
  },
  {
   "host": "claude",
   "rep": 1,
   "cli": "2.1.284 (Claude Code)",
   "effort": "high",
   "passed": true,
   "date": "2026-09-28",
   "total": 17.516,
   "ready": 1.111,
   "tools": 0.477,
   "first_token": 1.059,
   "answer_start": 8.793,
   "answer_span": 4.704,
   "answer_pieces": 219,
   "out": 1203,
   "out_per_s": 75.527,
   "requests": 2,
   "waiting": 15.928,
   "model": "claude-fable-5-1",
   "extras_off": false,
   "replaced_by": 13
  },
  {
   "host": "codex",
   "rep": 1,
   "cli": "codex-cli 0.157.1",
   "effort": "medium",
   "passed": true,
   "date": "2026-09-28",
   "total": 12.497,
   "ready": 2.803,
   "tools": 3.021,
   "first_token": 0.929,
   "answer_start": 1.842,
   "answer_span": 1.973,
   "answer_pieces": 118,
   "out": 319,
   "out_per_s": 47.8,
   "requests": 2,
   "waiting": 6.674,
   "model": "gpt-5.6-luna",
   "server_tbt_ms": 6.196,
   "extras_off": false,
   "replaced_by": 13
  },
  {
   "host": "codex",
   "rep": 1,
   "cli": "codex-cli 0.157.1",
   "effort": "medium",
   "passed": true,
   "date": "2026-09-28",
   "total": 10.008,
   "ready": 2.329,
   "tools": 1.743,
   "first_token": 0.87,
   "answer_start": 0.513,
   "answer_span": 2.843,
   "answer_pieces": 113,
   "out": 186,
   "out_per_s": 31.336,
   "requests": 2,
   "waiting": 5.936,
   "model": "gpt-6-luna",
   "server_tbt_ms": 25.685,
   "extras_off": false,
   "replaced_by": 13
  },
  {
   "host": "codex",
   "rep": 1,
   "cli": "codex-cli 0.157.1",
   "effort": "medium",
   "passed": true,
   "date": "2026-09-28",
   "total": 11.972,
   "ready": 2.903,
   "tools": 1.202,
   "first_token": 0.868,
   "answer_start": 3.269,
   "answer_span": 1.753,
   "answer_pieces": 153,
   "out": 365,
   "out_per_s": 46.39,
   "requests": 2,
   "waiting": 7.868,
   "model": "gpt-5.6-sol",
   "server_tbt_ms": 13.295,
   "extras_off": false,
   "replaced_by": 13
  },
  {
   "host": "codex",
   "rep": 1,
   "cli": "codex-cli 0.157.1",
   "effort": "medium",
   "passed": true,
   "date": "2026-09-28",
   "total": 14.893,
   "ready": 3.016,
   "tools": 1.142,
   "first_token": 3.549,
   "answer_start": 3.363,
   "answer_span": 2.095,
   "answer_pieces": 152,
   "out": 356,
   "out_per_s": 33.163,
   "requests": 2,
   "waiting": 10.735,
   "model": "gpt-6-sol",
   "server_tbt_ms": 18.347,
   "extras_off": false,
   "replaced_by": 13
  },
  {
   "host": "codex",
   "rep": 1,
   "cli": "codex-cli 0.157.1",
   "effort": "medium",
   "passed": true,
   "date": "2026-09-28",
   "total": 17.281,
   "ready": 6.718,
   "tools": 1.423,
   "first_token": 1.352,
   "answer_start": 1.089,
   "answer_span": 5.016,
   "answer_pieces": 179,
   "out": 238,
   "out_per_s": 26.04,
   "requests": 2,
   "waiting": 9.14,
   "model": "gpt-6-astra",
   "server_tbt_ms": 15.093,
   "extras_off": false,
   "replaced_by": 13
  },
  {
   "host": "claude",
   "rep": 2,
   "cli": "2.1.284 (Claude Code)",
   "effort": "medium",
   "passed": true,
   "date": "2026-09-28",
   "total": 10.546,
   "ready": 1.176,
   "tools": 0.571,
   "first_token": 3.033,
   "answer_start": 5.098,
   "answer_span": 0.001,
   "answer_pieces": 145,
   "out": 795,
   "out_per_s": 90.351,
   "requests": 2,
   "waiting": 8.799,
   "model": "claude-sonnet-5-5",
   "extras_off": false,
   "replaced_by": 13
  },
  {
   "host": "claude",
   "rep": 2,
   "cli": "2.1.284 (Claude Code)",
   "effort": "high",
   "passed": true,
   "date": "2026-09-28",
   "total": 10.892,
   "ready": 1.193,
   "tools": 0.733,
   "first_token": 1.073,
   "answer_start": 0.593,
   "answer_span": 5.36,
   "answer_pieces": 127,
   "out": 791,
   "out_per_s": 88.222,
   "requests": 3,
   "waiting": 8.966,
   "model": "claude-sonnet-5",
   "extras_off": false,
   "replaced_by": 13
  },
  {
   "host": "claude",
   "rep": 2,
   "cli": "2.1.284 (Claude Code)",
   "effort": "medium",
   "passed": true,
   "date": "2026-09-28",
   "total": 18.115,
   "ready": 1.007,
   "tools": 0.427,
   "first_token": 0.767,
   "answer_start": 11.792,
   "answer_span": 3.225,
   "answer_pieces": 197,
   "out": 1516,
   "out_per_s": 90.882,
   "requests": 2,
   "waiting": 16.681,
   "model": "claude-opus-5-5",
   "extras_off": false,
   "replaced_by": 13
  },
  {
   "host": "claude",
   "rep": 2,
   "cli": "2.1.284 (Claude Code)",
   "effort": "high",
   "passed": true,
   "date": "2026-09-28",
   "total": 19.445,
   "ready": 1.064,
   "tools": 0.397,
   "first_token": 3.514,
   "answer_start": 9.68,
   "answer_span": 3.235,
   "answer_pieces": 197,
   "out": 1105,
   "out_per_s": 61.444,
   "requests": 2,
   "waiting": 17.984,
   "model": "claude-fable-5-1",
   "extras_off": false,
   "replaced_by": 13
  },
  {
   "host": "codex",
   "rep": 2,
   "cli": "codex-cli 0.157.1",
   "effort": "medium",
   "passed": true,
   "date": "2026-09-28",
   "total": 13.614,
   "ready": 4.722,
   "tools": 2.196,
   "first_token": 1.362,
   "answer_start": 1.502,
   "answer_span": 1.748,
   "answer_pieces": 105,
   "out": 296,
   "out_per_s": 44.202,
   "requests": 2,
   "waiting": 6.697,
   "model": "gpt-5.6-luna",
   "server_tbt_ms": 11.163,
   "extras_off": false,
   "replaced_by": 13
  },
  {
   "host": "codex",
   "rep": 2,
   "cli": "codex-cli 0.157.1",
   "effort": "medium",
   "passed": true,
   "date": "2026-09-28",
   "total": 7.03,
   "ready": 1.836,
   "tools": 1.078,
   "first_token": 0.713,
   "answer_start": 0.83,
   "answer_span": 1.267,
   "answer_pieces": 83,
   "out": 159,
   "out_per_s": 38.628,
   "requests": 2,
   "waiting": 4.116,
   "model": "gpt-6-luna",
   "server_tbt_ms": 17.115,
   "extras_off": false,
   "replaced_by": 13
  },
  {
   "host": "codex",
   "rep": 2,
   "cli": "codex-cli 0.157.1",
   "effort": "medium",
   "passed": true,
   "date": "2026-09-28",
   "total": 11.391,
   "ready": 2.287,
   "tools": 1.054,
   "first_token": 0.683,
   "answer_start": 3.006,
   "answer_span": 2.563,
   "answer_pieces": 150,
   "out": 401,
   "out_per_s": 49.812,
   "requests": 2,
   "waiting": 8.05,
   "model": "gpt-5.6-sol",
   "server_tbt_ms": 12.238,
   "extras_off": false,
   "replaced_by": 13
  },
  {
   "host": "codex",
   "rep": 2,
   "cli": "codex-cli 0.157.1",
   "effort": "medium",
   "passed": true,
   "date": "2026-09-28",
   "total": 14.571,
   "ready": 3.765,
   "tools": 1.3,
   "first_token": 1.267,
   "answer_start": 3.835,
   "answer_span": 1.872,
   "answer_pieces": 124,
   "out": 300,
   "out_per_s": 31.561,
   "requests": 2,
   "waiting": 9.505,
   "model": "gpt-6-sol",
   "server_tbt_ms": 18.413,
   "extras_off": false,
   "replaced_by": 13
  },
  {
   "host": "codex",
   "rep": 2,
   "cli": "codex-cli 0.157.1",
   "effort": "medium",
   "passed": true,
   "date": "2026-09-28",
   "total": 12.932,
   "ready": 2.779,
   "tools": 1.781,
   "first_token": 1.212,
   "answer_start": 1.032,
   "answer_span": 4.417,
   "answer_pieces": 159,
   "out": 218,
   "out_per_s": 26.041,
   "requests": 2,
   "waiting": 8.371,
   "model": "gpt-6-astra",
   "server_tbt_ms": 15.838,
   "extras_off": false,
   "replaced_by": 13
  },
  {
   "model": "gpt-6.1-sol",
   "host": "codex",
   "rep": 1,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-09-29",
   "total": 12.751,
   "ready": 3.125,
   "waiting": 8.286,
   "tools": 1.34,
   "first_token": 0.979,
   "answer_start": 0.717,
   "answer_span": 4.614,
   "answer_pieces": 157,
   "out": 220,
   "out_per_s": 26.552,
   "requests": 2,
   "server_tbt_ms": 7.592,
   "round": 10,
   "extras_off": false,
   "replaced_by": 13
  },
  {
   "model": "gpt-6-sol",
   "host": "codex",
   "rep": 1,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-09-29",
   "total": 14.598,
   "ready": 2.578,
   "waiting": 10.591,
   "tools": 1.429,
   "first_token": 3.304,
   "answer_start": 3.343,
   "answer_span": 2.019,
   "answer_pieces": 98,
   "out": 300,
   "out_per_s": 28.326,
   "requests": 2,
   "server_tbt_ms": 20.235,
   "round": 10,
   "extras_off": false,
   "replaced_by": 13
  },
  {
   "model": "gpt-6.1-sol",
   "host": "codex",
   "rep": 2,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-09-29",
   "total": 10.713,
   "ready": 2.03,
   "waiting": 7.556,
   "tools": 1.127,
   "first_token": 0.799,
   "answer_start": 0.486,
   "answer_span": 4.451,
   "answer_pieces": 152,
   "out": 211,
   "out_per_s": 27.925,
   "requests": 2,
   "server_tbt_ms": 7.928,
   "round": 10,
   "extras_off": false,
   "replaced_by": 13
  },
  {
   "model": "gpt-6-sol",
   "host": "codex",
   "rep": 2,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-09-29",
   "total": 12.633,
   "ready": 3.039,
   "waiting": 8.453,
   "tools": 1.141,
   "first_token": 1.26,
   "answer_start": 3.455,
   "answer_span": 2.512,
   "answer_pieces": 133,
   "out": 286,
   "out_per_s": 33.834,
   "requests": 2,
   "server_tbt_ms": 21.901,
   "round": 10,
   "extras_off": false,
   "replaced_by": 13
  },
  {
   "model": "claude-opus-5-5",
   "host": "claude",
   "rep": 1,
   "cli": "2.1.284 (Claude Code)",
   "effort": "medium",
   "passed": true,
   "date": "2026-09-29",
   "total": 18.927,
   "ready": 0.445,
   "waiting": 18.241,
   "tools": 0.241,
   "first_token": 0.899,
   "answer_start": 12.869,
   "answer_span": 3.344,
   "answer_pieces": 204,
   "out": 1659,
   "out_per_s": 90.949,
   "requests": 2,
   "round": 10,
   "extras_off": false,
   "replaced_by": 13
  },
  {
   "model": "claude-opus-5-5",
   "host": "claude",
   "rep": 2,
   "cli": "2.1.284 (Claude Code)",
   "effort": "medium",
   "passed": true,
   "date": "2026-09-29",
   "total": 14.931,
   "ready": 0.34,
   "waiting": 14.182,
   "tools": 0.409,
   "first_token": 0.777,
   "answer_start": 11.133,
   "answer_span": 1.097,
   "answer_pieces": 154,
   "out": 1286,
   "out_per_s": 90.678,
   "requests": 2,
   "round": 10,
   "extras_off": false,
   "replaced_by": 13
  },
  {
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "rep": 1,
   "cli": "2.1.284 (Claude Code)",
   "effort": "medium",
   "passed": true,
   "date": "2026-09-30",
   "total": 8.476,
   "ready": 0.445,
   "waiting": 7.646,
   "tools": 0.385,
   "first_token": 1.215,
   "answer_start": 5.528,
   "answer_span": 0.089,
   "answer_pieces": 156,
   "out": 827,
   "out_per_s": 108.161,
   "requests": 2,
   "round": 13,
   "extras_off": true
  },
  {
   "model": "claude-sonnet-5",
   "host": "claude",
   "rep": 1,
   "cli": "2.1.284 (Claude Code)",
   "effort": "high",
   "passed": true,
   "date": "2026-09-30",
   "total": 10.436,
   "ready": 0.302,
   "waiting": 9.625,
   "tools": 0.51,
   "first_token": 1.034,
   "answer_start": 0.986,
   "answer_span": 5.87,
   "answer_pieces": 13,
   "out": 892,
   "out_per_s": 92.675,
   "requests": 3,
   "round": 13,
   "extras_off": true
  },
  {
   "model": "claude-opus-5-5",
   "host": "claude",
   "rep": 1,
   "cli": "2.1.284 (Claude Code)",
   "effort": "medium",
   "passed": true,
   "date": "2026-09-30",
   "total": 15.88,
   "ready": 0.535,
   "waiting": 15.066,
   "tools": 0.28,
   "first_token": 0.828,
   "answer_start": 11.226,
   "answer_span": 1.965,
   "answer_pieces": 177,
   "out": 1281,
   "out_per_s": 85.026,
   "requests": 2,
   "round": 13,
   "extras_off": true
  },
  {
   "model": "claude-fable-5-1",
   "host": "claude",
   "rep": 1,
   "cli": "2.1.284 (Claude Code)",
   "effort": "high",
   "passed": true,
   "date": "2026-09-30",
   "total": 13.268,
   "ready": 0.467,
   "waiting": 12.356,
   "tools": 0.444,
   "first_token": 1.114,
   "answer_start": 8.478,
   "answer_span": 1.192,
   "answer_pieces": 150,
   "out": 857,
   "out_per_s": 69.359,
   "requests": 2,
   "round": 13,
   "extras_off": true
  },
  {
   "model": "gpt-5.6-luna",
   "host": "codex",
   "rep": 1,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-09-30",
   "total": 11.364,
   "ready": 2.03,
   "waiting": 7.749,
   "tools": 1.585,
   "first_token": 0.982,
   "answer_start": 1.691,
   "answer_span": 1.377,
   "answer_pieces": 84,
   "out": 279,
   "out_per_s": 36.007,
   "requests": 2,
   "server_tbt_ms": 6.072,
   "round": 13,
   "extras_off": true
  },
  {
   "model": "gpt-6-luna",
   "host": "codex",
   "rep": 1,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-09-30",
   "total": 13.626,
   "ready": 3.663,
   "waiting": 8.394,
   "tools": 1.569,
   "first_token": 1.066,
   "answer_start": 0.704,
   "answer_span": 4.164,
   "answer_pieces": 107,
   "out": 180,
   "out_per_s": 21.444,
   "requests": 2,
   "server_tbt_ms": 39.618,
   "round": 13,
   "extras_off": true
  },
  {
   "model": "gpt-5.6-sol",
   "host": "codex",
   "rep": 1,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-09-30",
   "total": 23.59,
   "ready": 1.776,
   "waiting": 20.72,
   "tools": 1.094,
   "first_token": 1.345,
   "answer_start": 6.334,
   "answer_span": 8.029,
   "answer_pieces": 178,
   "out": 377,
   "out_per_s": 18.195,
   "requests": 2,
   "server_tbt_ms": 10.28,
   "round": 13,
   "extras_off": true
  },
  {
   "model": "gpt-6-sol",
   "host": "codex",
   "rep": 1,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-09-30",
   "total": 23.414,
   "ready": 2.579,
   "waiting": 19.654,
   "tools": 1.181,
   "first_token": 2.271,
   "answer_start": 5.765,
   "answer_span": 6.835,
   "answer_pieces": 116,
   "out": 329,
   "out_per_s": 16.739,
   "requests": 2,
   "server_tbt_ms": 22.37,
   "round": 13,
   "extras_off": true
  },
  {
   "model": "gpt-6-astra",
   "host": "codex",
   "rep": 1,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-09-30",
   "total": 13.82,
   "ready": 1.965,
   "waiting": 10.051,
   "tools": 1.803,
   "first_token": 1.452,
   "answer_start": 1.804,
   "answer_span": 4.734,
   "answer_pieces": 165,
   "out": 224,
   "out_per_s": 22.285,
   "requests": 2,
   "server_tbt_ms": 16.41,
   "round": 13,
   "extras_off": true
  },
  {
   "model": "gpt-6.1-sol",
   "host": "codex",
   "rep": 1,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-09-30",
   "total": 24.893,
   "ready": 2.345,
   "waiting": 20.555,
   "tools": 1.993,
   "first_token": 3.314,
   "answer_start": 4.881,
   "answer_span": 8.486,
   "answer_pieces": 189,
   "out": 250,
   "out_per_s": 12.162,
   "requests": 2,
   "server_tbt_ms": 40.605,
   "round": 13,
   "extras_off": true
  },
  {
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "rep": 2,
   "cli": "2.1.284 (Claude Code)",
   "effort": "medium",
   "passed": true,
   "date": "2026-09-30",
   "total": 7.81,
   "ready": 0.494,
   "waiting": 6.922,
   "tools": 0.394,
   "first_token": 0.65,
   "answer_start": 5.481,
   "answer_span": 0.022,
   "answer_pieces": 43,
   "out": 730,
   "out_per_s": 105.461,
   "requests": 2,
   "round": 13,
   "extras_off": true
  },
  {
   "model": "claude-sonnet-5",
   "host": "claude",
   "rep": 2,
   "cli": "2.1.284 (Claude Code)",
   "effort": "high",
   "passed": true,
   "date": "2026-09-30",
   "total": 13.1,
   "ready": 0.455,
   "waiting": 12.318,
   "tools": 0.327,
   "first_token": 2.063,
   "answer_start": 1.049,
   "answer_span": 5.699,
   "answer_pieces": 96,
   "out": 672,
   "out_per_s": 54.554,
   "requests": 3,
   "round": 13,
   "extras_off": true
  },
  {
   "model": "claude-opus-5-5",
   "host": "claude",
   "rep": 2,
   "cli": "2.1.284 (Claude Code)",
   "effort": "medium",
   "passed": true,
   "date": "2026-09-30",
   "total": 15.502,
   "ready": 0.471,
   "waiting": 14.742,
   "tools": 0.288,
   "first_token": 0.831,
   "answer_start": 9.8,
   "answer_span": 3.017,
   "answer_pieces": 200,
   "out": 1480,
   "out_per_s": 100.393,
   "requests": 2,
   "round": 13,
   "extras_off": true
  },
  {
   "model": "claude-fable-5-1",
   "host": "claude",
   "rep": 2,
   "cli": "2.1.284 (Claude Code)",
   "effort": "high",
   "passed": true,
   "date": "2026-09-30",
   "total": 21.313,
   "ready": 0.499,
   "waiting": 20.628,
   "tools": 0.186,
   "first_token": 2.287,
   "answer_start": 12.442,
   "answer_span": 4.881,
   "answer_pieces": 221,
   "out": 1260,
   "out_per_s": 61.082,
   "requests": 2,
   "round": 13,
   "extras_off": true
  },
  {
   "model": "gpt-5.6-luna",
   "host": "codex",
   "rep": 2,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-09-30",
   "total": 9.384,
   "ready": 2.014,
   "waiting": 6.231,
   "tools": 1.14,
   "first_token": 1.097,
   "answer_start": 1.381,
   "answer_span": 1.87,
   "answer_pieces": 112,
   "out": 292,
   "out_per_s": 46.863,
   "requests": 2,
   "server_tbt_ms": 8.172,
   "round": 13,
   "extras_off": true
  },
  {
   "model": "gpt-6-luna",
   "host": "codex",
   "rep": 2,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-09-30",
   "total": 15.403,
   "ready": 1.817,
   "waiting": 9.697,
   "tools": 3.889,
   "first_token": 0.712,
   "answer_start": 4.216,
   "answer_span": 1.926,
   "answer_pieces": 104,
   "out": 184,
   "out_per_s": 18.975,
   "requests": 2,
   "server_tbt_ms": 29.852,
   "round": 13,
   "extras_off": true
  },
  {
   "model": "gpt-5.6-sol",
   "host": "codex",
   "rep": 2,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-09-30",
   "total": 24.322,
   "ready": 3.382,
   "waiting": 19.282,
   "tools": 1.658,
   "first_token": 0.868,
   "answer_start": 4.938,
   "answer_span": 8.189,
   "answer_pieces": 173,
   "out": 370,
   "out_per_s": 19.189,
   "requests": 2,
   "server_tbt_ms": 15.068,
   "round": 13,
   "extras_off": true
  },
  {
   "model": "gpt-6-sol",
   "host": "codex",
   "rep": 2,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-09-30",
   "total": 22.556,
   "ready": 5.419,
   "waiting": 15.74,
   "tools": 1.397,
   "first_token": 2.0,
   "answer_start": 4.718,
   "answer_span": 4.373,
   "answer_pieces": 97,
   "out": 294,
   "out_per_s": 18.678,
   "requests": 2,
   "server_tbt_ms": 21.168,
   "round": 13,
   "extras_off": true
  },
  {
   "model": "gpt-6-astra",
   "host": "codex",
   "rep": 2,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-09-30",
   "total": 16.768,
   "ready": 1.793,
   "waiting": 13.849,
   "tools": 1.126,
   "first_token": 2.371,
   "answer_start": 2.283,
   "answer_span": 7.477,
   "answer_pieces": 208,
   "out": 267,
   "out_per_s": 19.28,
   "requests": 2,
   "server_tbt_ms": 21.438,
   "round": 13,
   "extras_off": true
  },
  {
   "model": "gpt-6.1-sol",
   "host": "codex",
   "rep": 2,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-09-30",
   "total": 22.249,
   "ready": 1.923,
   "waiting": 19.014,
   "tools": 1.312,
   "first_token": 1.779,
   "answer_start": 1.865,
   "answer_span": 8.131,
   "answer_pieces": 179,
   "out": 290,
   "out_per_s": 15.252,
   "requests": 3,
   "server_tbt_ms": 30.279,
   "round": 13,
   "extras_off": true
  }
 ],
 "speed_recheck_runs": [
  {
   "model": "gpt-6.1-sol",
   "host": "codex",
   "rep": 1,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-10-01",
   "total": 19.591,
   "ready": 3.203,
   "waiting": 14.657,
   "tools": 1.731,
   "first_token": 1.753,
   "answer_start": 1.887,
   "answer_span": 6.596,
   "answer_pieces": 196,
   "out": 302,
   "out_per_s": 20.605,
   "requests": 3,
   "server_tbt_ms": 32.595,
   "round": 14,
   "extras_off": true,
   "session": "morning"
  },
  {
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "rep": 1,
   "cli": "2.1.284 (Claude Code)",
   "effort": "medium",
   "passed": true,
   "date": "2026-10-01",
   "total": 7.777,
   "ready": 0.634,
   "waiting": 6.755,
   "tools": 0.388,
   "first_token": 0.875,
   "answer_start": 5.184,
   "answer_span": 0.019,
   "answer_pieces": 140,
   "out": 787,
   "out_per_s": 116.506,
   "requests": 2,
   "round": 14,
   "extras_off": true,
   "session": "morning"
  },
  {
   "model": "gpt-6.1-sol",
   "host": "codex",
   "rep": 2,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-10-01",
   "total": 16.642,
   "ready": 1.785,
   "waiting": 13.763,
   "tools": 1.095,
   "first_token": 2.777,
   "answer_start": 2.153,
   "answer_span": 6.858,
   "answer_pieces": 143,
   "out": 204,
   "out_per_s": 14.823,
   "requests": 2,
   "server_tbt_ms": 38.361,
   "round": 14,
   "extras_off": true,
   "session": "morning"
  },
  {
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "rep": 2,
   "cli": "2.1.284 (Claude Code)",
   "effort": "medium",
   "passed": true,
   "date": "2026-10-01",
   "total": 7.43,
   "ready": 0.337,
   "waiting": 6.69,
   "tools": 0.403,
   "first_token": 0.736,
   "answer_start": 5.206,
   "answer_span": 0.001,
   "answer_pieces": 135,
   "out": 737,
   "out_per_s": 110.164,
   "requests": 2,
   "round": 14,
   "extras_off": true,
   "session": "morning"
  },
  {
   "model": "gpt-6.1-sol",
   "host": "codex",
   "rep": 1,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-10-01",
   "total": 16.613,
   "ready": 4.455,
   "waiting": 10.913,
   "tools": 1.245,
   "first_token": 1.497,
   "answer_start": 1.26,
   "answer_span": 6.478,
   "answer_pieces": 194,
   "out": 253,
   "out_per_s": 23.184,
   "requests": 2,
   "server_tbt_ms": 32.845,
   "round": 14,
   "extras_off": false,
   "session": "morning"
  },
  {
   "model": "gpt-6.1-sol",
   "host": "codex",
   "rep": 2,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-10-01",
   "total": 19.708,
   "ready": 2.138,
   "waiting": 16.132,
   "tools": 1.438,
   "first_token": 1.558,
   "answer_start": 2.965,
   "answer_span": 6.878,
   "answer_pieces": 182,
   "out": 282,
   "out_per_s": 17.481,
   "requests": 3,
   "server_tbt_ms": 34.376,
   "round": 14,
   "extras_off": false,
   "session": "morning"
  },
  {
   "model": "gpt-6.1-sol",
   "host": "codex",
   "rep": 1,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-10-01",
   "total": 14.998,
   "ready": 2.754,
   "waiting": 10.899,
   "tools": 1.345,
   "first_token": 1.917,
   "answer_start": 1.989,
   "answer_span": 5.251,
   "answer_pieces": 157,
   "out": 216,
   "out_per_s": 19.818,
   "requests": 2,
   "server_tbt_ms": 33.621,
   "round": 15,
   "extras_off": true,
   "session": "afternoon"
  },
  {
   "model": "gpt-6.1-sol",
   "host": "codex",
   "rep": 2,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-10-01",
   "total": 13.519,
   "ready": 1.65,
   "waiting": 10.603,
   "tools": 1.266,
   "first_token": 1.779,
   "answer_start": 1.847,
   "answer_span": 5.298,
   "answer_pieces": 149,
   "out": 208,
   "out_per_s": 19.618,
   "requests": 2,
   "server_tbt_ms": 33.199,
   "round": 15,
   "extras_off": true,
   "session": "afternoon"
  },
  {
   "model": "gpt-6.1-sol",
   "host": "codex",
   "rep": 1,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-10-01",
   "total": 16.82,
   "ready": 2.956,
   "waiting": 12.369,
   "tools": 1.494,
   "first_token": 1.452,
   "answer_start": 1.263,
   "answer_span": 5.089,
   "answer_pieces": 173,
   "out": 274,
   "out_per_s": 22.152,
   "requests": 3,
   "server_tbt_ms": 27.705,
   "round": 15,
   "extras_off": false,
   "session": "afternoon"
  },
  {
   "model": "gpt-6.1-sol",
   "host": "codex",
   "rep": 2,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-10-01",
   "total": 15.771,
   "ready": 2.269,
   "waiting": 12.274,
   "tools": 1.228,
   "first_token": 1.591,
   "answer_start": 2.384,
   "answer_span": 6.403,
   "answer_pieces": 198,
   "out": 257,
   "out_per_s": 20.939,
   "requests": 2,
   "server_tbt_ms": 31.353,
   "round": 15,
   "extras_off": false,
   "session": "afternoon"
  },
  {
   "model": "gpt-6.1-sol",
   "host": "codex",
   "rep": 1,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-10-03",
   "total": 21.751,
   "ready": 3.173,
   "waiting": 16.819,
   "tools": 1.759,
   "first_token": 2.385,
   "answer_start": 2.733,
   "answer_span": 5.949,
   "answer_pieces": 192,
   "out": 312,
   "out_per_s": 18.55,
   "requests": 3,
   "server_tbt_ms": 32.191,
   "round": 17,
   "extras_off": true,
   "session": "afternoon"
  },
  {
   "model": "gpt-6.1-sol",
   "host": "codex",
   "rep": 2,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-10-03",
   "total": 20.577,
   "ready": 3.675,
   "waiting": 14.677,
   "tools": 2.225,
   "first_token": 1.586,
   "answer_start": 1.776,
   "answer_span": 6.189,
   "answer_pieces": 181,
   "out": 290,
   "out_per_s": 19.759,
   "requests": 3,
   "server_tbt_ms": 33.195,
   "round": 17,
   "extras_off": true,
   "session": "afternoon"
  },
  {
   "model": "gpt-6.1-sol",
   "host": "codex",
   "rep": 3,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-10-03",
   "total": 15.619,
   "ready": 2.074,
   "waiting": 12.459,
   "tools": 1.086,
   "first_token": 2.477,
   "answer_start": 2.583,
   "answer_span": 5.51,
   "answer_pieces": 149,
   "out": 212,
   "out_per_s": 17.016,
   "requests": 2,
   "server_tbt_ms": 36.214,
   "round": 17,
   "extras_off": true,
   "session": "afternoon"
  },
  {
   "model": "gpt-6.1-sol",
   "host": "codex",
   "rep": 4,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-10-03",
   "total": 26.55,
   "ready": 3.166,
   "waiting": 19.915,
   "tools": 3.469,
   "first_token": 4.005,
   "answer_start": 3.247,
   "answer_span": 7.099,
   "answer_pieces": 196,
   "out": 306,
   "out_per_s": 15.365,
   "requests": 3,
   "server_tbt_ms": 34.119,
   "round": 17,
   "extras_off": true,
   "session": "afternoon"
  },
  {
   "model": "gpt-6.1-sol",
   "host": "codex",
   "rep": 1,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-10-04",
   "total": 22.864,
   "ready": 4.974,
   "waiting": 13.096,
   "tools": 4.794,
   "first_token": 1.71,
   "answer_start": 1.615,
   "answer_span": 5.55,
   "answer_pieces": 191,
   "out": 297,
   "out_per_s": 22.678,
   "requests": 3,
   "server_tbt_ms": 24.466,
   "round": 18,
   "extras_off": true,
   "session": "afternoon"
  },
  {
   "model": "gpt-6.1-sol",
   "host": "codex",
   "rep": 2,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-10-04",
   "total": 20.087,
   "ready": 2.459,
   "waiting": 12.871,
   "tools": 4.757,
   "first_token": 1.875,
   "answer_start": 3.68,
   "answer_span": 5.555,
   "answer_pieces": 197,
   "out": 258,
   "out_per_s": 20.045,
   "requests": 2,
   "server_tbt_ms": 21.762,
   "round": 18,
   "extras_off": true,
   "session": "afternoon"
  },
  {
   "model": "gpt-6.1-sol",
   "host": "codex",
   "rep": 3,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-10-04",
   "total": 16.132,
   "ready": 3.686,
   "waiting": 11.48,
   "tools": 0.966,
   "first_token": 2.067,
   "answer_start": 1.876,
   "answer_span": 5.359,
   "answer_pieces": 188,
   "out": 247,
   "out_per_s": 21.516,
   "requests": 2,
   "server_tbt_ms": 24.147,
   "round": 18,
   "extras_off": true,
   "session": "afternoon"
  },
  {
   "model": "gpt-6.1-sol",
   "host": "codex",
   "rep": 4,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-10-04",
   "total": 13.198,
   "ready": 1.803,
   "waiting": 8.515,
   "tools": 2.88,
   "first_token": 1.303,
   "answer_start": 1.164,
   "answer_span": 4.229,
   "answer_pieces": 151,
   "out": 213,
   "out_per_s": 25.013,
   "requests": 2,
   "server_tbt_ms": 23.256,
   "round": 18,
   "extras_off": true,
   "session": "afternoon"
  },
  {
   "model": "gpt-6.1-sol",
   "host": "codex",
   "rep": 1,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-10-05",
   "total": 16.997,
   "ready": 3.084,
   "waiting": 12.8,
   "tools": 1.114,
   "first_token": 1.522,
   "answer_start": 1.984,
   "answer_span": 7.13,
   "answer_pieces": 206,
   "out": 267,
   "out_per_s": 20.86,
   "requests": 2,
   "server_tbt_ms": 22.46,
   "round": 19,
   "extras_off": true,
   "session": "before the post"
  },
  {
   "model": "gpt-6.1-sol",
   "host": "codex",
   "rep": 2,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-10-05",
   "total": 18.613,
   "ready": 2.518,
   "waiting": 14.519,
   "tools": 1.576,
   "first_token": 4.021,
   "answer_start": 1.448,
   "answer_span": 7.381,
   "answer_pieces": 163,
   "out": 222,
   "out_per_s": 15.29,
   "requests": 2,
   "server_tbt_ms": 21.564,
   "round": 19,
   "extras_off": true,
   "session": "before the post"
  },
  {
   "model": "gpt-6.1-sol",
   "host": "codex",
   "rep": 3,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-10-05",
   "total": 14.064,
   "ready": 2.349,
   "waiting": 10.877,
   "tools": 0.838,
   "first_token": 1.781,
   "answer_start": 1.531,
   "answer_span": 5.289,
   "answer_pieces": 150,
   "out": 213,
   "out_per_s": 19.583,
   "requests": 2,
   "server_tbt_ms": 21.156,
   "round": 19,
   "extras_off": true,
   "session": "before the post"
  },
  {
   "model": "gpt-6.1-sol",
   "host": "codex",
   "rep": 4,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-10-05",
   "total": 15.393,
   "ready": 2.421,
   "waiting": 11.764,
   "tools": 1.208,
   "first_token": 1.861,
   "answer_start": 1.279,
   "answer_span": 6.363,
   "answer_pieces": 179,
   "out": 241,
   "out_per_s": 20.486,
   "requests": 2,
   "server_tbt_ms": 21.983,
   "round": 19,
   "extras_off": true,
   "session": "before the post"
  },
  {
   "model": "gpt-6.1-sol",
   "host": "codex",
   "rep": 1,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-10-05",
   "total": 15.093,
   "ready": 2.61,
   "waiting": 11.373,
   "tools": 1.11,
   "first_token": 1.817,
   "answer_start": 0.961,
   "answer_span": 6.335,
   "answer_pieces": 180,
   "out": 243,
   "out_per_s": 21.367,
   "requests": 2,
   "server_tbt_ms": 23.809,
   "round": 20,
   "extras_off": true,
   "session": "right after the post"
  },
  {
   "model": "gpt-6.1-sol",
   "host": "codex",
   "rep": 2,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-10-05",
   "total": 14.825,
   "ready": 2.989,
   "waiting": 10.574,
   "tools": 1.262,
   "first_token": 1.658,
   "answer_start": 1.219,
   "answer_span": 5.446,
   "answer_pieces": 152,
   "out": 211,
   "out_per_s": 19.955,
   "requests": 2,
   "server_tbt_ms": 15.335,
   "round": 20,
   "extras_off": true,
   "session": "right after the post"
  },
  {
   "model": "gpt-6.1-sol",
   "host": "codex",
   "rep": 3,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-10-05",
   "total": 19.48,
   "ready": 2.334,
   "waiting": 15.358,
   "tools": 1.788,
   "first_token": 1.863,
   "answer_start": 1.414,
   "answer_span": 7.365,
   "answer_pieces": 201,
   "out": 302,
   "out_per_s": 19.664,
   "requests": 3,
   "server_tbt_ms": 14.656,
   "round": 20,
   "extras_off": true,
   "session": "right after the post"
  },
  {
   "model": "gpt-6.1-sol",
   "host": "codex",
   "rep": 4,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-10-05",
   "total": 14.339,
   "ready": 1.819,
   "waiting": 11.562,
   "tools": 0.958,
   "first_token": 2.068,
   "answer_start": 1.367,
   "answer_span": 5.92,
   "answer_pieces": 165,
   "out": 224,
   "out_per_s": 19.373,
   "requests": 2,
   "server_tbt_ms": 20.982,
   "round": 20,
   "extras_off": true,
   "session": "right after the post"
  },
  {
   "model": "gpt-6.1-sol",
   "host": "codex",
   "rep": 1,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-10-05",
   "total": 12.354,
   "ready": 2.543,
   "waiting": 8.466,
   "tools": 1.346,
   "first_token": 1.519,
   "answer_start": 1.06,
   "answer_span": 4.697,
   "answer_pieces": 203,
   "answer_out": 207,
   "answer_thinking": 0,
   "answer_max_gap": 0.173,
   "out": 263,
   "out_per_s": 31.067,
   "requests": 2,
   "server_tbt_ms": 22.877,
   "round": 21,
   "extras_off": true,
   "session": "retest"
  },
  {
   "model": "gpt-6.1-sol",
   "host": "codex",
   "rep": 2,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-10-05",
   "total": 11.808,
   "ready": 2.167,
   "waiting": 8.55,
   "tools": 1.091,
   "first_token": 1.028,
   "answer_start": 1.906,
   "answer_span": 4.537,
   "answer_pieces": 171,
   "answer_out": 175,
   "answer_thinking": 0,
   "answer_max_gap": 2.774,
   "out": 232,
   "out_per_s": 27.134,
   "requests": 2,
   "server_tbt_ms": 20.917,
   "round": 21,
   "extras_off": true,
   "session": "retest"
  },
  {
   "model": "gpt-6.1-sol",
   "host": "codex",
   "rep": 3,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-10-05",
   "total": 20.838,
   "ready": 2.575,
   "waiting": 15.604,
   "tools": 2.659,
   "first_token": 2.091,
   "answer_start": 1.473,
   "answer_span": 5.077,
   "answer_pieces": 194,
   "answer_out": 198,
   "answer_thinking": 0,
   "answer_max_gap": 0.47,
   "out": 305,
   "out_per_s": 19.546,
   "requests": 3,
   "server_tbt_ms": 22.493,
   "round": 21,
   "extras_off": true,
   "session": "retest"
  },
  {
   "model": "gpt-6.1-sol",
   "host": "codex",
   "rep": 4,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-10-05",
   "total": 11.335,
   "ready": 2.055,
   "waiting": 7.88,
   "tools": 1.4,
   "first_token": 1.401,
   "answer_start": 1.387,
   "answer_span": 3.922,
   "answer_pieces": 190,
   "answer_out": 194,
   "answer_thinking": 0,
   "answer_max_gap": 0.157,
   "out": 253,
   "out_per_s": 32.108,
   "requests": 2,
   "server_tbt_ms": 20.554,
   "round": 21,
   "extras_off": true,
   "session": "retest"
  },
  {
   "model": "gpt-6.1-sol",
   "host": "codex",
   "rep": 1,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-10-05",
   "total": 15.296,
   "ready": 2.956,
   "waiting": 10.623,
   "tools": 1.718,
   "first_token": 1.648,
   "answer_start": 1.642,
   "answer_span": 3.991,
   "answer_pieces": 203,
   "answer_out": 207,
   "answer_thinking": 0,
   "answer_max_gap": 0.217,
   "out": 306,
   "out_per_s": 28.806,
   "requests": 3,
   "server_tbt_ms": 20.261,
   "round": 22,
   "extras_off": true,
   "session": "after the two hours"
  },
  {
   "model": "gpt-6.1-sol",
   "host": "codex",
   "rep": 2,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-10-05",
   "total": 17.682,
   "ready": 2.145,
   "waiting": 13.038,
   "tools": 2.499,
   "first_token": 1.25,
   "answer_start": 1.927,
   "answer_span": 4.998,
   "answer_pieces": 196,
   "answer_out": 200,
   "answer_thinking": 0,
   "answer_max_gap": 0.402,
   "out": 299,
   "out_per_s": 22.933,
   "requests": 3,
   "server_tbt_ms": 22.98,
   "round": 22,
   "extras_off": true,
   "session": "after the two hours"
  },
  {
   "model": "gpt-6.1-sol",
   "host": "codex",
   "rep": 3,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-10-05",
   "total": 17.171,
   "ready": 3.791,
   "waiting": 11.14,
   "tools": 2.24,
   "first_token": 1.663,
   "answer_start": 1.572,
   "answer_span": 3.273,
   "answer_pieces": 173,
   "answer_out": 177,
   "answer_thinking": 0,
   "answer_max_gap": 0.088,
   "out": 284,
   "out_per_s": 25.494,
   "requests": 3,
   "server_tbt_ms": 19.178,
   "round": 22,
   "extras_off": true,
   "session": "after the two hours"
  },
  {
   "model": "gpt-6.1-sol",
   "host": "codex",
   "rep": 4,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-10-05",
   "total": 11.404,
   "ready": 2.107,
   "waiting": 7.898,
   "tools": 1.398,
   "first_token": 1.272,
   "answer_start": 1.017,
   "answer_span": 4.351,
   "answer_pieces": 195,
   "answer_out": 199,
   "answer_thinking": 0,
   "answer_max_gap": 0.931,
   "out": 254,
   "out_per_s": 32.159,
   "requests": 2,
   "server_tbt_ms": 21.048,
   "round": 22,
   "extras_off": true,
   "session": "after the two hours"
  },
  {
   "model": "gpt-6.1-sol",
   "host": "codex",
   "rep": 1,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-10-06",
   "total": 15.718,
   "ready": 2.338,
   "waiting": 11.787,
   "tools": 1.593,
   "first_token": 2.448,
   "answer_start": 2.462,
   "answer_span": 3.918,
   "answer_pieces": 166,
   "answer_out": 170,
   "answer_thinking": 0,
   "answer_max_gap": 0.615,
   "out": 281,
   "out_per_s": 23.84,
   "requests": 3,
   "server_tbt_ms": 19.73,
   "round": 23,
   "extras_off": true
  },
  {
   "model": "gpt-6.1-sol",
   "host": "codex",
   "rep": 2,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-10-06",
   "total": 10.305,
   "ready": 1.851,
   "waiting": 7.474,
   "tools": 0.979,
   "first_token": 1.382,
   "answer_start": 1.358,
   "answer_span": 3.803,
   "answer_pieces": 184,
   "answer_out": 188,
   "answer_thinking": 0,
   "answer_max_gap": 0.165,
   "out": 243,
   "out_per_s": 32.511,
   "requests": 2,
   "server_tbt_ms": 21.253,
   "round": 23,
   "extras_off": true
  },
  {
   "model": "gpt-6.1-sol",
   "host": "codex",
   "rep": 3,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-10-06",
   "total": 15.062,
   "ready": 2.187,
   "waiting": 9.855,
   "tools": 3.019,
   "first_token": 1.146,
   "answer_start": 1.501,
   "answer_span": 3.801,
   "answer_pieces": 172,
   "answer_out": 176,
   "answer_thinking": 0,
   "answer_max_gap": 0.188,
   "out": 271,
   "out_per_s": 27.498,
   "requests": 3,
   "server_tbt_ms": 22.59,
   "round": 23,
   "extras_off": true
  },
  {
   "model": "gpt-6.1-sol",
   "host": "codex",
   "rep": 4,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-10-06",
   "total": 11.59,
   "ready": 2.169,
   "waiting": 8.456,
   "tools": 0.965,
   "first_token": 1.765,
   "answer_start": 1.748,
   "answer_span": 3.709,
   "answer_pieces": 186,
   "answer_out": 190,
   "answer_thinking": 0,
   "answer_max_gap": 0.231,
   "out": 245,
   "out_per_s": 28.973,
   "requests": 2,
   "server_tbt_ms": 20.84,
   "round": 23,
   "extras_off": true
  }
 ],
 "sol_test_runs": [
  {
   "model": "gpt-6.1-sol",
   "host": "codex",
   "rep": 1,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-10-05",
   "total": 18.368,
   "ready": 4.503,
   "waiting": 12.25,
   "tools": 1.615,
   "first_token": 1.519,
   "answer_start": 1.4,
   "answer_span": 5.307,
   "answer_pieces": 188,
   "answer_out": 192,
   "out": 293,
   "out_per_s": 23.919,
   "requests": 3,
   "server_tbt_ms": 26.09,
   "session": "between the sitting right after the post and the retest"
  },
  {
   "model": "gpt-6.1-sol",
   "host": "codex",
   "rep": 2,
   "cli": "codex-cli 0.159.0",
   "effort": "medium",
   "passed": true,
   "date": "2026-10-05",
   "total": 11.739,
   "ready": 2.391,
   "waiting": 7.707,
   "tools": 1.642,
   "first_token": 1.697,
   "answer_start": 0.995,
   "answer_span": 3.798,
   "answer_pieces": 161,
   "answer_out": 165,
   "out": 219,
   "out_per_s": 28.417,
   "requests": 2,
   "server_tbt_ms": 23.341,
   "session": "between the sitting right after the post and the retest"
  }
 ],
 "sol_compare_runs": [
  {
   "model": "claude-opus-5-5",
   "host": "claude",
   "rep": 1,
   "cli": "2.1.284 (Claude Code)",
   "effort": "medium",
   "passed": true,
   "date": "2026-10-06",
   "total": 17.678,
   "ready": 0.663,
   "waiting": 16.797,
   "tools": 0.218,
   "first_token": 0.736,
   "answer_start": 11.015,
   "answer_span": 4.166,
   "answer_pieces": 220,
   "answer_out": 1570,
   "answer_thinking": 387,
   "answer_max_gap": null,
   "out": 1658,
   "out_per_s": 98.708,
   "requests": 2,
   "round": 24
  },
  {
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "rep": 1,
   "cli": "2.1.284 (Claude Code)",
   "effort": "medium",
   "passed": true,
   "date": "2026-10-06",
   "total": 8.081,
   "ready": 0.309,
   "waiting": 7.448,
   "tools": 0.324,
   "first_token": 0.938,
   "answer_start": 5.868,
   "answer_span": 0.003,
   "answer_pieces": 147,
   "answer_out": 697,
   "answer_thinking": 0,
   "answer_max_gap": null,
   "out": 787,
   "out_per_s": 105.666,
   "requests": 2,
   "round": 24
  },
  {
   "model": "claude-opus-5-5",
   "host": "claude",
   "rep": 2,
   "cli": "2.1.284 (Claude Code)",
   "effort": "medium",
   "passed": true,
   "date": "2026-10-06",
   "total": 17.425,
   "ready": 0.459,
   "waiting": 16.404,
   "tools": 0.562,
   "first_token": 2.613,
   "answer_start": 10.544,
   "answer_span": 2.532,
   "answer_pieces": 187,
   "answer_out": 1372,
   "answer_thinking": 319,
   "answer_max_gap": null,
   "out": 1460,
   "out_per_s": 89.003,
   "requests": 2,
   "round": 24
  },
  {
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "rep": 2,
   "cli": "2.1.284 (Claude Code)",
   "effort": "medium",
   "passed": true,
   "date": "2026-10-06",
   "total": 7.787,
   "ready": 0.457,
   "waiting": 6.996,
   "tools": 0.334,
   "first_token": 0.996,
   "answer_start": 5.321,
   "answer_span": 0.0,
   "answer_pieces": 144,
   "answer_out": 661,
   "answer_thinking": 0,
   "answer_max_gap": null,
   "out": 751,
   "out_per_s": 107.347,
   "requests": 2,
   "round": 24
  },
  {
   "model": "claude-opus-5-5",
   "host": "claude",
   "rep": 3,
   "cli": "2.1.284 (Claude Code)",
   "effort": "medium",
   "passed": true,
   "date": "2026-10-06",
   "total": 15.183,
   "ready": 0.44,
   "waiting": 14.288,
   "tools": 0.455,
   "first_token": 0.802,
   "answer_start": 11.221,
   "answer_span": 1.338,
   "answer_pieces": 170,
   "answer_out": 1168,
   "answer_thinking": 243,
   "answer_max_gap": null,
   "out": 1256,
   "out_per_s": 87.906,
   "requests": 2,
   "round": 24
  },
  {
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "rep": 3,
   "cli": "2.1.284 (Claude Code)",
   "effort": "medium",
   "passed": true,
   "date": "2026-10-06",
   "total": 9.812,
   "ready": 0.458,
   "waiting": 9.034,
   "tools": 0.321,
   "first_token": 1.198,
   "answer_start": 6.33,
   "answer_span": 0.86,
   "answer_pieces": 179,
   "answer_out": 851,
   "answer_thinking": 0,
   "answer_max_gap": null,
   "out": 940,
   "out_per_s": 104.051,
   "requests": 2,
   "round": 24
  },
  {
   "model": "claude-opus-5-5",
   "host": "claude",
   "rep": 4,
   "cli": "2.1.284 (Claude Code)",
   "effort": "medium",
   "passed": true,
   "date": "2026-10-06",
   "total": 17.528,
   "ready": 0.473,
   "waiting": 16.737,
   "tools": 0.319,
   "first_token": 0.761,
   "answer_start": 12.409,
   "answer_span": 2.665,
   "answer_pieces": 200,
   "answer_out": 1384,
   "answer_thinking": 340,
   "answer_max_gap": null,
   "out": 1471,
   "out_per_s": 87.889,
   "requests": 2,
   "round": 24
  },
  {
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "rep": 4,
   "cli": "2.1.284 (Claude Code)",
   "effort": "medium",
   "passed": true,
   "date": "2026-10-06",
   "total": 8.174,
   "ready": 0.468,
   "waiting": 7.358,
   "tools": 0.348,
   "first_token": 1.052,
   "answer_start": 5.641,
   "answer_span": 0.001,
   "answer_pieces": 157,
   "answer_out": 679,
   "answer_thinking": 0,
   "answer_max_gap": null,
   "out": 769,
   "out_per_s": 104.512,
   "requests": 2,
   "round": 24
  }
 ],
 "bughunt_runs": [
  {
   "run": "claude-opus-5-5__control__B1__r1__fa5630",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "B1",
   "skill": "control",
   "rep": 1,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 15890,
   "input": 12,
   "cache_read": 184673,
   "cache_write": 53163,
   "warmup": null,
   "cache_write_1h": 53163,
   "cost_usd": 0.7800866000000001,
   "wall_s": 153.1,
   "turns": 8,
   "prompt_last": 57664,
   "lines_added": 288,
   "lines_deleted": 33,
   "lines": 321,
   "files": 14,
   "thinking": 5879,
   "commands": null,
   "extras_off": true,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "regressions": 0,
   "visible_ok": true,
   "tests_edited": 0,
   "tests_added": 20,
   "lines_changed": 86,
   "score": 20,
   "of": 20,
   "api_cost": 0.780087
  },
  {
   "run": "claude-opus-5-5__control__B1__r2__a72733",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "B1",
   "skill": "control",
   "rep": 2,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 16464,
   "input": 12,
   "cache_read": 183083,
   "cache_write": 53973,
   "warmup": null,
   "cache_write_1h": 53973,
   "cost_usd": 0.7977285999999999,
   "wall_s": 156.5,
   "turns": 8,
   "prompt_last": 58477,
   "lines_added": 340,
   "lines_deleted": 38,
   "lines": 378,
   "files": 15,
   "thinking": 5149,
   "commands": null,
   "extras_off": true,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "regressions": 0,
   "visible_ok": true,
   "tests_edited": 0,
   "tests_added": 27,
   "lines_changed": 84,
   "score": 20,
   "of": 20,
   "api_cost": 0.797729
  },
  {
   "run": "claude-opus-5-5__control__B1__r3__14cb38",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "B1",
   "skill": "control",
   "rep": 3,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 16059,
   "input": 12,
   "cache_read": 181961,
   "cache_write": 53110,
   "warmup": null,
   "cache_write_1h": 53110,
   "cost_usd": 0.7825002,
   "wall_s": 157.7,
   "turns": 8,
   "prompt_last": 57703,
   "lines_added": 316,
   "lines_deleted": 28,
   "lines": 344,
   "files": 14,
   "thinking": 5314,
   "commands": null,
   "extras_off": true,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "regressions": 0,
   "visible_ok": true,
   "tests_edited": 0,
   "tests_added": 25,
   "lines_changed": 71,
   "score": 20,
   "of": 20,
   "api_cost": 0.7825
  },
  {
   "run": "claude-sonnet-5-5__control__B1__r1__2c9ea9",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "B1",
   "skill": "control",
   "rep": 1,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 2625,
   "input": 10,
   "cache_read": 106201,
   "cache_write": 25936,
   "warmup": null,
   "cache_write_1h": 25936,
   "cost_usd": 0.1512542,
   "wall_s": 23.0,
   "turns": 6,
   "prompt_last": 36172,
   "lines_added": 20,
   "lines_deleted": 20,
   "lines": 40,
   "files": 12,
   "thinking": 645,
   "commands": null,
   "extras_off": true,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "regressions": 0,
   "visible_ok": true,
   "tests_edited": 0,
   "tests_added": 0,
   "lines_changed": 40,
   "score": 19,
   "of": 20,
   "api_cost": 0.151254
  },
  {
   "run": "claude-sonnet-5-5__control__B1__r2__01857a",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "B1",
   "skill": "control",
   "rep": 2,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 2832,
   "input": 10,
   "cache_read": 113080,
   "cache_write": 25612,
   "warmup": null,
   "cache_write_1h": 25612,
   "cost_usd": 0.15340399999999998,
   "wall_s": 26.8,
   "turns": 6,
   "prompt_last": 35848,
   "lines_added": 20,
   "lines_deleted": 20,
   "lines": 40,
   "files": 12,
   "thinking": 653,
   "commands": null,
   "extras_off": true,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "regressions": 0,
   "visible_ok": true,
   "tests_edited": 0,
   "tests_added": 0,
   "lines_changed": 40,
   "score": 19,
   "of": 20,
   "api_cost": 0.153404
  },
  {
   "run": "claude-sonnet-5-5__control__B1__r3__60dcb8",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "B1",
   "skill": "control",
   "rep": 3,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 2709,
   "input": 8,
   "cache_read": 77754,
   "cache_write": 25465,
   "warmup": null,
   "cache_write_1h": 25465,
   "cost_usd": 0.1445168,
   "wall_s": 24.8,
   "turns": 5,
   "prompt_last": 35701,
   "lines_added": 20,
   "lines_deleted": 20,
   "lines": 40,
   "files": 12,
   "thinking": 686,
   "commands": null,
   "extras_off": true,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "regressions": 0,
   "visible_ok": true,
   "tests_edited": 0,
   "tests_added": 0,
   "lines_changed": 40,
   "score": 19,
   "of": 20,
   "api_cost": 0.144517
  },
  {
   "run": "gpt-5.6-sol__control__B1__r1__b943bf",
   "model": "gpt-5.6-sol",
   "host": "codex",
   "task": "B1",
   "skill": "control",
   "rep": 1,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 13044,
   "input": 571201,
   "cache_read": 533632,
   "cache_write": 48022,
   "warmup": 10453,
   "cache_write_1h": null,
   "cost_usd": 0.666421,
   "wall_s": 691.4,
   "turns": null,
   "prompt_last": null,
   "lines_added": 203,
   "lines_deleted": 26,
   "lines": 229,
   "files": 16,
   "thinking": 5067,
   "commands": 10,
   "extras_off": true,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "regressions": 0,
   "visible_ok": true,
   "tests_edited": 0,
   "tests_added": 16,
   "lines_changed": 58,
   "score": 20,
   "of": 20,
   "api_cost": 0.666421
  },
  {
   "run": "gpt-5.6-sol__control__B1__r2__4c7f6f",
   "model": "gpt-5.6-sol",
   "host": "codex",
   "task": "B1",
   "skill": "control",
   "rep": 2,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 10313,
   "input": 406421,
   "cache_read": 361088,
   "cache_write": 55786,
   "warmup": 10453,
   "cache_write_1h": null,
   "cost_usd": 0.573839,
   "wall_s": 468.2,
   "turns": null,
   "prompt_last": null,
   "lines_added": 190,
   "lines_deleted": 24,
   "lines": 214,
   "files": 15,
   "thinking": 3420,
   "commands": 9,
   "extras_off": true,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "regressions": 0,
   "visible_ok": true,
   "tests_edited": 0,
   "tests_added": 14,
   "lines_changed": 51,
   "score": 20,
   "of": 20,
   "api_cost": 0.573839
  },
  {
   "run": "gpt-5.6-sol__control__B1__r3__d81f87",
   "model": "gpt-5.6-sol",
   "host": "codex",
   "task": "B1",
   "skill": "control",
   "rep": 3,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 10170,
   "input": 289863,
   "cache_read": 257152,
   "cache_write": 43164,
   "warmup": 10453,
   "cache_write_1h": null,
   "cost_usd": 0.478917,
   "wall_s": 517.4,
   "turns": null,
   "prompt_last": null,
   "lines_added": 208,
   "lines_deleted": 24,
   "lines": 232,
   "files": 15,
   "thinking": 3791,
   "commands": 7,
   "extras_off": true,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "regressions": 0,
   "visible_ok": true,
   "tests_edited": 0,
   "tests_added": 14,
   "lines_changed": 54,
   "score": 20,
   "of": 20,
   "api_cost": 0.478917
  },
  {
   "run": "gpt-6-sol__control__B1__r1__a11374",
   "model": "gpt-6-sol",
   "host": "codex",
   "task": "B1",
   "skill": "control",
   "rep": 1,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 9263,
   "input": 502678,
   "cache_read": 469760,
   "cache_write": 43921,
   "warmup": 11003,
   "cache_write_1h": null,
   "cost_usd": 0.274424,
   "wall_s": 512.7,
   "turns": null,
   "prompt_last": null,
   "lines_added": 189,
   "lines_deleted": 28,
   "lines": 217,
   "files": 15,
   "thinking": 2366,
   "commands": 21,
   "extras_off": true,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "regressions": 0,
   "visible_ok": true,
   "tests_edited": 0,
   "tests_added": 12,
   "lines_changed": 64,
   "score": 20,
   "of": 20,
   "api_cost": 0.274424
  },
  {
   "run": "gpt-6-sol__control__B1__r2__972de6",
   "model": "gpt-6-sol",
   "host": "codex",
   "task": "B1",
   "skill": "control",
   "rep": 2,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 11835,
   "input": 629077,
   "cache_read": 590464,
   "cache_write": 49616,
   "warmup": 11003,
   "cache_write_1h": null,
   "cost_usd": 0.335675,
   "wall_s": 614.9,
   "turns": null,
   "prompt_last": null,
   "lines_added": 249,
   "lines_deleted": 37,
   "lines": 286,
   "files": 15,
   "thinking": 3498,
   "commands": 24,
   "extras_off": true,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "regressions": 0,
   "visible_ok": true,
   "tests_edited": 0,
   "tests_added": 14,
   "lines_changed": 100,
   "score": 20,
   "of": 20,
   "api_cost": 0.335675
  },
  {
   "run": "gpt-6-sol__control__B1__r3__b460bc",
   "model": "gpt-6-sol",
   "host": "codex",
   "task": "B1",
   "skill": "control",
   "rep": 3,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 7500,
   "input": 553799,
   "cache_read": 513408,
   "cache_write": 51394,
   "warmup": 11003,
   "cache_write_1h": null,
   "cost_usd": 0.28047,
   "wall_s": 403.6,
   "turns": null,
   "prompt_last": null,
   "lines_added": 158,
   "lines_deleted": 27,
   "lines": 185,
   "files": 15,
   "thinking": 1208,
   "commands": 21,
   "extras_off": true,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "regressions": 0,
   "visible_ok": true,
   "tests_edited": 0,
   "tests_added": 10,
   "lines_changed": 56,
   "score": 20,
   "of": 20,
   "api_cost": 0.28047
  },
  {
   "run": "gpt-6.1-sol__control__B1__r1__670458",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "B1",
   "skill": "control",
   "rep": 1,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 9185,
   "input": 372270,
   "cache_read": 335104,
   "cache_write": 48682,
   "warmup": 11516,
   "cache_write_1h": null,
   "cost_usd": 0.222724,
   "wall_s": 491.1,
   "turns": null,
   "prompt_last": null,
   "lines_added": 311,
   "lines_deleted": 36,
   "lines": 347,
   "files": 15,
   "thinking": 1186,
   "commands": 13,
   "extras_off": true,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "regressions": 0,
   "visible_ok": true,
   "tests_edited": 0,
   "tests_added": 20,
   "lines_changed": 94,
   "score": 20,
   "of": 20,
   "api_cost": 0.222724
  },
  {
   "run": "gpt-6.1-sol__control__B1__r2__80f5e3",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "B1",
   "skill": "control",
   "rep": 2,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 11714,
   "input": 395743,
   "cache_read": 341504,
   "cache_write": 65755,
   "warmup": 11516,
   "cache_write_1h": null,
   "cost_usd": 0.2828,
   "wall_s": 602.8,
   "turns": null,
   "prompt_last": null,
   "lines_added": 340,
   "lines_deleted": 34,
   "lines": 374,
   "files": 15,
   "thinking": 1959,
   "commands": 13,
   "extras_off": true,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "regressions": 0,
   "visible_ok": true,
   "tests_edited": 0,
   "tests_added": 22,
   "lines_changed": 82,
   "score": 20,
   "of": 20,
   "api_cost": 0.2828
  },
  {
   "run": "gpt-6.1-sol__control__B1__r3__c478eb",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "B1",
   "skill": "control",
   "rep": 3,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 8672,
   "input": 291347,
   "cache_read": 255488,
   "cache_write": 47375,
   "warmup": 11516,
   "cache_write_1h": null,
   "cost_usd": 0.207019,
   "wall_s": 451.0,
   "turns": null,
   "prompt_last": null,
   "lines_added": 273,
   "lines_deleted": 32,
   "lines": 305,
   "files": 15,
   "thinking": 940,
   "commands": 11,
   "extras_off": true,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "regressions": 0,
   "visible_ok": true,
   "tests_edited": 0,
   "tests_added": 21,
   "lines_changed": 82,
   "score": 20,
   "of": 20,
   "api_cost": 0.207019
  },
  {
   "run": "claude-opus-5-5__control__B2__r1__bc8273",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "B2",
   "skill": "control",
   "rep": 1,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 20862,
   "input": 16,
   "cache_read": 282512,
   "cache_write": 60486,
   "warmup": null,
   "cache_write_1h": 60486,
   "cost_usd": 0.9576944,
   "wall_s": 192.0,
   "turns": 9,
   "prompt_last": 65000,
   "lines_added": 524,
   "lines_deleted": 19,
   "lines": 543,
   "files": 14,
   "thinking": 5485,
   "commands": null,
   "extras_off": true,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "regressions": 0,
   "visible_ok": true,
   "tests_edited": 0,
   "tests_added": 23,
   "lines_changed": 252,
   "score": 17,
   "of": 17,
   "tickets": 3,
   "tickets_of": 3,
   "api_cost": 0.957694
  },
  {
   "run": "claude-opus-5-5__control__B2__r2__94b8e9",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "B2",
   "skill": "control",
   "rep": 2,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 22864,
   "input": 14,
   "cache_read": 238216,
   "cache_write": 58175,
   "warmup": null,
   "cache_write_1h": 58175,
   "cost_usd": 0.9703792,
   "wall_s": 212.4,
   "turns": 8,
   "prompt_last": 62779,
   "lines_added": 564,
   "lines_deleted": 21,
   "lines": 585,
   "files": 16,
   "thinking": 6354,
   "commands": null,
   "extras_off": true,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "regressions": 0,
   "visible_ok": true,
   "tests_edited": 0,
   "tests_added": 26,
   "lines_changed": 268,
   "score": 17,
   "of": 17,
   "tickets": 3,
   "tickets_of": 3,
   "api_cost": 0.970379
  },
  {
   "run": "claude-opus-5-5__control__B2__r3__e7cc6a",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "B2",
   "skill": "control",
   "rep": 3,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 21047,
   "input": 12,
   "cache_read": 191643,
   "cache_write": 57326,
   "warmup": null,
   "cache_write_1h": 57326,
   "cost_usd": 0.9179246,
   "wall_s": 192.9,
   "turns": 8,
   "prompt_last": 61929,
   "lines_added": 495,
   "lines_deleted": 15,
   "lines": 510,
   "files": 16,
   "thinking": 6129,
   "commands": null,
   "extras_off": true,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "regressions": 0,
   "visible_ok": true,
   "tests_edited": 0,
   "tests_added": 19,
   "lines_changed": 223,
   "score": 17,
   "of": 17,
   "tickets": 3,
   "tickets_of": 3,
   "api_cost": 0.917925
  },
  {
   "run": "claude-sonnet-5-5__control__B2__r1__62f839",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "B2",
   "skill": "control",
   "rep": 1,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 9551,
   "input": 14,
   "cache_read": 195270,
   "cache_write": 34363,
   "warmup": null,
   "cache_write_1h": 34363,
   "cost_usd": 0.272044,
   "wall_s": 63.0,
   "turns": 8,
   "prompt_last": 44599,
   "lines_added": 286,
   "lines_deleted": 30,
   "lines": 316,
   "files": 9,
   "thinking": 1232,
   "commands": null,
   "extras_off": true,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "regressions": 0,
   "visible_ok": true,
   "tests_edited": 0,
   "tests_added": 6,
   "lines_changed": 234,
   "score": 17,
   "of": 17,
   "tickets": 3,
   "tickets_of": 3,
   "api_cost": 0.272044
  },
  {
   "run": "claude-sonnet-5-5__control__B2__r2__2b1af1",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "B2",
   "skill": "control",
   "rep": 2,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 8098,
   "input": 12,
   "cache_read": 151191,
   "cache_write": 32485,
   "warmup": null,
   "cache_write_1h": 32485,
   "cost_usd": 0.24118219999999999,
   "wall_s": 54.2,
   "turns": 7,
   "prompt_last": 42721,
   "lines_added": 215,
   "lines_deleted": 13,
   "lines": 228,
   "files": 9,
   "thinking": 1405,
   "commands": null,
   "extras_off": true,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "regressions": 0,
   "visible_ok": true,
   "tests_edited": 0,
   "tests_added": 5,
   "lines_changed": 181,
   "score": 17,
   "of": 17,
   "tickets": 3,
   "tickets_of": 3,
   "api_cost": 0.241182
  },
  {
   "run": "claude-sonnet-5-5__control__B2__r3__47160b",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "B2",
   "skill": "control",
   "rep": 3,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 8511,
   "input": 18,
   "cache_read": 246891,
   "cache_write": 32872,
   "warmup": null,
   "cache_write_1h": 32872,
   "cost_usd": 0.26601220000000003,
   "wall_s": 77.3,
   "turns": 11,
   "prompt_last": 43108,
   "lines_added": 239,
   "lines_deleted": 11,
   "lines": 250,
   "files": 8,
   "thinking": 1189,
   "commands": null,
   "extras_off": true,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "regressions": 0,
   "visible_ok": true,
   "tests_edited": 0,
   "tests_added": 6,
   "lines_changed": 185,
   "score": 17,
   "of": 17,
   "tickets": 3,
   "tickets_of": 3,
   "api_cost": 0.266012
  },
  {
   "run": "gpt-5.6-sol__control__B2__r1__c47cb5",
   "model": "gpt-5.6-sol",
   "host": "codex",
   "task": "B2",
   "skill": "control",
   "rep": 1,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 10160,
   "input": 361327,
   "cache_read": 319360,
   "cache_write": 52420,
   "warmup": 10453,
   "cache_write_1h": null,
   "cost_usd": 0.540624,
   "wall_s": 528.2,
   "turns": null,
   "prompt_last": null,
   "lines_added": 277,
   "lines_deleted": 18,
   "lines": 295,
   "files": 10,
   "thinking": 2672,
   "commands": 10,
   "extras_off": true,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "regressions": 0,
   "visible_ok": true,
   "tests_edited": 0,
   "tests_added": 6,
   "lines_changed": 206,
   "score": 17,
   "of": 17,
   "tickets": 3,
   "tickets_of": 3,
   "api_cost": 0.540624
  },
  {
   "run": "gpt-5.6-sol__control__B2__r2__554763",
   "model": "gpt-5.6-sol",
   "host": "codex",
   "task": "B2",
   "skill": "control",
   "rep": 2,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 10904,
   "input": 402154,
   "cache_read": 370048,
   "cache_write": 42559,
   "warmup": 10453,
   "cache_write_1h": null,
   "cost_usd": 0.536335,
   "wall_s": 547.4,
   "turns": null,
   "prompt_last": null,
   "lines_added": 271,
   "lines_deleted": 12,
   "lines": 283,
   "files": 11,
   "thinking": 3914,
   "commands": 12,
   "extras_off": true,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "regressions": 0,
   "visible_ok": true,
   "tests_edited": 0,
   "tests_added": 6,
   "lines_changed": 189,
   "score": 17,
   "of": 17,
   "tickets": 3,
   "tickets_of": 3,
   "api_cost": 0.536335
  },
  {
   "run": "gpt-5.6-sol__control__B2__r3__66a519",
   "model": "gpt-5.6-sol",
   "host": "codex",
   "task": "B2",
   "skill": "control",
   "rep": 3,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 10640,
   "input": 507676,
   "cache_read": 475648,
   "cache_write": 42481,
   "warmup": 10453,
   "cache_write_1h": null,
   "cost_usd": 0.572983,
   "wall_s": 550.6,
   "turns": null,
   "prompt_last": null,
   "lines_added": 322,
   "lines_deleted": 18,
   "lines": 340,
   "files": 11,
   "thinking": 2415,
   "commands": 9,
   "extras_off": true,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "regressions": 0,
   "visible_ok": true,
   "tests_edited": 0,
   "tests_added": 7,
   "lines_changed": 219,
   "score": 17,
   "of": 17,
   "tickets": 3,
   "tickets_of": 3,
   "api_cost": 0.572983
  },
  {
   "run": "gpt-6-sol__control__B2__r1__3cfaa1",
   "model": "gpt-6-sol",
   "host": "codex",
   "task": "B2",
   "skill": "control",
   "rep": 1,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 7705,
   "input": 421996,
   "cache_read": 372736,
   "cache_write": 60263,
   "warmup": 11003,
   "cache_write_1h": null,
   "cost_usd": 0.272123,
   "wall_s": 406.7,
   "turns": null,
   "prompt_last": null,
   "lines_added": 229,
   "lines_deleted": 16,
   "lines": 245,
   "files": 9,
   "thinking": 1228,
   "commands": 14,
   "extras_off": true,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "regressions": 0,
   "visible_ok": true,
   "tests_edited": 0,
   "tests_added": 5,
   "lines_changed": 168,
   "score": 17,
   "of": 17,
   "tickets": 3,
   "tickets_of": 3,
   "api_cost": 0.272123
  },
  {
   "run": "gpt-6-sol__control__B2__r2__118bb7",
   "model": "gpt-6-sol",
   "host": "codex",
   "task": "B2",
   "skill": "control",
   "rep": 2,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 7900,
   "input": 372026,
   "cache_read": 343040,
   "cache_write": 39989,
   "warmup": 11003,
   "cache_write_1h": null,
   "cost_usd": 0.227586,
   "wall_s": 416.5,
   "turns": null,
   "prompt_last": null,
   "lines_added": 253,
   "lines_deleted": 14,
   "lines": 267,
   "files": 10,
   "thinking": 1644,
   "commands": 17,
   "extras_off": true,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "regressions": 0,
   "visible_ok": true,
   "tests_edited": 0,
   "tests_added": 7,
   "lines_changed": 176,
   "score": 17,
   "of": 17,
   "tickets": 3,
   "tickets_of": 3,
   "api_cost": 0.227586
  },
  {
   "run": "gpt-6-sol__control__B2__r3__450d9d",
   "model": "gpt-6-sol",
   "host": "codex",
   "task": "B2",
   "skill": "control",
   "rep": 3,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 7583,
   "input": 384968,
   "cache_read": 319104,
   "cache_write": 76867,
   "warmup": 11003,
   "cache_write_1h": null,
   "cost_usd": 0.293385,
   "wall_s": 427.5,
   "turns": null,
   "prompt_last": null,
   "lines_added": 198,
   "lines_deleted": 22,
   "lines": 220,
   "files": 9,
   "thinking": 1411,
   "commands": 17,
   "extras_off": true,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "regressions": 0,
   "visible_ok": true,
   "tests_edited": 0,
   "tests_added": 5,
   "lines_changed": 167,
   "score": 17,
   "of": 17,
   "tickets": 3,
   "tickets_of": 3,
   "api_cost": 0.293385
  },
  {
   "run": "gpt-6.1-sol__control__B2__r1__1968e4",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "B2",
   "skill": "control",
   "rep": 1,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 9215,
   "input": 261766,
   "cache_read": 233344,
   "cache_write": 39938,
   "warmup": 11516,
   "cache_write_1h": null,
   "cost_usd": 0.19536,
   "wall_s": 494.0,
   "turns": null,
   "prompt_last": null,
   "lines_added": 387,
   "lines_deleted": 19,
   "lines": 406,
   "files": 10,
   "thinking": 492,
   "commands": 9,
   "extras_off": true,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "regressions": 0,
   "visible_ok": true,
   "tests_edited": 0,
   "tests_added": 13,
   "lines_changed": 166,
   "score": 17,
   "of": 17,
   "tickets": 3,
   "tickets_of": 3,
   "api_cost": 0.19536
  },
  {
   "run": "gpt-6.1-sol__control__B2__r2__96eb23",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "B2",
   "skill": "control",
   "rep": 2,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 8808,
   "input": 303258,
   "cache_read": 270592,
   "cache_write": 44182,
   "warmup": 11516,
   "cache_write_1h": null,
   "cost_usd": 0.203503,
   "wall_s": 456.9,
   "turns": null,
   "prompt_last": null,
   "lines_added": 363,
   "lines_deleted": 14,
   "lines": 377,
   "files": 10,
   "thinking": 487,
   "commands": 9,
   "extras_off": true,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "regressions": 0,
   "visible_ok": true,
   "tests_edited": 0,
   "tests_added": 11,
   "lines_changed": 152,
   "score": 17,
   "of": 17,
   "tickets": 3,
   "tickets_of": 3,
   "api_cost": 0.203503
  },
  {
   "run": "gpt-6.1-sol__control__B2__r3__9f4c15",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "B2",
   "skill": "control",
   "rep": 3,
   "round": 13,
   "date": "2026-09-30",
   "exit": 0,
   "passed": true,
   "output": 7157,
   "input": 222430,
   "cache_read": 195968,
   "cache_write": 37978,
   "warmup": 11516,
   "cache_write_1h": null,
   "cost_usd": 0.167123,
   "wall_s": 372.4,
   "turns": null,
   "prompt_last": null,
   "lines_added": 302,
   "lines_deleted": 20,
   "lines": 322,
   "files": 10,
   "thinking": 167,
   "commands": 9,
   "extras_off": true,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "regressions": 0,
   "visible_ok": true,
   "tests_edited": 0,
   "tests_added": 11,
   "lines_changed": 175,
   "score": 17,
   "of": 17,
   "tickets": 3,
   "tickets_of": 3,
   "api_cost": 0.167123
  }
 ],
 "replaced_runs": [
  {
   "run": "claude-opus-5-5__control__B1__r1__69a5b8",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "B1",
   "skill": "control",
   "rep": 1,
   "round": 11,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 9009,
   "input": 16,
   "cache_read": 259378,
   "cache_write": 38004,
   "warmup": null,
   "cache_write_1h": 38004,
   "cost_usd": 0.5361516,
   "wall_s": 82.1,
   "turns": 9,
   "prompt_last": 48124,
   "lines_added": 183,
   "lines_deleted": 24,
   "lines": 207,
   "files": 15,
   "thinking": 2022,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "replaced_by": 13,
   "tests_edited": 0,
   "score": 20,
   "of": 20,
   "api_cost": 0.536152
  },
  {
   "run": "claude-opus-5-5__control__B1__r2__e56011",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "B1",
   "skill": "control",
   "rep": 2,
   "round": 11,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 7285,
   "input": 16,
   "cache_read": 242964,
   "cache_write": 34013,
   "warmup": null,
   "cache_write_1h": 34013,
   "cost_usd": 0.4664608,
   "wall_s": 69.9,
   "turns": 9,
   "prompt_last": 44133,
   "lines_added": 115,
   "lines_deleted": 23,
   "lines": 138,
   "files": 13,
   "thinking": 1737,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "replaced_by": 13,
   "tests_edited": 0,
   "score": 20,
   "of": 20,
   "api_cost": 0.466461
  },
  {
   "run": "claude-opus-5-5__control__B1__r3__a6d592",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "B1",
   "skill": "control",
   "rep": 3,
   "round": 11,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 7209,
   "input": 18,
   "cache_read": 288805,
   "cache_write": 35370,
   "warmup": null,
   "cache_write_1h": 35370,
   "cost_usd": 0.484973,
   "wall_s": 69.3,
   "turns": 10,
   "prompt_last": 45490,
   "lines_added": 109,
   "lines_deleted": 25,
   "lines": 134,
   "files": 14,
   "thinking": 1464,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "replaced_by": 13,
   "tests_edited": 0,
   "score": 20,
   "of": 20,
   "api_cost": 0.484973
  },
  {
   "run": "claude-opus-5-5__control__B2__r1__87bdc9",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "B2",
   "skill": "control",
   "rep": 1,
   "round": 12,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 16101,
   "input": 20,
   "cache_read": 371537,
   "cache_write": 47895,
   "warmup": null,
   "cache_write_1h": 47895,
   "cost_usd": 0.7795673999999999,
   "wall_s": 134.0,
   "turns": 11,
   "prompt_last": 58015,
   "lines_added": 400,
   "lines_deleted": 21,
   "lines": 421,
   "files": 13,
   "thinking": 2499,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "replaced_by": 13,
   "tests_edited": 0,
   "score": 17,
   "of": 17,
   "api_cost": 0.779567
  },
  {
   "run": "claude-opus-5-5__control__B2__r2__e71ba6",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "B2",
   "skill": "control",
   "rep": 2,
   "round": 12,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 14888,
   "input": 20,
   "cache_read": 400424,
   "cache_write": 48392,
   "warmup": null,
   "cache_write_1h": 48392,
   "cost_usd": 0.7650608,
   "wall_s": 128.4,
   "turns": 11,
   "prompt_last": 58512,
   "lines_added": 378,
   "lines_deleted": 19,
   "lines": 397,
   "files": 13,
   "thinking": 2215,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "replaced_by": 13,
   "tests_edited": 0,
   "score": 17,
   "of": 17,
   "api_cost": 0.765061
  },
  {
   "run": "claude-opus-5-5__control__B2__r3__3fa085",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "B2",
   "skill": "control",
   "rep": 3,
   "round": 12,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 16585,
   "input": 26,
   "cache_read": 535551,
   "cache_write": 47712,
   "warmup": null,
   "cache_write_1h": 47712,
   "cost_usd": 0.8206102,
   "wall_s": 145.0,
   "turns": 14,
   "prompt_last": 57832,
   "lines_added": 415,
   "lines_deleted": 23,
   "lines": 438,
   "files": 13,
   "thinking": 2675,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "replaced_by": 13,
   "tests_edited": 0,
   "score": 17,
   "of": 17,
   "api_cost": 0.82061
  },
  {
   "run": "claude-opus-5-5__caveman__T3__r1__3b54cf",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "caveman",
   "rep": 1,
   "round": 6,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1200,
   "input": 4,
   "cache_read": 39035,
   "cache_write": 19603,
   "warmup": null,
   "cache_write_1h": 19603,
   "cost_usd": 0.188647,
   "wall_s": 15.3,
   "turns": 2,
   "prompt_last": 29723,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 298,
   "commands": null,
   "answer": "**Monthly totals**\n\n`monthly_report(entries, year, month)` works in three steps. First, `entries_in_month` keeps only entries whose `day.year` and `day.month` match the arguments. Second, `total` sums `amount` over those entries. Third, `sum_by_category` loops over the same entries and adds each amount into a dict keyed by category, starting each category at `0.0`. The report returns a dict with `count`, `total`, and `by_category`.\n\nCategories are already lowercase because `add_entry` lowercases them on insert. So \"Food\" and \"food\" land in one bucket. Amounts are plain floats, so totals can carry float rounding error, for example `0.1 + 0.2`. The month arguments are not validated. `month=13` or a month with no entries returns count 0, total `0`, and an empty `by_category`. That `total` is the int `0` from `sum()` on an empty list, not `0.0`.\n\n**Invalid date strings**\n\n`parse_date` splits the text on `-`. It then requires exactly three parts, all passing `str.isdigit()`. Anything else raises `ValueError(\"bad date: '...'\")`. Examples: `\"2024/01/05\"`, `\"2024-1\"`, `\"2024--05\"` (empty part), and `\"-2024-01-05\"` (four parts).\n\nSome strings pass that check but still fail later, with different messages:\n- Correct shape but impossible date, such as `\"2024-02-30\"` or `\"2024-13-01\"`. `datetime.date` raises its own `ValueError`, for example `day is out of range for month`. The message is not \"bad date\".\n- Unicode digits such as `\"\uff12\uff10\uff12\uff14-01-01\"` (fullwidth) pass `isdigit()` and `int()` accepts them, so the date parses. Characters such as `\"\u00b2\"` pass `isdigit()`, but `int()` rejects them with `ValueError: invalid literal for int()`.\n- Non-string input such as `None` raises `AttributeError` from `.split`, not `ValueError`.\n\nThe format is not strictly checked. `\"2024-1-5\"` and `\"0002024-01-05\"` both parse fine.\n\nIn `add_entry`, the amount check runs first. A non-positive amount raises `\"amount must be positive\"` even when the date is also bad. `parse_date` runs while the `Entry` is built, before `append`. So any date error leaves the `entries` list unchanged. No error is caught anywhere, so every error reaches the caller.\n\nI did not change any files.",
   "account_remark": false,
   "replaced_by": 13,
   "api_cost": 0.188647
  },
  {
   "run": "claude-opus-5-5__caveman__T3__r2__add1c0",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "caveman",
   "rep": 2,
   "round": 6,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1364,
   "input": 4,
   "cache_read": 28329,
   "cache_write": 10847,
   "warmup": null,
   "cache_write_1h": 10847,
   "cost_usd": 0.1197378,
   "wall_s": 14.4,
   "turns": 2,
   "prompt_last": 20967,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 412,
   "commands": null,
   "answer": "`monthly_report(entries, year, month)` at `ledger/core.py:51` computes monthly totals in two steps. First, `entries_in_month` keeps only entries whose `day.year` and `day.month` match the arguments. Each `Entry.day` is already a `datetime.date`, so this step compares plain integers and parses no strings. Second, the function builds a dict from the filtered list. `count` is the list length. `total` comes from `total()`, which sums each `amount` with `sum()`. `by_category` comes from `sum_by_category()`, which adds amounts into a dict keyed by category. Categories are lowercased once in `add_entry`, so \"Food\" and \"food\" land in the same bucket. Amounts are floats and nothing rounds them, so float drift can show up, for example `0.1 + 0.2` gives `0.30000000000000004`. A month with no entries returns `count` 0, `total` 0 (the integer 0 from an empty `sum`, not 0.0) and an empty `by_category`. The function does not check `year` or `month`, so `month=13` gives an empty report, not an error.\n\nDate strings are validated only once, when `add_entry` calls `parse_date` (`ledger/core.py:15`). `parse_date` splits the text on `-`. It needs exactly three parts, and each part must pass `str.isdigit()`. Otherwise it raises `ValueError(\"bad date: '<text>'\")`. Inputs like `\"2026/01/05\"`, `\"2026-01\"`, `\"2026--05\"` and `\"-2026-01-05\"` fail this way. Some strings pass the shape check but still fail when `date(y, m, d)` builds the date. In that case Python's own `ValueError` comes back, for example `\"day is out of range for month\"` for `\"2026-02-30\"`, or a year error for `\"12345-01-01\"`. Unicode digits such as `\"\u00b2\"` pass `isdigit()` but make `int()` raise a `ValueError`. The format is looser than the docstring says: `\"2026-1-5\"` is accepted because nothing enforces zero-padding.\n\n`add_entry` raises before `append`, so a bad date leaves the ledger unchanged. The error goes to the caller because nothing catches it. The amount check runs first, so a non-positive amount with a bad date reports `\"amount must be positive\"`. Bad dates never reach `monthly_report`, because stored entries always hold valid `date` objects.\n\nI changed no files.",
   "account_remark": true,
   "replaced_by": 13,
   "api_cost": 0.119738,
   "light": true
  },
  {
   "run": "claude-opus-5-5__caveman__T3__r3__ba56c5",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "caveman",
   "rep": 3,
   "round": 6,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1384,
   "input": 4,
   "cache_read": 39064,
   "cache_write": 19632,
   "warmup": null,
   "cache_write_1h": 19632,
   "cost_usd": 0.1925648,
   "wall_s": 17.0,
   "turns": 2,
   "prompt_last": 29752,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 445,
   "commands": null,
   "answer": "## Monthly totals\n\n`monthly_report(entries, year, month)` computes the report for one month that the caller picks. It does not group every month at once.\n\n1. `entries_in_month` filters the list. It keeps each entry where `e.day.year == year` and `e.day.month == month`. Each `day` is already a `date` object, because `add_entry` parsed it when the entry was stored.\n2. `total` runs `sum(e.amount for e in ...)` over those entries. Each amount is a `float`, so normal float rounding applies. For example, 0.1 + 0.2 does not come out as exactly 0.3.\n3. `sum_by_category` builds a dict that maps each category to the sum of its amounts, starting from `0.0`. Categories are lowercased in `add_entry`, so \"Food\" and \"food\" fall into the same bucket.\n4. The return value is `{\"count\": ..., \"total\": ..., \"by_category\": ...}`.\n\nFor a month with no entries, the result is `count` 0, `total` 0 and an empty `by_category`. That `total` is the integer `0`, not `0.0`, because `sum` of an empty sequence returns `0`.\n\n## Invalid date strings\n\n`parse_date` checks the string in two stages.\n\n1. **Shape check.** It splits the string on `\"-\"`. It raises `ValueError(\"bad date: '...'\")` unless there are exactly 3 parts and every part passes `isdigit()`. So `\"2024/01/05\"`, `\"2024-01\"`, `\"-2024-01-05\"` and `\" 2024-01-05\"` (leading space) all fail here.\n2. **Range check.** It converts the parts to `int` and calls `date(y, m, d)`. That call raises `ValueError` with Python's own message, not \"bad date\". For example, `\"2024-02-30\"` gives \"day is out of range for month\", and year 0 also fails.\n\nThe check is looser than the \"YYYY-MM-DD\" docstring says. `\"2024-1-5\"` and `\"02024-01-05\"` both pass, because only the digit count per part is unchecked. A non-string input, such as `None`, raises `AttributeError` from `.split`, not `ValueError`.\n\nInside `add_entry`, the error stops the whole call:\n- The amount check runs first. If both the amount and the date are bad, you get \"amount must be positive\".\n- `parse_date` runs while the `Entry` is built, before `append`. On a bad date, the error goes up to the caller and the `entries` list stays unchanged. No partial entry is stored.\n- Nothing catches the error inside `core.py`.\n\nI didn't change any files.",
   "account_remark": false,
   "replaced_by": 13,
   "api_cost": 0.192565
  },
  {
   "run": "claude-opus-5-5__control__T3__r1__6a9b02",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "control",
   "rep": 1,
   "round": 6,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1666,
   "input": 4,
   "cache_read": 25776,
   "cache_write": 10297,
   "warmup": null,
   "cache_write_1h": 10297,
   "cost_usd": 0.12086720000000001,
   "wall_s": 16.9,
   "turns": 2,
   "prompt_last": 20417,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 463,
   "commands": null,
   "answer": "## How monthly totals are computed\n\n`monthly_report(entries, year, month)` (`ledger/core.py:51`) works in three steps.\n\n1. **Pick the month's entries.** `entries_in_month` (`ledger/core.py:40`) keeps every entry whose `day.year` and `day.month` equal the given `year` and `month`. It never checks those arguments. A month like `13` or a year with no data simply matches nothing and gives an empty list, not an error.\n\n2. **Add them up.** The report returns a dict with three keys:\n   - `count`: how many entries fall in that month.\n   - `total`: the sum of their `amount` fields, from `total()` (`ledger/core.py:31`). Amounts are stored as floats, so the total can pick up normal float rounding error. For example, `0.1 + 0.2` gives `0.30000000000000004`.\n   - `by_category`: a dict mapping each category to its summed amount, from `sum_by_category` (`ledger/core.py:44`). Categories are lowercased when an entry is added, so `\"Food\"` and `\"food\"` are grouped together.\n\n3. **Empty months.** If no entries match, the result is `{\"count\": 0, \"total\": 0, \"by_category\": {}}`. Here `total` is the integer `0`, not `0.0`, because `sum()` of an empty sequence returns an int. When at least one entry matches, `total` is a float.\n\n## What happens when a date string is invalid\n\nDates are checked only when an entry is added. `add_entry` calls `parse_date` (`ledger/core.py:15`), which checks the string in two stages.\n\n1. **Shape check.** It splits the string on `-` and requires exactly three parts, each made only of digits. Anything else raises `ValueError(\"bad date: '...'\")`. That covers `\"2024/01/05\"`, `\"2024-1\"`, `\" 2024-01-05\"` (leading space), `\"2024-01-05T10:00\"` and `\"-2024-01-01\"`. The check doesn't require four-digit years or two-digit months, so `\"24-1-5\"` passes and becomes the year 24.\n\n2. **Calendar check.** Strings that pass the shape check go to `datetime.date(y, m, d)`. That raises its own `ValueError` for impossible dates such as `\"2024-02-30\"`, `\"2024-13-01\"` or `\"0000-01-01\"`. The message is Python's own, for example \"day is out of range for month\", not the \"bad date\" text.\n\nIn both cases the error reaches the caller and **the entry is not added**. `parse_date` runs while the `Entry` is being built, before `entries.append`, so the ledger is never partly updated.\n\nSome edge cases:\n- **Order of checks:** `add_entry` checks `amount <= 0` before it parses the date. A call with both a bad amount and a bad date reports only the amount error.\n- **Non-string dates:** passing `None` or a `date` object raises `AttributeError` (no `.split`), not `ValueError`.\n- **Unusual digits:** `str.isdigit()` accepts some Unicode digit characters, such as Arabic-Indic numerals, and `int()` parses them. So some non-ASCII date strings are accepted.\n\nI didn't change any files.",
   "account_remark": true,
   "replaced_by": 13,
   "api_cost": 0.120867,
   "light": true
  },
  {
   "run": "claude-opus-5-5__control__T3__r2__cc0d17",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "control",
   "rep": 2,
   "round": 6,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1645,
   "input": 4,
   "cache_read": 36384,
   "cache_write": 16950,
   "warmup": null,
   "cache_write_1h": 16950,
   "cost_usd": 0.1757928,
   "wall_s": 18.5,
   "turns": 2,
   "prompt_last": 27070,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 269,
   "commands": null,
   "answer": "Monthly totals come from `monthly_report` in `ledger/core.py`, and an invalid date string always ends in a `ValueError`. Nothing gets stored when that happens. I didn't change any files.\n\n## How monthly totals are computed\n\n`monthly_report(entries, year, month)` (`ledger/core.py:51`) works in three steps:\n\n1. **Pick the month's entries.** `entries_in_month` (`ledger/core.py:40`) keeps every entry whose `day` has the requested year and month. It compares the year and month numbers directly, so `month` must be an integer from 1 to 12. Anything else just returns no entries instead of raising an error. Entries keep their original order.\n\n2. **Add up the amounts.** `total` (`ledger/core.py:31`) sums the `amount` of each matching entry. Amounts were converted to `float` when added, so the result is a plain float. It isn't rounded or held to cents, so you can see floating-point artifacts like `0.30000000000000004`. For a month with no entries, the total is the integer `0` rather than `0.0`, because `sum` of an empty sequence returns `0`.\n\n3. **Break it down by category.** `sum_by_category` (`ledger/core.py:44`) builds a dictionary that maps each category to the sum of its amounts, with each running total starting at `0.0`. Categories were lowercased when entries were added, so \"Food\" and \"food\" land in the same bucket. A month with no entries gives an empty dict.\n\nThe report is a dict with three keys: `count` (how many entries matched), `total`, and `by_category`. The category values always add up to `total`, apart from small differences from the order the floats were added in.\n\n## What happens when a date string is invalid\n\nDates are only parsed in one place: `add_entry` calls `parse_date` (`ledger/core.py:15`), which checks the input in two stages:\n\n1. **Shape check.** It splits the string on `-` and requires exactly three parts, each made only of digits. Anything else raises `ValueError(\"bad date: '...'\")`. That covers `2024/01/05`, `2024-01`, `2024-01-05T10:00`, surrounding spaces, empty parts like `2024--05`, and negative numbers. The message quotes the bad input.\n\n2. **Calendar check.** Once the shape is right, it converts the three parts to integers and calls `datetime.date(y, m, d)`. Impossible dates such as `2024-13-01`, `2023-02-29` or `2024-00-10` pass the shape check but fail here. `date` raises its own `ValueError`, e.g. \"month must be in 1..12\", so the message is different and doesn't quote the input.\n\nSome less obvious behaviours:\n\n- **The format is looser than \"YYYY-MM-DD\" suggests.** Zero-padding isn't required, so `2024-1-5` is accepted. Short or odd years like `24-01-05` also pass, as long as the year is between 1 and 9999.\n- **Unusual digit characters still raise `ValueError`.** `str.isdigit()` returns true for characters like `\u00b2`, which then make `int()` raise `ValueError`. The error type stays the same, but the message differs.\n- **Non-string input gives a different error.** Passing `None` or a `date` object raises `AttributeError` (there's no `.split`), not `ValueError`.\n- **The ledger is left untouched.** `add_entry` checks the amount first, then parses the date inside the call that builds the `Entry`. If the date is bad, the exception is raised before `entries.append` runs, so the list isn't changed. The error goes straight to the caller, because nothing in the module catches it.\n- **A bad amount is reported first.** If both the amount (\u2264 0) and the date are invalid, you get the \"amount must be positive\" error, because that check runs first.\n\nEvery stored entry already holds a valid `date`, so the monthly report never deals with date strings and can't fail because of a bad date.",
   "account_remark": false,
   "replaced_by": 13,
   "api_cost": 0.175793
  },
  {
   "run": "claude-opus-5-5__control__T3__r3__fd24b3",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "control",
   "rep": 3,
   "round": 6,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1576,
   "input": 4,
   "cache_read": 36386,
   "cache_write": 16952,
   "warmup": null,
   "cache_write_1h": 16952,
   "cost_usd": 0.1744292,
   "wall_s": 17.2,
   "turns": 2,
   "prompt_last": 27072,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 407,
   "commands": null,
   "answer": "Nothing was changed. Here's how `ledger/core.py` behaves.\n\n## How monthly totals are computed\n\nThe main function is `monthly_report(entries, year, month)` (`ledger/core.py:51`). It works in two steps.\n\n**1. Pick out the month's entries.** `entries_in_month` (`ledger/core.py:40`) goes through the whole list and keeps each entry whose date has the same year and month as the arguments. It checks both, so March 2024 and March 2025 are kept apart. It never checks that `year` and `month` are sensible: asking for month 13 doesn't raise an error, it just finds nothing.\n\n**2. Add up that smaller list three ways:**\n- **`count`** is the number of entries found.\n- **`total`** comes from `total()` (`ledger/core.py:31`), which adds up every entry's `amount`. Amounts are stored as `float`, so large sums can pick up small rounding errors (e.g. `0.1 + 0.2` gives `0.30000000000000004`).\n- **`by_category`** comes from `sum_by_category()` (`ledger/core.py:44`). It builds a dictionary that maps each category to its running total, starting each category at `0.0`. Categories are lowercased when an entry is added, so \"Food\" and \"food\" end up in the same bucket.\n\nFor a month with no entries, the report is `{\"count\": 0, \"total\": 0, \"by_category\": {}}`. That `total` is the integer `0`, not `0.0`, because Python's `sum()` of an empty list returns `0`.\n\n## What happens with an invalid date string\n\nDates are parsed by `parse_date` (`ledger/core.py:15`), which only `add_entry` calls. An invalid date can fail in two ways, and both raise `ValueError`:\n\n1. **Wrong format.** The string is split on `-`. If it doesn't give exactly three parts made only of digits, the function raises `ValueError(\"bad date: '...'\")`. This catches things like `\"2024/01/05\"`, `\"2024-01\"`, `\"Jan 5\"`, extra spaces, or a trailing time such as `\"2024-01-05T10:00\"`.\n2. **Right format, impossible date.** If the format passes, the numbers go to `datetime.date(y, m, d)`. That raises its own `ValueError` for dates that don't exist, like `\"2024-02-30\"`, `\"2024-13-01\"` or year `0000`. Its message is different, something like \"day is out of range for month\", not the \"bad date\" text.\n\nThe format check is looser than the docstring says: it doesn't require zero-padding or a four-digit year. So `\"2024-1-5\"` and `\"99-1-1\"` are accepted (the second becomes year 99 AD).\n\nIn `add_entry` (`ledger/core.py:24`), the error is not caught. It goes straight back to the caller. Two things follow:\n- **The amount is checked first.** If the amount is zero or negative and the date is also bad, the caller gets the \"amount must be positive\" error, not the date error.\n- **The list is left untouched.** The date is parsed while the `Entry` is being built, before `entries.append` runs. So a bad date means nothing is added, and there's no half-added entry.\n\nBecause every stored entry has already passed this check, the month-grouping and summing functions never see an invalid date.",
   "account_remark": false,
   "replaced_by": 13,
   "api_cost": 0.174429
  },
  {
   "run": "claude-opus-5-5__karpathy__T3__r1__5d6dce",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "karpathy",
   "rep": 1,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1688,
   "input": 4,
   "cache_read": 26794,
   "cache_write": 11318,
   "warmup": null,
   "cache_write_1h": 11318,
   "cost_usd": 0.12967879999999998,
   "wall_s": 19.2,
   "turns": 2,
   "prompt_last": 21438,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 462,
   "commands": null,
   "answer": "## How monthly totals are computed\n\n`monthly_report(entries, year, month)` (`ledger/core.py:51`) works in three steps:\n\n1. **Filter.** `entries_in_month` keeps entries whose `day.year` and `day.month` equal the requested values. It's a linear scan, and entries are never grouped or indexed.\n2. **Grand total.** `total` adds up the `amount` of the filtered entries with Python's `sum`.\n3. **Per-category totals.** `sum_by_category` builds a dict keyed by category. Each category starts at `0.0` and gains the amount of each entry in it.\n\nIt returns `{\"count\": ..., \"total\": ..., \"by_category\": {...}}`.\n\nSome details that follow from this:\n\n- **Categories are lowercased once, when an entry is added** (`add_entry`, line 27). So \"Food\" and \"food\" end up in the same bucket. Nothing in the report step normalizes them again, so an `Entry` built directly without `add_entry` could keep a mixed-case category.\n- **Amounts are floats**, so totals can pick up rounding error (for example, `0.1 + 0.2` gives `0.30000000000000004`). No rounding or `Decimal` is used.\n- **An empty month returns `total` as the integer `0`, not `0.0`.** That's because `sum` of an empty sequence starts at `0`. `by_category` is `{}`.\n- **`year` and `month` aren't validated.** `month=13` doesn't raise an error. It just matches nothing and returns an empty report.\n\n## What happens with an invalid date string\n\nDates are only parsed in `add_entry`, which calls `parse_date` (line 15). There are two ways a date can fail:\n\n1. **Wrong format.** The string is split on `-`. If there aren't exactly three parts, or any part isn't all digits, the function raises `ValueError(\"bad date: '...'\")`. This catches things like `\"2024/01/05\"`, `\"2024-1\"`, `\"2024-01-05T00:00\"`, `\" 2024-01-05\"` and `\"2024--05\"` (an empty part fails `isdigit`).\n2. **Right format, impossible date.** Strings like `\"2024-02-30\"` or `\"2024-13-01\"` pass the format check. The error then comes from `datetime.date(...)` itself. It's still a `ValueError`, but with the standard library's message (e.g. \"day is out of range for month\") rather than \"bad date\".\n\nEither way, the exception goes straight up to the caller. `add_entry` doesn't catch it, and the `Entry` is built inside the `append(...)` call. So a bad date means **nothing is appended and the list stays unchanged**. Also, the amount check runs first, so an entry with both a non-positive amount and a bad date reports the amount error.\n\nSome edge cases the format check doesn't catch:\n\n- **Field lengths aren't checked.** `\"24-1-5\"` is accepted as the year 24.\n- **Unicode digits.** `str.isdigit()` accepts characters like `\"\u00b2\"`, which `int()` then rejects. The result is a `ValueError` with a different message.\n- **Non-string input.** Passing something like a `date` object fails with an `AttributeError` on `.split` instead of a `ValueError`.\n\nI didn't change any files.",
   "account_remark": true,
   "replaced_by": 13,
   "api_cost": 0.129679,
   "light": true
  },
  {
   "run": "claude-opus-5-5__karpathy__T3__r2__4aad8e",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "karpathy",
   "rep": 2,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1738,
   "input": 4,
   "cache_read": 37427,
   "cache_write": 17996,
   "warmup": null,
   "cache_write_1h": 17996,
   "cost_usd": 0.18622940000000002,
   "wall_s": 19.6,
   "turns": 2,
   "prompt_last": 28116,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 513,
   "commands": null,
   "answer": "The monthly report and the date parsing both live in `ledger/core.py`. I didn't change any files.\n\n## How monthly totals are computed\n\n`monthly_report(entries, year, month)` (`ledger/core.py:51`) works in three steps:\n\n1. **Pick the month's entries.** `entries_in_month` (`ledger/core.py:40`) keeps every entry whose date has the same `year` and `month` as the ones passed in. It checks the fields of the stored `date` object directly, so a month's boundaries are exact and no date ranges are involved.\n2. **Add up the amounts.** `total` (`ledger/core.py:31`) calls `sum()` on the `amount` values of those entries. Amounts were converted to `float` when they were added, so this is ordinary floating-point addition. Small rounding errors can show up, e.g. `0.1 + 0.2` gives `0.30000000000000004`. If the month has no entries, `sum()` of nothing returns the integer `0`, not `0.0`.\n3. **Break it down by category.** `sum_by_category` (`ledger/core.py:44`) builds a dict from category to running total, starting each category at `0.0`. `add_entry` lowercases categories when it stores them, so \"Food\" and \"food\" end up in the same bucket.\n\nThe result is a dict with `count` (how many entries), `total` and `by_category`. `year` and `month` are never checked. Asking for month 13, for example, doesn't raise an error; it just returns `{\"count\": 0, \"total\": 0, \"by_category\": {}}`.\n\n## What happens with an invalid date string\n\nDates are parsed only in `parse_date` (`ledger/core.py:15`), which `add_entry` calls. A string can fail in two ways, and both raise `ValueError`:\n\n- **Wrong shape.** The string is split on `-`. If there aren't exactly three pieces, or any piece isn't all digits, it raises `ValueError(\"bad date: '...'\")`. This rejects things like `\"2024/01/05\"`, `\"2024-01\"`, `\" 2024-01-05\"` (leading space), `\"2024-01-05T00:00\"` and anything with a minus sign.\n- **Right shape, impossible date.** If the pieces are all digits, they're turned into integers and passed to `datetime.date(y, m, d)`. That raises its own `ValueError` for impossible dates, such as `\"2024-02-30\"`, `\"2024-13-01\"` or year `0000`. These messages come from Python (e.g. \"day is out of range for month\"), not from the ledger's \"bad date\" text.\n\nSome less obvious behaviour:\n\n- **Loose format.** Despite the docstring saying YYYY-MM-DD, the format isn't strict. `\"2024-1-5\"` works (no zero-padding needed), and so do digits from other scripts, such as Arabic-Indic numerals, because both `str.isdigit()` and `int()` accept them.\n- **Non-string input.** If `day` is `None` or already a `date` object, calling `.split` fails with `AttributeError`, not `ValueError`.\n- **Order of checks.** In `add_entry` the amount is checked first, so an entry with a bad amount and a bad date reports the amount error.\n- **The ledger stays unchanged on failure.** `parse_date` runs before `entries.append` while the `Entry` is being built. If it raises, nothing is added and the error goes straight back to the caller, since nothing in the module catches it.",
   "account_remark": false,
   "replaced_by": 13,
   "api_cost": 0.186229
  },
  {
   "run": "claude-opus-5-5__karpathy__T3__r3__f72642",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "karpathy",
   "rep": 3,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1630,
   "input": 4,
   "cache_read": 37423,
   "cache_write": 17990,
   "warmup": null,
   "cache_write_1h": 17990,
   "cost_usd": 0.1840206,
   "wall_s": 18.6,
   "turns": 2,
   "prompt_last": 28110,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 360,
   "commands": null,
   "answer": "I read `ledger/core.py` and didn't change anything.\n\n## How monthly totals are computed\n\n`monthly_report(entries, year, month)` (`ledger/core.py:51`) works in two steps.\n\n1. **Pick the month's entries.** `entries_in_month` (`ledger/core.py:40`) goes through the list and keeps each entry whose `day.year` and `day.month` match the arguments. Every entry holds a real `datetime.date`, so this is a plain integer comparison with no string handling. The year and month you pass in aren't checked. A request for month 13 or month 0 doesn't raise an error; it just matches nothing.\n\n2. **Add them up.** From that subset, the function returns a dict with three keys:\n   - `count`: how many entries are in the subset.\n   - `total`: the result of `total()`, which is `sum(e.amount for e in entries)`. The amounts are floats and nothing rounds them, so the usual floating-point drift can appear (for example, `0.1 + 0.2` gives `0.30000000000000004`).\n   - `by_category`: the result of `sum_by_category()`. It keeps a running float total for each category name in a dict, starting each category at `0.0`.\n\n   A month with no entries returns `{\"count\": 0, \"total\": 0, \"by_category\": {}}`. That `total` is the integer `0`, not `0.0`, because `sum()` of an empty sequence starts at `0`.\n\nCategory totals are case-insensitive in practice, because `add_entry` lowercases the category before storing it. So \"Food\" and \"food\" land in the same bucket. The totals are simply sums of the amounts. `add_entry` rejects zero and negative amounts, so there are no refunds or credits to subtract.\n\n## What happens with an invalid date string\n\nDates are only parsed in one place: `add_entry` calls `parse_date` (`ledger/core.py:15`). The date is parsed before the new `Entry` is appended, so a bad date raises an exception and leaves the list unchanged. Nothing is skipped or replaced with a default value.\n\n`parse_date` can fail in two ways, and both raise `ValueError`:\n\n1. **Wrong format.** The string is split on `-`. If it doesn't have exactly three parts, or any part contains a non-digit, the function raises `ValueError(\"bad date: '...'\")`. This catches inputs like `\"2024/01/05\"`, `\"Jan 5\"`, an empty string, a string with surrounding spaces, or `\"2024-01-05T00:00\"`.\n2. **Right format, impossible date.** Something like `\"2024-13-01\"` or `\"2023-02-30\"` passes the format check. It then fails inside `date(y, m, d)`, which raises its own `ValueError` (for example, \"month must be in 1..12\"). The message is different from the \"bad date\" one.\n\nA few edge cases:\n- **Loose format.** The function doesn't check digit counts or zero-padding, so `\"2024-1-5\"` and `\"24-01-05\"` are both accepted. The second becomes the year 24 AD, not 2024.\n- **Non-string input.** Passing `None` or a `date` object raises `AttributeError` from `.split`, not `ValueError`.\n- **Check order.** `add_entry` checks the amount before the date. A call with both a bad amount and a bad date reports the amount error (\"amount must be positive\").\n- **No date check when building `Entry` directly.** Creating an `Entry(...)` yourself skips `parse_date`. If its `day` isn't a real `date`, the monthly report fails later with an `AttributeError` when it reads `.year`.",
   "account_remark": false,
   "replaced_by": 13,
   "api_cost": 0.184021
  },
  {
   "run": "claude-opus-5-5__placebo__T3__r1__1dc7c2",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "placebo",
   "rep": 1,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1422,
   "input": 4,
   "cache_read": 37169,
   "cache_write": 17738,
   "warmup": null,
   "cache_write_1h": 17738,
   "cost_usd": 0.1777938,
   "wall_s": 14.9,
   "turns": 2,
   "prompt_last": 27858,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 273,
   "commands": null,
   "answer": "I read `ledger/core.py` and didn't change anything.\n\n**How monthly totals are computed**\n\n`monthly_report(entries, year, month)` (`ledger/core.py:51`) works in two steps. First, `entries_in_month` keeps only the entries whose `day` has the matching year and month. It checks every entry in the list, and the entries don't need to be sorted. Then it builds a dictionary from that smaller list with three fields:\n\n- **`count`**: how many entries fall in that month.\n- **`total`**: the sum of their `amount` values, from `total()`.\n- **`by_category`**: a dictionary mapping each category to the sum of its amounts, from `sum_by_category`. Categories are lowercased when an entry is added, so \"Food\" and \"food\" are added together.\n\nA few things about the results:\n\n- Amounts are stored as floats, so totals can have the usual rounding errors (for example, 0.1 + 0.2 won't be exactly 0.3).\n- The function never checks the `year` and `month` you pass in. A month with no entries, or a nonsense month like 13, just gives `count` 0, `total` 0 and an empty `by_category`. There's no error. That `total` is the integer `0`, not `0.0`, because `sum()` of nothing returns 0.\n\n**What happens when a date string is invalid**\n\nDates are only parsed when an entry is added. `add_entry` calls `parse_date` (`ledger/core.py:15`), and stored entries hold real `date` objects, so `monthly_report` never sees a date string. `parse_date` rejects bad dates in two stages:\n\n1. **Shape check:** it splits the text on `-`. If it doesn't get exactly three parts, or any part isn't all digits, it raises `ValueError(\"bad date: '...'\")` with the original text in the message. This catches things like `\"2024/01/05\"`, `\"2024-01\"`, `\"2024-01-05T10:00\"` and `\"-2024-01-05\"`, which gives an empty first part.\n2. **Calendar check:** the three numbers go to `datetime.date(y, m, d)`. If they aren't a real date, such as `\"2024-02-30\"` or `\"2024-13-01\"`, Python's `date` raises its own `ValueError` (for example \"day is out of range for month\"). This message doesn't include the bad input.\n\nSome consequences:\n\n- **The entry isn't added when the date is bad.** The date is parsed while the entry is being built, before it's appended, so the list stays unchanged and the error goes back to the caller. Nothing catches it inside the module.\n- **The amount is checked before the date.** If a call has both a non-positive amount and a bad date, you get the \"amount must be positive\" error.\n- **The format is looser than the docstring says.** The docstring says YYYY-MM-DD, but the code doesn't require zero-padding or a set number of digits, so `\"2024-1-5\"` is accepted as 5 January 2024.\n- **Not every failure is a `ValueError`.** Some Unicode characters count as digits to `isdigit()` but `int()` can't convert them, and those still end up as a `ValueError`. But passing something that isn't a string, like `None` or a `date` object, fails at `.split` with an `AttributeError`.",
   "account_remark": false,
   "replaced_by": 13,
   "api_cost": 0.177794
  },
  {
   "run": "claude-opus-5-5__placebo__T3__r2__a7b22d",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "placebo",
   "rep": 2,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1265,
   "input": 4,
   "cache_read": 37165,
   "cache_write": 17733,
   "warmup": null,
   "cache_write_1h": 17733,
   "cost_usd": 0.174613,
   "wall_s": 17.0,
   "turns": 2,
   "prompt_last": 27853,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 231,
   "commands": null,
   "answer": "## How monthly totals are computed\n\n`monthly_report(entries, year, month)` (`ledger/core.py:51`) builds the monthly figures in two steps:\n\n1. **Pick the month's entries.** `entries_in_month` keeps every entry whose `day.year` and `day.month` match the arguments. It filters on stored `date` objects, so it never parses strings. It doesn't check `year` or `month` either: asking for month 13 just returns an empty list, not an error.\n\n2. **Add up that list** and return a dict with three keys:\n   - `\"count\"`: the number of entries in the month.\n   - `\"total\"`: the sum of their `amount` values, from `total()`. For a month with no entries this is the integer `0`, not `0.0`, because `sum()` of an empty sequence starts at `0`.\n   - `\"by_category\"`: a dict of category \u2192 summed amount, from `sum_by_category`. Each category's sum starts at `0.0`, so these values are always floats. Categories are already lowercased when entries are added, so \"Food\" and \"food\" land in the same bucket.\n\nAmounts are Python floats, so totals can show normal floating-point rounding (for example, `0.1 + 0.2` gives `0.30000000000000004`). Nothing rounds them to cents.\n\n## What happens with an invalid date string\n\nDates are only parsed when you add an entry. `add_entry` passes the `day` string to `parse_date` (`ledger/core.py:15`), which rejects bad input in two layers:\n\n1. **Format check.** It splits the string on `-` and requires exactly three parts, all made of digits. If that fails, it raises `ValueError(\"bad date: '...'\")`. This catches things like `\"2024/01/05\"`, `\"2024-01\"`, `\"Jan 5\"`, `\" 2024-01-05\"` (because of the leading space) and `\"\"`.\n\n2. **Calendar check.** If the format is fine, the three numbers go to `datetime.date(y, m, d)`. An impossible date like `\"2024-02-30\"` or `\"2024-13-01\"` makes `date` raise its own `ValueError`, such as `\"day is out of range for month\"`. That message doesn't include the input string, so it's less helpful than the format-check message.\n\nEither way, the `ValueError` passes straight out of `add_entry`. Nothing catches it. Because the exception happens before `entries.append`, **the list is not changed**: no partial or placeholder entry is added.\n\nA few edge cases:\n- **The amount is checked first.** A call with both a non-positive amount and a bad date reports `\"amount must be positive\"`.\n- **Zero-padding isn't required.** `\"2024-1-5\"` is accepted.\n- **Non-string input gives a different error.** Passing `None`, or passing a `date` object instead of a string, raises `AttributeError` (no `.split`), not `ValueError`.\n\nI didn't change any files.",
   "account_remark": false,
   "replaced_by": 13,
   "api_cost": 0.174613
  },
  {
   "run": "claude-opus-5-5__placebo__T3__r3__b96e71",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "placebo",
   "rep": 3,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1471,
   "input": 4,
   "cache_read": 37162,
   "cache_write": 17729,
   "warmup": null,
   "cache_write_1h": 17729,
   "cost_usd": 0.17870039999999998,
   "wall_s": 17.8,
   "turns": 2,
   "prompt_last": 27849,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 414,
   "commands": null,
   "answer": "## How monthly totals are computed\n\n`monthly_report(entries, year, month)` (`ledger/core.py:51`) works in two steps.\n\n1. **Filter.** `entries_in_month` keeps each entry whose `day.year` and `day.month` match the arguments. `year` and `month` must be integers. No strings are parsed here. The month is also not checked: asking for month 13 or month 0 doesn't raise an error, it just returns an empty report.\n2. **Aggregate.** The filtered list is used three ways:\n   - `count` is the number of matching entries.\n   - `total` comes from `total()`, which adds up the `amount` fields.\n   - `by_category` comes from `sum_by_category()`, which builds a dict of category \u2192 running sum, starting each category at `0.0`.\n\nSome details worth knowing:\n\n- Amounts are stored as Python floats (`float(amount)` in `add_entry`), so totals can show floating-point artifacts. For example, 0.1 + 0.2 gives `0.30000000000000004`. Nothing rounds the result or uses `Decimal`.\n- For a month with no entries, `total` is the integer `0`, not `0.0`, because `sum()` of an empty sequence returns `0`. `by_category` is `{}`.\n- Categories are lowercased when an entry is added, so \"Food\" and \"food\" are summed together in `by_category`.\n\n## What happens with an invalid date string\n\nDate strings are only parsed when an entry is added: `add_entry` calls `parse_date` (`ledger/core.py:15`). `parse_date` splits the text on `-` and requires exactly three parts, each made up only of digits. If that check fails, it raises `ValueError(\"bad date: '...'\")`. So `\"2024/01/05\"`, `\"2024-01\"`, `\" 2024-01-05\"` (with a leading space) and `\"2024--05\"` are all rejected with that message.\n\nSome invalid dates get past that first check and fail somewhere else:\n\n- **Correct shape, impossible values**, like `\"2024-13-01\"` or `\"2023-02-30\"`. These pass the digit check, and then `datetime.date(...)` raises its own `ValueError`, such as \"month must be in 1..12\". It's still a `ValueError`, but the message doesn't mention \"bad date\" or show the input.\n- **Unpadded dates are accepted.** `\"2024-1-5\"` parses fine, even though the docstring says the format is YYYY-MM-DD.\n- **Input that isn't a string**, like `None` or a `date` object, raises `AttributeError` from `.split`, not `ValueError`.\n\nIn every failing case the ledger stays unchanged. `parse_date` runs while the `Entry` is being built, before `append`, so a failure means nothing is added. The exception goes straight to the caller; nothing catches or logs it. Also, the amount is checked before the date, so an entry with both a non-positive amount and a bad date reports \"amount must be positive\".",
   "account_remark": false,
   "replaced_by": 13,
   "api_cost": 0.1787
  },
  {
   "run": "claude-opus-5-5__ponytail__T3__r1__14ecee",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "ponytail",
   "rep": 1,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1429,
   "input": 4,
   "cache_read": 37438,
   "cache_write": 18006,
   "warmup": null,
   "cache_write_1h": 18006,
   "cost_usd": 0.18013159999999998,
   "wall_s": 17.2,
   "turns": 2,
   "prompt_last": 28126,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 276,
   "commands": null,
   "answer": "I read the file and changed nothing.\n\n## How monthly totals are computed\n\nMonthly totals come from `monthly_report(entries, year, month)` (`ledger/core.py:51`), which works in two steps.\n\n1. **Pick the month's entries.** `entries_in_month` (`ledger/core.py:40`) keeps every entry whose `day.year` equals `year` and whose `day.month` equals `month`. Each entry's `day` is already a `datetime.date`, so it just compares numbers and never re-reads a string. The `year` and `month` arguments aren't checked, so `month=13` or a year with no entries doesn't raise an error. It just returns an empty report: count 0, total 0, and an empty category dict.\n\n2. **Add them up.** The report is a dict with three keys:\n   - `count`: how many entries matched.\n   - `total`: `total()` (`ledger/core.py:31`) adds up the `amount` fields with the built-in `sum`, so an empty month gives `0`.\n   - `by_category`: `sum_by_category` (`ledger/core.py:44`) builds a dict from category to running total, starting each category at `0.0`. Categories are lowercased in `add_entry`, so \"Food\" and \"food\" are added together.\n\nAmounts are plain `float`s. Totals can therefore show normal rounding error: three entries of 0.10 add up to `0.30000000000000004`, not exactly 0.30.\n\n## What happens with an invalid date string\n\nDate strings are only parsed once, when an entry is added. `add_entry` calls `parse_date` (`ledger/core.py:15`):\n\n- **Wrong shape:** the string is split on `-`. If there aren't exactly three parts, or any part isn't all digits, it raises `ValueError(\"bad date: '...'\")`. This rejects `\"2024/01/05\"`, `\"2024-01\"`, `\"2024-01-05 \"` (trailing space), `\"\"`, and anything with a minus sign.\n- **Right shape, impossible date:** strings like `\"2024-02-30\"` or `\"2024-13-01\"` pass the shape check. `datetime.date()` then raises its own `ValueError`, such as \"day is out of range for month\". That message is different from the \"bad date\" one, but it's the same exception type.\n- **No data is lost:** `add_entry` checks the amount first, then builds the `Entry`, which parses the date, before it calls `append`. If the date is bad, the exception is raised before anything is added, so the list stays unchanged. The error goes straight to the caller, since nothing catches it.\n\nSome quirks:\n- **No zero-padding required:** `\"2024-1-5\"` is accepted, so the \"YYYY-MM-DD\" in the docstring isn't strictly enforced.\n- **Non-string input:** passing something other than a string, like `None` or a `date`, raises `AttributeError` (no `.split`), not `ValueError`.\n- **Unusual Unicode digits:** `str.isdigit()` accepts characters like `\"\u00b2\"`, which `int()` then rejects. That case still ends in a `ValueError`, just from a different line.\n\nBad dates never reach the monthly calculation. Every stored entry already has a valid `date`, so `monthly_report` can't fail because of a date.",
   "account_remark": false,
   "replaced_by": 13,
   "api_cost": 0.180132
  },
  {
   "run": "claude-opus-5-5__ponytail__T3__r2__370cc0",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "ponytail",
   "rep": 2,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1510,
   "input": 4,
   "cache_read": 37438,
   "cache_write": 18006,
   "warmup": null,
   "cache_write_1h": 18006,
   "cost_usd": 0.18175159999999999,
   "wall_s": 15.1,
   "turns": 2,
   "prompt_last": 28126,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 368,
   "commands": null,
   "answer": "Here's how `ledger/core.py` works. I didn't change any files.\n\n## How monthly totals are computed\n\n`monthly_report(entries, year, month)` works in three steps:\n\n1. **Filter.** `entries_in_month` keeps only the entries whose `day.year` and `day.month` match the arguments, in one pass over the list. `year` and `month` are never checked, so `month=13` returns an empty report rather than an error.\n2. **Total.** `total` adds up the `amount` of every matching entry with a plain `sum()`. Amounts are Python floats, so totals can pick up small rounding errors (for example, 0.1 + 0.2 gives 0.30000000000000004). Money isn't stored as `Decimal` or as whole cents.\n3. **Break down by category.** `sum_by_category` builds a dictionary that maps each category to the sum of its amounts, starting each category at `0.0`. Category names are lowercased when an entry is added (`add_entry`), so \"Food\" and \"food\" end up in the same bucket.\n\nThe result is `{\"count\": ..., \"total\": ..., \"by_category\": {...}}`. For a month with no entries, `total` is the integer `0` rather than `0.0`, because `sum()` of nothing returns `0`. `count` is `0` and `by_category` is `{}`.\n\n## What happens with an invalid date string\n\nDates are only parsed in `add_entry`, which calls `parse_date(day)`. `parse_date` rejects input in two ways:\n\n- **Shape check.** It splits the text on `-`. It raises `ValueError(\"bad date: '...'\")` unless there are exactly three parts and every part is all digits. So `\"2024/01/05\"`, `\"2024-01\"`, `\" 2024-01-05\"` (leading space) and `\"2024-01-05T00:00\"` are all rejected here.\n- **Calendar check.** If the shape is fine, the three parts become integers and go to `datetime.date(y, m, d)`. That rejects dates that don't exist, like `\"2024-02-30\"`, `\"2024-13-01\"` or year `0000`, with its own `ValueError` (for example \"day is out of range for month\").\n\nWhen `parse_date` raises, the error passes straight through `add_entry`. The entry is built only after the date parses, so **the entries list is left unchanged**. Nothing is added and nothing is caught or logged.\n\nSome edge cases to know about:\n\n- **Looser than the docstring says.** It accepts dates without leading zeros, like `\"2024-1-5\"`, even though the docstring says the format is strictly YYYY-MM-DD.\n- **Non-ASCII digits.** `str.isdigit()` counts full-width digits such as `\uff12\uff10\uff12\uff14` as digits, and `int()` converts them, so those are accepted. Characters like `\u00b2` pass `isdigit()`, but `int()` rejects them, which still gives a `ValueError`.\n- **Non-string input.** Passing something like `None` or a `date` object raises `AttributeError` (it has no `.split`), not `ValueError`. Code that only catches `ValueError` would miss it.\n- **Check order.** `add_entry` checks the amount first. A call with both a bad amount and a bad date reports the amount error.",
   "account_remark": false,
   "replaced_by": 13,
   "api_cost": 0.181752
  },
  {
   "run": "claude-opus-5-5__ponytail__T3__r3__5c70e9",
   "model": "claude-opus-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "ponytail",
   "rep": 3,
   "round": 7,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 1217,
   "input": 4,
   "cache_read": 37471,
   "cache_write": 18040,
   "warmup": null,
   "cache_write_1h": 18040,
   "cost_usd": 0.1761702,
   "wall_s": 15.2,
   "turns": 2,
   "prompt_last": 28160,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 246,
   "commands": null,
   "answer": "## How monthly totals are computed\n\n`monthly_report(entries, year, month)` (`ledger/core.py:51`) works in three steps:\n\n1. **Filter.** `entries_in_month` keeps the entries whose `day.year` and `day.month` match the arguments. It compares against real `date` objects, so the date text isn't involved at this point. `year` and `month` aren't checked, so `month=13` doesn't raise an error; it just matches nothing.\n2. **Total.** `total` adds up `e.amount` over the filtered entries. Amounts are stored as `float`, so totals can pick up small rounding errors (for example, `0.1 + 0.2` gives `0.30000000000000004`). If no entries match, `sum` returns the integer `0`, not `0.0`.\n3. **Per category.** `sum_by_category` builds a dict of category \u2192 running float sum. Categories were lowercased when the entries were added, so \"Food\" and \"food\" count as one category. If nothing matches, the result is `{}`.\n\nThe report it returns looks like `{\"count\": n, \"total\": x, \"by_category\": {...}}`.\n\n## What happens with an invalid date string\n\nDate strings are only parsed when an entry is added: `add_entry` calls `parse_date` (`ledger/core.py:15`).\n\n- `parse_date` splits the text on `-`. If there aren't exactly three parts, or any part isn't all digits, it raises `ValueError(\"bad date: '...'\")`. Strings like `\"2024/01/05\"`, `\"2024-01\"`, `\" 2024-01-05\"` (leading space) and `\"-2024-01-05\"` are all rejected this way.\n- If the format passes but the date doesn't exist, such as `\"2024-02-30\"` or `\"2024-13-01\"`, the `date(y, m, d)` constructor raises the `ValueError` instead. The message then comes from Python (for example, \"day is out of range for month\") rather than \"bad date\".\n- The check is looser than the docstring's \"YYYY-MM-DD\" suggests. `\"2024-1-5\"` is accepted, and so is `\"99-01-01\"` (read as the year 99).\n- If you pass something that isn't a string, like `None`, you get an `AttributeError` from `.split`, not a `ValueError`.\n\n**Effect on the ledger:** the date is parsed inside the `Entry(...)` call, before `append`. When parsing fails, the exception passes straight up to the caller and the list stays unchanged, so a bad date can never end up in a monthly report. Nothing catches the error inside the module. `add_entry` checks the amount before the date, so if both are bad you'll see the \"amount must be positive\" error first.\n\nI didn't change any files.",
   "account_remark": false,
   "replaced_by": 13,
   "api_cost": 0.17617
  },
  {
   "run": "claude-sonnet-5__caveman__T3__r1__0ee7d2",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "caveman",
   "rep": 1,
   "round": 1,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 294,
   "input": 4,
   "cache_read": 59226,
   "cache_write": 22769,
   "warmup": null,
   "cache_write_1h": 22769,
   "cost_usd": 0.1058692,
   "wall_s": 7.0,
   "turns": 2,
   "prompt_last": 41410,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": null,
   "thinking": 0,
   "commands": null,
   "answer": "Monthly totals: `monthly_report` filter entries by year/month via `entries_in_month` (`ledger/core.py:40-41`), then `total()` sum `amount` field over those (`ledger/core.py:31-32`). Also return count and per-category breakdown via `sum_by_category`.\n\nInvalid date string: `parse_date` raise `ValueError(f\"bad date: {text!r}\")` (`ledger/core.py:19`) when split on `-` not give 3 numeric parts. Called from `add_entry`, so entry with bad date string raise at insert time, before entry ever get added or counted in totals.",
   "account_remark": false,
   "replaced_by": 13,
   "api_cost": 0.105869
  },
  {
   "run": "claude-sonnet-5__caveman__T3__r2__c7912e",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "caveman",
   "rep": 2,
   "round": 2,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 375,
   "input": 6,
   "cache_read": 100019,
   "cache_write": 22981,
   "warmup": null,
   "cache_write_1h": 22981,
   "cost_usd": 0.1156898,
   "wall_s": 12.3,
   "turns": 3,
   "prompt_last": 41622,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": null,
   "thinking": 6,
   "commands": null,
   "answer": "Monthly total: `monthly_report` filter entries by year+month via `entries_in_month`, then sum amount field with `total()`. Also break down by category with `sum_by_category`.\n\nInvalid date string: `parse_date` split on \"-\", need exactly 3 parts, all digit. Fail that check, raise `ValueError(f\"bad date: {text!r}\")`. Called from `add_entry`, so bad date reject entry before add, error propagate to caller.",
   "account_remark": false,
   "replaced_by": 13,
   "api_cost": 0.11569
  },
  {
   "run": "claude-sonnet-5__caveman__T3__r3__e0c0f0",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "caveman",
   "rep": 3,
   "round": 2,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 289,
   "input": 4,
   "cache_read": 59223,
   "cache_write": 22779,
   "warmup": null,
   "cache_write_1h": 22779,
   "cost_usd": 0.1058586,
   "wall_s": 5.9,
   "turns": 2,
   "prompt_last": 41420,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": null,
   "thinking": 12,
   "commands": null,
   "answer": "Monthly totals: `monthly_report` call `entries_in_month` filter entries by `year`/`month` on `Entry.day`. Then `total` sum `amount` over filtered list. `sum_by_category` also run on filtered list, group amount by category.\n\nInvalid date: `parse_date` split text on \"-\". Need exactly 3 parts, all digit. Fail either check \u2192 raise `ValueError(f\"bad date: {text!r}\")`. `add_entry` call `parse_date` before append, so bad date entry never enter ledger, error propagate up to caller.",
   "account_remark": false,
   "replaced_by": 13,
   "api_cost": 0.105859
  },
  {
   "run": "claude-sonnet-5__control__T3__r1__6b1bcd",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "control",
   "rep": 1,
   "round": 1,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 985,
   "input": 6,
   "cache_read": 94669,
   "cache_write": 20272,
   "warmup": null,
   "cache_write_1h": 20272,
   "cost_usd": 0.1098838,
   "wall_s": 14.4,
   "turns": 3,
   "prompt_last": 38913,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": null,
   "thinking": 16,
   "commands": null,
   "answer": "## How monthly totals are computed\n\n`monthly_report(entries, year, month)` (ledger/core.py:51-53) does it in three steps:\n\n1. `entries_in_month` (line 40) filters the full entry list down to only those whose `day.year` and `day.month` match the requested year/month.\n2. `total` (line 31-32) sums the `amount` field of that filtered subset \u2014 a plain `sum()` over floats.\n3. `sum_by_category` (line 44-48) builds a per-category breakdown of the same filtered subset, accumulating into a dict keyed by lowercased category name.\n\nThe result is `{\"count\": ..., \"total\": ..., \"by_category\": ...}` \u2014 count and total for the month, plus a category-level breakdown, all computed only from entries already parsed into `Entry` objects (with real `date` values) that were previously added.\n\n## What happens with an invalid date string\n\nDate validity is only checked at insertion time, in `parse_date` (lines 15-21), which is called from `add_entry` (line 27) \u2014 `monthly_report` and its helpers never re-validate dates, they just trust the `date` objects already stored on each `Entry`.\n\n`parse_date` splits the string on `-` and requires exactly 3 parts that are all-digit substrings (`str.isdigit()`). If that check fails \u2014 wrong number of segments, non-numeric characters, empty parts, etc. \u2014 it raises `ValueError(f\"bad date: {text!r}\")` immediately, and `add_entry` never appends the entry.\n\nNote this validation is fairly shallow: it doesn't catch every \"invalid\" case itself \u2014 e.g. `\"2024-13-40\"` passes the digit/segment check and only fails later when `date(y, m, d)` is constructed, raising Python's own `ValueError: month must be in 1..12` (or similar) instead of the custom \"bad date\" message. Also, a leading `-` (negative-looking input, e.g. `\"-2024-01-01\"`) would split into more than 3 parts and correctly fail, but formats like `\"2024-1-1\"` (no zero-padding) pass fine since `isdigit()` doesn't require fixed width.\n\nEither way, whether it's the custom `ValueError` or the one raised by `date()`, the exception propagates up through `add_entry` uncaught \u2014 there's no try/except anywhere in this module \u2014 so the entry is never added to the list, and it simply can't appear in any later `total`/`monthly_report` calculation.",
   "account_remark": false,
   "replaced_by": 13,
   "api_cost": 0.109884
  },
  {
   "run": "claude-sonnet-5__control__T3__r2__1c8bc8",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "control",
   "rep": 2,
   "round": 2,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 839,
   "input": 6,
   "cache_read": 94702,
   "cache_write": 20305,
   "warmup": null,
   "cache_write_1h": 20305,
   "cost_usd": 0.1085624,
   "wall_s": 11.3,
   "turns": 3,
   "prompt_last": 38946,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": null,
   "thinking": 37,
   "commands": null,
   "answer": "Here's how it works:\n\n**Monthly totals.** `monthly_report(entries, year, month)` first calls `entries_in_month`, which filters the full entry list down to those whose `day.year` and `day.month` match the requested year/month (core.py:40-41). That filtered list is then passed to `total()`, which simply sums the `amount` field across those entries (core.py:31-32). The report also computes `by_category` via `sum_by_category`, which builds a dict accumulating amounts per (lowercased) category for the same filtered month entries (core.py:44-48). So a monthly total is just \"sum of amounts for every entry whose date falls in that year/month\" \u2014 no rounding, no handling of entries with missing/None dates since `day` is always a real `date` object by the time it's stored.\n\n**Invalid date strings.** Dates only enter the system through `add_entry`, which calls `parse_date(day)` (core.py:27). `parse_date` splits the string on `-` and requires exactly 3 parts that are all digit-only (core.py:17-19); if that check fails, it raises `ValueError(f\"bad date: {text!r}\")` immediately \u2014 no entry is created and nothing is appended to `entries`. If the string passes that structural check but represents an impossible calendar date (e.g. `\"2024-13-40\"`), `parse_date` still calls `date(y, m, d)` (core.py:21), and Python's `date` constructor itself raises `ValueError` (e.g. \"month must be in 1..12\"), which propagates uncaught out of `add_entry`. Either way, an invalid date string means the entry is rejected outright via an uncaught `ValueError` rather than being silently skipped or stored with a bad date \u2014 so it can never affect `monthly_report`'s totals, since only successfully parsed entries ever make it into the `entries` list.",
   "account_remark": false,
   "replaced_by": 13,
   "api_cost": 0.108562
  },
  {
   "run": "claude-sonnet-5__control__T3__r3__853149",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "control",
   "rep": 3,
   "round": 2,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 891,
   "input": 6,
   "cache_read": 94672,
   "cache_write": 20280,
   "warmup": null,
   "cache_write_1h": 20280,
   "cost_usd": 0.10897640000000001,
   "wall_s": 11.0,
   "turns": 3,
   "prompt_last": 38921,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": null,
   "thinking": 16,
   "commands": null,
   "answer": "## How monthly totals are computed\n\n`monthly_report(entries, year, month)` (core.py:51-53) does this in three steps:\n\n1. **Filter to the month** \u2014 `entries_in_month` (core.py:40-41) keeps only entries whose `day.year` and `day.month` match the requested year/month.\n2. **Sum the amounts** \u2014 `total()` (core.py:31-32) sums the `amount` field across those filtered entries to produce `\"total\"`.\n3. **Break down by category** \u2014 `sum_by_category()` (core.py:44-48) builds a dict accumulating `amount` per `category` for the same filtered entries, giving `\"by_category\"`.\n\nThe report also includes `\"count\"`, the number of entries in that month. Note that entries only get their `day` as a `date` object at creation time via `add_entry`, which calls `parse_date` \u2014 so by the time an entry exists in the list, its date is already a valid `date`.\n\n## What happens with an invalid date string\n\nValidation happens at insertion, not at report time. `parse_date` (core.py:15-21) is called from `add_entry` (core.py:27) when an entry is created:\n\n- It splits the string on `-` and requires exactly 3 parts, all of which are purely digit strings (`str.isdigit()`).\n- If that check fails (wrong number of parts, non-digit characters, e.g. `\"2024/01/01\"`, `\"abc\"`, or an empty string), it raises `ValueError(f\"bad date: {text!r}\")` immediately \u2014 `add_entry` does not catch this, so the exception propagates to the caller and the entry is never added to the list.\n- If the split/digit check passes but the numbers don't form a real calendar date (e.g. `\"2024-02-30\"` or `\"2024-13-01\"`), `date(y, m, d)` itself raises a `ValueError` (e.g. \"day is out of range for month\"), which also propagates uncaught.\n\nEither way, an invalid date string means `add_entry` raises before appending anything, so `monthly_report` never sees a malformed entry \u2014 it only ever operates on entries whose dates were already validated at creation time.",
   "account_remark": false,
   "replaced_by": 13,
   "api_cost": 0.108976
  },
  {
   "run": "claude-sonnet-5__karpathy__T3__r1__2d3e7b",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "karpathy",
   "rep": 1,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 877,
   "input": 4,
   "cache_read": 57587,
   "cache_write": 21146,
   "warmup": null,
   "cache_write_1h": 21146,
   "cost_usd": 0.1048794,
   "wall_s": 10.8,
   "turns": 2,
   "prompt_last": 39787,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 53,
   "commands": null,
   "answer": "## How monthly totals are computed\n\n`monthly_report(entries, year, month)` (line 51) does this in three steps:\n\n1. **Filter to the month** \u2014 `entries_in_month` (line 40) keeps only entries whose `day.year` and `day.month` match the requested year/month.\n2. **Sum the total** \u2014 `total()` (line 31) sums the `amount` field across those filtered entries via `sum(e.amount for e in entries)`.\n3. **Break down by category** \u2014 `sum_by_category()` (line 44) builds a dict accumulating amounts per `e.category` for the same filtered entries.\n\nThe report returns `{\"count\": ..., \"total\": ..., \"by_category\": ...}` \u2014 the count and total of matching entries, plus a per-category subtotal.\n\nNote that each `Entry.day` is a `date` object, not the original string \u2014 so date parsing/validation already happened when the entry was created via `add_entry`, not at report time.\n\n## What happens with an invalid date string\n\nValidation happens in `parse_date()` (line 15), called from `add_entry()` (line 27) when an entry is first added:\n\n- It splits the string on `\"-\"`.\n- If there aren't exactly 3 parts, or any part isn't all digits (`str.isdigit()`), it raises `ValueError(f\"bad date: {text!r}\")` immediately (line 19).\n- If the split/digit check passes but the values are semantically invalid (e.g. month `13`, day `32`, or day `\"31\"` for a 30-day month), `date(y, m, d)` (line 21) itself raises a `ValueError` from the stdlib (e.g. `\"month must be in 1..12\"`).\n\nEither way, an invalid date string means `add_entry` never appends the entry \u2014 the `ValueError` propagates up to the caller uncaught, so no bad entry can end up in the ledger to affect a later monthly total. There's no try/except anywhere in this file, so it's entirely on the caller to handle that exception.\n\nNote: `isdigit()` doesn't reject a leading `-` sign the way you might expect (it just fails the split-into-3-parts differently), and it also accepts non-ASCII digit characters \u2014 edge cases worth knowing about but not relevant to normal `YYYY-MM-DD` input.",
   "account_remark": false,
   "replaced_by": 13,
   "api_cost": 0.104879
  },
  {
   "run": "claude-sonnet-5__karpathy__T3__r2__ac2f83",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "karpathy",
   "rep": 2,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 635,
   "input": 6,
   "cache_read": 96763,
   "cache_write": 21360,
   "warmup": null,
   "cache_write_1h": 21360,
   "cost_usd": 0.11115459999999999,
   "wall_s": 8.5,
   "turns": 3,
   "prompt_last": 40001,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 16,
   "commands": null,
   "answer": "**Monthly totals:** `monthly_report(entries, year, month)` first calls `entries_in_month`, which filters `entries` down to those whose `day.year` and `day.month` match the given year/month. It then calls `total()` on that filtered list, which just sums the `amount` field of each matching `Entry`. The report also includes `count` (number of matching entries) and `by_category` (a dict summing amounts per category, via `sum_by_category`), all computed over that same filtered subset.\n\n**Invalid date strings:** Dates only get parsed when an entry is created, via `add_entry` \u2192 `parse_date`. `parse_date` splits the input on `-` and requires exactly 3 parts that are all digit-only strings; otherwise it raises `ValueError(f\"bad date: {text!r}\")`. If the parts pass that check but form an impossible date (e.g. month 13, Feb 30), the subsequent `date(y, m, d)` call raises its own `ValueError` from the standard library. Either way, `add_entry` doesn't catch the exception, so it propagates up to the caller and the entry is never appended to the list \u2014 meaning invalid dates never reach `monthly_report`'s calculations; they fail at insertion time instead.",
   "account_remark": false,
   "replaced_by": 13,
   "api_cost": 0.111155
  },
  {
   "run": "claude-sonnet-5__karpathy__T3__r3__21fd23",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "karpathy",
   "rep": 3,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 911,
   "input": 6,
   "cache_read": 96757,
   "cache_write": 21355,
   "warmup": null,
   "cache_write_1h": 21355,
   "cost_usd": 0.1138934,
   "wall_s": 11.0,
   "turns": 3,
   "prompt_last": 39996,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 16,
   "commands": null,
   "answer": "## How monthly totals are computed\n\n`monthly_report(entries, year, month)` (core.py:51-53) is the entry point:\n\n1. `entries_in_month` (core.py:40-41) filters the full entry list down to those whose `day.year` and `day.month` match the requested year/month.\n2. `total` (core.py:31-32) sums the `amount` field across that filtered list to produce the `\"total\"` value.\n3. `sum_by_category` (core.py:44-48) separately builds a dict of category \u2192 summed amount for the same filtered entries, returned as `\"by_category\"`.\n4. The report also includes `\"count\"`, the number of matching entries.\n\nSo the monthly total is just `sum(e.amount for e in entries if e.day.year == year and e.day.month == month)`, computed via those two helper passes.\n\n## What happens with an invalid date string\n\nInvalid dates never make it into the ledger in the first place \u2014 validation happens at insertion, not at reporting time:\n\n- `parse_date` (core.py:15-21) splits the string on `-` and requires exactly 3 numeric parts. If that check fails, it raises `ValueError(f\"bad date: {text!r}\")` immediately.\n- Even if the string has the right shape (3 numeric parts), `date(y, m, d)` (core.py:21) will itself raise `ValueError` for an out-of-range value (e.g. month 13, day 31 of February) \u2014 Python's `date` constructor validates the components.\n- `add_entry` (core.py:24-28) calls `parse_date(day)` before constructing the `Entry`, so a bad date string prevents the entry from being added at all \u2014 the exception propagates up to whoever called `add_entry`, and no entry is appended to `entries`.\n\nThere's no try/except anywhere in this file, so the `ValueError` is uncaught here \u2014 it's the caller's responsibility to handle it (or let it crash). Since bad dates are rejected at `add_entry` time, `monthly_report` and its helpers never have to deal with malformed dates \u2014 every `Entry.day` in the list is guaranteed to be a valid `date` object.",
   "account_remark": false,
   "replaced_by": 13,
   "api_cost": 0.113893
  },
  {
   "run": "claude-sonnet-5__placebo__T3__r1__3ed580",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "placebo",
   "rep": 1,
   "round": 4,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 956,
   "input": 6,
   "cache_read": 96229,
   "cache_write": 21083,
   "warmup": null,
   "cache_write_1h": 21083,
   "cost_usd": 0.11314980000000001,
   "wall_s": 12.8,
   "turns": 3,
   "prompt_last": 39724,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 20,
   "commands": null,
   "answer": "Here's how it works:\n\n**Monthly totals** are computed by `monthly_report(entries, year, month)` (line 51-53):\n1. It calls `entries_in_month`, which filters the full entry list down to just those whose `day.year` and `day.month` match the requested year/month (line 40-41).\n2. It then calls `total()` on that filtered subset, which is just `sum(e.amount for e in entries)` \u2014 a plain sum of the `amount` field across the matching entries (line 31-32).\n3. It also computes `count` (number of matching entries) and `by_category` (via `sum_by_category`, which buckets amounts into a dict keyed by category, line 44-48).\n\nThe result is a dict: `{\"count\": ..., \"total\": ..., \"by_category\": {...}}`.\n\nNote that `total()`/`entries_in_month()` operate on already-constructed `Entry` objects, whose `day` is a real `date` object \u2014 the date string parsing has already happened by the time an entry exists.\n\n**Invalid date strings** are handled at entry-creation time, not at report time. `parse_date(text)` (line 15-21) splits the string on `-` and requires exactly 3 parts that are all-digit substrings (line 18). If that check fails \u2014 e.g. wrong number of parts, non-numeric parts, or a format like `\"2024/01/01\"` \u2014 it raises `ValueError(f\"bad date: {text!r}\")` immediately (line 19).\n\nIf the split/digit check passes but the actual values are out of range (e.g. `\"2024-13-40\"`), `parse_date` doesn't catch that itself \u2014 it passes `y, m, d` straight to `date(y, m, d)` (line 21), and the `date` constructor raises its own `ValueError` (e.g. `\"month must be in 1..12\"`) instead of the custom \"bad date\" message.\n\nSince `add_entry` calls `parse_date(day)` before appending to the ledger (line 27), an entry with an invalid date string never gets added \u2014 the `ValueError` propagates up to the caller, and `entries` is left unmodified. So by the time `monthly_report` runs, every entry's `day` is guaranteed valid; invalid dates simply never make it into the ledger in the first place.",
   "account_remark": false,
   "replaced_by": 13,
   "api_cost": 0.11315
  },
  {
   "run": "claude-sonnet-5__placebo__T3__r2__d6b3ca",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "placebo",
   "rep": 2,
   "round": 4,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 731,
   "input": 6,
   "cache_read": 96244,
   "cache_write": 21092,
   "warmup": null,
   "cache_write_1h": 21092,
   "cost_usd": 0.1109388,
   "wall_s": 9.7,
   "turns": 3,
   "prompt_last": 39733,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "Here's how it works:\n\n**Monthly totals:** `monthly_report(entries, year, month)` first calls `entries_in_month`, which filters the full entry list down to those whose `day.year` and `day.month` match the requested year/month. That filtered list is then passed to `total`, which simply sums the `amount` field across those entries. The same filtered list also feeds `sum_by_category`, which builds a dict of category \u2192 summed amount. The report returns a dict with `count` (number of matching entries), `total` (the sum), and `by_category` (the per-category breakdown).\n\n**Invalid date strings:** Date parsing happens in `parse_date`, called from `add_entry` when a new entry is created (it's not used at query time \u2014 by then entries already hold real `date` objects). `parse_date` splits the input on `-` and requires exactly 3 parts, all of which must be purely digit strings (`str.isdigit()`). If that check fails \u2014 e.g. wrong number of parts, non-numeric parts, or empty parts \u2014 it raises `ValueError(f\"bad date: {text!r}\")` immediately, and no entry is added. If the split parts are digit strings but form an impossible date (e.g. `\"2024-02-30\"` or `\"2024-13-01\"`), `isdigit()` passes but `date(y, m, d)` itself raises `ValueError` (from the standard library, with its own message like \"day is out of range for month\"), which propagates up uncaught. So invalid dates always surface as a `ValueError`, just with two different message sources depending on whether the format or the actual calendar value is what's wrong.",
   "account_remark": false,
   "replaced_by": 13,
   "api_cost": 0.110939
  },
  {
   "run": "claude-sonnet-5__placebo__T3__r3__9109ac",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "placebo",
   "rep": 3,
   "round": 4,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 835,
   "input": 6,
   "cache_read": 96232,
   "cache_write": 21084,
   "warmup": null,
   "cache_write_1h": 21084,
   "cost_usd": 0.11194439999999999,
   "wall_s": 10.6,
   "turns": 3,
   "prompt_last": 39725,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "## How monthly totals are computed\n\n`monthly_report(entries, year, month)` (core.py:51-53) is the entry point. It first calls `entries_in_month`, which filters the full entry list down to those whose `day.year` and `day.month` match the requested year/month (core.py:40-41). That filtered list is then run through:\n\n- `total()` \u2014 sums `e.amount` over the filtered entries (core.py:31-32)\n- `sum_by_category()` \u2014 builds a dict accumulating `e.amount` per `e.category` (core.py:44-48)\n\nThe report returned is `{\"count\": ..., \"total\": ..., \"by_category\": ...}`.\n\nNote that this all operates on already-constructed `Entry` objects, whose `day` field is a real `datetime.date`. So by the time an entry reaches `monthly_report`, its date has already been validated \u2014 the month-total logic itself does no date parsing or validation.\n\n## What happens when a date string is invalid\n\nDate strings are only ever parsed at entry-creation time, in `add_entry` \u2192 `parse_date` (core.py:15-21, called at line 27):\n\n- `parse_date` splits the text on `-` and requires exactly 3 parts, each all-digit (`str.isdigit()`). If that check fails, it raises `ValueError(f\"bad date: {text!r}\")` \u2014 a clear, explicit error naming the bad input.\n- If the split/digit check passes but the numbers are out of range (e.g. month 13, day 31 in February), `date(y, m, d)` itself raises a `ValueError` (from the standard library), just with the stdlib's own message rather than the custom one.\n\nEither way, `add_entry` never appends a bad entry \u2014 the exception propagates up and the entry is never added to the list. So invalid dates can never reach `monthly_report`/`entries_in_month`; there's no silent fallback or skipping, just a raised `ValueError` at insertion time.",
   "account_remark": false,
   "replaced_by": 13,
   "api_cost": 0.111944
  },
  {
   "run": "claude-sonnet-5__ponytail__T3__r1__e8de45",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "ponytail",
   "rep": 1,
   "round": 1,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 537,
   "input": 4,
   "cache_read": 57603,
   "cache_write": 21145,
   "warmup": null,
   "cache_write_1h": 21145,
   "cost_usd": 0.1014786,
   "wall_s": 12.9,
   "turns": 2,
   "prompt_last": 39786,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": null,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals:** `monthly_report(entries, year, month)` filters `entries` down to those whose `day.year`/`day.month` match via `entries_in_month`, then computes three things from that subset: `count` (number of matching entries), `total` (a plain sum of `e.amount` across them via `total()`), and `by_category` (a dict built by `sum_by_category` that accumulates `amount` per lowercased `category` string, defaulting missing keys to `0.0`).\n\n**Invalid date strings:** Dates only get validated at entry-creation time, in `parse_date`, which is called from `add_entry`. It splits the string on `-` and requires exactly 3 numeric parts; if the format doesn't match (wrong number of parts, non-digit characters) it raises `ValueError(f\"bad date: {text!r}\")` immediately \u2014 `add_entry` doesn't catch it, so the error propagates to the caller and the entry is never appended. Note that even if the split/isdigit check passes, an out-of-range value (e.g. month `13`, day `32`) will still raise a `ValueError` from the `date(y, m, d)` constructor itself, uncaught. Since dates are stored as `date` objects on each `Entry` once created, `monthly_report` and everything downstream never has to deal with invalid dates \u2014 bad input is rejected at construction, not during reporting.",
   "account_remark": false,
   "replaced_by": 13,
   "api_cost": 0.101479
  },
  {
   "run": "claude-sonnet-5__ponytail__T3__r2__c90aa8",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "ponytail",
   "rep": 2,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 737,
   "input": 4,
   "cache_read": 47468,
   "cache_write": 12250,
   "warmup": null,
   "cache_write_1h": 12250,
   "cost_usd": 0.0658716,
   "wall_s": 8.2,
   "turns": 2,
   "prompt_last": 30891,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 46,
   "commands": null,
   "answer": "## Monthly totals\n\n`monthly_report(entries, year, month)` (ledger/core.py:51-53) computes the total by first filtering entries down to the given month via `entries_in_month` (ledger/core.py:40-41), which keeps only entries whose `day.year` and `day.month` match. It then calls `total()` on that filtered list, which is just `sum(e.amount for e in entries)` (ledger/core.py:31-32) \u2014 a plain sum of the dollar amounts, no rounding or currency handling. The report also returns the count of matching entries and a per-category breakdown via `sum_by_category`, which buckets amounts into a dict keyed by (lowercased) category.\n\n## Invalid date strings\n\nDates only enter the ledger through `add_entry`, which calls `parse_date(day)` (ledger/core.py:15-21). `parse_date` splits the string on `-` and requires exactly 3 numeric parts; if that check fails it raises `ValueError(f\"bad date: {text!r}\")` immediately. If the string *does* split into 3 numeric parts but describes a nonexistent date (e.g. `\"2024-02-30\"` or `\"2024-13-01\"`), the digit check passes but `date(y, m, d)` on ledger/core.py:21 raises its own `ValueError` (from the stdlib) since `parse_date` doesn't catch or re-validate that.\n\nEither way, an invalid date string causes `add_entry` to raise `ValueError` and the entry is never appended \u2014 there's no silent failure or default date. Since `monthly_report` only ever operates on already-constructed `Entry` objects (which by construction have valid `date` instances), invalid date strings can't corrupt a monthly total; they simply prevent the entry from being added in the first place, and the exception propagates to whoever called `add_entry`.",
   "account_remark": false,
   "replaced_by": 13,
   "api_cost": 0.065872,
   "light": true
  },
  {
   "run": "claude-sonnet-5__ponytail__T3__r3__a3984b",
   "model": "claude-sonnet-5",
   "host": "claude",
   "task": "T3",
   "skill": "ponytail",
   "rep": 3,
   "round": 3,
   "date": "2026-09-22",
   "exit": 0,
   "passed": true,
   "output": 743,
   "input": 4,
   "cache_read": 57579,
   "cache_write": 21143,
   "warmup": null,
   "cache_write_1h": 21143,
   "cost_usd": 0.1035258,
   "wall_s": 16.6,
   "turns": 2,
   "prompt_last": 39784,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 21,
   "commands": null,
   "answer": "## How monthly totals are computed\n\n`monthly_report(entries, year, month)` (ledger/core.py:51) does it in three steps:\n\n1. `entries_in_month` (line 40) filters the full entry list down to those whose `day.year` and `day.month` match the requested year/month \u2014 a plain list comprehension, no date-range math.\n2. `total` (line 31) sums the `amount` field across that filtered list with `sum(...)`.\n3. `sum_by_category` (line 44) walks the same filtered list and accumulates amounts into a dict keyed by `e.category`, starting each new category at `0.0`.\n\nThe report bundles these into `{\"count\", \"total\", \"by_category\"}`. Entries only get into `entries` via `add_entry` (line 24), which stores each `day` as an already-parsed `date` object (via `parse_date`) rather than a raw string \u2014 so by the time totals run, invalid dates can't be present in the data at all.\n\n## What happens with an invalid date string\n\nThe validation happens earlier, at insertion time, not at report time:\n\n- `add_entry` calls `parse_date(day)` (line 27) before ever constructing an `Entry`.\n- `parse_date` (line 15) splits on `-` and requires exactly 3 numeric parts; anything else \u2014 wrong format, non-numeric parts, empty string \u2014 raises `ValueError(f\"bad date: {text!r}\")` immediately (line 19).\n- If the parts are numeric but out of range (e.g. month `13`, day `31` in February), `date(y, m, d)` (line 21) itself raises `ValueError` (e.g. \"month must be in 1..12\"), which propagates uncaught.\n\nEither way, `add_entry` never appends the entry \u2014 the exception bubbles straight to the caller, so a bad date string never makes it into the ledger, and `monthly_report`/`total` never have to deal with malformed dates since they only ever operate on entries that already passed `parse_date`.",
   "account_remark": false,
   "replaced_by": 13,
   "api_cost": 0.103526
  },
  {
   "run": "claude-sonnet-5-5__control__B1__r1__5184b5",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "B1",
   "skill": "control",
   "rep": 1,
   "round": 11,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 3072,
   "input": 8,
   "cache_read": 79268,
   "cache_write": 26748,
   "warmup": null,
   "cache_write_1h": 26748,
   "cost_usd": 0.1535816,
   "wall_s": 23.0,
   "turns": 5,
   "prompt_last": 36984,
   "lines_added": 21,
   "lines_deleted": 21,
   "lines": 42,
   "files": 12,
   "thinking": 947,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "replaced_by": 13,
   "tests_edited": 0,
   "score": 20,
   "of": 20,
   "api_cost": 0.153582
  },
  {
   "run": "claude-sonnet-5-5__control__B1__r2__7392f3",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "B1",
   "skill": "control",
   "rep": 2,
   "round": 11,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 2915,
   "input": 10,
   "cache_read": 116611,
   "cache_write": 28895,
   "warmup": null,
   "cache_write_1h": 28895,
   "cost_usd": 0.1680722,
   "wall_s": 30.1,
   "turns": 5,
   "prompt_last": 39131,
   "lines_added": 20,
   "lines_deleted": 20,
   "lines": 40,
   "files": 12,
   "thinking": 711,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "replaced_by": 13,
   "tests_edited": 0,
   "score": 19,
   "of": 20,
   "api_cost": 0.168072
  },
  {
   "run": "claude-sonnet-5-5__control__B1__r3__cb45cc",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "B1",
   "skill": "control",
   "rep": 3,
   "round": 11,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 3245,
   "input": 10,
   "cache_read": 118084,
   "cache_write": 27699,
   "warmup": null,
   "cache_write_1h": 27699,
   "cost_usd": 0.1668828,
   "wall_s": 28.1,
   "turns": 6,
   "prompt_last": 37935,
   "lines_added": 22,
   "lines_deleted": 22,
   "lines": 44,
   "files": 12,
   "thinking": 815,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "replaced_by": 13,
   "tests_edited": 0,
   "score": 19,
   "of": 20,
   "api_cost": 0.166883
  },
  {
   "run": "claude-sonnet-5-5__control__B2__r1__4e4e0b",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "B2",
   "skill": "control",
   "rep": 1,
   "round": 12,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 8561,
   "input": 10,
   "cache_read": 117957,
   "cache_write": 33653,
   "warmup": null,
   "cache_write_1h": 33653,
   "cost_usd": 0.2438334,
   "wall_s": 54.0,
   "turns": 6,
   "prompt_last": 43889,
   "lines_added": 202,
   "lines_deleted": 16,
   "lines": 218,
   "files": 9,
   "thinking": 1436,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "replaced_by": 13,
   "tests_edited": 0,
   "score": 17,
   "of": 17,
   "api_cost": 0.243833
  },
  {
   "run": "claude-sonnet-5-5__control__B2__r2__5c5cd0",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "B2",
   "skill": "control",
   "rep": 2,
   "round": 12,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 11618,
   "input": 20,
   "cache_read": 342735,
   "cache_write": 41288,
   "warmup": null,
   "cache_write_1h": 41288,
   "cost_usd": 0.34991900000000004,
   "wall_s": 86.9,
   "turns": 12,
   "prompt_last": 51524,
   "lines_added": 282,
   "lines_deleted": 19,
   "lines": 301,
   "files": 8,
   "thinking": 1793,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "replaced_by": 13,
   "tests_edited": 0,
   "score": 17,
   "of": 17,
   "api_cost": 0.349919
  },
  {
   "run": "claude-sonnet-5-5__control__B2__r3__5a9735",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "B2",
   "skill": "control",
   "rep": 3,
   "round": 12,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 8445,
   "input": 16,
   "cache_read": 247205,
   "cache_write": 36058,
   "warmup": null,
   "cache_write_1h": 36058,
   "cost_usd": 0.278155,
   "wall_s": 58.7,
   "turns": 9,
   "prompt_last": 46294,
   "lines_added": 231,
   "lines_deleted": 14,
   "lines": 245,
   "files": 8,
   "thinking": 1120,
   "commands": null,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "replaced_by": 13,
   "tests_edited": 0,
   "score": 17,
   "of": 17,
   "api_cost": 0.278155
  },
  {
   "run": "claude-sonnet-5-5__caveman__T3__r1__d38e6a",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "caveman",
   "rep": 1,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 653,
   "input": 4,
   "cache_read": 38360,
   "cache_write": 18698,
   "warmup": null,
   "cache_write_1h": 18698,
   "cost_usd": 0.089002,
   "wall_s": 6.8,
   "turns": 2,
   "prompt_last": 28934,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` calls `entries_in_month`. That function keeps only entries whose `day.year` and `day.month` both match. The report is a dict with three keys:\n- `count`: number of matching entries.\n- `total`: `total()` sums `amount` over the matching entries.\n- `by_category`: `sum_by_category()` groups matching amounts by category. Category is already lowercased at insert time.\n\nAn empty month gives `count` 0, `total` 0 (int, not float, since `sum` of empty starts at 0), and `by_category` `{}`.\n\n**Invalid date.** Dates are parsed only in `add_entry`, through `parse_date`. `parse_date` splits the text on `-`. It raises `ValueError(f\"bad date: {text!r}\")` when either of these holds:\n- The text does not have exactly 3 parts.\n- Any part is not all digits. This also catches empty parts and negative signs.\n\nValues that pass the format check but are impossible, like `2024-13-45`, make `date(y, m, d)` raise its own `ValueError`. The message differs, for example \"month must be in 1..12\".\n\nOrder in `add_entry` matters. The amount check runs first. Then `parse_date` runs while the `Entry` is built, before `append`. So an invalid date raises before any change to `entries`, and the list stays unmodified. Nothing catches the error. It goes to the caller.\n\nNon-string input, like `None`, raises `AttributeError` from `.split`, not `ValueError`.\n\nNote: the `isdigit()` check accepts non-ASCII digits, such as Arabic-Indic numerals. `int()` also accepts them, so those dates parse.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "replaced_by": 13,
   "api_cost": 0.089002
  },
  {
   "run": "claude-sonnet-5-5__caveman__T3__r2__0b5098",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "caveman",
   "rep": 2,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 746,
   "input": 4,
   "cache_read": 38357,
   "cache_write": 18694,
   "warmup": null,
   "cache_write_1h": 18694,
   "cost_usd": 0.08991539999999999,
   "wall_s": 9.2,
   "turns": 2,
   "prompt_last": 28930,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` computes them. It first calls `entries_in_month`. That function keeps each entry whose `day.year` equals `year` and whose `day.month` equals `month`. The report then returns a dict with three keys:\n\n- `count`: number of entries in that month.\n- `total`: sum of their `amount` values, from `total`.\n- `by_category`: dict from `sum_by_category`, mapping each category to its summed amount.\n\nCategories are already lowercase, because `add_entry` lowercases them when it stores an entry. So \"Food\" and \"food\" land in the same bucket. An empty month gives `count` 0, `total` 0 and an empty `by_category`. Amounts are plain `float`, so sums can carry normal floating-point rounding error.\n\n**Invalid date strings.** `monthly_report` never sees date strings. Dates are parsed earlier, in `add_entry`, through `parse_date`. `parse_date` splits the text on `-`. It raises `ValueError(f\"bad date: {text!r}\")` if either of these holds:\n\n- The text does not split into exactly 3 parts.\n- Any part is not all digits. This also rejects negative numbers and empty parts.\n\nIf the format passes, it calls `date(y, m, d)`. That call raises its own `ValueError` for out-of-range values, such as month 13 or Feb 30. The message is different, for example \"month must be in 1..12\".\n\n`parse_date` runs inside the `Entry(...)` constructor call in `add_entry`. A bad date therefore raises before `entries.append`, and the list stays unchanged. The `amount <= 0` check runs first. A bad amount is reported before a bad date.\n\nThe code does not catch the error. The caller must handle it.\n\nSome inputs that look wrong still pass the format check. `\"2024-1-5\"` is accepted because `isdigit` does not require zero padding. A non-string `text` fails with `AttributeError` on `.split`, not `ValueError`.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "replaced_by": 13,
   "api_cost": 0.089915
  },
  {
   "run": "claude-sonnet-5-5__caveman__T3__r3__4afd58",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "caveman",
   "rep": 3,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 747,
   "input": 4,
   "cache_read": 28407,
   "cache_write": 9067,
   "warmup": null,
   "cache_write_1h": 9067,
   "cost_usd": 0.049427399999999996,
   "wall_s": 8.9,
   "turns": 2,
   "prompt_last": 19303,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` computes them. It calls `entries_in_month`, which keeps only entries whose `day.year` and `day.month` match the arguments. It returns a dict with three values:\n- `count`: number of matching entries.\n- `total`: `total()` over the matching entries, a plain float sum of `amount`.\n- `by_category`: `sum_by_category()` over the matching entries, a dict of category to summed amount. Categories are already lowercased by `add_entry`.\n\nEntries from other months or years never enter the sums. An empty month gives count 0, total 0 (the int from `sum` of an empty sequence), and an empty dict.\n\n**Invalid date strings.** `monthly_report` never parses strings. `Entry.day` is already a `date` object. Parsing happens earlier, in `add_entry`, which calls `parse_date(day)`.\n\n`parse_date` splits the text on `-`. It raises `ValueError(f\"bad date: {text!r}\")` if there are not exactly three parts, or if any part is not all digits. That check rejects `\"2024/01/05\"`, `\"2024-1\"`, `\"-2024-01-05\"` and `\"2024-01-x5\"`.\n\nIf the shape is valid but the values are impossible, such as `\"2024-13-01\"` or `\"2024-02-30\"`, the `date(y, m, d)` constructor raises its own `ValueError`. The message is different, for example \"month must be in 1..12\".\n\nEither way `ValueError` propagates out of `add_entry`. The check happens in the `Entry(...)` argument list, before `append`, so the bad entry is not added. The amount check runs first: a non-positive amount raises a different `ValueError` before the date is parsed.\n\nTwo edge cases:\n- `parse_date` accepts non-padded parts such as `\"2024-1-5\"`.\n- `parse_date` passes `None` or other non-strings through to `.split`, which raises `AttributeError`, not `ValueError`.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "replaced_by": 13,
   "api_cost": 0.049427,
   "light": true
  },
  {
   "run": "claude-sonnet-5-5__control__T3__r1__a75a74",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "control",
   "rep": 1,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 964,
   "input": 4,
   "cache_read": 25759,
   "cache_write": 6417,
   "warmup": null,
   "cache_write_1h": 6417,
   "cost_usd": 0.0404678,
   "wall_s": 10.5,
   "turns": 2,
   "prompt_last": 16653,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 73,
   "commands": null,
   "answer": "**How monthly totals are computed**\n\n`monthly_report(entries, year, month)` produces the monthly figures in two steps:\n\n1. It calls `entries_in_month`. That function keeps only the entries whose `day.year` and `day.month` both match the arguments. It compares the parsed `date` object's fields directly and does no string handling.\n2. It builds a dict from that filtered list:\n   - `count` is the number of entries.\n   - `total` is the sum of their `amount` values, from `total()`.\n   - `by_category` maps each category to its summed amount, from `sum_by_category()`.\n\nA month with no entries gives a count of 0, a total of 0 and an empty `by_category` dict. Categories are lowercased in `add_entry`, so \"Food\" and \"food\" land in the same bucket.\n\n**What happens with an invalid date string**\n\nDate strings are only handled when an entry is added. `add_entry` calls `parse_date(day)`, which:\n\n- splits the string on `-`;\n- raises `ValueError(f\"bad date: ...\")` if there aren't exactly three parts or any part isn't all digits.\n\nOtherwise it passes the integers to `date(y, m, d)`. That constructor raises its own `ValueError` for out-of-range values such as month 13 or Feb 30.\n\nSince `add_entry` evaluates `parse_date` while constructing the `Entry`, a bad date raises before anything is appended. The `entries` list is left unchanged, and the error propagates to the caller. The check for `amount <= 0` runs first, so a call with both a bad amount and a bad date reports the amount error.\n\nSome inputs slip through `parse_date`'s digit check and fail later or parse oddly:\n\n- `\"2024-1-5\"` is accepted, because the parts aren't required to be zero-padded.\n- A non-string input such as `None` raises `AttributeError` on `.split`, not `ValueError`.\n- A negative-looking part like `\"2024--1-5\"` produces an empty part, which fails `isdigit`, so it is rejected.\n\n`monthly_report` never sees date strings. Once entries are stored their dates are valid `date` objects, so it can't fail on a bad date.",
   "account_remark": true,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "replaced_by": 13,
   "api_cost": 0.040468,
   "light": true
  },
  {
   "run": "claude-sonnet-5-5__control__T3__r2__c4fa7b",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "control",
   "rep": 2,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 858,
   "input": 4,
   "cache_read": 35712,
   "cache_write": 16049,
   "warmup": null,
   "cache_write_1h": 16049,
   "cost_usd": 0.07992640000000001,
   "wall_s": 8.3,
   "turns": 2,
   "prompt_last": 26285,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` in `ledger/core.py:51` does the work. It first calls `entries_in_month`, which keeps only entries whose `day.year` and `day.month` both match. It then builds a dict from that filtered list:\n\n- `count` is the number of matching entries.\n- `total` is the sum of their `amount` values, computed by `total()`.\n- `by_category` is a dict from `sum_by_category()`. It maps each category to the sum of its amounts, starting each category at `0.0`.\n\nCategories are lowercased when an entry is added, so \"Food\" and \"food\" land in the same bucket. A month with no entries gives `count` 0, `total` 0 (the integer, since `sum` of an empty sequence returns 0) and an empty `by_category`.\n\n**Invalid date strings.** Dates are parsed once, when an entry is added. `add_entry` calls `parse_date(day)` at `ledger/core.py:27`. `parse_date` splits the text on `-` and raises `ValueError(f\"bad date: ...\")` if there are not exactly three parts or if any part is not all digits. Examples are `\"2024/01/05\"`, `\"2024-1\"`, `\"abc\"` and `\"2024-01--5\"`. Because the parse happens inside the `Entry(...)` construction, the error is raised before `entries.append` runs, so the bad entry is never added and the list is unchanged.\n\nTwo related behaviours:\n\n- **Format-valid but impossible dates.** Something like `\"2024-13-45\"` passes the digit check. It then fails inside `date(y, m, d)`, which raises its own `ValueError` (\"month must be in 1..12\" and similar). Callers see a `ValueError` either way, but with a different message.\n- **Loose formatting.** Because the check only requires digits, `\"2024-1-5\"` is accepted, as are unpadded parts. Negative numbers are rejected, since `-` is the separator and leaves an empty part.\n\nIn `add_entry`, the `amount <= 0` check runs before date parsing. An entry with both a bad amount and a bad date therefore reports the amount error first.\n\nThe monthly functions never see raw strings, because `Entry.day` is already a `date`. An invalid date can't reach the monthly totals. It fails at insertion.\n\nI didn't change any files.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "replaced_by": 13,
   "api_cost": 0.079926
  },
  {
   "run": "claude-sonnet-5-5__control__T3__r3__70757d",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "control",
   "rep": 3,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 854,
   "input": 4,
   "cache_read": 25850,
   "cache_write": 9342,
   "warmup": null,
   "cache_write_1h": 9342,
   "cost_usd": 0.051086,
   "wall_s": 7.5,
   "turns": 2,
   "prompt_last": 19578,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals**\n\n`monthly_report(entries, year, month)` (line 51) builds the report in three steps:\n\n1. `entries_in_month` filters the list, keeping entries whose `day.year` and `day.month` both match the arguments. Each `Entry.day` is already a `date` object.\n2. It returns a dict with three keys:\n   - `count` is the number of entries in that month.\n   - `total` comes from `total()`, which sums `amount` over the filtered entries.\n   - `by_category` comes from `sum_by_category()`. It walks the entries and adds each amount into a dict keyed by category, starting from 0.0.\n3. Categories are lowercased in `add_entry`, so \"Food\" and \"food\" land in the same bucket.\n\nThe sums are plain `float` addition, with no rounding. Amounts like 0.1 and 0.2 can therefore give results with floating-point noise.\n\n**Invalid date strings**\n\nDate parsing happens only when an entry is added, in `add_entry` (line 27), which calls `parse_date`. The reporting functions never see strings.\n\n`parse_date` splits the text on `-`. It raises `ValueError(f\"bad date: {text!r}\")` if either of these holds:\n- There aren't exactly three parts.\n- Any part isn't all digits.\n\nThat check catches `\"2024/01/05\"`, `\"2024-1\"`, `\"abc\"`, and negative-looking or blank parts.\n\nA string that passes the check but isn't a real calendar date, such as `\"2024-13-01\"` or `\"2024-02-30\"`, gets past the format check. It then fails inside `date(y, m, d)`, which raises its own `ValueError` (e.g. \"month must be in 1..12\"). The message differs from the \"bad date\" one, but the exception type is the same.\n\nNon-string input, such as `None`, fails earlier with an `AttributeError` on `.split`, not a `ValueError`.\n\nNothing catches the error, so it propagates to the caller of `add_entry`. Two details about that call:\n- `amount` is validated before the date, so a non-positive amount raises its own `ValueError` first.\n- `parse_date` runs while the `Entry(...)` arguments are being evaluated, before `entries.append` executes. A bad date therefore never adds a partial entry, and the list is left unchanged.\n\nI didn't change any files.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "replaced_by": 13,
   "api_cost": 0.051086,
   "light": true
  },
  {
   "run": "claude-sonnet-5-5__karpathy__T3__r1__97de84",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "karpathy",
   "rep": 1,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 785,
   "input": 4,
   "cache_read": 36727,
   "cache_write": 17064,
   "warmup": null,
   "cache_write_1h": 17064,
   "cost_usd": 0.08345939999999999,
   "wall_s": 8.0,
   "turns": 2,
   "prompt_last": 27300,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` in `ledger/core.py:51` does the work. It first calls `entries_in_month`, which keeps only the entries whose `day.year` and `day.month` both match the arguments. It then builds a dict from that filtered list with three values:\n- `count` is the number of matching entries.\n- `total` is the sum of their `amount` values, from `total()`.\n- `by_category` is a dict of per-category sums, from `sum_by_category()`. Categories are lowercased when an entry is added, so \"Food\" and \"food\" land in the same bucket.\n\nA month with no entries gives a count of 0, a total of 0 (the integer, since `sum` of an empty sequence is 0) and an empty `by_category`.\n\n**Invalid date strings.** Date parsing happens only in `add_entry` (`ledger/core.py:24`), which calls `parse_date` on the incoming string. `parse_date` splits the text on `-`. It raises `ValueError(\"bad date: ...\")` if there aren't exactly three parts or if any part isn't all digits. That means it rejects things like `\"2024/01/05\"`, `\"2024-1\"` and `\"2024-01-xx\"`, and also negative numbers, because the minus sign becomes a separator.\n\nIf the string has three digit groups but isn't a real date, such as `\"2024-13-40\"`, the `date(y, m, d)` constructor raises its own `ValueError`. The message is different, for example \"month must be in 1..12\".\n\nIn `add_entry` the amount check runs first. A non-positive amount raises before the date is parsed. If parsing fails, the exception propagates and nothing is appended. The `Entry` is built as an argument to `append`, so the list is left unchanged.\n\nOnce an entry exists, its `day` is always a valid `date`. The monthly report never has to handle bad dates, and an invalid date can only fail at insertion time.\n\n`parse_date` accepts non-padded parts like `\"2024-1-5\"`, since it only checks that each part is digits. The docstring says \"YYYY-MM-DD\", so this is looser than it claims.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "replaced_by": 13,
   "api_cost": 0.083459
  },
  {
   "run": "claude-sonnet-5-5__karpathy__T3__r2__559943",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "karpathy",
   "rep": 2,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 868,
   "input": 4,
   "cache_read": 36725,
   "cache_write": 17061,
   "warmup": null,
   "cache_write_1h": 17061,
   "cost_usd": 0.084277,
   "wall_s": 9.1,
   "turns": 2,
   "prompt_last": 27297,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` in `ledger/core.py:51` does the work in three steps:\n\n1. It calls `entries_in_month`, which keeps only entries whose `day.year` and `day.month` both match the arguments.\n2. It passes that filtered list to `total`, which adds up every `amount`. This gives the month's overall figure.\n3. It passes the same list to `sum_by_category`, which builds a dict of category to running sum.\n\nIt returns a dict with three keys: `count` (the number of entries that month), `total`, and `by_category`. If no entries match, `count` is 0, `total` is `0` (the integer that `sum` returns for an empty sequence, not `0.0`), and `by_category` is `{}`.\n\nNothing is stored between calls. Each call re-filters and re-sums the whole list. Amounts are plain floats, so the sums can pick up ordinary floating-point rounding error.\n\n**Invalid date strings.** Dates are only parsed when an entry is added, in `add_entry` (line 27), which calls `parse_date`. `parse_date` splits the text on `-`. It raises `ValueError(f\"bad date: {text!r}\")` if there aren't exactly three parts or if any part contains a non-digit character. A negative year or a string like `\"2024-1-5\"` gets past the digit check. Only the part count and the digits matter here, so single-digit months and days are accepted.\n\nThere are two more failure modes:\n- If the parts are all digits but don't form a real date, such as `2024-02-30` or month `13`, the `date(y, m, d)` constructor raises its own `ValueError`, with a different message such as \"day is out of range for month\".\n- If `text` isn't a string (for example `None`), `text.split` raises `AttributeError`, not `ValueError`, despite the docstring.\n\nNeither `add_entry` nor `parse_date` catches any of these, so the exception goes straight to the caller. `add_entry` parses the date while building the `Entry`, before `entries.append` runs, so a bad date never adds a partial entry to the list. It does check `amount <= 0` first, so a call with both a bad amount and a bad date reports the amount error.\n\nBecause dates are validated on the way in, `monthly_report` and the other functions can assume every `Entry.day` is a real `date`. They have no invalid-date handling of their own.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "replaced_by": 13,
   "api_cost": 0.084277
  },
  {
   "run": "claude-sonnet-5-5__karpathy__T3__r3__920bca",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "karpathy",
   "rep": 3,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 840,
   "input": 4,
   "cache_read": 36726,
   "cache_write": 17063,
   "warmup": null,
   "cache_write_1h": 17063,
   "cost_usd": 0.0840052,
   "wall_s": 8.3,
   "turns": 2,
   "prompt_last": 27299,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` in `ledger/core.py:51` first calls `entries_in_month`. That function keeps only the entries whose `day.year` and `day.month` both match the arguments. The report is a dict with three values:\n- `count` is the number of entries left after filtering.\n- `total` is `total()` applied to those entries, a plain sum of their `amount` floats.\n- `by_category` comes from `sum_by_category()`. It builds a dict mapping each category to its running sum, starting from 0.0.\n\nCategory names are lowercased when an entry is added (`add_entry`), so \"Food\" and \"food\" land in the same bucket. The sums are plain float addition with no rounding. If no entries match the month, you get `count` 0, `total` 0 (an int, since `sum` of an empty sequence is 0), and an empty `by_category` dict.\n\n**Invalid date strings.** Dates are only parsed in `add_entry`, through `parse_date` (`core.py:15`). The monthly functions work on already-parsed `date` objects and never see strings. `parse_date` splits the text on `-`. It raises `ValueError(\"bad date: ...\")` if there aren't exactly three parts, or if any part isn't all digits. It then converts the parts to ints and calls `date(y, m, d)`. If the format is right but the values are impossible, such as `2024-13-01` or `2024-02-30`, `date()` raises its own `ValueError`, with a different message.\n\n`add_entry` validates the amount before it parses the date. A bad date therefore propagates as an uncaught `ValueError` and nothing is appended to the list. Nothing catches or logs the error, so callers must handle it.\n\nSome edge cases follow from the digit check:\n- `\"2024-1-5\"` is accepted, because the parts aren't required to be zero-padded.\n- Negative parts like `\"-2024-01-01\"` are rejected, since the split gives an empty first part.\n- A non-string input, such as `None`, raises `AttributeError` at `.split` instead of `ValueError`.\n- `str.isdigit()` also accepts some non-ASCII digit characters, for example superscripts like `\"\u00b2\"`. `int()` can't convert those, so they raise a `ValueError` with a different message.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "replaced_by": 13,
   "api_cost": 0.084005
  },
  {
   "run": "claude-sonnet-5-5__placebo__T3__r1__87ea61",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "placebo",
   "rep": 1,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 889,
   "input": 4,
   "cache_read": 36497,
   "cache_write": 16834,
   "warmup": null,
   "cache_write_1h": 16834,
   "cost_usd": 0.08353340000000001,
   "wall_s": 10.6,
   "turns": 2,
   "prompt_last": 27070,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` in `ledger/core.py:51` builds the monthly figures in three steps.\n\n1. `entries_in_month` (line 40) keeps only the entries whose `day.year` and `day.month` both match the arguments.\n2. `total` (line 31) sums the `amount` of those entries. For an empty month, `sum` of nothing returns `0`.\n3. `sum_by_category` (line 44) builds a dict from category to running sum, starting each category at `0.0`.\n\nThe function returns a dict with three keys:\n- `count`: the number of entries in the month.\n- `total`: the sum of their amounts.\n- `by_category`: the per-category sums.\n\nCategories are lowercased when an entry is added, so `\"Food\"` and `\"food\"` land in the same bucket. `monthly_report` doesn't parse any dates itself. It compares against the `date` objects already stored on each `Entry`.\n\n**Invalid date strings.** Date strings are only handled when an entry is added, through `add_entry` (line 24), which calls `parse_date` (line 15).\n\n- `parse_date` splits the text on `-`. It raises `ValueError(f\"bad date: {text!r}\")` if there aren't exactly three parts or if any part isn't made up only of digits. That rejects things like `\"2024/01/05\"`, `\"2024-1\"`, `\"2024-01-xx\"` and negative numbers.\n- If the format passes, it calls `date(y, m, d)`. A well-formed but impossible date such as `\"2024-13-40\"` or `\"2023-02-29\"` makes `datetime.date` raise its own `ValueError`, for example \"month must be in 1..12\". That message is less specific than the \"bad date\" one.\n- `parse_date` also assumes `text` is a string. `None` or another non-string type would raise `AttributeError` at `.split`, not `ValueError`.\n- Nothing catches these errors, so they propagate to the caller of `add_entry`. Because `parse_date` runs while the `Entry` is being constructed, the `append` never happens. A bad date therefore never adds a partial entry, and the list is left unchanged. The `amount <= 0` check runs first, so an entry with a non-positive amount and a bad date reports the amount error.\n\nInvalid dates can't reach the monthly totals. They are rejected at entry time, so `monthly_report` only ever sees valid `date` objects.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "replaced_by": 13,
   "api_cost": 0.083533
  },
  {
   "run": "claude-sonnet-5-5__placebo__T3__r2__744d38",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "placebo",
   "rep": 2,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 910,
   "input": 4,
   "cache_read": 26604,
   "cache_write": 10098,
   "warmup": null,
   "cache_write_1h": 10098,
   "cost_usd": 0.0548208,
   "wall_s": 9.4,
   "turns": 2,
   "prompt_last": 20334,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals**\n\n`monthly_report(entries, year, month)` in `ledger/core.py` builds the monthly figures in three steps:\n\n1. `entries_in_month` keeps only the entries whose `day.year` and `day.month` both match the arguments. Each `Entry` stores `day` as a real `date` object, so this is a plain attribute comparison.\n2. `total` adds up the `amount` of those entries to give the month's overall spend.\n3. `sum_by_category` walks the same entries and builds a dict from category to running sum. Categories were lowercased when the entry was added, so `Food` and `food` end up in the same bucket.\n\nThe function returns a dict with `count` (the number of entries in the month), `total`, and `by_category`. A month with no entries gives a count of 0, a total of `0` (the int that `sum` returns for an empty sequence) and an empty category dict. Nothing is raised in that case.\n\n**Invalid date strings**\n\n`monthly_report` never sees date strings. Parsing happens earlier, in `add_entry`, which calls `parse_date(day)`.\n\n`parse_date` expects `YYYY-MM-DD`. It splits the text on `-` and raises `ValueError(f\"bad date: {text!r}\")` if either of these is true:\n- there aren't exactly three parts\n- any part contains a non-digit character (this includes empty parts, and negative years, because the sign splits into an extra part)\n\nIf the shape is right, it converts the parts to ints and calls `date(y, m, d)`. Values that are numeric but impossible, such as `2024-13-01` or `2023-02-30`, fail inside `date()`. That raises a `ValueError` too, with the standard library's message (for example \"month must be in 1..12\") instead of the \"bad date\" one. The check also doesn't enforce digit counts, so `2024-1-5` is accepted.\n\nIn `add_entry`, `parse_date` is evaluated while the `Entry` is being constructed, which happens before `entries.append`. A bad date therefore propagates the `ValueError` to the caller and the entry is not added. The list is left unchanged. The amount check (`amount <= 0` raises `ValueError`) also runs first, so an entry with both a bad amount and a bad date reports the amount error. `parse_date` is not wrapped in a try/except, so callers have to handle the exception themselves.\n\nA non-string `day` (such as `None`) would fail earlier with an `AttributeError` on `.split`, not a `ValueError`.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "replaced_by": 13,
   "api_cost": 0.054821,
   "light": true
  },
  {
   "run": "claude-sonnet-5-5__placebo__T3__r3__39a254",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "placebo",
   "rep": 3,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 930,
   "input": 4,
   "cache_read": 36461,
   "cache_write": 16797,
   "warmup": null,
   "cache_write_1h": 16797,
   "cost_usd": 0.08378820000000001,
   "wall_s": 8.7,
   "turns": 2,
   "prompt_last": 27033,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` in `ledger/core.py:51` builds the monthly figures in three steps:\n\n1. It calls `entries_in_month`, which keeps only the entries whose `day.year` and `day.month` both match the arguments.\n2. It passes that filtered list to `total`, which adds up the `amount` of each entry. That gives the month's overall total.\n3. It also passes the list to `sum_by_category`, which builds a dict mapping each category name to the sum of its amounts.\n\nIt returns a dict with three keys:\n- `count`: the number of entries in the month.\n- `total`: the month's overall total.\n- `by_category`: the per-category sums.\n\nA month with no entries gives `count` 0, `total` 0 and an empty `by_category`. The `total` is the integer 0 in that case, because `sum` over an empty sequence returns `0`. Amounts are plain floats, so sums can carry ordinary floating-point rounding noise. Nothing rounds them.\n\n**Invalid date strings.** Dates are only parsed when an entry is added. `add_entry` calls `parse_date`, which splits the text on `-`. It raises `ValueError(f\"bad date: ...\")` if there aren't exactly three parts or if any part isn't all digits. That check catches things like `\"2024/01/05\"`, `\"2024-1\"`, `\"\"` and `\"2024-01--5\"`.\n\nOnce the format passes, the values go to `date(y, m, d)`. If they are out of range, for example `\"2024-13-01\"` or `\"2024-02-30\"`, the `datetime.date` constructor raises its own `ValueError`, such as \"month must be in 1..12\". That message doesn't include the original string.\n\n`add_entry` doesn't catch either error, so the caller sees the `ValueError`. Because `parse_date` runs inside the `Entry(...)` construction, which happens before `entries.append`, a bad date never adds a partial entry to the list.\n\nThe amount check runs first. A non-positive amount raises `ValueError(\"amount must be positive\")` before the date is parsed. So an entry with both problems reports the amount error.\n\nSome inputs fail differently. If `text` isn't a string, for example `None`, `text.split` raises `AttributeError` instead of `ValueError`. Also, `str.isdigit()` accepts some non-ASCII digit characters, and `int()` converts them, so a few odd inputs get through the format check.\n\nSince `monthly_report` works on `Entry` objects that already hold `date` values, it never sees invalid dates.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "replaced_by": 13,
   "api_cost": 0.083788
  },
  {
   "run": "claude-sonnet-5-5__ponytail__T3__r1__7b8509",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "ponytail",
   "rep": 1,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 818,
   "input": 4,
   "cache_read": 36741,
   "cache_write": 17079,
   "warmup": null,
   "cache_write_1h": 17079,
   "cost_usd": 0.08385219999999999,
   "wall_s": 9.9,
   "turns": 2,
   "prompt_last": 27315,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` in `ledger/core.py:51` does three things:\n\n1. `entries_in_month` keeps only the entries whose `day.year` and `day.month` both match the arguments.\n2. `total` adds up the `amount` of those entries with `sum`.\n3. `sum_by_category` builds a dict that maps each category to its running total. It uses `out.get(cat, 0.0) + amount`.\n\nIt returns `{\"count\": ..., \"total\": ..., \"by_category\": {...}}`. A month with no entries gives count 0, total 0 and an empty dict.\n\nAmounts are plain floats, so the totals can carry small rounding error. Categories are lowercased when an entry is added (`add_entry`), so \"Food\" and \"food\" land in the same bucket.\n\n**Invalid date strings.** Dates are parsed only in `add_entry`, through `parse_date` (`ledger/core.py:15`). That function splits the text on `-` and raises `ValueError(\"bad date: ...\")` in two cases:\n\n- The text doesn't have exactly three parts.\n- Any part contains something other than digits.\n\nOtherwise it converts the parts to ints and calls `date(y, m, d)`. That call raises its own `ValueError` for out-of-range values such as month 13 or Feb 30. Its message is different from the \"bad date\" one.\n\n`add_entry` calls `parse_date` inside the `Entry(...)` construction. An invalid date therefore raises before `entries.append` runs, and the list is left unchanged. Nothing catches the error, so the caller has to handle it.\n\n`add_entry` checks `amount <= 0` before it parses the date. If both the amount and the date are bad, you get the amount error.\n\nA few inputs behave in ways you might not expect:\n\n- `\"2024-1-5\"` is accepted, because there is no width check.\n- `\" 2024-01-05\"` is rejected, because the space makes the first part fail `isdigit`.\n- Unicode digits such as `\"\u00b2\"` pass `isdigit` but then make `int()` raise a `ValueError` with yet another message.\n\n`monthly_report` never sees date strings, only `date` objects, so bad dates can't reach the totals.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "replaced_by": 13,
   "api_cost": 0.083852
  },
  {
   "run": "claude-sonnet-5-5__ponytail__T3__r2__ca430d",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "ponytail",
   "rep": 2,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 724,
   "input": 4,
   "cache_read": 36740,
   "cache_write": 17078,
   "warmup": null,
   "cache_write_1h": 17078,
   "cost_usd": 0.08290799999999998,
   "wall_s": 7.1,
   "turns": 2,
   "prompt_last": 27314,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` in `ledger/core.py:51` does the work in three steps:\n\n1. It calls `entries_in_month`, which keeps only the entries whose `day.year` and `day.month` both match the arguments.\n2. It passes that filtered list to `total`, which adds up each entry's `amount` with `sum`.\n3. It passes the same list to `sum_by_category`, which builds a dict of category to running sum.\n\nIt returns `{\"count\", \"total\", \"by_category\"}`. A month with no entries gives count 0, total 0 and an empty dict. Nothing is stored or cached. The report is recomputed from the raw entry list on every call.\n\n**Invalid date strings.** Dates are parsed only when an entry is added, in `add_entry` (`core.py:27`), which calls `parse_date`. `parse_date` splits the text on `-`. It raises `ValueError(\"bad date: ...\")` if the result isn't exactly three parts or any part isn't all digits.\n\nOtherwise it passes the integers to `datetime.date(y, m, d)`. That raises its own `ValueError` for out-of-range values such as month 13 or Feb 30. So malformed strings and impossible dates both raise `ValueError`, with different messages.\n\nThe error is not caught anywhere in this file. In `add_entry`, the amount check runs first. Then `parse_date` runs while the `Entry` is being built, before `append`. A bad date therefore leaves the list unchanged, and the caller has to handle the exception.\n\n`monthly_report` never sees a bad date. By the time an entry is in the list, its `day` is already a valid `date` object.\n\nTwo quirks in `parse_date`:\n- `str.isdigit()` accepts some non-ASCII digits, and `date()` would still take those.\n- Inputs like `\"2024-1-5\"` are accepted, because the parts aren't checked for a fixed width.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "replaced_by": 13,
   "api_cost": 0.082908
  },
  {
   "run": "claude-sonnet-5-5__ponytail__T3__r3__72b8dc",
   "model": "claude-sonnet-5-5",
   "host": "claude",
   "task": "T3",
   "skill": "ponytail",
   "rep": 3,
   "round": 9,
   "date": "2026-09-28",
   "exit": 0,
   "passed": true,
   "output": 785,
   "input": 4,
   "cache_read": 36743,
   "cache_write": 17082,
   "warmup": null,
   "cache_write_1h": 17082,
   "cost_usd": 0.08353460000000001,
   "wall_s": 10.0,
   "turns": 2,
   "prompt_last": 27318,
   "lines_added": 0,
   "lines_deleted": 0,
   "lines": 0,
   "files": 0,
   "thinking": 0,
   "commands": null,
   "answer": "**Monthly totals.** `monthly_report(entries, year, month)` in `ledger/core.py:51` builds the totals in three steps:\n\n1. `entries_in_month` (line 40) keeps only the entries whose `day.year` and `day.month` match the arguments.\n2. `total` (line 31) adds up the `amount` of those entries. `sum_by_category` (line 44) does the same, grouped by each entry's lowercased category.\n3. It returns a dict with `count`, `total` and `by_category`.\n\nAn empty month gives `count` 0, `total` 0 and an empty `by_category`. The function reads only `Entry.day`, which is already a `date`, so it never sees date strings.\n\n**Invalid date strings.** Strings are only parsed when an entry is added. `add_entry` (line 24) calls `parse_date` (line 15), which splits the text on `-`. It raises `ValueError(f\"bad date: ...\")` if there aren't exactly three parts or any part isn't all digits. That rejects `\"2024/01/05\"`, `\"2024-1\"`, `\"2024-01-xx\"` and negative numbers.\n\nIf the format passes, the parts become ints and go to `date(y, m, d)`. That constructor raises its own `ValueError` for impossible values, such as month 13 or `2024-02-30`. Its message is different, for example \"month must be in 1..12\".\n\n`add_entry` doesn't catch either error. The exception goes straight to the caller. It also fires before `entries.append`, so a bad date never adds a partial entry. The one earlier check is `amount <= 0`, which raises its own `ValueError` first.\n\nTwo things I noticed:\n- `parse_date` calls `text.split`, so `None` or another non-string raises `AttributeError` rather than `ValueError`.\n- `isdigit()` accepts some non-ASCII digit characters. `int()` usually handles them, but it is looser than the docstring's \"YYYY-MM-DD\" suggests. It also doesn't require four-digit years or two-digit months, so `\"2024-1-5\"` is accepted.\n\nI didn't change any files.",
   "account_remark": false,
   "effort": null,
   "cli": "2.1.284 (Claude Code)",
   "replaced_by": 13,
   "api_cost": 0.083535
  },
  {
   "run": "gpt-5.6-sol__control__B1__r1__bc6bde",
   "model": "gpt-5.6-sol",
   "host": "codex",
   "task": "B1",
   "skill": "control",
   "rep": 1,
   "round": 11,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 11107,
   "input": 461239,
   "cache_read": 423296,
   "cache_write": 49377,
   "warmup": 11434,
   "cache_write_1h": null,
   "cost_usd": 0.588966,
   "wall_s": 239.6,
   "turns": null,
   "prompt_last": null,
   "lines_added": 205,
   "lines_deleted": 27,
   "lines": 232,
   "files": 15,
   "thinking": 3751,
   "commands": 11,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "replaced_by": 13,
   "tests_edited": 0,
   "score": 20,
   "of": 20,
   "api_cost": 0.588966
  },
  {
   "run": "gpt-5.6-sol__control__B1__r2__6efac3",
   "model": "gpt-5.6-sol",
   "host": "codex",
   "task": "B1",
   "skill": "control",
   "rep": 2,
   "round": 11,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 9863,
   "input": 381559,
   "cache_read": 346624,
   "cache_write": 46369,
   "warmup": 11434,
   "cache_write_1h": null,
   "cost_usd": 0.521386,
   "wall_s": 205.5,
   "turns": null,
   "prompt_last": null,
   "lines_added": 191,
   "lines_deleted": 23,
   "lines": 214,
   "files": 15,
   "thinking": 3479,
   "commands": 8,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "replaced_by": 13,
   "tests_edited": 0,
   "score": 20,
   "of": 20,
   "api_cost": 0.521386
  },
  {
   "run": "gpt-5.6-sol__control__B1__r3__15fbb7",
   "model": "gpt-5.6-sol",
   "host": "codex",
   "task": "B1",
   "skill": "control",
   "rep": 3,
   "round": 11,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 10278,
   "input": 470902,
   "cache_read": 435840,
   "cache_write": 46496,
   "warmup": 11434,
   "cache_write_1h": null,
   "cost_usd": 0.56588,
   "wall_s": 221.5,
   "turns": null,
   "prompt_last": null,
   "lines_added": 204,
   "lines_deleted": 25,
   "lines": 229,
   "files": 15,
   "thinking": 3308,
   "commands": 8,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "replaced_by": 13,
   "tests_edited": 0,
   "score": 20,
   "of": 20,
   "api_cost": 0.56588
  },
  {
   "run": "gpt-5.6-sol__control__B2__r1__ec2ee7",
   "model": "gpt-5.6-sol",
   "host": "codex",
   "task": "B2",
   "skill": "control",
   "rep": 1,
   "round": 12,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 11778,
   "input": 597392,
   "cache_read": 557056,
   "cache_write": 51770,
   "warmup": 11434,
   "cache_write_1h": null,
   "cost_usd": 0.665462,
   "wall_s": 249.2,
   "turns": null,
   "prompt_last": null,
   "lines_added": 323,
   "lines_deleted": 13,
   "lines": 336,
   "files": 10,
   "thinking": 3255,
   "commands": 15,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "replaced_by": 13,
   "tests_edited": 0,
   "score": 17,
   "of": 17,
   "api_cost": 0.665462
  },
  {
   "run": "gpt-5.6-sol__control__B2__r2__97cd20",
   "model": "gpt-5.6-sol",
   "host": "codex",
   "task": "B2",
   "skill": "control",
   "rep": 2,
   "round": 12,
   "date": "2026-09-29",
   "exit": 0,
   "passed": false,
   "output": 10728,
   "input": 492542,
   "cache_read": 459392,
   "cache_write": 44584,
   "warmup": 11434,
   "cache_write_1h": null,
   "cost_usd": 0.576653,
   "wall_s": 234.1,
   "turns": null,
   "prompt_last": null,
   "lines_added": 291,
   "lines_deleted": 17,
   "lines": 308,
   "files": 11,
   "thinking": 3044,
   "commands": 11,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "replaced_by": 13,
   "tests_edited": 1,
   "score": 17,
   "of": 17,
   "api_cost": 0.576653
  },
  {
   "run": "gpt-5.6-sol__control__B2__r3__f0f57e",
   "model": "gpt-5.6-sol",
   "host": "codex",
   "task": "B2",
   "skill": "control",
   "rep": 3,
   "round": 12,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 12017,
   "input": 615756,
   "cache_read": 581632,
   "cache_write": 45558,
   "warmup": 11434,
   "cache_write_1h": null,
   "cost_usd": 0.655225,
   "wall_s": 264.2,
   "turns": null,
   "prompt_last": null,
   "lines_added": 354,
   "lines_deleted": 16,
   "lines": 370,
   "files": 11,
   "thinking": 3657,
   "commands": 12,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "replaced_by": 13,
   "tests_edited": 0,
   "score": 17,
   "of": 17,
   "api_cost": 0.655225
  },
  {
   "run": "gpt-6-sol__control__B1__r1__b64084",
   "model": "gpt-6-sol",
   "host": "codex",
   "task": "B1",
   "skill": "control",
   "rep": 1,
   "round": 11,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 11504,
   "input": 833165,
   "cache_read": 771968,
   "cache_write": 73181,
   "warmup": 11984,
   "cache_write_1h": null,
   "cost_usd": 0.415796,
   "wall_s": 308.3,
   "turns": null,
   "prompt_last": null,
   "lines_added": 223,
   "lines_deleted": 32,
   "lines": 255,
   "files": 15,
   "thinking": 3123,
   "commands": 16,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "replaced_by": 13,
   "tests_edited": 0,
   "score": 20,
   "of": 20,
   "api_cost": 0.415796
  },
  {
   "run": "gpt-6-sol__control__B1__r2__31ba85",
   "model": "gpt-6-sol",
   "host": "codex",
   "task": "B1",
   "skill": "control",
   "rep": 2,
   "round": 11,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 11808,
   "input": 909252,
   "cache_read": 865408,
   "cache_write": 55828,
   "warmup": 11984,
   "cache_write_1h": null,
   "cost_usd": 0.402818,
   "wall_s": 259.3,
   "turns": null,
   "prompt_last": null,
   "lines_added": 209,
   "lines_deleted": 34,
   "lines": 243,
   "files": 15,
   "thinking": 2413,
   "commands": 21,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "replaced_by": 13,
   "tests_edited": 0,
   "score": 20,
   "of": 20,
   "api_cost": 0.402818
  },
  {
   "run": "gpt-6-sol__control__B1__r3__891c98",
   "model": "gpt-6-sol",
   "host": "codex",
   "task": "B1",
   "skill": "control",
   "rep": 3,
   "round": 11,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 9965,
   "input": 550109,
   "cache_read": 516992,
   "cache_write": 45101,
   "warmup": 11984,
   "cache_write_1h": null,
   "cost_usd": 0.29325,
   "wall_s": 223.6,
   "turns": null,
   "prompt_last": null,
   "lines_added": 201,
   "lines_deleted": 26,
   "lines": 227,
   "files": 15,
   "thinking": 3110,
   "commands": 17,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "replaced_by": 13,
   "tests_edited": 0,
   "score": 20,
   "of": 20,
   "api_cost": 0.29325
  },
  {
   "run": "gpt-6-sol__control__B2__r1__770340",
   "model": "gpt-6-sol",
   "host": "codex",
   "task": "B2",
   "skill": "control",
   "rep": 1,
   "round": 12,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 7328,
   "input": 472582,
   "cache_read": 444544,
   "cache_write": 40022,
   "warmup": 11984,
   "cache_write_1h": null,
   "cost_usd": 0.242233,
   "wall_s": 173.6,
   "turns": null,
   "prompt_last": null,
   "lines_added": 205,
   "lines_deleted": 11,
   "lines": 216,
   "files": 9,
   "thinking": 1292,
   "commands": 23,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "replaced_by": 13,
   "tests_edited": 0,
   "score": 17,
   "of": 17,
   "api_cost": 0.242233
  },
  {
   "run": "gpt-6-sol__control__B2__r2__8f260d",
   "model": "gpt-6-sol",
   "host": "codex",
   "task": "B2",
   "skill": "control",
   "rep": 2,
   "round": 12,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 9183,
   "input": 741646,
   "cache_read": 698112,
   "cache_write": 55518,
   "warmup": 11984,
   "cache_write_1h": null,
   "cost_usd": 0.342488,
   "wall_s": 231.2,
   "turns": null,
   "prompt_last": null,
   "lines_added": 272,
   "lines_deleted": 15,
   "lines": 287,
   "files": 9,
   "thinking": 1623,
   "commands": 9,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "replaced_by": 13,
   "tests_edited": 0,
   "score": 17,
   "of": 17,
   "api_cost": 0.342488
  },
  {
   "run": "gpt-6-sol__control__B2__r3__39a2a5",
   "model": "gpt-6-sol",
   "host": "codex",
   "task": "B2",
   "skill": "control",
   "rep": 3,
   "round": 12,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 7491,
   "input": 417216,
   "cache_read": 390528,
   "cache_write": 38672,
   "warmup": 11984,
   "cache_write_1h": null,
   "cost_usd": 0.23036,
   "wall_s": 167.0,
   "turns": null,
   "prompt_last": null,
   "lines_added": 240,
   "lines_deleted": 14,
   "lines": 254,
   "files": 9,
   "thinking": 1336,
   "commands": 30,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "replaced_by": 13,
   "tests_edited": 0,
   "score": 17,
   "of": 17,
   "api_cost": 0.23036
  },
  {
   "run": "gpt-6.1-sol__control__B1__r1__4abb21",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "B1",
   "skill": "control",
   "rep": 1,
   "round": 11,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 10895,
   "input": 393441,
   "cache_read": 352256,
   "cache_write": 53682,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.25154,
   "wall_s": 382.8,
   "turns": null,
   "prompt_last": null,
   "lines_added": 367,
   "lines_deleted": 37,
   "lines": 404,
   "files": 15,
   "thinking": 1162,
   "commands": 14,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "replaced_by": 13,
   "tests_edited": 0,
   "score": 20,
   "of": 20,
   "api_cost": 0.25154
  },
  {
   "run": "gpt-6.1-sol__control__B1__r2__f035c8",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "B1",
   "skill": "control",
   "rep": 2,
   "round": 11,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 9591,
   "input": 351044,
   "cache_read": 310400,
   "cache_write": 53141,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.233232,
   "wall_s": 340.4,
   "turns": null,
   "prompt_last": null,
   "lines_added": 311,
   "lines_deleted": 39,
   "lines": 350,
   "files": 15,
   "thinking": 588,
   "commands": 14,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "replaced_by": 13,
   "tests_edited": 0,
   "score": 20,
   "of": 20,
   "api_cost": 0.233232
  },
  {
   "run": "gpt-6.1-sol__control__B1__r3__3778f8",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "B1",
   "skill": "control",
   "rep": 3,
   "round": 11,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 16473,
   "input": 417669,
   "cache_read": 373120,
   "cache_write": 57046,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.316134,
   "wall_s": 565.9,
   "turns": null,
   "prompt_last": null,
   "lines_added": 463,
   "lines_deleted": 47,
   "lines": 510,
   "files": 17,
   "thinking": 2816,
   "commands": 14,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "replaced_by": 13,
   "tests_edited": 0,
   "score": 20,
   "of": 20,
   "api_cost": 0.316134
  },
  {
   "run": "gpt-6.1-sol__control__B2__r1__9f5edd",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "B2",
   "skill": "control",
   "rep": 1,
   "round": 12,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 10363,
   "input": 355992,
   "cache_read": 318592,
   "cache_write": 49897,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.235283,
   "wall_s": 365.8,
   "turns": null,
   "prompt_last": null,
   "lines_added": 434,
   "lines_deleted": 16,
   "lines": 450,
   "files": 12,
   "thinking": 535,
   "commands": 10,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "replaced_by": 13,
   "tests_edited": 0,
   "score": 17,
   "of": 17,
   "api_cost": 0.235283
  },
  {
   "run": "gpt-6.1-sol__control__B2__r2__197bff",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "B2",
   "skill": "control",
   "rep": 2,
   "round": 12,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 7638,
   "input": 273320,
   "cache_read": 243968,
   "cache_write": 41849,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.184475,
   "wall_s": 271.4,
   "turns": null,
   "prompt_last": null,
   "lines_added": 335,
   "lines_deleted": 19,
   "lines": 354,
   "files": 11,
   "thinking": 285,
   "commands": 11,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "replaced_by": 13,
   "tests_edited": 0,
   "score": 17,
   "of": 17,
   "api_cost": 0.184475
  },
  {
   "run": "gpt-6.1-sol__control__B2__r3__dcfb29",
   "model": "gpt-6.1-sol",
   "host": "codex",
   "task": "B2",
   "skill": "control",
   "rep": 3,
   "round": 12,
   "date": "2026-09-29",
   "exit": 0,
   "passed": true,
   "output": 7231,
   "input": 203655,
   "cache_read": 178048,
   "cache_write": 38104,
   "warmup": 12497,
   "cache_write_1h": null,
   "cost_usd": 0.166323,
   "wall_s": 251.6,
   "turns": null,
   "prompt_last": null,
   "lines_added": 326,
   "lines_deleted": 19,
   "lines": 345,
   "files": 12,
   "thinking": 225,
   "commands": 8,
   "effort": "medium",
   "cli": "codex-cli 0.159.0",
   "replaced_by": 13,
   "tests_edited": 0,
   "score": 17,
   "of": 17,
   "api_cost": 0.166323
  }
 ]
}
