From 3a419f972972094546d7ed0eebab854b4e7f0f5d Mon Sep 17 00:00:00 2001 From: JSONbored <49853598+JSONbored@users.noreply.github.com> Date: Fri, 31 Jul 2026 07:59:49 -0700 Subject: [PATCH 1/2] fix(miner): count the prompt-cache tiers as input tokens in the engine's own parsers #10251 fixed this on the ORB side. Both of the engine's deliberately-parallel copies were left reading only `input_tokens`, so the miner still records a near-constant handful for every cached claude attempt. Anthropic, and therefore Claude Code and the Agent SDK, splits one prompt across three counters: input_tokens carries only the portion neither read from nor written to the prompt cache, with the rest in cache_read_input_tokens and cache_creation_input_tokens. With caching active -- which it is for every attempt the CLI runs -- essentially the whole prompt lands in the latter two. Sum the three, matching the ORB's totalInputTokens exactly, so the two parsers cannot report different numbers for the same envelope. Each tier keeps its own alias group because the three are additive components of one prompt, not names for one value; folding them into INPUT_TOKEN_KEYS would take the maximum and under-report again, just less severely. Absence stays absence: an envelope with no input counter at all yields undefined rather than a fabricated 0, and a tier present but zero contributes a real zero. Providers emitting no cache keys -- codex and the OpenAI-compatible bindings -- are byte-identical to before, which is what makes the shared extraction point safe to change. Closes #10246 --- .../src/miner/agent-sdk-driver.ts | 13 ++++++- .../src/miner/cli-subprocess-driver.ts | 26 ++++++++++++- .../test/agent-sdk-driver.test.ts | 38 ++++++++++++++++++ test/unit/agent-sdk-driver.test.ts | 39 +++++++++++++++++++ test/unit/cli-subprocess-driver.test.ts | 38 ++++++++++++++++++ 5 files changed, 152 insertions(+), 2 deletions(-) diff --git a/packages/loopover-engine/src/miner/agent-sdk-driver.ts b/packages/loopover-engine/src/miner/agent-sdk-driver.ts index c42a5b5b53..9cf9dde7be 100644 --- a/packages/loopover-engine/src/miner/agent-sdk-driver.ts +++ b/packages/loopover-engine/src/miner/agent-sdk-driver.ts @@ -101,7 +101,18 @@ function finiteNonNegativeNumber(value: unknown): number | undefined { * aggregate. */ function tokensFromResultMessage(resultMessage: Record | null): CodingAgentTokenUsage { const usage = asRecord(resultMessage?.usage); - const inputTokens = finiteNonNegativeNumber(usage?.input_tokens); + // #10246: Anthropic splits one prompt across three counters -- `input_tokens` carries only the portion + // neither read from nor written to the prompt cache. Summed, matching the ORB side's own totalInputTokens + // (#10251), so both parsers report the same thing for the same envelope. Absence stays absence; a tier that + // is present but zero contributes a real zero. + const inputTiers = [ + finiteNonNegativeNumber(usage?.input_tokens), + finiteNonNegativeNumber(usage?.cache_read_input_tokens), + finiteNonNegativeNumber(usage?.cache_creation_input_tokens), + ]; + const inputTokens = inputTiers.every((tier) => tier === undefined) + ? undefined + : inputTiers.reduce((sum, tier) => sum + (tier ?? 0), 0); const outputTokens = finiteNonNegativeNumber(usage?.output_tokens); if (inputTokens === undefined && outputTokens === undefined) return {}; return { diff --git a/packages/loopover-engine/src/miner/cli-subprocess-driver.ts b/packages/loopover-engine/src/miner/cli-subprocess-driver.ts index 4508e54ba2..c2189f330d 100644 --- a/packages/loopover-engine/src/miner/cli-subprocess-driver.ts +++ b/packages/loopover-engine/src/miner/cli-subprocess-driver.ts @@ -140,7 +140,19 @@ function resolveDefaultBuildArgs(command: string): (task: CodingAgentDriverTask) * key spellings are tolerated, and usage/token_usage/tokenUsage/usage_metadata sub-objects are all checked, same * as src/selfhost/ai.ts). A missing/malformed field means "no signal", never an error -- never fabricated. */ const COST_KEYS = ["total_cost_usd", "totalCostUsd", "cost_usd", "costUsd"] as const; +// ALIASES of one value -- different providers' names for the same number. The uncached portion of the prompt +// only; see the two cache tiers directly below. const INPUT_TOKEN_KEYS = ["input_tokens", "inputTokens", "prompt_tokens", "promptTokens"] as const; +/** #10246: the same three-counter split #10251 fixed on the ORB side, which left this deliberately-parallel + * copy behind. Anthropic (and therefore the claude CLI) puts only the neither-read-nor-written portion of a + * prompt in `input_tokens`; with caching active -- which it is for every attempt the CLI runs -- essentially + * the whole prompt lands in these two instead, and reading the first alone degenerates to a near-constant. + * + * Each tier is its OWN alias group because the three are ADDITIVE COMPONENTS of one prompt, not names for one + * value: they are summed with each other and max'd only within a group. Folding them into INPUT_TOKEN_KEYS + * would take the maximum of the three and under-report again, just less severely. */ +const CACHE_READ_INPUT_TOKEN_KEYS = ["cache_read_input_tokens", "cacheReadInputTokens"] as const; +const CACHE_CREATION_INPUT_TOKEN_KEYS = ["cache_creation_input_tokens", "cacheCreationInputTokens"] as const; const OUTPUT_TOKEN_KEYS = ["output_tokens", "outputTokens", "completion_tokens", "completionTokens"] as const; const TOTAL_TOKEN_KEYS = ["total_tokens", "totalTokens"] as const; @@ -164,6 +176,18 @@ function asPlainRecord(value: unknown): Record | null { return value && typeof value === "object" && !Array.isArray(value) ? (value as Record) : null; } +/** Sum the three input tiers (#10246). Absence stays absence: an envelope reporting no input counter at all + * yields undefined rather than a fabricated 0, while a tier that is present but zero contributes a real zero. */ +function totalInputTokens(entry: Record): number | undefined { + const tiers = [ + maxNumber(entry, INPUT_TOKEN_KEYS), + maxNumber(entry, CACHE_READ_INPUT_TOKEN_KEYS), + maxNumber(entry, CACHE_CREATION_INPUT_TOKEN_KEYS), + ]; + if (tiers.every((tier) => tier === undefined)) return undefined; + return tiers.reduce((sum, tier) => sum + (tier ?? 0), 0); +} + function mergeCliUsage(out: CliUsage, record: Record): void { const nested = [ record, @@ -176,7 +200,7 @@ function mergeCliUsage(out: CliUsage, record: Record): void { for (const entry of nested) { const costUsd = maxNumber(entry, COST_KEYS); if (costUsd !== undefined) out.costUsd = Math.max(out.costUsd ?? 0, costUsd); - const inputTokens = maxNumber(entry, INPUT_TOKEN_KEYS); + const inputTokens = totalInputTokens(entry); if (inputTokens !== undefined) out.inputTokens = Math.max(out.inputTokens ?? 0, inputTokens); const outputTokens = maxNumber(entry, OUTPUT_TOKEN_KEYS); if (outputTokens !== undefined) out.outputTokens = Math.max(out.outputTokens ?? 0, outputTokens); diff --git a/packages/loopover-engine/test/agent-sdk-driver.test.ts b/packages/loopover-engine/test/agent-sdk-driver.test.ts index da692d6d5b..f13bca40d0 100644 --- a/packages/loopover-engine/test/agent-sdk-driver.test.ts +++ b/packages/loopover-engine/test/agent-sdk-driver.test.ts @@ -473,3 +473,41 @@ test("a side the provider did not report stays ABSENT rather than being zeroed ( assert.equal(noUsageResult.inputTokens, undefined); assert.equal(noUsageResult.outputTokens, undefined); }); + +// #10246: the miner half of the three-counter prompt split #10251 fixed on the ORB side. +test("counts the prompt-cache tiers as input tokens (#10246)", async () => { + const driver = driverWith({ + query: queryYielding([ + { + type: "result", + subtype: "success", + is_error: false, + num_turns: 2, + result: "done", + usage: { input_tokens: 2, output_tokens: 787, cache_read_input_tokens: 48210, cache_creation_input_tokens: 1536 }, + }, + ]), + }); + + const result = await driver.run(task); + + assert.equal(result.inputTokens, 49748); + assert.equal(result.outputTokens, 787); + assert.equal(result.tokensUsed, 50535); +}); + +test("keeps a genuinely-zero input tier, and reports no input counter at all as undefined (#10246)", async () => { + const cacheOnly = driverWith({ + query: queryYielding([ + { type: "result", subtype: "success", is_error: false, num_turns: 1, result: "done", usage: { input_tokens: 0, cache_read_input_tokens: 900 } }, + ]), + }); + assert.equal((await cacheOnly.run(task)).inputTokens, 900); + + const outputOnly = driverWith({ + query: queryYielding([ + { type: "result", subtype: "success", is_error: false, num_turns: 1, result: "done", usage: { output_tokens: 50 } }, + ]), + }); + assert.equal((await outputOnly.run(task)).inputTokens, undefined); +}); diff --git a/test/unit/agent-sdk-driver.test.ts b/test/unit/agent-sdk-driver.test.ts index 2d52550714..bbb4eb058e 100644 --- a/test/unit/agent-sdk-driver.test.ts +++ b/test/unit/agent-sdk-driver.test.ts @@ -192,6 +192,45 @@ describe("createAgentSdkCodingAgentDriver", () => { expect(result.tokensUsed).toBe(1234); }); + it("counts the prompt-cache tiers as input tokens (#10246)", async () => { + // Matches the ORB side's totalInputTokens (#10251) so both parsers report the same thing for the same + // envelope -- the two are deliberately parallel copies and must not diverge on arithmetic. + const driver = driverWith({ + query: queryYielding([ + { + type: "result", + subtype: "success", + is_error: false, + num_turns: 2, + result: "done", + usage: { input_tokens: 2, output_tokens: 787, cache_read_input_tokens: 48210, cache_creation_input_tokens: 1536 }, + }, + ]), + }); + + const result = await driver.run(task); + + expect(result.inputTokens).toBe(49748); + expect(result.outputTokens).toBe(787); + expect(result.tokensUsed).toBe(50535); + }); + + it("keeps a genuinely-zero tier, and reports no input at all as undefined (#10246)", async () => { + const cacheOnly = driverWith({ + query: queryYielding([ + { type: "result", subtype: "success", is_error: false, num_turns: 1, result: "done", usage: { input_tokens: 0, cache_read_input_tokens: 900 } }, + ]), + }); + expect((await cacheOnly.run(task)).inputTokens).toBe(900); + + const outputOnly = driverWith({ + query: queryYielding([ + { type: "result", subtype: "success", is_error: false, num_turns: 1, result: "done", usage: { output_tokens: 50 } }, + ]), + }); + expect((await outputOnly.run(task)).inputTokens).toBeUndefined(); + }); + it("still reports real tokens on a non-success subtype -- the session was billed either way, same as costUsd", async () => { const driver = driverWith({ query: queryYielding([ diff --git a/test/unit/cli-subprocess-driver.test.ts b/test/unit/cli-subprocess-driver.test.ts index c0d5a14eb8..d94a144aeb 100644 --- a/test/unit/cli-subprocess-driver.test.ts +++ b/test/unit/cli-subprocess-driver.test.ts @@ -575,6 +575,44 @@ describe("createCliSubprocessCodingAgentDriver (#4266)", () => { expect(result.outputTokens).toBeUndefined(); }); + // #10246: the miner half of the same three-counter split #10251 fixed on the ORB. With caching active -- + // every attempt the CLI runs -- reading input_tokens alone degenerates to a near-constant handful. + it("counts the prompt-cache tiers as input tokens (#10246)", async () => { + const { spawn } = fakeSpawn({ + stdout: JSON.stringify({ + type: "result", + usage: { input_tokens: 2, output_tokens: 787, cache_read_input_tokens: 48210, cache_creation_input_tokens: 1536 }, + }), + code: 0, + }); + const driver = createCliSubprocessCodingAgentDriver({ command: "claude", spawn }); + const result = await driver.run(TASK); + // 2 + 48210 + 1536 -- the three tiers are additive components of ONE prompt, not aliases. + expect(result.inputTokens).toBe(49748); + expect(result.outputTokens).toBe(787); + expect(result.tokensUsed).toBe(50535); + }); + + it("leaves input absent when NO input counter is reported, but keeps a genuinely-zero tier (#10246)", async () => { + const { none } = { none: fakeSpawn({ stdout: JSON.stringify({ output_tokens: 50 }), code: 0 }) }; + const noInput = createCliSubprocessCodingAgentDriver({ command: "claude", spawn: none.spawn }); + expect((await noInput.run(TASK)).inputTokens).toBeUndefined(); + + const { spawn } = fakeSpawn({ stdout: JSON.stringify({ usage: { input_tokens: 0, cache_read_input_tokens: 900 } }), code: 0 }); + const zeroTier = createCliSubprocessCodingAgentDriver({ command: "claude", spawn }); + expect((await zeroTier.run(TASK)).inputTokens).toBe(900); + }); + + it("is byte-identical for a provider that emits no cache keys at all (#10246)", async () => { + // codex and the OpenAI-compatible bindings never populate them -- this is what makes the shared + // extraction point safe to change. + const { spawn } = fakeSpawn({ stdout: JSON.stringify({ prompt_tokens: 2706, completion_tokens: 544 }), code: 0 }); + const driver = createCliSubprocessCodingAgentDriver({ command: "codex", spawn }); + const result = await driver.run(TASK); + expect(result.inputTokens).toBe(2706); + expect(result.outputTokens).toBe(544); + }); + it("sums claude's top-level input_tokens + output_tokens from its single JSON result on success", async () => { const { spawn } = fakeSpawn({ stdout: JSON.stringify({ type: "result", subtype: "success", result: "done", input_tokens: 1000, output_tokens: 234 }), From 82b01ad4a4c340ed2f90047a4aff10367d58a87f Mon Sep 17 00:00:00 2001 From: JSONbored Date: Fri, 31 Jul 2026 16:22:05 -0700 Subject: [PATCH 2/2] test(engine): cover the cli driver's input-tier sum in the engine's own suite codecov/patch failed at 83.78% because totalInputTokens' body and its mergeCliUsage call site had zero hits in the engine flag's c8 report: the driver's behavior tests live entirely in the root vitest copy, which the engine flag never sees. Port the tier-sum scenarios into packages/loopover-engine/test/, the suite that actually carries this file's coverage. Verified locally: every changed line in both drivers now has hits in the engine lcov. --- .../test/cli-subprocess-driver.test.ts | 79 +++++++++++++++++++ 1 file changed, 79 insertions(+) create mode 100644 packages/loopover-engine/test/cli-subprocess-driver.test.ts diff --git a/packages/loopover-engine/test/cli-subprocess-driver.test.ts b/packages/loopover-engine/test/cli-subprocess-driver.test.ts new file mode 100644 index 0000000000..de6386991a --- /dev/null +++ b/packages/loopover-engine/test/cli-subprocess-driver.test.ts @@ -0,0 +1,79 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { + createCliSubprocessCodingAgentDriver, + type CliSubprocessSpawnFn, + type CodingAgentDriverTask, +} from "../dist/index.js"; + +// #10246: this driver had NO test in the engine's own suite -- its behavior coverage lived entirely in +// the root vitest copy (test/unit/cli-subprocess-driver.test.ts). That suite is invisible to the +// `engine` Codecov flag (c8 over dist-test), so the three-counter input-tier sum landed as uncovered +// changed lines and failed codecov/patch. These are the tier-sum scenarios from the vitest copy, ported +// here so the suite that carries this file's coverage actually executes the new path. The vitest copy +// stays -- the two suites are deliberately parallel (see agent-sdk-driver's pairing). + +const task: CodingAgentDriverTask = { + attemptId: "attempt-7", + workingDirectory: "/tmp/worktrees/attempt-7", + acceptanceCriteriaPath: "/tmp/worktrees/attempt-7/ACCEPTANCE-CRITERIA.md", + instructions: "Apply the fix described in ACCEPTANCE-CRITERIA.md.", + maxTurns: 6, +}; + +function spawnPrinting(stdout: string): CliSubprocessSpawnFn { + return async () => ({ stdout, code: 0 }); +} + +test("counts the prompt-cache tiers as input tokens (#10246)", async () => { + const driver = createCliSubprocessCodingAgentDriver({ + command: "claude", + spawn: spawnPrinting( + JSON.stringify({ + type: "result", + usage: { input_tokens: 2, output_tokens: 787, cache_read_input_tokens: 48210, cache_creation_input_tokens: 1536 }, + }), + ), + }); + const result = await driver.run(task); + // 2 + 48210 + 1536 -- the tiers are additive components of ONE prompt, not aliases of one value. + assert.equal(result.inputTokens, 49748); + assert.equal(result.outputTokens, 787); + assert.equal(result.tokensUsed, 50535); +}); + +test("leaves input absent when NO input counter is reported, but keeps a genuinely-zero tier (#10246)", async () => { + const noInput = createCliSubprocessCodingAgentDriver({ + command: "claude", + spawn: spawnPrinting(JSON.stringify({ output_tokens: 50 })), + }); + assert.equal((await noInput.run(task)).inputTokens, undefined); + + const zeroTier = createCliSubprocessCodingAgentDriver({ + command: "claude", + spawn: spawnPrinting(JSON.stringify({ usage: { input_tokens: 0, cache_read_input_tokens: 900 } })), + }); + assert.equal((await zeroTier.run(task)).inputTokens, 900); +}); + +test("is byte-identical for a provider that emits no cache keys at all (#10246)", async () => { + const driver = createCliSubprocessCodingAgentDriver({ + command: "codex", + spawn: spawnPrinting(JSON.stringify({ prompt_tokens: 2706, completion_tokens: 544 })), + }); + const result = await driver.run(task); + assert.equal(result.inputTokens, 2706); + assert.equal(result.outputTokens, 544); +}); + +test("tolerates camelCase cache-tier spellings across a JSONL stream (#10246)", async () => { + const driver = createCliSubprocessCodingAgentDriver({ + command: "codex", + spawn: spawnPrinting( + '{"type":"start"}\n{"tokenUsage":{"inputTokens":50,"cacheReadInputTokens":200,"cacheCreationInputTokens":30,"outputTokens":25}}\n{"type":"end"}', + ), + }); + const result = await driver.run(task); + assert.equal(result.inputTokens, 280); + assert.equal(result.tokensUsed, 305); +});