fix(hooks): dedupe transcript usage by message.id in cost-tracker (~2.5-3x inflation) (#2483)

Claude Code writes one transcript JSONL line per content block, so a
single API response (one message.id) spans multiple assistant lines that
each repeat the same message.usage. sumUsageFromTranscript summed every
line, inflating token totals and estimated_cost_usd roughly 2.5-3x.

Verified on a real session: 704 assistant lines but only 286 unique
message.ids (2.46 lines/response on average); line-summing reported
$866.52 while the deduped total is $332.62. Usage payloads are identical
across lines of the same id (0/286 varied), so counting once per id is
equivalent to taking the last line per id.

Fix: collect usage into a Map keyed by message.id (last line wins) and
sum unique entries. Lines without a message.id (older transcript shapes)
keep the previous per-line behavior via a synthetic key, so existing
tests and old transcripts are unaffected.

Adds a regression test: a response split into 3 content-block lines with
the same message.id is counted exactly once.

Note: rows already written to ~/.claude/metrics/costs.jsonl by the old
code carry inflated token counts and estimates (except rows whose cost
came from the harness-cost cache, where cost is authoritative but token
counts are still inflated). Downstream consumers may want to annotate
history; this change intentionally does not rewrite the raw log.

Co-authored-by: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
AlbertChiu777 2026-07-28 02:14:47 +08:00 committed by GitHub
parent ecb45c1764
commit 536221cf7a
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
2 changed files with 61 additions and 7 deletions

View file

@ -92,6 +92,13 @@ function toNumber(v) {
* Scan the session JSONL and sum token usage across all assistant turns.
* Returns { inputTokens, outputTokens, cacheWriteTokens, cacheReadTokens, model }
* or null on read failure.
*
* Claude Code writes one JSONL line per content block, so a single API
* response (one message.id) spans multiple assistant lines that each repeat
* the same message.usage. Summing every line inflates totals ~2.5-3x
* (verified: a session with 704 assistant lines had only 286 unique
* message.ids $867 line-summed vs $333 deduped). Usage is therefore
* counted once per message.id, keeping the last line seen for each id.
*/
function sumUsageFromTranscript(transcriptPath) {
let content;
@ -101,10 +108,8 @@ function sumUsageFromTranscript(transcriptPath) {
return null;
}
let inputTokens = 0;
let outputTokens = 0;
let cacheWriteTokens = 0;
let cacheReadTokens = 0;
const usageById = new Map();
let syntheticKey = 0;
let model = 'unknown';
for (const line of content.split('\n')) {
@ -116,13 +121,26 @@ function sumUsageFromTranscript(transcriptPath) {
const msg = entry.message;
if (!msg || !msg.usage) continue;
const u = msg.usage;
// Lines without a message.id (older transcript shapes) keep the previous
// per-line behavior via a synthetic key.
const key = (typeof msg.id === 'string' && msg.id)
? msg.id
: `__line_${++syntheticKey}`;
usageById.set(key, msg.usage);
if (msg.model && msg.model !== 'unknown') model = msg.model;
}
let inputTokens = 0;
let outputTokens = 0;
let cacheWriteTokens = 0;
let cacheReadTokens = 0;
for (const u of usageById.values()) {
inputTokens += toNumber(u.input_tokens);
outputTokens += toNumber(u.output_tokens);
cacheWriteTokens += toNumber(u.cache_creation_input_tokens);
cacheReadTokens += toNumber(u.cache_read_input_tokens);
if (msg.model && msg.model !== 'unknown') model = msg.model;
}
return { inputTokens, outputTokens, cacheWriteTokens, cacheReadTokens, model };