Files
claude-code-haha/src/utils/usageAccounting.test.ts
T
程序员阿江(Relakkes) 09120df85a fix(stats): count one assistant reply once instead of once per content block
Claude Code writes an assistant message as one JSONL line per content
block — thinking, text, and each tool_use separately — and every one of
those lines repeats the same complete `usage` object. Both stats paths
summed them, so a reply cost as many times as it had blocks. One real
message in this repo's own transcripts spans 52 lines carrying 50.8K
tokens, and was counted as 2.65M.

Scanning ~/.claude/projects with 1927 sources: the total drops from
9.76B to 4.59B tokens, and the busiest day from 2.60B to 1.46B. 39,290
of 69,306 usage lines were repeats.

Usage records now deduplicate on (message.id, requestId), scoped per
transcript, and survive an incremental read. Lines with no message id
are still counted, matching ccusage.

Two paths compute these stats — the local index reducer and the direct
scan in stats.ts — and a parity test pins them to identical output. Both
carried the bug, so the rules that decide what counts now live in one
module, `utils/usageAccounting.ts`, rather than being written twice and
drifting.

Also fixed, all surfaced while verifying the above:

Cost was dead code. `costUSD` and `webSearchRequests` were initialized
to 0 in the reducer and never assigned, so 0 travelled to SQLite and out
again. Now estimated from the rates in modelCost.ts — but with an
unknown model returning null instead of falling back to the default
model's rates, because 12.7% of the tokens here come from third-party
providers (k3, glm, MiniMax, deepseek, grok) that would otherwise be
billed at Claude prices. Those models keep their tokens in the activity
totals and are named in `unpricedModels` so the UI can say the dollar
figure is a floor.

`MODEL_COSTS` has no entry for claude-opus-5 — the canonical-name
resolver maps it to `claude-opus`, which isn't a key either — so the
largest model by usage priced through the unknown-model fallback. The
new module handles it explicitly. The CLI's own calculateUSDCost still
takes that fallback; left alone here, it reaches well beyond activity
stats.

"Longest task" measured `last - first`, a calendar span rather than a
duration: a session resumed the next morning reported the whole night as
time on task. The panel read 436 hours for a 17-message session, and the
worst case on this machine was 1137 hours. Sessions now accumulate
working time across gaps under 30 minutes, stored in a new
`active_duration_ms` column.

Workflow subagent transcripts were never indexed. Discovery read one
level of `subagents/`, but workflow agents nest a group deeper at
`subagents/workflows/<id>/`; 79 files here were invisible to both the
index and the direct scan, which additionally attributed them to a
session named "workflows".

Ties in "longest session" now break on session id. The two paths iterate
sessions in different orders, and ties became reachable once an
out-of-order session scored 0 rather than a distinct negative span.

The parser version bump is what makes any of this reach an existing
install: `detectSourceChange` rebuilds a source only when its stored
version differs.
2026-07-27 03:35:12 +08:00

109 lines
4.0 KiB
TypeScript

import { describe, expect, it } from 'bun:test'
import { estimateCostUSD, resolveModelCosts } from './usageAccounting.js'
const ONE_MILLION = 1_000_000
function tokens(overrides: Partial<Parameters<typeof estimateCostUSD>[1]> = {}) {
return {
inputTokens: 0,
outputTokens: 0,
cacheReadInputTokens: 0,
cacheCreationInputTokens: 0,
...overrides,
}
}
describe('resolveModelCosts', () => {
it('prices every Claude family the app can run', () => {
expect(resolveModelCosts('claude-opus-5')).toMatchObject({
inputTokens: 5,
outputTokens: 25,
promptCacheReadTokens: 0.5,
promptCacheWriteTokens: 6.25,
})
expect(resolveModelCosts('claude-opus-4-8')?.inputTokens).toBe(5)
expect(resolveModelCosts('claude-opus-4-1')?.inputTokens).toBe(15)
expect(resolveModelCosts('claude-fable-5')?.outputTokens).toBe(50)
expect(resolveModelCosts('claude-sonnet-5')?.outputTokens).toBe(15)
expect(resolveModelCosts('claude-haiku-4-5')?.outputTokens).toBe(5)
})
it('sees through the decorations gateways and dated snapshots add', () => {
const opus = resolveModelCosts('claude-opus-4-8')
expect(resolveModelCosts('claude-opus-4-8-r')).toEqual(opus!)
expect(resolveModelCosts('CLAUDE-OPUS-4-8')).toEqual(opus!)
expect(resolveModelCosts('anthropic/claude-opus-4-8')).toEqual(opus!)
expect(resolveModelCosts('claude-haiku-4-5-20251001')?.outputTokens).toBe(5)
})
it('picks the longest matching prefix rather than the first', () => {
// `claude-opus-4-1` bills at the old Opus tier; a bare `claude-opus-4` must not swallow it.
expect(resolveModelCosts('claude-opus-4-1')?.inputTokens).toBe(15)
expect(resolveModelCosts('claude-opus-4-5')?.inputTokens).toBe(5)
})
it('returns null for third-party models instead of guessing Claude rates', () => {
for (const model of [
'k3',
'glm-5.2',
'MiniMax-M3',
'deepseek-v4-flash',
'kimi-k2.7-code',
'gpt-5.6-sol',
'grok-4.5',
'google/gemini-3.6-flash',
'doubao-seed-2.0-code',
'<synthetic>',
'',
' ',
]) {
expect(resolveModelCosts(model)).toBeNull()
}
})
it('bills fast mode at its own rate only where fast mode exists', () => {
expect(resolveModelCosts('claude-opus-5', 'fast')?.inputTokens).toBe(10)
expect(resolveModelCosts('claude-opus-5', 'standard')?.inputTokens).toBe(5)
// Sonnet has no fast mode — a stray `speed` must not change what it costs.
expect(resolveModelCosts('claude-sonnet-5', 'fast')?.inputTokens).toBe(3)
})
})
describe('estimateCostUSD', () => {
it('bills each token bucket at its own rate', () => {
const cost = estimateCostUSD('claude-opus-5', tokens({
inputTokens: ONE_MILLION,
outputTokens: ONE_MILLION,
cacheReadInputTokens: ONE_MILLION,
cacheCreationInputTokens: ONE_MILLION,
}))
// 5 input + 25 output + 0.50 cache read + 6.25 cache write
expect(cost).toBeCloseTo(36.75, 10)
})
it('prices cache reads at a tenth of input, which is why token totals overstate spend', () => {
const cacheRead = estimateCostUSD('claude-opus-5', tokens({ cacheReadInputTokens: ONE_MILLION }))
const input = estimateCostUSD('claude-opus-5', tokens({ inputTokens: ONE_MILLION }))
expect(cacheRead).toBeCloseTo(input! / 10, 10)
})
it('charges per web search request', () => {
expect(estimateCostUSD('claude-opus-5', { ...tokens(), webSearchRequests: 100 }))
.toBeCloseTo(1, 10)
})
it('returns null — never zero — for an unpriceable model', () => {
const cost = estimateCostUSD('glm-5.2', tokens({
inputTokens: ONE_MILLION,
outputTokens: ONE_MILLION,
}))
// A zero here would silently understate spend for anyone on third-party providers; callers
// must be able to tell "no rates published" apart from "this turn was free".
expect(cost).toBeNull()
})
it('costs nothing for a zeroed usage record on a known model', () => {
expect(estimateCostUSD('claude-opus-5', tokens())).toBe(0)
})
})