mirror of
https://github.com/NanmiCoder/claude-code-haha.git
synced 2026-10-10 20:03:13 +08:00
09120df85a
Claude Code writes an assistant message as one JSONL line per content block — thinking, text, and each tool_use separately — and every one of those lines repeats the same complete `usage` object. Both stats paths summed them, so a reply cost as many times as it had blocks. One real message in this repo's own transcripts spans 52 lines carrying 50.8K tokens, and was counted as 2.65M. Scanning ~/.claude/projects with 1927 sources: the total drops from 9.76B to 4.59B tokens, and the busiest day from 2.60B to 1.46B. 39,290 of 69,306 usage lines were repeats. Usage records now deduplicate on (message.id, requestId), scoped per transcript, and survive an incremental read. Lines with no message id are still counted, matching ccusage. Two paths compute these stats — the local index reducer and the direct scan in stats.ts — and a parity test pins them to identical output. Both carried the bug, so the rules that decide what counts now live in one module, `utils/usageAccounting.ts`, rather than being written twice and drifting. Also fixed, all surfaced while verifying the above: Cost was dead code. `costUSD` and `webSearchRequests` were initialized to 0 in the reducer and never assigned, so 0 travelled to SQLite and out again. Now estimated from the rates in modelCost.ts — but with an unknown model returning null instead of falling back to the default model's rates, because 12.7% of the tokens here come from third-party providers (k3, glm, MiniMax, deepseek, grok) that would otherwise be billed at Claude prices. Those models keep their tokens in the activity totals and are named in `unpricedModels` so the UI can say the dollar figure is a floor. `MODEL_COSTS` has no entry for claude-opus-5 — the canonical-name resolver maps it to `claude-opus`, which isn't a key either — so the largest model by usage priced through the unknown-model fallback. The new module handles it explicitly. The CLI's own calculateUSDCost still takes that fallback; left alone here, it reaches well beyond activity stats. "Longest task" measured `last - first`, a calendar span rather than a duration: a session resumed the next morning reported the whole night as time on task. The panel read 436 hours for a 17-message session, and the worst case on this machine was 1137 hours. Sessions now accumulate working time across gaps under 30 minutes, stored in a new `active_duration_ms` column. Workflow subagent transcripts were never indexed. Discovery read one level of `subagents/`, but workflow agents nest a group deeper at `subagents/workflows/<id>/`; 79 files here were invisible to both the index and the direct scan, which additionally attributed them to a session named "workflows". Ties in "longest session" now break on session id. The two paths iterate sessions in different orders, and ties became reachable once an out-of-order session scored 0 rather than a distinct negative span. The parser version bump is what makes any of this reach an existing install: `detectSourceChange` rebuilds a source only when its stored version differs.
109 lines
4.0 KiB
TypeScript
109 lines
4.0 KiB
TypeScript
import { describe, expect, it } from 'bun:test'
|
|
import { estimateCostUSD, resolveModelCosts } from './usageAccounting.js'
|
|
|
|
const ONE_MILLION = 1_000_000
|
|
|
|
function tokens(overrides: Partial<Parameters<typeof estimateCostUSD>[1]> = {}) {
|
|
return {
|
|
inputTokens: 0,
|
|
outputTokens: 0,
|
|
cacheReadInputTokens: 0,
|
|
cacheCreationInputTokens: 0,
|
|
...overrides,
|
|
}
|
|
}
|
|
|
|
describe('resolveModelCosts', () => {
|
|
it('prices every Claude family the app can run', () => {
|
|
expect(resolveModelCosts('claude-opus-5')).toMatchObject({
|
|
inputTokens: 5,
|
|
outputTokens: 25,
|
|
promptCacheReadTokens: 0.5,
|
|
promptCacheWriteTokens: 6.25,
|
|
})
|
|
expect(resolveModelCosts('claude-opus-4-8')?.inputTokens).toBe(5)
|
|
expect(resolveModelCosts('claude-opus-4-1')?.inputTokens).toBe(15)
|
|
expect(resolveModelCosts('claude-fable-5')?.outputTokens).toBe(50)
|
|
expect(resolveModelCosts('claude-sonnet-5')?.outputTokens).toBe(15)
|
|
expect(resolveModelCosts('claude-haiku-4-5')?.outputTokens).toBe(5)
|
|
})
|
|
|
|
it('sees through the decorations gateways and dated snapshots add', () => {
|
|
const opus = resolveModelCosts('claude-opus-4-8')
|
|
expect(resolveModelCosts('claude-opus-4-8-r')).toEqual(opus!)
|
|
expect(resolveModelCosts('CLAUDE-OPUS-4-8')).toEqual(opus!)
|
|
expect(resolveModelCosts('anthropic/claude-opus-4-8')).toEqual(opus!)
|
|
expect(resolveModelCosts('claude-haiku-4-5-20251001')?.outputTokens).toBe(5)
|
|
})
|
|
|
|
it('picks the longest matching prefix rather than the first', () => {
|
|
// `claude-opus-4-1` bills at the old Opus tier; a bare `claude-opus-4` must not swallow it.
|
|
expect(resolveModelCosts('claude-opus-4-1')?.inputTokens).toBe(15)
|
|
expect(resolveModelCosts('claude-opus-4-5')?.inputTokens).toBe(5)
|
|
})
|
|
|
|
it('returns null for third-party models instead of guessing Claude rates', () => {
|
|
for (const model of [
|
|
'k3',
|
|
'glm-5.2',
|
|
'MiniMax-M3',
|
|
'deepseek-v4-flash',
|
|
'kimi-k2.7-code',
|
|
'gpt-5.6-sol',
|
|
'grok-4.5',
|
|
'google/gemini-3.6-flash',
|
|
'doubao-seed-2.0-code',
|
|
'<synthetic>',
|
|
'',
|
|
' ',
|
|
]) {
|
|
expect(resolveModelCosts(model)).toBeNull()
|
|
}
|
|
})
|
|
|
|
it('bills fast mode at its own rate only where fast mode exists', () => {
|
|
expect(resolveModelCosts('claude-opus-5', 'fast')?.inputTokens).toBe(10)
|
|
expect(resolveModelCosts('claude-opus-5', 'standard')?.inputTokens).toBe(5)
|
|
// Sonnet has no fast mode — a stray `speed` must not change what it costs.
|
|
expect(resolveModelCosts('claude-sonnet-5', 'fast')?.inputTokens).toBe(3)
|
|
})
|
|
})
|
|
|
|
describe('estimateCostUSD', () => {
|
|
it('bills each token bucket at its own rate', () => {
|
|
const cost = estimateCostUSD('claude-opus-5', tokens({
|
|
inputTokens: ONE_MILLION,
|
|
outputTokens: ONE_MILLION,
|
|
cacheReadInputTokens: ONE_MILLION,
|
|
cacheCreationInputTokens: ONE_MILLION,
|
|
}))
|
|
// 5 input + 25 output + 0.50 cache read + 6.25 cache write
|
|
expect(cost).toBeCloseTo(36.75, 10)
|
|
})
|
|
|
|
it('prices cache reads at a tenth of input, which is why token totals overstate spend', () => {
|
|
const cacheRead = estimateCostUSD('claude-opus-5', tokens({ cacheReadInputTokens: ONE_MILLION }))
|
|
const input = estimateCostUSD('claude-opus-5', tokens({ inputTokens: ONE_MILLION }))
|
|
expect(cacheRead).toBeCloseTo(input! / 10, 10)
|
|
})
|
|
|
|
it('charges per web search request', () => {
|
|
expect(estimateCostUSD('claude-opus-5', { ...tokens(), webSearchRequests: 100 }))
|
|
.toBeCloseTo(1, 10)
|
|
})
|
|
|
|
it('returns null — never zero — for an unpriceable model', () => {
|
|
const cost = estimateCostUSD('glm-5.2', tokens({
|
|
inputTokens: ONE_MILLION,
|
|
outputTokens: ONE_MILLION,
|
|
}))
|
|
// A zero here would silently understate spend for anyone on third-party providers; callers
|
|
// must be able to tell "no rates published" apart from "this turn was free".
|
|
expect(cost).toBeNull()
|
|
})
|
|
|
|
it('costs nothing for a zeroed usage record on a known model', () => {
|
|
expect(estimateCostUSD('claude-opus-5', tokens())).toBe(0)
|
|
})
|
|
})
|