mirror of
https://github.com/NanmiCoder/claude-code-haha.git
synced 2026-10-10 11:53:10 +08:00
fix: recover oversized conversations using provider token limits
This commit is contained in:
@@ -137,6 +137,38 @@ afterEach(async () => {
|
||||
})
|
||||
|
||||
describe('session messages HTTP surface', () => {
|
||||
it('keeps a transcript above 20 MiB readable when checkpoint previews exceed their budget (#1373)', async () => {
|
||||
const sessionId = await seedSessionWithSubagent()
|
||||
const filePath = path.join(tmpDir, 'projects', '-tmp-http-invariant', `${sessionId}.jsonl`)
|
||||
const content = 'x'.repeat(8 * 1024)
|
||||
const count = 2700
|
||||
await fs.appendFile(filePath, Array.from({ length: count }, (_, n) => JSON.stringify({
|
||||
type: 'assistant', uuid: `issue-1373-${n}`, timestamp: '2026-01-02T00:00:00Z',
|
||||
message: { role: 'assistant', content },
|
||||
})).join('\n') + '\n')
|
||||
expect((await fs.stat(filePath)).size).toBeGreaterThan(20 * 1024 * 1024)
|
||||
|
||||
const checkpoint = await api('GET', `/api/sessions/${sessionId}/turn-checkpoints`)
|
||||
expect(checkpoint.status).toBe(413)
|
||||
expect((await checkpoint.json() as { error: string }).error).toBe('HISTORY_CHECKPOINT_PREVIEW_LIMIT')
|
||||
|
||||
for (const suffix of ['', '/messages', '/messages?mode=full']) {
|
||||
const response = await api('GET', `/api/sessions/${sessionId}${suffix}`)
|
||||
expect(response.status).toBe(200)
|
||||
const body = await response.json() as {
|
||||
messages: Array<{ id: string; content: unknown }>
|
||||
page: { historyComplete: boolean; hasMore: boolean }
|
||||
}
|
||||
expect(body.messages.at(-1)?.id).toBe(`issue-1373-${count - 1}`)
|
||||
if (suffix.endsWith('mode=full')) {
|
||||
expect(body.messages.filter(message => message.id.startsWith('issue-1373-'))).toHaveLength(count)
|
||||
expect(body.page.historyComplete).toBe(true)
|
||||
} else {
|
||||
expect(body.page.hasMore).toBe(true)
|
||||
}
|
||||
}
|
||||
})
|
||||
|
||||
it('bounds both public history endpoints and returns a continuation without canonical hydration', async () => {
|
||||
const sessionId = await seedSessionWithSubagent()
|
||||
const filePath = path.join(tmpDir, 'projects', '-tmp-http-invariant', `${sessionId}.jsonl`)
|
||||
|
||||
@@ -3,10 +3,12 @@ import { APIError } from '@anthropic-ai/sdk'
|
||||
import { BUSINESS_ERROR_CODES } from '../../constants/businessErrors.js'
|
||||
import {
|
||||
getAssistantMessageFromError,
|
||||
getPromptTooLongTokenGap,
|
||||
getImageUnsupportedErrorMessage,
|
||||
isContextOverflowErrorText,
|
||||
isUnsupportedImageInputErrorMessage,
|
||||
PROMPT_TOO_LONG_ERROR_MESSAGE,
|
||||
parsePromptTooLongTokenCounts,
|
||||
} from './errors.js'
|
||||
|
||||
describe('image unsupported API errors', () => {
|
||||
@@ -142,6 +144,50 @@ describe('image unsupported API errors', () => {
|
||||
})
|
||||
|
||||
describe('context overflow errors', () => {
|
||||
test('uses the full requested DeepSeek token count to recover an oversized session (#1373)', () => {
|
||||
const message = "This model's maximum context length is 1048576 tokens. However, you requested 3763011 tokens (3731011 in the messages, 32000 in the completion)."
|
||||
const error = new APIError(400, {
|
||||
error: { type: 'invalid_request_error', message },
|
||||
}, message, undefined)
|
||||
const assistant = getAssistantMessageFromError(error, 'deepseek-v4-flash')
|
||||
|
||||
expect(parsePromptTooLongTokenCounts(message)).toEqual({
|
||||
actualTokens: 3763011,
|
||||
limitTokens: 1048576,
|
||||
})
|
||||
expect(getPromptTooLongTokenGap(assistant)).toBe(2714435)
|
||||
})
|
||||
|
||||
test('parses wrapped and case-insensitive Anthropic and OpenAI token counts', () => {
|
||||
for (const message of [
|
||||
'400 {"error":{"message":"PROMPT IS TOO LONG: 137500 tokens > 135000 maximum"}}',
|
||||
'400 {"error":{"message":"This model\'s MAXIMUM CONTEXT LENGTH IS 135000 tokens. However, you REQUESTED 137500 tokens."}}',
|
||||
]) {
|
||||
expect(parsePromptTooLongTokenCounts(message)).toEqual({
|
||||
actualTokens: 137500,
|
||||
limitTokens: 135000,
|
||||
})
|
||||
}
|
||||
})
|
||||
|
||||
test('leaves missing, malformed, and invalid token counts unparsed', () => {
|
||||
for (const message of [
|
||||
'Prompt is too long',
|
||||
'maximum context length is 1048576 tokens',
|
||||
'you requested 3763011 tokens',
|
||||
'maximum context length is -1 tokens. However, you requested 3 tokens.',
|
||||
'maximum context length is 1.5 tokens. However, you requested 3 tokens.',
|
||||
'maximum context length is 0 tokens. However, you requested 3 tokens.',
|
||||
'maximum context length is 1 tokens. However, you requested 9007199254740992 tokens.',
|
||||
'prompt is too long: 0 tokens > 135000 maximum',
|
||||
]) {
|
||||
expect(parsePromptTooLongTokenCounts(message)).toEqual({
|
||||
actualTokens: undefined,
|
||||
limitTokens: undefined,
|
||||
})
|
||||
}
|
||||
})
|
||||
|
||||
test('matches provider-specific overflow wordings', () => {
|
||||
const overflowMessages = [
|
||||
'prompt is too long: 137500 tokens > 135000 maximum',
|
||||
|
||||
@@ -146,7 +146,8 @@ export function isPromptTooLongMessage(msg: AssistantMessage): boolean {
|
||||
|
||||
/**
|
||||
* Parse actual/limit token counts from a raw prompt-too-long API error
|
||||
* message like "prompt is too long: 137500 tokens > 135000 maximum".
|
||||
* message in Anthropic or OpenAI-compatible format. Requested tokens include
|
||||
* the completion allowance so compaction also leaves room for the response.
|
||||
* The raw string may be wrapped in SDK prefixes or JSON envelopes, or
|
||||
* have different casing (Vertex), so this is intentionally lenient.
|
||||
*/
|
||||
@@ -154,12 +155,19 @@ export function parsePromptTooLongTokenCounts(rawMessage: string): {
|
||||
actualTokens: number | undefined
|
||||
limitTokens: number | undefined
|
||||
} {
|
||||
const match = rawMessage.match(
|
||||
const anthropic = rawMessage.match(
|
||||
/prompt is too long[^0-9]*(\d+)\s*tokens?\s*>\s*(\d+)/i,
|
||||
)
|
||||
const openAI = anthropic ? null : rawMessage.match(
|
||||
/maximum context length is\s+(\d+)\s+tokens?\.\s*(?:however,\s*)you requested\s+(\d+)\s+tokens?\b/i,
|
||||
)
|
||||
const actualTokens = Number(anthropic?.[1] ?? openAI?.[2])
|
||||
const limitTokens = Number(anthropic?.[2] ?? openAI?.[1])
|
||||
const valid = Number.isSafeInteger(actualTokens) && actualTokens > 0 &&
|
||||
Number.isSafeInteger(limitTokens) && limitTokens > 0
|
||||
return {
|
||||
actualTokens: match ? parseInt(match[1]!, 10) : undefined,
|
||||
limitTokens: match ? parseInt(match[2]!, 10) : undefined,
|
||||
actualTokens: valid ? actualTokens : undefined,
|
||||
limitTokens: valid ? limitTokens : undefined,
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -1,8 +1,8 @@
|
||||
import { describe, expect, test } from 'bun:test'
|
||||
|
||||
import { buildPostCompactMessages, type CompactionResult } from './compact.js'
|
||||
import { buildPostCompactMessages, truncateHeadForPTLRetry, type CompactionResult } from './compact.js'
|
||||
import { getCurrentUsage } from '../../utils/tokens.js'
|
||||
import type { Message } from '../../types/message.js'
|
||||
import type { AssistantMessage, Message } from '../../types/message.js'
|
||||
|
||||
const PRE_COMPACT_USAGE = {
|
||||
input_tokens: 150_000,
|
||||
@@ -124,3 +124,68 @@ describe('buildPostCompactMessages stale-usage stripping (#743)', () => {
|
||||
expect(getCurrentUsage(result)).toBeNull()
|
||||
})
|
||||
})
|
||||
|
||||
describe('oversized compaction recovery (#1373)', () => {
|
||||
function toolHistory(rounds: number): Message[] {
|
||||
const content = 'historical log data '.repeat(28_000)
|
||||
const messages: Message[] = [makePreservedUser()]
|
||||
for (let index = 0; index < rounds; index++) {
|
||||
messages.push({
|
||||
...makePreservedAssistant(),
|
||||
uuid: crypto.randomUUID(),
|
||||
message: {
|
||||
id: `round-${index}`, role: 'assistant', model: 'deepseek-flash',
|
||||
content: [{ type: 'tool_use', id: `read-${index}`, name: 'Read', input: { file_path: `/fixture/${index}` } }],
|
||||
stop_reason: 'tool_use', usage: { input_tokens: 0, output_tokens: 0 },
|
||||
},
|
||||
} as Message, {
|
||||
...makePreservedUser(),
|
||||
uuid: crypto.randomUUID(),
|
||||
message: { role: 'user', content: [{ type: 'tool_result', tool_use_id: `read-${index}`, content }] },
|
||||
} as Message)
|
||||
}
|
||||
messages.push(makePreservedAssistant(), makePreservedUser())
|
||||
return messages
|
||||
}
|
||||
|
||||
function overflow(errorDetails: string): AssistantMessage {
|
||||
return {
|
||||
...makePreservedAssistant(), isApiErrorMessage: true, errorDetails,
|
||||
message: { ...(makePreservedAssistant() as AssistantMessage).message,
|
||||
content: [{ type: 'text', text: 'Prompt is too long' }] },
|
||||
} as AssistantMessage
|
||||
}
|
||||
|
||||
test('fits the provider budget despite an overestimated local tokenizer, preserving recent tool pairs', () => {
|
||||
const messages = toolHistory(44)
|
||||
// Counts captured from the real DeepSeek request: chars/4 estimates this
|
||||
// repeated log much higher than the provider. Subtracting its raw token
|
||||
// gap from our estimate leaves too many rounds and exhausts the retries.
|
||||
const error = overflow("This model's maximum context length is 1048576 tokens. However, you requested 3763011 tokens (3731011 in the messages, 32000 in the completion).")
|
||||
const result = truncateHeadForPTLRetry(messages, error)!
|
||||
const uses = result.flatMap(message => message.type === 'assistant'
|
||||
? message.message.content.filter(block => block.type === 'tool_use') : [])
|
||||
const results = result.flatMap(message => message.type === 'user' && Array.isArray(message.message.content)
|
||||
? message.message.content.filter(block => block.type === 'tool_result') : [])
|
||||
// Real fixture rounds cost about 84k tokens each, plus fixed request
|
||||
// overhead. A retry must actually fit, not merely become smaller.
|
||||
expect(62_000 + uses.length * 84_114).toBeLessThan(1_048_576)
|
||||
expect(uses.length).toBeGreaterThan(0)
|
||||
expect(uses.map(block => block.id)).toEqual(results.map(block => block.tool_use_id))
|
||||
expect(uses.at(-1)?.id).toBe('read-43')
|
||||
expect(result.at(-1)).toBe(messages.at(-1))
|
||||
expect(messages).toHaveLength(91)
|
||||
expect(result[0]?.type).toBe('user')
|
||||
})
|
||||
|
||||
test('unparseable overflow still makes progress across retries and keeps the newest round', () => {
|
||||
const messages = toolHistory(8)
|
||||
const error = overflow('Provider rejected the prompt without token counts')
|
||||
const first = truncateHeadForPTLRetry(messages, error)!
|
||||
const second = truncateHeadForPTLRetry(first, error)!
|
||||
expect(first.length).toBeLessThan(messages.length)
|
||||
expect(second.length).toBeLessThan(first.length)
|
||||
expect(second.at(-1)).toBe(messages.at(-1))
|
||||
expect(second[0]?.type).toBe('user')
|
||||
})
|
||||
})
|
||||
|
||||
@@ -101,7 +101,8 @@ import {
|
||||
queryModelWithStreaming,
|
||||
} from '../api/claude.js'
|
||||
import {
|
||||
getPromptTooLongTokenGap,
|
||||
isPromptTooLongMessage,
|
||||
parsePromptTooLongTokenCounts,
|
||||
PROMPT_TOO_LONG_ERROR_MESSAGE,
|
||||
startsWithApiErrorPrefix,
|
||||
} from '../api/errors.js'
|
||||
@@ -228,7 +229,9 @@ const MAX_PTL_RETRIES = 3
|
||||
const PTL_RETRY_MARKER = '[earlier conversation truncated for compaction retry]'
|
||||
|
||||
/**
|
||||
* Drops the oldest API-round groups from messages until tokenGap is covered.
|
||||
* Drops the oldest API-round groups to fit the provider's reported budget.
|
||||
* Local token estimates only weight the groups: their absolute scale can be
|
||||
* very different from the provider's tokenizer, especially for large logs.
|
||||
* Falls back to dropping 20% of groups when the gap is unparseable (some
|
||||
* Vertex/Bedrock error formats). Returns null when nothing can be dropped
|
||||
* without leaving an empty summarize set.
|
||||
@@ -257,15 +260,26 @@ export function truncateHeadForPTLRetry(
|
||||
const groups = groupMessagesByApiRound(input)
|
||||
if (groups.length < 2) return null
|
||||
|
||||
const tokenGap = getPromptTooLongTokenGap(ptlResponse)
|
||||
const { actualTokens, limitTokens } = parsePromptTooLongTokenCounts(ptlResponse.errorDetails ?? '')
|
||||
let dropCount: number
|
||||
if (tokenGap !== undefined) {
|
||||
if (
|
||||
isPromptTooLongMessage(ptlResponse) &&
|
||||
actualTokens !== undefined &&
|
||||
limitTokens !== undefined &&
|
||||
actualTokens > limitTokens
|
||||
) {
|
||||
// Reserve room for the summary prompt, fixed request overhead and rounding
|
||||
// at API-round boundaries. Subtracting an actual token gap directly from
|
||||
// chars/4 estimates can leave oversized DeepSeek histories over limit
|
||||
// even after retries (3.76M requested tokens for a 1.05M window in #1373).
|
||||
const estimatedTokens = roughTokenCountEstimationForMessages(input)
|
||||
const tokensToDrop = estimatedTokens * (1 - (limitTokens * 0.9) / actualTokens)
|
||||
let acc = 0
|
||||
dropCount = 0
|
||||
for (const g of groups) {
|
||||
acc += roughTokenCountEstimationForMessages(g)
|
||||
dropCount++
|
||||
if (acc >= tokenGap) break
|
||||
if (acc >= tokensToDrop) break
|
||||
}
|
||||
} else {
|
||||
dropCount = Math.max(1, Math.floor(groups.length * 0.2))
|
||||
|
||||
Reference in New Issue
Block a user