fix(streaming): scale the overall stream cap with the user request timeout

The desktop injects CLAUDE_STREAM_MAX_DURATION_MS=600000, a wall-clock cap
that no incoming chunk resets. Slow local models (LM Studio / qwen3) and very
large generations legitimately keep streaming thinking_delta past ten minutes,
so the stream was killed mid-response even though it was alive and producing
content (#1307, #1275). Raising "请求超时" could not help: the cap was
hardcoded, and the first-token budget it was meant to raise was silently
bounded by that same 600s.

Derive the cap from the user's request timeout, never below the existing 600s
default so the #766 trickle protection is not tightened when a short first-byte
budget is configured. Push the derived value through the per-turn env hot
update as well — otherwise a session that is already running keeps its spawn
time 600s no matter what the user configures, which is exactly the retry path
the reports describe.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
程序员阿江(Relakkes)
2026-09-11 16:40:45 +08:00
parent 6d09d9f899
commit 446cc47846
8 changed files with 143 additions and 8 deletions
+1 -1
View File
@@ -1787,7 +1787,7 @@ Row 9, all 8 cells: continuing from straight down, turning left through lower-le
'settings.general.networkProxyUrlRequired': 'Enter a proxy URL.',
'settings.general.networkTimeout': 'AI request timeout',
'settings.general.networkTimeoutValue': '{seconds}s',
'settings.general.networkTimeoutHint': 'Applies to provider requests, streaming first responses, and provider connection tests. Supports 30-1800 seconds; slow providers may need several minutes before the first streamed byte.',
'settings.general.networkTimeoutHint': 'Applies to provider requests, streaming first responses, and provider connection tests. Supports 30-1800 seconds; slow providers may need several minutes before the first streamed byte. Raising it also extends the overall time limit for a single streamed response (never below 600 seconds).',
'settings.general.networkTimeoutUnit': 'sec',
'settings.general.networkTimeoutDecrease': 'Decrease by 30 seconds',
'settings.general.networkTimeoutIncrease': 'Increase by 30 seconds',
+1 -1
View File
@@ -1789,7 +1789,7 @@ export const jp: Record<TranslationKey, string> = {
'settings.general.networkProxyUrlRequired': 'プロキシ URL を入力してください。',
'settings.general.networkTimeout': 'AI リクエストのタイムアウト',
'settings.general.networkTimeoutValue': '{seconds}秒',
'settings.general.networkTimeoutHint': 'プロバイダーへのリクエスト、ストリーミングの最初の応答、プロバイダー接続テストに適用されます。30〜1800 秒に対応します。大きなコンテキストでは最初のストリーミングバイトまで数分かかる場合があります。',
'settings.general.networkTimeoutHint': 'プロバイダーへのリクエスト、ストリーミングの最初の応答、プロバイダー接続テストに適用されます。30〜1800 秒に対応します。大きなコンテキストでは最初のストリーミングバイトまで数分かかる場合があります。値を大きくすると、1 回のストリーミング応答の総時間上限も広がります(最低 600 秒)。',
'settings.general.networkTimeoutUnit': '秒',
'settings.general.networkTimeoutDecrease': '30 秒減らす',
'settings.general.networkTimeoutIncrease': '30 秒増やす',
+1 -1
View File
@@ -1789,7 +1789,7 @@ export const kr: Record<TranslationKey, string> = {
'settings.general.networkProxyUrlRequired': '프록시 URL을 입력하세요.',
'settings.general.networkTimeout': 'AI 요청 시간 초과',
'settings.general.networkTimeoutValue': '{seconds}초',
'settings.general.networkTimeoutHint': '공급자 요청, 스트리밍 첫 응답, 공급자 연결 테스트에 적용됩니다. 30~1800초를 지원합니다. 큰 컨텍스트에서는 첫 스트리밍 바이트까지 몇 분이 걸릴 수 있습니다.',
'settings.general.networkTimeoutHint': '공급자 요청, 스트리밍 첫 응답, 공급자 연결 테스트에 적용됩니다. 30~1800초를 지원합니다. 큰 컨텍스트에서는 첫 스트리밍 바이트까지 몇 분이 걸릴 수 있습니다. 값을 늘리면 단일 스트리밍 응답의 총 시간 상한도 함께 늘어납니다(최소 600초).',
'settings.general.networkTimeoutUnit': '초',
'settings.general.networkTimeoutDecrease': '30초 줄이기',
'settings.general.networkTimeoutIncrease': '30초 늘리기',
+1 -1
View File
@@ -1788,7 +1788,7 @@ export const zh: Record<TranslationKey, string> = {
'settings.general.networkProxyUrlRequired': '請輸入代理地址。',
'settings.general.networkTimeout': 'AI 請求超時',
'settings.general.networkTimeoutValue': '{seconds} 秒',
'settings.general.networkTimeoutHint': '用於服務商請求、流式首個響應,以及服務商連線測試。支援 30-1800 秒;部分服務商在大上下文下首個流式位元組可能需要等待數分鐘。',
'settings.general.networkTimeoutHint': '用於服務商請求、流式首個響應,以及服務商連線測試。支援 30-1800 秒;部分服務商在大上下文下首個流式位元組可能需要等待數分鐘。調大後,單次流式回應的總時長上限同步放寬(不低於 600 秒)。',
'settings.general.networkTimeoutUnit': '秒',
'settings.general.networkTimeoutDecrease': '減少 30 秒',
'settings.general.networkTimeoutIncrease': '增加 30 秒',
+1 -1
View File
@@ -1788,7 +1788,7 @@ export const zh: Record<TranslationKey, string> = {
'settings.general.networkProxyUrlRequired': '请输入代理地址。',
'settings.general.networkTimeout': 'AI 请求超时',
'settings.general.networkTimeoutValue': '{seconds} 秒',
'settings.general.networkTimeoutHint': '用于服务商请求、流式首个响应,以及服务商连接测试。支持 30-1800 秒;部分服务商在大上下文下首个流式字节可能需要等待数分钟。',
'settings.general.networkTimeoutHint': '用于服务商请求、流式首个响应,以及服务商连接测试。支持 30-1800 秒;部分服务商在大上下文下首个流式字节可能需要等待数分钟。调大后,单次流式回复的总时长上限同步放宽(不低于 600 秒)。',
'settings.general.networkTimeoutUnit': '秒',
'settings.general.networkTimeoutDecrease': '减少 30 秒',
'settings.general.networkTimeoutIncrease': '增加 30 秒',
@@ -184,11 +184,13 @@ describe('ConversationService', () => {
sessionId: string,
sent: string[],
networkDerivedFirstTokenTimeout = true,
networkDerivedStreamMaxDuration = true,
) {
const session = {
outputCallbacks: [],
networkRoutingFingerprint: '',
networkDerivedFirstTokenTimeout,
networkDerivedStreamMaxDuration,
sdkSocket: {
send(line: string) {
sent.push(line)
@@ -473,6 +475,51 @@ describe('ConversationService', () => {
}
})
test('buildChildEnv raises the overall stream cap with the user request timeout so an active long response is not killed (#1307)', async () => {
const prev = process.env.CLAUDE_STREAM_MAX_DURATION_MS
delete process.env.CLAUDE_STREAM_MAX_DURATION_MS
await fs.writeFile(
path.join(tmpDir, 'settings.json'),
JSON.stringify({ network: { aiRequestTimeoutMs: 1_800_000 } }),
'utf-8',
)
try {
const service = new ConversationService() as any
const env = (await service.buildChildEnv('/tmp')) as Record<string, string>
// The overall cap is NOT reset by incoming chunks, so a local model that
// keeps streaming thinking_delta events past it is killed mid-response.
// Raising "请求超时" must therefore extend the cap too — otherwise the
// user's timeout setting is silently capped at 600s (#1307).
expect(env.CLAUDE_STREAM_MAX_DURATION_MS).toBe('1800000')
} finally {
if (prev === undefined) delete process.env.CLAUDE_STREAM_MAX_DURATION_MS
else process.env.CLAUDE_STREAM_MAX_DURATION_MS = prev
}
})
test('buildChildEnv keeps the overall stream cap floor for a short request timeout (#766)', async () => {
const prev = process.env.CLAUDE_STREAM_MAX_DURATION_MS
delete process.env.CLAUDE_STREAM_MAX_DURATION_MS
await fs.writeFile(
path.join(tmpDir, 'settings.json'),
JSON.stringify({ network: { aiRequestTimeoutMs: 30_000 } }),
'utf-8',
)
try {
const service = new ConversationService() as any
const env = (await service.buildChildEnv('/tmp')) as Record<string, string>
// Shrinking the cap for a short first-byte budget would re-open #766:
// a stream that trickles content deltas just under the idle window must
// still be freed after a fixed overall duration.
expect(env.CLAUDE_STREAM_MAX_DURATION_MS).toBe('600000')
} finally {
if (prev === undefined) delete process.env.CLAUDE_STREAM_MAX_DURATION_MS
else process.env.CLAUDE_STREAM_MAX_DURATION_MS = prev
}
})
test('buildChildEnv lets caller env override the first-token watchdog (#826)', async () => {
const prev = process.env.CLAUDE_STREAM_FIRST_TOKEN_TIMEOUT_MS
process.env.CLAUDE_STREAM_FIRST_TOKEN_TIMEOUT_MS = '900000'
@@ -777,10 +824,51 @@ describe('ConversationService', () => {
all_proxy: 'http://127.0.0.1:17892',
API_TIMEOUT_MS: '180000',
CLAUDE_STREAM_FIRST_TOKEN_TIMEOUT_MS: '180000',
// The floor still holds on the hot-update path: lowering the request
// timeout must not shrink the overall cap below its 600s default.
CLAUDE_STREAM_MAX_DURATION_MS: '600000',
})
expect(JSON.parse(sent[1]!).type).toBe('user')
})
test('sendMessage hot-applies a raised request timeout to the stream cap so a running session is not stuck at 600s (#1307)', async () => {
await fs.writeFile(
path.join(tmpDir, 'settings.json'),
JSON.stringify({
network: {
aiRequestTimeoutMs: 600_000,
proxy: { mode: 'direct', url: '' },
},
}),
'utf-8',
)
const service = new ConversationService() as any
const sent: string[] = []
const session = installNetworkTestSession(service, 'raised-timeout', sent)
await service.refreshNetworkEnvironmentBeforeTurn('raised-timeout', session)
await fs.writeFile(
path.join(tmpDir, 'settings.json'),
JSON.stringify({
network: {
aiRequestTimeoutMs: 1_800_000,
proxy: { mode: 'direct', url: '' },
},
}),
'utf-8',
)
expect(await service.sendMessage('raised-timeout', 'retry the long prompt')).toBe(true)
expect(sent).toHaveLength(2)
const update = JSON.parse(sent[0]!)
expect(update.type).toBe('update_environment_variables')
// The reported path is: user hits the 600s error, raises the timeout, and
// retries in the SAME conversation. The live CLI re-reads this per request,
// so it has to be pushed down — otherwise the retry dies at 600s again.
expect(update.variables.CLAUDE_STREAM_MAX_DURATION_MS).toBe('1800000')
expect(JSON.parse(sent[1]!).type).toBe('user')
})
test('sendMessage does not resend network env when the system bridge fingerprint is unchanged', async () => {
const originalBridgeUrl = process.env.CC_HAHA_SYSTEM_PROXY_URL
process.env.CC_HAHA_SYSTEM_PROXY_URL = 'http://127.0.0.1:17893'
+35 -3
View File
@@ -58,6 +58,7 @@ import { attributionHeaderEnvForModel } from './attributionHeaderPolicy.js'
import {
buildNetworkEnvironment,
loadNetworkSettings,
resolveStreamMaxDurationMs,
SYSTEM_PROXY_URL_ENV,
type NetworkSettings,
} from './networkSettings.js'
@@ -200,6 +201,7 @@ type SessionProcess = {
permissionMode: string
networkRoutingFingerprint: string
networkDerivedFirstTokenTimeout: boolean
networkDerivedStreamMaxDuration: boolean
sdkToken: string
sdkSocket: { send(data: string): void } | null
sdkAttached: Promise<void>
@@ -426,7 +428,10 @@ export class ConversationService {
// chdir 后落到正确目录。
//
const networkSettings = await loadNetworkSettings()
const networkRuntimeMetadata = { firstTokenTimeoutDerived: false }
const networkRuntimeMetadata = {
firstTokenTimeoutDerived: false,
streamMaxDurationDerived: false,
}
const childEnv = await this.buildChildEnv(
launchWorkDir,
sdkUrl,
@@ -472,6 +477,7 @@ export class ConversationService {
permissionMode: options?.permissionMode || 'default',
networkRoutingFingerprint: networkRoutingFingerprint(networkSettings, childEnv),
networkDerivedFirstTokenTimeout: networkRuntimeMetadata.firstTokenTimeoutDerived,
networkDerivedStreamMaxDuration: networkRuntimeMetadata.streamMaxDurationDerived,
sdkToken: this.getSdkTokenFromUrl(sdkUrl),
sdkSocket: null,
seenSdkMessageUuids: new Set<string>(),
@@ -662,6 +668,14 @@ export class ConversationService {
if (session.networkDerivedFirstTokenTimeout) {
variables.CLAUDE_STREAM_FIRST_TOKEN_TIMEOUT_MS = networkEnv.API_TIMEOUT_MS
}
// The CLI re-reads this per request, so the fix for #1307 must reach a
// session that is already running — otherwise the user sees the 600s error,
// raises the timeout, retries in the same conversation and hits it again.
if (session.networkDerivedStreamMaxDuration) {
variables.CLAUDE_STREAM_MAX_DURATION_MS = String(
resolveStreamMaxDurationMs(networkEnv.API_TIMEOUT_MS),
)
}
const sent = this.sendSdkMessage(sessionId, {
type: 'update_environment_variables',
@@ -1522,7 +1536,10 @@ export class ConversationService {
sdkUrl?: string,
options?: SessionStartOptions,
networkSettingsOverride?: NetworkSettings,
networkRuntimeMetadata?: { firstTokenTimeoutDerived: boolean },
networkRuntimeMetadata?: {
firstTokenTimeoutDerived: boolean
streamMaxDurationDerived: boolean
},
): Promise<Record<string, string>> {
// Provider isolation: when Desktop has its own provider config/index,
// strip inherited provider env vars so the child CLI reads fresh values
@@ -1568,6 +1585,8 @@ export class ConversationService {
if (networkRuntimeMetadata) {
networkRuntimeMetadata.firstTokenTimeoutDerived =
!cleanEnv.CLAUDE_STREAM_FIRST_TOKEN_TIMEOUT_MS
networkRuntimeMetadata.streamMaxDurationDerived =
!cleanEnv.CLAUDE_STREAM_MAX_DURATION_MS
}
delete cleanEnv.CLAUDE_CODE_OAUTH_TOKEN
if (options?.resumeInterruptedTurn === false) {
@@ -1603,6 +1622,14 @@ export class ConversationService {
networkSettingsOverride ?? await loadNetworkSettings(),
cleanEnv,
)
// The overall-duration cap has to scale with the user's "请求超时" or raising
// that setting can never extend a long response — the cap is a wall-clock
// budget that no chunk resets, so a slow local model that legitimately
// thinks past it is killed mid-stream (#1307). The floor keeps the existing
// 600s protection from being TIGHTENED when the user configures a short
// first-byte budget (a lower cap would kill legitimate long responses even
// earlier); the per-turn hot update below mirrors this for live turns.
const streamMaxDurationMs = resolveStreamMaxDurationMs(networkEnv.API_TIMEOUT_MS)
const traceCaptureEnabled = (await readTraceCaptureSettings()).enabled
if (explicitProviderEnv && options?.model?.trim()) {
explicitProviderEnv.ANTHROPIC_MODEL = options.model.trim()
@@ -1639,7 +1666,12 @@ export class ConversationService {
// just under 240s apart keeps it alive forever and the request hangs with
// no completion (#766: "卡住" with slowly growing tokens). This independent
// cap frees such a stream after a fixed duration regardless of trickle.
CLAUDE_STREAM_MAX_DURATION_MS: cleanEnv.CLAUDE_STREAM_MAX_DURATION_MS || '600000',
// It tracks the user's "请求超时" (never below MIN_STREAM_MAX_DURATION_MS)
// so that raising the timeout also extends legitimately long responses,
// and a provider preset's own CLAUDE_STREAM_MAX_DURATION_MS still wins
// via the explicitProviderEnv spread below (#1307).
CLAUDE_STREAM_MAX_DURATION_MS:
cleanEnv.CLAUDE_STREAM_MAX_DURATION_MS || String(streamMaxDurationMs),
// Abort a local tool call when its JSON arguments stop making progress.
// Healthy input_json_delta events reset this budget; the independent full
// response cap above still bounds a stream that trickles forever.
+15
View File
@@ -27,6 +27,21 @@ export type NetworkSettings = {
export const DEFAULT_AI_REQUEST_TIMEOUT_MS = 600_000
export const MIN_AI_REQUEST_TIMEOUT_MS = 30_000
export const MAX_AI_REQUEST_TIMEOUT_MS = 1_800_000
// Floor for the CLI's overall stream-duration cap (CLAUDE_STREAM_MAX_DURATION_MS).
// That cap is what frees an endlessly-trickling provider stream (#766), but it is
// a wall-clock budget that no incoming chunk resets. It must therefore never be
// LOWER than the user's own "请求超时": a local model that legitimately spends
// longer than that thinking or generating would otherwise be killed mid-response
// no matter how far the user raises the timeout (#1307). The floor keeps the
// #766 protection intact when the user configures a very short first-byte budget.
export const MIN_STREAM_MAX_DURATION_MS = 600_000
// Shared by the spawn-time child env and the per-turn hot update so the two
// cannot drift apart.
export function resolveStreamMaxDurationMs(
apiTimeoutMs: string | number | undefined,
): number {
return Math.max(MIN_STREAM_MAX_DURATION_MS, Number(apiTimeoutMs) || 0)
}
export const SYSTEM_PROXY_URL_ENV = 'CC_HAHA_SYSTEM_PROXY_URL'
export const SYSTEM_PROXY_ERROR_ENV = 'CC_HAHA_SYSTEM_PROXY_ERROR'