From ca43297daeed7a6a501aef23187f3c232913454d Mon Sep 17 00:00:00 2001 From: MAC Date: Wed, 2 Sep 2026 21:53:58 +0800 Subject: [PATCH 1/4] feat(models): add Fable 5.1 with reasoning effort support Add claude-fable-5-1 (Fable 5.1) to the official model catalog and the server fallback model list, with reasoning effort enabled (low/medium/high/xhigh/max, default high) to match Opus 4.8 / Sonnet 5. Previously only Fable 5 was selectable. Update the affected fixtures in ModelSelector and settings API tests. Co-Authored-By: Claude Opus 4.8 Claude-Session: https://claude.ai/code/session_019MxSMcKdN6WWQjtJEUhLGR --- desktop/src/components/controls/ModelSelector.test.tsx | 2 +- desktop/src/constants/modelCatalog.ts | 8 ++++++++ src/server/__tests__/settings.test.ts | 9 +++++++++ src/server/api/models.ts | 8 ++++++++ 4 files changed, 26 insertions(+), 1 deletion(-) diff --git a/desktop/src/components/controls/ModelSelector.test.tsx b/desktop/src/components/controls/ModelSelector.test.tsx index 7899f1f1..330ecec5 100644 --- a/desktop/src/components/controls/ModelSelector.test.tsx +++ b/desktop/src/components/controls/ModelSelector.test.tsx @@ -93,7 +93,7 @@ describe('ModelSelector', () => { await clickByRole(/Opus 4\.7/i) - expect(screen.getByRole('button', { name: /Fable 5/ })).toBeInTheDocument() + expect(screen.getByRole('button', { name: /Fable 5\.1/ })).toBeInTheDocument() expect(screen.getByRole('button', { name: /Opus 4\.8/ })).toBeInTheDocument() expect(screen.getByRole('button', { name: /Sonnet 5/ })).toBeInTheDocument() expect(screen.getAllByRole('button', { name: /Opus 4\.7/ }).length).toBeGreaterThan(0) diff --git a/desktop/src/constants/modelCatalog.ts b/desktop/src/constants/modelCatalog.ts index f97aca24..a87cb823 100644 --- a/desktop/src/constants/modelCatalog.ts +++ b/desktop/src/constants/modelCatalog.ts @@ -3,6 +3,14 @@ import type { ModelInfo } from '../types/settings' export const OFFICIAL_DEFAULT_MODEL_ID = 'claude-opus-4-8' export const OFFICIAL_MODELS: ModelInfo[] = [ + { + id: 'claude-fable-5-1', + name: 'Fable 5.1', + description: 'Highest capability for long-running tasks', + context: '1m', + defaultReasoningEffort: 'high', + supportedReasoningEfforts: ['low', 'medium', 'high', 'xhigh', 'max'], + }, { id: 'claude-fable-5', name: 'Fable 5', diff --git a/src/server/__tests__/settings.test.ts b/src/server/__tests__/settings.test.ts index c1431e3e..99ceb622 100644 --- a/src/server/__tests__/settings.test.ts +++ b/src/server/__tests__/settings.test.ts @@ -713,6 +713,14 @@ describe('Models API', () => { expect(res.status).toBe(200) const body = await res.json() expect(body.models).toEqual([ + { + id: 'claude-fable-5-1', + name: 'Fable 5.1', + description: 'Highest capability for long-running tasks', + context: '1m', + defaultReasoningEffort: 'high', + supportedReasoningEfforts: ['low', 'medium', 'high', 'xhigh', 'max'], + }, { id: 'claude-fable-5', name: 'Fable 5', @@ -971,6 +979,7 @@ describe('Models API', () => { ) const listBody = await listResponse.json() expect(listBody.models.map((model: { id: string }) => model.id)).toEqual([ + 'claude-fable-5-1', 'claude-fable-5', 'claude-opus-4-8', 'claude-sonnet-5', diff --git a/src/server/api/models.ts b/src/server/api/models.ts index 20608eea..5e11bcff 100644 --- a/src/server/api/models.ts +++ b/src/server/api/models.ts @@ -50,6 +50,14 @@ import { // ─── Fallback models (used when no provider is configured) ──────────────────── const DEFAULT_MODELS = [ + { + id: 'claude-fable-5-1', + name: 'Fable 5.1', + description: 'Highest capability for long-running tasks', + context: '1m', + defaultReasoningEffort: 'high', + supportedReasoningEfforts: ['low', 'medium', 'high', 'xhigh', 'max'], + }, { id: 'claude-fable-5', name: 'Fable 5', From 9c081af7462edb2c76e89b05248c37ed19982163 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=E7=A8=8B=E5=BA=8F=E5=91=98=E9=98=BF=E6=B1=9F=28Relakkes?= =?UTF-8?q?=29?= Date: Mon, 7 Sep 2026 21:41:31 +0800 Subject: [PATCH 2/4] fix(models): complete Fable 5.1 runtime compatibility --- src/cli/handlers/autoMode.test.ts | 21 ++- src/constants/betas.ts | 1 + .../__tests__/e2e/business-flow.test.ts | 16 +- src/server/__tests__/e2e/full-flow.test.ts | 4 +- src/services/api/claude.ts | 18 +- .../api/claudeRequiredThinking.test.ts | 130 +++++++++++--- src/utils/commitAttribution.ts | 1 + src/utils/model/configs.ts | 9 + src/utils/model/fable.test.ts | 37 ++++ src/utils/model/model.ts | 10 ++ src/utils/model/modelContextWindows.ts | 1 + src/utils/model/modelOptions.ts | 11 +- src/utils/modelCost.test.ts | 32 ++++ src/utils/modelCost.ts | 9 + src/utils/permissions/permissionExplainer.ts | 8 +- .../permissions/permissions.autoMode.test.ts | 17 +- src/utils/sideQuery.test.ts | 162 ++++++++++++++++++ src/utils/sideQuery.ts | 61 ++++++- src/utils/thinking.ts | 5 + src/utils/usageAccounting.test.ts | 13 ++ src/utils/usageAccounting.ts | 2 + 21 files changed, 519 insertions(+), 49 deletions(-) create mode 100644 src/utils/modelCost.test.ts create mode 100644 src/utils/sideQuery.test.ts diff --git a/src/cli/handlers/autoMode.test.ts b/src/cli/handlers/autoMode.test.ts index 0f051a43..9f3edbb5 100644 --- a/src/cli/handlers/autoMode.test.ts +++ b/src/cli/handlers/autoMode.test.ts @@ -1,4 +1,4 @@ -import { afterEach, describe, expect, mock, spyOn, test } from 'bun:test' +import { afterEach, beforeEach, describe, expect, mock, spyOn, test } from 'bun:test' import { feature } from 'bun:bundle' import { resetSettingsCache, @@ -8,14 +8,23 @@ import { process.env.ANTHROPIC_API_KEY = 'test-key' let capturedQuery: Record | undefined -mock.module('../../utils/sideQuery.js', () => ({ - sideQuery: async (options: Record) => { +const sideQueryModule = await import('../../utils/sideQuery.js') + +beforeEach(() => { + spyOn(sideQueryModule, 'sideQuery').mockImplementation(async options => { capturedQuery = options return { + id: 'msg_critique', + type: 'message', + role: 'assistant', + model: 'test-model', + stop_reason: 'end_turn', + stop_sequence: null, + usage: { input_tokens: 1, output_tokens: 1 }, content: [{ type: 'text', text: 'critique complete' }], - } - }, -})) + } as Awaited> + }) +}) const transcriptClassifierEnabled = feature('TRANSCRIPT_CLASSIFIER') ? true diff --git a/src/constants/betas.ts b/src/constants/betas.ts index dc1383b0..9b2e0d90 100644 --- a/src/constants/betas.ts +++ b/src/constants/betas.ts @@ -3,6 +3,7 @@ import { feature } from 'bun:bundle' export const CLAUDE_CODE_20250219_BETA_HEADER = 'claude-code-20250219' export const INTERLEAVED_THINKING_BETA_HEADER = 'interleaved-thinking-2025-05-14' +export const THINKING_BINDING_CONTROLS_BETA_HEADER = 'thinking-binding-controls-2026-08-01' export const CONTEXT_1M_BETA_HEADER = 'context-1m-2025-08-07' export const CONTEXT_MANAGEMENT_BETA_HEADER = 'context-management-2025-06-27' export const STRUCTURED_OUTPUTS_BETA_HEADER = 'structured-outputs-2025-12-15' diff --git a/src/server/__tests__/e2e/business-flow.test.ts b/src/server/__tests__/e2e/business-flow.test.ts index fafa9d8e..e0746434 100644 --- a/src/server/__tests__/e2e/business-flow.test.ts +++ b/src/server/__tests__/e2e/business-flow.test.ts @@ -374,8 +374,9 @@ describe('Business Flow: Models & Effort', () => { it('should return available fallback models', async () => { const { data } = await api('GET', '/api/models') - expect(data.models.length).toBe(4) + expect(data.models.length).toBe(5) const names = data.models.map((m: any) => m.name) + expect(names).toContain('Fable 5.1') expect(names).toContain('Fable 5') expect(names).toContain('Opus 4.8') expect(names).toContain('Sonnet 5') @@ -398,6 +399,19 @@ describe('Business Flow: Models & Effort', () => { expect(data.model.name).toBe('Opus 4.8') }) + it('should select Fable 5.1 with its reasoning catalog intact', async () => { + const { status } = await api('PUT', '/api/models/current', { + modelId: 'claude-fable-5-1', + }) + expect(status).toBe(200) + const { data } = await api('GET', '/api/models/current') + expect(data.model).toMatchObject({ + id: 'claude-fable-5-1', name: 'Fable 5.1', context: '1m', + defaultReasoningEffort: 'high', + supportedReasoningEfforts: ['low', 'medium', 'high', 'xhigh', 'max'], + }) + }) + it('should switch to Haiku 4.5', async () => { await api('PUT', '/api/models/current', { modelId: 'claude-haiku-4-5' }) const { data } = await api('GET', '/api/models/current') diff --git a/src/server/__tests__/e2e/full-flow.test.ts b/src/server/__tests__/e2e/full-flow.test.ts index 8c93984e..ddd10a4a 100644 --- a/src/server/__tests__/e2e/full-flow.test.ts +++ b/src/server/__tests__/e2e/full-flow.test.ts @@ -216,8 +216,8 @@ describe('E2E: Full Flow', () => { it('should list available models', async () => { const { data } = await api('GET', '/api/models') - expect(data.models.length).toBe(4) - expect(data.models[0].name).toBe('Fable 5') + expect(data.models.length).toBe(5) + expect(data.models[0].name).toBe('Fable 5.1') }) it('should switch model', async () => { diff --git a/src/services/api/claude.ts b/src/services/api/claude.ts index 5f74d7d4..1f3bacef 100644 --- a/src/services/api/claude.ts +++ b/src/services/api/claude.ts @@ -134,6 +134,7 @@ import { } from "src/bootstrap/state.js"; import { AFK_MODE_BETA_HEADER, + THINKING_BINDING_CONTROLS_BETA_HEADER, CONTEXT_1M_BETA_HEADER, CONTEXT_MANAGEMENT_BETA_HEADER, EFFORT_BETA_HEADER, @@ -191,6 +192,7 @@ import { import { endQueryProfile, queryCheckpoint } from "src/utils/queryProfiler.js"; import { modelSupportsAdaptiveThinking, + modelUsesBoundThinking, modelSupportsThinking, resolveModelThinkingEnabled, shouldSendExplicitDisabledThinking, @@ -1751,7 +1753,8 @@ async function* queryModel( // setting that can greatly affect model quality and bashing. if (hasThinking && modelCanThink) { if ( - !isEnvTruthy(process.env.CLAUDE_CODE_DISABLE_ADAPTIVE_THINKING) && + (modelUsesBoundThinking(options.model) || + !isEnvTruthy(process.env.CLAUDE_CODE_DISABLE_ADAPTIVE_THINKING)) && modelSupportsAdaptiveThinking(options.model) ) { // For models that support adaptive thinking, always use adaptive @@ -1781,6 +1784,19 @@ async function* queryModel( } as unknown as BetaMessageStreamParams['thinking'] } + if (thinking?.type === 'adaptive' && modelUsesBoundThinking(options.model)) { + // Directory, tool and compacted-history updates can invalidate old thinking. + // Let the API retain valid blocks and drop only those bound to an old prefix. + // This header is required for compatibility even when optional betas are disabled. + thinking = { + ...thinking, + block_binding: { prefix_mismatch_behavior: 'drop_block' }, + } as typeof thinking + if (!betasParams.includes(THINKING_BINDING_CONTROLS_BETA_HEADER)) { + betasParams.push(THINKING_BINDING_CONTROLS_BETA_HEADER) + } + } + // Get API context management strategies if enabled const contextManagement = getAPIContextManagement({ hasThinking, diff --git a/src/services/api/claudeRequiredThinking.test.ts b/src/services/api/claudeRequiredThinking.test.ts index 1dfb1a48..6208ec40 100644 --- a/src/services/api/claudeRequiredThinking.test.ts +++ b/src/services/api/claudeRequiredThinking.test.ts @@ -10,13 +10,18 @@ import { enableConfigs } from '../../utils/config.js' import { get3PModelCapabilityOverride } from '../../utils/model/modelSupportOverrides.js' import { buildProviderManagedEnv } from '../../server/services/providerRuntimeEnv.js' import type { SavedProvider } from '../../server/types/provider.js' -import { queryWithModel } from './claude.js' +import { queryModelWithStreaming, queryWithModel } from './claude.js' +import { createUserMessage } from '../../utils/messages.js' +import { asSystemPrompt } from '../../utils/systemPromptType.js' +import { getEmptyToolPermissionContext } from '../../Tool.js' +import type { Message } from '../../types/message.js' function sseEvent(name: string, data: unknown): string { return `event: ${name}\ndata: ${JSON.stringify(data)}\n\n` } -function successfulResponse(model: string): string { +function successfulResponse(model: string, withThinking = false): string { + const textIndex = withThinking ? 1 : 0 return [ sseEvent('message_start', { type: 'message_start', @@ -31,17 +36,28 @@ function successfulResponse(model: string): string { usage: { input_tokens: 1, output_tokens: 0 }, }, }), + ...(withThinking ? [ + sseEvent('content_block_start', { + type: 'content_block_start', index: 0, + content_block: { type: 'thinking', thinking: '', signature: '' }, + }), + sseEvent('content_block_delta', { + type: 'content_block_delta', index: 0, + delta: { type: 'signature_delta', signature: 'fixture-signature' }, + }), + sseEvent('content_block_stop', { type: 'content_block_stop', index: 0 }), + ] : []), sseEvent('content_block_start', { type: 'content_block_start', - index: 0, + index: textIndex, content_block: { type: 'text', text: '' }, }), sseEvent('content_block_delta', { type: 'content_block_delta', - index: 0, + index: textIndex, delta: { type: 'text_delta', text: 'OK' }, }), - sseEvent('content_block_stop', { type: 'content_block_stop', index: 0 }), + sseEvent('content_block_stop', { type: 'content_block_stop', index: textIndex }), sseEvent('message_delta', { type: 'message_delta', delta: { stop_reason: 'end_turn', stop_sequence: null }, @@ -298,6 +314,7 @@ async function captureQueryRequest({ globalThinkingEnabled, provider, responseFactory, + continuationSystemPrompts, env, }: { model: string @@ -307,7 +324,8 @@ async function captureQueryRequest({ configureCapabilityOverrides?: boolean globalThinkingEnabled?: boolean provider?: SavedProvider - responseFactory?: (model: string, body: Record) => Response + responseFactory?: (model: string, body: Record, headers: Headers) => Response + continuationSystemPrompts?: string[][] env?: Readonly> }): Promise<{ content: unknown @@ -325,7 +343,7 @@ async function captureQueryRequest({ requestHeaders.push(new Headers(request.headers)) const body = await request.json() as Record requests.push(body) - return responseFactory?.(model, body) ?? new Response(successfulResponse(model), { + return responseFactory?.(model, body, request.headers) ?? new Response(successfulResponse(model), { headers: { 'content-type': 'text/event-stream' }, }) }, @@ -386,19 +404,46 @@ async function captureQueryRequest({ clearCapabilityCache() enableConfigs() - const result = await queryWithModel({ - userPrompt: 'Reply exactly OK', - signal: new AbortController().signal, - options: { - model, - querySource: 'insights', - agents: [], - isNonInteractiveSession: true, - hasAppendSystemPrompt: false, - mcpTools: [], - effortValue, - }, - }) + const options = { + model, + querySource: 'insights' as const, + agents: [], + isNonInteractiveSession: true, + hasAppendSystemPrompt: false, + mcpTools: [], + effortValue, + } + let result + if (continuationSystemPrompts) { + const history: Message[] = [] + for (const [index, systemPrompt] of [[], ...continuationSystemPrompts].entries()) { + history.push(createUserMessage({ content: index === 0 ? 'Reply exactly OK' : 'Continue' })) + const assistants = [] + for await (const message of queryModelWithStreaming({ + messages: history, + systemPrompt: asSystemPrompt(systemPrompt), + thinkingConfig: { type: 'disabled' }, + tools: [], + signal: new AbortController().signal, + options: { + ...options, + enablePromptCaching: false, + getToolPermissionContext: async () => getEmptyToolPermissionContext(), + }, + })) { + if (message.type === 'assistant') assistants.push(message) + } + history.push(...assistants) + result = assistants.at(-1) + } + } else { + result = await queryWithModel({ + userPrompt: 'Reply exactly OK', + signal: new AbortController().signal, + options, + }) + } + if (!result) throw new Error('No assistant response received') return { content: result.message.content, @@ -541,6 +586,51 @@ test('normalizes a disabled parent thinking mode to adaptive for Fable', async ( expect(requests[0]?.thinking).not.toEqual({ type: 'disabled' }) }, 10_000) +for (const [disableBetas, disableAdaptive] of [[false, false], [true, false], [false, true]]) { + test(`keeps Fable 5.1 replay usable after a system change (disable betas: ${disableBetas}, adaptive: ${disableAdaptive})`, async () => { + let originalSystem: string | undefined + let responseIndex = 0 + const { content, requests, requestHeaders } = await captureQueryRequest({ + model: 'claude-fable-5-1', + configureCapabilityOverrides: false, + continuationSystemPrompts: [[], ['Additional working directories: /tmp/project-two']], + env: { + CLAUDE_CODE_DISABLE_EXPERIMENTAL_BETAS: disableBetas ? '1' : undefined, + CLAUDE_CODE_DISABLE_ADAPTIVE_THINKING: disableAdaptive ? '1' : undefined, + }, + responseFactory: (model, body, headers) => { + const system = JSON.stringify(body.system) + const changed = originalSystem !== undefined && system !== originalSystem + originalSystem ??= system + const thinking = body.thinking as { block_binding?: { prefix_mismatch_behavior?: string } } + if (changed && ( + thinking?.block_binding?.prefix_mismatch_behavior !== 'drop_block' || + !headers.get('anthropic-beta')?.includes('thinking-binding-controls-2026-08-01') + )) { + return Response.json({ type: 'error', error: { + type: 'invalid_request_error', message: 'The block is bound to a different conversation', + } }, { status: 400 }) + } + const response = successfulResponse(model, true).replaceAll('msg_required_thinking', `msg_replay_${responseIndex++}`) + return new Response(response, { headers: { 'content-type': 'text/event-stream' } }) + }, + }) + + expect(content).toContainEqual({ type: 'text', text: 'OK' }) + expect(requests).toHaveLength(3) + for (let index = 0; index < requests.length; index++) { + expect(requests[index]?.model).toBe('claude-fable-5-1') + expect(requests[index]?.thinking).toEqual({ + type: 'adaptive', block_binding: { prefix_mismatch_behavior: 'drop_block' }, + }) + expect(requestHeaders[index]?.get('anthropic-beta')).toContain('thinking-binding-controls-2026-08-01') + } + for (const request of requests.slice(1)) { + expect(JSON.stringify(request.messages)).toContain('fixture-signature') + } + }, 10_000) +} + test('drops a tool call truncated at the output-token boundary', async () => { const { content, apiError, error, requests } = await captureQueryRequest({ model: 'deepseek-v4-flash', diff --git a/src/utils/commitAttribution.ts b/src/utils/commitAttribution.ts index aedbf2f6..226e011a 100644 --- a/src/utils/commitAttribution.ts +++ b/src/utils/commitAttribution.ts @@ -153,6 +153,7 @@ export function sanitizeSurfaceKey(surfaceKey: string): string { */ export function sanitizeModelName(shortName: string): string { // Map internal variants to public equivalents based on model family + if (shortName.includes('fable-5-1')) return 'claude-fable-5-1' if (shortName.includes('fable-5')) return 'claude-fable-5' if (shortName.includes('opus-4-8')) return 'claude-opus-4-8' if (shortName.includes('opus-4-6')) return 'claude-opus-4-7' diff --git a/src/utils/model/configs.ts b/src/utils/model/configs.ts index 1331d080..2034b2d9 100644 --- a/src/utils/model/configs.ts +++ b/src/utils/model/configs.ts @@ -46,6 +46,14 @@ export const CLAUDE_FABLE_5_CONFIG = { azureOpenAI: 'claude-fable-5', } as const satisfies ModelConfig +export const CLAUDE_FABLE_5_1_CONFIG = { + firstParty: 'claude-fable-5-1', + bedrock: 'anthropic.claude-fable-5-1', + vertex: 'claude-fable-5-1', + foundry: 'claude-fable-5-1', + azureOpenAI: 'claude-fable-5-1', +} as const satisfies ModelConfig + export const CLAUDE_SONNET_5_CONFIG = { firstParty: 'claude-sonnet-5', bedrock: 'anthropic.claude-sonnet-5', @@ -145,6 +153,7 @@ export const GPT_5_4_CODEX_CONFIG = { // @[MODEL LAUNCH]: Register the new config here. export const ALL_MODEL_CONFIGS = { fable5: CLAUDE_FABLE_5_CONFIG, + fable51: CLAUDE_FABLE_5_1_CONFIG, haiku35: CLAUDE_3_5_HAIKU_CONFIG, haiku45: CLAUDE_HAIKU_4_5_CONFIG, sonnet35: CLAUDE_3_5_V2_SONNET_CONFIG, diff --git a/src/utils/model/fable.test.ts b/src/utils/model/fable.test.ts index b6037e6f..49b5a1e6 100644 --- a/src/utils/model/fable.test.ts +++ b/src/utils/model/fable.test.ts @@ -7,6 +7,7 @@ import { getHardcodedTeammateModelFallback } from '../swarm/teammateModel.js' import { getAgentModel, getAgentModelOptions } from './agent.js' import { isModelAlias, isModelFamilyAlias } from './aliases.js' import { + CLAUDE_FABLE_5_1_CONFIG, CLAUDE_FABLE_5_CONFIG, CLAUDE_OPUS_4_6_CONFIG, CLAUDE_OPUS_4_8_CONFIG, @@ -21,11 +22,13 @@ import { renderDefaultModelSetting, } from './model.js' import { getModelStrings } from './modelStrings.js' +import { getModelOptions } from './modelOptions.js' import { get3PModelCapabilityOverride } from './modelSupportOverrides.js' const ENV_KEYS = [ 'ANTHROPIC_BASE_URL', 'ANTHROPIC_API_KEY', + 'ANTHROPIC_MODEL', 'ANTHROPIC_DEFAULT_FABLE_MODEL', 'ANTHROPIC_DEFAULT_FABLE_MODEL_SUPPORTED_CAPABILITIES', 'CLAUDE_CODE_DISABLE_1M_CONTEXT', @@ -50,6 +53,14 @@ afterEach(() => { describe('Fable model configuration', () => { test('uses the official provider IDs', () => { + expect(CLAUDE_FABLE_5_1_CONFIG).toEqual({ + firstParty: 'claude-fable-5-1', + bedrock: 'anthropic.claude-fable-5-1', + vertex: 'claude-fable-5-1', + foundry: 'claude-fable-5-1', + azureOpenAI: 'claude-fable-5-1', + }) + expect(getModelStrings().fable51).toBe('claude-fable-5-1') expect(CLAUDE_FABLE_5_CONFIG).toEqual({ firstParty: 'claude-fable-5', bedrock: 'anthropic.claude-fable-5', @@ -121,6 +132,31 @@ describe('Fable model configuration', () => { ) }) + test('keeps Fable 5.1 identity distinct from the Fable 5 default', () => { + expect(firstPartyNameToCanonical('us.anthropic.claude-fable-5-1-v1:0')).toBe( + 'claude-fable-5-1', + ) + expect(getPublicModelDisplayName('claude-fable-5-1')).toBe('Fable 5.1') + expect(getPublicModelDisplayName('claude-fable-5-1[1m]')).toBe('Fable 5.1 (1M context)') + expect(getMarketingNameForModel('claude-fable-5-1[1m]')).toBe( + 'Fable 5.1 (with 1M context)', + ) + expect(parseUserSpecifiedModel('fable')).toBe('claude-fable-5') + expect(parseUserSpecifiedModel('claude-fable-5-1')).toBe('claude-fable-5-1') + expect(modelSupports1M('claude-fable-5-1')).toBe(true) + expect(getContextWindowForModel('claude-fable-5-1')).toBe(1_000_000) + }) + + test('does not recommend downgrading a pinned Fable 5.1 to the Fable 5 alias', () => { + process.env.ANTHROPIC_API_KEY = 'test-api-key' + process.env.ANTHROPIC_MODEL = 'claude-fable-5-1' + expect(getModelOptions()).toContainEqual({ + value: 'claude-fable-5-1', + label: 'Fable 5.1', + description: 'claude-fable-5-1', + }) + }) + test('publishes current model IDs and knowledge cutoffs in environment context', async () => { expect(SKILL_MODEL_VARS).toMatchObject({ OPUS_ID: 'claude-opus-4-8', @@ -140,6 +176,7 @@ describe('Fable model configuration', () => { }) test('sanitizes new model trailers and keeps teammate fallbacks provider-safe', () => { + expect(sanitizeModelName('claude-fable-5-1-experimental')).toBe('claude-fable-5-1') expect(sanitizeModelName('claude-fable-5-experimental')).toBe('claude-fable-5') expect(sanitizeModelName('claude-opus-4-8-experimental')).toBe('claude-opus-4-8') expect(sanitizeModelName('claude-sonnet-5-experimental')).toBe('claude-sonnet-5') diff --git a/src/utils/model/model.ts b/src/utils/model/model.ts index 3ebdb1d4..3f3c34a2 100644 --- a/src/utils/model/model.ts +++ b/src/utils/model/model.ts @@ -267,6 +267,9 @@ export function firstPartyNameToCanonical(name: ModelName): ModelShortName { name = name.toLowerCase() // Special cases for Claude 4+ models to differentiate versions // Order matters: check more specific versions first (4-5 before 4) + if (name.includes('claude-fable-5-1')) { + return 'claude-fable-5-1' + } if (name.includes('claude-fable-5')) { return 'claude-fable-5' } @@ -416,6 +419,10 @@ export function getPublicModelDisplayName(model: ModelName): string | null { return openAIModelName } switch (model) { + case getModelStrings().fable51: + return 'Fable 5.1' + case getModelStrings().fable51 + '[1m]': + return 'Fable 5.1 (1M context)' case getModelStrings().fable5: return 'Fable 5' case getModelStrings().fable5 + '[1m]': @@ -662,6 +669,9 @@ export function getMarketingNameForModel(modelId: string): string | undefined { const has1m = modelId.toLowerCase().includes('[1m]') const canonical = getCanonicalName(modelId) + if (canonical.includes('claude-fable-5-1')) { + return has1m ? 'Fable 5.1 (with 1M context)' : 'Fable 5.1' + } if (canonical.includes('claude-fable-5')) { return has1m ? 'Fable 5 (with 1M context)' : 'Fable 5' } diff --git a/src/utils/model/modelContextWindows.ts b/src/utils/model/modelContextWindows.ts index 47ba4751..33156263 100644 --- a/src/utils/model/modelContextWindows.ts +++ b/src/utils/model/modelContextWindows.ts @@ -3,6 +3,7 @@ export const MODEL_CONTEXT_WINDOW_MIN = 16_000 export const MODEL_CONTEXT_WINDOW_MAX = 10_000_000 const DIRECT_MODEL_CONTEXT_WINDOWS: Record = { + 'claude-fable-5-1': 1_000_000, 'claude-fable-5': 1_000_000, 'claude-opus-4-8': 1_000_000, 'claude-opus-4-7': 1_000_000, diff --git a/src/utils/model/modelOptions.ts b/src/utils/model/modelOptions.ts index 9ab9d366..6185c349 100644 --- a/src/utils/model/modelOptions.ts +++ b/src/utils/model/modelOptions.ts @@ -539,7 +539,16 @@ function getModelFamilyInfo( // Fable family if (canonical.includes('claude-fable-5')) { - const currentName = getMarketingNameForModel(getDefaultFableModel()) + const defaultModel = getDefaultFableModel() + // The alias can lag a newly selectable release; a different version is not + // necessarily an upgrade for a user who has already pinned Fable 5.1. + if ( + canonical === 'claude-fable-5-1' && + getCanonicalName(defaultModel) === 'claude-fable-5' + ) { + return null + } + const currentName = getMarketingNameForModel(defaultModel) if (currentName) { return { alias: 'Fable', currentVersionName: currentName } } diff --git a/src/utils/modelCost.test.ts b/src/utils/modelCost.test.ts new file mode 100644 index 00000000..d0cb0148 --- /dev/null +++ b/src/utils/modelCost.test.ts @@ -0,0 +1,32 @@ +import { describe, expect, test } from 'bun:test' +import { calculateCostFromTokens, getModelPricingString } from './modelCost.js' + +describe('Fable 5.1 query costs', () => { + const cacheReads = { + inputTokens: 0, + outputTokens: 0, + cacheReadInputTokens: 1_000_000, + cacheCreationInputTokens: 0, + } + + test('charges $0.25 per million cached tokens for Fable 5.1', () => { + for (const model of [ + 'claude-fable-5-1', + 'claude-fable-5-1[1m]', + 'us.anthropic.claude-fable-5-1-v1:0', + ]) { + expect(calculateCostFromTokens(model, cacheReads)).toBe(0.25) + } + expect(calculateCostFromTokens('claude-fable-5', cacheReads)).toBe(1) + }) + + test('keeps input, output and cache-write pricing at the published rates', () => { + expect(calculateCostFromTokens('claude-fable-5-1', { + inputTokens: 1_000_000, + outputTokens: 1_000_000, + cacheReadInputTokens: 1_000_000, + cacheCreationInputTokens: 1_000_000, + })).toBe(72.75) + expect(getModelPricingString('claude-fable-5-1')).toBe('$10/$50 per Mtok') + }) +}) diff --git a/src/utils/modelCost.ts b/src/utils/modelCost.ts index c395c9ca..fd4035c1 100644 --- a/src/utils/modelCost.ts +++ b/src/utils/modelCost.ts @@ -8,6 +8,7 @@ import { CLAUDE_3_5_V2_SONNET_CONFIG, CLAUDE_3_7_SONNET_CONFIG, CLAUDE_FABLE_5_CONFIG, + CLAUDE_FABLE_5_1_CONFIG, CLAUDE_HAIKU_4_5_CONFIG, CLAUDE_OPUS_4_1_CONFIG, CLAUDE_OPUS_4_5_CONFIG, @@ -71,6 +72,12 @@ export const COST_TIER_10_50 = { webSearchRequests: 0.01, } as const satisfies ModelCosts +// Fable 5.1 keeps Fable 5 input/output pricing but reduces cache reads by 75%. +export const COST_FABLE_51 = { + ...COST_TIER_10_50, + promptCacheReadTokens: 0.25, +} as const satisfies ModelCosts + // Fast mode pricing for Opus 4.7: $30 input / $150 output per Mtok export const COST_TIER_30_150 = { inputTokens: 30, @@ -114,6 +121,8 @@ export function getOpus46CostTier(fastMode: boolean): ModelCosts { // Costs from https://platform.claude.com/docs/en/about-claude/pricing // Web search cost: $10 per 1000 requests = $0.01 per request export const MODEL_COSTS: Record = { + [firstPartyNameToCanonical(CLAUDE_FABLE_5_1_CONFIG.firstParty)]: + COST_FABLE_51, [firstPartyNameToCanonical(CLAUDE_FABLE_5_CONFIG.firstParty)]: COST_TIER_10_50, [firstPartyNameToCanonical(CLAUDE_3_5_HAIKU_CONFIG.firstParty)]: diff --git a/src/utils/permissions/permissionExplainer.ts b/src/utils/permissions/permissionExplainer.ts index e2ef3ed0..25cb0139 100644 --- a/src/utils/permissions/permissionExplainer.ts +++ b/src/utils/permissions/permissionExplainer.ts @@ -141,7 +141,7 @@ export function isPermissionExplainerEnabled(): boolean { } /** - * Generate a permission explanation using Haiku with structured output. + * Generate a permission explanation using the selected model with structured output. * Returns null if the feature is disabled, request is aborted, or an error occurs. */ export async function generatePermissionExplanation({ @@ -174,7 +174,7 @@ Explain this command in context.` const model = getMainLoopModel() - // Use sideQuery with forced tool choice for guaranteed structured output + // sideQuery adapts forced tool choice for models with required thinking. const response = await sideQuery({ model, system: SYSTEM_PROMPT, @@ -191,7 +191,9 @@ Explain this command in context.` ) // Extract structured data from tool use block - const toolUseBlock = response.content.find(c => c.type === 'tool_use') + const toolUseBlock = response.content.find(c => + c.type === 'tool_use' && c.name === EXPLAIN_COMMAND_TOOL.name, + ) if (toolUseBlock && toolUseBlock.type === 'tool_use') { logForDebugging( `Permission explainer: tool input: ${jsonStringify(toolUseBlock.input).slice(0, 500)}`, diff --git a/src/utils/permissions/permissions.autoMode.test.ts b/src/utils/permissions/permissions.autoMode.test.ts index ee705a30..bf14783a 100644 --- a/src/utils/permissions/permissions.autoMode.test.ts +++ b/src/utils/permissions/permissions.autoMode.test.ts @@ -1,4 +1,4 @@ -import { afterAll, afterEach, beforeAll, describe, expect, mock, test } from 'bun:test' +import { afterAll, afterEach, beforeAll, beforeEach, describe, expect, mock, spyOn, test } from 'bun:test' import { feature } from 'bun:bundle' import { mkdtemp, rm } from 'node:fs/promises' import { tmpdir } from 'node:os' @@ -22,11 +22,11 @@ let classifierMode: 'allow' | 'block' | 'parse-failure' | 'unavailable' = let configDir = '' const actualSideQuery = await import('../sideQuery.js') -mock.module('../sideQuery.js', () => ({ - ...actualSideQuery, - sideQuery: async () => { + +beforeEach(() => { + spyOn(actualSideQuery, 'sideQuery').mockImplementation(async () => { if (classifierMode === 'unavailable') throw new Error('classifier offline') - const content = + const content: Awaited>['content'] = classifierMode === 'parse-failure' ? [{ type: 'text', text: 'not structured' }] : [ @@ -55,9 +55,9 @@ mock.module('../sideQuery.js', () => ({ cache_read_input_tokens: 0, cache_creation_input_tokens: 0, }, - } - }, -})) + } as Awaited> + }) +}) const { hasPermissionsToUseTool } = await import('./permissions.js') @@ -67,6 +67,7 @@ beforeAll(async () => { }) afterEach(() => { + mock.restore() classifierMode = 'allow' resetSettingsCache() resetAutoModeState() diff --git a/src/utils/sideQuery.test.ts b/src/utils/sideQuery.test.ts new file mode 100644 index 00000000..e0e42f3e --- /dev/null +++ b/src/utils/sideQuery.test.ts @@ -0,0 +1,162 @@ +import { expect, test } from 'bun:test' +import { mkdtemp, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { createSandboxedTestEnvironment } from '../../scripts/pr/test-environment.js' +import { getIsInteractive, setIsInteractive } from '../bootstrap/state.js' +import { enableConfigs } from './config.js' +import { get3PModelCapabilityOverride } from './model/modelSupportOverrides.js' +import { generatePermissionExplanation } from './permissions/permissionExplainer.js' +import { sideQuery } from './sideQuery.js' + +const explanation = { + explanation: 'Lists the working directory.', + reasoning: 'I need to inspect the project files.', + risk: 'No files are changed.', + riskLevel: 'LOW', +} + +async function withCapturedRequests( + model: string, + run: () => Promise, + responseContent: unknown[] = [{ + type: 'tool_use', id: 'toolu_fixture', name: 'explain_command', input: explanation, + }], +) { + const requests: Record[] = [] + const headers: Headers[] = [] + const server = Bun.serve({ + hostname: '127.0.0.1', + port: 0, + async fetch(request) { + const body = await request.json() as Record + requests.push(body) + headers.push(request.headers) + if (model === 'claude-fable-5-1' && + (body.tool_choice?.type === 'tool' || body.tool_choice?.type === 'any' || + body.thinking?.type === 'disabled' || 'temperature' in body)) { + return Response.json({ type: 'error', error: { + type: 'invalid_request_error', message: 'Unsupported Fable request controls', + } }, { status: 400 }) + } + return Response.json({ + id: 'msg_fixture', type: 'message', role: 'assistant', model, + content: responseContent, stop_reason: 'end_turn', stop_sequence: null, + usage: { input_tokens: 1, output_tokens: 1 }, + }) + }, + }) + const sandbox = await mkdtemp(join(tmpdir(), 'cc-haha-side-query-')) + const originalEnv = { ...process.env } + const interactive = getIsInteractive() + try { + for (const key of Object.keys(process.env)) delete process.env[key] + Object.assign(process.env, createSandboxedTestEnvironment(sandbox, { + NODE_ENV: 'production', + ANTHROPIC_API_KEY: 'loopback-test-key', + ANTHROPIC_BASE_URL: `http://127.0.0.1:${server.port}`, + ANTHROPIC_MODEL: model, + CC_HAHA_SEND_DISABLED_THINKING: '1', + CLAUDE_CODE_DISABLE_NONESSENTIAL_TRAFFIC: '1', + }, originalEnv)) + setIsInteractive(false) + get3PModelCapabilityOverride.cache.clear() + enableConfigs() + return { result: await run(), requests, headers } + } finally { + for (const key of Object.keys(process.env)) delete process.env[key] + Object.assign(process.env, originalEnv) + setIsInteractive(interactive) + get3PModelCapabilityOverride.cache.clear() + server.stop(true) + await rm(sandbox, { recursive: true, force: true }) + } +} + +test('permission explanations work with Fable 5.1 without forced tool choice', async () => { + const { result, requests, headers } = await withCapturedRequests('claude-fable-5-1', () => + generatePermissionExplanation({ + toolName: 'Bash', toolInput: { command: 'ls' }, + signal: new AbortController().signal, + })) + expect(result).toEqual(explanation) + expect(requests).toHaveLength(1) + expect(requests[0]?.tool_choice).toEqual({ type: 'auto', disable_parallel_tool_use: true }) + expect(requests[0]?.thinking).toEqual({ + type: 'adaptive', block_binding: { prefix_mismatch_behavior: 'drop_block' }, + }) + expect(headers[0]?.get('anthropic-beta')).toContain('thinking-binding-controls-2026-08-01') + expect(JSON.stringify(requests[0]?.system)).toContain('explain_command') + expect(requests[0]?.tools.map((tool: any) => tool.name)).toEqual(['explain_command']) +}) + +test('Fable side classifiers retain output headroom and omit incompatible sampling controls', async () => { + const { requests } = await withCapturedRequests('claude-fable-5-1', () => sideQuery({ + model: 'claude-fable-5-1', messages: [{ role: 'user', content: 'Classify action' }], + thinking: false, temperature: 0, max_tokens: 64, + stop_sequences: [''], querySource: 'auto_mode', + }), [{ type: 'text', text: 'no' }]) + expect(requests[0]?.thinking).toMatchObject({ type: 'adaptive' }) + expect(requests[0]).not.toHaveProperty('temperature') + expect(requests[0]?.max_tokens).toBeGreaterThanOrEqual(2112) + expect(requests[0]?.stop_sequences).toEqual(['']) +}) + +test('Fable side-query thinking headroom respects the output limit without reducing caller budgets', async () => { + for (const [requested, expected] of [[127_000, 128_000], [128_000, 128_000], [130_000, 130_000]]) { + const { requests } = await withCapturedRequests('claude-fable-5-1', () => sideQuery({ + model: 'claude-fable-5-1', messages: [{ role: 'user', content: 'Return result' }], + max_tokens: requested, querySource: 'permission_explainer', + })) + expect(requests[0]?.max_tokens).toBe(expected) + } +}) + +test('permission explanation rejects missing or invalid tool results', async () => { + for (const content of [ + [{ type: 'text', text: 'The command is safe' }], + [{ type: 'tool_use', id: 'toolu_bad', name: 'explain_command', input: { riskLevel: 'LOW' } }], + [{ type: 'tool_use', id: 'toolu_wrong', name: 'unrelated_tool', input: explanation }], + ]) { + const { result, requests } = await withCapturedRequests('claude-fable-5-1', () => + generatePermissionExplanation({ toolName: 'Bash', toolInput: 'ls', + signal: new AbortController().signal }), content) + expect(result).toBeNull() + expect(requests).toHaveLength(1) + } +}) + +test('Fable any-tool side queries retain all choices and require a tool result in the prompt', async () => { + const tools = ['first_result', 'second_result'].map(name => ({ + name, input_schema: { type: 'object' as const, properties: {} }, + })) + const { requests } = await withCapturedRequests('claude-fable-5-1', () => sideQuery({ + model: 'claude-fable-5-1', messages: [{ role: 'user', content: 'Return result' }], + tools, tool_choice: { type: 'any' }, thinking: 2048, max_tokens: 4096, + querySource: 'permission_explainer', + })) + expect(requests[0]?.tool_choice).toEqual({ type: 'auto', disable_parallel_tool_use: true }) + expect(requests[0]?.tools).toEqual(tools) + expect(JSON.stringify(requests[0]?.system)).toContain('calling one of the provided tools') + expect(requests[0]?.thinking.type).toBe('adaptive') + expect(requests[0]?.max_tokens).toBe(4096) +}) + +test('Fable 5 side queries require thinking without the 5.1 binding controls', async () => { + const { requests, headers } = await withCapturedRequests('claude-fable-5', () => sideQuery({ + model: 'claude-fable-5', messages: [{ role: 'user', content: 'Return result' }], + thinking: false, querySource: 'permission_explainer', + })) + expect(requests[0]?.thinking).toEqual({ type: 'adaptive' }) + expect(headers[0]?.get('anthropic-beta') ?? '').not.toContain('thinking-binding-controls') +}) + +test('existing non-Fable side queries preserve forced tools and disabled thinking', async () => { + const { result, requests } = await withCapturedRequests('claude-sonnet-4-6', () => + generatePermissionExplanation({ toolName: 'Bash', toolInput: 'ls', + signal: new AbortController().signal })) + expect(result).toEqual(explanation) + expect(requests[0]?.tool_choice).toEqual({ type: 'tool', name: 'explain_command' }) + expect(requests[0]?.thinking).toEqual({ type: 'disabled' }) + expect(requests[0]?.max_tokens).toBe(1024) +}) diff --git a/src/utils/sideQuery.ts b/src/utils/sideQuery.ts index 76b7983e..f54812ef 100644 --- a/src/utils/sideQuery.ts +++ b/src/utils/sideQuery.ts @@ -4,7 +4,10 @@ import { getLastApiCompletionTimestamp, setLastApiCompletionTimestamp, } from '../bootstrap/state.js' -import { STRUCTURED_OUTPUTS_BETA_HEADER } from '../constants/betas.js' +import { + STRUCTURED_OUTPUTS_BETA_HEADER, + THINKING_BINDING_CONTROLS_BETA_HEADER, +} from '../constants/betas.js' import { CLAUDE_CODE_COMPAT_VERSION } from '../constants/claudeCodeCompatibility.js' import type { QuerySource } from '../constants/querySource.js' import { @@ -17,9 +20,15 @@ import { getAPIMetadata } from '../services/api/claude.js' import { getAnthropicClient } from '../services/api/client.js' import { normalizeUsage } from '../services/api/emptyUsage.js' import { getModelBetas, modelSupportsStructuredOutputs } from './betas.js' +import { getModelMaxOutputTokens } from './context.js' import { computeFingerprint } from './fingerprint.js' import { normalizeModelStringForAPI } from './model/model.js' -import { shouldSendExplicitDisabledThinking } from './thinking.js' +import { + modelRequiresThinking, + modelSupportsAdaptiveThinking, + modelUsesBoundThinking, + shouldSendExplicitDisabledThinking, +} from './thinking.js' type MessageParam = Anthropic.MessageParam type TextBlockParam = Anthropic.TextBlockParam @@ -169,7 +178,34 @@ export async function sideQuery(opts: SideQueryOptions): Promise { : []), ].filter((block): block is TextBlockParam => block !== null) - const thinkingConfig = resolveSideQueryThinkingConfig(thinking, max_tokens) + const requiredAdaptiveThinking = + modelRequiresThinking(model) && modelSupportsAdaptiveThinking(model) + // Side-query budgets normally cover only the structured/text answer. Required + // thinking must have headroom, including classifiers with a 64-token answer. + const maxTokens = requiredAdaptiveThinking && typeof thinking !== 'number' + ? Math.max(max_tokens, Math.min(max_tokens + 2048, getModelMaxOutputTokens(model).upperLimit)) + : max_tokens + const thinkingConfig = resolveSideQueryThinkingConfig(thinking, maxTokens, model) + if (requiredAdaptiveThinking && modelUsesBoundThinking(model)) { + betas.push(THINKING_BINDING_CONTROLS_BETA_HEADER) + } + + // Adaptive thinking cannot be combined with a forced tool choice. Keep the + // same tool-result contract, explicitly request it, and let callers validate + // the result (the permission classifier still fails closed on missing output). + const useAutoToolChoice = requiredAdaptiveThinking && + (tool_choice?.type === 'tool' || tool_choice?.type === 'any') + const requestTools = useAutoToolChoice && tool_choice?.type === 'tool' + ? tools?.filter(tool => 'name' in tool && tool.name === tool_choice.name) + : tools + if (useAutoToolChoice) { + systemBlocks.push({ + type: 'text', + text: tool_choice.type === 'tool' + ? `Respond by calling the ${tool_choice.name} tool exactly once with the complete result. Do not substitute a text response for the tool call.` + : 'Respond by calling one of the provided tools with the complete result. Do not substitute a text response for the tool call.', + }) + } const normalizedModel = normalizeModelStringForAPI(model) const start = Date.now() @@ -177,13 +213,15 @@ export async function sideQuery(opts: SideQueryOptions): Promise { const response = await client.beta.messages.create( { model: normalizedModel, - max_tokens, + max_tokens: maxTokens, system: systemBlocks, messages, - ...(tools && { tools }), - ...(tool_choice && { tool_choice }), + ...(requestTools && { tools: requestTools }), + ...(tool_choice && { tool_choice: useAutoToolChoice + ? { type: 'auto' as const, disable_parallel_tool_use: true } + : tool_choice }), ...(output_format && { output_config: { format: output_format } }), - ...(temperature !== undefined && { temperature }), + ...(temperature !== undefined && !requiredAdaptiveThinking && { temperature }), ...(stop_sequences && { stop_sequences }), ...(thinkingConfig && { thinking: thinkingConfig }), ...(betas.length > 0 && { betas }), @@ -220,7 +258,16 @@ export async function sideQuery(opts: SideQueryOptions): Promise { export function resolveSideQueryThinkingConfig( thinking: SideQueryOptions['thinking'], maxTokens: number, + model?: string, ): BetaThinkingConfigParam | undefined { + if (model && modelRequiresThinking(model) && modelSupportsAdaptiveThinking(model)) { + return { + type: 'adaptive', + ...(modelUsesBoundThinking(model) && { + block_binding: { prefix_mismatch_behavior: 'drop_block' }, + }), + } + } if ( thinking === false || (thinking === undefined && shouldSendExplicitDisabledThinking()) diff --git a/src/utils/thinking.ts b/src/utils/thinking.ts index f2abe816..5d9300ea 100644 --- a/src/utils/thinking.ts +++ b/src/utils/thinking.ts @@ -129,6 +129,11 @@ export function modelRequiresThinking(model: string): boolean { return getCanonicalName(model).includes('claude-fable-5') } +/** Fable 5.1 binds replayed thinking to the preceding system, tools and history. */ +export function modelUsesBoundThinking(model: string): boolean { + return getCanonicalName(model) === 'claude-fable-5-1' +} + export function resolveModelThinkingEnabled( model: string, requestedEnabled: boolean, diff --git a/src/utils/usageAccounting.test.ts b/src/utils/usageAccounting.test.ts index 5656a6d1..0eea2cb8 100644 --- a/src/utils/usageAccounting.test.ts +++ b/src/utils/usageAccounting.test.ts @@ -95,6 +95,19 @@ describe('isBillableUsageRecord', () => { }) describe('estimateCostUSD', () => { + it('uses the reduced Fable 5.1 cache-read rate without changing Fable 5', () => { + const cacheReads = tokens({ cacheReadInputTokens: ONE_MILLION }) + expect(estimateCostUSD('claude-fable-5-1', cacheReads)).toBe(0.25) + expect(estimateCostUSD('anthropic/claude-fable-5-1-20260825', cacheReads)).toBe(0.25) + expect(estimateCostUSD('claude-fable-5', cacheReads)).toBe(1) + expect(estimateCostUSD('claude-fable-5-1', tokens({ + inputTokens: ONE_MILLION, + outputTokens: ONE_MILLION, + cacheReadInputTokens: ONE_MILLION, + cacheCreationInputTokens: ONE_MILLION, + }))).toBe(72.75) + }) + it('bills each token bucket at its own rate', () => { const cost = estimateCostUSD('claude-opus-5', tokens({ inputTokens: ONE_MILLION, diff --git a/src/utils/usageAccounting.ts b/src/utils/usageAccounting.ts index ca8602a5..0ddf5b84 100644 --- a/src/utils/usageAccounting.ts +++ b/src/utils/usageAccounting.ts @@ -1,4 +1,5 @@ import { + COST_FABLE_51, COST_HAIKU_35, COST_HAIKU_45, COST_TIER_3_15, @@ -51,6 +52,7 @@ const MODEL_TIERS: ReadonlyArray = ['claude-opus-4-1', COST_TIER_15_75], ['claude-opus-4', COST_TIER_15_75], ['claude-fable-5', COST_TIER_10_50], + ['claude-fable-5-1', COST_FABLE_51], ['claude-mythos-5', COST_TIER_10_50], ['claude-sonnet-5', COST_TIER_3_15], ['claude-sonnet-4-6', COST_TIER_3_15], From 9e4f83bbe329345cef4c52dd61d658f02088cdf0 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=E7=A8=8B=E5=BA=8F=E5=91=98=E9=98=BF=E6=B1=9F=28Relakkes?= =?UTF-8?q?=29?= Date: Mon, 7 Sep 2026 21:41:31 +0800 Subject: [PATCH 3/4] test: stabilize PR validation and install server adapter dependencies --- .github/workflows/pr-quality.yml | 3 ++ desktop/src/pages/TraceSession.test.tsx | 37 +++++++++++++++++++------ scripts/pr/pr-quality-workflow.test.ts | 13 ++++++++- 3 files changed, 43 insertions(+), 10 deletions(-) diff --git a/.github/workflows/pr-quality.yml b/.github/workflows/pr-quality.yml index 86ef6846..d6789e14 100644 --- a/.github/workflows/pr-quality.yml +++ b/.github/workflows/pr-quality.yml @@ -128,6 +128,9 @@ jobs: bun-version-file: package.json - name: Install root dependencies run: bun install --frozen-lockfile + - name: Install adapter dependencies + working-directory: adapters + run: bun install --frozen-lockfile - name: Run root runtime checks run: bun run check:server diff --git a/desktop/src/pages/TraceSession.test.tsx b/desktop/src/pages/TraceSession.test.tsx index 684f0078..22f03c33 100644 --- a/desktop/src/pages/TraceSession.test.tsx +++ b/desktop/src/pages/TraceSession.test.tsx @@ -1,4 +1,4 @@ -import { cleanup, fireEvent, render, screen, waitFor, within } from '@testing-library/react' +import { act, cleanup, fireEvent, render, screen, waitFor, within } from '@testing-library/react' import '@testing-library/jest-dom' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import { TraceSession } from './TraceSession' @@ -503,16 +503,35 @@ describe('TraceSession', () => { }, calls: [baseTrace.calls[0]!, makeCall({ id: 'call-2', startedAt: '2026-06-09T10:00:08.000Z' })], }) - await renderReady(20) + // Advance polls only after each state is asserted. A real 20ms interval can + // invalidate the detail cache before the initial assertion under coverage. + vi.useFakeTimers({ toFake: ['setInterval', 'clearInterval'] }) + try { + await renderReady(20) - fireEvent.click(within(screen.getByTestId('trace-tree')).getByText('claude-sonnet-4-5')) - await waitFor(() => expect(sessionsApi.getTraceCall).toHaveBeenCalledTimes(1)) - await waitFor(() => expect(vi.mocked(sessionsApi.getTrace).mock.calls.length).toBeGreaterThanOrEqual(3)) + fireEvent.click(within(screen.getByTestId('trace-tree')).getByText('claude-sonnet-4-5')) + await waitFor(() => expect(sessionsApi.getTraceCall).toHaveBeenCalledTimes(1)) + expect(sessionsApi.getTrace).toHaveBeenCalledTimes(1) + expect(sessionsApi.getMessages).toHaveBeenCalledTimes(1) - await screen.findByText('claude-sonnet-4-5 x2') - expect(vi.mocked(sessionsApi.getTraceCall).mock.calls.length).toBeGreaterThan(1) - const detail = within(screen.getByTestId('trace-detail')) - expect(detail.getByRole('heading', { level: 2, name: 'claude-sonnet-4-5' })).toBeInTheDocument() + for (const expectedTraceCalls of [2, 3]) { + await act(async () => { await vi.advanceTimersByTimeAsync(20) }) + expect(sessionsApi.getTrace).toHaveBeenCalledTimes(expectedTraceCalls) + expect(sessionsApi.getMessages).toHaveBeenCalledTimes(1) + expect(screen.queryByText('claude-sonnet-4-5 x2')).not.toBeInTheDocument() + } + + await act(async () => { await vi.advanceTimersByTimeAsync(20) }) + expect(sessionsApi.getTrace).toHaveBeenCalledTimes(4) + expect(sessionsApi.getMessages).toHaveBeenCalledTimes(2) + expect(screen.getByText('claude-sonnet-4-5 x2')).toBeInTheDocument() + expect(vi.mocked(sessionsApi.getTraceCall).mock.calls.length).toBeGreaterThan(1) + const detail = within(screen.getByTestId('trace-detail')) + expect(detail.getByRole('heading', { level: 2, name: 'claude-sonnet-4-5' })).toBeInTheDocument() + } finally { + cleanup() + vi.useRealTimers() + } }) it('does not refetch full messages for identical trace polls', async () => { diff --git a/scripts/pr/pr-quality-workflow.test.ts b/scripts/pr/pr-quality-workflow.test.ts index 9acff32c..95ef9eb5 100644 --- a/scripts/pr/pr-quality-workflow.test.ts +++ b/scripts/pr/pr-quality-workflow.test.ts @@ -4,7 +4,7 @@ import { parse } from 'yaml' type WorkflowJob = { needs?: string | string[] - steps?: Array<{ name?: string; run?: string }> + steps?: Array<{ name?: string; run?: string; 'working-directory'?: string }> } function workflowJobs(workflow: string) { @@ -70,6 +70,17 @@ describe('PR quality workflow', () => { } }) + test('installs adapter dependencies before root server tests that import adapters', () => { + const jobs = workflowJobs(readFileSync('.github/workflows/pr-quality.yml', 'utf8')) + const steps = jobs['server-checks'].steps ?? [] + const adapterInstall = steps.findIndex(step => + step['working-directory'] === 'adapters' && step.run === 'bun install --frozen-lockfile', + ) + const serverCheck = steps.findIndex(step => step.run === 'bun run check:server') + expect(adapterInstall).toBeGreaterThanOrEqual(0) + expect(adapterInstall).toBeLessThan(serverCheck) + }) + test('keeps coverage artifacts observable in CI', () => { const workflow = readFileSync('.github/workflows/pr-quality.yml', 'utf8') From 8a4be0dee7888cd27ae82a66688e0557b3fbd3fa Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=E7=A8=8B=E5=BA=8F=E5=91=98=E9=98=BF=E6=B1=9F=28Relakkes?= =?UTF-8?q?=29?= Date: Mon, 7 Sep 2026 21:56:11 +0800 Subject: [PATCH 4/4] fix(ci): make runtime validation portable to Linux --- .github/workflows/pr-quality.yml | 7 +++++ scripts/pr/pr-quality-workflow.test.ts | 27 +++++++++++++------ src/server/__tests__/teams.test.ts | 12 +++++---- src/tools/TaskTools.eager.test.ts | 15 ++++++----- .../cuHelperDaemon.process.test.ts | 20 +++++++++++--- src/utils/computerUse/setup.test.ts | 15 +++++++++-- src/utils/computerUse/setup.ts | 2 +- src/utils/swarm/inProcessRunner.test.ts | 7 ++--- 8 files changed, 77 insertions(+), 28 deletions(-) diff --git a/.github/workflows/pr-quality.yml b/.github/workflows/pr-quality.yml index d6789e14..61a6c826 100644 --- a/.github/workflows/pr-quality.yml +++ b/.github/workflows/pr-quality.yml @@ -126,8 +126,13 @@ jobs: - uses: oven-sh/setup-bun@v2 with: bun-version-file: package.json + - name: Install ripgrep + run: sudo apt-get update && sudo apt-get install -y ripgrep - name: Install root dependencies run: bun install --frozen-lockfile + - name: Install desktop dependencies + working-directory: desktop + run: bun install --frozen-lockfile - name: Install adapter dependencies working-directory: adapters run: bun install --frozen-lockfile @@ -290,6 +295,8 @@ jobs: - uses: oven-sh/setup-bun@v2 with: bun-version-file: package.json + - name: Install ripgrep + run: sudo apt-get update && sudo apt-get install -y ripgrep - name: Install root dependencies run: bun install --frozen-lockfile - name: Install desktop dependencies diff --git a/scripts/pr/pr-quality-workflow.test.ts b/scripts/pr/pr-quality-workflow.test.ts index 95ef9eb5..705e432d 100644 --- a/scripts/pr/pr-quality-workflow.test.ts +++ b/scripts/pr/pr-quality-workflow.test.ts @@ -70,15 +70,26 @@ describe('PR quality workflow', () => { } }) - test('installs adapter dependencies before root server tests that import adapters', () => { + test('installs imported workspace dependencies and ripgrep before runtime and coverage tests', () => { const jobs = workflowJobs(readFileSync('.github/workflows/pr-quality.yml', 'utf8')) - const steps = jobs['server-checks'].steps ?? [] - const adapterInstall = steps.findIndex(step => - step['working-directory'] === 'adapters' && step.run === 'bun install --frozen-lockfile', - ) - const serverCheck = steps.findIndex(step => step.run === 'bun run check:server') - expect(adapterInstall).toBeGreaterThanOrEqual(0) - expect(adapterInstall).toBeLessThan(serverCheck) + for (const [job, command] of [ + ['server-checks', 'bun run check:server'], + ['coverage-checks', 'bun run check:coverage'], + ]) { + const steps = jobs[job].steps ?? [] + const check = steps.findIndex(step => step.run === command) + expect(check).toBeGreaterThanOrEqual(0) + for (const workspace of ['desktop', 'adapters']) { + const install = steps.findIndex(step => + step['working-directory'] === workspace && step.run === 'bun install --frozen-lockfile', + ) + expect(install).toBeGreaterThanOrEqual(0) + expect(install).toBeLessThan(check) + } + const ripgrep = steps.findIndex(step => step.run?.includes('apt-get install -y ripgrep')) + expect(ripgrep).toBeGreaterThanOrEqual(0) + expect(ripgrep).toBeLessThan(check) + } }) test('keeps coverage artifacts observable in CI', () => { diff --git a/src/server/__tests__/teams.test.ts b/src/server/__tests__/teams.test.ts index e102bba3..b420997c 100644 --- a/src/server/__tests__/teams.test.ts +++ b/src/server/__tests__/teams.test.ts @@ -2250,12 +2250,14 @@ describe('TeamService', () => { blocks: [], blockedBy: [], }) - expect(await readTaskListSnapshot(teamName)).toMatchObject({ + const activeSnapshot = await readTaskListSnapshot(teamName) + expect(activeSnapshot.tasks).toHaveLength(2) + expect(activeSnapshot).toMatchObject({ revision: 2, - tasks: [ - { id: '1', subject: 'Only generation-two task' }, - { id: '2', subject: 'Active generation-two writer' }, - ], + tasks: expect.arrayContaining([ + expect.objectContaining({ id: '1', subject: 'Only generation-two task' }), + expect.objectContaining({ id: '2', subject: 'Active generation-two writer' }), + ]), }) } finally { writerResource.emitDestroy() diff --git a/src/tools/TaskTools.eager.test.ts b/src/tools/TaskTools.eager.test.ts index 21f6b194..63154e72 100644 --- a/src/tools/TaskTools.eager.test.ts +++ b/src/tools/TaskTools.eager.test.ts @@ -478,27 +478,30 @@ describe('Task tool execution ordering', () => { expect(listed.data.taskListSnapshotRevision).toBe(2) expect(updated.data.taskListMutationRevision).toBe(3) expect(afterUpdate.data.taskListSnapshotRevision).toBe(3) - expect(listed.data.tasks).toEqual([ + expect(listed.data.tasks).toHaveLength(2) + expect(listed.data.tasks).toEqual(expect.arrayContaining([ expect.objectContaining({ id: taskId, status: 'pending' }), expect.objectContaining({ id: dependent.data.task.id, status: 'pending' }), - ]) - expect(afterUpdate.data.tasks).toEqual([ + ])) + expect(afterUpdate.data.tasks).toHaveLength(2) + expect(afterUpdate.data.tasks).toEqual(expect.arrayContaining([ expect.objectContaining({ id: taskId, status: 'in_progress' }), expect.objectContaining({ id: dependent.data.task.id, blockedBy: [taskId], }), - ]) + ])) appState = { ...appState, teamContext: { teamName: taskListId }, } const deleted = await TeamDeleteTool.call({}, context) + expect(deleted.data.finalTasks).toHaveLength(2) expect(deleted.data).toMatchObject({ success: true, team_name: taskListId, - finalTasks: [ + finalTasks: expect.arrayContaining([ expect.objectContaining({ id: taskId, status: 'in_progress', @@ -508,7 +511,7 @@ describe('Task tool execution ordering', () => { id: dependent.data.task.id, blockedBy: [taskId], }), - ], + ]), }) expect(Number.isFinite(Date.parse(deleted.data.taskListSnapshotAt ?? ''))).toBe(true) expect(deleted.data.taskListSnapshotRevision).toBe(3) diff --git a/src/utils/computerUse/cuHelperDaemon.process.test.ts b/src/utils/computerUse/cuHelperDaemon.process.test.ts index 1df529fa..f65c759f 100644 --- a/src/utils/computerUse/cuHelperDaemon.process.test.ts +++ b/src/utils/computerUse/cuHelperDaemon.process.test.ts @@ -34,6 +34,11 @@ const originalSessionId = getSessionId() function isAlive(pid: number): boolean { try { process.kill(pid, 0) + // A Linux container's init may defer reaping an exited orphan. A zombie + // cannot hold the socket open and is no longer a running daemon. + if (process.platform === 'linux') { + return !/^\d+ \(.*\) Z /.test(fs.readFileSync(`/proc/${pid}/stat`, 'utf8')) + } return true } catch { return false @@ -75,7 +80,14 @@ async function launchDetachedDaemon(socketPath: string): Promise { encoding: 'utf8', }) expect(parent.status).toBe(0) - expect(Number.parseInt(parent.stdout.trim(), 10)).toBe(1) + const parentPid = Number.parseInt(parent.stdout.trim(), 10) + if (process.platform === 'darwin') { + expect(parentPid).toBe(1) + } else { + // Linux can reparent to a container subreaper rather than PID 1. + expect(parentPid).toBeGreaterThan(0) + expect(parentPid).not.toBe(result.pid) + } return pid } @@ -143,8 +155,10 @@ afterEach(async () => { runtimeRoot = undefined }) -describe.skipIf(process.platform !== 'darwin')( - 'cu-helper launchd-owned daemon lifecycle', +// The fixture models launchd ownership with a detached POSIX process; it does +// not invoke launchctl or native Computer Use, so Linux exercises it too. +describe.skipIf(process.platform === 'win32')( + 'cu-helper detached daemon lifecycle', () => { test('startup enumeration cannot poison the first resumed turn across shutdown and restart', async () => { runtimeRoot = fs.mkdtempSync(path.join(os.tmpdir(), 'cc-haha-cu-process-')) diff --git a/src/utils/computerUse/setup.test.ts b/src/utils/computerUse/setup.test.ts index d24396bc..6f8f8754 100644 --- a/src/utils/computerUse/setup.test.ts +++ b/src/utils/computerUse/setup.test.ts @@ -17,7 +17,8 @@ describe('setupComputerUseMCP runtime capability', () => { }) expect(Object.keys(result.mcpConfig)).toEqual(['computer-use']) - expect(result.allowedTools.length).toBeGreaterThan(0) + expect(result.allowedTools).toContain('mcp__computer-use__get_app_state') + expect(result.allowedTools).not.toContain('mcp__computer-use__screenshot') }) test('keeps the Windows compatibility engine available without a macOS helper', () => { @@ -29,6 +30,16 @@ describe('setupComputerUseMCP runtime capability', () => { }) expect(Object.keys(result.mcpConfig)).toEqual(['computer-use']) - expect(result.allowedTools.length).toBeGreaterThan(0) + expect(result.allowedTools).toContain('mcp__computer-use__screenshot') + expect(result.allowedTools).not.toContain('mcp__computer-use__get_app_state') + }) + + test('does not resolve a helper or advertise tools on unsupported platforms', () => { + expect(setupComputerUseMCP({ + platform: 'linux', + resolveMacosNativeBinary: () => { + throw new Error('must not resolve a macOS helper on Linux') + }, + })).toEqual({ mcpConfig: {}, allowedTools: [] }) }) }) diff --git a/src/utils/computerUse/setup.ts b/src/utils/computerUse/setup.ts index 15ad54e5..2b8a7ca3 100644 --- a/src/utils/computerUse/setup.ts +++ b/src/utils/computerUse/setup.ts @@ -41,7 +41,7 @@ export function setupComputerUseMCP(deps: SetupComputerUseDeps = {}): { } const allowedTools = buildPlatformComputerUseTools( - getCliComputerUseCapabilities(), + getCliComputerUseCapabilities(platform), getChicagoCoordinateMode(), ).map(t => buildMcpToolName(COMPUTER_USE_MCP_SERVER_NAME, t.name)) diff --git a/src/utils/swarm/inProcessRunner.test.ts b/src/utils/swarm/inProcessRunner.test.ts index cbc4fe61..3714559f 100644 --- a/src/utils/swarm/inProcessRunner.test.ts +++ b/src/utils/swarm/inProcessRunner.test.ts @@ -692,7 +692,7 @@ describe('in-process teammate task claiming', () => { status: 'in_progress', }) await updateTask(taskListId, explicitAssignment, { status: 'completed' }) - await createTask(taskListId, { + const followUpTaskId = await createTask(taskListId, { subject: 'Audit workflow', description: 'Audit workflow changes', status: 'pending', @@ -711,8 +711,9 @@ describe('in-process teammate task claiming', () => { expect(prompt).toContain('Audit workflow') const claimedTasks = await listTasks(taskListId) - expect(claimedTasks[1]?.owner).toBe(agentName) - expect(claimedTasks[1]?.status).toBe('in_progress') + const claimedTask = claimedTasks.find(task => task.id === followUpTaskId) + expect(claimedTask?.owner).toBe(agentName) + expect(claimedTask?.status).toBe('in_progress') const unrelatedTasks = await listTasks(parentSessionId) expect(unrelatedTasks).toHaveLength(1)