mirror of
https://github.com/NanmiCoder/claude-code-haha.git
synced 2026-10-10 20:03:13 +08:00
Merge pull request #1294 from li-new123/feat/add-fable-5-1
feat(models): support Fable 5.1 with runtime compatibility
This commit is contained in:
@@ -126,8 +126,16 @@ jobs:
|
||||
- uses: oven-sh/setup-bun@v2
|
||||
with:
|
||||
bun-version-file: package.json
|
||||
- name: Install ripgrep
|
||||
run: sudo apt-get update && sudo apt-get install -y ripgrep
|
||||
- name: Install root dependencies
|
||||
run: bun install --frozen-lockfile
|
||||
- name: Install desktop dependencies
|
||||
working-directory: desktop
|
||||
run: bun install --frozen-lockfile
|
||||
- name: Install adapter dependencies
|
||||
working-directory: adapters
|
||||
run: bun install --frozen-lockfile
|
||||
- name: Run root runtime checks
|
||||
run: bun run check:server
|
||||
|
||||
@@ -287,6 +295,8 @@ jobs:
|
||||
- uses: oven-sh/setup-bun@v2
|
||||
with:
|
||||
bun-version-file: package.json
|
||||
- name: Install ripgrep
|
||||
run: sudo apt-get update && sudo apt-get install -y ripgrep
|
||||
- name: Install root dependencies
|
||||
run: bun install --frozen-lockfile
|
||||
- name: Install desktop dependencies
|
||||
|
||||
@@ -93,7 +93,7 @@ describe('ModelSelector', () => {
|
||||
|
||||
await clickByRole(/Opus 4\.7/i)
|
||||
|
||||
expect(screen.getByRole('button', { name: /Fable 5/ })).toBeInTheDocument()
|
||||
expect(screen.getByRole('button', { name: /Fable 5\.1/ })).toBeInTheDocument()
|
||||
expect(screen.getByRole('button', { name: /Opus 4\.8/ })).toBeInTheDocument()
|
||||
expect(screen.getByRole('button', { name: /Sonnet 5/ })).toBeInTheDocument()
|
||||
expect(screen.getAllByRole('button', { name: /Opus 4\.7/ }).length).toBeGreaterThan(0)
|
||||
|
||||
@@ -3,6 +3,14 @@ import type { ModelInfo } from '../types/settings'
|
||||
export const OFFICIAL_DEFAULT_MODEL_ID = 'claude-opus-4-8'
|
||||
|
||||
export const OFFICIAL_MODELS: ModelInfo[] = [
|
||||
{
|
||||
id: 'claude-fable-5-1',
|
||||
name: 'Fable 5.1',
|
||||
description: 'Highest capability for long-running tasks',
|
||||
context: '1m',
|
||||
defaultReasoningEffort: 'high',
|
||||
supportedReasoningEfforts: ['low', 'medium', 'high', 'xhigh', 'max'],
|
||||
},
|
||||
{
|
||||
id: 'claude-fable-5',
|
||||
name: 'Fable 5',
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
import { cleanup, fireEvent, render, screen, waitFor, within } from '@testing-library/react'
|
||||
import { act, cleanup, fireEvent, render, screen, waitFor, within } from '@testing-library/react'
|
||||
import '@testing-library/jest-dom'
|
||||
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
|
||||
import { TraceSession } from './TraceSession'
|
||||
@@ -503,16 +503,35 @@ describe('TraceSession', () => {
|
||||
},
|
||||
calls: [baseTrace.calls[0]!, makeCall({ id: 'call-2', startedAt: '2026-06-09T10:00:08.000Z' })],
|
||||
})
|
||||
await renderReady(20)
|
||||
// Advance polls only after each state is asserted. A real 20ms interval can
|
||||
// invalidate the detail cache before the initial assertion under coverage.
|
||||
vi.useFakeTimers({ toFake: ['setInterval', 'clearInterval'] })
|
||||
try {
|
||||
await renderReady(20)
|
||||
|
||||
fireEvent.click(within(screen.getByTestId('trace-tree')).getByText('claude-sonnet-4-5'))
|
||||
await waitFor(() => expect(sessionsApi.getTraceCall).toHaveBeenCalledTimes(1))
|
||||
await waitFor(() => expect(vi.mocked(sessionsApi.getTrace).mock.calls.length).toBeGreaterThanOrEqual(3))
|
||||
fireEvent.click(within(screen.getByTestId('trace-tree')).getByText('claude-sonnet-4-5'))
|
||||
await waitFor(() => expect(sessionsApi.getTraceCall).toHaveBeenCalledTimes(1))
|
||||
expect(sessionsApi.getTrace).toHaveBeenCalledTimes(1)
|
||||
expect(sessionsApi.getMessages).toHaveBeenCalledTimes(1)
|
||||
|
||||
await screen.findByText('claude-sonnet-4-5 x2')
|
||||
expect(vi.mocked(sessionsApi.getTraceCall).mock.calls.length).toBeGreaterThan(1)
|
||||
const detail = within(screen.getByTestId('trace-detail'))
|
||||
expect(detail.getByRole('heading', { level: 2, name: 'claude-sonnet-4-5' })).toBeInTheDocument()
|
||||
for (const expectedTraceCalls of [2, 3]) {
|
||||
await act(async () => { await vi.advanceTimersByTimeAsync(20) })
|
||||
expect(sessionsApi.getTrace).toHaveBeenCalledTimes(expectedTraceCalls)
|
||||
expect(sessionsApi.getMessages).toHaveBeenCalledTimes(1)
|
||||
expect(screen.queryByText('claude-sonnet-4-5 x2')).not.toBeInTheDocument()
|
||||
}
|
||||
|
||||
await act(async () => { await vi.advanceTimersByTimeAsync(20) })
|
||||
expect(sessionsApi.getTrace).toHaveBeenCalledTimes(4)
|
||||
expect(sessionsApi.getMessages).toHaveBeenCalledTimes(2)
|
||||
expect(screen.getByText('claude-sonnet-4-5 x2')).toBeInTheDocument()
|
||||
expect(vi.mocked(sessionsApi.getTraceCall).mock.calls.length).toBeGreaterThan(1)
|
||||
const detail = within(screen.getByTestId('trace-detail'))
|
||||
expect(detail.getByRole('heading', { level: 2, name: 'claude-sonnet-4-5' })).toBeInTheDocument()
|
||||
} finally {
|
||||
cleanup()
|
||||
vi.useRealTimers()
|
||||
}
|
||||
})
|
||||
|
||||
it('does not refetch full messages for identical trace polls', async () => {
|
||||
|
||||
@@ -4,7 +4,7 @@ import { parse } from 'yaml'
|
||||
|
||||
type WorkflowJob = {
|
||||
needs?: string | string[]
|
||||
steps?: Array<{ name?: string; run?: string }>
|
||||
steps?: Array<{ name?: string; run?: string; 'working-directory'?: string }>
|
||||
}
|
||||
|
||||
function workflowJobs(workflow: string) {
|
||||
@@ -70,6 +70,28 @@ describe('PR quality workflow', () => {
|
||||
}
|
||||
})
|
||||
|
||||
test('installs imported workspace dependencies and ripgrep before runtime and coverage tests', () => {
|
||||
const jobs = workflowJobs(readFileSync('.github/workflows/pr-quality.yml', 'utf8'))
|
||||
for (const [job, command] of [
|
||||
['server-checks', 'bun run check:server'],
|
||||
['coverage-checks', 'bun run check:coverage'],
|
||||
]) {
|
||||
const steps = jobs[job].steps ?? []
|
||||
const check = steps.findIndex(step => step.run === command)
|
||||
expect(check).toBeGreaterThanOrEqual(0)
|
||||
for (const workspace of ['desktop', 'adapters']) {
|
||||
const install = steps.findIndex(step =>
|
||||
step['working-directory'] === workspace && step.run === 'bun install --frozen-lockfile',
|
||||
)
|
||||
expect(install).toBeGreaterThanOrEqual(0)
|
||||
expect(install).toBeLessThan(check)
|
||||
}
|
||||
const ripgrep = steps.findIndex(step => step.run?.includes('apt-get install -y ripgrep'))
|
||||
expect(ripgrep).toBeGreaterThanOrEqual(0)
|
||||
expect(ripgrep).toBeLessThan(check)
|
||||
}
|
||||
})
|
||||
|
||||
test('keeps coverage artifacts observable in CI', () => {
|
||||
const workflow = readFileSync('.github/workflows/pr-quality.yml', 'utf8')
|
||||
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
import { afterEach, describe, expect, mock, spyOn, test } from 'bun:test'
|
||||
import { afterEach, beforeEach, describe, expect, mock, spyOn, test } from 'bun:test'
|
||||
import { feature } from 'bun:bundle'
|
||||
import {
|
||||
resetSettingsCache,
|
||||
@@ -8,14 +8,23 @@ import {
|
||||
process.env.ANTHROPIC_API_KEY = 'test-key'
|
||||
|
||||
let capturedQuery: Record<string, unknown> | undefined
|
||||
mock.module('../../utils/sideQuery.js', () => ({
|
||||
sideQuery: async (options: Record<string, unknown>) => {
|
||||
const sideQueryModule = await import('../../utils/sideQuery.js')
|
||||
|
||||
beforeEach(() => {
|
||||
spyOn(sideQueryModule, 'sideQuery').mockImplementation(async options => {
|
||||
capturedQuery = options
|
||||
return {
|
||||
id: 'msg_critique',
|
||||
type: 'message',
|
||||
role: 'assistant',
|
||||
model: 'test-model',
|
||||
stop_reason: 'end_turn',
|
||||
stop_sequence: null,
|
||||
usage: { input_tokens: 1, output_tokens: 1 },
|
||||
content: [{ type: 'text', text: 'critique complete' }],
|
||||
}
|
||||
},
|
||||
}))
|
||||
} as Awaited<ReturnType<typeof sideQueryModule.sideQuery>>
|
||||
})
|
||||
})
|
||||
|
||||
const transcriptClassifierEnabled = feature('TRANSCRIPT_CLASSIFIER')
|
||||
? true
|
||||
|
||||
@@ -3,6 +3,7 @@ import { feature } from 'bun:bundle'
|
||||
export const CLAUDE_CODE_20250219_BETA_HEADER = 'claude-code-20250219'
|
||||
export const INTERLEAVED_THINKING_BETA_HEADER =
|
||||
'interleaved-thinking-2025-05-14'
|
||||
export const THINKING_BINDING_CONTROLS_BETA_HEADER = 'thinking-binding-controls-2026-08-01'
|
||||
export const CONTEXT_1M_BETA_HEADER = 'context-1m-2025-08-07'
|
||||
export const CONTEXT_MANAGEMENT_BETA_HEADER = 'context-management-2025-06-27'
|
||||
export const STRUCTURED_OUTPUTS_BETA_HEADER = 'structured-outputs-2025-12-15'
|
||||
|
||||
@@ -374,8 +374,9 @@ describe('Business Flow: Models & Effort', () => {
|
||||
|
||||
it('should return available fallback models', async () => {
|
||||
const { data } = await api('GET', '/api/models')
|
||||
expect(data.models.length).toBe(4)
|
||||
expect(data.models.length).toBe(5)
|
||||
const names = data.models.map((m: any) => m.name)
|
||||
expect(names).toContain('Fable 5.1')
|
||||
expect(names).toContain('Fable 5')
|
||||
expect(names).toContain('Opus 4.8')
|
||||
expect(names).toContain('Sonnet 5')
|
||||
@@ -398,6 +399,19 @@ describe('Business Flow: Models & Effort', () => {
|
||||
expect(data.model.name).toBe('Opus 4.8')
|
||||
})
|
||||
|
||||
it('should select Fable 5.1 with its reasoning catalog intact', async () => {
|
||||
const { status } = await api('PUT', '/api/models/current', {
|
||||
modelId: 'claude-fable-5-1',
|
||||
})
|
||||
expect(status).toBe(200)
|
||||
const { data } = await api('GET', '/api/models/current')
|
||||
expect(data.model).toMatchObject({
|
||||
id: 'claude-fable-5-1', name: 'Fable 5.1', context: '1m',
|
||||
defaultReasoningEffort: 'high',
|
||||
supportedReasoningEfforts: ['low', 'medium', 'high', 'xhigh', 'max'],
|
||||
})
|
||||
})
|
||||
|
||||
it('should switch to Haiku 4.5', async () => {
|
||||
await api('PUT', '/api/models/current', { modelId: 'claude-haiku-4-5' })
|
||||
const { data } = await api('GET', '/api/models/current')
|
||||
|
||||
@@ -216,8 +216,8 @@ describe('E2E: Full Flow', () => {
|
||||
|
||||
it('should list available models', async () => {
|
||||
const { data } = await api('GET', '/api/models')
|
||||
expect(data.models.length).toBe(4)
|
||||
expect(data.models[0].name).toBe('Fable 5')
|
||||
expect(data.models.length).toBe(5)
|
||||
expect(data.models[0].name).toBe('Fable 5.1')
|
||||
})
|
||||
|
||||
it('should switch model', async () => {
|
||||
|
||||
@@ -713,6 +713,14 @@ describe('Models API', () => {
|
||||
expect(res.status).toBe(200)
|
||||
const body = await res.json()
|
||||
expect(body.models).toEqual([
|
||||
{
|
||||
id: 'claude-fable-5-1',
|
||||
name: 'Fable 5.1',
|
||||
description: 'Highest capability for long-running tasks',
|
||||
context: '1m',
|
||||
defaultReasoningEffort: 'high',
|
||||
supportedReasoningEfforts: ['low', 'medium', 'high', 'xhigh', 'max'],
|
||||
},
|
||||
{
|
||||
id: 'claude-fable-5',
|
||||
name: 'Fable 5',
|
||||
@@ -971,6 +979,7 @@ describe('Models API', () => {
|
||||
)
|
||||
const listBody = await listResponse.json()
|
||||
expect(listBody.models.map((model: { id: string }) => model.id)).toEqual([
|
||||
'claude-fable-5-1',
|
||||
'claude-fable-5',
|
||||
'claude-opus-4-8',
|
||||
'claude-sonnet-5',
|
||||
|
||||
@@ -2250,12 +2250,14 @@ describe('TeamService', () => {
|
||||
blocks: [],
|
||||
blockedBy: [],
|
||||
})
|
||||
expect(await readTaskListSnapshot(teamName)).toMatchObject({
|
||||
const activeSnapshot = await readTaskListSnapshot(teamName)
|
||||
expect(activeSnapshot.tasks).toHaveLength(2)
|
||||
expect(activeSnapshot).toMatchObject({
|
||||
revision: 2,
|
||||
tasks: [
|
||||
{ id: '1', subject: 'Only generation-two task' },
|
||||
{ id: '2', subject: 'Active generation-two writer' },
|
||||
],
|
||||
tasks: expect.arrayContaining([
|
||||
expect.objectContaining({ id: '1', subject: 'Only generation-two task' }),
|
||||
expect.objectContaining({ id: '2', subject: 'Active generation-two writer' }),
|
||||
]),
|
||||
})
|
||||
} finally {
|
||||
writerResource.emitDestroy()
|
||||
|
||||
@@ -50,6 +50,14 @@ import {
|
||||
// ─── Fallback models (used when no provider is configured) ────────────────────
|
||||
|
||||
const DEFAULT_MODELS = [
|
||||
{
|
||||
id: 'claude-fable-5-1',
|
||||
name: 'Fable 5.1',
|
||||
description: 'Highest capability for long-running tasks',
|
||||
context: '1m',
|
||||
defaultReasoningEffort: 'high',
|
||||
supportedReasoningEfforts: ['low', 'medium', 'high', 'xhigh', 'max'],
|
||||
},
|
||||
{
|
||||
id: 'claude-fable-5',
|
||||
name: 'Fable 5',
|
||||
|
||||
@@ -134,6 +134,7 @@ import {
|
||||
} from "src/bootstrap/state.js";
|
||||
import {
|
||||
AFK_MODE_BETA_HEADER,
|
||||
THINKING_BINDING_CONTROLS_BETA_HEADER,
|
||||
CONTEXT_1M_BETA_HEADER,
|
||||
CONTEXT_MANAGEMENT_BETA_HEADER,
|
||||
EFFORT_BETA_HEADER,
|
||||
@@ -191,6 +192,7 @@ import {
|
||||
import { endQueryProfile, queryCheckpoint } from "src/utils/queryProfiler.js";
|
||||
import {
|
||||
modelSupportsAdaptiveThinking,
|
||||
modelUsesBoundThinking,
|
||||
modelSupportsThinking,
|
||||
resolveModelThinkingEnabled,
|
||||
shouldSendExplicitDisabledThinking,
|
||||
@@ -1751,7 +1753,8 @@ async function* queryModel(
|
||||
// setting that can greatly affect model quality and bashing.
|
||||
if (hasThinking && modelCanThink) {
|
||||
if (
|
||||
!isEnvTruthy(process.env.CLAUDE_CODE_DISABLE_ADAPTIVE_THINKING) &&
|
||||
(modelUsesBoundThinking(options.model) ||
|
||||
!isEnvTruthy(process.env.CLAUDE_CODE_DISABLE_ADAPTIVE_THINKING)) &&
|
||||
modelSupportsAdaptiveThinking(options.model)
|
||||
) {
|
||||
// For models that support adaptive thinking, always use adaptive
|
||||
@@ -1781,6 +1784,19 @@ async function* queryModel(
|
||||
} as unknown as BetaMessageStreamParams['thinking']
|
||||
}
|
||||
|
||||
if (thinking?.type === 'adaptive' && modelUsesBoundThinking(options.model)) {
|
||||
// Directory, tool and compacted-history updates can invalidate old thinking.
|
||||
// Let the API retain valid blocks and drop only those bound to an old prefix.
|
||||
// This header is required for compatibility even when optional betas are disabled.
|
||||
thinking = {
|
||||
...thinking,
|
||||
block_binding: { prefix_mismatch_behavior: 'drop_block' },
|
||||
} as typeof thinking
|
||||
if (!betasParams.includes(THINKING_BINDING_CONTROLS_BETA_HEADER)) {
|
||||
betasParams.push(THINKING_BINDING_CONTROLS_BETA_HEADER)
|
||||
}
|
||||
}
|
||||
|
||||
// Get API context management strategies if enabled
|
||||
const contextManagement = getAPIContextManagement({
|
||||
hasThinking,
|
||||
|
||||
@@ -10,13 +10,18 @@ import { enableConfigs } from '../../utils/config.js'
|
||||
import { get3PModelCapabilityOverride } from '../../utils/model/modelSupportOverrides.js'
|
||||
import { buildProviderManagedEnv } from '../../server/services/providerRuntimeEnv.js'
|
||||
import type { SavedProvider } from '../../server/types/provider.js'
|
||||
import { queryWithModel } from './claude.js'
|
||||
import { queryModelWithStreaming, queryWithModel } from './claude.js'
|
||||
import { createUserMessage } from '../../utils/messages.js'
|
||||
import { asSystemPrompt } from '../../utils/systemPromptType.js'
|
||||
import { getEmptyToolPermissionContext } from '../../Tool.js'
|
||||
import type { Message } from '../../types/message.js'
|
||||
|
||||
function sseEvent(name: string, data: unknown): string {
|
||||
return `event: ${name}\ndata: ${JSON.stringify(data)}\n\n`
|
||||
}
|
||||
|
||||
function successfulResponse(model: string): string {
|
||||
function successfulResponse(model: string, withThinking = false): string {
|
||||
const textIndex = withThinking ? 1 : 0
|
||||
return [
|
||||
sseEvent('message_start', {
|
||||
type: 'message_start',
|
||||
@@ -31,17 +36,28 @@ function successfulResponse(model: string): string {
|
||||
usage: { input_tokens: 1, output_tokens: 0 },
|
||||
},
|
||||
}),
|
||||
...(withThinking ? [
|
||||
sseEvent('content_block_start', {
|
||||
type: 'content_block_start', index: 0,
|
||||
content_block: { type: 'thinking', thinking: '', signature: '' },
|
||||
}),
|
||||
sseEvent('content_block_delta', {
|
||||
type: 'content_block_delta', index: 0,
|
||||
delta: { type: 'signature_delta', signature: 'fixture-signature' },
|
||||
}),
|
||||
sseEvent('content_block_stop', { type: 'content_block_stop', index: 0 }),
|
||||
] : []),
|
||||
sseEvent('content_block_start', {
|
||||
type: 'content_block_start',
|
||||
index: 0,
|
||||
index: textIndex,
|
||||
content_block: { type: 'text', text: '' },
|
||||
}),
|
||||
sseEvent('content_block_delta', {
|
||||
type: 'content_block_delta',
|
||||
index: 0,
|
||||
index: textIndex,
|
||||
delta: { type: 'text_delta', text: 'OK' },
|
||||
}),
|
||||
sseEvent('content_block_stop', { type: 'content_block_stop', index: 0 }),
|
||||
sseEvent('content_block_stop', { type: 'content_block_stop', index: textIndex }),
|
||||
sseEvent('message_delta', {
|
||||
type: 'message_delta',
|
||||
delta: { stop_reason: 'end_turn', stop_sequence: null },
|
||||
@@ -298,6 +314,7 @@ async function captureQueryRequest({
|
||||
globalThinkingEnabled,
|
||||
provider,
|
||||
responseFactory,
|
||||
continuationSystemPrompts,
|
||||
env,
|
||||
}: {
|
||||
model: string
|
||||
@@ -307,7 +324,8 @@ async function captureQueryRequest({
|
||||
configureCapabilityOverrides?: boolean
|
||||
globalThinkingEnabled?: boolean
|
||||
provider?: SavedProvider
|
||||
responseFactory?: (model: string, body: Record<string, unknown>) => Response
|
||||
responseFactory?: (model: string, body: Record<string, unknown>, headers: Headers) => Response
|
||||
continuationSystemPrompts?: string[][]
|
||||
env?: Readonly<Record<string, string | undefined>>
|
||||
}): Promise<{
|
||||
content: unknown
|
||||
@@ -325,7 +343,7 @@ async function captureQueryRequest({
|
||||
requestHeaders.push(new Headers(request.headers))
|
||||
const body = await request.json() as Record<string, unknown>
|
||||
requests.push(body)
|
||||
return responseFactory?.(model, body) ?? new Response(successfulResponse(model), {
|
||||
return responseFactory?.(model, body, request.headers) ?? new Response(successfulResponse(model), {
|
||||
headers: { 'content-type': 'text/event-stream' },
|
||||
})
|
||||
},
|
||||
@@ -386,19 +404,46 @@ async function captureQueryRequest({
|
||||
clearCapabilityCache()
|
||||
enableConfigs()
|
||||
|
||||
const result = await queryWithModel({
|
||||
userPrompt: 'Reply exactly OK',
|
||||
signal: new AbortController().signal,
|
||||
options: {
|
||||
model,
|
||||
querySource: 'insights',
|
||||
agents: [],
|
||||
isNonInteractiveSession: true,
|
||||
hasAppendSystemPrompt: false,
|
||||
mcpTools: [],
|
||||
effortValue,
|
||||
},
|
||||
})
|
||||
const options = {
|
||||
model,
|
||||
querySource: 'insights' as const,
|
||||
agents: [],
|
||||
isNonInteractiveSession: true,
|
||||
hasAppendSystemPrompt: false,
|
||||
mcpTools: [],
|
||||
effortValue,
|
||||
}
|
||||
let result
|
||||
if (continuationSystemPrompts) {
|
||||
const history: Message[] = []
|
||||
for (const [index, systemPrompt] of [[], ...continuationSystemPrompts].entries()) {
|
||||
history.push(createUserMessage({ content: index === 0 ? 'Reply exactly OK' : 'Continue' }))
|
||||
const assistants = []
|
||||
for await (const message of queryModelWithStreaming({
|
||||
messages: history,
|
||||
systemPrompt: asSystemPrompt(systemPrompt),
|
||||
thinkingConfig: { type: 'disabled' },
|
||||
tools: [],
|
||||
signal: new AbortController().signal,
|
||||
options: {
|
||||
...options,
|
||||
enablePromptCaching: false,
|
||||
getToolPermissionContext: async () => getEmptyToolPermissionContext(),
|
||||
},
|
||||
})) {
|
||||
if (message.type === 'assistant') assistants.push(message)
|
||||
}
|
||||
history.push(...assistants)
|
||||
result = assistants.at(-1)
|
||||
}
|
||||
} else {
|
||||
result = await queryWithModel({
|
||||
userPrompt: 'Reply exactly OK',
|
||||
signal: new AbortController().signal,
|
||||
options,
|
||||
})
|
||||
}
|
||||
if (!result) throw new Error('No assistant response received')
|
||||
|
||||
return {
|
||||
content: result.message.content,
|
||||
@@ -541,6 +586,51 @@ test('normalizes a disabled parent thinking mode to adaptive for Fable', async (
|
||||
expect(requests[0]?.thinking).not.toEqual({ type: 'disabled' })
|
||||
}, 10_000)
|
||||
|
||||
for (const [disableBetas, disableAdaptive] of [[false, false], [true, false], [false, true]]) {
|
||||
test(`keeps Fable 5.1 replay usable after a system change (disable betas: ${disableBetas}, adaptive: ${disableAdaptive})`, async () => {
|
||||
let originalSystem: string | undefined
|
||||
let responseIndex = 0
|
||||
const { content, requests, requestHeaders } = await captureQueryRequest({
|
||||
model: 'claude-fable-5-1',
|
||||
configureCapabilityOverrides: false,
|
||||
continuationSystemPrompts: [[], ['Additional working directories: /tmp/project-two']],
|
||||
env: {
|
||||
CLAUDE_CODE_DISABLE_EXPERIMENTAL_BETAS: disableBetas ? '1' : undefined,
|
||||
CLAUDE_CODE_DISABLE_ADAPTIVE_THINKING: disableAdaptive ? '1' : undefined,
|
||||
},
|
||||
responseFactory: (model, body, headers) => {
|
||||
const system = JSON.stringify(body.system)
|
||||
const changed = originalSystem !== undefined && system !== originalSystem
|
||||
originalSystem ??= system
|
||||
const thinking = body.thinking as { block_binding?: { prefix_mismatch_behavior?: string } }
|
||||
if (changed && (
|
||||
thinking?.block_binding?.prefix_mismatch_behavior !== 'drop_block' ||
|
||||
!headers.get('anthropic-beta')?.includes('thinking-binding-controls-2026-08-01')
|
||||
)) {
|
||||
return Response.json({ type: 'error', error: {
|
||||
type: 'invalid_request_error', message: 'The block is bound to a different conversation',
|
||||
} }, { status: 400 })
|
||||
}
|
||||
const response = successfulResponse(model, true).replaceAll('msg_required_thinking', `msg_replay_${responseIndex++}`)
|
||||
return new Response(response, { headers: { 'content-type': 'text/event-stream' } })
|
||||
},
|
||||
})
|
||||
|
||||
expect(content).toContainEqual({ type: 'text', text: 'OK' })
|
||||
expect(requests).toHaveLength(3)
|
||||
for (let index = 0; index < requests.length; index++) {
|
||||
expect(requests[index]?.model).toBe('claude-fable-5-1')
|
||||
expect(requests[index]?.thinking).toEqual({
|
||||
type: 'adaptive', block_binding: { prefix_mismatch_behavior: 'drop_block' },
|
||||
})
|
||||
expect(requestHeaders[index]?.get('anthropic-beta')).toContain('thinking-binding-controls-2026-08-01')
|
||||
}
|
||||
for (const request of requests.slice(1)) {
|
||||
expect(JSON.stringify(request.messages)).toContain('fixture-signature')
|
||||
}
|
||||
}, 10_000)
|
||||
}
|
||||
|
||||
test('drops a tool call truncated at the output-token boundary', async () => {
|
||||
const { content, apiError, error, requests } = await captureQueryRequest({
|
||||
model: 'deepseek-v4-flash',
|
||||
|
||||
@@ -478,27 +478,30 @@ describe('Task tool execution ordering', () => {
|
||||
expect(listed.data.taskListSnapshotRevision).toBe(2)
|
||||
expect(updated.data.taskListMutationRevision).toBe(3)
|
||||
expect(afterUpdate.data.taskListSnapshotRevision).toBe(3)
|
||||
expect(listed.data.tasks).toEqual([
|
||||
expect(listed.data.tasks).toHaveLength(2)
|
||||
expect(listed.data.tasks).toEqual(expect.arrayContaining([
|
||||
expect.objectContaining({ id: taskId, status: 'pending' }),
|
||||
expect.objectContaining({ id: dependent.data.task.id, status: 'pending' }),
|
||||
])
|
||||
expect(afterUpdate.data.tasks).toEqual([
|
||||
]))
|
||||
expect(afterUpdate.data.tasks).toHaveLength(2)
|
||||
expect(afterUpdate.data.tasks).toEqual(expect.arrayContaining([
|
||||
expect.objectContaining({ id: taskId, status: 'in_progress' }),
|
||||
expect.objectContaining({
|
||||
id: dependent.data.task.id,
|
||||
blockedBy: [taskId],
|
||||
}),
|
||||
])
|
||||
]))
|
||||
|
||||
appState = {
|
||||
...appState,
|
||||
teamContext: { teamName: taskListId },
|
||||
}
|
||||
const deleted = await TeamDeleteTool.call({}, context)
|
||||
expect(deleted.data.finalTasks).toHaveLength(2)
|
||||
expect(deleted.data).toMatchObject({
|
||||
success: true,
|
||||
team_name: taskListId,
|
||||
finalTasks: [
|
||||
finalTasks: expect.arrayContaining([
|
||||
expect.objectContaining({
|
||||
id: taskId,
|
||||
status: 'in_progress',
|
||||
@@ -508,7 +511,7 @@ describe('Task tool execution ordering', () => {
|
||||
id: dependent.data.task.id,
|
||||
blockedBy: [taskId],
|
||||
}),
|
||||
],
|
||||
]),
|
||||
})
|
||||
expect(Number.isFinite(Date.parse(deleted.data.taskListSnapshotAt ?? ''))).toBe(true)
|
||||
expect(deleted.data.taskListSnapshotRevision).toBe(3)
|
||||
|
||||
@@ -153,6 +153,7 @@ export function sanitizeSurfaceKey(surfaceKey: string): string {
|
||||
*/
|
||||
export function sanitizeModelName(shortName: string): string {
|
||||
// Map internal variants to public equivalents based on model family
|
||||
if (shortName.includes('fable-5-1')) return 'claude-fable-5-1'
|
||||
if (shortName.includes('fable-5')) return 'claude-fable-5'
|
||||
if (shortName.includes('opus-4-8')) return 'claude-opus-4-8'
|
||||
if (shortName.includes('opus-4-6')) return 'claude-opus-4-7'
|
||||
|
||||
@@ -34,6 +34,11 @@ const originalSessionId = getSessionId()
|
||||
function isAlive(pid: number): boolean {
|
||||
try {
|
||||
process.kill(pid, 0)
|
||||
// A Linux container's init may defer reaping an exited orphan. A zombie
|
||||
// cannot hold the socket open and is no longer a running daemon.
|
||||
if (process.platform === 'linux') {
|
||||
return !/^\d+ \(.*\) Z /.test(fs.readFileSync(`/proc/${pid}/stat`, 'utf8'))
|
||||
}
|
||||
return true
|
||||
} catch {
|
||||
return false
|
||||
@@ -75,7 +80,14 @@ async function launchDetachedDaemon(socketPath: string): Promise<number> {
|
||||
encoding: 'utf8',
|
||||
})
|
||||
expect(parent.status).toBe(0)
|
||||
expect(Number.parseInt(parent.stdout.trim(), 10)).toBe(1)
|
||||
const parentPid = Number.parseInt(parent.stdout.trim(), 10)
|
||||
if (process.platform === 'darwin') {
|
||||
expect(parentPid).toBe(1)
|
||||
} else {
|
||||
// Linux can reparent to a container subreaper rather than PID 1.
|
||||
expect(parentPid).toBeGreaterThan(0)
|
||||
expect(parentPid).not.toBe(result.pid)
|
||||
}
|
||||
return pid
|
||||
}
|
||||
|
||||
@@ -143,8 +155,10 @@ afterEach(async () => {
|
||||
runtimeRoot = undefined
|
||||
})
|
||||
|
||||
describe.skipIf(process.platform !== 'darwin')(
|
||||
'cu-helper launchd-owned daemon lifecycle',
|
||||
// The fixture models launchd ownership with a detached POSIX process; it does
|
||||
// not invoke launchctl or native Computer Use, so Linux exercises it too.
|
||||
describe.skipIf(process.platform === 'win32')(
|
||||
'cu-helper detached daemon lifecycle',
|
||||
() => {
|
||||
test('startup enumeration cannot poison the first resumed turn across shutdown and restart', async () => {
|
||||
runtimeRoot = fs.mkdtempSync(path.join(os.tmpdir(), 'cc-haha-cu-process-'))
|
||||
|
||||
@@ -17,7 +17,8 @@ describe('setupComputerUseMCP runtime capability', () => {
|
||||
})
|
||||
|
||||
expect(Object.keys(result.mcpConfig)).toEqual(['computer-use'])
|
||||
expect(result.allowedTools.length).toBeGreaterThan(0)
|
||||
expect(result.allowedTools).toContain('mcp__computer-use__get_app_state')
|
||||
expect(result.allowedTools).not.toContain('mcp__computer-use__screenshot')
|
||||
})
|
||||
|
||||
test('keeps the Windows compatibility engine available without a macOS helper', () => {
|
||||
@@ -29,6 +30,16 @@ describe('setupComputerUseMCP runtime capability', () => {
|
||||
})
|
||||
|
||||
expect(Object.keys(result.mcpConfig)).toEqual(['computer-use'])
|
||||
expect(result.allowedTools.length).toBeGreaterThan(0)
|
||||
expect(result.allowedTools).toContain('mcp__computer-use__screenshot')
|
||||
expect(result.allowedTools).not.toContain('mcp__computer-use__get_app_state')
|
||||
})
|
||||
|
||||
test('does not resolve a helper or advertise tools on unsupported platforms', () => {
|
||||
expect(setupComputerUseMCP({
|
||||
platform: 'linux',
|
||||
resolveMacosNativeBinary: () => {
|
||||
throw new Error('must not resolve a macOS helper on Linux')
|
||||
},
|
||||
})).toEqual({ mcpConfig: {}, allowedTools: [] })
|
||||
})
|
||||
})
|
||||
|
||||
@@ -41,7 +41,7 @@ export function setupComputerUseMCP(deps: SetupComputerUseDeps = {}): {
|
||||
}
|
||||
|
||||
const allowedTools = buildPlatformComputerUseTools(
|
||||
getCliComputerUseCapabilities(),
|
||||
getCliComputerUseCapabilities(platform),
|
||||
getChicagoCoordinateMode(),
|
||||
).map(t => buildMcpToolName(COMPUTER_USE_MCP_SERVER_NAME, t.name))
|
||||
|
||||
|
||||
@@ -46,6 +46,14 @@ export const CLAUDE_FABLE_5_CONFIG = {
|
||||
azureOpenAI: 'claude-fable-5',
|
||||
} as const satisfies ModelConfig
|
||||
|
||||
export const CLAUDE_FABLE_5_1_CONFIG = {
|
||||
firstParty: 'claude-fable-5-1',
|
||||
bedrock: 'anthropic.claude-fable-5-1',
|
||||
vertex: 'claude-fable-5-1',
|
||||
foundry: 'claude-fable-5-1',
|
||||
azureOpenAI: 'claude-fable-5-1',
|
||||
} as const satisfies ModelConfig
|
||||
|
||||
export const CLAUDE_SONNET_5_CONFIG = {
|
||||
firstParty: 'claude-sonnet-5',
|
||||
bedrock: 'anthropic.claude-sonnet-5',
|
||||
@@ -145,6 +153,7 @@ export const GPT_5_4_CODEX_CONFIG = {
|
||||
// @[MODEL LAUNCH]: Register the new config here.
|
||||
export const ALL_MODEL_CONFIGS = {
|
||||
fable5: CLAUDE_FABLE_5_CONFIG,
|
||||
fable51: CLAUDE_FABLE_5_1_CONFIG,
|
||||
haiku35: CLAUDE_3_5_HAIKU_CONFIG,
|
||||
haiku45: CLAUDE_HAIKU_4_5_CONFIG,
|
||||
sonnet35: CLAUDE_3_5_V2_SONNET_CONFIG,
|
||||
|
||||
@@ -7,6 +7,7 @@ import { getHardcodedTeammateModelFallback } from '../swarm/teammateModel.js'
|
||||
import { getAgentModel, getAgentModelOptions } from './agent.js'
|
||||
import { isModelAlias, isModelFamilyAlias } from './aliases.js'
|
||||
import {
|
||||
CLAUDE_FABLE_5_1_CONFIG,
|
||||
CLAUDE_FABLE_5_CONFIG,
|
||||
CLAUDE_OPUS_4_6_CONFIG,
|
||||
CLAUDE_OPUS_4_8_CONFIG,
|
||||
@@ -21,11 +22,13 @@ import {
|
||||
renderDefaultModelSetting,
|
||||
} from './model.js'
|
||||
import { getModelStrings } from './modelStrings.js'
|
||||
import { getModelOptions } from './modelOptions.js'
|
||||
import { get3PModelCapabilityOverride } from './modelSupportOverrides.js'
|
||||
|
||||
const ENV_KEYS = [
|
||||
'ANTHROPIC_BASE_URL',
|
||||
'ANTHROPIC_API_KEY',
|
||||
'ANTHROPIC_MODEL',
|
||||
'ANTHROPIC_DEFAULT_FABLE_MODEL',
|
||||
'ANTHROPIC_DEFAULT_FABLE_MODEL_SUPPORTED_CAPABILITIES',
|
||||
'CLAUDE_CODE_DISABLE_1M_CONTEXT',
|
||||
@@ -50,6 +53,14 @@ afterEach(() => {
|
||||
|
||||
describe('Fable model configuration', () => {
|
||||
test('uses the official provider IDs', () => {
|
||||
expect(CLAUDE_FABLE_5_1_CONFIG).toEqual({
|
||||
firstParty: 'claude-fable-5-1',
|
||||
bedrock: 'anthropic.claude-fable-5-1',
|
||||
vertex: 'claude-fable-5-1',
|
||||
foundry: 'claude-fable-5-1',
|
||||
azureOpenAI: 'claude-fable-5-1',
|
||||
})
|
||||
expect(getModelStrings().fable51).toBe('claude-fable-5-1')
|
||||
expect(CLAUDE_FABLE_5_CONFIG).toEqual({
|
||||
firstParty: 'claude-fable-5',
|
||||
bedrock: 'anthropic.claude-fable-5',
|
||||
@@ -121,6 +132,31 @@ describe('Fable model configuration', () => {
|
||||
)
|
||||
})
|
||||
|
||||
test('keeps Fable 5.1 identity distinct from the Fable 5 default', () => {
|
||||
expect(firstPartyNameToCanonical('us.anthropic.claude-fable-5-1-v1:0')).toBe(
|
||||
'claude-fable-5-1',
|
||||
)
|
||||
expect(getPublicModelDisplayName('claude-fable-5-1')).toBe('Fable 5.1')
|
||||
expect(getPublicModelDisplayName('claude-fable-5-1[1m]')).toBe('Fable 5.1 (1M context)')
|
||||
expect(getMarketingNameForModel('claude-fable-5-1[1m]')).toBe(
|
||||
'Fable 5.1 (with 1M context)',
|
||||
)
|
||||
expect(parseUserSpecifiedModel('fable')).toBe('claude-fable-5')
|
||||
expect(parseUserSpecifiedModel('claude-fable-5-1')).toBe('claude-fable-5-1')
|
||||
expect(modelSupports1M('claude-fable-5-1')).toBe(true)
|
||||
expect(getContextWindowForModel('claude-fable-5-1')).toBe(1_000_000)
|
||||
})
|
||||
|
||||
test('does not recommend downgrading a pinned Fable 5.1 to the Fable 5 alias', () => {
|
||||
process.env.ANTHROPIC_API_KEY = 'test-api-key'
|
||||
process.env.ANTHROPIC_MODEL = 'claude-fable-5-1'
|
||||
expect(getModelOptions()).toContainEqual({
|
||||
value: 'claude-fable-5-1',
|
||||
label: 'Fable 5.1',
|
||||
description: 'claude-fable-5-1',
|
||||
})
|
||||
})
|
||||
|
||||
test('publishes current model IDs and knowledge cutoffs in environment context', async () => {
|
||||
expect(SKILL_MODEL_VARS).toMatchObject({
|
||||
OPUS_ID: 'claude-opus-4-8',
|
||||
@@ -140,6 +176,7 @@ describe('Fable model configuration', () => {
|
||||
})
|
||||
|
||||
test('sanitizes new model trailers and keeps teammate fallbacks provider-safe', () => {
|
||||
expect(sanitizeModelName('claude-fable-5-1-experimental')).toBe('claude-fable-5-1')
|
||||
expect(sanitizeModelName('claude-fable-5-experimental')).toBe('claude-fable-5')
|
||||
expect(sanitizeModelName('claude-opus-4-8-experimental')).toBe('claude-opus-4-8')
|
||||
expect(sanitizeModelName('claude-sonnet-5-experimental')).toBe('claude-sonnet-5')
|
||||
|
||||
@@ -267,6 +267,9 @@ export function firstPartyNameToCanonical(name: ModelName): ModelShortName {
|
||||
name = name.toLowerCase()
|
||||
// Special cases for Claude 4+ models to differentiate versions
|
||||
// Order matters: check more specific versions first (4-5 before 4)
|
||||
if (name.includes('claude-fable-5-1')) {
|
||||
return 'claude-fable-5-1'
|
||||
}
|
||||
if (name.includes('claude-fable-5')) {
|
||||
return 'claude-fable-5'
|
||||
}
|
||||
@@ -416,6 +419,10 @@ export function getPublicModelDisplayName(model: ModelName): string | null {
|
||||
return openAIModelName
|
||||
}
|
||||
switch (model) {
|
||||
case getModelStrings().fable51:
|
||||
return 'Fable 5.1'
|
||||
case getModelStrings().fable51 + '[1m]':
|
||||
return 'Fable 5.1 (1M context)'
|
||||
case getModelStrings().fable5:
|
||||
return 'Fable 5'
|
||||
case getModelStrings().fable5 + '[1m]':
|
||||
@@ -662,6 +669,9 @@ export function getMarketingNameForModel(modelId: string): string | undefined {
|
||||
const has1m = modelId.toLowerCase().includes('[1m]')
|
||||
const canonical = getCanonicalName(modelId)
|
||||
|
||||
if (canonical.includes('claude-fable-5-1')) {
|
||||
return has1m ? 'Fable 5.1 (with 1M context)' : 'Fable 5.1'
|
||||
}
|
||||
if (canonical.includes('claude-fable-5')) {
|
||||
return has1m ? 'Fable 5 (with 1M context)' : 'Fable 5'
|
||||
}
|
||||
|
||||
@@ -3,6 +3,7 @@ export const MODEL_CONTEXT_WINDOW_MIN = 16_000
|
||||
export const MODEL_CONTEXT_WINDOW_MAX = 10_000_000
|
||||
|
||||
const DIRECT_MODEL_CONTEXT_WINDOWS: Record<string, number> = {
|
||||
'claude-fable-5-1': 1_000_000,
|
||||
'claude-fable-5': 1_000_000,
|
||||
'claude-opus-4-8': 1_000_000,
|
||||
'claude-opus-4-7': 1_000_000,
|
||||
|
||||
@@ -539,7 +539,16 @@ function getModelFamilyInfo(
|
||||
|
||||
// Fable family
|
||||
if (canonical.includes('claude-fable-5')) {
|
||||
const currentName = getMarketingNameForModel(getDefaultFableModel())
|
||||
const defaultModel = getDefaultFableModel()
|
||||
// The alias can lag a newly selectable release; a different version is not
|
||||
// necessarily an upgrade for a user who has already pinned Fable 5.1.
|
||||
if (
|
||||
canonical === 'claude-fable-5-1' &&
|
||||
getCanonicalName(defaultModel) === 'claude-fable-5'
|
||||
) {
|
||||
return null
|
||||
}
|
||||
const currentName = getMarketingNameForModel(defaultModel)
|
||||
if (currentName) {
|
||||
return { alias: 'Fable', currentVersionName: currentName }
|
||||
}
|
||||
|
||||
@@ -0,0 +1,32 @@
|
||||
import { describe, expect, test } from 'bun:test'
|
||||
import { calculateCostFromTokens, getModelPricingString } from './modelCost.js'
|
||||
|
||||
describe('Fable 5.1 query costs', () => {
|
||||
const cacheReads = {
|
||||
inputTokens: 0,
|
||||
outputTokens: 0,
|
||||
cacheReadInputTokens: 1_000_000,
|
||||
cacheCreationInputTokens: 0,
|
||||
}
|
||||
|
||||
test('charges $0.25 per million cached tokens for Fable 5.1', () => {
|
||||
for (const model of [
|
||||
'claude-fable-5-1',
|
||||
'claude-fable-5-1[1m]',
|
||||
'us.anthropic.claude-fable-5-1-v1:0',
|
||||
]) {
|
||||
expect(calculateCostFromTokens(model, cacheReads)).toBe(0.25)
|
||||
}
|
||||
expect(calculateCostFromTokens('claude-fable-5', cacheReads)).toBe(1)
|
||||
})
|
||||
|
||||
test('keeps input, output and cache-write pricing at the published rates', () => {
|
||||
expect(calculateCostFromTokens('claude-fable-5-1', {
|
||||
inputTokens: 1_000_000,
|
||||
outputTokens: 1_000_000,
|
||||
cacheReadInputTokens: 1_000_000,
|
||||
cacheCreationInputTokens: 1_000_000,
|
||||
})).toBe(72.75)
|
||||
expect(getModelPricingString('claude-fable-5-1')).toBe('$10/$50 per Mtok')
|
||||
})
|
||||
})
|
||||
@@ -8,6 +8,7 @@ import {
|
||||
CLAUDE_3_5_V2_SONNET_CONFIG,
|
||||
CLAUDE_3_7_SONNET_CONFIG,
|
||||
CLAUDE_FABLE_5_CONFIG,
|
||||
CLAUDE_FABLE_5_1_CONFIG,
|
||||
CLAUDE_HAIKU_4_5_CONFIG,
|
||||
CLAUDE_OPUS_4_1_CONFIG,
|
||||
CLAUDE_OPUS_4_5_CONFIG,
|
||||
@@ -71,6 +72,12 @@ export const COST_TIER_10_50 = {
|
||||
webSearchRequests: 0.01,
|
||||
} as const satisfies ModelCosts
|
||||
|
||||
// Fable 5.1 keeps Fable 5 input/output pricing but reduces cache reads by 75%.
|
||||
export const COST_FABLE_51 = {
|
||||
...COST_TIER_10_50,
|
||||
promptCacheReadTokens: 0.25,
|
||||
} as const satisfies ModelCosts
|
||||
|
||||
// Fast mode pricing for Opus 4.7: $30 input / $150 output per Mtok
|
||||
export const COST_TIER_30_150 = {
|
||||
inputTokens: 30,
|
||||
@@ -114,6 +121,8 @@ export function getOpus46CostTier(fastMode: boolean): ModelCosts {
|
||||
// Costs from https://platform.claude.com/docs/en/about-claude/pricing
|
||||
// Web search cost: $10 per 1000 requests = $0.01 per request
|
||||
export const MODEL_COSTS: Record<ModelShortName, ModelCosts> = {
|
||||
[firstPartyNameToCanonical(CLAUDE_FABLE_5_1_CONFIG.firstParty)]:
|
||||
COST_FABLE_51,
|
||||
[firstPartyNameToCanonical(CLAUDE_FABLE_5_CONFIG.firstParty)]:
|
||||
COST_TIER_10_50,
|
||||
[firstPartyNameToCanonical(CLAUDE_3_5_HAIKU_CONFIG.firstParty)]:
|
||||
|
||||
@@ -141,7 +141,7 @@ export function isPermissionExplainerEnabled(): boolean {
|
||||
}
|
||||
|
||||
/**
|
||||
* Generate a permission explanation using Haiku with structured output.
|
||||
* Generate a permission explanation using the selected model with structured output.
|
||||
* Returns null if the feature is disabled, request is aborted, or an error occurs.
|
||||
*/
|
||||
export async function generatePermissionExplanation({
|
||||
@@ -174,7 +174,7 @@ Explain this command in context.`
|
||||
|
||||
const model = getMainLoopModel()
|
||||
|
||||
// Use sideQuery with forced tool choice for guaranteed structured output
|
||||
// sideQuery adapts forced tool choice for models with required thinking.
|
||||
const response = await sideQuery({
|
||||
model,
|
||||
system: SYSTEM_PROMPT,
|
||||
@@ -191,7 +191,9 @@ Explain this command in context.`
|
||||
)
|
||||
|
||||
// Extract structured data from tool use block
|
||||
const toolUseBlock = response.content.find(c => c.type === 'tool_use')
|
||||
const toolUseBlock = response.content.find(c =>
|
||||
c.type === 'tool_use' && c.name === EXPLAIN_COMMAND_TOOL.name,
|
||||
)
|
||||
if (toolUseBlock && toolUseBlock.type === 'tool_use') {
|
||||
logForDebugging(
|
||||
`Permission explainer: tool input: ${jsonStringify(toolUseBlock.input).slice(0, 500)}`,
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
import { afterAll, afterEach, beforeAll, describe, expect, mock, test } from 'bun:test'
|
||||
import { afterAll, afterEach, beforeAll, beforeEach, describe, expect, mock, spyOn, test } from 'bun:test'
|
||||
import { feature } from 'bun:bundle'
|
||||
import { mkdtemp, rm } from 'node:fs/promises'
|
||||
import { tmpdir } from 'node:os'
|
||||
@@ -22,11 +22,11 @@ let classifierMode: 'allow' | 'block' | 'parse-failure' | 'unavailable' =
|
||||
let configDir = ''
|
||||
|
||||
const actualSideQuery = await import('../sideQuery.js')
|
||||
mock.module('../sideQuery.js', () => ({
|
||||
...actualSideQuery,
|
||||
sideQuery: async () => {
|
||||
|
||||
beforeEach(() => {
|
||||
spyOn(actualSideQuery, 'sideQuery').mockImplementation(async () => {
|
||||
if (classifierMode === 'unavailable') throw new Error('classifier offline')
|
||||
const content =
|
||||
const content: Awaited<ReturnType<typeof actualSideQuery.sideQuery>>['content'] =
|
||||
classifierMode === 'parse-failure'
|
||||
? [{ type: 'text', text: 'not structured' }]
|
||||
: [
|
||||
@@ -55,9 +55,9 @@ mock.module('../sideQuery.js', () => ({
|
||||
cache_read_input_tokens: 0,
|
||||
cache_creation_input_tokens: 0,
|
||||
},
|
||||
}
|
||||
},
|
||||
}))
|
||||
} as Awaited<ReturnType<typeof actualSideQuery.sideQuery>>
|
||||
})
|
||||
})
|
||||
|
||||
const { hasPermissionsToUseTool } = await import('./permissions.js')
|
||||
|
||||
@@ -67,6 +67,7 @@ beforeAll(async () => {
|
||||
})
|
||||
|
||||
afterEach(() => {
|
||||
mock.restore()
|
||||
classifierMode = 'allow'
|
||||
resetSettingsCache()
|
||||
resetAutoModeState()
|
||||
|
||||
@@ -0,0 +1,162 @@
|
||||
import { expect, test } from 'bun:test'
|
||||
import { mkdtemp, rm } from 'node:fs/promises'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { join } from 'node:path'
|
||||
import { createSandboxedTestEnvironment } from '../../scripts/pr/test-environment.js'
|
||||
import { getIsInteractive, setIsInteractive } from '../bootstrap/state.js'
|
||||
import { enableConfigs } from './config.js'
|
||||
import { get3PModelCapabilityOverride } from './model/modelSupportOverrides.js'
|
||||
import { generatePermissionExplanation } from './permissions/permissionExplainer.js'
|
||||
import { sideQuery } from './sideQuery.js'
|
||||
|
||||
const explanation = {
|
||||
explanation: 'Lists the working directory.',
|
||||
reasoning: 'I need to inspect the project files.',
|
||||
risk: 'No files are changed.',
|
||||
riskLevel: 'LOW',
|
||||
}
|
||||
|
||||
async function withCapturedRequests<T>(
|
||||
model: string,
|
||||
run: () => Promise<T>,
|
||||
responseContent: unknown[] = [{
|
||||
type: 'tool_use', id: 'toolu_fixture', name: 'explain_command', input: explanation,
|
||||
}],
|
||||
) {
|
||||
const requests: Record<string, any>[] = []
|
||||
const headers: Headers[] = []
|
||||
const server = Bun.serve({
|
||||
hostname: '127.0.0.1',
|
||||
port: 0,
|
||||
async fetch(request) {
|
||||
const body = await request.json() as Record<string, any>
|
||||
requests.push(body)
|
||||
headers.push(request.headers)
|
||||
if (model === 'claude-fable-5-1' &&
|
||||
(body.tool_choice?.type === 'tool' || body.tool_choice?.type === 'any' ||
|
||||
body.thinking?.type === 'disabled' || 'temperature' in body)) {
|
||||
return Response.json({ type: 'error', error: {
|
||||
type: 'invalid_request_error', message: 'Unsupported Fable request controls',
|
||||
} }, { status: 400 })
|
||||
}
|
||||
return Response.json({
|
||||
id: 'msg_fixture', type: 'message', role: 'assistant', model,
|
||||
content: responseContent, stop_reason: 'end_turn', stop_sequence: null,
|
||||
usage: { input_tokens: 1, output_tokens: 1 },
|
||||
})
|
||||
},
|
||||
})
|
||||
const sandbox = await mkdtemp(join(tmpdir(), 'cc-haha-side-query-'))
|
||||
const originalEnv = { ...process.env }
|
||||
const interactive = getIsInteractive()
|
||||
try {
|
||||
for (const key of Object.keys(process.env)) delete process.env[key]
|
||||
Object.assign(process.env, createSandboxedTestEnvironment(sandbox, {
|
||||
NODE_ENV: 'production',
|
||||
ANTHROPIC_API_KEY: 'loopback-test-key',
|
||||
ANTHROPIC_BASE_URL: `http://127.0.0.1:${server.port}`,
|
||||
ANTHROPIC_MODEL: model,
|
||||
CC_HAHA_SEND_DISABLED_THINKING: '1',
|
||||
CLAUDE_CODE_DISABLE_NONESSENTIAL_TRAFFIC: '1',
|
||||
}, originalEnv))
|
||||
setIsInteractive(false)
|
||||
get3PModelCapabilityOverride.cache.clear()
|
||||
enableConfigs()
|
||||
return { result: await run(), requests, headers }
|
||||
} finally {
|
||||
for (const key of Object.keys(process.env)) delete process.env[key]
|
||||
Object.assign(process.env, originalEnv)
|
||||
setIsInteractive(interactive)
|
||||
get3PModelCapabilityOverride.cache.clear()
|
||||
server.stop(true)
|
||||
await rm(sandbox, { recursive: true, force: true })
|
||||
}
|
||||
}
|
||||
|
||||
test('permission explanations work with Fable 5.1 without forced tool choice', async () => {
|
||||
const { result, requests, headers } = await withCapturedRequests('claude-fable-5-1', () =>
|
||||
generatePermissionExplanation({
|
||||
toolName: 'Bash', toolInput: { command: 'ls' },
|
||||
signal: new AbortController().signal,
|
||||
}))
|
||||
expect(result).toEqual(explanation)
|
||||
expect(requests).toHaveLength(1)
|
||||
expect(requests[0]?.tool_choice).toEqual({ type: 'auto', disable_parallel_tool_use: true })
|
||||
expect(requests[0]?.thinking).toEqual({
|
||||
type: 'adaptive', block_binding: { prefix_mismatch_behavior: 'drop_block' },
|
||||
})
|
||||
expect(headers[0]?.get('anthropic-beta')).toContain('thinking-binding-controls-2026-08-01')
|
||||
expect(JSON.stringify(requests[0]?.system)).toContain('explain_command')
|
||||
expect(requests[0]?.tools.map((tool: any) => tool.name)).toEqual(['explain_command'])
|
||||
})
|
||||
|
||||
test('Fable side classifiers retain output headroom and omit incompatible sampling controls', async () => {
|
||||
const { requests } = await withCapturedRequests('claude-fable-5-1', () => sideQuery({
|
||||
model: 'claude-fable-5-1', messages: [{ role: 'user', content: 'Classify action' }],
|
||||
thinking: false, temperature: 0, max_tokens: 64,
|
||||
stop_sequences: ['</block>'], querySource: 'auto_mode',
|
||||
}), [{ type: 'text', text: '<block>no</block>' }])
|
||||
expect(requests[0]?.thinking).toMatchObject({ type: 'adaptive' })
|
||||
expect(requests[0]).not.toHaveProperty('temperature')
|
||||
expect(requests[0]?.max_tokens).toBeGreaterThanOrEqual(2112)
|
||||
expect(requests[0]?.stop_sequences).toEqual(['</block>'])
|
||||
})
|
||||
|
||||
test('Fable side-query thinking headroom respects the output limit without reducing caller budgets', async () => {
|
||||
for (const [requested, expected] of [[127_000, 128_000], [128_000, 128_000], [130_000, 130_000]]) {
|
||||
const { requests } = await withCapturedRequests('claude-fable-5-1', () => sideQuery({
|
||||
model: 'claude-fable-5-1', messages: [{ role: 'user', content: 'Return result' }],
|
||||
max_tokens: requested, querySource: 'permission_explainer',
|
||||
}))
|
||||
expect(requests[0]?.max_tokens).toBe(expected)
|
||||
}
|
||||
})
|
||||
|
||||
test('permission explanation rejects missing or invalid tool results', async () => {
|
||||
for (const content of [
|
||||
[{ type: 'text', text: 'The command is safe' }],
|
||||
[{ type: 'tool_use', id: 'toolu_bad', name: 'explain_command', input: { riskLevel: 'LOW' } }],
|
||||
[{ type: 'tool_use', id: 'toolu_wrong', name: 'unrelated_tool', input: explanation }],
|
||||
]) {
|
||||
const { result, requests } = await withCapturedRequests('claude-fable-5-1', () =>
|
||||
generatePermissionExplanation({ toolName: 'Bash', toolInput: 'ls',
|
||||
signal: new AbortController().signal }), content)
|
||||
expect(result).toBeNull()
|
||||
expect(requests).toHaveLength(1)
|
||||
}
|
||||
})
|
||||
|
||||
test('Fable any-tool side queries retain all choices and require a tool result in the prompt', async () => {
|
||||
const tools = ['first_result', 'second_result'].map(name => ({
|
||||
name, input_schema: { type: 'object' as const, properties: {} },
|
||||
}))
|
||||
const { requests } = await withCapturedRequests('claude-fable-5-1', () => sideQuery({
|
||||
model: 'claude-fable-5-1', messages: [{ role: 'user', content: 'Return result' }],
|
||||
tools, tool_choice: { type: 'any' }, thinking: 2048, max_tokens: 4096,
|
||||
querySource: 'permission_explainer',
|
||||
}))
|
||||
expect(requests[0]?.tool_choice).toEqual({ type: 'auto', disable_parallel_tool_use: true })
|
||||
expect(requests[0]?.tools).toEqual(tools)
|
||||
expect(JSON.stringify(requests[0]?.system)).toContain('calling one of the provided tools')
|
||||
expect(requests[0]?.thinking.type).toBe('adaptive')
|
||||
expect(requests[0]?.max_tokens).toBe(4096)
|
||||
})
|
||||
|
||||
test('Fable 5 side queries require thinking without the 5.1 binding controls', async () => {
|
||||
const { requests, headers } = await withCapturedRequests('claude-fable-5', () => sideQuery({
|
||||
model: 'claude-fable-5', messages: [{ role: 'user', content: 'Return result' }],
|
||||
thinking: false, querySource: 'permission_explainer',
|
||||
}))
|
||||
expect(requests[0]?.thinking).toEqual({ type: 'adaptive' })
|
||||
expect(headers[0]?.get('anthropic-beta') ?? '').not.toContain('thinking-binding-controls')
|
||||
})
|
||||
|
||||
test('existing non-Fable side queries preserve forced tools and disabled thinking', async () => {
|
||||
const { result, requests } = await withCapturedRequests('claude-sonnet-4-6', () =>
|
||||
generatePermissionExplanation({ toolName: 'Bash', toolInput: 'ls',
|
||||
signal: new AbortController().signal }))
|
||||
expect(result).toEqual(explanation)
|
||||
expect(requests[0]?.tool_choice).toEqual({ type: 'tool', name: 'explain_command' })
|
||||
expect(requests[0]?.thinking).toEqual({ type: 'disabled' })
|
||||
expect(requests[0]?.max_tokens).toBe(1024)
|
||||
})
|
||||
+54
-7
@@ -4,7 +4,10 @@ import {
|
||||
getLastApiCompletionTimestamp,
|
||||
setLastApiCompletionTimestamp,
|
||||
} from '../bootstrap/state.js'
|
||||
import { STRUCTURED_OUTPUTS_BETA_HEADER } from '../constants/betas.js'
|
||||
import {
|
||||
STRUCTURED_OUTPUTS_BETA_HEADER,
|
||||
THINKING_BINDING_CONTROLS_BETA_HEADER,
|
||||
} from '../constants/betas.js'
|
||||
import { CLAUDE_CODE_COMPAT_VERSION } from '../constants/claudeCodeCompatibility.js'
|
||||
import type { QuerySource } from '../constants/querySource.js'
|
||||
import {
|
||||
@@ -17,9 +20,15 @@ import { getAPIMetadata } from '../services/api/claude.js'
|
||||
import { getAnthropicClient } from '../services/api/client.js'
|
||||
import { normalizeUsage } from '../services/api/emptyUsage.js'
|
||||
import { getModelBetas, modelSupportsStructuredOutputs } from './betas.js'
|
||||
import { getModelMaxOutputTokens } from './context.js'
|
||||
import { computeFingerprint } from './fingerprint.js'
|
||||
import { normalizeModelStringForAPI } from './model/model.js'
|
||||
import { shouldSendExplicitDisabledThinking } from './thinking.js'
|
||||
import {
|
||||
modelRequiresThinking,
|
||||
modelSupportsAdaptiveThinking,
|
||||
modelUsesBoundThinking,
|
||||
shouldSendExplicitDisabledThinking,
|
||||
} from './thinking.js'
|
||||
|
||||
type MessageParam = Anthropic.MessageParam
|
||||
type TextBlockParam = Anthropic.TextBlockParam
|
||||
@@ -169,7 +178,34 @@ export async function sideQuery(opts: SideQueryOptions): Promise<BetaMessage> {
|
||||
: []),
|
||||
].filter((block): block is TextBlockParam => block !== null)
|
||||
|
||||
const thinkingConfig = resolveSideQueryThinkingConfig(thinking, max_tokens)
|
||||
const requiredAdaptiveThinking =
|
||||
modelRequiresThinking(model) && modelSupportsAdaptiveThinking(model)
|
||||
// Side-query budgets normally cover only the structured/text answer. Required
|
||||
// thinking must have headroom, including classifiers with a 64-token answer.
|
||||
const maxTokens = requiredAdaptiveThinking && typeof thinking !== 'number'
|
||||
? Math.max(max_tokens, Math.min(max_tokens + 2048, getModelMaxOutputTokens(model).upperLimit))
|
||||
: max_tokens
|
||||
const thinkingConfig = resolveSideQueryThinkingConfig(thinking, maxTokens, model)
|
||||
if (requiredAdaptiveThinking && modelUsesBoundThinking(model)) {
|
||||
betas.push(THINKING_BINDING_CONTROLS_BETA_HEADER)
|
||||
}
|
||||
|
||||
// Adaptive thinking cannot be combined with a forced tool choice. Keep the
|
||||
// same tool-result contract, explicitly request it, and let callers validate
|
||||
// the result (the permission classifier still fails closed on missing output).
|
||||
const useAutoToolChoice = requiredAdaptiveThinking &&
|
||||
(tool_choice?.type === 'tool' || tool_choice?.type === 'any')
|
||||
const requestTools = useAutoToolChoice && tool_choice?.type === 'tool'
|
||||
? tools?.filter(tool => 'name' in tool && tool.name === tool_choice.name)
|
||||
: tools
|
||||
if (useAutoToolChoice) {
|
||||
systemBlocks.push({
|
||||
type: 'text',
|
||||
text: tool_choice.type === 'tool'
|
||||
? `Respond by calling the ${tool_choice.name} tool exactly once with the complete result. Do not substitute a text response for the tool call.`
|
||||
: 'Respond by calling one of the provided tools with the complete result. Do not substitute a text response for the tool call.',
|
||||
})
|
||||
}
|
||||
|
||||
const normalizedModel = normalizeModelStringForAPI(model)
|
||||
const start = Date.now()
|
||||
@@ -177,13 +213,15 @@ export async function sideQuery(opts: SideQueryOptions): Promise<BetaMessage> {
|
||||
const response = await client.beta.messages.create(
|
||||
{
|
||||
model: normalizedModel,
|
||||
max_tokens,
|
||||
max_tokens: maxTokens,
|
||||
system: systemBlocks,
|
||||
messages,
|
||||
...(tools && { tools }),
|
||||
...(tool_choice && { tool_choice }),
|
||||
...(requestTools && { tools: requestTools }),
|
||||
...(tool_choice && { tool_choice: useAutoToolChoice
|
||||
? { type: 'auto' as const, disable_parallel_tool_use: true }
|
||||
: tool_choice }),
|
||||
...(output_format && { output_config: { format: output_format } }),
|
||||
...(temperature !== undefined && { temperature }),
|
||||
...(temperature !== undefined && !requiredAdaptiveThinking && { temperature }),
|
||||
...(stop_sequences && { stop_sequences }),
|
||||
...(thinkingConfig && { thinking: thinkingConfig }),
|
||||
...(betas.length > 0 && { betas }),
|
||||
@@ -220,7 +258,16 @@ export async function sideQuery(opts: SideQueryOptions): Promise<BetaMessage> {
|
||||
export function resolveSideQueryThinkingConfig(
|
||||
thinking: SideQueryOptions['thinking'],
|
||||
maxTokens: number,
|
||||
model?: string,
|
||||
): BetaThinkingConfigParam | undefined {
|
||||
if (model && modelRequiresThinking(model) && modelSupportsAdaptiveThinking(model)) {
|
||||
return {
|
||||
type: 'adaptive',
|
||||
...(modelUsesBoundThinking(model) && {
|
||||
block_binding: { prefix_mismatch_behavior: 'drop_block' },
|
||||
}),
|
||||
}
|
||||
}
|
||||
if (
|
||||
thinking === false ||
|
||||
(thinking === undefined && shouldSendExplicitDisabledThinking())
|
||||
|
||||
@@ -692,7 +692,7 @@ describe('in-process teammate task claiming', () => {
|
||||
status: 'in_progress',
|
||||
})
|
||||
await updateTask(taskListId, explicitAssignment, { status: 'completed' })
|
||||
await createTask(taskListId, {
|
||||
const followUpTaskId = await createTask(taskListId, {
|
||||
subject: 'Audit workflow',
|
||||
description: 'Audit workflow changes',
|
||||
status: 'pending',
|
||||
@@ -711,8 +711,9 @@ describe('in-process teammate task claiming', () => {
|
||||
|
||||
expect(prompt).toContain('Audit workflow')
|
||||
const claimedTasks = await listTasks(taskListId)
|
||||
expect(claimedTasks[1]?.owner).toBe(agentName)
|
||||
expect(claimedTasks[1]?.status).toBe('in_progress')
|
||||
const claimedTask = claimedTasks.find(task => task.id === followUpTaskId)
|
||||
expect(claimedTask?.owner).toBe(agentName)
|
||||
expect(claimedTask?.status).toBe('in_progress')
|
||||
|
||||
const unrelatedTasks = await listTasks(parentSessionId)
|
||||
expect(unrelatedTasks).toHaveLength(1)
|
||||
|
||||
@@ -129,6 +129,11 @@ export function modelRequiresThinking(model: string): boolean {
|
||||
return getCanonicalName(model).includes('claude-fable-5')
|
||||
}
|
||||
|
||||
/** Fable 5.1 binds replayed thinking to the preceding system, tools and history. */
|
||||
export function modelUsesBoundThinking(model: string): boolean {
|
||||
return getCanonicalName(model) === 'claude-fable-5-1'
|
||||
}
|
||||
|
||||
export function resolveModelThinkingEnabled(
|
||||
model: string,
|
||||
requestedEnabled: boolean,
|
||||
|
||||
@@ -95,6 +95,19 @@ describe('isBillableUsageRecord', () => {
|
||||
})
|
||||
|
||||
describe('estimateCostUSD', () => {
|
||||
it('uses the reduced Fable 5.1 cache-read rate without changing Fable 5', () => {
|
||||
const cacheReads = tokens({ cacheReadInputTokens: ONE_MILLION })
|
||||
expect(estimateCostUSD('claude-fable-5-1', cacheReads)).toBe(0.25)
|
||||
expect(estimateCostUSD('anthropic/claude-fable-5-1-20260825', cacheReads)).toBe(0.25)
|
||||
expect(estimateCostUSD('claude-fable-5', cacheReads)).toBe(1)
|
||||
expect(estimateCostUSD('claude-fable-5-1', tokens({
|
||||
inputTokens: ONE_MILLION,
|
||||
outputTokens: ONE_MILLION,
|
||||
cacheReadInputTokens: ONE_MILLION,
|
||||
cacheCreationInputTokens: ONE_MILLION,
|
||||
}))).toBe(72.75)
|
||||
})
|
||||
|
||||
it('bills each token bucket at its own rate', () => {
|
||||
const cost = estimateCostUSD('claude-opus-5', tokens({
|
||||
inputTokens: ONE_MILLION,
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
import {
|
||||
COST_FABLE_51,
|
||||
COST_HAIKU_35,
|
||||
COST_HAIKU_45,
|
||||
COST_TIER_3_15,
|
||||
@@ -51,6 +52,7 @@ const MODEL_TIERS: ReadonlyArray<readonly [prefix: string, costs: ModelCosts]> =
|
||||
['claude-opus-4-1', COST_TIER_15_75],
|
||||
['claude-opus-4', COST_TIER_15_75],
|
||||
['claude-fable-5', COST_TIER_10_50],
|
||||
['claude-fable-5-1', COST_FABLE_51],
|
||||
['claude-mythos-5', COST_TIER_10_50],
|
||||
['claude-sonnet-5', COST_TIER_3_15],
|
||||
['claude-sonnet-4-6', COST_TIER_3_15],
|
||||
|
||||
Reference in New Issue
Block a user