diff --git a/desktop/src/components/chat/AskUserQuestion.test.tsx b/desktop/src/components/chat/AskUserQuestion.test.tsx index a691668d..eeafb13b 100644 --- a/desktop/src/components/chat/AskUserQuestion.test.tsx +++ b/desktop/src/components/chat/AskUserQuestion.test.tsx @@ -966,4 +966,150 @@ describe('AskUserQuestion', () => { expect(storedDraft()).toBeUndefined() }) }) + + // Issue #1400. The input is whatever the model emitted. The server validates it and + // answers a bad call with InputValidationError, but the transcript keeps the + // tool_use as sent, so a card is rebuilt from it on every replay. One provider's + // tool-call parser turned "`rm `" inside a description into an object + // ({ $text, path }); rendering that as a child threw React #31, which took the + // whole app down — again on every launch, since the open tab is restored. + describe('input the model got wrong', () => { + // The exact shape from the report: text around an inline became an object. + const BROKEN_TEXT = { + path: '`(一次一个,不用 wildcard)。CLAUDE.md 禁止 wildcard 删多个文件。', + $text: '我会逐个执行 `rm ', + } + const INVALID_INPUT_RESULT = + 'InputValidationError: AskUserQuestion failed due to the following issue:\n' + + 'The parameter `questions[0].options[0].description` type is expected as `string` but provided as `object`' + + function singleQuestion(question: Record) { + return { questions: [{ question: 'Pick one?', ...question }] } + } + + it('renders the rest of a question whose option description is an object', () => { + render() + + expect(screen.getByText('Delete these 5 files?')).toBeTruthy() + // The broken description is dropped with nothing rendered in its place. + expect(screen.getByRole('button', { name: /^Delete all 5$/ })).toBeTruthy() + // Its sound sibling keeps its own. + expect(screen.getByText('Back up first.')).toBeTruthy() + }) + + it('still answers a question that has a broken option, handing the original input back', () => { + const input = singleQuestion({ + question: 'Delete these 5 files?', + options: [{ label: 'Delete all 5', description: BROKEN_TEXT }, { label: 'Wait' }], + }) + render() + + fireEvent.click(screen.getByRole('button', { name: /^Wait$/ })) + fireEvent.click(screen.getByRole('button', { name: /submit/i })) + + // Only what is shown is cleaned up; the answer travels with the input untouched. + expect(sendMock).toHaveBeenCalledWith(ACTIVE_TAB, { + type: 'permission_response', + requestId: 'perm-1', + allowed: true, + updatedInput: { ...input, answers: { 'Delete these 5 files?': 'Wait' } }, + }) + }) + + it('shows a failed call from history together with its error result', () => { + render() + + expect(screen.getByText('Delete these 5 files?')).toBeTruthy() + expect(screen.getByText(/InputValidationError/)).toBeTruthy() + }) + + it('renders nothing for a question whose text is an object', () => { + const { container } = render() + + expect(container.textContent).toBe('') + }) + + it('drops an option whose label is an object and keeps its siblings', () => { + render() + + expect(screen.getByRole('button', { name: /^B$/ })).toBeTruthy() + expect(screen.queryByText('goes with its label')).toBeNull() + }) + + it.each([ + ['a string', 'A or B'], + ['an object', { A: 'first', B: 'second' }], + ])('renders the question when options is %s instead of a list', (_shape, options) => { + render() + + expect(screen.getByText('Pick one?')).toBeTruthy() + expect(screen.getByRole('textbox')).toBeTruthy() + }) + + it('skips null and non-object entries among the options', () => { + render() + + expect(screen.getByRole('button', { name: /^Real$/ })).toBeTruthy() + expect(screen.queryByText('stray text')).toBeNull() + }) + + it('skips entries of questions that are not questions', () => { + render() + + expect(screen.getByText('Real?')).toBeTruthy() + // One question is left, so there is no tab strip to number. + expect(screen.queryByRole('button', { name: 'Q1' })).toBeNull() + }) + + it('numbers the tab of a question whose header is not text', () => { + render() + + expect(screen.getByRole('button', { name: 'Q1' })).toBeTruthy() + expect(screen.getByRole('button', { name: 'Sound' })).toBeTruthy() + }) + + // chatStore rebuilds tool_use messages under a stable id, so a mounted card can be + // handed a repaired input after it was handed a broken one. + it('recovers when a broken input is replaced by a sound one while mounted', () => { + const { container, rerender } = render() + expect(container.textContent).toBe('') + + rerender() + + expect(screen.getByText('Ship it?')).toBeTruthy() + }) + }) }) diff --git a/desktop/src/components/chat/AskUserQuestion.tsx b/desktop/src/components/chat/AskUserQuestion.tsx index d4c72c41..810eb5cf 100644 --- a/desktop/src/components/chat/AskUserQuestion.tsx +++ b/desktop/src/components/chat/AskUserQuestion.tsx @@ -24,14 +24,6 @@ type Question = { multiSelect?: boolean } -type AskUserInput = { - questions?: Question[] - question?: string - header?: string - options?: QuestionOption[] - multiSelect?: boolean -} - type Props = { sessionId?: string | null toolUseId: string @@ -45,29 +37,51 @@ type Props = { supersededByUserMessage?: boolean } +function isRecord(value: unknown): value is Record { + return typeof value === 'object' && value !== null && !Array.isArray(value) +} + +// The input is whatever the model emitted, and the transcript keeps it even when the +// server rejected the call as invalid — so a card is rebuilt from it on every replay +// (issue #1400: an option description that came back as an object threw React #31 and +// took the whole app down, again on every launch). Only string fields are trusted, +// because a non-string child throws while rendering; anything else is left out rather +// than shown. That cleanup is display-only: the answer still travels with the +// original input (see handleSubmit). + +function toOption(value: unknown): QuestionOption | null { + if (!isRecord(value) || typeof value.label !== 'string') return null + return typeof value.description === 'string' + ? { label: value.label, description: value.description } + : { label: value.label } +} + +function toQuestion(value: unknown): Question | null { + if (!isRecord(value) || typeof value.question !== 'string') return null + return { + question: value.question, + header: typeof value.header === 'string' ? value.header : undefined, + options: Array.isArray(value.options) + ? value.options.flatMap((option) => toOption(option) ?? []) + : undefined, + multiSelect: value.multiSelect === true, + } +} + /** * Parse the AskUserQuestion input which may come in different shapes. */ function parseInput(input: unknown): Question[] { - if (!input || typeof input !== 'object') return [] - const obj = input as AskUserInput + if (!isRecord(input)) return [] // Shape 1: { questions: [...] } - if (Array.isArray(obj.questions)) { - return obj.questions + if (Array.isArray(input.questions)) { + return input.questions.flatMap((question) => toQuestion(question) ?? []) } // Shape 2: { question: "...", options: [...] } - if (typeof obj.question === 'string') { - return [{ - question: obj.question, - header: obj.header, - options: obj.options, - multiSelect: obj.multiSelect, - }] - } - - return [] + const single = toQuestion(input) + return single ? [single] : [] } type QuestionSelections = Record @@ -100,7 +114,7 @@ export function AskUserQuestion({ const sessionConnectionState = useChatStore((s) => targetSessionId ? s.sessions[targetSessionId]?.connectionState : undefined) const t = useTranslation() - const questions = parseInput(input) + const questions = useMemo(() => parseInput(input), [input]) const inputObject = (input && typeof input === 'object') ? input as Record : {} // Read once instead of subscribing: this card writes the draft on every // change, and a subscription would feed its own writes back as re-renders. diff --git a/desktop/src/components/chat/MessageList.itemBoundary.test.tsx b/desktop/src/components/chat/MessageList.itemBoundary.test.tsx new file mode 100644 index 00000000..0a8f6d61 --- /dev/null +++ b/desktop/src/components/chat/MessageList.itemBoundary.test.tsx @@ -0,0 +1,163 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' +import { render, screen } from '@testing-library/react' + +const { reportMock } = vi.hoisted(() => ({ + reportMock: vi.fn(async () => undefined), +})) + +vi.mock('../../lib/diagnosticsCapture', async (importOriginal) => ({ + ...(await importOriginal()), + reportReactError: reportMock, +})) + +// A card that fails the way an unforeseen stored record would. What breaks a real one +// is beside the point here; this file is about where the failure is allowed to land. +vi.mock('./AskUserQuestion', () => ({ + AskUserQuestion: ({ toolUseId }: { toolUseId: string }) => { + if (toolUseId === 'tool-poisoned') throw new Error('card exploded') + return
question card {toolUseId}
+ }, +})) + +import { MessageList } from './MessageList' +import { sessionsApi } from '../../api/sessions' +import { useChatStore, type PerSessionState } from '../../stores/chatStore' +import { useSettingsStore } from '../../stores/settingsStore' +import { useSessionStore } from '../../stores/sessionStore' +import { useTabStore } from '../../stores/tabStore' +import { useTeamStore } from '../../stores/teamStore' +import { useWorkspaceChatContextStore } from '../../stores/workspaceChatContextStore' +import { useWorkspaceStore } from '../../stores/workspaceStore' +import type { UIMessage } from '../../types/chat' + +const ACTIVE_TAB = 'active-tab' + +function makeSessionState(messages: UIMessage[]): PerSessionState { + return { + messages, + chatState: 'idle', + connectionState: 'connected', + historyStatus: 'ready', + historyHydrated: true, + streamingText: '', + streamingToolInput: '', + activeToolUseId: null, + activeToolName: null, + activeThinkingId: null, + pendingPermission: null, + pendingComputerUsePermission: null, + tokenUsage: { input_tokens: 0, output_tokens: 0 }, + streamingResponseChars: 0, + elapsedSeconds: 0, + statusVerb: '', + apiRetry: null, + slashCommands: [], + agentTaskNotifications: {}, + elapsedTimer: null, + composerPrefill: null, + } +} + +function askMessage(toolUseId: string, timestamp: number): UIMessage { + return { + id: `ask-${toolUseId}`, + type: 'tool_use', + toolName: 'AskUserQuestion', + toolUseId, + input: { questions: [{ question: 'Which scope?', options: [{ label: 'A' }, { label: 'B' }] }] }, + timestamp, + } +} + +describe('MessageList item containment', () => { + beforeEach(() => { + vi.restoreAllMocks() + reportMock.mockClear() + // React logs every caught render error; keep the output readable. + vi.spyOn(console, 'error').mockImplementation(() => {}) + useSettingsStore.setState({ locale: 'en' }) + useTabStore.setState({ + activeTabId: ACTIVE_TAB, + tabs: [{ sessionId: ACTIVE_TAB, title: 'Test', type: 'session' as const, status: 'idle' }], + }) + useSessionStore.setState({ sessions: [], activeSessionId: null, isLoading: false, error: null }) + useTeamStore.getState().clearTeam() + useWorkspaceChatContextStore.setState(useWorkspaceChatContextStore.getInitialState(), true) + useWorkspaceStore.setState(useWorkspaceStore.getInitialState(), true) + vi.spyOn(sessionsApi, 'getTurnCheckpoints').mockImplementation(() => new Promise(() => {})) + vi.spyOn(sessionsApi, 'getWorkspaceStatus').mockResolvedValue({ + state: 'ok', + workDir: '/tmp/example-project', + repoName: 'example-project', + branch: null, + isGitRepo: false, + changedFiles: [], + }) + }) + + // Issue #1400: one poisoned record in a saved transcript used to replace the whole + // app with the root error page, again on every launch. + it('keeps a row that fails to render from taking the transcript down', () => { + useChatStore.setState({ + sessions: { + [ACTIVE_TAB]: makeSessionState([ + { id: 'user-1', type: 'user_text', content: 'clean up the temp files', timestamp: 1 }, + { id: 'assistant-1', type: 'assistant_text', content: 'reply before the bad card', timestamp: 2 }, + askMessage('tool-poisoned', 3), + { id: 'assistant-2', type: 'assistant_text', content: 'reply after the bad card', timestamp: 4 }, + ]), + }, + }) + + render() + + expect(screen.getByText('clean up the temp files')).toBeTruthy() + expect(screen.getByText('reply before the bad card')).toBeTruthy() + expect(screen.getByText('reply after the bad card')).toBeTruthy() + expect(screen.getByText(/This item couldn't be displayed/)).toBeTruthy() + expect(reportMock).toHaveBeenCalledTimes(1) + }) + + it('leaves healthy cards in the same transcript rendered', () => { + useChatStore.setState({ + sessions: { + [ACTIVE_TAB]: makeSessionState([ + // Answered, so it stays in the history; only the latest unresolved question + // is ever shown, which would hide it otherwise. + askMessage('tool-fine', 1), + { + id: 'result-fine', + type: 'tool_result', + toolUseId: 'tool-fine', + content: { answers: { 'Which scope?': 'A' } }, + isError: false, + timestamp: 2, + }, + askMessage('tool-poisoned', 3), + ]), + }, + }) + + render() + + expect(screen.getByText('question card tool-fine')).toBeTruthy() + expect(screen.getAllByText(/This item couldn't be displayed/)).toHaveLength(1) + }) + + it('shows no notice and reports nothing for a transcript that renders', () => { + useChatStore.setState({ + sessions: { + [ACTIVE_TAB]: makeSessionState([ + { id: 'assistant-1', type: 'assistant_text', content: 'all good here', timestamp: 1 }, + askMessage('tool-fine', 2), + ]), + }, + }) + + render() + + expect(screen.getByText('all good here')).toBeTruthy() + expect(screen.queryByText(/couldn't be displayed/)).toBeNull() + expect(reportMock).not.toHaveBeenCalled() + }) +}) diff --git a/desktop/src/components/chat/MessageList.test.tsx b/desktop/src/components/chat/MessageList.test.tsx index 127d958b..c5292da7 100644 --- a/desktop/src/components/chat/MessageList.test.tsx +++ b/desktop/src/components/chat/MessageList.test.tsx @@ -7942,6 +7942,43 @@ describe('MessageList nested tool calls', () => { expect(screen.queryByText(/This model does not support images/)).toBeNull() }) + it.each([ + ['en', /too large for your provider or relay/, /removed automatically/], + ['zh', /服务商或中转站允许的大小/, /自动移除/], + ] as const)( + 'explains a request-too-large rejection as a provider limit and the recovery that follows (%s)', + (locale, limit, recovery) => { + useSettingsStore.setState({ locale }) + useChatStore.setState({ + sessions: { + [ACTIVE_TAB]: makeSessionState({ + messages: [ + { + id: 'error-1', + type: 'error', + code: 'invalid_request', + businessErrorCode: 'request_too_large', + message: + 'Request too large: your provider or relay rejected it (HTTP 413). This conversation is about 24MB.', + timestamp: 1, + }, + ], + }), + }, + }) + + render() + + expect(screen.getByText(limit)).toBeTruthy() + expect(screen.getByText(recovery)).toBeTruthy() + // The old copy blamed the selected model and asked users to delete files + // by hand, although the limit belongs to the provider and old media is + // now dropped automatically. + expect(screen.queryByText(/selected model|当前模型/)).toBeNull() + expect(screen.queryByText(/Remove large files|移除大文件/)).toBeNull() + }, + ) + it('restores opener focus without scrolling when its render item remains fully visible', async () => { useChatStore.setState({ sessions: { diff --git a/desktop/src/components/chat/MessageList.tsx b/desktop/src/components/chat/MessageList.tsx index 9318e6ba..a6c13d1c 100644 --- a/desktop/src/components/chat/MessageList.tsx +++ b/desktop/src/components/chat/MessageList.tsx @@ -28,6 +28,7 @@ import type { ActivityStep } from './activityGroupModel' import { ToolResultBlock } from './ToolResultBlock' import { PermissionDialog } from './PermissionDialog' import { AskUserQuestion } from './AskUserQuestion' +import { RenderItemBoundary } from './RenderItemBoundary' import { StreamingIndicator } from './StreamingIndicator' import { InlineTaskSummary } from './InlineTaskSummary' import { CurrentTurnChangeCard } from './CurrentTurnChangeCard' @@ -3555,70 +3556,72 @@ export function MessageList({ return ( <> - {item.kind === 'tool_group' ? ( - !toolResultMap.has(tc.toolUseId)) - } - // Only the tail of a live turn can still grow. Everything above it - // is finished, whatever any individual tool's state looks like this - // instant — which is why this, and not `isStreaming`, decides - // whether a run stands open. - isLive={chatState !== 'idle' && index === renderItems.length - 1 && !hasTrailingStreamingItem} - disclosureKey={getRenderItemKey(item)} - /> - ) : item.kind === 'team_card' ? ( - resolvedSessionId ? (() => { - const cardSnapshot = snapshotForTeamCard(teamSnapshot, item) - const fallbackPhase = item.endedAt !== undefined || teamTaskWindows.some((window) => ( - item.startedAt >= window.startedAt && - window.endedAt !== undefined && - item.startedAt <= window.endedAt - )) ? 'completed' : 'forming' - return ( - openTeamWorkbench(resolvedSessionId, cardSnapshot.team.name) - : undefined} - > - - - ) - })() : null - ) : ( - - )} + + {item.kind === 'tool_group' ? ( + !toolResultMap.has(tc.toolUseId)) + } + // Only the tail of a live turn can still grow. Everything above it + // is finished, whatever any individual tool's state looks like this + // instant — which is why this, and not `isStreaming`, decides + // whether a run stands open. + isLive={chatState !== 'idle' && index === renderItems.length - 1 && !hasTrailingStreamingItem} + disclosureKey={getRenderItemKey(item)} + /> + ) : item.kind === 'team_card' ? ( + resolvedSessionId ? (() => { + const cardSnapshot = snapshotForTeamCard(teamSnapshot, item) + const fallbackPhase = item.endedAt !== undefined || teamTaskWindows.some((window) => ( + item.startedAt >= window.startedAt && + window.endedAt !== undefined && + item.startedAt <= window.endedAt + )) ? 'completed' : 'forming' + return ( + openTeamWorkbench(resolvedSessionId, cardSnapshot.team.name) + : undefined} + > + + + ) + })() : null + ) : ( + + )} + {resolvedSessionId && cardsForItem.map((card) => { diff --git a/desktop/src/components/chat/PlanModePermissionDialog.test.tsx b/desktop/src/components/chat/PlanModePermissionDialog.test.tsx index c61fb119..c5c78225 100644 --- a/desktop/src/components/chat/PlanModePermissionDialog.test.tsx +++ b/desktop/src/components/chat/PlanModePermissionDialog.test.tsx @@ -389,7 +389,8 @@ describe('plan mode permission UI', () => { await waitFor(() => { expect(screen.getByTestId('model-selector-dropdown')).toBeTruthy() }) - fireEvent.click(screen.getByRole('button', { name: /Sonnet 5/ })) + // Anchored so the official catalog's "Sonnet 5.5" is not also matched. + fireEvent.click(screen.getByRole('button', { name: /^Sonnet 5(?![.\d])/ })) await waitFor(() => { expect(screen.getByTestId('plan-execution-model').textContent).toContain('applies on approve') @@ -445,7 +446,7 @@ describe('plan mode permission UI', () => { await waitFor(() => { expect(screen.getByTestId('model-selector-dropdown')).toBeTruthy() }) - fireEvent.click(screen.getByRole('button', { name: /Sonnet 5/ })) + fireEvent.click(screen.getByRole('button', { name: /^Sonnet 5(?![.\d])/ })) await waitFor(() => { expect(screen.getByTestId('plan-execution-model').textContent).toContain('applies on approve') }) diff --git a/desktop/src/components/chat/RenderItemBoundary.test.tsx b/desktop/src/components/chat/RenderItemBoundary.test.tsx new file mode 100644 index 00000000..d5212e52 --- /dev/null +++ b/desktop/src/components/chat/RenderItemBoundary.test.tsx @@ -0,0 +1,73 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { render, screen } from '@testing-library/react' + +const { reportMock } = vi.hoisted(() => ({ + reportMock: vi.fn(async () => undefined), +})) + +vi.mock('../../lib/diagnosticsCapture', () => ({ + reportReactError: reportMock, +})) + +import { RenderItemBoundary } from './RenderItemBoundary' +import { useSettingsStore } from '../../stores/settingsStore' + +function Explodes(): never { + throw new Error('row exploded') +} + +describe('RenderItemBoundary', () => { + beforeEach(() => { + reportMock.mockClear() + useSettingsStore.setState({ locale: 'en' }) + // React logs every caught render error; keep the output readable. + vi.spyOn(console, 'error').mockImplementation(() => {}) + }) + + afterEach(() => { + vi.restoreAllMocks() + }) + + it('renders its children untouched when nothing throws', () => { + render(fine) + + expect(screen.getByText('fine')).toBeTruthy() + expect(screen.queryByText(/couldn't be displayed/)).toBeNull() + expect(reportMock).not.toHaveBeenCalled() + }) + + // The point of the boundary: the failure stays where it happened. + it('replaces only the item that throws and leaves its neighbours alone', () => { + render( +
+ before + + after +
, + ) + + expect(screen.getByText('before')).toBeTruthy() + expect(screen.getByText('after')).toBeTruthy() + expect(screen.getByText(/couldn't be displayed/)).toBeTruthy() + }) + + it('still records the error in Diagnostics, with the component stack', () => { + render() + + expect(reportMock).toHaveBeenCalledTimes(1) + const [error, info] = reportMock.mock.calls[0] as unknown as [Error, { componentStack: string }] + expect(error.message).toBe('row exploded') + expect(info.componentStack).toContain('Explodes') + }) + + it('does not report a failed item again when its parent re-renders', () => { + const { rerender } = render() + expect(reportMock).toHaveBeenCalledTimes(1) + + rerender() + rerender() + + expect(reportMock).toHaveBeenCalledTimes(1) + expect(screen.getByText(/couldn't be displayed/)).toBeTruthy() + }) +}) diff --git a/desktop/src/components/chat/RenderItemBoundary.tsx b/desktop/src/components/chat/RenderItemBoundary.tsx new file mode 100644 index 00000000..42c8055f --- /dev/null +++ b/desktop/src/components/chat/RenderItemBoundary.tsx @@ -0,0 +1,48 @@ +import React from 'react' +import { TriangleAlert } from 'lucide-react' +import { t } from '../../i18n' +import { reportReactError } from '../../lib/diagnosticsCapture' + +type Props = { + children: React.ReactNode +} + +type State = { + failed: boolean +} + +/** + * Keeps one transcript item that fails to render from taking the whole app with it. + * + * The root ErrorBoundary replaces everything, and a transcript is rebuilt from saved + * history — so a single bad record (issue #1400: an AskUserQuestion input the model + * got wrong) crashed the app again on every launch, because the open tab is restored. + * Here the failure stays in its row: the rest of the conversation, the sidebar and the + * composer keep working, and the error still goes to Diagnostics. + * + * There is deliberately no automatic retry. A stored record fails the same way every + * time, and retrying on each render would report it on every update of a live list. + * The row is tried again whenever it is mounted afresh (switching session, reloading). + */ +export class RenderItemBoundary extends React.Component { + state: State = { failed: false } + + static getDerivedStateFromError(): State { + return { failed: true } + } + + componentDidCatch(error: unknown, errorInfo: React.ErrorInfo) { + void reportReactError(error, errorInfo) + } + + render() { + if (!this.state.failed) return this.props.children + + return ( +
+ + {t('errorBoundary.item')} +
+ ) + } +} diff --git a/desktop/src/components/controls/ModelSelector.test.tsx b/desktop/src/components/controls/ModelSelector.test.tsx index 2b85c717..6dc35d7d 100644 --- a/desktop/src/components/controls/ModelSelector.test.tsx +++ b/desktop/src/components/controls/ModelSelector.test.tsx @@ -194,7 +194,7 @@ describe('ModelSelector', () => { }, ) - it('keeps the current Claude Official catalog visible when the API returns legacy settings models', async () => { + function renderClaudeOfficialWithLegacyModels() { const legacyModels: ModelInfo[] = [ { id: 'claude-opus-4-7', name: 'Opus 4.7', description: 'Legacy Opus', context: '1m' }, { id: 'claude-sonnet-4-6', name: 'Sonnet 4.6', description: 'Legacy Sonnet', context: '200k' }, @@ -224,6 +224,11 @@ describe('ModelSelector', () => { const onRuntimeChange = vi.fn() render() + return onRuntimeChange + } + + it('keeps the current Claude Official catalog visible when the API returns legacy settings models', async () => { + const onRuntimeChange = renderClaudeOfficialWithLegacyModels() await clickByRole(/Opus 4\.7/i) @@ -231,7 +236,8 @@ describe('ModelSelector', () => { expect(screen.getByRole('button', { name: /Opus 5\.5/ })).toBeInTheDocument() expect(screen.getByRole('button', { name: /^Opus 5 / })).toBeInTheDocument() expect(screen.getByRole('button', { name: /Opus 4\.8/ })).toBeInTheDocument() - expect(screen.getByRole('button', { name: /Sonnet 5/ })).toBeInTheDocument() + expect(screen.getByRole('button', { name: /Sonnet 5\.5/ })).toBeInTheDocument() + expect(screen.getByRole('button', { name: /^Sonnet 5 / })).toBeInTheDocument() expect(screen.getAllByRole('button', { name: /Opus 4\.7/ }).length).toBeGreaterThan(0) await clickByRole(/Opus 5\.5/) expect(onRuntimeChange).toHaveBeenCalledWith(expect.objectContaining({ @@ -239,6 +245,20 @@ describe('ModelSelector', () => { })) }) + it('selects Sonnet 5.5 from the Claude Official catalog instead of the Sonnet 5 it supersedes', async () => { + const onRuntimeChange = renderClaudeOfficialWithLegacyModels() + + await clickByRole(/Opus 4\.7/i) + await clickByRole(/Sonnet 5\.5/) + + expect(onRuntimeChange).toHaveBeenCalledWith(expect.objectContaining({ + providerId: null, modelId: 'claude-sonnet-5-5', + })) + expect(onRuntimeChange).not.toHaveBeenCalledWith(expect.objectContaining({ + modelId: 'claude-sonnet-5', + })) + }) + it('does not query official OAuth status when mounted', () => { const fetchClaudeStatus = vi.fn(async () => {}) const fetchOpenAIStatus = vi.fn(async () => {}) diff --git a/desktop/src/components/layout/AppShell.test.tsx b/desktop/src/components/layout/AppShell.test.tsx index a4699c9e..612436d3 100644 --- a/desktop/src/components/layout/AppShell.test.tsx +++ b/desktop/src/components/layout/AppShell.test.tsx @@ -122,6 +122,15 @@ vi.mock('./ContentRouter', () => ({ ContentRouter: () =>
content loaded
, })) +// The real one subscribes to the chat store, which this file replaces with a +// `{ getState }` stub. Its behaviour has its own test; here only the wiring — +// which session it is told is on screen, and where it sits — is checked. +vi.mock('./MobileAttentionDot', () => ({ + MobileAttentionDot: ({ activeSessionId }: { activeSessionId: string | null }) => ( + + ), +})) + vi.mock('./TabBar', () => ({ TabBar: () => , })) @@ -666,6 +675,22 @@ describe('AppShell boot flow', () => { // platform minimum for primary touch targets is 44, not the 40 it shipped // at (IconButton's own size doc had the two tiers reversed). expect(screen.getByTestId('mobile-sidebar-toggle')).toHaveClass('h-11', 'w-11') + + // The hamburger's corner is where the drawer's waiting marks get announced + // from, and the dot is told which session is on screen so it can leave that + // one out. It shares a wrapper with the button so it can sit on its corner. + const dot = screen.getByTestId('mobile-attention-dot-stub') + expect(dot).toHaveAttribute('data-active-session', 'session-mobile') + expect(screen.getByTestId('mobile-sidebar-toggle').parentElement).toContainElement(dot) + }) + + it('does not put the waiting dot on a desktop window, which has the tab strip instead', async () => { + mocks.isMobile = false + + render() + + await screen.findByText('content loaded') + expect(screen.queryByTestId('mobile-attention-dot-stub')).not.toBeInTheDocument() }) it('keeps browser H5 settings active alongside existing chat tabs', async () => { diff --git a/desktop/src/components/layout/AppShell.tsx b/desktop/src/components/layout/AppShell.tsx index 9237ae62..47cc94c9 100644 --- a/desktop/src/components/layout/AppShell.tsx +++ b/desktop/src/components/layout/AppShell.tsx @@ -27,6 +27,7 @@ import { } from '../../stores/projectDisplayNameStore' import { openDesktopNotificationTarget } from '../../lib/desktopNotificationNavigation' import { TabBar } from './TabBar' +import { MobileAttentionDot } from './MobileAttentionDot' import { WorkspaceHeaderProvider } from './WorkspaceHeaderContext' import { StartupErrorView } from './StartupErrorView' import { useTabStore, SETTINGS_TAB_ID } from '../../stores/tabStore' @@ -362,15 +363,20 @@ export function AppShell() { data-testid="mobile-session-header" className="flex shrink-0 items-center gap-3 border-b border-[var(--color-border)] bg-[var(--color-surface)] px-3 py-2" > - + + + {/* 手机没有 tab 栏,抽屉是切换会话的唯一入口,也是等待标志唯一能被 + 找到的地方:别的会话在等人时,在汉堡按钮上提一下。 */} + + {activeTab?.type === 'settings' ? (

{t('sidebar.settings')}

) : isActiveChatTab ? ( diff --git a/desktop/src/components/layout/MobileAttentionDot.test.tsx b/desktop/src/components/layout/MobileAttentionDot.test.tsx new file mode 100644 index 00000000..ff282bc0 --- /dev/null +++ b/desktop/src/components/layout/MobileAttentionDot.test.tsx @@ -0,0 +1,97 @@ +import { act, cleanup, render, screen } from '@testing-library/react' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import '@testing-library/jest-dom' +import { createDefaultSessionState, useChatStore, type PerSessionState } from '../../stores/chatStore' +import { MobileAttentionDot } from './MobileAttentionDot' + +vi.mock('../../i18n', () => ({ + useTranslation: () => (key: string) => key, +})) + +function session(overrides: Partial = {}): PerSessionState { + return { ...createDefaultSessionState(), ...overrides } +} + +const openRequest: Partial = { + pendingPermission: { requestId: 'r1', toolName: 'Bash', toolUseId: 'tu-1', input: {} }, + pendingPermissions: { r1: { requestId: 'r1', toolName: 'Bash', toolUseId: 'tu-1', input: {} } }, +} + +function seed(sessions: Record) { + act(() => { + useChatStore.setState({ sessions }) + }) +} + +describe('MobileAttentionDot', () => { + beforeEach(() => { + useChatStore.setState({ sessions: {} }) + }) + + afterEach(() => { + cleanup() + useChatStore.setState({ sessions: {} }) + }) + + it('renders nothing while no session is waiting', () => { + seed({ a: session(), b: session({ chatState: 'thinking' }) }) + const { container } = render() + + expect(container).toBeEmptyDOMElement() + }) + + it('lights when a session other than the one on screen is waiting', () => { + seed({ a: session(), b: session(openRequest) }) + render() + + expect(screen.getByTestId('mobile-attention-dot')).toBeInTheDocument() + expect(screen.getByText('sidebar.sessionNeedsAttention')).toHaveClass('sr-only') + }) + + it('stays dark when the only waiting session is the one on screen', () => { + seed({ a: session(openRequest), b: session() }) + const { container } = render() + + // Its card is already in front of the person. + expect(container).toBeEmptyDOMElement() + }) + + it('lights from a page that is not a session, such as Settings', () => { + seed({ a: session(openRequest) }) + render() + + expect(screen.getByTestId('mobile-attention-dot')).toBeInTheDocument() + }) + + it('counts sessions that have no tab, since the drawer lists them all', () => { + // Nothing here mentions tabs: the dot reads the chat store, not the strip. + seed({ a: session(), 'no-tab': session(openRequest) }) + render() + + expect(screen.getByTestId('mobile-attention-dot')).toBeInTheDocument() + }) + + it('does not light for chatState alone, since no card exists for it', () => { + seed({ a: session(), b: session({ chatState: 'permission_pending' }) }) + const { container } = render() + + expect(container).toBeEmptyDOMElement() + }) + + it('goes out once the request is answered', () => { + seed({ a: session(), b: session(openRequest) }) + render() + expect(screen.getByTestId('mobile-attention-dot')).toBeInTheDocument() + + seed({ a: session(), b: session({ pendingPermission: null, pendingPermissions: {} }) }) + + expect(screen.queryByTestId('mobile-attention-dot')).not.toBeInTheDocument() + }) + + it('is decorative for a screen reader, which gets the hidden text instead', () => { + seed({ a: session(), b: session(openRequest) }) + render() + + expect(screen.getByTestId('mobile-attention-dot')).toHaveAttribute('aria-hidden', 'true') + }) +}) diff --git a/desktop/src/components/layout/MobileAttentionDot.tsx b/desktop/src/components/layout/MobileAttentionDot.tsx new file mode 100644 index 00000000..dd22c0ce --- /dev/null +++ b/desktop/src/components/layout/MobileAttentionDot.tsx @@ -0,0 +1,39 @@ +import { StatusDot } from '@/components/ui/Badge' +import { useTranslation } from '../../i18n' +import { sessionNeedsAttention } from '../../lib/sessionAttention' +import { useChatStore } from '../../stores/chatStore' + +/** + * A dot in the corner of the phone header's hamburger when a session other than + * the one on screen is waiting for the person. + * + * The phone has no tab strip: the sidebar drawer behind that button is the only + * way to change session, so it is the only place the marks on the session rows + * can be found from. On H5 neither a notification click nor `requestAttention` + * exists, so without this the drawer would have to be opened on a guess. + * + * Its own session is left out — the card is already in front of the person — + * and every session counts, not only ones with a tab, because the drawer lists + * them all. It renders inside a `relative` wrapper around the button. + */ +export function MobileAttentionDot({ activeSessionId }: { activeSessionId: string | null }) { + const t = useTranslation() + const waitingElsewhere = useChatStore((state) => + Object.entries(state.sessions).some( + ([sessionId, session]) => sessionId !== activeSessionId && sessionNeedsAttention(session), + )) + + if (!waitingElsewhere) return null + + return ( + <> + + {t('sidebar.sessionNeedsAttention')} + + ) +} diff --git a/desktop/src/components/layout/SessionAttentionMark.test.tsx b/desktop/src/components/layout/SessionAttentionMark.test.tsx new file mode 100644 index 00000000..a70d0ddc --- /dev/null +++ b/desktop/src/components/layout/SessionAttentionMark.test.tsx @@ -0,0 +1,61 @@ +import { cleanup, render, screen } from '@testing-library/react' +import { afterEach, describe, expect, it } from 'vitest' +import '@testing-library/jest-dom' +import { SessionAttentionMark } from './SessionAttentionMark' + +afterEach(cleanup) + +describe('SessionAttentionMark', () => { + it('is one image with the given name', () => { + render() + + expect(screen.getByRole('img', { name: 'Waiting for your approval' })).toBeInTheDocument() + }) + + it('repeats the name as a native tooltip', () => { + render() + + expect(screen.getByRole('img')).toHaveAttribute('title', 'Waiting for your approval') + }) + + it('is not a live region, which would promise an announcement it cannot make', () => { + render() + + expect(screen.queryByRole('status')).not.toBeInTheDocument() + }) + + it('hides the ligature text so a screen reader does not say "warning" beside the label', () => { + const { container } = render() + + const glyph = container.querySelector('.material-symbols-outlined') + expect(glyph).toHaveTextContent('warning') + expect(glyph).toHaveAttribute('aria-hidden', 'true') + }) + + it('draws a filled glyph', () => { + const { container } = render() + + // Outlined at 14px the exclamation mark is a hairline; filled it is a shape. + expect(container.querySelector('.material-symbols-outlined')).toHaveStyle({ + fontVariationSettings: "'FILL' 1", + }) + }) + + it('pulses three times and then holds still instead of blinking for hours', () => { + const { container } = render() + + const glyph = container.querySelector('.material-symbols-outlined') + expect(glyph).toHaveClass('animate-pulse-dot') + // Inline, so it overrides the `infinite` the class carries. + expect(glyph?.style.animationIterationCount).toBe('3') + }) + + it('takes its colour from the darker warning token and hard-codes none', () => { + const { container } = render() + + // `--color-warning` is 2.92:1 on warm-classic's hovered sidebar row, and a + // bare glyph has no text to fall back on; contrast.test.ts measures this token. + expect(screen.getByRole('img').className).toContain('var(--color-on-warning-container)') + expect(container.innerHTML).not.toMatch(/#[0-9a-f]{3,8}\b/i) + }) +}) diff --git a/desktop/src/components/layout/SessionAttentionMark.tsx b/desktop/src/components/layout/SessionAttentionMark.tsx new file mode 100644 index 00000000..47dc8ba4 --- /dev/null +++ b/desktop/src/components/layout/SessionAttentionMark.tsx @@ -0,0 +1,48 @@ +/** + * "This session is waiting for you". The tab strip and both sidebar views draw + * it through this one component, so `sessionNeedsAttention` keeping them in step + * in state is matched by them staying in step in pixels. + * + * A glyph, not a coloured dot. A session parked on a permission card is still + * running as far as `chatState` goes, and the running marker is a dot; warning + * and brand are within 1.4–1.7:1 of each other in most themes, so a dot against + * a dot leaves colour alone to tell "working" from "needs you", which is no + * answer for anyone who cannot rely on it. The filled triangle has a different + * outline from every dot in the app. + * + * It pulses three times on arrival and then holds still. This state can sit + * for hours, and `pulse-dot` dips to 0.3 opacity — a thin glyph would spend half + * of every cycle unreadable, and an animation that never ends is not something + * to leave running beside someone's work (WCAG 2.2.2). Reduced-motion users + * already get the slower 3s cycle from the global rule. + * + * `role="img"`, not `status`: a live region announces changes to its content, + * and an empty span's aria-label is not one, so `status` promised an + * announcement nothing delivers. The inner ligature text is hidden for the same + * reason — left in, a screen reader would say "warning" next to the label. + * + * The colour is `--color-on-warning-container`, the darker of the two warning + * tokens, not `--color-warning`. This is a bare graphic with no text beside it + * to fall back on, so it has to clear 3:1 on every ground it can sit on, and + * `--color-warning` does not: it is 2.92:1 on warm-classic's hovered sidebar + * row. The two are the same colour in the ink themes, where nothing changes. + * `contrast.test.ts` measures all four grounds in all six themes. + */ +export function SessionAttentionMark({ label }: { label: string }) { + return ( + + + + ) +} diff --git a/desktop/src/components/layout/Sidebar.test.tsx b/desktop/src/components/layout/Sidebar.test.tsx index 5ec9ff57..16178368 100644 --- a/desktop/src/components/layout/Sidebar.test.tsx +++ b/desktop/src/components/layout/Sidebar.test.tsx @@ -296,6 +296,13 @@ function makeChatSessionState(overrides: Partial = {}): PerSess } } +// 一条挂起的授权请求,形状与 store 里的一致:`pendingPermission` 是最新一条的镜像, +// `pendingPermissions` 是全集。 +const openRequest: Partial = { + pendingPermission: { requestId: 'r1', toolName: 'Bash', toolUseId: 'tu-1', input: {} }, + pendingPermissions: { r1: { requestId: 'r1', toolName: 'Bash', toolUseId: 'tu-1', input: {} } }, +} + function createDeferred() { let resolve!: (value: T) => void let reject!: (reason?: unknown) => void @@ -572,6 +579,75 @@ describe('Sidebar', () => { expect(screen.getByRole('button', { name: 'Collapse display' })).toBeInTheDocument() }) + describe('a folded project with a session waiting on the user', () => { + // Eleven sessions against a six-row preview: the last two are past the fold. + function seedFoldedProject() { + const base = new Date('2026-05-15T10:00:00.000Z').getTime() + useSessionStore.setState({ + sessions: Array.from({ length: 11 }, (_, index) => ( + makeSession( + `alpha-${index + 1}`, + index === 10 ? 'Alpha waiting' : index === 9 ? 'Alpha folded' : `Alpha ${index + 1}`, + '/workspace/alpha', + new Date(base - index * 1000).toISOString(), + ) + )), + }) + } + + it('keeps the waiting row on screen while its quiet neighbour stays folded away', () => { + seedFoldedProject() + useChatStore.setState({ + sessions: { 'alpha-11': makeChatSessionState({ chatState: 'tool_executing', ...openRequest }) }, + } as Partial>) + + render() + + // On a phone the dot on the menu button points at this row, and there is no + // tab strip to fall back on: a row folded out of the drawer would end the + // signal one step short. + const waitingRow = screen.getByRole('button', { name: /Alpha waiting/ }) + expect(within(waitingRow).getByLabelText('Waiting for your approval')).toBeInTheDocument() + expect(screen.queryByRole('button', { name: /Alpha folded/ })).not.toBeInTheDocument() + }) + + it('folds the row away again once it stops waiting', () => { + seedFoldedProject() + useChatStore.setState({ + sessions: { 'alpha-11': makeChatSessionState({ chatState: 'tool_executing', ...openRequest }) }, + } as Partial>) + render() + expect(screen.getByRole('button', { name: /Alpha waiting/ })).toBeInTheDocument() + + act(() => { + useChatStore.setState({ + sessions: { + 'alpha-11': makeChatSessionState({ chatState: 'tool_executing', pendingPermission: null, pendingPermissions: {} }), + }, + } as Partial>) + }) + + expect(screen.queryByRole('button', { name: /Alpha waiting/ })).not.toBeInTheDocument() + }) + + it('keeps the open session and a waiting one visible together, in list order', () => { + seedFoldedProject() + useTabStore.setState({ + tabs: [{ sessionId: 'alpha-10', title: 'Alpha folded', type: 'session', status: 'idle' }], + activeTabId: 'alpha-10', + }) + useChatStore.setState({ + sessions: { 'alpha-11': makeChatSessionState({ chatState: 'tool_executing', ...openRequest }) }, + } as Partial>) + + render() + + const rows = screen.getAllByRole('button', { name: /^Alpha / }).map((row) => row.textContent ?? '') + expect(rows.at(-2)).toContain('Alpha folded') + expect(rows.at(-1)).toContain('Alpha waiting') + }) + }) + it('does not show a fold control when a project is at or below the collapse threshold', () => { const base = new Date('2026-05-15T10:00:00.000Z').getTime() useSessionStore.setState({ @@ -2807,7 +2883,9 @@ describe('Sidebar', () => { seedSessions() useChatStore.setState({ sessions: { - 'today-1': makeChatSessionState({ chatState: 'permission_pending' }), + // chatState 故意不是 'permission_pending':`status` 消息会在卡片还开着时 + // 把它改成别的,只看 chatState 的实现在这里会熄灯。 + 'today-1': makeChatSessionState({ chatState: 'tool_executing', ...openRequest }), }, } as Partial>) @@ -2820,6 +2898,71 @@ describe('Sidebar', () => { expect(within(waitingRow).queryByLabelText('Session running')).not.toBeInTheDocument() }) + it('does not call a session waiting from chatState alone, since no card exists for it', async () => { + seedSessions() + useChatStore.setState({ + sessions: { + 'today-1': makeChatSessionState({ chatState: 'permission_pending' }), + }, + } as Partial>) + + render() + await act(async () => { toggleBell() }) + + const runningGroup = screen.getByTestId('sidebar-task-group-running') + const row = within(runningGroup).getByRole('button', { name: /Today Session/ }) + expect(within(row).queryByLabelText('Waiting for your approval')).not.toBeInTheDocument() + expect(within(row).getByLabelText('Session running')).toBeInTheDocument() + }) + + it('marks a waiting session in the project view too, in place of the running spinner', () => { + seedSessions() + // 没有为它开 tab:侧边栏要能标出没有 tab 的会话,tab 栏那边看不到它。 + useChatStore.setState({ + sessions: { + 'today-1': makeChatSessionState({ chatState: 'tool_executing', ...openRequest }), + 'yesterday-1': makeChatSessionState({ chatState: 'thinking' }), + }, + } as Partial>) + + render() + + const waitingRow = screen.getByRole('button', { name: /Today Session/ }) + expect(within(waitingRow).getByLabelText('Waiting for your approval')).toBeInTheDocument() + expect(within(waitingRow).queryByLabelText('Session running')).not.toBeInTheDocument() + + // 只是在跑、没有在等的会话仍然转圈。 + const busyRow = screen.getByRole('button', { name: /^Yesterday Session/ }) + expect(within(busyRow).getByLabelText('Session running')).toBeInTheDocument() + expect(within(busyRow).queryByLabelText('Waiting for your approval')).not.toBeInTheDocument() + }) + + it('hands a project-view row back to the spinner once the request is answered', () => { + seedSessions() + useChatStore.setState({ + sessions: { + 'today-1': makeChatSessionState({ chatState: 'tool_executing', ...openRequest }), + }, + } as Partial>) + + render() + expect(within(screen.getByRole('button', { name: /Today Session/ })).getByLabelText('Waiting for your approval')) + .toBeInTheDocument() + + // 往返:授权被处理后标志要撤掉,会话还在跑,所以回到转圈。 + act(() => { + useChatStore.setState({ + sessions: { + 'today-1': makeChatSessionState({ chatState: 'tool_executing', pendingPermission: null, pendingPermissions: {} }), + }, + } as Partial>) + }) + + const row = screen.getByRole('button', { name: /Today Session/ }) + expect(within(row).queryByLabelText('Waiting for your approval')).not.toBeInTheDocument() + expect(within(row).getByLabelText('Session running')).toBeInTheDocument() + }) + it('lights the bell when the organize menu picks by time, because they are one state', async () => { seedSessions() render() diff --git a/desktop/src/components/layout/Sidebar.tsx b/desktop/src/components/layout/Sidebar.tsx index 91b16d89..fcc6dc03 100644 --- a/desktop/src/components/layout/Sidebar.tsx +++ b/desktop/src/components/layout/Sidebar.tsx @@ -43,7 +43,9 @@ import { } from '../../api/desktopUiPreferences' import { getDesktopHost } from '../../lib/desktopHost' import { hasRunningBackgroundTasks } from '../../lib/backgroundTasks' +import { collectAttentionIds } from '../../lib/sessionAttention' import { getSessionWorkspaceState, getSessionSeedWorkDir } from '../../lib/sessionWorkspace' +import { SessionAttentionMark } from './SessionAttentionMark' const desktopHost = getDesktopHost() const isDesktopRuntime = desktopHost.isDesktop @@ -297,14 +299,11 @@ export function Sidebar({ return ids }, [chatSessions, tabs]) // 停在权限请求上的会话在 `runningSessionIds` 里也算「没结束」,但它不是在 - // 干活而是在等人。任务视图要把这两种状态分开显示。 - const attentionSessionIds = useMemo(() => { - const ids = new Set() - for (const [sessionId, sessionState] of Object.entries(chatSessions)) { - if (sessionState.chatState === 'permission_pending') ids.add(sessionId) - } - return ids - }, [chatSessions]) + // 干活而是在等人。两个视图都要把这两种状态分开显示。 + // 判定看挂起的请求记录而不是 `chatState`:`status` / `session_state` 消息会在 + // 卡片还开着的时候把 chatState 改掉,按它判定会在有卡的会话上熄灯。tab 栏与 + // 这里共用同一个 `sessionNeedsAttention`,不要各写各的。 + const attentionSessionIds = useMemo(() => new Set(collectAttentionIds(chatSessions)), [chatSessions]) const taskGroups = useMemo(() => { if (!isTaskView) return [] // 隐藏的项目在任务视图里也要隐藏,否则两个视图对「有哪些会话」说法不一致。 @@ -1212,7 +1211,7 @@ export function Sidebar({ const sessionsExpanded = expandedProjectKeys.has(project.key) const visibleItems = projectCollapsed ? [] - : getVisibleProjectSessions(project.sessions, sessionsExpanded, activeTabId) + : getVisibleProjectSessions(project.sessions, sessionsExpanded, activeTabId, attentionSessionIds) const hiddenCount = project.sessions.length - visibleItems.length const projectSessionTotal = projectSessionTotals[project.key] const hasUnloadedSessions = projectSessionTotal === undefined @@ -1418,6 +1417,7 @@ export function Sidebar({ )} , ): SessionListItem[] { if (expanded || sessions.length <= PROJECT_GROUP_VISIBLE_COUNT) return sessions const visible = sessions.slice(0, PROJECT_GROUP_VISIBLE_COUNT) - if (!activeSessionId || visible.some((session) => session.id === activeSessionId)) return visible - - const activeSession = sessions.find((session) => session.id === activeSessionId) - return activeSession ? [...visible, activeSession] : visible + // 折叠掉的行是要人自己去翻的。当前打开的会话不该被翻到看不见,正在等人的会话 + // 更不该:手机上没有 tab 栏,汉堡按钮上的提示点指向的就是抽屉里这一行, + // 折叠起来等于信号在最后一步断了。它们按列表原有顺序排在前几行之后。 + // 只管已经加载进列表的行:侧边栏每个项目只预取最近 + // `SIDEBAR_PROJECT_SESSION_PREVIEW_LIMIT` 条,更早的会话在列表里根本没有行, + // 桌面上由 tab 栏兜底,手机上要展开历史后才看得到。 + const pinned = sessions + .slice(PROJECT_GROUP_VISIBLE_COUNT) + .filter((session) => session.id === activeSessionId || attentionSessionIds.has(session.id)) + return pinned.length > 0 ? [...visible, ...pinned] : visible } function compareSessionsByTimestamp( @@ -2318,11 +2325,13 @@ function ProjectMenuItem({ function SessionRowMeta({ isRunning, + needsAttention, isWorktree, modifiedAt, t, }: { isRunning: boolean + needsAttention: boolean isWorktree: boolean modifiedAt: string t: (key: TranslationKey, params?: Record) => string @@ -2335,7 +2344,14 @@ function SessionRowMeta({ className="ml-auto flex h-5 flex-shrink-0 items-center justify-end gap-1.5 whitespace-nowrap text-[10px] font-medium tabular-nums text-[var(--color-text-tertiary)]" title={updatedLabel} > - {isRunning && ( + {/* 等人比在跑更要紧:停在卡片上的会话按 chatState 也算「在跑」,但转圈会 + 让人以为可以不管它。 */} + {needsAttention && ( + + + + )} + {isRunning && !needsAttention && ( {title} {needsAttention ? ( - - {/* 名字挂在 StatusDot 自己身上:它带 `role="status"`,裸 span 上的 - aria-label 多数读屏并不播报。 */} - + + {/* 与 tab 栏、项目视图共用同一个标志;名字与 title 都在它自己身上。 */} + ) : isRunning ? ( { + it('is named by what it does, not by the number on it', () => { + render() + + expect(screen.getByRole('button', { name: 'Jump to the next waiting session (3 waiting)' })).toBeInTheDocument() + }) + + it('shows how many sessions a press can reach', () => { + render() + + expect(screen.getByTestId('tab-attention-jump')).toHaveTextContent('3') + }) + + it.each([ + [9, '9'], + [10, '9+'], + [42, '9+'], + ])('shows %i as %s so the pill never outgrows the toolbar', (count, shown) => { + render() + + expect(screen.getByTestId('tab-attention-jump')).toHaveTextContent(new RegExp(`^warning${shown.replace('+', '\\+')}$`)) + }) + + it('jumps when pressed', () => { + const onJump = vi.fn() + render() + + fireEvent.click(screen.getByRole('button', { name: 'Jump' })) + + expect(onJump).toHaveBeenCalledTimes(1) + }) + + it('repeats the label as a native tooltip', () => { + render() + + expect(screen.getByRole('button')).toHaveAttribute('title', 'Jump to the next waiting session') + }) + + it('hides the glyph text from a screen reader', () => { + const { container } = render() + + expect(container.querySelector('.material-symbols-outlined')).toHaveAttribute('aria-hidden', 'true') + }) + + it('does not submit a surrounding form', () => { + render() + + expect(screen.getByRole('button')).toHaveAttribute('type', 'button') + }) + + it('is marked interactive so the window drag region never swallows the press', () => { + render() + + expect(screen.getByRole('button')).toHaveClass('tab-bar-interactive') + }) + + it('wears the warning container pair and hard-codes no colour', () => { + const { container } = render() + + expect(container.innerHTML).toContain('var(--color-warning-container)') + expect(container.innerHTML).toContain('var(--color-on-warning-container)') + expect(container.innerHTML).not.toMatch(/#[0-9a-f]{3,8}\b/i) + }) +}) diff --git a/desktop/src/components/layout/TabAttentionJump.tsx b/desktop/src/components/layout/TabAttentionJump.tsx new file mode 100644 index 00000000..a019002d --- /dev/null +++ b/desktop/src/components/layout/TabAttentionJump.tsx @@ -0,0 +1,55 @@ +import { Badge } from '@/components/ui/Badge' + +/** + * "Some sessions are waiting for you — take me to the next one." + * + * The mark on a tab says which tab; this says how many, and turns "find them + * in a strip that scrolls" into one press. A person coming back to a run of + * reviewers all stopped on the same kind of card can clear them one after + * another without hunting for each. + * + * `count` is the waiting sessions other than the one on screen — the places a + * press can actually go — and the caller only renders this when it is above + * zero. The pill is a `Badge` so it borrows the warning container pair that + * the contrast tests already cover, inside a real button that owns the focus + * ring and the press. Static: the pulse is the tab mark's, and a second thing + * breathing at the other end of the strip would only compete with it. + */ +export function TabAttentionJump({ + count, + label, + onJump, +}: { + count: number + label: string + onJump: () => void +}) { + return ( + + ) +} diff --git a/desktop/src/components/layout/TabBar.test.tsx b/desktop/src/components/layout/TabBar.test.tsx index 686e46b7..761f30df 100644 --- a/desktop/src/components/layout/TabBar.test.tsx +++ b/desktop/src/components/layout/TabBar.test.tsx @@ -1,8 +1,8 @@ -import { act, cleanup, fireEvent, render, screen, waitFor } from '@testing-library/react' +import { act, cleanup, fireEvent, render, screen, waitFor, within } from '@testing-library/react' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import '@testing-library/jest-dom' import type { PerSessionState } from '../../stores/chatStore' -import type { ChatState, UIMessage } from '../../types/chat' +import type { BackgroundAgentTask, ChatState, ServerMessage, UIMessage } from '../../types/chat' import type { TeamWorkbenchSessionTimeline, TeamWorkbenchTask, TeamWorkbenchTimeline } from '../../types/team' import type { WorkflowRun } from '../../types/workflow' import { browserHost } from '../../lib/desktopHost/browserHost' @@ -58,7 +58,7 @@ function fireStripResize() { }) } -function makeChatSession(chatState: ChatState): PerSessionState { +function makeChatSession(chatState: ChatState, overrides: Partial = {}): PerSessionState { return { messages: [], chatState, @@ -81,6 +81,30 @@ function makeChatSession(chatState: ChatState): PerSessionState { elapsedTimer: null, composerPrefill: null, composerDraft: null, + ...overrides, + } +} + +function makeBackgroundTask( + taskId: string, + overrides: Partial = {}, +): BackgroundAgentTask { + return { + taskId, + toolUseId: `tool-${taskId}`, + status: 'running', + taskType: 'local_bash', + description: `Task ${taskId}`, + startedAt: 1, + updatedAt: 2, + ...overrides, + } +} + +function makeSessionWithTasks(chatState: ChatState, tasks: BackgroundAgentTask[]): PerSessionState { + return { + ...makeChatSession(chatState), + backgroundAgentTasks: Object.fromEntries(tasks.map((task) => [task.taskId, task])), } } @@ -144,6 +168,7 @@ vi.mock('../../i18n', () => ({ useTranslation: () => (key: string, params?: Record) => { const translations: Record = { 'sidebar.extensions': 'Extension Market', + 'sidebar.sessionNeedsAttention': 'Waiting for your approval', 'tabs.close': 'Close', 'tabs.closeOthers': 'Close Others', 'tabs.closeLeft': 'Close Left', @@ -167,6 +192,7 @@ vi.mock('../../i18n', () => ({ 'agentTeams.hideReport': 'Hide Agent Teams Run Report', 'tabs.scrollLeft': 'Scroll tabs left', 'tabs.scrollRight': 'Scroll tabs right', + 'tabs.jumpToAttention': 'Jump to the next waiting session ({count} waiting)', 'tabs.closeTab': 'Close {title}', 'tabs.untitled': 'Untitled', 'settings.title': 'Localized Settings', @@ -2883,6 +2909,250 @@ describe('TabBar', () => { expect(useTabStore.getState().tabs).toEqual([]) }) + // The "Session running" dialog counts background tasks as work in progress, so + // its Stop & Close has to stop them. `stopGeneration` alone does not: the server + // only interrupts the foreground turn and Agent tasks, which left a background + // shell command (and the session it kept busy) alive after "Stop & Close" and + // made the reopened session ask the same question again (issue #1398). + // + // These tests run the real chat-store actions on top of a recorded socket so + // they see what would really leave the client, and in which order. + describe('stopping background tasks when closing a running session', () => { + async function recordSocketTraffic() { + const { wsManager } = await import('../../api/websocket') + const traffic: string[] = [] + vi.spyOn(wsManager, 'send').mockImplementation((sessionId, message) => { + traffic.push( + `send ${sessionId} ${message.type}${message.type === 'stop_background_task' ? ` ${message.taskId}` : ''}`, + ) + }) + vi.spyOn(wsManager, 'disconnect').mockImplementation((sessionId) => { + traffic.push(`disconnect ${sessionId}`) + }) + return traffic + } + + it('stops the background shell task that kept an otherwise idle session running', async () => { + const { TabBar } = await import('./TabBar') + const { useTabStore } = await import('../../stores/tabStore') + const { useChatStore } = await import('../../stores/chatStore') + const traffic = await recordSocketTraffic() + + useTabStore.setState({ + tabs: [{ sessionId: 'tab-shell', title: 'Shell Session', type: 'session', status: 'idle' }], + activeTabId: 'tab-shell', + }) + useChatStore.setState({ + sessions: { + 'tab-shell': makeSessionWithTasks('idle', [makeBackgroundTask('shell-1')]), + }, + } as Partial>) + + await act(async () => { + render() + }) + + fireEvent.click(screen.getByLabelText('Close Shell Session')) + + expect(screen.getByRole('dialog', { name: 'Session Running' })).toBeInTheDocument() + expect(traffic).toEqual([]) + + fireEvent.click(screen.getByText('Stop & Close')) + + expect(traffic).toEqual([ + 'send tab-shell stop_generation', + 'send tab-shell stop_background_task shell-1', + 'disconnect tab-shell', + ]) + expect(useTabStore.getState().tabs).toEqual([]) + }) + + it('stops only the tasks that are still running and count as session activity', async () => { + const { TabBar } = await import('./TabBar') + const { useTabStore } = await import('../../stores/tabStore') + const { useChatStore } = await import('../../stores/chatStore') + const traffic = await recordSocketTraffic() + + useTabStore.setState({ + tabs: [{ sessionId: 'tab-mixed', title: 'Mixed Session', type: 'session', status: 'running' }], + activeTabId: 'tab-mixed', + }) + useChatStore.setState({ + sessions: { + 'tab-mixed': makeSessionWithTasks('thinking', [ + makeBackgroundTask('shell-running'), + makeBackgroundTask('shell-done', { status: 'completed' }), + makeBackgroundTask('shell-failed', { status: 'failed' }), + makeBackgroundTask('shell-stopped', { status: 'stopped' }), + // AutoDream is detached maintenance: it never counted as the session running. + makeBackgroundTask('dream-running', { taskType: 'dream' }), + // A teammate runtime is a container, not an activity row the user can stop. + makeBackgroundTask('teammate-running', { taskType: 'in_process_teammate' }), + ]), + }, + } as Partial>) + + await act(async () => { + render() + }) + + fireEvent.click(screen.getByLabelText('Close Mixed Session')) + fireEvent.click(screen.getByText('Stop & Close')) + + expect(traffic).toEqual([ + 'send tab-mixed stop_generation', + 'send tab-mixed stop_background_task shell-running', + 'disconnect tab-mixed', + ]) + }) + + it('does not stop a running Agent a second time after the session-level stop', async () => { + const { TabBar } = await import('./TabBar') + const { useTabStore } = await import('../../stores/tabStore') + const { useChatStore } = await import('../../stores/chatStore') + const traffic = await recordSocketTraffic() + + useTabStore.setState({ + tabs: [{ sessionId: 'tab-agent', title: 'Agent Session', type: 'session', status: 'idle' }], + activeTabId: 'tab-agent', + }) + useChatStore.setState({ + sessions: { + 'tab-agent': makeSessionWithTasks('idle', [ + makeBackgroundTask('agent-1', { taskType: 'local_agent' }), + makeBackgroundTask('shell-1'), + ]), + }, + } as Partial>) + + await act(async () => { + render() + }) + + fireEvent.click(screen.getByLabelText('Close Agent Session')) + fireEvent.click(screen.getByText('Stop & Close')) + + // `stopGeneration` already marks the Agent as stopping; only the shell is left to stop. + expect(traffic).toEqual([ + 'send tab-agent stop_generation', + 'send tab-agent stop_background_task shell-1', + 'disconnect tab-agent', + ]) + }) + + it('sends no stop message when the user keeps the session running', async () => { + const { TabBar } = await import('./TabBar') + const { useTabStore } = await import('../../stores/tabStore') + const { useChatStore } = await import('../../stores/chatStore') + const traffic = await recordSocketTraffic() + + useTabStore.setState({ + tabs: [{ sessionId: 'tab-shell', title: 'Shell Session', type: 'session', status: 'idle' }], + activeTabId: 'tab-shell', + }) + useChatStore.setState({ + sessions: { + 'tab-shell': makeSessionWithTasks('idle', [makeBackgroundTask('shell-1')]), + }, + } as Partial>) + + await act(async () => { + render() + }) + + fireEvent.click(screen.getByLabelText('Close Shell Session')) + fireEvent.click(screen.getByText('Keep Running')) + + expect(traffic).toEqual([]) + expect(useTabStore.getState().tabs).toEqual([]) + expect(useChatStore.getState().sessions['tab-shell']?.backgroundAgentTasks?.['shell-1']?.status).toBe('running') + }) + + it('stops the background tasks of every running session when closing all tabs', async () => { + const { TabBar } = await import('./TabBar') + const { useTabStore } = await import('../../stores/tabStore') + const { useChatStore } = await import('../../stores/chatStore') + const traffic = await recordSocketTraffic() + + useTabStore.setState({ + tabs: [ + { sessionId: 'tab-a', title: 'Session A', type: 'session', status: 'idle' }, + { sessionId: 'tab-b', title: 'Session B', type: 'session', status: 'idle' }, + { sessionId: 'tab-idle', title: 'Idle Session', type: 'session', status: 'idle' }, + ], + activeTabId: 'tab-a', + }) + useChatStore.setState({ + sessions: { + 'tab-a': makeSessionWithTasks('idle', [makeBackgroundTask('a-1')]), + 'tab-b': makeSessionWithTasks('idle', [makeBackgroundTask('b-1'), makeBackgroundTask('b-2')]), + 'tab-idle': makeSessionWithTasks('idle', [makeBackgroundTask('old', { status: 'completed' })]), + }, + } as Partial>) + + await act(async () => { + render() + }) + + fireEvent.contextMenu(screen.getByText('Session A')) + fireEvent.click(screen.getByText('Close All')) + + expect(screen.getByRole('dialog', { name: 'Sessions Running' })).toBeInTheDocument() + expect(traffic).toEqual([]) + + fireEvent.click(screen.getByText('Stop All & Close')) + + expect(traffic).toEqual([ + 'send tab-a stop_generation', + 'send tab-a stop_background_task a-1', + 'disconnect tab-a', + 'send tab-b stop_generation', + 'send tab-b stop_background_task b-1', + 'send tab-b stop_background_task b-2', + 'disconnect tab-b', + 'disconnect tab-idle', + ]) + expect(useTabStore.getState().tabs).toEqual([]) + }) + + it('leaves the tabs that stay open alone when closing the others', async () => { + const { TabBar } = await import('./TabBar') + const { useTabStore } = await import('../../stores/tabStore') + const { useChatStore } = await import('../../stores/chatStore') + const traffic = await recordSocketTraffic() + + useTabStore.setState({ + tabs: [ + { sessionId: 'tab-other', title: 'Other Session', type: 'session', status: 'idle' }, + { sessionId: 'tab-kept', title: 'Kept Session', type: 'session', status: 'idle' }, + ], + activeTabId: 'tab-kept', + }) + useChatStore.setState({ + sessions: { + 'tab-other': makeSessionWithTasks('idle', [makeBackgroundTask('other-1')]), + 'tab-kept': makeSessionWithTasks('idle', [makeBackgroundTask('kept-1')]), + }, + } as Partial>) + + await act(async () => { + render() + }) + + fireEvent.contextMenu(screen.getByText('Kept Session')) + fireEvent.click(screen.getByText('Close Others')) + fireEvent.click(screen.getByText('Stop & Close')) + + expect(traffic).toEqual([ + 'send tab-other stop_generation', + 'send tab-other stop_background_task other-1', + 'disconnect tab-other', + ]) + expect(useTabStore.getState().tabs.map((tab) => tab.sessionId)).toEqual(['tab-kept']) + expect(useChatStore.getState().sessions['tab-kept']?.backgroundAgentTasks?.['kept-1']?.status).toBe('running') + }) + }) + it('shows a running marker on tabs from tab status, live chat state, or background tasks', async () => { const { TabBar } = await import('./TabBar') const { useTabStore } = await import('../../stores/tabStore') @@ -2926,4 +3196,506 @@ describe('TabBar', () => { expect(screen.getAllByLabelText('Session running')).toHaveLength(3) expect(screen.getByText('Idle').closest('[data-dragging]')?.querySelector('[aria-label="Session running"]')).toBeNull() }) + + describe('a session waiting on the user', () => { + // A tool approval, an AskUserQuestion card and an ExitPlanMode review all + // arrive as `permission_request`; only the tool name differs. + const request = (requestId: string, toolName = 'Bash'): ServerMessage => ({ + type: 'permission_request', + requestId, + toolName, + toolUseId: `tu-${requestId}`, + input: {}, + }) + const resolved = (requestId: string, allowed = true): ServerMessage => ({ + type: 'permission_resolved', + requestId, + permissionType: 'tool', + allowed, + }) + // The store keeps the newest request in `pendingPermission` as a + // compatibility mirror and every outstanding one in `pendingPermissions`. + const waiting = (...requestIds: string[]): Partial => { + const requests = requestIds.map((requestId) => ({ + requestId, + toolName: 'Bash', + toolUseId: `tu-${requestId}`, + input: {}, + })) + return { + pendingPermission: requests.at(-1) ?? null, + pendingPermissions: Object.fromEntries(requests.map((request) => [request.requestId, request])), + } + } + const tabOf = (title: string) => screen.getByText(title).closest('[data-dragging]')! + const attentionMark = (title: string) => + within(tabOf(title)).queryByRole('img', { name: 'Waiting for your approval' }) + const runningMark = (title: string) => within(tabOf(title)).queryByLabelText('Session running') + + async function renderStrip( + titles: Record, + activeTabId: string, + sessions: Record = {}, + ) { + const { TabBar } = await import('./TabBar') + const { useTabStore } = await import('../../stores/tabStore') + const { useChatStore } = await import('../../stores/chatStore') + + useTabStore.setState({ + tabs: Object.entries(titles).map(([sessionId, title]) => ({ + sessionId, + title, + type: 'session' as const, + status: 'idle' as const, + })), + activeTabId, + }) + useChatStore.setState({ + sessions: Object.fromEntries(Object.keys(titles).map((id) => [id, sessions[id] ?? makeChatSession('idle')])), + disconnectSession: vi.fn(), + } as Partial>) + + await act(async () => { + render() + }) + + return { + useChatStore, + useTabStore, + // The real reducer, driven the way the socket drives it. + receive: (sessionId: string, message: ServerMessage) => act(() => { + useChatStore.getState().handleServerMessage(sessionId, message) + }), + } + } + + it('marks the tab while a request is open and gives it back to the running marker once answered', async () => { + const { receive } = await renderStrip({ waiting: 'Waiting', quiet: 'Quiet' }, 'quiet') + + await receive('waiting', request('r1')) + + expect(tabOf('Waiting')).toHaveAttribute('data-attention', 'true') + expect(attentionMark('Waiting')).toBeInTheDocument() + // Parked on a card is "running" by chatState as well. The brand dot said + // the session was working and could be left alone, which is exactly what + // was not true of it, so it must not be drawn beside the mark. + expect(runningMark('Waiting')).toBeNull() + expect(tabOf('Quiet')).toHaveAttribute('data-attention', 'false') + expect(attentionMark('Quiet')).toBeNull() + + await receive('waiting', resolved('r1')) + + expect(tabOf('Waiting')).toHaveAttribute('data-attention', 'false') + expect(attentionMark('Waiting')).toBeNull() + // An allowed tool runs on, so the tab is back to reporting that. + expect(runningMark('Waiting')).toBeInTheDocument() + }) + + it('stays lit when a status message overwrites chatState under a card that is still open', async () => { + const { receive, useChatStore } = await renderStrip({ waiting: 'Waiting' }, 'waiting') + + await receive('waiting', request('r1')) + expect(useChatStore.getState().sessions.waiting?.chatState).toBe('permission_pending') + await receive('waiting', { type: 'status', state: 'tool_executing' }) + + // The precondition that makes this a regression test: a rule that read + // chatState would see "not waiting" from here on, under a live card. + expect(useChatStore.getState().sessions.waiting?.chatState).toBe('tool_executing') + expect(tabOf('Waiting')).toHaveAttribute('data-attention', 'true') + expect(attentionMark('Waiting')).toBeInTheDocument() + }) + + it('does not light for chatState alone, because nothing renders a card for it', async () => { + await renderStrip({ empty: 'Empty' }, 'empty', { empty: makeChatSession('permission_pending') }) + + expect(tabOf('Empty')).toHaveAttribute('data-attention', 'false') + expect(attentionMark('Empty')).toBeNull() + // It is still not idle, so the tab keeps saying it is running. + expect(runningMark('Empty')).toBeInTheDocument() + }) + + it.each(['Bash', 'AskUserQuestion', 'ExitPlanMode'])('lights for a %s request', async (toolName) => { + const { receive } = await renderStrip({ waiting: 'Waiting' }, 'waiting') + + await receive('waiting', request('r1', toolName)) + + expect(attentionMark('Waiting')).toBeInTheDocument() + }) + + it('does not light for a Computer Use request, which has no card to answer', async () => { + const computerUse = { requestId: 'cu-1', request: {} as never } + await renderStrip({ cu: 'Computer use' }, 'cu', { + cu: makeChatSession('idle', { + pendingComputerUsePermission: computerUse, + pendingComputerUsePermissions: { 'cu-1': computerUse }, + }), + }) + + expect(tabOf('Computer use')).toHaveAttribute('data-attention', 'false') + expect(attentionMark('Computer use')).toBeNull() + }) + + it('outranks the running marker for a session that is also mid-turn', async () => { + await renderStrip({ busy: 'Busy' }, 'busy', { busy: makeChatSession('thinking', waiting('r1')) }) + + expect(attentionMark('Busy')).toBeInTheDocument() + expect(runningMark('Busy')).toBeNull() + }) + + it('still shows the error dot for a failed tab that is neither running nor waiting', async () => { + const { useTabStore } = await renderStrip({ failed: 'Failed', quiet: 'Quiet' }, 'quiet') + + act(() => { + useTabStore.getState().updateTabStatus('failed', 'error') + }) + + // The dot is decorative and has no name, so it is found by its danger colour. + expect(tabOf('Failed').querySelector('[class*="--color-error"]')).toBeInTheDocument() + expect(attentionMark('Failed')).toBeNull() + expect(runningMark('Failed')).toBeNull() + expect(tabOf('Quiet').querySelector('[class*="--color-error"]')).toBeNull() + }) + + it('ranks waiting above an earlier failure: the failed turn is over, the card is not', async () => { + const { receive, useTabStore } = await renderStrip({ failed: 'Failed', quiet: 'Quiet' }, 'quiet') + act(() => { + useTabStore.getState().updateTabStatus('failed', 'error') + }) + + await receive('failed', request('rf')) + + expect(attentionMark('Failed')).toBeInTheDocument() + expect(tabOf('Failed').querySelector('[class*="--color-error"]')).toBeNull() + + // And it goes once the card is answered. + await receive('failed', resolved('rf', false)) + expect(attentionMark('Failed')).toBeNull() + }) + + it('lights a background tab and the active tab alike, and only the ones that are waiting', async () => { + const { receive } = await renderStrip({ a: 'Alpha', b: 'Bravo', c: 'Charlie' }, 'a') + + await receive('a', request('ra')) + await receive('c', request('rc')) + + expect(tabOf('Alpha')).toHaveAttribute('data-attention', 'true') + expect(tabOf('Bravo')).toHaveAttribute('data-attention', 'false') + expect(tabOf('Charlie')).toHaveAttribute('data-attention', 'true') + expect(screen.getAllByRole('img', { name: 'Waiting for your approval' })).toHaveLength(2) + }) + + it('stays lit until the last of several requests is answered', async () => { + const { receive } = await renderStrip({ waiting: 'Waiting' }, 'waiting') + + await receive('waiting', request('r1')) + await receive('waiting', request('r2', 'AskUserQuestion')) + await receive('waiting', resolved('r1')) + + expect(attentionMark('Waiting')).toBeInTheDocument() + + await receive('waiting', resolved('r2')) + + expect(attentionMark('Waiting')).toBeNull() + }) + + it('swaps the mark into the same 14px slot the running dot uses, so the title does not move', async () => { + const { receive } = await renderStrip({ waiting: 'Waiting' }, 'waiting') + const slot = () => tabOf('Waiting').firstElementChild as HTMLElement + + await receive('waiting', { type: 'status', state: 'thinking' }) + expect(runningMark('Waiting')).toBeInTheDocument() + expect(slot()).toHaveClass('w-[14px]', 'mr-1.5') + + await receive('waiting', request('r1')) + expect(attentionMark('Waiting')).toBeInTheDocument() + expect(slot()).toHaveClass('w-[14px]', 'mr-1.5') + }) + + it('does not rerender when requests change without changing which tabs are waiting', async () => { + const { TabBar } = await import('./TabBar') + const { useTabStore } = await import('../../stores/tabStore') + const { useChatStore } = await import('../../stores/chatStore') + const { useSessionStore } = await import('../../stores/sessionStore') + + useTabStore.setState({ + tabs: [{ sessionId: 'tab-1', title: 'Workspace Session', type: 'session', status: 'idle' }], + activeTabId: 'tab-1', + }) + useChatStore.setState({ + sessions: { + 'tab-1': makeChatSession('idle', waiting('r1')), + ghost: makeChatSession('idle'), + }, + disconnectSession: vi.fn(), + } as Partial>) + useSessionStore.setState({ + sessions: [{ + id: 'tab-1', + title: 'Workspace Session', + createdAt: '2026-05-13T00:00:00.000Z', + modifiedAt: '2026-05-13T00:00:00.000Z', + messageCount: 0, + projectPath: '/repo', + workDir: '/repo/worktree', + workDirExists: true, + }], + activeSessionId: 'tab-1', + }) + + await act(async () => { + render() + }) + expect(openProjectMenuMock.paths[openProjectMenuMock.paths.length - 1]).toBe('/repo/worktree') + + openProjectMenuMock.paths = [] + await act(async () => { + useChatStore.setState((state) => ({ + sessions: { + ...state.sessions, + // Same request, fresh objects: what a replayed permission_request does. + 'tab-1': { ...state.sessions['tab-1']!, ...waiting('r1') }, + // A session with no tab is not this strip's business. + ghost: { ...state.sessions.ghost!, ...waiting('r2') }, + }, + })) + }) + expect(openProjectMenuMock.paths).toEqual([]) + + // Positive control: a change that does alter the set does rerender. + await act(async () => { + useChatStore.getState().handleServerMessage('tab-1', resolved('r1')) + }) + expect(openProjectMenuMock.paths.length).toBeGreaterThan(0) + }) + + describe('scrolled out of view', () => { + // The strip is 840 wide and scrolled to 300, so both chevrons are in. Each + // case moves only the tabs' own rects against that. + async function renderScrolledStrip() { + const view = await renderStrip({ a: 'Alpha', b: 'Bravo', c: 'Charlie', d: 'Delta' }, 'b') + const strip = screen.getByTestId('tab-bar-scroll-region') + stubRect(strip, 0, 840) + Object.defineProperty(strip, 'clientWidth', { configurable: true, get: () => 840 }) + Object.defineProperty(strip, 'scrollWidth', { configurable: true, get: () => 1600 }) + Object.defineProperty(strip, 'scrollLeft', { configurable: true, get: () => 300 }) + Object.defineProperty(strip, 'scrollBy', { configurable: true, value: vi.fn() }) + + const place = (title: string, left: number, right: number) => stubRect(tabOf(title), left, right) + place('Alpha', -400, -260) + place('Bravo', 20, 160) + place('Charlie', 600, 740) + place('Delta', 900, 1040) + const scrolled = () => act(() => { fireEvent.scroll(strip) }) + scrolled() + + return { ...view, place, scrolled } + } + const hintOn = (side: 'left' | 'right') => screen.queryByTestId(`tab-strip-attention-${side}`) + + it('hints on the right chevron when a waiting tab is cut off past the right edge', async () => { + const { receive } = await renderScrolledStrip() + + await receive('d', request('rd')) + + expect(hintOn('right')).toBeInTheDocument() + expect(hintOn('left')).not.toBeInTheDocument() + }) + + it('hints on the left chevron when a waiting tab is cut off past the left edge', async () => { + const { receive } = await renderScrolledStrip() + + await receive('a', request('ra')) + + expect(hintOn('left')).toBeInTheDocument() + expect(hintOn('right')).not.toBeInTheDocument() + }) + + it('hints on both sides when waiting tabs are cut off on both', async () => { + const { receive } = await renderScrolledStrip() + + await receive('a', request('ra')) + await receive('d', request('rd')) + + expect(hintOn('left')).toBeInTheDocument() + expect(hintOn('right')).toBeInTheDocument() + }) + + it('describes the hint to a screen reader without renaming the chevron', async () => { + const { receive } = await renderScrolledStrip() + + await receive('d', request('rd')) + + const chevron = screen.getByRole('button', { name: 'Scroll tabs right' }) + const describedBy = chevron.getAttribute('aria-describedby') + expect(describedBy).toBeTruthy() + expect(document.getElementById(describedBy!)).toHaveTextContent('Waiting for your approval') + expect(chevron).toHaveAttribute('title', 'Waiting for your approval') + // The other chevron has nothing to say and says nothing. + expect(screen.getByRole('button', { name: 'Scroll tabs left' })).not.toHaveAttribute('aria-describedby') + }) + + it('lets go of the hint once the waiting tab is scrolled fully into view', async () => { + const { receive, place, scrolled } = await renderScrolledStrip() + await receive('d', request('rd')) + expect(hintOn('right')).toBeInTheDocument() + + place('Delta', 500, 640) + scrolled() + + expect(hintOn('right')).not.toBeInTheDocument() + }) + + it('says nothing for a tab that is cut off but not waiting', async () => { + const { receive } = await renderScrolledStrip() + + // Someone else is waiting, and in view. Alpha and Delta are out of view + // but idle: being out of view is not what the hint is about. + await receive('c', request('rc')) + + expect(hintOn('left')).not.toBeInTheDocument() + expect(hintOn('right')).not.toBeInTheDocument() + }) + + it('says nothing for a waiting tab that is whole, because its own mark is on screen', async () => { + const { receive } = await renderScrolledStrip() + + await receive('b', request('rb')) + await receive('c', request('rc')) + + expect(hintOn('left')).not.toBeInTheDocument() + expect(hintOn('right')).not.toBeInTheDocument() + }) + + it('gives a subpixel of slack, and no more than that', async () => { + const { receive, place, scrolled } = await renderScrolledStrip() + await receive('d', request('rd')) + + // One pixel past the strip's 840 edge is layout rounding, not a clip. + place('Delta', 700, 841) + scrolled() + expect(hintOn('right')).not.toBeInTheDocument() + + place('Delta', 700, 842) + scrolled() + expect(hintOn('right')).toBeInTheDocument() + }) + + it('follows a request that arrives and is answered while the strip stands still', async () => { + const { receive } = await renderScrolledStrip() + // No scroll and no resize from here on: only the store moves. + expect(hintOn('right')).not.toBeInTheDocument() + + await receive('d', request('rd')) + expect(hintOn('right')).toBeInTheDocument() + + await receive('d', resolved('rd')) + expect(hintOn('right')).not.toBeInTheDocument() + }) + }) + + describe('jumping to the next one', () => { + const jump = () => screen.queryByTestId('tab-attention-jump') + + it('is not offered while nothing is waiting', async () => { + await renderStrip({ a: 'Alpha', b: 'Bravo' }, 'a') + + expect(jump()).not.toBeInTheDocument() + }) + + it('is not offered when the only waiting session is the one on screen', async () => { + const { receive } = await renderStrip({ a: 'Alpha', b: 'Bravo' }, 'a') + + await receive('a', request('ra')) + + // Its own mark is lit, but there is nowhere else to go. + expect(attentionMark('Alpha')).toBeInTheDocument() + expect(jump()).not.toBeInTheDocument() + }) + + it('counts the other waiting sessions and sits with the toolbar buttons', async () => { + const { receive } = await renderStrip({ a: 'Alpha', b: 'Bravo', c: 'Charlie', d: 'Delta' }, 'a') + + await receive('a', request('ra')) + await receive('c', request('rc')) + await receive('d', request('rd')) + + // Alpha is on screen, so it is not somewhere to jump to. + const button = within(screen.getByTestId('workspace-window-header')).getByTestId('tab-attention-jump') + expect(button).toHaveTextContent('2') + expect(button).toHaveAccessibleName('Jump to the next waiting session (2 waiting)') + }) + + it('takes you to the next waiting tab after the active one, in strip order', async () => { + const { receive, useTabStore } = await renderStrip({ a: 'Alpha', b: 'Bravo', c: 'Charlie', d: 'Delta' }, 'a') + await receive('c', request('rc')) + await receive('d', request('rd')) + + fireEvent.click(jump()!) + expect(useTabStore.getState().activeTabId).toBe('c') + + // From Charlie the only other waiting tab is Delta, and then back again. + fireEvent.click(jump()!) + expect(useTabStore.getState().activeTabId).toBe('d') + fireEvent.click(jump()!) + expect(useTabStore.getState().activeTabId).toBe('c') + }) + + it('wraps past the end of the strip', async () => { + const { receive, useTabStore } = await renderStrip({ a: 'Alpha', b: 'Bravo', c: 'Charlie', d: 'Delta' }, 'd') + await receive('a', request('ra')) + await receive('c', request('rc')) + + fireEvent.click(jump()!) + + expect(useTabStore.getState().activeTabId).toBe('a') + }) + + it('brings the tab it lands on into view', async () => { + const { receive } = await renderStrip({ a: 'Alpha', b: 'Bravo', c: 'Charlie' }, 'a') + await receive('c', request('rc')) + scrollIntoViewMock.mockClear() + + fireEvent.click(jump()!) + + expect(scrollIntoViewMock).toHaveBeenCalled() + }) + + it('counts down as requests are answered and leaves with the last one', async () => { + const { receive } = await renderStrip({ a: 'Alpha', b: 'Bravo', c: 'Charlie' }, 'a') + await receive('b', request('rb')) + await receive('c', request('rc')) + expect(jump()).toHaveTextContent('2') + + await receive('b', resolved('rb')) + expect(jump()).toHaveTextContent('1') + + await receive('c', resolved('rc')) + expect(jump()).not.toBeInTheDocument() + }) + + it('is offered from a tab that is not a session, such as Settings', async () => { + const { TabBar } = await import('./TabBar') + const { useTabStore } = await import('../../stores/tabStore') + const { useChatStore } = await import('../../stores/chatStore') + useTabStore.setState({ + tabs: [ + { sessionId: '__settings__', title: 'Settings', type: 'settings', status: 'idle' }, + { sessionId: 'a', title: 'Alpha', type: 'session', status: 'idle' }, + ], + activeTabId: '__settings__', + }) + useChatStore.setState({ + sessions: { a: makeChatSession('idle', waiting('ra')) }, + disconnectSession: vi.fn(), + } as Partial>) + await act(async () => { + render() + }) + + fireEvent.click(jump()!) + + expect(useTabStore.getState().activeTabId).toBe('a') + }) + }) + }) }) diff --git a/desktop/src/components/layout/TabBar.tsx b/desktop/src/components/layout/TabBar.tsx index 50a1eefb..50fc65c9 100644 --- a/desktop/src/components/layout/TabBar.tsx +++ b/desktop/src/components/layout/TabBar.tsx @@ -1,4 +1,4 @@ -import { forwardRef, useMemo, useRef, useState, useEffect, useCallback } from 'react' +import { forwardRef, useMemo, useRef, useState, useEffect, useLayoutEffect, useCallback, useId } from 'react' import { useShallow } from 'zustand/react/shallow' import { SCHEDULED_TAB_ID, @@ -28,7 +28,10 @@ import { IconButton } from '@/components/ui/IconButton' import { useDismissable } from '@/hooks/useDismissable' import { useTranslation } from '../../i18n' import { getDesktopHost } from '../../lib/desktopHost' -import { hasRunningBackgroundTasks } from '../../lib/backgroundTasks' +import { hasRunningBackgroundTasks, listRunningBackgroundTasks } from '../../lib/backgroundTasks' +import { collectAttentionIds, nextAttentionSessionId } from '../../lib/sessionAttention' +import { SessionAttentionMark } from './SessionAttentionMark' +import { TabAttentionJump } from './TabAttentionJump' import { WindowControls, showWindowControls } from './WindowControls' import { OpenProjectMenu } from './OpenProjectMenu' import { SquareTerminal } from 'lucide-react' @@ -58,6 +61,26 @@ const REVEAL_ACTIVE_TAB: ScrollIntoViewOptions = { // whose edges land on fractional pixels reports the tab as clipped on every // single resize and re-scrolls forever. const TAB_VISIBILITY_TOLERANCE = 1 + +type ClippedSide = 'left' | 'right' + +/** + * Which edge of the strip a tab is cut off by, or null when it is whole. This + * is the one definition of "whole": re-revealing the active tab and hinting at + * waiting tabs the strip has scrolled away both ask it, so the two cannot + * disagree about what is out of view. A tab that is only partly clipped counts — + * its glyph sits at the left edge, and a hint that sometimes stays quiet for a + * tab the person cannot fully see is worse than one that speaks a little early. + */ +function clippedSide( + strip: Pick, + tab: Pick, +): ClippedSide | null { + if (tab.left < strip.left - TAB_VISIBILITY_TOLERANCE) return 'left' + if (tab.right > strip.right + TAB_VISIBILITY_TOLERANCE) return 'right' + return null +} + // One glyph per *non-chat* tab kind: the glyph says "this tab is not a // conversation". Chat tabs deliberately have none — a bubble on every tab in a // strip that is mostly chats is pure noise, and the slot it occupied is worth @@ -125,6 +148,15 @@ export function TabBar() { (sessionState.chatState !== 'idle' || hasRunningBackgroundTasks(sessionState.backgroundAgentTasks)) }) )) + // Tabs that are stopped on a decision only the user can make. Read from the + // outstanding requests rather than from `chatState` (see + // `sessionNeedsAttention`), and `useShallow` for the same reason as above: a + // fresh array from every store update would loop the subscription. + const attentionList = useChatStore(useShallow((s) => collectAttentionIds(s.sessions, sessionTabIds))) + const attentionSet = useMemo(() => new Set(attentionList), [attentionList]) + // The waiting tabs other than the one on screen: the places a jump can take + // you, and so the number on the button that offers it. + const otherAttentionCount = attentionList.filter((sessionId) => sessionId !== activeTabId).length const disconnectSession = useChatStore((s) => s.disconnectSession) const activeTab = tabs.find((tab) => tab.sessionId === activeTabId) ?? null const isActiveSessionTab = isSessionTab(activeTab) || isSessionTabId(activeTabId) @@ -203,6 +235,14 @@ export function TabBar() { const userScrolledRef = useRef(false) const [canScrollLeft, setCanScrollLeft] = useState(false) const [canScrollRight, setCanScrollRight] = useState(false) + // Whether a waiting tab is scrolled out of view on each side. The mark on the + // tab itself is no use to someone who cannot see the tab, and "tab 多了" is + // exactly when that happens. + const [attentionOffscreen, setAttentionOffscreen] = useState({ left: false, right: false }) + // `updateScrollState` is a stable callback with no dependencies, so it reads + // the waiting tabs through a ref that the layout effect below keeps current. + const attentionListRef = useRef([]) + const attentionHintId = useId() const [tabHitWidth, setTabHitWidth] = useState(0) const [contextMenu, setContextMenu] = useState<{ sessionId: string; x: number; y: number } | null>(null) const [pendingCloseRequest, setPendingCloseRequest] = useState(null) @@ -239,8 +279,33 @@ export function TabBar() { const last = el.lastElementChild as HTMLElement | null const contentWidth = first && last ? last.offsetLeft + last.offsetWidth - first.offsetLeft : 0 setTabHitWidth(Math.max(0, Math.min(el.clientWidth, contentWidth - el.scrollLeft))) + + // Only the waiting tabs are measured, so a scroll costs a few rect reads + // rather than one per tab. State is left alone when nothing changed: this + // runs on every scroll event, and a fresh object each time would rerender + // the strip for nothing. + const strip = el.getBoundingClientRect() + let left = false + let right = false + for (const sessionId of attentionListRef.current) { + const tabEl = tabRefs.current.get(sessionId) + if (!tabEl) continue + const side = clippedSide(strip, tabEl.getBoundingClientRect()) + if (side === 'left') left = true + else if (side === 'right') right = true + } + setAttentionOffscreen((prev) => (prev.left === left && prev.right === right ? prev : { left, right })) }, []) + // A request can arrive, or be answered, with the strip standing still — no + // scroll, no resize — so nothing in the observer path would notice. Measuring + // in a layout effect keeps the chevron from showing the previous answer for + // a frame, and the tabs are in the DOM by then, so their rects are current. + useLayoutEffect(() => { + attentionListRef.current = attentionList + updateScrollState() + }, [attentionList, tabs, updateScrollState]) + // Keeping the active tab whole is an invariant the strip has to re-establish // after its own width changes, not something a single scroll on activation // can settle. The chevrons are why: they are `w-7` siblings of this region, @@ -276,10 +341,7 @@ export function TabBar() { // Already whole. The tolerance is for subpixel layout, which would // otherwise report a clip on every resize and scroll forever. - if ( - tab.left >= strip.left - TAB_VISIBILITY_TOLERANCE && - tab.right <= strip.right + TAB_VISIBILITY_TOLERANCE - ) return + if (!clippedSide(strip, tab)) return activeTabEl.scrollIntoView(REVEAL_ACTIVE_TAB) }, []) @@ -371,7 +433,17 @@ export function TabBar() { if (isSessionTab(tab)) { const isRunning = runningSessionSet.has(tab.sessionId) if (isRunning && stopRunning) { - useChatStore.getState().stopGeneration(tab.sessionId) + const chat = useChatStore.getState() + const runningTasks = listRunningBackgroundTasks(chat.sessions[tab.sessionId]?.backgroundAgentTasks) + chat.stopGeneration(tab.sessionId) + // The dialog counts background tasks as the session running, but + // stopGeneration only reaches the foreground turn and Agent tasks (and + // marks the latter as stopping, which stopBackgroundTask skips). A shell + // command would outlive "Stop & Close" and keep the reopened session + // running. Both must go out before disconnectSession closes the socket. + for (const task of runningTasks) { + chat.stopBackgroundTask(tab.sessionId, task.taskId) + } } if (!isRunning || stopRunning) { // Auto-delete only when both server metadata and the loaded transcript @@ -543,9 +615,34 @@ export function TabBar() { setActiveTab(sessionId) } + // Activation does the rest: the effect on `activeTabId` above scrolls the tab + // into view and hands the strip's position back to it. + const jumpToAttention = () => { + const next = nextAttentionSessionId(sessionTabIds, attentionSet, activeTabId) + if (next) setActiveTab(next) + } + + // The chevron on the side a waiting tab has scrolled off gets a dot in its + // corner. Static on purpose: the pulse belongs to the mark on the tab, and + // several things pulsing out of phase across one strip read as noise. The + // sr-only text is the description, not the name — `aria-label` keeps naming + // the button by what it does. + const attentionHint = (side: ClippedSide) => attentionOffscreen[side] ? ( + <> + + {t('sidebar.sessionNeedsAttention')} + + ) : null + const rightScrollControl = canScrollRight && ( - ) @@ -574,8 +671,9 @@ export function TabBar() {
{canScrollLeft && ( - )} @@ -605,11 +703,13 @@ export function TabBar() { displayTitle={displayTitle} closeLabel={t('tabs.closeTab', { title: displayTitle })} isRunning={runningSessionIds.has(tab.sessionId)} + needsAttention={attentionSet.has(tab.sessionId)} isActive={tab.sessionId === activeTabId} isDragOver={dragOverIndex === index} isDragging={tab.sessionId === draggingSessionId} dragOffsetX={tab.sessionId === draggingSessionId ? dragOffsetX : 0} runningLabel={t('tabs.sessionRunning')} + attentionLabel={t('sidebar.sessionNeedsAttention')} onClick={() => handleTabClick(tab.sessionId)} onClose={() => handleClose(tab.sessionId)} onContextMenu={(e) => handleContextMenu(e, tab.sessionId)} @@ -672,6 +772,13 @@ export function TabBar() { className="flex h-[52px] min-w-0 flex-1" /> ) : null} + {otherAttentionCount > 0 && ( + + )} {showActivityButton && activeTabId && ( )} @@ -800,24 +907,33 @@ const TabItem = forwardRef void onClose: () => void onContextMenu: (e: React.MouseEvent) => void onMouseDown: (event: React.MouseEvent) => void -}>(({ tab, displayTitle, closeLabel, isRunning, isActive, isDragOver, isDragging, dragOffsetX, runningLabel, onClick, onClose, onContextMenu, onMouseDown }, ref) => { +}>(({ tab, displayTitle, closeLabel, isRunning, needsAttention, isActive, isDragOver, isDragging, dragOffsetX, runningLabel, attentionLabel, onClick, onClose, onContextMenu, onMouseDown }, ref) => { // Chat tabs carry no glyph at all; the dot only appears when there is // something to say. Everything else identifies its section with one. + // + // Waiting on the user outranks running. A session parked on a permission card + // is also "running" by `chatState`, so before this the brand dot told the + // person the one thing that was not true of it: that it was working and + // could be left alone. const leadingGlyph = isSessionTab(tab) - ? (isRunning - ? - : tab.status === 'error' - ? - : null) + ? (needsAttention + ? + : isRunning + ? + : tab.status === 'error' + ? + : null) : ( {TAB_TYPE_ICON[tab.type] ?? TAB_TYPE_ICON_FALLBACK} @@ -829,6 +945,7 @@ const TabItem = forwardRef { }) }) + it('offers Sonnet 5.5 with the Claude Code default effort and context for OAuth selection', () => { + expect(OFFICIAL_MODELS.find(model => model.id === 'claude-sonnet-5-5')).toMatchObject({ + name: 'Sonnet 5.5', + context: '1m', + defaultReasoningEffort: 'medium', + supportedReasoningEfforts: ['low', 'medium', 'high', 'xhigh', 'max'], + }) + }) + + it('lists Sonnet 5.5 ahead of the Sonnet 5 it supersedes', () => { + const ids = OFFICIAL_MODELS.map(model => model.id) + expect(ids.indexOf('claude-sonnet-5-5')).toBeGreaterThanOrEqual(0) + expect(ids.indexOf('claude-sonnet-5-5')).toBeLessThan(ids.indexOf('claude-sonnet-5')) + }) + it('keeps older explicit model selections available', () => { expect(OFFICIAL_MODELS.map(model => model.id)).toEqual(expect.arrayContaining([ 'claude-opus-5', 'claude-opus-4-8', 'claude-fable-5-1', 'claude-sonnet-5', diff --git a/desktop/src/constants/modelCatalog.ts b/desktop/src/constants/modelCatalog.ts index 49339f52..d39cffdc 100644 --- a/desktop/src/constants/modelCatalog.ts +++ b/desktop/src/constants/modelCatalog.ts @@ -41,6 +41,14 @@ export const OFFICIAL_MODELS: ModelInfo[] = [ defaultReasoningEffort: 'high', supportedReasoningEfforts: ['low', 'medium', 'high', 'xhigh', 'max'], }, + { + id: 'claude-sonnet-5-5', + name: 'Sonnet 5.5', + description: 'Best combination of speed and intelligence', + context: '1m', + defaultReasoningEffort: 'medium', + supportedReasoningEfforts: ['low', 'medium', 'high', 'xhigh', 'max'], + }, { id: 'claude-sonnet-5', name: 'Sonnet 5', diff --git a/desktop/src/i18n/locales/en.ts b/desktop/src/i18n/locales/en.ts index 4e731867..9d4e68c8 100644 --- a/desktop/src/i18n/locales/en.ts +++ b/desktop/src/i18n/locales/en.ts @@ -1349,6 +1349,7 @@ Row 9, all 8 cells: continuing from straight down, turning left through lower-le 'settings.diagnostics.doctorNoKeys': 'None', 'errorBoundary.title': 'Something went wrong.', 'errorBoundary.description': 'The error was recorded in Diagnostics.', + 'errorBoundary.item': "This item couldn't be displayed. The error was recorded in Diagnostics.", // Settings > Claude Official Login 'settings.claudeOfficialLogin.intro': 'Using official Claude models requires signing in to your Claude.ai account. Click the button below to open the official Claude login page in your browser; you\'ll be returned here after authorizing.', @@ -3651,7 +3652,7 @@ Row 9, all 8 cells: continuing from straight down, turning left through lower-le 'businessError.pdf_invalid': 'The PDF file is not valid. Convert it to text or send a different file.', 'businessError.image_too_large': 'The image is too large for the selected model. Resize it or send a smaller image.', 'businessError.image_unsupported': 'This model does not support images. Continue with text, or switch to a vision-capable model and send the image again.', - 'businessError.request_too_large': 'The request is too large for the selected model. Remove large files or retry with a smaller message.', + 'businessError.request_too_large': 'The request is too large for your provider or relay. Earlier images and documents are removed automatically on your next message; if it still fails, compact the conversation or start a new session.', 'businessError.prompt_too_long': 'The prompt is too long for the selected model. Compact the conversation or retry with less context.', 'businessError.auto_mode_unavailable': 'Auto mode is unavailable for your current plan.', @@ -3724,6 +3725,7 @@ Row 9, all 8 cells: continuing from straight down, turning left through lower-le 'tabs.hideBrowser': 'Hide Browser', 'tabs.scrollLeft': 'Scroll tabs left', 'tabs.scrollRight': 'Scroll tabs right', + 'tabs.jumpToAttention': 'Jump to the next waiting session ({count} waiting)', 'tabs.closeTab': 'Close {title}', 'tabs.untitled': 'Untitled', diff --git a/desktop/src/i18n/locales/jp.ts b/desktop/src/i18n/locales/jp.ts index c8a371f9..99f3d741 100644 --- a/desktop/src/i18n/locales/jp.ts +++ b/desktop/src/i18n/locales/jp.ts @@ -1350,6 +1350,7 @@ export const jp: Record = { 'settings.diagnostics.doctorNoKeys': 'なし', 'errorBoundary.title': '問題が発生しました。', 'errorBoundary.description': 'エラーは診断に記録されました。', + 'errorBoundary.item': 'この項目は表示できませんでした。エラーは診断に記録されました。', // Settings > Claude Official Login 'settings.claudeOfficialLogin.intro': '公式 Claude モデルを使用するには、Claude.ai アカウントにサインインする必要があります。下のボタンをクリックすると、ブラウザで公式 Claude ログインページが開きます。承認後、ここに戻ります。', @@ -3652,7 +3653,7 @@ export const jp: Record = { 'businessError.pdf_invalid': 'PDF ファイルが無効です。テキストに変換するか、別のファイルを送信してください。', 'businessError.image_too_large': '画像が選択したモデルには大きすぎます。サイズを変更するか、より小さい画像を送信してください。', 'businessError.image_unsupported': 'このモデルは画像をサポートしていません。テキストで続行するか、ビジョン対応モデルに切り替えて画像を再送信してください。', - 'businessError.request_too_large': 'リクエストが選択したモデルには大きすぎます。大きなファイルを削除するか、より短いメッセージで再試行してください。', + 'businessError.request_too_large': 'リクエストがプロバイダーまたは中継サービスの許容サイズを超えました。次のメッセージで以前の画像とドキュメントが自動的に削除されます。それでも失敗する場合は、会話を圧縮するか、新しいセッションを開始してください。', 'businessError.prompt_too_long': 'プロンプトが選択したモデルには長すぎます。会話を圧縮するか、コンテキストを減らして再試行してください。', 'businessError.auto_mode_unavailable': '自動モードは現在のプランでは利用できません。', @@ -3725,6 +3726,7 @@ export const jp: Record = { 'tabs.hideBrowser': 'ブラウザを非表示', 'tabs.scrollLeft': 'タブを左にスクロール', 'tabs.scrollRight': 'タブを右にスクロール', + 'tabs.jumpToAttention': '次の待機中のセッションへ移動({count} 件)', 'tabs.closeTab': '{title} を閉じる', 'tabs.untitled': '無題', diff --git a/desktop/src/i18n/locales/kr.ts b/desktop/src/i18n/locales/kr.ts index 8ef79095..9bf2bf4e 100644 --- a/desktop/src/i18n/locales/kr.ts +++ b/desktop/src/i18n/locales/kr.ts @@ -1352,6 +1352,7 @@ export const kr: Record = { 'settings.diagnostics.doctorNoKeys': '없음', 'errorBoundary.title': '문제가 발생했습니다.', 'errorBoundary.description': '오류가 진단에 기록되었습니다.', + 'errorBoundary.item': '이 항목을 표시할 수 없습니다. 오류가 진단에 기록되었습니다.', // Settings > Claude Official Login 'settings.claudeOfficialLogin.intro': '공식 Claude 모델을 사용하려면 Claude.ai 계정에 로그인해야 합니다. 아래 버튼을 클릭하면 브라우저에서 공식 Claude 로그인 페이지가 열립니다. 승인 후 이곳으로 돌아옵니다.', @@ -3654,7 +3655,7 @@ export const kr: Record = { 'businessError.pdf_invalid': 'PDF 파일이 유효하지 않습니다. 텍스트로 변환하거나 다른 파일을 보내세요.', 'businessError.image_too_large': '이미지가 선택한 모델에 비해 너무 큽니다. 크기를 조정하거나 더 작은 이미지를 보내세요.', 'businessError.image_unsupported': '이 모델은 이미지를 지원하지 않습니다. 텍스트로 계속하거나, 비전 지원 모델로 전환하여 이미지를 다시 보내세요.', - 'businessError.request_too_large': '요청이 선택한 모델에 비해 너무 큽니다. 큰 파일을 제거하거나 더 짧은 메시지로 다시 시도하세요.', + 'businessError.request_too_large': '요청이 공급자 또는 중계 서버가 허용하는 크기를 초과했습니다. 다음 메시지에서 이전 이미지와 문서가 자동으로 제거됩니다. 그래도 실패하면 대화를 압축하거나 새 세션을 시작하세요.', 'businessError.prompt_too_long': '프롬프트가 선택한 모델에 비해 너무 깁니다. 대화를 압축하거나 컨텍스트를 줄여 다시 시도하세요.', 'businessError.auto_mode_unavailable': '자동 모드는 현재 요금제에서 사용할 수 없습니다.', @@ -3727,6 +3728,7 @@ export const kr: Record = { 'tabs.hideBrowser': '브라우저 숨기기', 'tabs.scrollLeft': '탭 왼쪽으로 스크롤', 'tabs.scrollRight': '탭 오른쪽으로 스크롤', + 'tabs.jumpToAttention': '다음 대기 중인 세션으로 이동 ({count}개)', 'tabs.closeTab': '{title} 닫기', 'tabs.untitled': '제목 없음', diff --git a/desktop/src/i18n/locales/zh-TW.ts b/desktop/src/i18n/locales/zh-TW.ts index 0402f18e..49fc158c 100644 --- a/desktop/src/i18n/locales/zh-TW.ts +++ b/desktop/src/i18n/locales/zh-TW.ts @@ -1349,6 +1349,7 @@ export const zh: Record = { 'settings.diagnostics.doctorNoKeys': '無', 'errorBoundary.title': '出現異常。', 'errorBoundary.description': '錯誤已記錄到診斷日誌。', + 'errorBoundary.item': '此則內容無法顯示。錯誤已記錄到診斷日誌。', // Settings > Claude Official Login 'settings.claudeOfficialLogin.intro': '使用官方 Claude 模型需要登入你的 Claude.ai 賬號。點選下方按鈕,瀏覽器會開啟 Claude 官方登入頁面,授權後自動回到這裡。', @@ -3651,7 +3652,7 @@ export const zh: Record = { 'businessError.pdf_invalid': '這個 PDF 檔案無效。請先轉成文字,或換一個檔案傳送。', 'businessError.image_too_large': '這張圖片超出了當前模型可處理的大小。請壓縮圖片,或換一張更小的圖片。', 'businessError.image_unsupported': '當前模型不支援圖片。請繼續使用文字,或切換到支援視覺的模型後重新傳送圖片。', - 'businessError.request_too_large': '這次請求超出了當前模型可處理的大小。請移除大檔案,或縮短訊息後重試。', + 'businessError.request_too_large': '這次請求超出了服務商或中轉站允許的大小。下一則訊息會自動移除較早的圖片和文件;如果仍然失敗,請壓縮會話或新開會話。', 'businessError.prompt_too_long': '當前上下文超出了模型限制。請先壓縮會話,或減少上下文後重試。', 'businessError.auto_mode_unavailable': '當前套餐不支援自動模式。', @@ -3724,6 +3725,7 @@ export const zh: Record = { 'tabs.hideBrowser': '隱藏瀏覽器', 'tabs.scrollLeft': '分頁向左捲動', 'tabs.scrollRight': '分頁向右捲動', + 'tabs.jumpToAttention': '跳轉到下一個待處理會話(共 {count} 個)', 'tabs.closeTab': '關閉 {title}', 'tabs.untitled': '未命名', diff --git a/desktop/src/i18n/locales/zh.ts b/desktop/src/i18n/locales/zh.ts index efc1373d..ecdd8962 100644 --- a/desktop/src/i18n/locales/zh.ts +++ b/desktop/src/i18n/locales/zh.ts @@ -1348,6 +1348,7 @@ export const zh: Record = { 'settings.diagnostics.doctorNoKeys': '无', 'errorBoundary.title': '出现异常。', 'errorBoundary.description': '错误已记录到诊断日志。', + 'errorBoundary.item': '此条内容无法显示。错误已记录到诊断日志。', // Settings > Claude Official Login 'settings.claudeOfficialLogin.intro': '使用官方 Claude 模型需要登录你的 Claude.ai 账号。点击下方按钮,浏览器会打开 Claude 官方登录页面,授权后自动回到这里。', @@ -3650,7 +3651,7 @@ export const zh: Record = { 'businessError.pdf_invalid': '这个 PDF 文件无效。请先转成文本,或换一个文件发送。', 'businessError.image_too_large': '这张图片超出了当前模型可处理的大小。请压缩图片,或换一张更小的图片。', 'businessError.image_unsupported': '当前模型不支持图片。请继续使用文字,或切换到支持视觉的模型后重新发送图片。', - 'businessError.request_too_large': '这次请求超出了当前模型可处理的大小。请移除大文件,或缩短消息后重试。', + 'businessError.request_too_large': '这次请求超出了服务商或中转站允许的大小。下一条消息会自动移除较早的图片和文档;如果仍然失败,请压缩会话或新开会话。', 'businessError.prompt_too_long': '当前上下文超出了模型限制。请先压缩会话,或减少上下文后重试。', 'businessError.auto_mode_unavailable': '当前套餐不支持自动模式。', @@ -3723,6 +3724,7 @@ export const zh: Record = { 'tabs.hideBrowser': '隐藏浏览器', 'tabs.scrollLeft': '标签页向左滚动', 'tabs.scrollRight': '标签页向右滚动', + 'tabs.jumpToAttention': '跳转到下一个待处理会话(共 {count} 个)', 'tabs.closeTab': '关闭 {title}', 'tabs.untitled': '未命名', diff --git a/desktop/src/lib/backgroundTasks.test.ts b/desktop/src/lib/backgroundTasks.test.ts index a7a3b29a..41a46a9c 100644 --- a/desktop/src/lib/backgroundTasks.test.ts +++ b/desktop/src/lib/backgroundTasks.test.ts @@ -6,6 +6,7 @@ import { hasRunningBackgroundTasks, hasRunningSubagentTasks, isVisibleSessionBackgroundTask, + listRunningBackgroundTasks, } from './backgroundTasks' import { translate } from '../i18n' @@ -44,6 +45,48 @@ describe('hasRunningBackgroundTasks', () => { }) }) +describe('listRunningBackgroundTasks', () => { + it('lists user-started tasks that are still running, in insertion order', () => { + const tasks = { + shell: task('shell', { taskType: 'local_bash' }), + agent: task('agent', { taskType: 'local_agent' }), + workflow: task('workflow', { taskType: 'local_workflow' }), + } + expect(listRunningBackgroundTasks(tasks).map((entry) => entry.taskId)).toEqual(['shell', 'agent', 'workflow']) + }) + + it('leaves out finished tasks, AutoDream and teammate runtime containers', () => { + const tasks = { + shell: task('shell', { taskType: 'local_bash' }), + completed: task('completed', { taskType: 'local_bash', status: 'completed' }), + failed: task('failed', { taskType: 'local_bash', status: 'failed' }), + stopped: task('stopped', { taskType: 'local_bash', status: 'stopped' }), + dream: task('dream', { taskType: 'dream' }), + teammate: task('teammate', { taskType: 'in_process_teammate' }), + } + expect(listRunningBackgroundTasks(tasks).map((entry) => entry.taskId)).toEqual(['shell']) + }) + + it('is empty for a missing task record', () => { + expect(listRunningBackgroundTasks(undefined)).toEqual([]) + expect(listRunningBackgroundTasks({})).toEqual([]) + }) + + it('agrees with hasRunningBackgroundTasks about what counts as running', () => { + const records: Array> = [ + {}, + { dream: task('dream', { taskType: 'dream' }) }, + { teammate: task('teammate', { taskType: 'in_process_teammate' }) }, + { done: task('done', { status: 'completed' }) }, + { shell: task('shell', { taskType: 'local_bash' }) }, + { shell: task('shell', { taskType: 'local_bash' }), dream: task('dream', { taskType: 'dream' }) }, + ] + for (const record of records) { + expect(listRunningBackgroundTasks(record).length > 0).toBe(hasRunningBackgroundTasks(record)) + } + }) +}) + describe('hasRunningSubagentTasks', () => { it.each(['local_agent', 'remote_agent'])('reports a running %s as stoppable', (taskType) => { expect(hasRunningSubagentTasks({ diff --git a/desktop/src/lib/backgroundTasks.ts b/desktop/src/lib/backgroundTasks.ts index e16afffb..0a19ce55 100644 --- a/desktop/src/lib/backgroundTasks.ts +++ b/desktop/src/lib/backgroundTasks.ts @@ -10,14 +10,26 @@ export function isVisibleSessionBackgroundTask( return task.taskType !== 'in_process_teammate' } -export function hasRunningBackgroundTasks(tasks?: Record): boolean { +function isBusyBackgroundTask(task: BackgroundAgentTask): boolean { // AutoDream is detached maintenance work: it remains visible and stoppable // in Activity, but must not keep the foreground conversation marked busy. - return Object.values(tasks ?? {}).some( - (task) => isVisibleSessionBackgroundTask(task) && - task.status === 'running' && - task.taskType !== 'dream', - ) + return isVisibleSessionBackgroundTask(task) && + task.status === 'running' && + task.taskType !== 'dream' +} + +export function hasRunningBackgroundTasks(tasks?: Record): boolean { + return Object.values(tasks ?? {}).some(isBusyBackgroundTask) +} + +/** + * The tasks `hasRunningBackgroundTasks` counted, for a caller that has been told + * to stop them. Asking "is this session running?" and stopping it share one + * definition so a confirmation can never offer a Stop that leaves running the + * very task that raised it. + */ +export function listRunningBackgroundTasks(tasks?: Record): BackgroundAgentTask[] { + return Object.values(tasks ?? {}).filter(isBusyBackgroundTask) } export function hasRunningSubagentTasks(tasks?: Record): boolean { diff --git a/desktop/src/lib/sessionAttention.test.ts b/desktop/src/lib/sessionAttention.test.ts new file mode 100644 index 00000000..328006fe --- /dev/null +++ b/desktop/src/lib/sessionAttention.test.ts @@ -0,0 +1,130 @@ +import { describe, expect, it } from 'vitest' +import { createDefaultSessionState, type PendingPermission, type PerSessionState } from '../stores/chatStore' +import type { ChatState } from '../types/chat' +import { collectAttentionIds, nextAttentionSessionId, sessionNeedsAttention } from './sessionAttention' + +const CHAT_STATES: ChatState[] = [ + 'idle', + 'thinking', + 'compacting', + 'tool_executing', + 'streaming', + 'permission_pending', +] + +function perm(requestId: string, toolName = 'Bash'): PendingPermission { + return { requestId, toolName, toolUseId: `tu-${requestId}`, input: {} } +} + +function session(overrides: Partial = {}): PerSessionState { + return { ...createDefaultSessionState(), ...overrides } +} + +// The store keeps the newest request in `pendingPermission` as a compatibility +// mirror and every outstanding one in `pendingPermissions`. +function waiting(...requests: PendingPermission[]): Partial { + return { + pendingPermission: requests.at(-1) ?? null, + pendingPermissions: Object.fromEntries(requests.map((request) => [request.requestId, request])), + } +} + +describe('sessionNeedsAttention', () => { + it.each(CHAT_STATES)('lights for an outstanding request while chatState is %s', (chatState) => { + // `status` and `session_state` messages overwrite chatState under a card + // that is still open, so the records have to decide, whatever chatState says. + expect(sessionNeedsAttention(session({ chatState, ...waiting(perm('r1')) }))).toBe(true) + }) + + it.each(CHAT_STATES)('stays dark without a request while chatState is %s', (chatState) => { + // Including `permission_pending`: nothing renders a card for it, so a + // marker there would point at an empty tab. + expect(sessionNeedsAttention(session({ chatState }))).toBe(false) + }) + + it('reads a legacy session that only has the singular mirror', () => { + expect(sessionNeedsAttention(session({ pendingPermission: perm('r1'), pendingPermissions: undefined }))).toBe(true) + }) + + it('reads a session that only has the plural set', () => { + expect(sessionNeedsAttention(session({ + pendingPermission: null, + pendingPermissions: { r1: perm('r1') }, + }))).toBe(true) + }) + + it('stays dark for an emptied set with no mirror', () => { + expect(sessionNeedsAttention(session({ pendingPermission: null, pendingPermissions: {} }))).toBe(false) + }) + + it.each(['Bash', 'AskUserQuestion', 'ExitPlanMode'])('lights for a %s request', (toolName) => { + expect(sessionNeedsAttention(session(waiting(perm('r1', toolName))))).toBe(true) + }) + + it('does not count a Computer Use request, which has no card to answer', () => { + expect(sessionNeedsAttention(session({ + pendingComputerUsePermission: { requestId: 'cu-1', request: {} as never }, + pendingComputerUsePermissions: { 'cu-1': { requestId: 'cu-1', request: {} as never } }, + }))).toBe(false) + }) + + it('stays dark for a session the store does not know', () => { + expect(sessionNeedsAttention(undefined)).toBe(false) + }) +}) + +describe('collectAttentionIds', () => { + const sessions = { + a: session(waiting(perm('a1'))), + b: session(), + c: session(waiting(perm('c1', 'AskUserQuestion'))), + } + + it('scans every session by default', () => { + expect(collectAttentionIds(sessions)).toEqual(['a', 'c']) + }) + + it('limits the scan to the ids it is given, in that order', () => { + expect(collectAttentionIds(sessions, ['c', 'b', 'a'])).toEqual(['c', 'a']) + }) + + it('ignores ids the store has no session for', () => { + expect(collectAttentionIds(sessions, ['ghost', 'a'])).toEqual(['a']) + }) +}) + +describe('nextAttentionSessionId', () => { + const order = ['a', 'b', 'c', 'd'] + + it('takes the first waiting session after the active one', () => { + expect(nextAttentionSessionId(order, new Set(['c', 'd']), 'a')).toBe('c') + }) + + it('wraps past the end of the strip', () => { + expect(nextAttentionSessionId(order, new Set(['a', 'b']), 'c')).toBe('a') + }) + + it('never returns the active session, even when it is the only one waiting', () => { + expect(nextAttentionSessionId(order, new Set(['b']), 'b')).toBeNull() + }) + + it('skips the active session when others are waiting too', () => { + expect(nextAttentionSessionId(order, new Set(['b', 'd']), 'b')).toBe('d') + expect(nextAttentionSessionId(order, new Set(['b', 'd']), 'd')).toBe('b') + }) + + it('returns null when nothing is waiting or the strip is empty', () => { + expect(nextAttentionSessionId(order, new Set(), 'a')).toBeNull() + expect(nextAttentionSessionId([], new Set(['a']), null)).toBeNull() + }) + + it('starts from the first tab when the active one is not in the strip', () => { + // Settings, Market and friends are tabs but not sessions. + expect(nextAttentionSessionId(order, new Set(['a', 'c']), '__settings__')).toBe('a') + expect(nextAttentionSessionId(order, new Set(['c']), null)).toBe('c') + }) + + it('ignores waiting ids that have no tab', () => { + expect(nextAttentionSessionId(order, new Set(['ghost']), 'a')).toBeNull() + }) +}) diff --git a/desktop/src/lib/sessionAttention.ts b/desktop/src/lib/sessionAttention.ts new file mode 100644 index 00000000..6c27d2cd --- /dev/null +++ b/desktop/src/lib/sessionAttention.ts @@ -0,0 +1,49 @@ +import { listPendingPermissions, type PerSessionState } from '../stores/chatStore' + +type AttentionSource = Pick + +/** + * Whether a session is parked on a decision only the user can make: a tool + * approval, an AskUserQuestion card or an ExitPlanMode review. All three arrive + * as `permission_request` and land in the same records that PermissionDialog, + * AskUserQuestion and the composer render from, so a marker built on this can + * never disagree with the card behind the tab. + * + * It reads the records, not `chatState`. `status(tool_executing)`, + * `session_state(running)` and `content_start` overwrite `chatState` while the + * request is still open, which switched a state-based marker off under a card + * that was still waiting. Computer Use records are left out on purpose: nothing + * renders them, so counting them would light a marker with no card to answer. + * + * This is the one place that decides. The tab strip, both sidebar views and the + * mobile header all call it; a surface that grows its own rule will drift. + */ +export function sessionNeedsAttention(session: AttentionSource | undefined): boolean { + return listPendingPermissions(session).length > 0 +} + +/** The ids among `ids` (every session by default) that are waiting on the user. */ +export function collectAttentionIds( + sessions: Readonly>, + ids: readonly string[] = Object.keys(sessions), +): string[] { + return ids.filter((id) => sessionNeedsAttention(sessions[id])) +} + +/** + * The next waiting session after `activeId` in strip order, wrapping past the + * end and never returning `activeId` itself. An `activeId` that is not in the + * strip (Settings, a session with no tab) starts from the first entry. + */ +export function nextAttentionSessionId( + orderedIds: readonly string[], + attentionIds: ReadonlySet, + activeId: string | null, +): string | null { + const start = activeId ? orderedIds.indexOf(activeId) : -1 + for (let step = 1; step <= orderedIds.length; step += 1) { + const id = orderedIds[(start + step) % orderedIds.length] + if (id !== undefined && id !== activeId && attentionIds.has(id)) return id + } + return null +} diff --git a/desktop/src/stores/chatStore.test.ts b/desktop/src/stores/chatStore.test.ts index 0e92cd81..ad9b7ce6 100644 --- a/desktop/src/stores/chatStore.test.ts +++ b/desktop/src/stores/chatStore.test.ts @@ -3735,7 +3735,158 @@ describe('chatStore history mapping', () => { ]) }) - it('filters task-notification turns and resumes at the next real user message', () => { + it('hides the task-notification message but keeps every model reply that follows it', () => { + const notice = (taskId: string, toolUseId: string, name: string) => + `\n${taskId}\n${toolUseId}\ncompleted\nBackground command "Background sleep then write ${name}" completed (exit code 0)\n` + const messages: MessageEntry[] = [ + { id: 'user-real-1', type: 'user', timestamp: '2026-09-29T06:41:00.000Z', content: '启动两个后台命令' }, + { + id: 'assistant-dispatched', + type: 'assistant', + timestamp: '2026-09-29T06:41:03.000Z', + content: [{ type: 'text', text: '已派发' }], + }, + { id: 'notice-alpha', type: 'user', timestamp: '2026-09-29T06:42:25.000Z', content: notice('bt1', 'call_alpha', 'alpha') }, + { + id: 'assistant-alpha', + type: 'assistant', + timestamp: '2026-09-29T06:42:27.000Z', + content: [{ type: 'text', text: 'alpha 任务完成:ALPHA_DONE' }], + }, + { id: 'notice-bravo', type: 'user', timestamp: '2026-09-29T06:43:55.000Z', content: notice('bt2', 'call_bravo', 'bravo') }, + { + id: 'assistant-bravo', + type: 'assistant', + timestamp: '2026-09-29T06:43:57.000Z', + content: [{ type: 'text', text: 'bravo 任务完成:BRAVO_DONE' }], + }, + { id: 'user-real-2', type: 'user', timestamp: '2026-09-29T06:50:00.000Z', content: '继续' }, + { + id: 'assistant-real-2', + type: 'assistant', + timestamp: '2026-09-29T06:50:02.000Z', + content: [{ type: 'text', text: '好的' }], + }, + ] + + const mapped = mapHistoryMessagesToUiMessages(messages) + + expect(mapped.map((message) => [message.type, 'content' in message ? message.content : undefined])).toEqual([ + ['user_text', '启动两个后台命令'], + ['assistant_text', '已派发'], + ['assistant_text', 'alpha 任务完成:ALPHA_DONE'], + ['assistant_text', 'bravo 任务完成:BRAVO_DONE'], + ['user_text', '继续'], + ['assistant_text', '好的'], + ]) + expect(JSON.stringify(mapped)).not.toContain('') + }) + + it('keeps thinking, tool calls and tool results the model produces after a task notification', () => { + const messages: MessageEntry[] = [ + { id: 'user-real-1', type: 'user', timestamp: '2026-09-29T07:00:00.000Z', content: '构建项目' }, + { + id: 'assistant-started', + type: 'assistant', + timestamp: '2026-09-29T07:00:01.000Z', + content: [{ type: 'text', text: '构建已在后台启动' }], + }, + { + id: 'notice-build', + type: 'user', + timestamp: '2026-09-29T07:03:00.000Z', + content: '\nbuild-task\ntoolu_build\nfailed\nBackground command "npm run build" failed with exit code 2\n', + }, + { + id: 'assistant-thinking', + type: 'assistant', + timestamp: '2026-09-29T07:03:02.000Z', + content: [{ type: 'thinking', thinking: '构建失败,先看日志。' }], + }, + { + id: 'assistant-tool', + type: 'tool_use', + timestamp: '2026-09-29T07:03:03.000Z', + content: [{ type: 'tool_use', id: 'toolu_fix_1', name: 'Bash', input: { command: 'tail -n 20 build.log' } }], + }, + { + id: 'tool-result', + type: 'tool_result', + timestamp: '2026-09-29T07:03:04.000Z', + content: [{ type: 'tool_result', tool_use_id: 'toolu_fix_1', content: 'error TS2322: type mismatch' }], + }, + { + id: 'assistant-fix', + type: 'assistant', + timestamp: '2026-09-29T07:03:06.000Z', + content: [{ type: 'text', text: '类型错误,正在修复。' }], + }, + ] + + const mapped = mapHistoryMessagesToUiMessages(messages) + + expect(mapped.map((message) => message.type)).toEqual([ + 'user_text', + 'assistant_text', + 'thinking', + 'tool_use', + 'tool_result', + 'assistant_text', + ]) + expect(mapped[3]).toMatchObject({ toolUseId: 'toolu_fix_1', toolName: 'Bash' }) + expect(mapped[4]).toMatchObject({ toolUseId: 'toolu_fix_1' }) + expect(mapped[5]).toMatchObject({ content: '类型错误,正在修复。' }) + expect(JSON.stringify(mapped)).not.toContain('') + }) + + it('keeps a notification follow-up turn that is stored as raw assistant and user transcript blocks', () => { + const messages: MessageEntry[] = [ + { + id: 'notice-raw', + type: 'user', + timestamp: '2026-09-29T07:10:00.000Z', + content: '\nraw-task\ntoolu_raw\ncompleted\nBackground command completed\n', + }, + { + id: 'assistant-raw', + type: 'assistant', + timestamp: '2026-09-29T07:10:02.000Z', + content: [ + { type: 'thinking', thinking: '读取输出文件。' }, + { type: 'text', text: '先看一下输出。' }, + { type: 'tool_use', name: 'Read', id: 'toolu_read_1', input: { file_path: '/tmp/out.txt' } }, + ], + }, + { + id: 'user-tool-result', + type: 'user', + timestamp: '2026-09-29T07:10:03.000Z', + content: [{ type: 'tool_result', tool_use_id: 'toolu_read_1', content: 'all green' }], + }, + { + id: 'assistant-raw-final', + type: 'assistant', + timestamp: '2026-09-29T07:10:04.000Z', + content: [{ type: 'text', text: '全部通过。' }], + }, + ] + + const mapped = mapHistoryMessagesToUiMessages(messages) + + expect(mapped.map((message) => message.type)).toEqual([ + 'thinking', + 'assistant_text', + 'tool_use', + 'tool_result', + 'assistant_text', + ]) + expect(mapped[4]).toMatchObject({ content: '全部通过。' }) + expect(JSON.stringify(mapped)).not.toContain('') + }) + + it('keeps even a no-action reply to a stale task notification rather than hiding model output', () => { + // The transcript cannot tell a filler reply from a useful one, so it never + // guesses: only the injected notification prompt is hidden. const messages: MessageEntry[] = [ { id: 'user-real-1', @@ -3781,6 +3932,10 @@ describe('chatStore history mapping', () => { type: 'assistant_text', content: '项目创建好了', }, + { + type: 'assistant_text', + content: '旧后台任务通知,无需处理', + }, { id: 'user-real-2', type: 'user_text', @@ -3788,7 +3943,6 @@ describe('chatStore history mapping', () => { }, ]) expect(JSON.stringify(mapped)).not.toContain('') - expect(JSON.stringify(mapped)).not.toContain('旧后台任务通知') }) it('reconstructs task notifications from transcript XML before filtering it from UI', () => { @@ -12965,8 +13119,8 @@ describe('chatStore history mapping', () => { expect(useChatStore.getState().sessions[TEST_SESSION_ID]?.messages).toHaveLength(2) }) - // The task_notification that should suppress this output arrived while - // the renderer was disconnected, so only the late follow-up is observed. + // Whatever started this turn (a task_notification, for example) arrived + // while the renderer was disconnected, so only its late thinking is observed. useChatStore.getState().handleServerMessage(TEST_SESSION_ID, { type: 'thinking', text: 'orphan background follow-up thinking', @@ -13195,7 +13349,213 @@ describe('chatStore history mapping', () => { ])) }) - it('suppresses assistant output for a task-notification-only follow-up turn', () => { + it('shows the model reply to a background command completion as soon as its turn ends, and keeps it after the transcript reload', async () => { + // Replays a callback turn captured from a real desktop server (issue #1389): + // the task_notification lands while the session is idle, then the CLI starts + // a follow-up turn on its own and answers it. Nothing here comes from the user. + const send = (message: ServerMessage) => + useChatStore.getState().handleServerMessage(TEST_SESSION_ID, message) + vi.mocked(sessionsApi.getFullHistory).mockResolvedValue({ + messages: [ + { + id: 'user-1', + type: 'user', + timestamp: '2026-09-29T06:41:00.000Z', + content: 'Start the background commands', + }, + { + id: 'assistant-dispatched', + type: 'assistant', + timestamp: '2026-09-29T06:41:03.000Z', + content: [{ type: 'text', text: '已派发' }], + }, + { + id: 'assistant-callback', + type: 'assistant', + timestamp: '2026-09-29T06:42:29.000Z', + content: [{ type: 'text', text: 'alpha 任务完成:ALPHA_DONE' }], + }, + ], + }) + useChatStore.setState({ + sessions: { + [TEST_SESSION_ID]: makeSession({ + chatState: 'idle', + messages: [ + { id: 'user-1', type: 'user_text', content: 'Start the background commands', timestamp: 1 }, + { id: 'assistant-dispatched', type: 'assistant_text', content: '已派发', timestamp: 2 }, + ], + }), + }, + }) + + send({ + type: 'system_notification', + subtype: 'task_notification', + data: { + type: 'system', + subtype: 'task_notification', + task_id: 'bt1uz0yz1', + tool_use_id: 'call_00_ddp9AzDwk9lWwuLP9pom8301', + status: 'completed', + output_file: '/tmp/bt1uz0yz1.output', + summary: 'Background command "Background sleep 25 then write alpha" completed (exit code 0)', + }, + }) + send({ + type: 'system_notification', + subtype: 'init', + message: 'Model: deepseek-flash[1m]', + data: { model: 'deepseek-flash[1m]' }, + }) + send({ + type: 'system_notification', + subtype: 'slash_commands', + data: [{ name: 'update-config', description: 'Configure the harness' }], + }) + send({ type: 'status', state: 'thinking', attemptStart: true }) + send({ type: 'status', state: 'thinking', verb: 'Thinking' }) + send({ type: 'content_start', blockType: 'text' }) + for (const text of ['alpha', ' ', '任务', '完成', ':', 'AL', 'P', 'HA', '_D', 'ONE']) { + send({ type: 'content_delta', text }) + } + send({ + type: 'message_complete', + usage: { input_tokens: 249, output_tokens: 11, cache_read_tokens: 30592 }, + timing: { duration_ms: 500, duration_api_ms: 3026, ttft_ms: 0, decode_ms: 0 }, + }) + + // The reply must not wait for a transcript round trip to become visible. + const live = useChatStore.getState().sessions[TEST_SESSION_ID] + expect(live?.chatState).toBe('idle') + expect(live?.messages).toEqual(expect.arrayContaining([ + expect.objectContaining({ type: 'assistant_text', content: 'alpha 任务完成:ALPHA_DONE' }), + expect.objectContaining({ + type: 'background_task', + task: expect.objectContaining({ taskId: 'bt1uz0yz1', status: 'completed' }), + }), + ])) + // The desktop notification previews the same reply the transcript shows. + expect(notifyDesktopMock).toHaveBeenCalledWith(expect.objectContaining({ + title: 'Claude Code Haha 已完成回复', + body: expect.stringContaining('alpha 任务完成'), + })) + + // The authoritative reload carries the same reply and must not duplicate it. + await vi.waitFor(() => { + expect(sessionsApi.getFullHistory).toHaveBeenCalled() + const settled = useChatStore.getState().sessions[TEST_SESSION_ID] + expect(settled?.historyStatus).toBe('ready') + expect(settled?.messages.filter((message) => message.type === 'assistant_text') + .map((message) => message.type === 'assistant_text' ? message.content : '')) + .toEqual(['已派发', 'alpha 任务完成:ALPHA_DONE']) + }) + }) + + it('shows the thinking and tool work the model performs in response to a background task notification', async () => { + const send = (message: ServerMessage) => + useChatStore.getState().handleServerMessage(TEST_SESSION_ID, message) + const notification = [ + '', + 'build-task', + 'toolu_build', + 'failed', + 'Background command "npm run build" failed with exit code 2', + '', + ].join('\n') + vi.mocked(sessionsApi.getFullHistory).mockResolvedValue({ + messages: [ + { id: 'user-1', type: 'user', timestamp: '2026-09-29T07:00:00.000Z', content: 'Build the project' }, + { + id: 'assistant-started', + type: 'assistant', + timestamp: '2026-09-29T07:00:01.000Z', + content: [{ type: 'text', text: 'Build started in the background' }], + }, + { id: 'notification-user', type: 'user', timestamp: '2026-09-29T07:03:00.000Z', content: notification }, + { + id: 'assistant-thinking', + type: 'assistant', + timestamp: '2026-09-29T07:03:02.000Z', + content: [{ type: 'thinking', thinking: 'The build failed, so I will read its log.' }], + }, + { + id: 'assistant-tool', + type: 'tool_use', + timestamp: '2026-09-29T07:03:03.000Z', + content: [{ type: 'tool_use', id: 'toolu_fix_1', name: 'Bash', input: { command: 'tail -n 20 build.log' } }], + }, + { + id: 'tool-result', + type: 'tool_result', + timestamp: '2026-09-29T07:03:04.000Z', + content: [{ type: 'tool_result', tool_use_id: 'toolu_fix_1', content: 'error TS2322: type mismatch' }], + }, + { + id: 'assistant-fix', + type: 'assistant', + timestamp: '2026-09-29T07:03:06.000Z', + content: [{ type: 'text', text: 'The build fails on a type error; fixing it now.' }], + }, + ], + }) + useChatStore.setState({ + sessions: { + [TEST_SESSION_ID]: makeSession({ + chatState: 'idle', + messages: [ + { id: 'user-1', type: 'user_text', content: 'Build the project', timestamp: 1 }, + { id: 'assistant-started', type: 'assistant_text', content: 'Build started in the background', timestamp: 2 }, + ], + }), + }, + }) + + send({ + type: 'system_notification', + subtype: 'task_notification', + data: { + task_id: 'build-task', + tool_use_id: 'toolu_build', + status: 'failed', + summary: 'Background command "npm run build" failed with exit code 2', + }, + }) + send({ type: 'status', state: 'thinking', attemptStart: true }) + send({ type: 'thinking', text: 'The build failed, so I will read its log.' }) + send({ type: 'content_start', blockType: 'tool_use', toolName: 'Bash', toolUseId: 'toolu_fix_1' }) + send({ + type: 'tool_use_complete', + toolName: 'Bash', + toolUseId: 'toolu_fix_1', + input: { command: 'tail -n 20 build.log' }, + }) + send({ type: 'tool_result', toolUseId: 'toolu_fix_1', content: 'error TS2322: type mismatch', isError: false }) + send({ type: 'content_start', blockType: 'text' }) + send({ type: 'content_delta', text: 'The build fails on a type error; fixing it now.' }) + send({ type: 'message_complete', usage: { input_tokens: 40, output_tokens: 22 } }) + + const expectFollowUpWork = (messages: UIMessage[] | undefined) => { + expect(messages).toEqual(expect.arrayContaining([ + expect.objectContaining({ type: 'thinking', content: 'The build failed, so I will read its log.' }), + expect.objectContaining({ type: 'tool_use', toolUseId: 'toolu_fix_1', toolName: 'Bash' }), + expect.objectContaining({ type: 'tool_result', toolUseId: 'toolu_fix_1' }), + expect.objectContaining({ type: 'assistant_text', content: 'The build fails on a type error; fixing it now.' }), + ])) + } + expectFollowUpWork(useChatStore.getState().sessions[TEST_SESSION_ID]?.messages) + + await vi.waitFor(() => { + const settled = useChatStore.getState().sessions[TEST_SESSION_ID] + expect(settled?.historyStatus).toBe('ready') + expectFollowUpWork(settled?.messages) + expect(JSON.stringify(settled?.messages)).not.toContain('') + }) + }) + + it('shows the thinking and reply of a follow-up turn started only by a task notification', () => { + // Even a reply that says "nothing more to add" is model output: the store + // cannot tell it from a useful one, so it never hides it. useChatStore.setState({ sessions: { [TEST_SESSION_ID]: makeSession({ @@ -13243,17 +13603,20 @@ describe('chatStore history mapping', () => { status: 'completed', }, }, + { + type: 'thinking', + content: "The earlier monitoring command has already been handled by subsequent work, so there's nothing more to add here.", + }, + { + type: 'assistant_text', + content: '那是早前的监控命令收尾通知,已被后续的多核压测取代,无需处理。交付已全部完成并验证通过。', + }, ]) - expect(session?.messages).not.toEqual(expect.arrayContaining([ - expect.objectContaining({ type: 'thinking' }), - expect.objectContaining({ type: 'assistant_text' }), - expect.objectContaining({ type: 'system', content: 'Completed in 11m 58s' }), - ])) expect(session?.chatState).toBe('idle') expect(updateTabStatusMock).toHaveBeenLastCalledWith(TEST_SESSION_ID, 'idle') }) - it('does not suppress foreground skill output when a background task completes', () => { + it('keeps foreground skill output visible when a background task completes mid-turn', () => { useChatStore.setState({ sessions: { [TEST_SESSION_ID]: makeSession({ @@ -13298,8 +13661,6 @@ describe('chatStore history mapping', () => { }, }) - expect(useChatStore.getState().sessions[TEST_SESSION_ID]?.suppressNextTaskNotificationResponse).not.toBe(true) - useChatStore.getState().handleServerMessage(TEST_SESSION_ID, { type: 'content_start', blockType: 'text', diff --git a/desktop/src/stores/chatStore.ts b/desktop/src/stores/chatStore.ts index 64cfa067..d9b65071 100644 --- a/desktop/src/stores/chatStore.ts +++ b/desktop/src/stores/chatStore.ts @@ -202,7 +202,6 @@ export type PerSessionState = { historyMutationEpoch?: number /** Changes when a directed child stream starts or settles. */ agentStreamRevision?: number - suppressNextTaskNotificationResponse?: boolean replaceHistoryOnCompletion?: boolean activeGoal?: ActiveGoalState | null activeGoalRevision?: number @@ -256,7 +255,6 @@ const DEFAULT_SESSION_STATE: PerSessionState = { historyBootstrapDisabled: false, historyMutationEpoch: 0, agentStreamRevision: 0, - suppressNextTaskNotificationResponse: false, replaceHistoryOnCompletion: false, activeGoal: null, activeGoalRevision: 0, @@ -1332,15 +1330,6 @@ function isCancellableSubagentTask(task: BackgroundAgentTask): boolean { ) } -function shouldSuppressTaskNotificationResponse(session: PerSessionState): boolean { - if (session.chatState !== 'idle') return false - const lastMessage = session.messages[session.messages.length - 1] - const hasVisibleActiveOutput = - session.streamingText.trim().length > 0 || - Boolean(session.activeToolUseId) - return !hasVisibleActiveOutput && lastMessage?.type !== 'user_text' -} - function mergeRestoredTerminalGoalEvents( messages: UIMessage[], restoredMessages: UIMessage[], @@ -3337,7 +3326,6 @@ export const useChatStore = create((setState, get) => { isPreparingTurn: false, historyMutationEpoch: (session.historyMutationEpoch ?? 0) + 1, elapsedSeconds: 0, - suppressNextTaskNotificationResponse: false, replaceHistoryOnCompletion: false, streamingText: '', streamingResponseChars: 0, @@ -3560,7 +3548,6 @@ export const useChatStore = create((setState, get) => { pendingComputerUsePermissions: {}, apiRetry: null, streamingFallback: null, - suppressNextTaskNotificationResponse: false, stoppingBackgroundTaskIds, stopAllSubagentsRequested: true, elapsedTimer: null, @@ -4502,7 +4489,6 @@ export const useChatStore = create((setState, get) => { queuedUserMessages: (currentSession.queuedUserMessages ?? []) .filter((message) => message.id !== messageId), ...(pendingText.trim() ? { streamingText: '' } : {}), - suppressNextTaskNotificationResponse: false, replaceHistoryOnCompletion: false, } }), @@ -4537,7 +4523,6 @@ export const useChatStore = create((setState, get) => { preHydrationSocketGapPending: false, apiRetry: null, streamingFallback: null, - suppressNextTaskNotificationResponse: false, replaceHistoryOnCompletion: false, queuedUserMessages: [], })) })) @@ -4912,18 +4897,6 @@ export const useChatStore = create((setState, get) => { case 'content_start': { const session = get().sessions[sessionId] if (!session) break - if (session.suppressNextTaskNotificationResponse && msg.blockType === 'text') { - consumePendingDelta(sessionId) - update(() => ({ - streamingText: '', - activeThinkingId: null, - statusVerb: '', - })) - break - } - if (session.suppressNextTaskNotificationResponse) { - update(() => ({ suppressNextTaskNotificationResponse: false })) - } // The server keeps a stopped-turn fence until it attributes a replay // (or a pure local command's first output) to the replacement turn. // Mirror that boundary instead of clearing the SubAgent stop latch on @@ -5065,10 +5038,6 @@ export const useChatStore = create((setState, get) => { } case 'content_delta': - if (get().sessions[sessionId]?.suppressNextTaskNotificationResponse) { - consumePendingDelta(sessionId) - break - } let receivedLiveDelta = false if (msg.text !== undefined) { if (!get().sessions[sessionId]) break @@ -5129,15 +5098,6 @@ export const useChatStore = create((setState, get) => { break case 'thinking': { - if (get().sessions[sessionId]?.suppressNextTaskNotificationResponse) { - consumePendingDelta(sessionId) - update(() => ({ - streamingText: '', - activeThinkingId: null, - statusVerb: '', - })) - break - } // 重放/空块都不该冒出一个新的「已思考」气泡,也不该把会话拖回 thinking 态 // 或者启动计时器 —— 那正是"打开一个早就结束的会话,它自己开始输出"的观感。 let skippedThinkingBlock = false @@ -5466,39 +5426,6 @@ export const useChatStore = create((setState, get) => { void cliTaskStore.refreshTasks(sessionId) } } - if (session.suppressNextTaskNotificationResponse) { - consumePendingDelta(sessionId) - clearPendingToolInputDelta(sessionId) - if (session.elapsedTimer) clearInterval(session.elapsedTimer) - const hasRunningBackgroundAgents = hasRunningBackgroundTasks(session.backgroundAgentTasks) - update((current) => ({ - tokenUsage: msg.usage, - chatState: 'idle', - activeThinkingId: null, - pendingPermission: null, - pendingPermissions: {}, - pendingComputerUsePermission: null, - pendingComputerUsePermissions: {}, - elapsedTimer: null, - apiRetry: null, - streamingFallback: null, - streamingText: '', - streamingToolInput: '', - suppressNextTaskNotificationResponse: false, - replaceHistoryOnCompletion: false, - historyMutationEpoch: (current.historyMutationEpoch ?? 0) + 1, - })) - useTabStore.getState().updateTabStatus(sessionId, hasRunningBackgroundAgents ? 'running' : 'idle') - reconcileCompletedTranscriptHistory( - get, - sessionId, - session.replaceHistoryOnCompletion === true, - ) - for (const queuedMessage of get().sessions[sessionId]?.queuedUserMessages ?? []) { - get().sendQueuedUserMessage(sessionId, queuedMessage.id) - } - break - } const completedAt = Date.now() const wasAgentRunning = session.chatState !== 'idle' const text = `${session.streamingText}${consumePendingDelta(sessionId)}` @@ -5565,7 +5492,6 @@ export const useChatStore = create((setState, get) => { messages: appendReplayedUserMessage(baseMessages, msg.content, Date.now(), msg.sessionReferences, msg.collaboration), ...(pendingText.trim() ? { streamingText: '' } : {}), activeThinkingId: null, - suppressNextTaskNotificationResponse: false, replaceHistoryOnCompletion: false, stopAllSubagentsRequested: false, historyMutationEpoch: (session.historyMutationEpoch ?? 0) + 1, @@ -5613,7 +5539,6 @@ export const useChatStore = create((setState, get) => { pendingComputerUsePermissions: {}, apiRetry: null, streamingFallback: null, - suppressNextTaskNotificationResponse: false, historyMutationEpoch: (s.historyMutationEpoch ?? 0) + 1, } }) @@ -5977,18 +5902,11 @@ export const useChatStore = create((setState, get) => { hasRunningBackgroundAgentsAfterUpdate = hasRunningBackgroundTasks(backgroundAgentTasks) const task = backgroundAgentTasks[taskEvent.taskId] const accepted = Boolean(task) - const suppressNotificationResponse = - accepted && - (taskEvent.status === 'completed' || - taskEvent.status === 'failed' || - taskEvent.status === 'stopped') && - shouldSuppressTaskNotificationResponse(session) const stoppingBackgroundTaskIds = { ...session.stoppingBackgroundTaskIds } delete stoppingBackgroundTaskIds[taskEvent.taskId] return { ...buildBackgroundTaskSessionUpdate(session, backgroundAgentTasks, task, now), stoppingBackgroundTaskIds, - ...(suppressNotificationResponse ? { suppressNextTaskNotificationResponse: true } : {}), agentTaskNotifications: { ...session.agentTaskNotifications, ...(accepted && @@ -7774,14 +7692,14 @@ export function mapHistoryMessagesToUiMessages( ): UIMessage[] { const includeTeammateMessages = options?.includeTeammateMessages === true const uiMessages: UIMessage[] = [] - let suppressTaskNotificationResponse = false let pendingGoalCommand: { name: string; args: string } | null = null for (const msg of messages) { - if (msg.type === 'user' && isTaskNotificationContent(msg.content)) { - suppressTaskNotificationResponse = true - continue - } + // A task notification is a system-injected prompt: the background task + // cards already stand for it. What the assistant does in the turn that + // answers it (thinking, text, tool calls) is ordinary transcript content + // and stays visible. + if (msg.type === 'user' && isTaskNotificationContent(msg.content)) continue if (msg.type === 'user') { const commandDisplayText = getCommandMetadataDisplayText(msg.content) if (commandDisplayText) { @@ -7792,18 +7710,12 @@ export function mapHistoryMessagesToUiMessages( ...(msg.id ? { transcriptMessageId: msg.id } : {}), timestamp: new Date(msg.timestamp).getTime(), }) - suppressTaskNotificationResponse = false continue } if (shouldHideCommandMetadataContent(msg.content)) { continue } } - if (msg.type === 'user') { - suppressTaskNotificationResponse = false - } else if (suppressTaskNotificationResponse) { - continue - } const timestamp = new Date(msg.timestamp).getTime() if ( diff --git a/desktop/src/stores/teamStore.ts b/desktop/src/stores/teamStore.ts index 7db66e3b..b0e77539 100644 --- a/desktop/src/stores/teamStore.ts +++ b/desktop/src/stores/teamStore.ts @@ -1409,8 +1409,8 @@ export const useTeamStore = create((set, get) => ({ ), }, })) - // Mapping the complete durable transcript preserves suppression state - // across cursor pages (for example task-notification follow-up blocks). + // Mapping the complete durable transcript keeps state that spans entries + // intact across cursor pages (for example a goal command and its output). // Pending local sends are merged back only after that projection. const transcriptMessages = mapHistoryMessagesToUiMessages( mergedEntries, diff --git a/desktop/src/test/webStorage.ts b/desktop/src/test/webStorage.ts index 75667e2f..05d8af28 100644 --- a/desktop/src/test/webStorage.ts +++ b/desktop/src/test/webStorage.ts @@ -37,6 +37,24 @@ class MemoryStorage implements Storage { } function installStorage(name: 'localStorage' | 'sessionStorage'): void { + // StorageEvent validates storageArea against jsdom's Storage implementation. + // Node's experimental global accessor shadows window[name], so use jsdom's + // backing instance when it is available. + try { + const browserStorage = + (window as unknown as Record)[`_${name}`] ?? window[name] + if (browserStorage instanceof window.Storage) { + Object.defineProperty(globalThis, name, { + value: browserStorage, + configurable: true, + writable: true, + }) + return + } + } catch { + // An opaque jsdom origin may not expose storage; use the in-memory fallback. + } + let usable = false try { const existing = (globalThis as Record)[name] as Storage | undefined diff --git a/desktop/src/theme/contrast.test.ts b/desktop/src/theme/contrast.test.ts index ff092c45..5491409b 100644 --- a/desktop/src/theme/contrast.test.ts +++ b/desktop/src/theme/contrast.test.ts @@ -310,6 +310,46 @@ describe('link affordance contrast', () => { } }) +describe('waiting-on-you mark', () => { + /** + * `SessionAttentionMark` is a filled triangle and nothing else: no fill + * behind it, no text beside it to carry the meaning if the shape were lost. A + * graphic that conveys state answers to WCAG 1.4.11, so it needs 3:1 against + * every ground it can sit on: the strip's trough (idle tabs, and sidebar rows + * at rest), paper (the active tab), and the fills a sidebar row takes when + * hovered or selected. + * + * It is drawn in `--color-on-warning-container`. The plain `--color-warning` + * measured 2.92:1 on warm-classic's hovered sidebar row and was the reason + * for the switch; the two are the same colour in the ink themes. If a palette + * tweak drops a ground below the line, fix the colour, not this number. + */ + const MARK = '--color-on-warning-container' + const GROUNDS = [ + '--color-surface-sidebar', + '--color-surface', + '--color-sidebar-item-hover', + '--color-sidebar-item-active', + ] as const + + for (const [theme, selectors] of Object.entries(THEME_BLOCKS)) { + for (const ground of GROUNDS) { + it(`keeps the warning triangle visible on ${ground} in ${theme}`, () => { + // Dark's row fills are translucent; composite them on the trough they sit on. + const trough = parseColor(resolve('--color-surface-sidebar', selectors)) + const fill = flatten(parseColor(resolve(ground, selectors)), trough) + const mark = flatten(parseColor(resolve(MARK, selectors)), fill) + const ratio = contrast(mark, fill) + + expect( + Number(ratio.toFixed(2)), + `${theme}: ${MARK} on ${ground} is ${ratio.toFixed(2)}:1, needs ${AA_CONTROL_BOUNDARY}:1`, + ).toBeGreaterThanOrEqual(AA_CONTROL_BOUNDARY) + }) + } + } +}) + describe('tab strip outlines', () => { /** * The tab strip's trough is deliberately the sidebar's own ground rather diff --git a/docs/desktop/sessions.md b/docs/desktop/sessions.md index 8cee8533..d463f094 100644 --- a/docs/desktop/sessions.md +++ b/docs/desktop/sessions.md @@ -13,7 +13,9 @@ order: 1 点侧边栏的「新建会话」,或按 `⌘N`(Windows / Linux 是 `Ctrl+N`)。空会话首屏只要求你做一件事:选一个项目目录。选完就能提问了,模型和权限沿用你在设置里的默认值。 -会话开出来是一个标签页,可以像浏览器一样开很多个并排跑。标签上有小圆点表示这条会话还在运行;关一个正在跑的标签会先问你是「保持运行」还是「停止并关闭」。 +会话开出来是一个标签页,可以像浏览器一样开很多个并排跑。标签上有小圆点表示这条会话还在运行;关一个正在跑的标签会先问你是「保持运行」还是「停止并关闭」;「停止并关闭」会连这条会话里还在跑的后台任务一起停掉。 + +会话停在权限请求、提问或计划审批上等你处理时,标签上的小圆点会换成琥珀色的三角叹号,侧边栏里对应的会话行也一样。它会一直亮到你处理完为止,和系统通知开不开无关。标签多到滚出视野时,那一侧的滚动箭头上会出现一个琥珀色小点;标签栏右侧的数字按钮显示还有几个会话在等你,点一下跳到下一个。手机宽度下没有标签栏,汉堡按钮上的小点表示别的会话在等你,打开侧边栏找带三角的那一行。 会话标题下面那行小字是元信息:项目路径、分支、模型。想换项目就新建一条会话,一条会话绑定一个目录。 @@ -84,7 +86,7 @@ Claude 每完成一轮并修改文件,对话里会出现一张「{n} 个文件 - **任务** — Claude 自己维护的待办清单,顶部显示「任务进度 3/7」。 - **SubAgent** — 它派出去的子 Agent,点进去能看子 Agent 完整的运行记录。 -- **后台任务** — 挂在后台跑的命令和工作流,可以单独停掉某一个。 +- **后台任务** — 挂在后台跑的命令和工作流,可以单独停掉某一个。任务结束时 Claude 会收到通知并接着处理,它的回复照常出现在对话里。 - **团队** — 用到 Agent Team 时,每个成员一行,可以点进去直接给某个成员发消息。 后台跑的子 Agent 的工具活动也会冒泡到这里,不用等它跑完才知道它在干什么。 diff --git a/docs/en/desktop/sessions.md b/docs/en/desktop/sessions.md index fcdc8333..2a40a5af 100644 --- a/docs/en/desktop/sessions.md +++ b/docs/en/desktop/sessions.md @@ -13,7 +13,9 @@ A session is one complete collaboration: you describe what you want, Claude read Click **New session** in the sidebar, or press `⌘N` (`Ctrl+N` on Windows and Linux). The empty session asks you for exactly one thing: a project directory. After that you can start typing — the model and permission mode come from your defaults in Settings. -Each session opens as a tab, and you can run many side by side. A dot on the tab means that session is still running; closing a running tab asks whether you want to **Keep running** or **Stop and close**. +Each session opens as a tab, and you can run many side by side. A dot on the tab means that session is still running; closing a running tab asks whether you want to **Keep running** or **Stop and close**, and **Stop and close** also stops the background tasks that session still has running. + +When a session is stopped on a permission request, a question or a plan review and needs you, the dot on its tab becomes an amber warning triangle, and so does its row in the sidebar. It stays until you have dealt with it, whether or not system notifications are on. When so many tabs are open that some scroll out of view, the scroll arrow on that side gets an amber dot, and the number button on the right of the tab bar shows how many other sessions are waiting; press it to jump to the next one. On a phone there is no tab bar: a dot on the menu button means another session is waiting, and its row in the sidebar carries the triangle. The small line under the session title is metadata: project path, branch, model. A session is bound to one directory — to work on a different project, start a new session. @@ -84,7 +86,7 @@ The first button on the right of the tab bar opens the Activity panel, which lis - **Tasks** — the to-do list Claude maintains for itself, with "Task progress 3/7" at the top. - **SubAgents** — the agents it delegated to. Open one to read its full transcript. -- **Background tasks** — commands and workflows running in the background; each can be stopped individually. +- **Background tasks** — commands and workflows running in the background; each can be stopped individually. When one finishes, Claude is notified and carries on, and its reply appears in the conversation as usual. - **Team** — when an Agent Team is in play, one row per member, and you can message a member directly. Tool activity from background subagents bubbles up here too, so you don't have to wait for one to finish to see what it's doing. diff --git a/scripts/quality-gate/coverage.ts b/scripts/quality-gate/coverage.ts index d2f9923e..6828ce2c 100644 --- a/scripts/quality-gate/coverage.ts +++ b/scripts/quality-gate/coverage.ts @@ -156,6 +156,7 @@ const DESKTOP_SCOPE: CoverageScope = { includePrefixes: ['desktop/src/'], excludePrefixes: [ 'desktop/src/types/', + 'desktop/src/test/', // Dev-only tooling, same category as mocks/. `dev/` holds the component // gallery, which Vite never bundles (its build input is index.html alone) // and which exists precisely to be looked at by a person — unit-testing a diff --git a/scripts/quality-gate/desktop-smoke/deterministic.test.ts b/scripts/quality-gate/desktop-smoke/deterministic.test.ts index 6cc2a7c6..cdb02b92 100644 --- a/scripts/quality-gate/desktop-smoke/deterministic.test.ts +++ b/scripts/quality-gate/desktop-smoke/deterministic.test.ts @@ -5,6 +5,8 @@ import { tmpdir } from 'node:os' import { join } from 'node:path' import { DESKTOP_UI_SMOKE_ALLOW_SELECTOR, + DESKTOP_UI_SMOKE_ATTENTION_PROBE, + DESKTOP_UI_SMOKE_ATTENTION_TAB, DESKTOP_UI_SMOKE_LOCALE, buildDesktopUiSmokeBootstrap, buildDesktopUiSmokePrompt, @@ -34,6 +36,24 @@ describe('deterministic desktop UI smoke setup', () => { expect(DESKTOP_UI_SMOKE_ALLOW_SELECTOR).toBe('button[aria-label^="Allow: "]') }) + test('matches the waiting-tab marker the desktop actually renders', () => { + const tabBar = readFileSync('desktop/src/components/layout/TabBar.tsx', 'utf8') + const mark = readFileSync('desktop/src/components/layout/SessionAttentionMark.tsx', 'utf8') + const english = readFileSync('desktop/src/i18n/locales/en.ts', 'utf8') + + // Same idea as the approval button above: the selectors are derived from the + // production markup, so a rename fails here before the lane times out waiting. + expect(tabBar).toContain('data-testid="tab-bar"') + expect(tabBar).toContain("data-attention={needsAttention ? 'true' : 'false'}") + expect(mark).toContain('role="img"') + expect(english).toContain("'tabs.sessionRunning': 'Session running'") + expect(DESKTOP_UI_SMOKE_ATTENTION_TAB).toBe('[data-testid="tab-bar"] [data-attention="true"]') + expect(DESKTOP_UI_SMOKE_ATTENTION_PROBE).toContain(DESKTOP_UI_SMOKE_ATTENTION_TAB) + expect(DESKTOP_UI_SMOKE_ATTENTION_PROBE).toContain('[role="img"]') + // The probe must also reject a tab that carries the running dot. + expect(DESKTOP_UI_SMOKE_ATTENTION_PROBE).toContain('!tab.querySelector(\'[aria-label="Session running"]\')') + }) + test('asks the mock runtime to write inside the fixture copy only', () => { const projectDir = '/tmp/fixture-copy' const { target, prompt } = buildDesktopUiSmokePrompt(projectDir) diff --git a/scripts/quality-gate/desktop-smoke/deterministic.ts b/scripts/quality-gate/desktop-smoke/deterministic.ts index ae72099c..352b7fee 100644 --- a/scripts/quality-gate/desktop-smoke/deterministic.ts +++ b/scripts/quality-gate/desktop-smoke/deterministic.ts @@ -47,6 +47,22 @@ const SMOKE_MODEL_ID = 'desktop-ui-smoke-model' export const DESKTOP_UI_SMOKE_LOCALE = 'en' export const DESKTOP_UI_SMOKE_ALLOW_SELECTOR = 'button[aria-label^="Allow: "]' +/** + * The tab of a session stopped on a decision only the user can make. + * `data-attention` is set on the tab by TabBar; the strip is the only place it + * appears, so the selector is anchored there. + */ +export const DESKTOP_UI_SMOKE_ATTENTION_TAB = '[data-testid="tab-bar"] [data-attention="true"]' + +/** + * True while that tab carries the warning mark and has *not* also been given the + * running dot. A session parked on a permission card is "running" by chatState + * too, and the running dot said the one thing that is not true of it. + */ +export const DESKTOP_UI_SMOKE_ATTENTION_PROBE = + `(() => { const tab = document.querySelector('${DESKTOP_UI_SMOKE_ATTENTION_TAB}'); ` + + `return !!tab && !!tab.querySelector('[role="img"]') && !tab.querySelector('[aria-label="Session running"]') })()` + export function buildDesktopUiSmokeBootstrap(sessionId: string) { return [ `localStorage.setItem('cc-haha-locale', ${JSON.stringify(DESKTOP_UI_SMOKE_LOCALE)})`, @@ -262,7 +278,13 @@ export async function executeDeterministicDesktopSmoke( throw new Error('the tool wrote the fixture file before the permission dialog was answered') } await browserStep(['screenshot', join(artifactDir, 'permission-dialog.png')], { allowFailure: true }) + // While the dialog is open the session's tab has to say the session is + // waiting on the person, not running, and has to stop saying it once the + // request is answered. A marker derived from anything but the open request + // gets one of those two wrong. + await browserStep(['wait', '--fn', DESKTOP_UI_SMOKE_ATTENTION_PROBE], { timeoutMs: 15_000 }) await browserStep(['click', DESKTOP_UI_SMOKE_ALLOW_SELECTOR], { timeoutMs: 20_000 }) + await browserStep(['wait', '--fn', `!document.querySelector('${DESKTOP_UI_SMOKE_ATTENTION_TAB}')`], { timeoutMs: 20_000 }) await pollUntil( async () => existsSync(target) && readFileSync(target, 'utf8') === TARGET_CONTENT, diff --git a/src/constants/businessErrors.ts b/src/constants/businessErrors.ts index 542f1a6f..b26ebca8 100644 --- a/src/constants/businessErrors.ts +++ b/src/constants/businessErrors.ts @@ -12,6 +12,9 @@ export const BUSINESS_ERROR_CODES = { export type BusinessErrorCode = (typeof BUSINESS_ERROR_CODES)[keyof typeof BUSINESS_ERROR_CODES] +// Block types to strip from the single turn a rejection followed. +// REQUEST_TOO_LARGE is deliberately absent: a byte limit says nothing about +// which block was at fault, so normalizeMessagesForAPI resolves it history-wide. export const BUSINESS_ERROR_MEDIA_BLOCK_TYPES: Partial< Record > = { @@ -20,5 +23,14 @@ export const BUSINESS_ERROR_MEDIA_BLOCK_TYPES: Partial< [BUSINESS_ERROR_CODES.PDF_INVALID]: ['document'], [BUSINESS_ERROR_CODES.IMAGE_TOO_LARGE]: ['image'], [BUSINESS_ERROR_CODES.IMAGE_UNSUPPORTED]: ['image'], - [BUSINESS_ERROR_CODES.REQUEST_TOO_LARGE]: ['document', 'image'], } + +/** + * Wording of the request-too-large error before it reported measured sizes. + * Transcripts persisted it (the oldest without a businessErrorCode), so history + * normalization must keep recognizing both the SDK and the interactive variant. + */ +export const LEGACY_REQUEST_TOO_LARGE_ERROR_MESSAGES: readonly string[] = [ + 'Request too large (max 20MB). Try with a smaller file.', + 'Request too large (max 20MB). Double press esc to go back and try with a smaller file.', +] diff --git a/src/constants/prompts.ts b/src/constants/prompts.ts index 23e62bac..a51bd73a 100644 --- a/src/constants/prompts.ts +++ b/src/constants/prompts.ts @@ -114,15 +114,14 @@ export const CLAUDE_CODE_DOCS_MAP_URL = export const SYSTEM_PROMPT_DYNAMIC_BOUNDARY = '__SYSTEM_PROMPT_DYNAMIC_BOUNDARY__' -// @[MODEL LAUNCH]: Update the latest frontier model. -const FRONTIER_MODEL_NAME = 'Claude Opus 4.8' - -// @[MODEL LAUNCH]: Update the model family IDs below to the latest in each tier. +// @[MODEL LAUNCH]: Update the model family IDs below to the latest in each tier, and the +// matching names in the "most recent Claude models" sentence in computeSimpleEnvInfo. +// The Haiku ID is the pinned snapshot, as in the official Claude Code prompt. const LATEST_CLAUDE_MODEL_IDS = { - fable: 'claude-fable-5', - opus: 'claude-opus-4-8', - sonnet: 'claude-sonnet-5', - haiku: 'claude-haiku-4-5', + fable: 'claude-fable-5-1', + opus: 'claude-opus-5-5', + sonnet: 'claude-sonnet-5-5', + haiku: 'claude-haiku-4-5-20251001', } function getHooksSection(): string { @@ -694,13 +693,13 @@ export async function computeSimpleEnvInfo( knowledgeCutoffMessage, process.env.USER_TYPE === 'ant' && isUndercover() ? null - : `The most recent Claude flagship models are Claude Fable 5, Claude Opus 4.8, Claude Sonnet 5, and Claude Haiku 4.5. Model IDs — Fable 5: '${LATEST_CLAUDE_MODEL_IDS.fable}', Opus 4.8: '${LATEST_CLAUDE_MODEL_IDS.opus}', Sonnet 5: '${LATEST_CLAUDE_MODEL_IDS.sonnet}', Haiku 4.5: '${LATEST_CLAUDE_MODEL_IDS.haiku}'. When building AI applications, default to the latest and most capable Claude models.`, + : `The most recent Claude models are the Claude 5 family and Haiku 4.5. Model IDs — Fable 5.1: '${LATEST_CLAUDE_MODEL_IDS.fable}', Opus 5.5: '${LATEST_CLAUDE_MODEL_IDS.opus}', Sonnet 5.5: '${LATEST_CLAUDE_MODEL_IDS.sonnet}', Haiku 4.5: '${LATEST_CLAUDE_MODEL_IDS.haiku}'. When building AI applications, default to the latest and most capable Claude models.`, process.env.USER_TYPE === 'ant' && isUndercover() ? null : `Claude Code is available as a CLI in the terminal, desktop app (Mac/Windows), web app (claude.ai/code), and IDE extensions (VS Code, JetBrains).`, process.env.USER_TYPE === 'ant' && isUndercover() ? null - : `Fast mode for Claude Code uses the same ${FRONTIER_MODEL_NAME} model with faster output. It does NOT switch to a different model. It can be toggled with /fast.`, + : `Fast mode for Claude Code uses Claude Opus with faster output (it does not downgrade to a smaller model). It can be toggled with /fast.`, ].filter(item => item !== null) return [ @@ -713,16 +712,25 @@ export async function computeSimpleEnvInfo( // @[MODEL LAUNCH]: Add a knowledge cutoff date for the new model. function getKnowledgeCutoff(modelId: string): string | null { const canonical = getCanonicalName(modelId) + // Order matters: a point release (`-5-5`, `-5-1`) contains its whole-number id, so it is + // matched first. Values follow Anthropic's model pages and the official Claude Code catalog. if ( + canonical.includes('claude-fable-5-1') || + canonical.includes('claude-opus-5-5') || + canonical.includes('claude-sonnet-5-5') + ) { + return 'June 2026' + } else if (canonical.includes('claude-opus-5')) { + return 'May 2026' + } else if ( canonical.includes('claude-fable-5') || canonical.includes('claude-opus-4-8') || + canonical.includes('claude-opus-4-7') || canonical.includes('claude-sonnet-5') ) { return 'January 2026' } else if (canonical.includes('claude-sonnet-4-6')) { return 'August 2025' - } else if (canonical.includes('claude-opus-4-7')) { - return 'May 2025' } else if (canonical.includes('claude-opus-4-5')) { return 'May 2025' } else if (canonical.includes('claude-haiku-4')) { diff --git a/src/server/__tests__/blank-file-content.test.ts b/src/server/__tests__/blank-file-content.test.ts new file mode 100644 index 00000000..729e028d --- /dev/null +++ b/src/server/__tests__/blank-file-content.test.ts @@ -0,0 +1,34 @@ +import { describe, expect, it } from 'bun:test' +import { isBlankFileContent } from '../services/blankFileContent.js' + +describe('isBlankFileContent', () => { + // Nothing in these holds data a reader could lose. + const blank: Array<[string, string]> = [ + ['an empty string', ''], + ['spaces, tabs and newlines', ' \t\r\n '], + ['a lone UTF-8 BOM', ''], + ['NUL padding (a zero-filled file after a crash mid-write)', '\0'.repeat(713)], + ['a BOM followed by NULs and whitespace', '\0\0 \n\0'], + ] + for (const [name, content] of blank) { + it(`treats ${name} as blank`, () => { + expect(isBlankFileContent(content)).toBe(true) + }) + } + + // Anything with content stays a parse problem to be reported, never a blank file. + const notBlank: Array<[string, string]> = [ + ['valid JSON', '{"tasks": []}'], + ['truncated JSON', '{"tasks": ['], + ['NULs in front of real data', '\0\0{"tasks": []}'], + ['NULs behind real data', '{"tasks": []}\0\0'], + ['a BOM in front of real data', '{"tasks": []}'], + ['a bare JSON scalar', '0'], + ['the JSON literal null', 'null'], + ] + for (const [name, content] of notBlank) { + it(`does not treat ${name} as blank`, () => { + expect(isBlankFileContent(content)).toBe(false) + }) + } +}) diff --git a/src/server/__tests__/conversations.test.ts b/src/server/__tests__/conversations.test.ts index 0b5099b7..5f58c185 100644 --- a/src/server/__tests__/conversations.test.ts +++ b/src/server/__tests__/conversations.test.ts @@ -3570,12 +3570,12 @@ describe('WebSocket Chat Integration', () => { sessionId, options: { providerId: null, - model: 'claude-sonnet-5', + model: 'claude-sonnet-5-5', }, }) await expect(sessionService.getSessionLaunchInfo(sessionId)).resolves.toMatchObject({ runtimeProviderId: null, - runtimeModelId: 'claude-sonnet-5', + runtimeModelId: 'claude-sonnet-5-5', }) } finally { conversationService.startSession = originalStartSession diff --git a/src/server/__tests__/cron-scheduler.test.ts b/src/server/__tests__/cron-scheduler.test.ts index b099b151..8c8f31f9 100644 --- a/src/server/__tests__/cron-scheduler.test.ts +++ b/src/server/__tests__/cron-scheduler.test.ts @@ -581,6 +581,91 @@ describe('CronScheduler', () => { const runs = await scheduler.getRecentRuns(2) expect(runs.length).toBeLessThanOrEqual(2) }) + + // Issue #1400. The run log is read by a second reader, independent of the task + // file's, with the same exposure to a BOM (PowerShell 5.x) or a zero-filled file + // (crash mid-write). History queries used to fail with a bare 500 and the startup + // stale-run cleanup logged an error on every launch. + describe('run log written by another tool', () => { + const BOM = Buffer.from([0xef, 0xbb, 0xbf]) + const logPath = () => path.join(tmpDir, 'scheduled_tasks_log.json') + const finishedRun: TaskRun = { + id: 'run-1', + taskId: 'task-a', + taskName: 'Task A', + startedAt: '2026-01-01T00:00:00.000Z', + completedAt: '2026-01-01T00:00:01.000Z', + status: 'completed', + prompt: 'test prompt', + exitCode: 0, + durationMs: 1000, + } + const staleRun: TaskRun = { + id: 'run-stale', + taskId: 'task-a', + taskName: 'Task A', + startedAt: '2020-01-01T00:00:00.000Z', + status: 'running', + prompt: 'left behind by a crashed process', + } + const withBom = (runs: TaskRun[]) => + Buffer.concat([BOM, Buffer.from(JSON.stringify({ runs }))]) + + async function waitUntil(check: () => Promise, timeoutMs = 3000) { + const deadline = Date.now() + timeoutMs + while (Date.now() < deadline) { + if (await check().catch(() => false)) return + await Bun.sleep(10) + } + throw new Error('timed out waiting for the run log to be rewritten') + } + + it('reads run history from a log that starts with a UTF-8 BOM', async () => { + await fs.writeFile(logPath(), withBom([finishedRun])) + + expect((await scheduler.getRecentRuns()).map((run) => run.id)).toEqual(['run-1']) + expect((await scheduler.getTaskRuns('task-a')).map((run) => run.id)).toEqual(['run-1']) + }) + + const blankLogs: Array<[string, Buffer]> = [ + ['a zero-filled log', Buffer.alloc(200, 0)], + ['an empty log', Buffer.alloc(0)], + ['a BOM-only log', BOM], + ] + for (const [name, bytes] of blankLogs) { + it(`reports no runs for ${name}`, async () => { + await fs.writeFile(logPath(), bytes) + + expect(await scheduler.getRecentRuns()).toEqual([]) + expect(await scheduler.getTaskRuns('task-a')).toEqual([]) + }) + } + + it('still fails on a log with real content it cannot parse', async () => { + await fs.writeFile(logPath(), Buffer.from('{"runs": [{"id": "run-1"')) + + await expect(scheduler.getRecentRuns()).rejects.toThrow() + }) + + it('completes the startup stale-run cleanup on a BOM log without logging an error', async () => { + await fs.writeFile(logPath(), withBom([staleRun])) + const errors = spyOn(console, 'error').mockImplementation(() => {}) + let logged: string[] = [] + + try { + scheduler.start() + await waitUntil(async () => { + const rewritten = JSON.parse(await fs.readFile(logPath(), 'utf-8')) as { runs: TaskRun[] } + return rewritten.runs[0]?.status === 'failed' + }) + } finally { + logged = errors.mock.calls.map((args) => args.map(String).join(' ')) + errors.mockRestore() + } + + expect(logged.filter((line) => line.includes('cleaning up stale runs'))).toEqual([]) + }) + }) }) // ─── Execution log trimming ──────────────────────────────────────────────── diff --git a/src/server/__tests__/e2e/business-flow.test.ts b/src/server/__tests__/e2e/business-flow.test.ts index 2ddffeef..58645edb 100644 --- a/src/server/__tests__/e2e/business-flow.test.ts +++ b/src/server/__tests__/e2e/business-flow.test.ts @@ -374,20 +374,21 @@ describe('Business Flow: Models & Effort', () => { it('should return available fallback models', async () => { const { data } = await api('GET', '/api/models') - expect(data.models.length).toBe(7) + expect(data.models.length).toBe(8) const names = data.models.map((m: any) => m.name) expect(names).toContain('Fable 5.1') expect(names).toContain('Fable 5') expect(names).toContain('Opus 5.5') expect(names).toContain('Opus 5') expect(names).toContain('Opus 4.8') + expect(names).toContain('Sonnet 5.5') expect(names).toContain('Sonnet 5') expect(names).toContain('Haiku 4.5') }) it('should default to Opus model', async () => { const { data } = await api('GET', '/api/models/current') - expect(data.model.id).toBe('claude-opus-5') + expect(data.model.id).toBe('claude-opus-5-5') }) it('should switch to Opus 4.8', async () => { diff --git a/src/server/__tests__/e2e/full-flow.test.ts b/src/server/__tests__/e2e/full-flow.test.ts index 7c7ebd64..2a5268f0 100644 --- a/src/server/__tests__/e2e/full-flow.test.ts +++ b/src/server/__tests__/e2e/full-flow.test.ts @@ -216,7 +216,7 @@ describe('E2E: Full Flow', () => { it('should list available models', async () => { const { data } = await api('GET', '/api/models') - expect(data.models.length).toBe(7) + expect(data.models.length).toBe(8) expect(data.models[0].name).toBe('Fable 5.1') }) diff --git a/src/server/__tests__/scheduled-tasks.test.ts b/src/server/__tests__/scheduled-tasks.test.ts index 15d9287e..b3616651 100644 --- a/src/server/__tests__/scheduled-tasks.test.ts +++ b/src/server/__tests__/scheduled-tasks.test.ts @@ -287,6 +287,83 @@ describe('CronService', () => { const tasks = await service.listTasks() expect(tasks).toHaveLength(0) }) + + // Issue #1400. PowerShell 5.x `Set-Content -Encoding UTF8` prepends a BOM, and a + // crash mid-write on NTFS leaves a file at full size but zero-filled. Neither is + // damage to the user's tasks, yet each used to fail every read — and because every + // mutation reads first, the whole scheduled-task feature went down with it. + describe('task file written by another tool', () => { + const BOM = Buffer.from([0xef, 0xbb, 0xbf]) + const validFile = JSON.stringify({ + tasks: [{ id: 'abc12345', cron: '0 9 * * *', prompt: 'kept', createdAt: 1 }], + }) + const tasksPath = () => path.join(tmpDir, 'scheduled_tasks.json') + + it('reads a task file that starts with a UTF-8 BOM', async () => { + await fs.writeFile(tasksPath(), Buffer.concat([BOM, Buffer.from(validFile)])) + + const tasks = await service.listTasks() + + expect(tasks.map((task) => task.id)).toEqual(['abc12345']) + }) + + it('keeps the tasks of a BOM file when a new one is added', async () => { + await fs.writeFile(tasksPath(), Buffer.concat([BOM, Buffer.from(validFile)])) + + const added = await service.createTask({ cron: '* * * * *', prompt: 'new' }) + + const tasks = await service.listTasks() + expect(tasks.map((task) => task.id).sort()).toEqual(['abc12345', added.id].sort()) + // Written back as plain JSON: no reader has to know about the BOM again. + expect(JSON.parse(await fs.readFile(tasksPath(), 'utf-8')).tasks).toHaveLength(2) + }) + + const blankFiles: Array<[string, Buffer]> = [ + ['a zero-filled file', Buffer.alloc(713, 0)], + ['an empty file', Buffer.alloc(0)], + ['a whitespace-only file', Buffer.from(' \r\n\t\n')], + ['a BOM-only file', BOM], + ] + for (const [name, bytes] of blankFiles) { + it(`treats ${name} as having no tasks`, async () => { + await fs.writeFile(tasksPath(), bytes) + + expect(await service.listTasks()).toEqual([]) + }) + } + + it('lets a task be created over a zero-filled file and heals it', async () => { + await fs.writeFile(tasksPath(), Buffer.alloc(713, 0)) + + const created = await service.createTask({ cron: '* * * * *', prompt: 'fresh' }) + + expect((await service.listTasks()).map((task) => task.id)).toEqual([created.id]) + expect(JSON.parse(await fs.readFile(tasksPath(), 'utf-8')).tasks).toHaveLength(1) + }) + + // The other half of the contract: blank content holds nothing to lose, real + // content that will not parse might be recoverable. It must keep failing loudly + // and must not be overwritten by the next create. + it('still refuses content it cannot parse and leaves the file untouched', async () => { + const truncated = Buffer.from(validFile.slice(0, 30)) + await fs.writeFile(tasksPath(), truncated) + + await expect(service.listTasks()).rejects.toThrow('Failed to read scheduled tasks') + await expect( + service.createTask({ cron: '* * * * *', prompt: 'must not overwrite' }), + ).rejects.toThrow('Failed to read scheduled tasks') + + expect((await fs.readFile(tasksPath())).equals(truncated)).toBe(true) + }) + + it('does not mistake NUL bytes around real content for a blank file', async () => { + const padded = Buffer.concat([Buffer.alloc(8, 0), Buffer.from('{"tasks": [')]) + await fs.writeFile(tasksPath(), padded) + + await expect(service.listTasks()).rejects.toThrow('Failed to read scheduled tasks') + expect((await fs.readFile(tasksPath())).equals(padded)).toBe(true) + }) + }) }) // ─── SearchService tests ──────────────────────────────────────────────────── diff --git a/src/server/__tests__/sessions.test.ts b/src/server/__tests__/sessions.test.ts index a8637fef..f4131051 100644 --- a/src/server/__tests__/sessions.test.ts +++ b/src/server/__tests__/sessions.test.ts @@ -2223,7 +2223,7 @@ describe('SessionService', () => { ]) }) - it('should hide task-notification turns and their automatic responses from history', async () => { + it('should hide the task-notification turn itself but keep everything the assistant does after it', async () => { const sessionId = 'aaaaaaaa-bbbb-cccc-dddd-eeeeeeeeeeee' const firstUserId = crypto.randomUUID() const firstAssistantId = crypto.randomUUID() @@ -2253,7 +2253,7 @@ describe('SessionService', () => { parentUuid: firstAssistantId, }, { - ...makeAssistantEntry('旧后台任务通知,无需处理', taskNotificationId), + ...makeAssistantEntry('后台命令跑完了,我来重启服务', taskNotificationId), uuid: taskAssistantId, }, { @@ -2279,7 +2279,7 @@ describe('SessionService', () => { timestamp: '2026-01-01T00:03:00.000Z', }, { - ...makeAssistantEntry('后台任务触发的工具调用完成', taskToolResultId), + ...makeAssistantEntry('服务已重启', taskToolResultId), uuid: taskAfterToolId, }, { @@ -2298,13 +2298,17 @@ describe('SessionService', () => { expect(messages.map((message) => message.id)).toEqual([ firstUserId, firstAssistantId, + taskAssistantId, + taskToolUseMessageId, + taskToolResultId, + taskAfterToolId, realFollowUpId, realAssistantId, ]) expect(JSON.stringify(messages)).not.toContain('') - expect(JSON.stringify(messages)).not.toContain('旧后台任务通知') - expect(JSON.stringify(messages)).not.toContain('server restarted') - expect(JSON.stringify(messages)).not.toContain('后台任务触发的工具调用完成') + expect(JSON.stringify(messages)).toContain('后台命令跑完了,我来重启服务') + expect(JSON.stringify(messages)).toContain('server restarted') + expect(JSON.stringify(messages)).toContain('服务已重启') expect(taskNotifications).toEqual([ { taskId: 'bg-1', @@ -2316,6 +2320,77 @@ describe('SessionService', () => { ]) }) + it('keeps the model reply to a background-task notification and hides only the notification itself (#1389)', async () => { + // Shapes copied from a real desktop session: a background shell finished, the CLI queued a + // user turn, and DeepSeek answered it with an (empty) thinking block plus text. + const sessionId = 'bbbbbbbb-bbbb-4ccc-8ddd-eeeeeeeeeeee' + const userId = crypto.randomUUID() + const dispatchedId = crypto.randomUUID() + const notificationId = crypto.randomUUID() + const thinkingId = crypto.randomUUID() + const replyId = crypto.randomUUID() + const secondNotificationId = crypto.randomUUID() + const secondReplyId = crypto.randomUUID() + const notification = (taskId: string) => [ + '', + `${taskId}`, + `call_${taskId}`, + `/tmp/${taskId}.output`, + 'completed', + `Background command "sleep then write ${taskId}" completed (exit code 0)`, + '', + ].join('\n') + + await writeSessionFile('-tmp-notification-reply', sessionId, [ + makeSnapshotEntry(), + { ...makeUserEntry('start two background commands', userId), parentUuid: null }, + { ...makeAssistantEntry('dispatched', userId), uuid: dispatchedId }, + { + type: 'cc-haha-task-notification', + isMeta: true, + taskNotification: { + taskId: 'alpha', + toolUseId: 'call_alpha', + status: 'completed', + summary: 'Background command "sleep then write alpha" completed (exit code 0)', + timestamp: '2026-01-01T00:03:00.000Z', + }, + timestamp: '2026-01-01T00:03:00.000Z', + }, + { ...makeUserEntry(notification('alpha'), notificationId), parentUuid: dispatchedId }, + { + ...makeAssistantEntry('', notificationId), + uuid: thinkingId, + message: { + role: 'assistant', + model: 'deepseek-flash', + id: 'msg_thinking_alpha', + content: [{ type: 'thinking', thinking: '', signature: 'sig' }], + }, + }, + { ...makeAssistantEntry('alpha finished: ALPHA_DONE', thinkingId), uuid: replyId }, + { ...makeUserEntry(notification('bravo'), secondNotificationId), parentUuid: replyId }, + { ...makeAssistantEntry('bravo finished: BRAVO_DONE', secondNotificationId), uuid: secondReplyId }, + ]) + + const messages = await service.getSessionMessages(sessionId) + const rendered = JSON.stringify(messages) + + expect(rendered).not.toContain('') + expect(rendered).toContain('alpha finished: ALPHA_DONE') + expect(rendered).toContain('bravo finished: BRAVO_DONE') + expect(messages.map((message) => message.id)).toEqual(expect.arrayContaining([ + userId, + dispatchedId, + replyId, + secondReplyId, + ])) + // The cards for the finished tasks still come from the notification data, not from the hidden message. + expect((await service.getSessionTaskNotifications(sessionId)).map((item) => item.taskId)).toEqual( + expect.arrayContaining(['alpha', 'bravo']), + ) + }) + it('uses bounded locators for snapshots and task notifications with safe fallback', async () => { const sessionId = 'aaaaaaaa-bbbb-4ccc-8ddd-eeeeeeeeeeee' const projectDir = '-tmp-locator-consumers' @@ -2416,7 +2491,7 @@ describe('SessionService', () => { serveWrongEmptyFingerprint = false const callsBeforeFullHistory = locatorCalls - expect(await indexedService.getSessionMessages(sessionId)).toHaveLength(0) + expect(await indexedService.getSessionMessages(sessionId)).toHaveLength(1) expect(locatorCalls).toBe(callsBeforeFullHistory) mode = 'shadow' @@ -4304,7 +4379,7 @@ describe('Sessions API', () => { '\nbg-1\ntoolu_bg\nfailed\nBackground command failed & stopped\nStack trace & failed assertion\nC:\\Temp\\bg.output\n', crypto.randomUUID(), ), - makeAssistantEntry('internal task response'), + makeAssistantEntry('the background command failed, investigating'), ]) const res = await fetch(`${baseUrl}/api/sessions/${sessionId}/messages`) @@ -4314,11 +4389,15 @@ describe('Sessions API', () => { messages: unknown[] taskNotifications: unknown[] } - expect(body.messages).toHaveLength(2) + expect(body.messages).toHaveLength(3) expect(body.messages[1]).toMatchObject({ type: 'assistant', usage: { input_tokens: 1234, output_tokens: 56 }, }) + expect(body.messages[2]).toMatchObject({ + type: 'assistant', + content: [{ type: 'text', text: 'the background command failed, investigating' }], + }) expect(JSON.stringify(body.messages)).not.toContain('') expect(body.taskNotifications).toEqual([ { @@ -4403,7 +4482,7 @@ describe('Sessions API', () => { }, { type: 'assistant', - message: { role: 'assistant', content: 'Internal notification response' }, + message: { role: 'assistant', content: 'Child shell stopped, continuing the review' }, uuid: crypto.randomUUID(), timestamp: '2026-01-01T00:00:07.000Z', }, @@ -4430,9 +4509,9 @@ describe('Sessions API', () => { prompt: 'Read session routes', source: 'subagent-jsonl', }) - expect(body.messages).toHaveLength(2) + expect(body.messages).toHaveLength(3) expect(JSON.stringify(body.messages)).not.toContain('') - expect(JSON.stringify(body.messages)).not.toContain('Internal notification response') + expect(JSON.stringify(body.messages)).toContain('Child shell stopped, continuing the review') expect(body.taskNotifications).toEqual([{ taskId: 'child-shell-task', toolUseId: 'child-shell.call', diff --git a/src/server/__tests__/settings.test.ts b/src/server/__tests__/settings.test.ts index 0ee5930b..5f8b466c 100644 --- a/src/server/__tests__/settings.test.ts +++ b/src/server/__tests__/settings.test.ts @@ -968,6 +968,14 @@ describe('Models API', () => { defaultReasoningEffort: 'high', supportedReasoningEfforts: ['low', 'medium', 'high', 'xhigh', 'max'], }, + { + id: 'claude-sonnet-5-5', + name: 'Sonnet 5.5', + description: 'Best combination of speed and intelligence', + context: '1m', + defaultReasoningEffort: 'medium', + supportedReasoningEfforts: ['low', 'medium', 'high', 'xhigh', 'max'], + }, { id: 'claude-sonnet-5', name: 'Sonnet 5', @@ -1177,7 +1185,8 @@ describe('Models API', () => { expect(res.status).toBe(200) const body = await res.json() - expect(body.model.id).toBe('claude-opus-5') + // The CLI launches without an explicit model here and its `opus` alias is Opus 5.5. + expect(body.model).toMatchObject({ id: 'claude-opus-5-5', name: 'Opus 5.5' }) }) it('GET /api/models/current should replace the legacy opus[1m] default with the Claude OAuth Pro default', async () => { @@ -1200,8 +1209,9 @@ describe('Models API', () => { const currentBody = await currentResponse.json() expect(currentBody.model).toMatchObject({ - id: 'claude-sonnet-5', - name: 'Sonnet 5', + id: 'claude-sonnet-5-5', + name: 'Sonnet 5.5', + defaultReasoningEffort: 'medium', }) const listRequest = makeRequest('GET', '/api/models') @@ -1217,6 +1227,7 @@ describe('Models API', () => { 'claude-opus-5-5', 'claude-opus-5', 'claude-opus-4-8', + 'claude-sonnet-5-5', 'claude-sonnet-5', 'claude-haiku-4-5', ]) @@ -1574,16 +1585,17 @@ describe('Model Options', () => { beforeEach(setup) afterEach(teardown) - it('defaults Anthropic API users to Opus 5 and exposes the current official options once', () => { + it('defaults Anthropic API users to Opus 5.5 and exposes the current official options once', () => { process.env.ANTHROPIC_API_KEY = 'test-api-key' - expect(getDefaultMainLoopModelSetting()).toBe('claude-opus-5') + expect(getDefaultMainLoopModelSetting()).toBe('claude-opus-5-5') const options = getModelOptions() const values = options.map(option => option.value) - expect(options[0]?.description).toContain('Opus 5') - expect(options[0]?.description).toContain('$5') + expect(options[0]?.description).toContain('Opus 5.5') + // Opus 5.5 is $4/$20, not the $5/$25 of the Opus 5 it replaces as the default. + expect(options[0]?.description).toContain('$4/$20 per Mtok') expect(values).toContain('fable') expect(values).toContain('sonnet') expect(values).not.toContain('opus') @@ -1623,10 +1635,10 @@ describe('Model Options', () => { it('labels extended-context options with current first-party and conservative third-party names', () => { process.env.ANTHROPIC_API_KEY = 'test-api-key' - expect(getSonnet46_1MOption().description).toContain('Sonnet 5') - expect(getOpus46_1MOption().description).toContain('Opus 5') - expect(getMaxSonnet46_1MOption().description).toContain('Sonnet 5') - expect(getMaxOpus46_1MOption().description).toContain('Opus 5') + expect(getSonnet46_1MOption().description).toContain('Sonnet 5.5') + expect(getOpus46_1MOption().description).toContain('Opus 5.5') + expect(getMaxSonnet46_1MOption().description).toContain('Sonnet 5.5') + expect(getMaxOpus46_1MOption().description).toContain('Opus 5.5') process.env.ANTHROPIC_BASE_URL = 'https://api.deepseek.com/anthropic' diff --git a/src/server/api/models.ts b/src/server/api/models.ts index aef44ff2..b9af8ae5 100644 --- a/src/server/api/models.ts +++ b/src/server/api/models.ts @@ -89,6 +89,14 @@ const DEFAULT_MODELS = [ defaultReasoningEffort: 'high', supportedReasoningEfforts: ['low', 'medium', 'high', 'xhigh', 'max'], }, + { + id: 'claude-sonnet-5-5', + name: 'Sonnet 5.5', + description: 'Best combination of speed and intelligence', + context: '1m', + defaultReasoningEffort: 'medium', + supportedReasoningEfforts: ['low', 'medium', 'high', 'xhigh', 'max'], + }, { id: 'claude-sonnet-5', name: 'Sonnet 5', @@ -107,7 +115,10 @@ const DEFAULT_MODELS = [ const EFFORT_LEVELS = MODEL_REASONING_EFFORTS -const DEFAULT_MODEL = 'claude-opus-5' +// Standalone sessions (no provider, no Claude OAuth) start the CLI without an explicit model, so the +// CLI's own default runs. Keep this in step with what the `opus` alias resolves to there +// (getDefaultOpusModel) so the UI names the model that actually runs. +const DEFAULT_MODEL = 'claude-opus-5-5' const DEFAULT_EFFORT = 'max' const settingsService = new SettingsService() diff --git a/src/server/services/blankFileContent.ts b/src/server/services/blankFileContent.ts new file mode 100644 index 00000000..5dd8f4e9 --- /dev/null +++ b/src/server/services/blankFileContent.ts @@ -0,0 +1,13 @@ +/** + * Whether a state file's text carries no data at all: empty, whitespace (a BOM counts — + * U+FEFF is whitespace to `\s`), or NUL padding. + * + * A crash or power loss during a write can leave a file on NTFS at its full size but + * zero-filled, and a tool that wrote nothing but a BOM leaves just that. A reader loses + * nothing by treating such a file as never written — unlike content that merely fails to + * parse, which may still be recoverable and must not be silently replaced by the next + * write. NULs sitting next to real data are therefore not blank. + */ +export function isBlankFileContent(content: string): boolean { + return /^[\s\0]*$/.test(content) +} diff --git a/src/server/services/claudeOfficialRuntime.test.ts b/src/server/services/claudeOfficialRuntime.test.ts index dfd29b4f..e86ecc76 100644 --- a/src/server/services/claudeOfficialRuntime.test.ts +++ b/src/server/services/claudeOfficialRuntime.test.ts @@ -23,10 +23,33 @@ describe('Claude official runtime model selection', () => { }) } - test('selects Opus 5.5 for Max and preserves the Sonnet default for other subscriptions', () => { + function useProSubscription() { + tokenSpy.mockResolvedValue({ + accessToken: 'fake-access-token', + refreshToken: null, + expiresAt: null, + scopes: [], + subscriptionType: 'pro', + }) + } + + test('selects Opus 5.5 for Max and Sonnet 5.5 for other subscriptions', () => { expect(getClaudeOfficialDefaultModelId('max')).toBe('claude-opus-5-5') - expect(getClaudeOfficialDefaultModelId('pro')).toBe('claude-sonnet-5') - expect(getClaudeOfficialDefaultModelId(null)).toBe('claude-sonnet-5') + expect(getClaudeOfficialDefaultModelId('pro')).toBe('claude-sonnet-5-5') + expect(getClaudeOfficialDefaultModelId(null)).toBe('claude-sonnet-5-5') + }) + + test('resolves Sonnet aliases and the unset default to Sonnet 5.5', async () => { + useProSubscription() + for (const model of [undefined, 'sonnet', 'sonnet[1m]', 'SONNET:1m', 'claude-sonnet-5-5[1m]']) { + expect(await resolveClaudeOfficialRuntimeModel(model)).toBe('claude-sonnet-5-5') + } + }) + + test('preserves an explicitly selected Sonnet 5 instead of rewriting it to Sonnet 5.5', async () => { + useProSubscription() + expect(await resolveClaudeOfficialRuntimeModel('claude-sonnet-5')).toBe('claude-sonnet-5') + expect(await resolveClaudeOfficialRuntimeModel('claude-sonnet-5[1m]')).toBe('claude-sonnet-5') }) test('resolves Opus aliases and legacy defaults to Opus 5.5', async () => { diff --git a/src/server/services/claudeOfficialRuntime.ts b/src/server/services/claudeOfficialRuntime.ts index 4cff2724..3cf9f7e7 100644 --- a/src/server/services/claudeOfficialRuntime.ts +++ b/src/server/services/claudeOfficialRuntime.ts @@ -3,7 +3,7 @@ import type { ModelMapping } from '../types/provider.js' import { hahaOAuthService } from './hahaOAuthService.js' export const CLAUDE_OFFICIAL_OPUS_MODEL_ID = 'claude-opus-5-5' -export const CLAUDE_OFFICIAL_SONNET_MODEL_ID = 'claude-sonnet-5' +export const CLAUDE_OFFICIAL_SONNET_MODEL_ID = 'claude-sonnet-5-5' export const CLAUDE_OFFICIAL_DEFAULT_MODELS: ModelMapping = { main: CLAUDE_OFFICIAL_OPUS_MODEL_ID, diff --git a/src/server/services/cronScheduler.ts b/src/server/services/cronScheduler.ts index 42cb5398..1e930763 100644 --- a/src/server/services/cronScheduler.ts +++ b/src/server/services/cronScheduler.ts @@ -12,11 +12,13 @@ import { existsSync, readFileSync, realpathSync, statSync } from 'node:fs' import * as path from 'path' import * as os from 'os' import * as crypto from 'crypto' +import { isBlankFileContent } from './blankFileContent.js' import { CronService, type CronTask } from './cronService.js' import { SessionService } from './sessionService.js' import { sendTaskNotification } from './notificationService.js' import { ProviderService } from './providerService.js' import { SettingsService } from './settingsService.js' +import { stripBOM } from '../../utils/jsonRead.js' import { isProviderManagedEnvVar } from '../../utils/managedEnvConstants.js' import { buildClaudeCliArgs, @@ -254,8 +256,14 @@ function captureRunsFileMutationTarget(): RunsFileMutationTarget { } } +/** + * Same tolerance as the task file (CronService.readTasksFile): a BOM is stripped and + * a blank file — empty, whitespace or NUL padding — means no runs. Content that has + * data but will not parse still throws, so a later write cannot replace it unseen. + */ function parseRunsFile(raw: string): RunsFile { - const parsed = JSON.parse(raw) as RunsFile + if (isBlankFileContent(raw)) return { runs: [] } + const parsed = JSON.parse(stripBOM(raw)) as RunsFile return Array.isArray(parsed.runs) ? parsed : { runs: [] } } diff --git a/src/server/services/cronService.ts b/src/server/services/cronService.ts index 08bcbf38..c7b95bb7 100644 --- a/src/server/services/cronService.ts +++ b/src/server/services/cronService.ts @@ -9,7 +9,9 @@ import * as fs from 'fs/promises' import * as path from 'path' import * as os from 'os' import * as crypto from 'crypto' +import { stripBOM } from '../../utils/jsonRead.js' import { ApiError } from '../middleware/errorHandler.js' +import { isBlankFileContent } from './blankFileContent.js' export type TaskNotificationConfig = { enabled: boolean @@ -159,11 +161,18 @@ export class CronService { // 内部: 文件读写 // --------------------------------------------------------------------------- - /** 读取任务 JSON 文件。文件不存在时返回空列表。 */ + /** + * 读取任务 JSON 文件。文件不存在、或内容为空(空白 / 全 NUL,见 isBlankFileContent) + * 时返回空列表;带 UTF-8 BOM 的文件(PowerShell 5.x 默认写入)照常解析。 + * 有实际内容却无法解析时仍然抛错:此时文件可能还能人工恢复,绝不能让后续写入悄悄覆盖它。 + */ private async readTasksFile(): Promise { try { const raw = await fs.readFile(this.getTasksFilePath(), 'utf-8') - const parsed = JSON.parse(raw) as TasksFile + if (isBlankFileContent(raw)) { + return { tasks: [] } + } + const parsed = JSON.parse(stripBOM(raw)) as TasksFile // 兼容异常格式 if (!Array.isArray(parsed.tasks)) { return { tasks: [] } diff --git a/src/server/services/localIndex/coordinator.test.ts b/src/server/services/localIndex/coordinator.test.ts index 4edff93f..142a9058 100644 --- a/src/server/services/localIndex/coordinator.test.ts +++ b/src/server/services/localIndex/coordinator.test.ts @@ -1,4 +1,5 @@ import { afterEach, describe, expect, it } from 'bun:test' +import { Database } from 'bun:sqlite' import { getEventListeners } from 'node:events' import { appendFile, chmod, mkdir, mkdtemp, readdir, rm, stat, symlink, writeFile } from 'node:fs/promises' import { tmpdir } from 'node:os' @@ -777,6 +778,100 @@ describe('local index coordinator', () => { } }) + it('replaces persisted per-model dollars with the current rates after a parser upgrade', async () => { + // Version 9 corrected the Sonnet 5 / Sonnet 5.5 / Opus 5.5 rates. Dollars are persisted per model, + // so without a parser bump every already-indexed transcript would keep its old price forever. + expect(SESSION_SUMMARY_PARSER_VERSION).toBeGreaterThanOrEqual(9) + + const root = await createTempDir('coordinator-activity-cost-upgrade') + const configDir = join(root, 'config') + const databasePath = join(configDir, 'cc-haha', 'db', 'index-v1.sqlite') + const transcriptPath = join(configDir, 'projects', '-repo', 'cost-upgrade.jsonl') + await mkdir(dirname(transcriptPath), { recursive: true }) + await writeFile(transcriptPath, [ + { + type: 'user', + message: { role: 'user', content: 'Price this' }, + timestamp: '2026-01-01T00:00:00.000Z', + }, + { + type: 'assistant', + requestId: 'req_cost_upgrade', + message: { + id: 'msg_cost_upgrade', + role: 'assistant', + model: 'claude-sonnet-5', + content: [{ type: 'text', text: 'Priced' }], + usage: { input_tokens: 1_000_000, output_tokens: 1_000_000 }, + }, + timestamp: '2026-01-01T00:00:01.000Z', + }, + ].map(entry => JSON.stringify(entry)).join('\n') + '\n') + + const createIdleWatcher = (): ReconciliationWatcher => ({ + async start() {}, + async stop() {}, + queueTranscriptPath() {}, + queueFullSweep() {}, + getMetrics: () => ({ + queuedPaths: 0, + maxBatchSize: 0, + yielded: 0, + fullSweeps: 0, + watchFailures: 0, + }), + }) + const previousCoordinator = createLocalIndexCoordinator({ + resolveMode: () => ({ mode: 'on', warningCode: null }), + resolveScope: () => configDir, + resolveDatabasePath: () => databasePath, + createProjector: options => createSessionProjector({ + ...options, + parserVersion: SESSION_SUMMARY_PARSER_VERSION - 1, + }), + createWatcher: createIdleWatcher, + }) + await previousCoordinator.start() + await waitFor(() => previousCoordinator.isActivityScopeReady()) + await previousCoordinator.stop() + + // Leave behind what the previous release persisted: Sonnet 5 at the old $3/$15 rate. + const staleDatabase = new Database(databasePath) + try { + staleDatabase.run( + "UPDATE activity_daily_models SET cost_usd = 18 WHERE model = 'claude-sonnet-5'", + ) + expect(staleDatabase.query<{ cost_usd: number }, []>( + "SELECT cost_usd FROM activity_daily_models WHERE model = 'claude-sonnet-5'", + ).get()?.cost_usd).toBe(18) + } finally { + staleDatabase.close() + } + + let runScheduledDiscovery: (() => void) | undefined + const upgradedCoordinator = createLocalIndexCoordinator({ + resolveMode: () => ({ mode: 'on', warningCode: null }), + resolveScope: () => configDir, + resolveDatabasePath: () => databasePath, + schedule: operation => { runScheduledDiscovery = operation }, + createWatcher: createIdleWatcher, + }) + + try { + await upgradedCoordinator.start() + // The stale dollars are withheld until the rebuild has replaced them. + expect(upgradedCoordinator.getActivityStats('all')).toBeNull() + + runScheduledDiscovery?.() + await waitFor(() => upgradedCoordinator.isActivityScopeReady()) + // 1M input at $2 + 1M output at $10. + expect(upgradedCoordinator.getActivityStats('all')?.modelUsage['claude-sonnet-5']?.costUSD) + .toBeCloseTo(12, 6) + } finally { + await upgradedCoordinator.stop() + } + }) + it('keeps retained rows degraded after watch failure and ignores late batches after stop', async () => { const index = createFakeIndex([candidate(1)]) let watcherOptions!: ReconciliationWatcherOptions diff --git a/src/server/services/localIndex/sessionProjector.ts b/src/server/services/localIndex/sessionProjector.ts index c9848816..df5f77c5 100644 --- a/src/server/services/localIndex/sessionProjector.ts +++ b/src/server/services/localIndex/sessionProjector.ts @@ -38,7 +38,9 @@ import type { // 6: protocol enforcement was removed; rebuild v5 summaries without protocol restrictions. // 7: independent desktop team workers remain addressable but leave sidebar listings. // 8: complete runtime selections clear the previous effort when no override is saved. -export const SESSION_SUMMARY_PARSER_VERSION = 8 +// 9: usage cost rates were corrected (Sonnet 5, Sonnet 5.5, Opus 5.5, fast mode); rebuild the +// persisted per-model dollars. +export const SESSION_SUMMARY_PARSER_VERSION = 9 export type SessionSourceCandidate = { path: string diff --git a/src/server/services/sessionHistoryContext.test.ts b/src/server/services/sessionHistoryContext.test.ts index e9da132c..6188cedb 100644 --- a/src/server/services/sessionHistoryContext.test.ts +++ b/src/server/services/sessionHistoryContext.test.ts @@ -10,38 +10,48 @@ beforeEach(async () => { directory = await mkdtemp(join(tmpdir(), 'history-conte afterEach(async () => { await rm(directory, { recursive: true, force: true }) }) const row = (id: string, fields: object = {}) => JSON.stringify({ uuid: id, ...fields }) const version = async (filePath = file) => { const info = await stat(filePath, { bigint: true }); return `${info.dev}:${info.ino}:${info.size}:${info.mtimeNs}` } -const classify = (entry: Record) => ({ notification: entry.notification === true, reset: entry.reset === true, agentToolId: typeof entry.agent === 'string' ? entry.agent : undefined }) +const agentToolId = (entry: Record) => typeof entry.agent === 'string' ? entry.agent : undefined +const messageAgentToolId = (entry: Record) => { + const content = (entry.message as { content?: unknown } | undefined)?.content + const blocks = Array.isArray(content) ? content : [] + return blocks.find(block => block.type === 'tool_use' && ['Agent', 'Task'].includes(block.name))?.id as string | undefined +} -test('subagent visibility keeps unowned sidechains without bypassing cached notification suppression', async () => { +test('root reads hide unowned sidechains while a child transcript keeps them, from one shared scan', async () => { const entries = [ - row('user', { isSidechain: true, reset: true }), - row('notice', { isSidechain: true, notification: true }), - row('hidden', { isSidechain: true, parentUuid: 'notice' }), - row('next', { isSidechain: true, reset: true }), + row('owner', { agent: 'agent-tool' }), + row('owned', { isSidechain: true, parentUuid: 'owner' }), + row('orphan', { isSidechain: true }), + row('root'), ] await writeFile(file, entries.join('\n') + '\n') const offsets = entries.map((_, index) => Buffer.byteLength(entries.slice(0, index).map(value => value + '\n').join(''))) - const options = { filePath: file, sourceVersion: await version(), offsets, classify } + const options = { filePath: file, sourceVersion: await version(), offsets, agentToolId } const root = await readHistoryContexts(options) - expect([...root.contexts.values()].map(context => context.suppressed)).toEqual([true, true, true, true]) + expect([...root.contexts.values()]).toEqual([ + { owner: undefined, hidden: false }, + { owner: 'agent-tool', hidden: false }, + { owner: undefined, hidden: true }, + { owner: undefined, hidden: false }, + ]) const child = await readHistoryContexts({ ...options, includeUnownedSidechains: true }) expect(child.scannedBytes).toBe(0) - expect([...child.contexts.values()].map(context => context.suppressed)).toEqual([false, true, true, false]) - expect([...((await readHistoryContexts(options)).contexts.values())].every(context => context.suppressed)).toBe(true) + expect([...child.contexts.values()].map(context => context.hidden)).toEqual([false, false, false, false]) + expect([...((await readHistoryContexts(options)).contexts.values())].map(context => context.hidden)).toEqual([false, false, true, false]) }) test('indexes EOF records and resumes at their boundary when a newline and new records are appended', async () => { - const first = row('notification', { notification: true }) + '\n' - const partial = row('user', { reset: true }) + const first = row('owner', { agent: 'agent-tool' }) + '\n' + const partial = row('child', { isSidechain: true, parentUuid: 'owner' }) await writeFile(file, first + partial) - const before = await readHistoryContexts({ filePath: file, sourceVersion: await version(), offsets: [0, Buffer.byteLength(first)], classify }) - expect(before.contexts.get(0)?.suppressed).toBe(true) - expect(before.contexts.get(Buffer.byteLength(first))?.suppressed).toBe(false) + const before = await readHistoryContexts({ filePath: file, sourceVersion: await version(), offsets: [0, Buffer.byteLength(first)], agentToolId }) + expect(before.contexts.get(0)).toEqual({ owner: undefined, hidden: false }) + expect(before.contexts.get(Buffer.byteLength(first))).toEqual({ owner: 'agent-tool', hidden: false }) const suffix = '\n' + row('assistant') + '\n' await appendFile(file, suffix) - const after = await readHistoryContexts({ filePath: file, sourceVersion: await version(), offsets: [Buffer.byteLength(first + partial + '\n')], classify }) + const after = await readHistoryContexts({ filePath: file, sourceVersion: await version(), offsets: [Buffer.byteLength(first + partial + '\n')], agentToolId }) expect(after.scannedBytes).toBe(Buffer.byteLength(partial + suffix)) - expect([...after.contexts.values()]).toEqual([{ owner: undefined, suppressed: false }]) + expect([...after.contexts.values()]).toEqual([{ owner: undefined, hidden: false }]) }) test('joins concurrent builds and cancelling one subscriber does not cancel the remaining reader', async () => { @@ -49,118 +59,62 @@ test('joins concurrent builds and cancelling one subscriber does not cancel the const sourceVersion = await version() const controller = new AbortController() let visits = 0 - const observed = (entry: Record) => { visits++; if (visits === 1) controller.abort(); return classify(entry) } - const first = readHistoryContexts({ filePath: file, sourceVersion, offsets: [0], signal: controller.signal, classify: observed }).catch(error => error) - const second = readHistoryContexts({ filePath: file, sourceVersion, offsets: [0], classify: observed }) + const observed = (entry: Record) => { visits++; if (visits === 1) controller.abort(); return agentToolId(entry) } + const first = readHistoryContexts({ filePath: file, sourceVersion, offsets: [0], signal: controller.signal, agentToolId: observed }).catch(error => error) + const second = readHistoryContexts({ filePath: file, sourceVersion, offsets: [0], agentToolId: observed }) expect((await first).name).toBe('AbortError') - expect((await second).contexts.get(0)?.suppressed).toBe(false) + expect((await second).contexts.get(0)?.hidden).toBe(false) expect(visits).toBe(128) - expect((await readHistoryContexts({ filePath: file, sourceVersion, offsets: [0], classify })).scannedBytes).toBe(0) + expect((await readHistoryContexts({ filePath: file, sourceVersion, offsets: [0], agentToolId })).scannedBytes).toBe(0) }) test('replacing a transcript invalidates old scalar state and rejects stale page identities', async () => { - await writeFile(file, row('notice', { notification: true }) + '\n') + await writeFile(file, row('side', { isSidechain: true }) + '\n') const oldVersion = await version() - await readHistoryContexts({ filePath: file, sourceVersion: oldVersion, offsets: [0], classify }) + expect((await readHistoryContexts({ filePath: file, sourceVersion: oldVersion, offsets: [0], agentToolId })).contexts.get(0)?.hidden).toBe(true) await writeFile(file + '.new', row('new') + '\n') await rename(file + '.new', file) - const fresh = await readHistoryContexts({ filePath: file, sourceVersion: await version(), offsets: [0], classify }) - expect(fresh.contexts.get(0)?.suppressed).toBe(false) - await expect(readHistoryContexts({ filePath: file, sourceVersion: oldVersion, offsets: [0], classify })).rejects.toMatchObject({ statusCode: 409 }) + const fresh = await readHistoryContexts({ filePath: file, sourceVersion: await version(), offsets: [0], agentToolId }) + expect(fresh.contexts.get(0)?.hidden).toBe(false) + await expect(readHistoryContexts({ filePath: file, sourceVersion: oldVersion, offsets: [0], agentToolId })).rejects.toMatchObject({ statusCode: 409 }) }) test('bounds queued file builds and cleans a fully cancelled scan for subsequent retry', async () => { const files = Array.from({ length: 6 }, (_, index) => join(directory, `${index}.jsonl`)) await Promise.all(files.map(filePath => writeFile(filePath, row('one') + '\n' + row('two') + '\n'))) const versions = await Promise.all(files.map(filePath => version(filePath))) - const requests = files.slice(0, 5).map((filePath, index) => readHistoryContexts({ filePath, sourceVersion: versions[index]!, offsets: [0], classify })) - await expect(readHistoryContexts({ filePath: files[5]!, sourceVersion: versions[5]!, offsets: [0], classify })).rejects.toMatchObject({ statusCode: 429 }) + const requests = files.slice(0, 5).map((filePath, index) => readHistoryContexts({ filePath, sourceVersion: versions[index]!, offsets: [0], agentToolId })) + await expect(readHistoryContexts({ filePath: files[5]!, sourceVersion: versions[5]!, offsets: [0], agentToolId })).rejects.toMatchObject({ statusCode: 429 }) await Promise.all(requests) const controller = new AbortController() - await expect(readHistoryContexts({ filePath: files[5]!, sourceVersion: versions[5]!, offsets: [0], signal: controller.signal, classify: entry => { controller.abort(); return classify(entry) } })).rejects.toThrow() - expect((await readHistoryContexts({ filePath: files[5]!, sourceVersion: versions[5]!, offsets: [0], classify })).contexts.get(0)?.suppressed).toBe(false) + await expect(readHistoryContexts({ filePath: files[5]!, sourceVersion: versions[5]!, offsets: [0], signal: controller.signal, agentToolId: entry => { controller.abort(); return agentToolId(entry) } })).rejects.toThrow() + expect((await readHistoryContexts({ filePath: files[5]!, sourceVersion: versions[5]!, offsets: [0], agentToolId })).contexts.get(0)?.hidden).toBe(false) }) test('rebuilds scalar context after an in-place rewrite grows beyond the cached snapshot', async () => { - await writeFile(file, row('notice', { notification: true }) + '\n') - await readHistoryContexts({ filePath: file, sourceVersion: await version(), offsets: [0], classify }) - await writeFile(file, row('replacement', { reset: true }) + '\n' + row('more-records') + '\n') - const rebuilt = await readHistoryContexts({ filePath: file, sourceVersion: await version(), offsets: [0], classify }) - expect(rebuilt.contexts.get(0)?.suppressed).toBe(false) + await writeFile(file, row('side', { isSidechain: true }) + '\n') + await readHistoryContexts({ filePath: file, sourceVersion: await version(), offsets: [0], agentToolId }) + await writeFile(file, row('replacement') + '\n' + row('more-records') + '\n') + const rebuilt = await readHistoryContexts({ filePath: file, sourceVersion: await version(), offsets: [0], agentToolId }) + expect(rebuilt.contexts.get(0)?.hidden).toBe(false) expect(rebuilt.scannedBytes).toBe(Number((await stat(file)).size)) }) -const classifyMessage = (entry: Record) => { - const message = entry.message as { role?: string; content?: unknown } - const content = message?.content - const blocks = Array.isArray(content) ? content : [] - const texts = typeof content === 'string' ? [content] : blocks.filter(block => block.type === 'text').map(block => block.text) - const user = message?.role === 'user' && !entry.isMeta - return { - notification: user && texts.length > 0 && texts.every(text => //.test(text)), - reset: user && !blocks.some(block => block.type === 'tool_result'), - agentToolId: blocks.find(block => block.type === 'tool_use' && ['Agent', 'Task'].includes(block.name))?.id, - } -} - -test('an oversized ordinary image resets notification suppression and preserves following replies', async () => { - const notice = row('notice', { type: 'user', message: { role: 'user', content: 'done' } }) + '\n' - const image = row('image', { type: 'user', message: { role: 'user', content: [{ type: 'image', source: { type: 'base64', data: 'A'.repeat(9 * 1024 * 1024) } }] } }) + '\n' - const answer = row('answer', { type: 'assistant', message: { role: 'assistant', content: 'Visible reply' } }) + '\n' - await writeFile(file, notice + image + answer) - const answerOffset = Buffer.byteLength(notice + image) - const contexts = await readHistoryContexts({ filePath: file, sourceVersion: await version(), offsets: [0, Buffer.byteLength(notice), answerOffset], classify: classifyMessage }) - expect(contexts.contexts.get(0)?.suppressed).toBe(true) - expect(contexts.contexts.get(Buffer.byteLength(notice))?.suppressed).toBe(false) - expect(contexts.contexts.get(answerOffset)?.suppressed).toBe(false) - expect((await readHistoryContexts({ filePath: file, sourceVersion: await version(), offsets: [answerOffset], classify: classifyMessage })).scannedBytes).toBe(0) -}) - test('oversized Agent input preserves the sidechain owner for later small records', async () => { const parent = row('parent', { type: 'assistant', message: { role: 'assistant', content: [{ type: 'tool_use', id: 'agent-tool', name: 'Agent', input: { prompt: 'x'.repeat(9 * 1024 * 1024) } }] } }) + '\n' const child = row('child', { type: 'assistant', parentUuid: 'parent', isSidechain: true, message: { role: 'assistant', content: 'Child result' } }) + '\n' await writeFile(file, parent + child) const offset = Buffer.byteLength(parent) - const contexts = await readHistoryContexts({ filePath: file, sourceVersion: await version(), offsets: [offset], classify: classifyMessage }) - expect(contexts.contexts.get(offset)).toEqual({ owner: 'agent-tool', suppressed: false }) + const contexts = await readHistoryContexts({ filePath: file, sourceVersion: await version(), offsets: [offset], agentToolId: messageAgentToolId }) + expect(contexts.contexts.get(offset)).toEqual({ owner: 'agent-tool', hidden: false }) }) -test('an oversized tool result does not reset preceding notification suppression', async () => { - const notice = row('notice', { type: 'user', message: { role: 'user', content: 'done' } }) + '\n' - const result = row('result', { type: 'user', message: { role: 'user', content: [{ type: 'tool_result', tool_use_id: 'tool', content: 'x'.repeat(9 * 1024 * 1024) }] } }) + '\n' - const answer = row('answer', { type: 'assistant', message: { role: 'assistant', content: 'Hidden notification response' } }) + '\n' - await writeFile(file, notice + result + answer) - const offset = Buffer.byteLength(notice + result) - expect((await readHistoryContexts({ filePath: file, sourceVersion: await version(), offsets: [offset], classify: classifyMessage })).contexts.get(offset)?.suppressed).toBe(true) -}) - -test('a notification beyond a projected text prefix remains suppressed until an explicit user reset', async () => { - const longText = row('long', { type: 'user', message: { role: 'user', content: 'x'.repeat(9 * 1024 * 1024) + 'done' } }) + '\n' - const hidden = row('hidden', { type: 'assistant', message: { role: 'assistant', content: 'Notification response' } }) + '\n' - const reset = row('reset', { type: 'user', message: { role: 'user', content: 'New request' } }) + '\n' - const visible = row('visible', { type: 'assistant', message: { role: 'assistant', content: 'Normal reply' } }) + '\n' - await writeFile(file, longText + hidden + reset + visible) - const hiddenOffset = Buffer.byteLength(longText) - const visibleOffset = Buffer.byteLength(longText + hidden + reset) - const contexts = await readHistoryContexts({ filePath: file, sourceVersion: await version(), offsets: [hiddenOffset, visibleOffset], classify: classifyMessage }) - expect(contexts.contexts.get(hiddenOffset)?.suppressed).toBe(true) - expect(contexts.contexts.get(visibleOffset)?.suppressed).toBe(false) -}) - -test('malformed oversized records preserve uncertainty rather than exposing subsequent notification replies', async () => { +test('a malformed oversized record does not disturb the records after it', async () => { const malformed = '{"message":{"content":"' + 'x'.repeat(9 * 1024 * 1024) + '\n' - const reply = row('reply', { type: 'assistant', message: { role: 'assistant', content: 'Unknown context' } }) + '\n' + const reply = row('reply', { type: 'assistant', message: { role: 'assistant', content: 'Still visible' } }) + '\n' await writeFile(file, malformed + reply) - const contexts = await readHistoryContexts({ filePath: file, sourceVersion: await version(), offsets: [Buffer.byteLength(malformed)], classify: classifyMessage }) - expect(contexts.contexts.get(Buffer.byteLength(malformed))?.suppressed).toBe(true) -}) - -test('ordinary long text inside the existing semantic limit remains a complete reset', async () => { - const notice = row('notice', { type: 'user', message: { role: 'user', content: 'done' } }) + '\n' - const request = row('request', { type: 'user', message: { role: 'user', content: 'x'.repeat(60 * 1024) } }) + '\n' - const answer = row('answer', { type: 'assistant', message: { role: 'assistant', content: 'Visible reply' } }) + '\n' - await writeFile(file, notice + request + answer) - const offset = Buffer.byteLength(notice + request) - expect((await readHistoryContexts({ filePath: file, sourceVersion: await version(), offsets: [offset], classify: classifyMessage })).contexts.get(offset)?.suppressed).toBe(false) + const offset = Buffer.byteLength(malformed) + const contexts = await readHistoryContexts({ filePath: file, sourceVersion: await version(), offsets: [offset], agentToolId: messageAgentToolId }) + expect(contexts.contexts.get(offset)).toEqual({ owner: undefined, hidden: false }) }) diff --git a/src/server/services/sessionHistoryContext.ts b/src/server/services/sessionHistoryContext.ts index 91e2fa3a..3dd4f532 100644 --- a/src/server/services/sessionHistoryContext.ts +++ b/src/server/services/sessionHistoryContext.ts @@ -5,11 +5,12 @@ import { rmSync } from 'node:fs' import { tmpdir } from 'node:os' import { join } from 'node:path' import { withHistoryReadBudget } from './boundedSessionHistory.js' -import { isSessionMetadataTextTruncated, streamSessionMetadata } from './sessionMetadataReader.js' +import { streamSessionMetadata } from './sessionMetadataReader.js' import { ApiError } from '../middleware/errorHandler.js' -type Context = { owner?: string; suppressed: boolean } -type Cache = { database: Database; directory: string; identity: string; size: number; mtime: string; offset: number; suppressed: boolean | null; fingerprint?: string } +/** `hidden` marks an unowned sidechain record that a root-only view must not show. */ +type Context = { owner?: string; hidden: boolean } +type Cache = { database: Database; directory: string; identity: string; size: number; mtime: string; offset: number; fingerprint?: string } type Flight = { promise: Promise; controller: AbortController; users: number } const cache = new Map() const flights = new Map() @@ -42,7 +43,7 @@ export async function readHistoryContexts(options: { offsets: number[] signal?: AbortSignal includeUnownedSidechains?: boolean - classify: (entry: Record) => { notification: boolean; reset: boolean; agentToolId?: string } + agentToolId: (entry: Record) => string | undefined }): Promise<{ contexts: Map; scannedBytes: number }> { if (options.signal?.aborted) throw options.signal.reason ?? new DOMException('Aborted', 'AbortError') const [dev, ino, size, mtime] = options.sourceVersion.split(':') @@ -73,8 +74,8 @@ export async function readHistoryContexts(options: { } const directory = await mkdtemp(join(tmpdir(), 'claude-history-context-')) const database = new Database(join(directory, 'context.sqlite')) - database.exec('PRAGMA journal_mode=OFF; PRAGMA cache_size=-512; PRAGMA temp_store=FILE; CREATE TABLE parents (id TEXT PRIMARY KEY, chain TEXT); CREATE TABLE context (offset INTEGER PRIMARY KEY, owner TEXT, suppressed INTEGER, unowned_sidechain INTEGER)') - state = { database, directory, identity, size: 0, mtime: '', offset: 0, suppressed: false } + database.exec('PRAGMA journal_mode=OFF; PRAGMA cache_size=-512; PRAGMA temp_store=FILE; CREATE TABLE parents (id TEXT PRIMARY KEY, chain TEXT); CREATE TABLE context (offset INTEGER PRIMARY KEY, owner TEXT, unowned_sidechain INTEGER)') + state = { database, directory, identity, size: 0, mtime: '', offset: 0 } cache.set(options.filePath, state) } cache.delete(options.filePath) @@ -82,39 +83,27 @@ export async function readHistoryContexts(options: { if (state.size >= targetSize) return const getParent = state.database.query('SELECT chain FROM parents WHERE id = ?') const saveParent = state.database.query('INSERT OR REPLACE INTO parents VALUES (?, ?)') - const saveContext = state.database.query('INSERT OR REPLACE INTO context VALUES (?, ?, ?, ?)') + const saveContext = state.database.query('INSERT OR REPLACE INTO context VALUES (?, ?, ?)') const originalOffset = state.offset - let suppressed = state.suppressed - let completeSuppression = suppressed try { const fingerprint = await sourceAnchors(options.filePath, targetSize, signal) state.database.exec('BEGIN') - const result = await streamSessionMetadata(options.filePath, (entry, completeLine, offset) => { - const classification = options.classify(entry) + const result = await streamSessionMetadata(options.filePath, (entry, _completeLine, offset) => { const inherited = typeof entry.parentUuid === 'string' ? (getParent.get(entry.parentUuid) as { chain?: string } | null)?.chain : undefined const explicit = typeof entry.parent_tool_use_id === 'string' && entry.parent_tool_use_id ? entry.parent_tool_use_id : undefined const owner = explicit ?? (entry.isSidechain === true ? inherited : undefined) - const chain = classification.agentToolId ?? inherited + const chain = options.agentToolId(entry) ?? inherited if (typeof entry.uuid === 'string') saveParent.run(entry.uuid, chain ?? null) - const message = entry.message as { role?: unknown } | undefined - // A bounded text preview cannot prove whether a user record contains a - // task notification beyond its prefix. Keep uncertainty fail-closed; - // images and tool payloads do not affect this textual classification. - if (message?.role === 'user' && !entry.isMeta && isSessionMetadataTextTruncated(entry)) suppressed = null - else if (classification.notification) suppressed = true - else if (classification.reset) suppressed = false - // Keep root-only ownership filtering separate from notification state: - // a dedicated child transcript legitimately lacks its parent's Agent call. - saveContext.run(offset, owner ?? null, suppressed !== false ? 1 : 0, entry.isSidechain === true && !owner ? 1 : 0) - if (completeLine) completeSuppression = suppressed - }, signal, { startOffset: originalOffset, endOffset: targetSize, onSkipped: () => { suppressed = null; completeSuppression = null } }) + // Root-only ownership filtering: a dedicated child transcript legitimately + // lacks its parent's Agent call, so the flag is kept per record. + saveContext.run(offset, owner ?? null, entry.isSidechain === true && !owner ? 1 : 0) + }, signal, { startOffset: originalOffset, endOffset: targetSize }) if (fingerprint !== await sourceAnchors(options.filePath, targetSize, signal)) throw new ApiError(409, 'History was rewritten during context scan', 'HISTORY_CHANGED') state.database.exec('COMMIT') state.fingerprint = fingerprint state.size = targetSize state.mtime = mtime! state.offset = result.nextOffset - state.suppressed = completeSuppression scannedBytes += result.scannedBytes } catch (error) { try { state.database.exec('ROLLBACK') } catch { /* Validation may fail before BEGIN. */ } @@ -162,12 +151,12 @@ export async function readHistoryContexts(options: { }) } const state = cache.get(options.filePath)! - const query = state.database.query('SELECT owner, suppressed, unowned_sidechain FROM context WHERE offset = ?') + const query = state.database.query('SELECT owner, unowned_sidechain FROM context WHERE offset = ?') const contexts = new Map() for (const offset of options.offsets) { - const row = query.get(offset) as { owner: string | null; suppressed: number; unowned_sidechain: number } | null + const row = query.get(offset) as { owner: string | null; unowned_sidechain: number } | null if (!row) throw new ApiError(409, 'History context is unavailable; reload the page', 'HISTORY_CHANGED') - contexts.set(offset, { owner: row.owner ?? undefined, suppressed: row.suppressed === 1 || (!options.includeUnownedSidechains && row.unowned_sidechain === 1) }) + contexts.set(offset, { owner: row.owner ?? undefined, hidden: !options.includeUnownedSidechains && row.unowned_sidechain === 1 }) } return { contexts, scannedBytes } } diff --git a/src/server/services/sessionHistoryRecovery.test.ts b/src/server/services/sessionHistoryRecovery.test.ts index e2f5c256..df5c637f 100644 --- a/src/server/services/sessionHistoryRecovery.test.ts +++ b/src/server/services/sessionHistoryRecovery.test.ts @@ -104,16 +104,16 @@ test('launch metadata, title, work directory and metadata appends never material }) -test('history pages preserve cross-page notification suppression and sidechain ownership, and index only appended bytes', async () => { +test('history pages hide the notification turn but keep its reply, preserve sidechain ownership, and index only appended bytes', async () => { const notification = 'taskagentcompleted' await writeFile(file, [ entry('assistant', 'owner', [{ type: 'tool_use', id: 'agent', name: 'Agent', input: {} }]), entry('assistant', 'child', 'child response', { isSidechain: true, parentUuid: 'owner' }), entry('user', 'notice', notification), - entry('assistant', 'hidden', 'internal notification response'), + entry('assistant', 'reply', 'the agent finished, here is the result'), ].map(value => JSON.stringify(value)).join('\n') + '\n') const latest = await service.getSessionHistoryPage(id, { limit: 1 }) - expect(latest.messages).toEqual([]) + expect(latest.messages).toMatchObject([{ id: 'reply' }]) expect(latest.page.contextScanBytes).toBeGreaterThan(0) const noticePage = await service.getSessionHistoryPage(id, { limit: 1, cursor: latest.page.nextCursor! }) expect(noticePage.messages).toEqual([]) diff --git a/src/server/services/sessionHistoryRecovery.ts b/src/server/services/sessionHistoryRecovery.ts index 5e754922..da12c36d 100644 --- a/src/server/services/sessionHistoryRecovery.ts +++ b/src/server/services/sessionHistoryRecovery.ts @@ -56,7 +56,6 @@ export async function recoverBoundedSessionHistory(options: { let goalBase: Evidence | undefined let goalStatus: Evidence | undefined let omitted = 0 - let suppressTaskNotificationResponse = false const completeness = { goal: true, todos: true, activity: true, usage: true, workspace: true } const usage = { input_tokens: 0, output_tokens: 0, cache_read_tokens: 0, cache_creation_tokens: 0 } const saveActivity = (evidence: Evidence, category: 'activity' | 'workspace' = 'activity') => { @@ -80,11 +79,9 @@ export async function recoverBoundedSessionHistory(options: { saveNotice.run(JSON.stringify([notice.ownerAgentId ?? null, notice.toolUseId]), ordinal, json) } const rawMessage = entry.message as { role?: string; content?: unknown } | undefined - const notificationUser = rawMessage?.role === 'user' && notifications.length > 0 - const hasToolResult = Array.isArray(rawMessage?.content) && rawMessage.content.some((block: any) => block?.type === 'tool_result') - if (notificationUser) { suppressTaskNotificationResponse = true; return } - if (rawMessage?.role === 'user' && !hasToolResult) suppressTaskNotificationResponse = false - else if (suppressTaskNotificationResponse) return + // The queued notification turn is plumbing (its data was saved above); the + // assistant's response to it is ordinary conversation and is kept. + if (rawMessage?.role === 'user' && notifications.length > 0) return const message = options.toMessage(entry, owner) if (!message) return if (message.usage && (!message.usageKey || usageKey.run(message.usageKey).changes > 0)) { diff --git a/src/server/services/sessionService.ts b/src/server/services/sessionService.ts index 1f6889cf..c2fcea1d 100644 --- a/src/server/services/sessionService.ts +++ b/src/server/services/sessionService.ts @@ -1973,17 +1973,6 @@ export class SessionService { ) } - private isToolResultContent(content: unknown): boolean { - return ( - Array.isArray(content) && - content.some((block) => - block && - typeof block === 'object' && - (block as Record).type === 'tool_result' - ) - ) - } - private isTaskNotificationContent(content: unknown): boolean { const textBlocks = this.extractTextBlocks(content) return ( @@ -3990,19 +3979,11 @@ export class SessionService { offsets: result.entries.map(item => item.byteStart), signal, includeUnownedSidechains, - classify: raw => { - const entry = raw as RawEntry - const user = entry.message?.role === 'user' && !entry.isMeta - return { - notification: user && this.isTaskNotificationContent(entry.message?.content), - reset: user && !this.isToolResultContent(entry.message?.content), - agentToolId: this.extractAgentToolUseId(entry), - } - }, + agentToolId: raw => this.extractAgentToolUseId(raw as RawEntry), }) const visibleEntries = result.entries.flatMap(item => { const state = context.contexts.get(item.byteStart)! - if (state.suppressed && !this.isGoalLocalCommandEntry(item.entry as RawEntry)) return [] + if (state.hidden && !this.isGoalLocalCommandEntry(item.entry as RawEntry)) return [] return [{ ...item.entry, ...(state.owner ? { parent_tool_use_id: state.owner } : {}) } as RawEntry] }) return { entries: visibleEntries, contextScanBytes: context.scannedBytes } @@ -4152,18 +4133,13 @@ export class SessionService { const taskNotifications: SessionTaskNotification[] = [] let bytes = 0 let incomplete = false - let suppressTaskNotificationResponse = false const scan = await streamBoundedHistory(filePath, raw => { const entry = raw as RawEntry const message = raw.message as { role?: string; content?: unknown } | undefined - if (!entry.isMeta && message?.role === 'user') { - if (this.isTaskNotificationContent(message.content)) suppressTaskNotificationResponse = true - else if (!this.isToolResultContent(message.content)) suppressTaskNotificationResponse = false - } const content = Array.isArray(message?.content) ? message.content.filter((block: any) => block?.type === 'tool_use' ? ids.has(block.id) : block?.type === 'tool_result' && ids.has(block.tool_use_id)) : [] const notices = this.taskNotificationsFromEntries([entry]).filter(notice => ids.has(notice.toolUseId)) - const selected = content.length && !suppressTaskNotificationResponse + const selected = content.length ? displayPreview({ ...entry, message: { ...message, content } }) : undefined if (selected?.bodyTruncated) incomplete = true const selectedBytes = (selected ? Buffer.byteLength(JSON.stringify(selected)) : 0) + @@ -5238,7 +5214,6 @@ export class SessionService { const messages: MessageEntry[] = [] const entriesByUuid = new Map() const parentToolUseIdCache = new Map() - let suppressTaskNotificationResponse = false for (const entry of entries) { if (typeof entry.uuid === 'string' && entry.uuid.length > 0) { @@ -5261,23 +5236,9 @@ export class SessionService { // message that must render as an ordinary user-position bubble. if (entry.isMeta && !parseSessionCollaborationEnvelope(entry.message.content)) continue - const isTaskNotification = - entry.message.role === 'user' && - this.isTaskNotificationContent(entry.message.content) - if (isTaskNotification) { - suppressTaskNotificationResponse = true - continue - } - - if ( - entry.message.role === 'user' && - !this.isToolResultContent(entry.message.content) - ) { - suppressTaskNotificationResponse = false - } else if (suppressTaskNotificationResponse) { - continue - } - + // The queued turn is system plumbing and is hidden here (the + // Activity cards come from its notification data). What the assistant does in + // response is ordinary conversation, so it is never dropped with it. if (this.shouldHideTranscriptEntry(entry)) continue // Skip non-transcript entry types diff --git a/src/server/services/subagentRunService.test.ts b/src/server/services/subagentRunService.test.ts index f1dc0c2f..798c8f62 100644 --- a/src/server/services/subagentRunService.test.ts +++ b/src/server/services/subagentRunService.test.ts @@ -587,7 +587,7 @@ describe('getSubagentRunByTool', () => { ) }) - it('returns child shell notifications without exposing notification turns or their response', async () => { + it('returns child shell notifications while hiding the notification turn but keeping the response to it', async () => { await setupTmpConfigDir() const sessionId = '12121212-bbbb-cccc-dddd-eeeeeeeeeeee' const projectDir = '-tmp-subagent-run' @@ -639,7 +639,7 @@ describe('getSubagentRunByTool', () => { type: 'assistant', message: { role: 'assistant', - content: [{ type: 'text', text: 'Internal notification response' }], + content: [{ type: 'text', text: 'The child shell was stopped, continuing the review' }], }, uuid: 'child-notification-response', timestamp: '2026-01-01T00:00:08.000Z', @@ -648,9 +648,9 @@ describe('getSubagentRunByTool', () => { const result = await getSubagentRunByTool(sessionId, toolUseId) - expect(result?.messages).toHaveLength(2) + expect(result?.messages).toHaveLength(3) expect(JSON.stringify(result?.messages)).not.toContain('') - expect(JSON.stringify(result?.messages)).not.toContain('Internal notification response') + expect(JSON.stringify(result?.messages)).toContain('The child shell was stopped, continuing the review') expect(result?.taskNotifications).toEqual([{ taskId: 'shell-task-1', toolUseId: 'shell.call:0', @@ -659,7 +659,7 @@ describe('getSubagentRunByTool', () => { outputFile: '/tmp/shell-task-1.output', timestamp: '2026-01-01T00:00:07.000Z', }]) - expect(result?.updatedAt).toBe('2026-01-01T00:00:07.000Z') + expect(result?.updatedAt).toBe('2026-01-01T00:00:08.000Z') }) it('uses the live task id to resolve a running one-shot SubAgent transcript', async () => { @@ -1677,7 +1677,7 @@ describe('getSubagentRunByAgentId', () => { expect(result?.canSendMessage).toBe(false) }) - it('returns workflow-agent task notifications beside its filtered transcript', async () => { + it('returns workflow-agent task notifications beside its transcript without the notification turn', async () => { await setupTmpConfigDir() const sessionId = '90909090-bbbb-cccc-dddd-ffffffffffff' const projectDir = '-tmp-workflow-agent' @@ -1702,7 +1702,7 @@ describe('getSubagentRunByAgentId', () => { }, { type: 'assistant', - message: { role: 'assistant', content: 'Internal notification response' }, + message: { role: 'assistant', content: 'The workflow check passed, moving on' }, uuid: 'wf-notification-response', timestamp: '2026-01-01T00:00:07.000Z', }, @@ -1710,7 +1710,7 @@ describe('getSubagentRunByAgentId', () => { const result = await getSubagentRunByAgentId(sessionId, agentId) - expect(result?.messages.map(message => message.id)).toEqual(['wf-assistant']) + expect(result?.messages.map(message => message.id)).toEqual(['wf-assistant', 'wf-notification-response']) expect(result?.taskNotifications).toEqual([{ taskId: 'wf-shell-task', toolUseId: 'wf-shell-tool', @@ -1718,7 +1718,7 @@ describe('getSubagentRunByAgentId', () => { summary: 'Workflow check passed', timestamp: '2026-01-01T00:00:06.000Z', }]) - expect(result?.updatedAt).toBe('2026-01-01T00:00:06.000Z') + expect(result?.updatedAt).toBe('2026-01-01T00:00:07.000Z') }) it('returns null when no transcript exists for that agent', async () => { diff --git a/src/services/api/claudeRequiredThinking.test.ts b/src/services/api/claudeRequiredThinking.test.ts index ec337def..7966f359 100644 --- a/src/services/api/claudeRequiredThinking.test.ts +++ b/src/services/api/claudeRequiredThinking.test.ts @@ -855,6 +855,26 @@ for (const disableAdaptive of [false, true]) { }, 10_000) } +for (const disableAdaptive of [false, true]) { + test(`Sonnet 5.5 never sends disabled thinking, with optional adaptive disabled=${disableAdaptive}`, async () => { + const { content, requests } = await captureQueryRequest({ + model: 'claude-sonnet-5-5', + configureCapabilityOverrides: false, + globalThinkingEnabled: false, + env: { + CLAUDE_CODE_DISABLE_ADAPTIVE_THINKING: disableAdaptive ? '1' : undefined, + CLAUDE_CODE_ALWAYS_ENABLE_EFFORT: '1', + }, + }) + expect(content).toContainEqual({ type: 'text', text: 'OK' }) + expect(requests).toHaveLength(1) + expect(requests[0]?.model).toBe('claude-sonnet-5-5') + expect(requests[0]?.thinking).toEqual({ type: 'adaptive' }) + expect(requests[0]?.max_tokens).toBe(128_000) + expect(requests[0]?.output_config).toEqual({ effort: 'medium' }) + }, 10_000) +} + test('Opus 5.5 respects an explicit third-party capability opt-out', async () => { const { requests } = await captureQueryRequest({ model: 'claude-opus-5-5', diff --git a/src/services/api/errors.test.ts b/src/services/api/errors.test.ts index 226cefa5..f20676aa 100644 --- a/src/services/api/errors.test.ts +++ b/src/services/api/errors.test.ts @@ -1,5 +1,6 @@ import { describe, expect, test } from 'bun:test' import { APIError } from '@anthropic-ai/sdk' +import { getIsInteractive, setIsInteractive } from '../../bootstrap/state.js' import { BUSINESS_ERROR_CODES } from '../../constants/businessErrors.js' import { getAssistantMessageFromError, @@ -7,6 +8,7 @@ import { getImageUnsupportedErrorMessage, isContextOverflowErrorText, isUnsupportedImageInputErrorMessage, + measureRequestPayload, PROMPT_TOO_LONG_ERROR_MESSAGE, parsePromptTooLongTokenCounts, } from './errors.js' @@ -240,3 +242,167 @@ describe('context overflow errors', () => { }) }) }) + +describe('request too large (HTTP 413)', () => { + const MODEL = 'claude-sonnet-5-5' + const nginxHtml = + '\r\n413 Request Entity Too Large\r\n\r\n

413 Request Entity Too Large

\r\n
nginx/1.18.0
\r\n\r\n' + const relayBody = { + error: { message: 'request body too large, limit is 10 MB', type: 'invalid_request_error' }, + } + const anthropicBody = { + type: 'error', + error: { type: 'request_too_large', message: 'Request exceeds the maximum allowed number of bytes.' }, + } + const sources = [ + { + name: 'nginx in front of a relay', + error: new APIError(413, undefined, `413 ${nginxHtml}`, undefined), + upstream: 'nginx/1.18.0', + }, + { + name: 'a relay that names its own limit', + error: new APIError(413, relayBody, `413 ${JSON.stringify(relayBody)}`, undefined), + upstream: 'limit is 10 MB', + }, + { + name: 'the Anthropic API', + error: new APIError(413, anthropicBody, `413 ${JSON.stringify(anthropicBody)}`, undefined), + upstream: 'maximum allowed number of bytes', + }, + ] + const textOf = (msg: { message: { content: unknown[] } }) => + (msg.message.content[0] as { text: string }).text + const image = (chars: number) => ({ + type: 'image', + source: { type: 'base64', media_type: 'image/png', data: 'A'.repeat(chars) }, + }) + const userTurn = (content: unknown[]) => ({ + type: 'user' as const, + message: { role: 'user' as const, content }, + }) + const relayRejection = sources[1]!.error + + test.each(sources)( + 'does not invent a size limit when $name rejects the request', + ({ error, upstream }) => { + const msg = getAssistantMessageFromError(error, MODEL) + const text = textOf(msg) + + // The old wording hard-coded the PDF limit ("max 20MB") for every 413, + // which is wrong for a relay with its own limit and for the API's real one. + expect(text).not.toMatch(/\d\s?MB/) + expect(text).not.toMatch(/smaller file/i) + expect(text).toContain('HTTP 413') + expect(msg.businessErrorCode).toBe(BUSINESS_ERROR_CODES.REQUEST_TOO_LARGE) + // Whoever rejected the request said something; keep it for diagnosis. + expect(msg.errorDetails).toStartWith('request_too_large: ') + expect(msg.errorDetails).toContain(upstream) + }, + ) + + test('reports how much of the rejected request was images and documents', () => { + const messagesForAPI = [ + userTurn([{ type: 'text', text: 'take a screenshot' }]), + userTurn([ + { type: 'tool_result', tool_use_id: 't1', content: [image(2 * 1024 * 1024)] }, + ]), + ] + + const text = textOf( + getAssistantMessageFromError(relayRejection, MODEL, { + messagesForAPI: messagesForAPI as never, + }), + ) + + expect(text).toContain('about 2MB') + expect(text).toContain('2MB is images or documents') + expect(text).toContain('placeholders') + }) + + test('says so when the rejected request carried no images or documents', () => { + const messagesForAPI = [ + userTurn([ + { type: 'tool_result', tool_use_id: 't1', content: 'x'.repeat(1.5 * 1024 * 1024) }, + ]), + ] + + const text = textOf( + getAssistantMessageFromError(relayRejection, MODEL, { + messagesForAPI: messagesForAPI as never, + }), + ) + + expect(text).toContain('about 1.5MB') + expect(text).toContain('none of it is images or documents') + }) + + test('points interactive users at /compact and non-interactive callers at a new session', () => { + // Session mode is process-global and bun runs every file in one process: + // set both modes explicitly and put back whatever was there. + const original = getIsInteractive() + try { + setIsInteractive(true) + expect(textOf(getAssistantMessageFromError(relayRejection, MODEL))).toContain('/compact') + setIsInteractive(false) + expect(textOf(getAssistantMessageFromError(relayRejection, MODEL))).toContain( + 'start a new session', + ) + } finally { + setIsInteractive(original) + } + }) + + describe('measureRequestPayload', () => { + const toolUse = { + type: 'tool_use', + id: 't', + name: 'Read', + input: { file_path: '/a' }, + } + + test('counts media by data length and keeps it separate from everything else', () => { + const size = measureRequestPayload([ + userTurn([{ type: 'text', text: 'hello' }]), + userTurn([ + { + type: 'tool_result', + tool_use_id: 't', + content: [image(1000), { type: 'text', text: 'abc' }], + }, + ]), + userTurn([ + { + type: 'document', + source: { type: 'base64', media_type: 'application/pdf', data: 'B'.repeat(500) }, + }, + ]), + { type: 'assistant', message: { role: 'assistant', content: [toolUse] } }, + ] as never) + + expect(size?.mediaBytes).toBe(1500) + expect(size?.totalBytes).toBe( + 5 + 1003 + 500 + Buffer.byteLength(JSON.stringify(toolUse)), + ) + }) + + test('never throws while classifying: an unserializable payload falls back to the unmeasured wording', () => { + const circular: Record = {} + circular.self = circular + const messagesForAPI = [ + { + type: 'assistant', + message: { + role: 'assistant', + content: [{ type: 'tool_use', id: 't', name: 'Read', input: circular }], + }, + }, + ] as never + + expect(measureRequestPayload(messagesForAPI)).toBeUndefined() + expect( + textOf(getAssistantMessageFromError(relayRejection, MODEL, { messagesForAPI })), + ).toContain('The size limit is set by that endpoint') + }) + }) +}) diff --git a/src/services/api/errors.ts b/src/services/api/errors.ts index cb6cb5e2..a43a39a9 100644 --- a/src/services/api/errors.ts +++ b/src/services/api/errors.ts @@ -269,11 +269,100 @@ export function getImageUnsupportedErrorMessage(): string { ? 'This model does not support images. Continue with text, or switch to a vision-capable model and send the image again.' : 'This model does not support images. Double press esc to go back, switch to a vision-capable model, or continue with text.' } -export function getRequestTooLargeErrorMessage(): string { - const limits = `max ${formatFileSize(PDF_TARGET_RAW_SIZE)}` - return getIsNonInteractiveSession() - ? `Request too large (${limits}). Try with a smaller file.` - : `Request too large (${limits}). Double press esc to go back and try with a smaller file.` +// A 413 body can be a whole HTML page; keep enough to identify the server +// (e.g. "nginx/1.18.0") without bloating the transcript. +const REQUEST_TOO_LARGE_DETAIL_CHARS = 500 + +/** + * Rough size of a request that was rejected as too large. Only measured on the + * 413 error path, never while sending. Base64 media dominates real payloads, so + * image/document data is counted by string length instead of re-serialized. + * `mediaBytes` is what replacing images and documents with placeholders removes. + */ +export type RequestPayloadSize = { totalBytes: number; mediaBytes: number } + +function measureContent(content: unknown): RequestPayloadSize { + if (typeof content === 'string') { + return { totalBytes: Buffer.byteLength(content), mediaBytes: 0 } + } + if (!Array.isArray(content)) { + return { totalBytes: 0, mediaBytes: 0 } + } + let totalBytes = 0 + let mediaBytes = 0 + for (const block of content) { + const size = measureBlock(block) + totalBytes += size.totalBytes + mediaBytes += size.mediaBytes + } + return { totalBytes, mediaBytes } +} + +function measureBlock(block: unknown): RequestPayloadSize { + if (typeof block === 'string') return measureContent(block) + if (!block || typeof block !== 'object') { + return { totalBytes: 0, mediaBytes: 0 } + } + const { type, text, source, content } = block as { + type?: unknown + text?: unknown + source?: { data?: unknown; url?: unknown } + content?: unknown + } + if (type === 'image' || type === 'document') { + const bytes = + typeof source?.data === 'string' + ? source.data.length + : typeof source?.url === 'string' + ? source.url.length + : 0 + return { totalBytes: bytes, mediaBytes: bytes } + } + if (type === 'text' && typeof text === 'string') return measureContent(text) + if (type === 'tool_result') return measureContent(content) + return { totalBytes: Buffer.byteLength(JSON.stringify(block)), mediaBytes: 0 } +} + +/** + * Returns undefined instead of throwing: this runs while classifying an API + * failure, where a second failure would hide the first. + */ +export function measureRequestPayload( + messages: readonly (UserMessage | AssistantMessage)[], +): RequestPayloadSize | undefined { + try { + let totalBytes = 0 + let mediaBytes = 0 + for (const message of messages) { + const size = measureContent(message.message.content) + totalBytes += size.totalBytes + mediaBytes += size.mediaBytes + } + return { totalBytes, mediaBytes } + } catch { + return undefined + } +} + +export function getRequestTooLargeErrorMessage( + size?: RequestPayloadSize, +): string { + const interactive = !getIsNonInteractiveSession() + const rejected = + 'Request too large: your provider or relay rejected it (HTTP 413).' + const nextStep = interactive + ? 'run /compact or double press esc to go back past the large content' + : 'compact the conversation or start a new session' + const sentenceNextStep = nextStep.charAt(0).toUpperCase() + nextStep.slice(1) + + if (!size) { + return `${rejected} The size limit is set by that endpoint. ${sentenceNextStep}.` + } + const total = formatFileSize(size.totalBytes) + if (size.mediaBytes === 0) { + return `${rejected} This conversation is about ${total} and none of it is images or documents. ${sentenceNextStep}; a relay can enforce a lower request size limit than its upstream provider.` + } + return `${rejected} This conversation is about ${total}, of which ${formatFileSize(size.mediaBytes)} is images or documents. Earlier images and documents are replaced with placeholders in the next request; if it still fails, ${nextStep}.` } export const OAUTH_ORG_NOT_ALLOWED_ERROR_MESSAGE = 'Your account does not have access to Claude Code. Please run /login.' @@ -817,12 +906,20 @@ export function getAssistantMessageFromError( }) } - // Check for request too large errors (413 status) - // This typically happens when a large PDF + conversation context exceeds the 32MB API limit + // 413 comes from whatever sits in front of the model: the API itself (32MB + // request limit) or a relay/gateway with its own, usually smaller, limit. + // Which one and what limit is unknown here, so report what was sent and keep + // the upstream's own words rather than claiming a number. No sourceModel on + // purpose: a byte limit belongs to the provider or relay, not to one model + // (normalizeMessagesForAPI relies on that to keep stripping after a switch). if (error instanceof APIError && error.status === 413) { + const size = options?.messagesForAPI + ? measureRequestPayload(options.messagesForAPI) + : undefined return createAssistantAPIErrorMessage({ - content: getRequestTooLargeErrorMessage(), + content: getRequestTooLargeErrorMessage(size), error: 'invalid_request', + errorDetails: `request_too_large: ${(error.message ?? '').slice(0, REQUEST_TOO_LARGE_DETAIL_CHARS)}`, businessErrorCode: BUSINESS_ERROR_CODES.REQUEST_TOO_LARGE, }) } diff --git a/src/services/compact/compact.test.ts b/src/services/compact/compact.test.ts index 2983f463..3baa6f7b 100644 --- a/src/services/compact/compact.test.ts +++ b/src/services/compact/compact.test.ts @@ -1,6 +1,6 @@ import { describe, expect, test } from 'bun:test' -import { buildPostCompactMessages, truncateHeadForPTLRetry, type CompactionResult } from './compact.js' +import { buildPostCompactMessages, stripImagesFromMessages, truncateHeadForPTLRetry, type CompactionResult } from './compact.js' import { getCurrentUsage } from '../../utils/tokens.js' import type { AssistantMessage, Message } from '../../types/message.js' @@ -189,3 +189,54 @@ describe('oversized compaction recovery (#1373)', () => { expect(second[0]?.type).toBe('user') }) }) + +describe('stripImagesFromMessages', () => { + const image = { + type: 'image', + source: { type: 'base64', media_type: 'image/png', data: 'AAAA' }, + } + const userWith = (content: unknown) => + ({ + ...makePreservedUser(), + uuid: crypto.randomUUID(), + message: { role: 'user', content }, + }) as Message + + test('replaces images and documents with markers, including inside tool results', () => { + const [stripped] = stripImagesFromMessages([ + userWith([ + { type: 'text', text: 'look' }, + image, + { + type: 'document', + source: { type: 'base64', media_type: 'application/pdf', data: 'JVBERi0=' }, + }, + { type: 'tool_result', tool_use_id: 't1', content: [image, { type: 'text', text: 'caption' }] }, + ]), + ]) + + expect((stripped as { message: { content: unknown } }).message.content).toEqual([ + { type: 'text', text: 'look' }, + { type: 'text', text: '[image]' }, + { type: 'text', text: '[document]' }, + { + type: 'tool_result', + tool_use_id: 't1', + content: [ + { type: 'text', text: '[image]' }, + { type: 'text', text: 'caption' }, + ], + }, + ]) + }) + + test('returns messages without media untouched', () => { + const plain = userWith([{ type: 'text', text: 'plain' }]) + const assistant = makePreservedAssistant() + + const result = stripImagesFromMessages([plain, assistant]) + + expect(result[0]).toBe(plain) + expect(result[1]).toBe(assistant) + }) +}) diff --git a/src/services/compact/compact.ts b/src/services/compact/compact.ts index 29756e16..8e0c122e 100644 --- a/src/services/compact/compact.ts +++ b/src/services/compact/compact.ts @@ -65,6 +65,7 @@ import { getMessagesAfterCompactBoundary, isCompactBoundaryMessage, normalizeMessagesForAPI, + replaceMediaWithPlaceholders, } from '../../utils/messages.js' import { expandPath } from '../../utils/path.js' import { getPlan, getPlanFilePath } from '../../utils/plans.js' @@ -144,60 +145,9 @@ const MAX_COMPACT_STREAMING_RETRIES = 2 * and thinking blocks but not images. */ export function stripImagesFromMessages(messages: Message[]): Message[] { - return messages.map(message => { - if (message.type !== 'user') { - return message - } - - const content = message.message.content - if (!Array.isArray(content)) { - return message - } - - let hasMediaBlock = false - const newContent = content.flatMap(block => { - if (block.type === 'image') { - hasMediaBlock = true - return [{ type: 'text' as const, text: '[image]' }] - } - if (block.type === 'document') { - hasMediaBlock = true - return [{ type: 'text' as const, text: '[document]' }] - } - // Also strip images/documents nested inside tool_result content arrays - if (block.type === 'tool_result' && Array.isArray(block.content)) { - let toolHasMedia = false - const newToolContent = block.content.map(item => { - if (item.type === 'image') { - toolHasMedia = true - return { type: 'text' as const, text: '[image]' } - } - if (item.type === 'document') { - toolHasMedia = true - return { type: 'text' as const, text: '[document]' } - } - return item - }) - if (toolHasMedia) { - hasMediaBlock = true - return [{ ...block, content: newToolContent }] - } - } - return [block] - }) - - if (!hasMediaBlock) { - return message - } - - return { - ...message, - message: { - ...message.message, - content: newContent, - }, - } as typeof message - }) + return messages.map(message => + message.type === 'user' ? replaceMediaWithPlaceholders(message) : message, + ) } /** diff --git a/src/skills/bundled/claudeApiContent.ts b/src/skills/bundled/claudeApiContent.ts index 901cf3c8..3ade8abb 100644 --- a/src/skills/bundled/claudeApiContent.ts +++ b/src/skills/bundled/claudeApiContent.ts @@ -34,14 +34,14 @@ import typescriptClaudeApiToolUse from './claude-api/typescript/claude-api/tool- // - claude-api/SKILL.md (Current Models pricing table) // - claude-api/shared/models.md (full model catalog with legacy versions and alias mappings) export const SKILL_MODEL_VARS = { - OPUS_ID: 'claude-opus-4-8', - OPUS_NAME: 'Claude Opus 4.8', - SONNET_ID: 'claude-sonnet-5', - SONNET_NAME: 'Claude Sonnet 5', + OPUS_ID: 'claude-opus-5-5', + OPUS_NAME: 'Claude Opus 5.5', + SONNET_ID: 'claude-sonnet-5-5', + SONNET_NAME: 'Claude Sonnet 5.5', HAIKU_ID: 'claude-haiku-4-5', HAIKU_NAME: 'Claude Haiku 4.5', // Previous Sonnet ID — used in "do not append date suffixes" example in SKILL.md. - PREV_SONNET_ID: 'claude-sonnet-4-5', + PREV_SONNET_ID: 'claude-sonnet-5', } satisfies Record export const SKILL_PROMPT: string = skillPrompt diff --git a/src/utils/commitAttribution.ts b/src/utils/commitAttribution.ts index 4857074f..f4f9eea8 100644 --- a/src/utils/commitAttribution.ts +++ b/src/utils/commitAttribution.ts @@ -162,6 +162,7 @@ export function sanitizeModelName(shortName: string): string { if (shortName.includes('opus-4-5')) return 'claude-opus-4-5' if (shortName.includes('opus-4-1')) return 'claude-opus-4-1' if (shortName.includes('opus-4')) return 'claude-opus-4' + if (shortName.includes('sonnet-5-5')) return 'claude-sonnet-5-5' if (shortName.includes('sonnet-5')) return 'claude-sonnet-5' if (shortName.includes('sonnet-4-6')) return 'claude-sonnet-4-6' if (shortName.includes('sonnet-4-5')) return 'claude-sonnet-4-5' diff --git a/src/utils/context.ts b/src/utils/context.ts index c34e0732..7698396d 100644 --- a/src/utils/context.ts +++ b/src/utils/context.ts @@ -269,7 +269,7 @@ export function getModelMaxOutputTokens(model: string): { const m = getCanonicalName(model) - if (m === 'claude-opus-5-5') { + if (m === 'claude-opus-5-5' || m === 'claude-sonnet-5-5') { defaultTokens = 128_000 upperLimit = 128_000 } else if ( diff --git a/src/utils/effort.ts b/src/utils/effort.ts index 0b92abde..ed2f75f8 100644 --- a/src/utils/effort.ts +++ b/src/utils/effort.ts @@ -435,7 +435,10 @@ export function getDefaultEffortForModel( // the model launch DRI and research. Default effort is a sensitive setting // that can greatly affect model quality and bashing. - if (getCanonicalName(model) === 'claude-opus-5-5') { + // Claude Code runs Opus 5.5 and Sonnet 5.5 at medium; the API default for + // Sonnet 5.5 is high, so it must be sent explicitly. + const canonicalModel = getCanonicalName(model) + if (canonicalModel === 'claude-opus-5-5' || canonicalModel === 'claude-sonnet-5-5') { return 'medium' } diff --git a/src/utils/messages.test.ts b/src/utils/messages.test.ts index 818a4735..30740b28 100644 --- a/src/utils/messages.test.ts +++ b/src/utils/messages.test.ts @@ -1,12 +1,21 @@ import { describe, expect, test } from 'bun:test' +import { APIError } from '@anthropic-ai/sdk' import type { ContentBlockParam } from '@anthropic-ai/sdk/resources/index.mjs' +import { BUSINESS_ERROR_CODES } from '../constants/businessErrors.js' +import { + getAssistantMessageFromError, + getImageTooLargeErrorMessage, +} from '../services/api/errors.js' import type { Tool } from '../Tool.js' import type { AssistantMessage } from '../types/message.js' +import { createAttachmentMessage } from './attachments.js' import { + createAssistantAPIErrorMessage, createAssistantMessage, createUserMessage, normalizeMessagesForAPI, normalizeContentFromAPI, + replaceMediaWithPlaceholders, stripSignatureBlocksAfterModelChange, } from './messages.js' @@ -274,3 +283,242 @@ test('legacy ordinary tool-use JSON still round-trips without a migration', () = const replay = normalizeMessagesForAPI([historical, toolResult('legacy-read')], [tool]) expect(replay[0]!.message.content).toEqual(historical.message.content) }) + +type ApiBlock = { type: string; [key: string]: unknown } + +function imageBlock(chars: number) { + return { + type: 'image' as const, + source: { + type: 'base64' as const, + media_type: 'image/png' as const, + data: 'A'.repeat(chars), + }, + } +} + +function screenshotResult(id: string, chars: number) { + return createUserMessage({ + content: [ + { type: 'tool_result', tool_use_id: id, content: [imageBlock(chars)] }, + ] as ContentBlockParam[], + }) +} + +// The real classifier, so the anchor is exactly what a relay's 413 produces. +function requestTooLargeAnchor() { + return getAssistantMessageFromError( + new APIError(413, undefined, '413 Request Entity Too Large', undefined), + 'claude-sonnet-5-5', + ) +} + +function allBlocks( + messages: ReturnType, +): ApiBlock[] { + const blocks: ApiBlock[] = [] + for (const message of messages) { + const content = message.message.content + if (!Array.isArray(content)) continue + for (const block of content as ApiBlock[]) { + blocks.push(block) + if (block.type === 'tool_result' && Array.isArray(block.content)) { + blocks.push(...(block.content as ApiBlock[])) + } + } + } + return blocks +} + +const imagesIn = (blocks: ApiBlock[]) => blocks.filter(b => b.type === 'image') +// Attachment text is wrapped in a system reminder, which adds a trailing newline. +const placeholdersIn = (blocks: ApiBlock[]) => + blocks.filter(b => b.type === 'text' && String(b.text).trim() === '[image]') +const bytesOf = (value: unknown) => Buffer.byteLength(JSON.stringify(value)) + +describe('normalizeMessagesForAPI after a request-too-large rejection', () => { + test('replaces images from every earlier turn, including ones nested in tool results', () => { + const history = [createUserMessage({ content: 'do a long GUI task' })] + for (let i = 0; i < 3; i++) { + history.push(assistant(`a${i}`, [toolUse(`s${i}`)]), screenshotResult(`s${i}`, 300_000)) + } + const fresh = createUserMessage({ + content: [{ type: 'text', text: 'try again' }, imageBlock(1_000)] as ContentBlockParam[], + }) + + const normalized = normalizeMessagesForAPI([...history, requestTooLargeAnchor(), fresh]) + const blocks = allBlocks(normalized) + + // Only the image sent after the rejection is still a real image. + expect(imagesIn(blocks)).toHaveLength(1) + expect(placeholdersIn(blocks)).toHaveLength(3) + // Placeholders replace media in place, so tool_use/tool_result pairing holds. + expect( + blocks.filter(b => b.type === 'tool_result').map(b => b.tool_use_id), + ).toEqual(['s0', 's1', 's2']) + expect(bytesOf(normalized)).toBeLessThan(20_000) + }) + + test('covers images from @-mentioned files, which are rebuilt on every normalization', () => { + const mention = createAttachmentMessage({ + type: 'file', + filename: '/tmp/shot.png', + displayPath: 'shot.png', + content: { + type: 'image', + file: { base64: 'A'.repeat(300_000), type: 'image/png', originalSize: 225_000 }, + }, + } as never) + const history = [ + mention, + createUserMessage({ content: 'what is in @shot.png?' }), + assistant('a0', [{ type: 'text', text: 'A screenshot.' }]), + ] + const followUp = createUserMessage({ content: 'and now?' }) + + // Without a rejection the attachment really does contribute an image. + expect(imagesIn(allBlocks(normalizeMessagesForAPI([...history, followUp])))).toHaveLength(1) + + const blocks = allBlocks( + normalizeMessagesForAPI([...history, requestTooLargeAnchor(), followUp]), + ) + expect(imagesIn(blocks)).toHaveLength(0) + expect(placeholdersIn(blocks)).toHaveLength(1) + }) + + test('leaves a history without media exactly as it was', () => { + const history = [createUserMessage({ content: 'dump the logs' })] + for (let i = 0; i < 3; i++) { + history.push( + assistant(`a${i}`, [toolUse(`t${i}`)]), + createUserMessage({ + content: [ + { type: 'tool_result', tool_use_id: `t${i}`, content: 'x'.repeat(50_000) }, + ] as ContentBlockParam[], + }), + ) + } + const followUp = createUserMessage({ content: 'compact it' }) + + expect( + normalizeMessagesForAPI([...history, requestTooLargeAnchor(), followUp]).map( + m => m.message.content, + ), + ).toEqual(normalizeMessagesForAPI([...history, followUp]).map(m => m.message.content)) + }) + + test('holds for any model: the byte limit belongs to the provider, not to one model', () => { + const history = [ + createUserMessage({ content: 'screenshot it' }), + assistant('a0', [toolUse('s0')]), + screenshotResult('s0', 300_000), + ] + const followUp = createUserMessage({ content: 'continue' }) + + const normalized = normalizeMessagesForAPI( + [...history, requestTooLargeAnchor(), followUp], + [], + 'a-different-model', + ) + + expect(imagesIn(allBlocks(normalized))).toHaveLength(0) + }) + + test.each([ + 'Request too large (max 20MB). Try with a smaller file.', + 'Request too large (max 20MB). Double press esc to go back and try with a smaller file.', + ])('still recognizes transcripts saved with the old wording: %s', legacyText => { + const legacyAnchor = createAssistantAPIErrorMessage({ content: legacyText }) + expect(legacyAnchor.businessErrorCode).toBeUndefined() + const history = [ + createUserMessage({ content: 'screenshot it' }), + assistant('a0', [toolUse('s0')]), + screenshotResult('s0', 300_000), + assistant('a1', [toolUse('s1')]), + screenshotResult('s1', 300_000), + ] + + const blocks = allBlocks( + normalizeMessagesForAPI([...history, legacyAnchor, createUserMessage({ content: 'go on' })]), + ) + + expect(imagesIn(blocks)).toHaveLength(0) + expect(placeholdersIn(blocks)).toHaveLength(2) + }) + + test('other media rejections stay scoped to the turn they followed', () => { + const earlier = createUserMessage({ + content: [{ type: 'text', text: 'first' }, imageBlock(1_000)] as ContentBlockParam[], + }) + const oversized = createUserMessage({ + content: [{ type: 'text', text: 'second' }, imageBlock(1_000)] as ContentBlockParam[], + }) + const anchor = createAssistantAPIErrorMessage({ + content: getImageTooLargeErrorMessage(), + businessErrorCode: BUSINESS_ERROR_CODES.IMAGE_TOO_LARGE, + }) + + const blocks = allBlocks( + normalizeMessagesForAPI([ + earlier, + assistant('a0', [{ type: 'text', text: 'ok' }]), + oversized, + anchor, + createUserMessage({ content: 'retry' }), + ]), + ) + + // Only the rejected turn loses its image; earlier turns are not touched. + expect(imagesIn(blocks)).toHaveLength(1) + }) +}) + +describe('replaceMediaWithPlaceholders', () => { + test('replaces top-level and tool-result media without touching the input', () => { + const message = createUserMessage({ + content: [ + { type: 'text', text: 'see attached' }, + imageBlock(10), + { + type: 'document', + source: { type: 'base64', media_type: 'application/pdf', data: 'JVBERi0=' }, + }, + { + type: 'tool_result', + tool_use_id: 't1', + content: [imageBlock(10), { type: 'text', text: 'caption' }], + }, + ] as ContentBlockParam[], + }) + const before = JSON.stringify(message) + + const replaced = replaceMediaWithPlaceholders(message) + + expect(replaced.message.content).toEqual([ + { type: 'text', text: 'see attached' }, + { type: 'text', text: '[image]' }, + { type: 'text', text: '[document]' }, + { + type: 'tool_result', + tool_use_id: 't1', + content: [ + { type: 'text', text: '[image]' }, + { type: 'text', text: 'caption' }, + ], + }, + ]) + expect(JSON.stringify(message)).toBe(before) + }) + + test('returns the same message when it has no media', () => { + const text = createUserMessage({ content: 'plain text' }) + const blocks = createUserMessage({ + content: [ + { type: 'tool_result', tool_use_id: 't1', content: 'ok' }, + ] as ContentBlockParam[], + }) + + expect(replaceMediaWithPlaceholders(text)).toBe(text) + expect(replaceMediaWithPlaceholders(blocks)).toBe(blocks) + }) +}) diff --git a/src/utils/messages.ts b/src/utils/messages.ts index 812e9449..aa198f55 100644 --- a/src/utils/messages.ts +++ b/src/utils/messages.ts @@ -37,7 +37,9 @@ import { import { OUTPUT_STYLE_CONFIG } from '../constants/outputStyles.js' import { type BusinessErrorCode, + BUSINESS_ERROR_CODES, BUSINESS_ERROR_MEDIA_BLOCK_TYPES, + LEGACY_REQUEST_TOO_LARGE_ERROR_MESSAGES, } from '../constants/businessErrors.js' import { isAutoMemoryEnabled } from '../memdir/paths.js' import { @@ -50,7 +52,6 @@ import { getPdfInvalidErrorMessage, getPdfPasswordProtectedErrorMessage, getPdfTooLargeErrorMessage, - getRequestTooLargeErrorMessage, } from '../services/api/errors.js' import type { AnyObject, Progress } from '../Tool.js' import { isConnectorTextBlock } from '../types/connectorText.js' @@ -2020,6 +2021,80 @@ function relocateToolReferenceSiblings( return result } +/** + * Replaces image and document blocks with short text markers, including media + * nested in tool_result content. Markers (rather than removal) keep turn + * structure and tool_use/tool_result pairing valid, and let the model see that + * media was there. Returns the same object when there is nothing to replace. + * + * Only user messages carry media; assistant messages hold text, tool_use and + * thinking blocks. + */ +export function replaceMediaWithPlaceholders(message: UserMessage): UserMessage { + const content = message.message.content + if (!Array.isArray(content)) { + return message + } + + let hasMediaBlock = false + const newContent = content.flatMap(block => { + if (block.type === 'image') { + hasMediaBlock = true + return [{ type: 'text' as const, text: '[image]' }] + } + if (block.type === 'document') { + hasMediaBlock = true + return [{ type: 'text' as const, text: '[document]' }] + } + // Also replace media nested inside tool_result content arrays + if (block.type === 'tool_result' && Array.isArray(block.content)) { + let toolHasMedia = false + const newToolContent = block.content.map(item => { + if (item.type === 'image') { + toolHasMedia = true + return { type: 'text' as const, text: '[image]' } + } + if (item.type === 'document') { + toolHasMedia = true + return { type: 'text' as const, text: '[document]' } + } + return item + }) + if (toolHasMedia) { + hasMediaBlock = true + return [{ ...block, content: newToolContent }] + } + } + return [block] + }) + + if (!hasMediaBlock) { + return message + } + + return { + ...message, + message: { + ...message.message, + content: newContent, + }, + } as UserMessage +} + +// Legacy transcripts predate businessErrorCode and only carry the wording. +function isRequestTooLargeAnchor( + message: AssistantMessage, + errorText: string | undefined, +): boolean { + if (typeof message.businessErrorCode === 'string') { + return message.businessErrorCode === BUSINESS_ERROR_CODES.REQUEST_TOO_LARGE + } + return ( + errorText !== undefined && + LEGACY_REQUEST_TOO_LARGE_ERROR_MESSAGES.includes(errorText) + ) +} + export function normalizeMessagesForAPI( messages: Message[], tools: Tools = [], @@ -2044,12 +2119,18 @@ export function normalizeMessagesForAPI( [getPdfInvalidErrorMessage()]: new Set(['document']), [getImageTooLargeErrorMessage()]: new Set(['image']), [getImageUnsupportedErrorMessage()]: new Set(['image']), - [getRequestTooLargeErrorMessage()]: new Set(['document', 'image']), } // Walk the reordered messages to build a targeted strip map: // userMessageUUID → set of block types to strip from that message. const stripTargets = new Map>() + // Index of the last request-too-large anchor. Unlike the rejections above it + // says nothing about which block was at fault: the whole conversation was over + // an upstream byte limit. So media is dropped from everything the failed + // request carried, not just the turn before the anchor. It carries no + // sourceModel on purpose: a byte limit belongs to the provider or relay, so + // it must survive a model switch. + let mediaStrippedThrough = -1 for (let i = 0; i < reorderedMessages.length; i++) { const msg = reorderedMessages[i]! if (!isSyntheticApiErrorMessage(msg)) { @@ -2068,6 +2149,17 @@ export function normalizeMessagesForAPI( ) { continue } + const errorText = + Array.isArray(msg.message.content) && + msg.message.content[0]?.type === 'text' + ? msg.message.content[0].text + : undefined + if (isRequestTooLargeAnchor(msg, errorText)) { + // Anchors are visited in order, so the last one wins. + mediaStrippedThrough = i + continue + } + let blockTypesToStrip: Set | undefined const blockTypesFromCode = typeof msg.businessErrorCode === 'string' @@ -2078,11 +2170,6 @@ export function normalizeMessagesForAPI( } // Determine which legacy text error this is. - const errorText = - Array.isArray(msg.message.content) && - msg.message.content[0]?.type === 'text' - ? msg.message.content[0].text - : undefined if (!blockTypesToStrip && errorText) { blockTypesToStrip = errorToBlockTypes[errorText] } @@ -2114,6 +2201,16 @@ export function normalizeMessagesForAPI( } } + // Identity rather than uuid: attachment-derived messages (an @-mentioned + // image) are rebuilt on every normalization, so they have no stable uuid. + const mediaStripped = new Set() + for (let i = 0; i < mediaStrippedThrough; i++) { + const candidate = reorderedMessages[i]! + if (candidate.type === 'user' || candidate.type === 'attachment') { + mediaStripped.add(candidate) + } + } + const result: (UserMessage | AssistantMessage)[] = [] const assistantIndexByMessageId = new Map() let indexedResultLength = 0 @@ -2173,9 +2270,16 @@ export function normalizeMessagesForAPI( ) } + // Before the merge below: the rejected turn and the retry become + // adjacent once the synthetic error is filtered out, and only the + // rejected part may lose its media. + if (mediaStripped.has(message)) { + normalizedMessage = replaceMediaWithPlaceholders(normalizedMessage) + } + // Strip document/image blocks from the specific user message that - // preceded a PDF/image/request-too-large error, to prevent re-sending - // the problematic content on every subsequent API call. + // preceded a PDF/image error, to prevent re-sending the problematic + // content on every subsequent API call. const typesToStrip = stripTargets.get(normalizedMessage.uuid) if (typesToStrip) { const content = normalizedMessage.message.content @@ -2357,11 +2461,14 @@ export function normalizeMessagesForAPI( ) return } + const keptAttachmentMessage = mediaStripped.has(message) + ? rawAttachmentMessage.map(m => replaceMediaWithPlaceholders(m)) + : rawAttachmentMessage const attachmentMessage = checkStatsigFeatureGate_CACHED_MAY_BE_STALE( 'tengu_chair_sermon', ) - ? rawAttachmentMessage.map(ensureSystemReminderWrap) - : rawAttachmentMessage + ? keptAttachmentMessage.map(ensureSystemReminderWrap) + : keptAttachmentMessage // If the last message is also a user message, merge them const lastMessage = last(result) diff --git a/src/utils/model/configs.ts b/src/utils/model/configs.ts index 0895d68c..9948c9d7 100644 --- a/src/utils/model/configs.ts +++ b/src/utils/model/configs.ts @@ -62,6 +62,14 @@ export const CLAUDE_SONNET_5_CONFIG = { azureOpenAI: 'claude-sonnet-5', } as const satisfies ModelConfig +export const CLAUDE_SONNET_5_5_CONFIG = { + firstParty: 'claude-sonnet-5-5', + bedrock: 'us.anthropic.claude-sonnet-5-5', + vertex: 'claude-sonnet-5-5', + foundry: 'claude-sonnet-5-5', + azureOpenAI: 'claude-sonnet-5-5', +} as const satisfies ModelConfig + export const CLAUDE_SONNET_4_CONFIG = { firstParty: 'claude-sonnet-4-20250514', bedrock: 'us.anthropic.claude-sonnet-4-20250514-v1:0', @@ -178,6 +186,7 @@ export const ALL_MODEL_CONFIGS = { sonnet45: CLAUDE_SONNET_4_5_CONFIG, sonnet46: CLAUDE_SONNET_4_6_CONFIG, sonnet50: CLAUDE_SONNET_5_CONFIG, + sonnet55: CLAUDE_SONNET_5_5_CONFIG, opus40: CLAUDE_OPUS_4_CONFIG, opus41: CLAUDE_OPUS_4_1_CONFIG, opus45: CLAUDE_OPUS_4_5_CONFIG, diff --git a/src/utils/model/fable.test.ts b/src/utils/model/fable.test.ts index 8c10711a..d3c3eb71 100644 --- a/src/utils/model/fable.test.ts +++ b/src/utils/model/fable.test.ts @@ -129,7 +129,7 @@ describe('Fable model configuration', () => { expect(getMarketingNameForModel('claude-opus-4-8')).toBe('Opus 4.8') expect(getMarketingNameForModel('claude-sonnet-5')).toBe('Sonnet 5') expect(renderDefaultModelSetting('opusplan')).toBe( - 'Opus 5 in plan mode, else Sonnet 5', + 'Opus 5.5 in plan mode, else Sonnet 5.5', ) }) @@ -158,24 +158,48 @@ describe('Fable model configuration', () => { }) }) - test('publishes current model IDs and knowledge cutoffs in environment context', async () => { + test('publishes the current model lineup in environment context', async () => { expect(SKILL_MODEL_VARS).toMatchObject({ - OPUS_ID: 'claude-opus-4-8', - OPUS_NAME: 'Claude Opus 4.8', - SONNET_ID: 'claude-sonnet-5', - SONNET_NAME: 'Claude Sonnet 5', + OPUS_ID: 'claude-opus-5-5', + OPUS_NAME: 'Claude Opus 5.5', + SONNET_ID: 'claude-sonnet-5-5', + SONNET_NAME: 'Claude Sonnet 5.5', + PREV_SONNET_ID: 'claude-sonnet-5', }) - for (const model of ['claude-fable-5', 'claude-opus-4-8', 'claude-sonnet-5']) { + for (const model of ['claude-fable-5', 'claude-opus-4-8', 'claude-sonnet-5', 'claude-sonnet-5-5']) { const info = await computeSimpleEnvInfo(model) - expect(info).toContain('Assistant knowledge cutoff is January 2026.') - expect(info).toContain("Fable 5: 'claude-fable-5'") - expect(info).toContain("Opus 4.8: 'claude-opus-4-8'") - expect(info).toContain("Sonnet 5: 'claude-sonnet-5'") - expect(info).toContain('same Claude Opus 4.8 model') + expect(info).toContain('The most recent Claude models are the Claude 5 family and Haiku 4.5.') + expect(info).toContain("Fable 5.1: 'claude-fable-5-1'") + expect(info).toContain("Opus 5.5: 'claude-opus-5-5'") + expect(info).toContain("Sonnet 5.5: 'claude-sonnet-5-5'") + expect(info).toContain("Haiku 4.5: 'claude-haiku-4-5-20251001'") + // The fast-mode note no longer names an Opus generation that has since been superseded. + expect(info).toContain('Fast mode for Claude Code uses Claude Opus with faster output') + expect(info).not.toContain('same Claude Opus 4.8 model') } }) + // Reliable knowledge cutoffs: https://platform.claude.com/docs/en/models/overview and each + // model's page, which match the official Claude Code model catalog. + test.each([ + ['claude-fable-5-1', 'June 2026'], + ['claude-opus-5-5', 'June 2026'], + ['claude-sonnet-5-5', 'June 2026'], + ['claude-opus-5', 'May 2026'], + ['claude-fable-5', 'January 2026'], + ['claude-opus-4-8', 'January 2026'], + ['claude-opus-4-7', 'January 2026'], + ['claude-sonnet-5', 'January 2026'], + ['claude-sonnet-4-6', 'August 2025'], + ['claude-opus-4-5', 'May 2025'], + ['claude-haiku-4-5', 'February 2025'], + ])('tells %s its knowledge cutoff is %s', async (model, cutoff) => { + expect(await computeSimpleEnvInfo(model)).toContain(`Assistant knowledge cutoff is ${cutoff}.`) + // A context-window marker does not change what the model knows. + expect(await computeSimpleEnvInfo(`${model}[1m]`)).toContain(`Assistant knowledge cutoff is ${cutoff}.`) + }) + test('sanitizes new model trailers and keeps teammate fallbacks provider-safe', () => { expect(sanitizeModelName('claude-fable-5-1-experimental')).toBe('claude-fable-5-1') expect(sanitizeModelName('claude-fable-5-experimental')).toBe('claude-fable-5') diff --git a/src/utils/model/model.ts b/src/utils/model/model.ts index 90c8ffb4..fed3d9c9 100644 --- a/src/utils/model/model.ts +++ b/src/utils/model/model.ts @@ -22,7 +22,11 @@ import { } from '../context.js' import { isEnvTruthy } from '../envUtils.js' import { getModelStrings, resolveOverriddenModel } from './modelStrings.js' -import { formatModelPricing, getOpus46CostTier } from '../modelCost.js' +import { + formatModelPricing, + getModelPricingString, + getOpus46CostTier, +} from '../modelCost.js' import { getSettings_DEPRECATED } from '../settings/settings.js' import type { PermissionMode } from '../permissions/PermissionMode.js' import { @@ -144,7 +148,7 @@ export function getDefaultOpusModel(): ModelName { if (shouldUseThirdPartyAnthropicModelDefaults()) { return getModelStrings().opus46 } - return getModelStrings().opus50 + return getModelStrings().opus55 } // @[MODEL LAUNCH]: Update the default Sonnet model (3P providers may lag so keep defaults unchanged). @@ -159,7 +163,7 @@ export function getDefaultSonnetModel(): ModelName { if (shouldUseThirdPartyAnthropicModelDefaults()) { return getModelStrings().sonnet45 } - return getModelStrings().sonnet50 + return getModelStrings().sonnet55 } // @[MODEL LAUNCH]: Update the default Haiku model (3P providers may lag so keep defaults unchanged). @@ -296,6 +300,9 @@ export function firstPartyNameToCanonical(name: ModelName): ModelShortName { if (name.includes('claude-opus-4')) { return 'claude-opus-4' } + if (name.includes('claude-sonnet-5-5')) { + return 'claude-sonnet-5-5' + } if (name.includes('claude-sonnet-5')) { return 'claude-sonnet-5' } @@ -363,11 +370,11 @@ export function getClaudeAiUserDefaultModelDescription( } if (isMaxSubscriber() || isTeamPremiumSubscriber()) { if (isOpus1mMergeEnabled()) { - return `Opus 5 with 1M context · Most capable for complex work${fastMode ? getOpus46PricingSuffix(true) : ''}` + return `Opus 5.5 with 1M context · Most capable for complex work${fastMode ? getOpus46PricingSuffix(true) : ''}` } - return `Opus 5 · Most capable for complex work${fastMode ? getOpus46PricingSuffix(true) : ''}` + return `Opus 5.5 · Most capable for complex work${fastMode ? getOpus46PricingSuffix(true) : ''}` } - return 'Sonnet 5 · Best for everyday tasks' + return 'Sonnet 5.5 · Best for everyday tasks' } export function renderDefaultModelSetting( @@ -381,7 +388,12 @@ export function renderDefaultModelSetting( export function getOpus46PricingSuffix(fastMode: boolean): string { if (getAPIProvider() !== 'firstParty') return '' - const pricing = formatModelPricing(getOpus46CostTier(fastMode)) + // Standard rates follow whichever Opus the `opus` alias resolves to; only the fast-mode + // premium is still the tier published for Opus 4.7. + const pricing = fastMode + ? formatModelPricing(getOpus46CostTier(true)) + : (getModelPricingString(getDefaultOpusModel()) ?? + formatModelPricing(getOpus46CostTier(false))) const fastModeIndicator = fastMode ? ` (${LIGHTNING_BOLT})` : '' return ` ·${fastModeIndicator} ${pricing}` } @@ -461,6 +473,10 @@ export function getPublicModelDisplayName(model: ModelName): string | null { return 'Sonnet 4.6 (1M context)' case getModelStrings().sonnet46: return 'Sonnet 4.6' + case getModelStrings().sonnet55 + '[1m]': + return 'Sonnet 5.5 (1M context)' + case getModelStrings().sonnet55: + return 'Sonnet 5.5' case getModelStrings().sonnet50 + '[1m]': return 'Sonnet 5 (1M context)' case getModelStrings().sonnet50: @@ -712,6 +728,9 @@ export function getMarketingNameForModel(modelId: string): string | undefined { if (canonical.includes('claude-opus-4')) { return 'Opus 4' } + if (canonical.includes('claude-sonnet-5-5')) { + return has1m ? 'Sonnet 5.5 (with 1M context)' : 'Sonnet 5.5' + } if (canonical.includes('claude-sonnet-5')) { return has1m ? 'Sonnet 5 (with 1M context)' : 'Sonnet 5' } diff --git a/src/utils/model/modelContextWindows.ts b/src/utils/model/modelContextWindows.ts index 69d888b5..10938e56 100644 --- a/src/utils/model/modelContextWindows.ts +++ b/src/utils/model/modelContextWindows.ts @@ -9,6 +9,7 @@ const DIRECT_MODEL_CONTEXT_WINDOWS: Record = { 'claude-opus-5': 1_000_000, 'claude-opus-4-8': 1_000_000, 'claude-opus-4-7': 1_000_000, + 'claude-sonnet-5-5': 1_000_000, 'claude-sonnet-5': 1_000_000, 'claude-sonnet-4-6': 200_000, 'claude-haiku-4-5': 200_000, diff --git a/src/utils/model/modelOptions.ts b/src/utils/model/modelOptions.ts index b7b6490f..48d199ae 100644 --- a/src/utils/model/modelOptions.ts +++ b/src/utils/model/modelOptions.ts @@ -10,7 +10,7 @@ import { import { getModelStrings } from './modelStrings.js' import { COST_TIER_10_50, - COST_TIER_3_15, + COST_TIER_2_10, COST_HAIKU_35, COST_HAIKU_45, formatModelPricing, @@ -122,7 +122,7 @@ export function getDefaultOptionForUser(fastMode = false): ModelOption { ? '' : currentModel.includes('Opus') ? getOpus46PricingSuffix(fastMode) - : ` · ${formatModelPricing(COST_TIER_3_15)}` + : ` · ${formatModelPricing(COST_TIER_2_10)}` return { value: null, label: 'Default (recommended)', @@ -182,11 +182,11 @@ function getFable5Option(): ModelOption | undefined { function getSonnet46Option(): ModelOption { const is3P = shouldUseThirdPartyAnthropicOptions() - const modelName = is3P ? 'Sonnet 4.6' : 'Sonnet 5' + const modelName = is3P ? 'Sonnet 4.6' : 'Sonnet 5.5' return { value: is3P ? getModelStrings().sonnet46 : 'sonnet', label: 'Sonnet', - description: `${modelName} · Best for everyday tasks${is3P ? '' : ` · ${formatModelPricing(COST_TIER_3_15)}`}`, + description: `${modelName} · Best for everyday tasks${is3P ? '' : ` · ${formatModelPricing(COST_TIER_2_10)}`}`, descriptionForModel: `${modelName} - best for everyday tasks. Generally recommended for most coding tasks`, } @@ -223,18 +223,18 @@ function getOpus46Option(fastMode = false): ModelOption { return { value: is3P ? getModelStrings().opus46 : 'opus', label: 'Opus', - description: `${is3P ? 'Opus 4.7' : 'Opus 5'} · Most capable for complex work${getOpus46PricingSuffix(fastMode)}`, - descriptionForModel: `${is3P ? 'Opus 4.7' : 'Opus 5'} - most capable for complex work`, + description: `${is3P ? 'Opus 4.7' : 'Opus 5.5'} · Most capable for complex work${getOpus46PricingSuffix(fastMode)}`, + descriptionForModel: `${is3P ? 'Opus 4.7' : 'Opus 5.5'} - most capable for complex work`, } } export function getSonnet46_1MOption(): ModelOption { const is3P = shouldUseThirdPartyAnthropicOptions() - const modelName = is3P ? 'Sonnet 4.6' : 'Sonnet 5' + const modelName = is3P ? 'Sonnet 4.6' : 'Sonnet 5.5' return { value: is3P ? getModelStrings().sonnet46 + '[1m]' : 'sonnet[1m]', label: 'Sonnet (1M context)', - description: `${modelName} for long sessions${is3P ? '' : ` · ${formatModelPricing(COST_TIER_3_15)}`}`, + description: `${modelName} for long sessions${is3P ? '' : ` · ${formatModelPricing(COST_TIER_2_10)}`}`, descriptionForModel: `${modelName} with 1M context window - for long sessions with large codebases`, } @@ -242,7 +242,7 @@ export function getSonnet46_1MOption(): ModelOption { export function getOpus46_1MOption(fastMode = false): ModelOption { const is3P = shouldUseThirdPartyAnthropicOptions() - const modelName = is3P ? 'Opus 4.7' : 'Opus 5' + const modelName = is3P ? 'Opus 4.7' : 'Opus 5.5' return { value: is3P ? getModelStrings().opus46 + '[1m]' : 'opus[1m]', label: 'Opus (1M context)', @@ -302,7 +302,7 @@ function getMaxOpusOption(fastMode = false): ModelOption { return { value: 'opus', label: 'Opus', - description: `Opus 5 · Most capable for complex work${fastMode ? getOpus46PricingSuffix(true) : ''}`, + description: `Opus 5.5 · Most capable for complex work${fastMode ? getOpus46PricingSuffix(true) : ''}`, } } @@ -312,7 +312,7 @@ export function getMaxSonnet46_1MOption(): ModelOption { return { value: 'sonnet[1m]', label: 'Sonnet (1M context)', - description: `Sonnet 5 with 1M context${billingInfo}${is3P ? '' : ` · ${formatModelPricing(COST_TIER_3_15)}`}`, + description: `Sonnet 5.5 with 1M context${billingInfo}${is3P ? '' : ` · ${formatModelPricing(COST_TIER_2_10)}`}`, } } @@ -321,7 +321,7 @@ export function getMaxOpus46_1MOption(fastMode = false): ModelOption { return { value: 'opus[1m]', label: 'Opus (1M context)', - description: `Opus 5 with 1M context${billingInfo}${getOpus46PricingSuffix(fastMode)}`, + description: `Opus 5.5 with 1M context${billingInfo}${getOpus46PricingSuffix(fastMode)}`, } } @@ -330,16 +330,16 @@ function getMergedOpus1MOption(fastMode = false): ModelOption { return { value: is3P ? getModelStrings().opus46 + '[1m]' : 'opus[1m]', label: 'Opus (1M context)', - description: `${is3P ? 'Opus 4.7' : 'Opus 5'} with 1M context · Most capable for complex work${!is3P && fastMode ? getOpus46PricingSuffix(fastMode) : ''}`, + description: `${is3P ? 'Opus 4.7' : 'Opus 5.5'} with 1M context · Most capable for complex work${!is3P && fastMode ? getOpus46PricingSuffix(fastMode) : ''}`, descriptionForModel: - `${is3P ? 'Opus 4.7' : 'Opus 5'} with 1M context - most capable for complex work`, + `${is3P ? 'Opus 4.7' : 'Opus 5.5'} with 1M context - most capable for complex work`, } } const MaxSonnet46Option: ModelOption = { value: 'sonnet', label: 'Sonnet', - description: 'Sonnet 5 · Best for everyday tasks', + description: 'Sonnet 5.5 · Best for everyday tasks', } const MaxHaiku45Option: ModelOption = { @@ -476,8 +476,8 @@ function getModelOptionsBase(fastMode = false): ModelOption[] { return standardOptions } - // PAYG 1P API: Default (Opus 5) + Fable + Sonnet + Haiku. - // Opus 5 and Sonnet 5 already have native 1M context windows. + // PAYG 1P API: Default (Opus 5.5) + Fable + Sonnet + Haiku. + // Opus 5.x and Sonnet 5.x already have native 1M context windows. if (!shouldUseThirdPartyAnthropicOptions()) { const payg1POptions = [getDefaultOptionForUser(fastMode)] pushUniqueOption(payg1POptions, getFable5Option()) @@ -569,8 +569,13 @@ function getModelFamilyInfo( } } - // Opus family - if (canonical.includes('claude-opus-4')) { + // Opus family. A pinned Opus 5 only has a newer alias target where the alias has moved + // on to Opus 5.5; third-party defaults lag on Opus 4.7, which is not an upgrade. + if ( + canonical.includes('claude-opus-4') || + (canonical === 'claude-opus-5' && + getCanonicalName(getDefaultOpusModel()) === 'claude-opus-5-5') + ) { const currentName = getMarketingNameForModel(getDefaultOpusModel()) if (currentName) { return { alias: 'Opus', currentVersionName: currentName } diff --git a/src/utils/model/opus5.test.ts b/src/utils/model/opus5.test.ts index 80d033e2..f80539d7 100644 --- a/src/utils/model/opus5.test.ts +++ b/src/utils/model/opus5.test.ts @@ -6,11 +6,16 @@ import { clearOAuthTokenCache } from '../auth.js' import { resetSettingsCache } from '../settings/settingsCache.js' import { sanitizeModelName } from '../commitAttribution.js' import { ALL_MODEL_CONFIGS } from './configs.js' -import { getOpus46_1MOption, getMaxOpus46_1MOption } from './modelOptions.js' +import { + getModelOptions, + getOpus46_1MOption, + getMaxOpus46_1MOption, +} from './modelOptions.js' import { firstPartyNameToCanonical, getDefaultOpusModel, getMarketingNameForModel, + getOpus46PricingSuffix, getPublicModelDisplayName, isNonCustomOpusModel, parseUserSpecifiedModel, @@ -22,6 +27,7 @@ const envKeys = [ 'CLAUDE_CONFIG_DIR', 'ANTHROPIC_BASE_URL', 'ANTHROPIC_API_KEY', + 'ANTHROPIC_MODEL', 'ANTHROPIC_DEFAULT_OPUS_MODEL', 'CLAUDE_CODE_USE_BEDROCK', 'CLAUDE_CODE_USE_VERTEX', @@ -52,13 +58,15 @@ afterEach(() => { }) describe('Opus 5 runtime model identity', () => { - test('registers and resolves the latest official Opus without rewriting pinned models', () => { + test('registers Opus 5 and leaves a pinned Opus 5 alone now that the opus alias tracks Opus 5.5', () => { expect(Object.values(ALL_MODEL_CONFIGS).some(config => config.firstParty === 'claude-opus-5')).toBe(true) - expect(getDefaultOpusModel()).toBe('claude-opus-5') - expect(parseUserSpecifiedModel('opus')).toBe('claude-opus-5') - expect(parseUserSpecifiedModel('opus[1m]')).toBe('claude-opus-5[1m]') + expect(getDefaultOpusModel()).toBe('claude-opus-5-5') + expect(parseUserSpecifiedModel('opus')).toBe('claude-opus-5-5') + expect(parseUserSpecifiedModel('opus[1m]')).toBe('claude-opus-5-5[1m]') + expect(parseUserSpecifiedModel('claude-opus-5')).toBe('claude-opus-5') + expect(parseUserSpecifiedModel('claude-opus-5[1m]')).toBe('claude-opus-5[1m]') expect(parseUserSpecifiedModel('claude-opus-4-8')).toBe('claude-opus-4-8') - expect(renderDefaultModelSetting('opusplan')).toBe('Opus 5 in plan mode, else Sonnet 5') + expect(renderDefaultModelSetting('opusplan')).toBe('Opus 5.5 in plan mode, else Sonnet 5.5') }) test('preserves explicit provider overrides and third-party defaults', () => { @@ -70,13 +78,52 @@ describe('Opus 5 runtime model identity', () => { }) test('describes the Opus alias with its actual resolved generation', () => { - expect(getOpus46_1MOption().description).toContain('Opus 5') - expect(getMaxOpus46_1MOption().description).toContain('Opus 5') + expect(getOpus46_1MOption().description).toContain('Opus 5.5') + expect(getMaxOpus46_1MOption().description).toContain('Opus 5.5') process.env.ANTHROPIC_BASE_URL = 'https://provider.example.invalid' process.env.ANTHROPIC_API_KEY = 'fixture-key' expect(getOpus46_1MOption().description).toContain('Opus 4.7') }) + test('quotes the price of the Opus the alias resolves to', () => { + // Opus 5.5 is $4/$20; quoting the $5/$25 of Opus 5 or 4.7 would overstate it. + expect(getOpus46PricingSuffix(false)).toBe(' · $4/$20 per Mtok') + + process.env.ANTHROPIC_BASE_URL = 'https://provider.example.invalid' + process.env.ANTHROPIC_API_KEY = 'fixture-key' + // The lag-safe third-party default is Opus 4.7 ($5/$25). + expect(getOpus46PricingSuffix(false)).toBe(' · $5/$25 per Mtok') + + process.env.CLAUDE_CODE_USE_BEDROCK = '1' + expect(getOpus46PricingSuffix(false)).toBe('') + }) + + test('hints that a pinned Opus 5 has a newer alias target only where the alias moved to Opus 5.5', () => { + process.env.ANTHROPIC_API_KEY = 'fixture-key' + process.env.ANTHROPIC_MODEL = 'claude-opus-5' + expect(getModelOptions()).toContainEqual({ + value: 'claude-opus-5', + label: 'Opus 5', + description: 'Newer version available · select Opus for Opus 5.5', + }) + + process.env.ANTHROPIC_MODEL = 'claude-opus-5-5' + expect(getModelOptions()).toContainEqual({ + value: 'claude-opus-5-5', + label: 'Opus 5.5', + description: 'claude-opus-5-5', + }) + + // Third-party defaults lag on Opus 4.7, which is not an upgrade over a pinned Opus 5. + process.env.ANTHROPIC_BASE_URL = 'https://provider.example.invalid' + process.env.ANTHROPIC_MODEL = 'claude-opus-5' + expect(getModelOptions()).toContainEqual({ + value: 'claude-opus-5', + label: 'Opus 5', + description: 'claude-opus-5', + }) + }) + test('recognizes Opus 5 throughout canonicalization, public display, and attribution', () => { expect(firstPartyNameToCanonical('anthropic.claude-opus-5')).toBe('claude-opus-5') expect(isNonCustomOpusModel('claude-opus-5')).toBe(true) diff --git a/src/utils/model/sonnet55.test.ts b/src/utils/model/sonnet55.test.ts new file mode 100644 index 00000000..a6ecaf8e --- /dev/null +++ b/src/utils/model/sonnet55.test.ts @@ -0,0 +1,256 @@ +import { afterEach, beforeEach, describe, expect, test } from 'bun:test' +import { mkdtempSync, rmSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { computeSimpleEnvInfo } from '../../constants/prompts.js' +import { clearOAuthTokenCache } from '../auth.js' +import { sanitizeModelName } from '../commitAttribution.js' +import { + getContextWindowForModel, + getModelMaxOutputTokens, + modelSupports1M, +} from '../context.js' +import { + getDefaultEffortForModel, + modelSupportsEffort, + modelSupportsMaxEffort, + modelSupportsXHighEffort, +} from '../effort.js' +import { calculateCostFromTokens, getModelPricingString } from '../modelCost.js' +import { resetSettingsCache } from '../settings/settingsCache.js' +import { resolveSideQueryThinkingConfig } from '../sideQuery.js' +import { + modelRequiresThinking, + modelSupportsAdaptiveThinking, + modelSupportsThinking, + modelUsesBoundThinking, +} from '../thinking.js' +import { resolveModelCosts } from '../usageAccounting.js' +import { ALL_MODEL_CONFIGS } from './configs.js' +import { + firstPartyNameToCanonical, + getClaudeAiUserDefaultModelDescription, + getDefaultSonnetModel, + getMarketingNameForModel, + getPublicModelDisplayName, + parseUserSpecifiedModel, + renderDefaultModelSetting, +} from './model.js' +import { getMaxSonnet46_1MOption, getSonnet46_1MOption } from './modelOptions.js' +import { get3PModelCapabilityOverride } from './modelSupportOverrides.js' + +const envKeys = [ + 'HOME', + 'CLAUDE_CONFIG_DIR', + 'ANTHROPIC_BASE_URL', + 'ANTHROPIC_API_KEY', + 'ANTHROPIC_MODEL', + 'ANTHROPIC_DEFAULT_SONNET_MODEL', + 'ANTHROPIC_DEFAULT_SONNET_MODEL_SUPPORTED_CAPABILITIES', + 'CLAUDE_CODE_USE_BEDROCK', + 'CLAUDE_CODE_USE_VERTEX', + 'CLAUDE_CODE_USE_FOUNDRY', + 'CLAUDE_CODE_DISABLE_1M_CONTEXT', + 'CLAUDE_CODE_DISABLE_ADAPTIVE_THINKING', + 'CC_HAHA_SEND_DISABLED_THINKING', +] as const +let savedEnv: (string | undefined)[] +let temporaryHome: string + +function clearCapabilityCache() { + ;(get3PModelCapabilityOverride as typeof get3PModelCapabilityOverride & { + cache?: { clear?: () => void } + }).cache?.clear?.() +} + +beforeEach(() => { + savedEnv = envKeys.map(key => process.env[key]) + for (const key of envKeys) delete process.env[key] + temporaryHome = mkdtempSync(join(tmpdir(), 'sonnet55-model-test-')) + process.env.HOME = temporaryHome + process.env.CLAUDE_CONFIG_DIR = join(temporaryHome, 'config') + resetSettingsCache() + clearOAuthTokenCache() + clearCapabilityCache() +}) + +afterEach(() => { + resetSettingsCache() + clearOAuthTokenCache() + clearCapabilityCache() + rmSync(temporaryHome, { recursive: true, force: true }) + envKeys.forEach((key, index) => { + const value = savedEnv[index] + if (value === undefined) delete process.env[key] + else process.env[key] = value + }) +}) + +const SONNET_55_VARIANTS = [ + 'claude-sonnet-5-5', + 'claude-sonnet-5-5[1m]', + 'us.anthropic.claude-sonnet-5-5', + 'anthropic.claude-sonnet-5-5', +] + +describe('Sonnet 5.5 identity', () => { + test('registers the official IDs without absorbing the Sonnet 5 it supersedes', () => { + expect(ALL_MODEL_CONFIGS.sonnet55).toMatchObject({ + firstParty: 'claude-sonnet-5-5', + vertex: 'claude-sonnet-5-5', + foundry: 'claude-sonnet-5-5', + }) + expect(ALL_MODEL_CONFIGS.sonnet55.bedrock).toMatch(/anthropic\.claude-sonnet-5-5$/) + + for (const model of SONNET_55_VARIANTS) { + expect(firstPartyNameToCanonical(model)).toBe('claude-sonnet-5-5') + expect(sanitizeModelName(model)).toBe('claude-sonnet-5-5') + } + // `claude-sonnet-5` is a prefix of `claude-sonnet-5-5`; each keeps its own identity. + for (const model of ['claude-sonnet-5', 'anthropic.claude-sonnet-5']) { + expect(firstPartyNameToCanonical(model)).toBe('claude-sonnet-5') + expect(sanitizeModelName(model)).toBe('claude-sonnet-5') + } + }) + + test('renders distinct public and marketing names', () => { + expect(getPublicModelDisplayName('claude-sonnet-5-5')).toBe('Sonnet 5.5') + expect(getPublicModelDisplayName('claude-sonnet-5-5[1m]')).toBe('Sonnet 5.5 (1M context)') + expect(getMarketingNameForModel('claude-sonnet-5-5')).toBe('Sonnet 5.5') + expect(getMarketingNameForModel('claude-sonnet-5-5[1m]')).toBe('Sonnet 5.5 (with 1M context)') + expect(getPublicModelDisplayName('claude-sonnet-5')).toBe('Sonnet 5') + expect(getMarketingNameForModel('claude-sonnet-5')).toBe('Sonnet 5') + }) + + test('tells the model its own name, ID and June 2026 knowledge cutoff', async () => { + const info = await computeSimpleEnvInfo('claude-sonnet-5-5') + expect(info).toContain( + 'You are powered by the model named Sonnet 5.5. The exact model ID is claude-sonnet-5-5.', + ) + expect(info).toContain('Assistant knowledge cutoff is June 2026.') + }) +}) + +describe('Sonnet alias resolution', () => { + test('resolves the sonnet alias to Sonnet 5.5 and keeps explicit pins untouched', () => { + expect(getDefaultSonnetModel()).toBe('claude-sonnet-5-5') + expect(parseUserSpecifiedModel('sonnet')).toBe('claude-sonnet-5-5') + expect(parseUserSpecifiedModel('sonnet[1m]')).toBe('claude-sonnet-5-5[1m]') + expect(parseUserSpecifiedModel('claude-sonnet-5')).toBe('claude-sonnet-5') + expect(parseUserSpecifiedModel('claude-sonnet-5-5')).toBe('claude-sonnet-5-5') + expect(renderDefaultModelSetting('opusplan')).toBe('Opus 5.5 in plan mode, else Sonnet 5.5') + }) + + test('keeps lag-safe third-party defaults and honors provider overrides', () => { + process.env.ANTHROPIC_BASE_URL = 'https://provider.example.invalid' + process.env.ANTHROPIC_API_KEY = 'fixture-key' + expect(getDefaultSonnetModel()).toBe('claude-sonnet-4-5-20250929') + process.env.ANTHROPIC_DEFAULT_SONNET_MODEL = 'provider-custom-sonnet' + expect(parseUserSpecifiedModel('sonnet')).toBe('provider-custom-sonnet') + }) + + test('describes the alias with its actual resolved generation and price', () => { + expect(getClaudeAiUserDefaultModelDescription()).toBe('Sonnet 5.5 · Best for everyday tasks') + expect(getSonnet46_1MOption().description).toBe('Sonnet 5.5 for long sessions · $2/$10 per Mtok') + expect(getMaxSonnet46_1MOption().description).toBe('Sonnet 5.5 with 1M context · $2/$10 per Mtok') + + process.env.ANTHROPIC_BASE_URL = 'https://provider.example.invalid' + process.env.ANTHROPIC_API_KEY = 'fixture-key' + expect(getSonnet46_1MOption().description).toBe('Sonnet 4.6 for long sessions') + }) +}) + +describe('Sonnet 5.5 runtime limits and thinking', () => { + test('exposes the native 1M window and the 128K output limit', () => { + for (const model of SONNET_55_VARIANTS) { + expect(modelSupports1M(model)).toBe(true) + } + expect(getContextWindowForModel('claude-sonnet-5-5')).toBe(1_000_000) + expect(getContextWindowForModel('claude-sonnet-5-5[1m]')).toBe(1_000_000) + expect(getModelMaxOutputTokens('claude-sonnet-5-5')).toEqual({ + default: 128_000, + upperLimit: 128_000, + }) + expect(getModelMaxOutputTokens('claude-sonnet-5')).toEqual({ + default: 64_000, + upperLimit: 128_000, + }) + }) + + test('always runs adaptive thinking because the API rejects thinking: disabled', () => { + for (const model of SONNET_55_VARIANTS) { + expect(modelSupportsThinking(model)).toBe(true) + expect(modelSupportsAdaptiveThinking(model)).toBe(true) + expect(modelRequiresThinking(model)).toBe(true) + } + expect(modelUsesBoundThinking('claude-sonnet-5-5')).toBe(false) + // Sonnet 5 still accepts an explicit disable. + expect(modelRequiresThinking('claude-sonnet-5')).toBe(false) + }) + + test('never lets a side query send disabled thinking to Sonnet 5.5', () => { + expect(resolveSideQueryThinkingConfig(false, 1024, 'claude-sonnet-5-5')).toEqual({ + type: 'adaptive', + }) + expect(resolveSideQueryThinkingConfig(undefined, 1024, 'claude-sonnet-5-5[1m]')).toEqual({ + type: 'adaptive', + }) + expect(resolveSideQueryThinkingConfig(false, 1024, 'claude-sonnet-5')).toEqual({ + type: 'disabled', + }) + }) + + test('respects an explicit third-party capability opt-out', () => { + process.env.ANTHROPIC_BASE_URL = 'https://provider.example.invalid' + process.env.ANTHROPIC_API_KEY = 'fixture-key' + process.env.ANTHROPIC_DEFAULT_SONNET_MODEL = 'claude-sonnet-5-5' + process.env.ANTHROPIC_DEFAULT_SONNET_MODEL_SUPPORTED_CAPABILITIES = '' + clearCapabilityCache() + expect(modelRequiresThinking('claude-sonnet-5-5')).toBe(false) + expect(modelSupportsAdaptiveThinking('claude-sonnet-5-5')).toBe(false) + + // A provider that declares the capabilities keeps the required adaptive mode. + process.env.ANTHROPIC_DEFAULT_SONNET_MODEL_SUPPORTED_CAPABILITIES = + 'thinking,required_thinking,adaptive_thinking' + clearCapabilityCache() + expect(modelRequiresThinking('claude-sonnet-5-5')).toBe(true) + expect(modelSupportsAdaptiveThinking('claude-sonnet-5-5')).toBe(true) + }) + + test('defaults to medium effort while supporting every effort level', () => { + expect(getDefaultEffortForModel('claude-sonnet-5-5')).toBe('medium') + expect(getDefaultEffortForModel('claude-sonnet-5-5[1m]')).toBe('medium') + expect(modelSupportsEffort('claude-sonnet-5-5')).toBe(true) + expect(modelSupportsXHighEffort('claude-sonnet-5-5')).toBe(true) + expect(modelSupportsMaxEffort('claude-sonnet-5-5')).toBe(true) + }) +}) + +describe('Sonnet 5.5 pricing', () => { + test('uses the published $2/$10 pricing, $2.50 cache writes and $0.20 cached reads', () => { + expect(getModelPricingString('claude-sonnet-5-5')).toBe('$2/$10 per Mtok') + for (const model of SONNET_55_VARIANTS) { + expect( + calculateCostFromTokens(model, { + inputTokens: 1_000_000, + outputTokens: 1_000_000, + cacheReadInputTokens: 1_000_000, + cacheCreationInputTokens: 1_000_000, + }), + ).toBeCloseTo(14.7, 10) + } + }) + + test('prices activity stats at the Sonnet 5.5 rate instead of the Sonnet 5 prefix rate', () => { + const expected = { + inputTokens: 2, + outputTokens: 10, + promptCacheWriteTokens: 2.5, + promptCacheReadTokens: 0.2, + webSearchRequests: 0.01, + } + for (const model of ['claude-sonnet-5-5', 'anthropic/claude-sonnet-5-5', 'claude-sonnet-5-5-20260928']) { + expect(resolveModelCosts(model)).toEqual(expected) + } + }) +}) diff --git a/src/utils/modelCost.ts b/src/utils/modelCost.ts index 282e8827..2f3f25c8 100644 --- a/src/utils/modelCost.ts +++ b/src/utils/modelCost.ts @@ -16,9 +16,11 @@ import { CLAUDE_OPUS_4_8_CONFIG, CLAUDE_OPUS_4_CONFIG, CLAUDE_OPUS_5_5_CONFIG, + CLAUDE_OPUS_5_CONFIG, CLAUDE_SONNET_4_5_CONFIG, CLAUDE_SONNET_4_6_CONFIG, CLAUDE_SONNET_4_CONFIG, + CLAUDE_SONNET_5_5_CONFIG, CLAUDE_SONNET_5_CONFIG, } from './model/configs.js' import { @@ -37,7 +39,7 @@ export type ModelCosts = { webSearchRequests: number } -// Standard pricing tier for Sonnet models: $3 input / $15 output per Mtok +// Pricing tier for Sonnet 4.x and earlier: $3 input / $15 output per Mtok export const COST_TIER_3_15 = { inputTokens: 3, outputTokens: 15, @@ -46,6 +48,17 @@ export const COST_TIER_3_15 = { webSearchRequests: 0.01, } as const satisfies ModelCosts +// Pricing tier for Sonnet 5 and Sonnet 5.5: $2 input / $10 output per Mtok. +// Sonnet 5 launched at this price as introductory pricing; Anthropic made it the +// standard price and cancelled the announced $3/$15 increase, so do not restore it. +export const COST_TIER_2_10 = { + inputTokens: 2, + outputTokens: 10, + promptCacheWriteTokens: 2.5, + promptCacheReadTokens: 0.2, + webSearchRequests: 0.01, +} as const satisfies ModelCosts + // Pricing tier for Opus 4/4.1: $15 input / $75 output per Mtok export const COST_TIER_15_75 = { inputTokens: 15, @@ -88,6 +101,16 @@ export const COST_OPUS_55 = { webSearchRequests: 0.01, } as const satisfies ModelCosts +// Opus 5.5 fast mode (research preview): $8 input / $40 output per Mtok, with the cache +// multipliers stacked on top (1.25x writes, 0.05x reads on Opus 5.5). +export const COST_OPUS_55_FAST = { + inputTokens: 8, + outputTokens: 40, + promptCacheWriteTokens: 10, + promptCacheReadTokens: 0.4, + webSearchRequests: 0.01, +} as const satisfies ModelCosts + // Fast mode pricing for Opus 4.7: $30 input / $150 output per Mtok export const COST_TIER_30_150 = { inputTokens: 30, @@ -132,6 +155,7 @@ export function getOpus46CostTier(fastMode: boolean): ModelCosts { // Web search cost: $10 per 1000 requests = $0.01 per request export const MODEL_COSTS: Record = { [firstPartyNameToCanonical(CLAUDE_OPUS_5_5_CONFIG.firstParty)]: COST_OPUS_55, + [firstPartyNameToCanonical(CLAUDE_OPUS_5_CONFIG.firstParty)]: COST_TIER_5_25, [firstPartyNameToCanonical(CLAUDE_FABLE_5_1_CONFIG.firstParty)]: COST_FABLE_51, [firstPartyNameToCanonical(CLAUDE_FABLE_5_CONFIG.firstParty)]: @@ -150,8 +174,10 @@ export const MODEL_COSTS: Record = { COST_TIER_3_15, [firstPartyNameToCanonical(CLAUDE_SONNET_4_6_CONFIG.firstParty)]: COST_TIER_3_15, + [firstPartyNameToCanonical(CLAUDE_SONNET_5_5_CONFIG.firstParty)]: + COST_TIER_2_10, [firstPartyNameToCanonical(CLAUDE_SONNET_5_CONFIG.firstParty)]: - COST_TIER_3_15, + COST_TIER_2_10, [firstPartyNameToCanonical(CLAUDE_OPUS_4_CONFIG.firstParty)]: COST_TIER_15_75, [firstPartyNameToCanonical(CLAUDE_OPUS_4_1_CONFIG.firstParty)]: COST_TIER_15_75, diff --git a/src/utils/modelCostParity.test.ts b/src/utils/modelCostParity.test.ts new file mode 100644 index 00000000..7ddb119e --- /dev/null +++ b/src/utils/modelCostParity.test.ts @@ -0,0 +1,42 @@ +import { describe, expect, it } from 'bun:test' +import { ALL_MODEL_CONFIGS } from './model/configs.js' +import { firstPartyNameToCanonical } from './model/model.js' +import { MODEL_COSTS } from './modelCost.js' +import { resolveModelCosts } from './usageAccounting.js' + +// Two independent tables price Claude models: `MODEL_COSTS` (canonical-name lookup, used for the +// CLI's /cost) and the indexer's prefix table in usageAccounting (activity stats). A model added to +// only one, or a shorter prefix such as `claude-sonnet-5` shadowing `claude-sonnet-5-5`, misprices +// it without any error, so the two must agree for every model the app registers. +const claudeModels = Object.entries(ALL_MODEL_CONFIGS).filter(([, config]) => + config.firstParty.startsWith('claude-'), +) + +describe('Claude model pricing tables', () => { + it.each(claudeModels)('%s is priced by both tables, identically', (_key, config) => { + const canonical = firstPartyNameToCanonical(config.firstParty) + expect(MODEL_COSTS[canonical]).toBeDefined() + expect(resolveModelCosts(config.firstParty)).toEqual(MODEL_COSTS[canonical]!) + }) + + // https://platform.claude.com/docs/en/about-claude/pricing (checked 2026-09-29), $ per MTok. + it.each([ + ['claude-fable-5-1', 10, 50, 0.25], + ['claude-fable-5', 10, 50, 1], + ['claude-opus-5-5', 4, 20, 0.2], + ['claude-opus-5', 5, 25, 0.5], + ['claude-opus-4-8', 5, 25, 0.5], + ['claude-sonnet-5-5', 2, 10, 0.2], + ['claude-sonnet-5', 2, 10, 0.2], + ['claude-sonnet-4-6', 3, 15, 0.3], + ['claude-haiku-4-5', 1, 5, 0.1], + ])('%s matches the published input/output/cache-read prices', (model, input, output, cacheRead) => { + for (const costs of [MODEL_COSTS[firstPartyNameToCanonical(model)], resolveModelCosts(model)]) { + expect(costs).toMatchObject({ + inputTokens: input, + outputTokens: output, + promptCacheReadTokens: cacheRead, + }) + } + }) +}) diff --git a/src/utils/skills/__tests__/skillChangeDetector.test.ts b/src/utils/skills/__tests__/skillChangeDetector.test.ts index e1103db3..4fa24245 100644 --- a/src/utils/skills/__tests__/skillChangeDetector.test.ts +++ b/src/utils/skills/__tests__/skillChangeDetector.test.ts @@ -84,6 +84,7 @@ describe('skill change detector watch paths', () => { // chokidar on a non-existent path never fires, and would mask a real // directory appearing later; the loader tolerates the miss instead. await makeDir(tmpHome, '.claude', 'skills') + process.chdir(tmpHome) const paths = await getWatchablePaths() diff --git a/src/utils/thinking.ts b/src/utils/thinking.ts index fca8f4ae..74b112a8 100644 --- a/src/utils/thinking.ts +++ b/src/utils/thinking.ts @@ -86,6 +86,16 @@ export function getRainbowColor( return colors[charIndex % colors.length]! } +// Fable, Opus 5.5 and Sonnet 5.5 reject `thinking: disabled` and manual budgets; +// they always run adaptive thinking. +function isAdaptiveOnlyModel(canonical: string): boolean { + return ( + canonical.includes('claude-fable-5') || + canonical === 'claude-opus-5-5' || + canonical === 'claude-sonnet-5-5' + ) +} + // TODO(inigo): add support for probing unknown models via API error detection // Provider-aware thinking support detection (aligns with modelSupportsISP in betas.ts) export function modelSupportsThinking(model: string): boolean { @@ -101,9 +111,9 @@ export function modelSupportsThinking(model: string): boolean { // IMPORTANT: Do not change thinking support without notifying the model // launch DRI and research. This can greatly affect model quality and bashing. const canonical = getCanonicalName(model) - // Fable and Opus 5.5 use always-on adaptive thinking. Keep this after the provider + // Adaptive-only models use always-on adaptive thinking. Keep this after the provider // capability override so an explicitly incompatible 3P route can opt out. - if (canonical.includes('claude-fable-5') || canonical === 'claude-opus-5-5') { + if (isAdaptiveOnlyModel(canonical)) { return true } const provider = getAPIProvider() @@ -126,8 +136,7 @@ export function modelRequiresThinking(model: string): boolean { if (required3P !== undefined) { return required3P } - const canonical = getCanonicalName(model) - return canonical.includes('claude-fable-5') || canonical === 'claude-opus-5-5' + return isAdaptiveOnlyModel(getCanonicalName(model)) } /** Fable 5.1 binds replayed thinking to the preceding system, tools and history. */ @@ -154,9 +163,9 @@ export function modelSupportsAdaptiveThinking(model: string): boolean { return supported3P } const canonical = getCanonicalName(model) - // Fable and Opus 5.5 reject disabled/manual thinking and require adaptive thinking. + // Adaptive-only models reject disabled/manual thinking and require adaptive thinking. // Explicit 3P capability declarations above remain authoritative. - if (canonical.includes('claude-fable-5') || canonical === 'claude-opus-5-5') { + if (isAdaptiveOnlyModel(canonical)) { return true } const provider = getAPIProvider() diff --git a/src/utils/usageAccounting.test.ts b/src/utils/usageAccounting.test.ts index 0eea2cb8..81329e28 100644 --- a/src/utils/usageAccounting.test.ts +++ b/src/utils/usageAccounting.test.ts @@ -28,10 +28,33 @@ describe('resolveModelCosts', () => { expect(resolveModelCosts('claude-opus-4-8')?.inputTokens).toBe(5) expect(resolveModelCosts('claude-opus-4-1')?.inputTokens).toBe(15) expect(resolveModelCosts('claude-fable-5')?.outputTokens).toBe(50) - expect(resolveModelCosts('claude-sonnet-5')?.outputTokens).toBe(15) expect(resolveModelCosts('claude-haiku-4-5')?.outputTokens).toBe(5) }) + it('prices Sonnet 5 at its published $2/$10, which replaced the announced $3/$15', () => { + expect(resolveModelCosts('claude-sonnet-5')).toEqual({ + inputTokens: 2, + outputTokens: 10, + promptCacheWriteTokens: 2.5, + promptCacheReadTokens: 0.2, + webSearchRequests: 0.01, + }) + // Sonnet 4.x keeps the older Sonnet rate. + expect(resolveModelCosts('claude-sonnet-4-6')).toMatchObject({ inputTokens: 3, outputTokens: 15 }) + }) + + it('prices Opus 5.5 below Opus 5 rather than inheriting the shorter prefix', () => { + expect(resolveModelCosts('claude-opus-5-5')).toEqual({ + inputTokens: 4, + outputTokens: 20, + promptCacheWriteTokens: 5, + promptCacheReadTokens: 0.2, + webSearchRequests: 0.01, + }) + expect(resolveModelCosts('anthropic/claude-opus-5-5')).toEqual(resolveModelCosts('claude-opus-5-5')!) + expect(resolveModelCosts('claude-opus-5')).toMatchObject({ inputTokens: 5, outputTokens: 25 }) + }) + it('sees through the decorations gateways and dated snapshots add', () => { const opus = resolveModelCosts('claude-opus-4-8') expect(resolveModelCosts('claude-opus-4-8-r')).toEqual(opus!) @@ -44,6 +67,11 @@ describe('resolveModelCosts', () => { // `claude-opus-4-1` bills at the old Opus tier; a bare `claude-opus-4` must not swallow it. expect(resolveModelCosts('claude-opus-4-1')?.inputTokens).toBe(15) expect(resolveModelCosts('claude-opus-4-5')?.inputTokens).toBe(5) + // The same holds when a point release extends a whole-number model id. + expect(resolveModelCosts('claude-sonnet-5')?.inputTokens).toBe(2) + expect(resolveModelCosts('claude-sonnet-5-5')?.inputTokens).toBe(2) + expect(resolveModelCosts('claude-fable-5')?.promptCacheReadTokens).toBe(1) + expect(resolveModelCosts('claude-fable-5-1')?.promptCacheReadTokens).toBe(0.25) }) it('returns null for third-party models instead of guessing Claude rates', () => { @@ -69,7 +97,23 @@ describe('resolveModelCosts', () => { expect(resolveModelCosts('claude-opus-5', 'fast')?.inputTokens).toBe(10) expect(resolveModelCosts('claude-opus-5', 'standard')?.inputTokens).toBe(5) // Sonnet has no fast mode — a stray `speed` must not change what it costs. - expect(resolveModelCosts('claude-sonnet-5', 'fast')?.inputTokens).toBe(3) + expect(resolveModelCosts('claude-sonnet-5', 'fast')?.inputTokens).toBe(2) + expect(resolveModelCosts('claude-sonnet-5-5', 'fast')).toEqual(resolveModelCosts('claude-sonnet-5-5')!) + }) + + it('bills Opus 5.5 and Opus 4.8 fast mode at their published premiums', () => { + // https://platform.claude.com/docs/en/about-claude/pricing#fast-mode-pricing + expect(resolveModelCosts('claude-opus-5-5', 'fast')).toEqual({ + inputTokens: 8, + outputTokens: 40, + promptCacheWriteTokens: 10, + promptCacheReadTokens: 0.4, + webSearchRequests: 0.01, + }) + // Opus 5.5 must not inherit Opus 5's $10/$50 fast rate through the shorter prefix. + expect(resolveModelCosts('claude-opus-5', 'fast')).toMatchObject({ inputTokens: 10, outputTokens: 50 }) + expect(resolveModelCosts('claude-opus-4-8', 'fast')).toMatchObject({ inputTokens: 10, outputTokens: 50 }) + expect(resolveModelCosts('claude-opus-4-8', 'standard')).toMatchObject({ inputTokens: 5, outputTokens: 25 }) }) }) @@ -119,6 +163,37 @@ describe('estimateCostUSD', () => { expect(cost).toBeCloseTo(36.75, 10) }) + it('bills Sonnet 5 and the newer Sonnet 5.5 identically at $2/$10', () => { + const million = tokens({ + inputTokens: ONE_MILLION, + outputTokens: ONE_MILLION, + cacheReadInputTokens: ONE_MILLION, + cacheCreationInputTokens: ONE_MILLION, + }) + // 2 input + 10 output + 0.20 cache read + 2.50 cache write + expect(estimateCostUSD('claude-sonnet-5', million)).toBeCloseTo(14.7, 10) + expect(estimateCostUSD('claude-sonnet-5-5', million)).toBeCloseTo(14.7, 10) + }) + + it('bills Opus 5.5 at $4/$20 with the 5%-of-input cache read rate', () => { + // 4 input + 20 output + 0.20 cache read + 5 cache write + expect(estimateCostUSD('claude-opus-5-5', tokens({ + inputTokens: ONE_MILLION, + outputTokens: ONE_MILLION, + cacheReadInputTokens: ONE_MILLION, + cacheCreationInputTokens: ONE_MILLION, + }))).toBeCloseTo(29.2, 10) + expect(estimateCostUSD('claude-opus-5-5', tokens({ cacheReadInputTokens: ONE_MILLION }))) + .toBeCloseTo(0.2, 10) + }) + + it('bills fast-mode usage at the fast rate and standard usage at the standard rate', () => { + const million = tokens({ inputTokens: ONE_MILLION, outputTokens: ONE_MILLION }) + expect(estimateCostUSD('claude-opus-5-5', million, 'fast')).toBeCloseTo(48, 10) + expect(estimateCostUSD('claude-opus-5-5', million, 'standard')).toBeCloseTo(24, 10) + expect(estimateCostUSD('claude-opus-4-8', million, 'fast')).toBeCloseTo(60, 10) + }) + it('prices cache reads at a tenth of input, which is why token totals overstate spend', () => { const cacheRead = estimateCostUSD('claude-opus-5', tokens({ cacheReadInputTokens: ONE_MILLION })) const input = estimateCostUSD('claude-opus-5', tokens({ inputTokens: ONE_MILLION })) diff --git a/src/utils/usageAccounting.ts b/src/utils/usageAccounting.ts index 0ddf5b84..529e8d28 100644 --- a/src/utils/usageAccounting.ts +++ b/src/utils/usageAccounting.ts @@ -2,6 +2,9 @@ import { COST_FABLE_51, COST_HAIKU_35, COST_HAIKU_45, + COST_OPUS_55, + COST_OPUS_55_FAST, + COST_TIER_2_10, COST_TIER_3_15, COST_TIER_5_25, COST_TIER_10_50, @@ -44,6 +47,7 @@ export type PricedTokens = { // @[MODEL LAUNCH]: add the new model's canonical prefix here alongside its entry in MODEL_COSTS. // Longest match wins, so a bare family name may sit next to its versioned variants. const MODEL_TIERS: ReadonlyArray = [ + ['claude-opus-5-5', COST_OPUS_55], ['claude-opus-5', COST_TIER_5_25], ['claude-opus-4-8', COST_TIER_5_25], ['claude-opus-4-7', COST_TIER_5_25], @@ -54,7 +58,8 @@ const MODEL_TIERS: ReadonlyArray = ['claude-fable-5', COST_TIER_10_50], ['claude-fable-5-1', COST_FABLE_51], ['claude-mythos-5', COST_TIER_10_50], - ['claude-sonnet-5', COST_TIER_3_15], + ['claude-sonnet-5-5', COST_TIER_2_10], + ['claude-sonnet-5', COST_TIER_2_10], ['claude-sonnet-4-6', COST_TIER_3_15], ['claude-sonnet-4-5', COST_TIER_3_15], ['claude-sonnet-4', COST_TIER_3_15], @@ -68,7 +73,9 @@ const MODEL_TIERS: ReadonlyArray = // Fast mode bills at its own rate on the models that offer it; everything else ignores `speed`. const FAST_MODE_TIERS: ReadonlyArray = [ + ['claude-opus-5-5', COST_OPUS_55_FAST], ['claude-opus-5', COST_TIER_10_50], + ['claude-opus-4-8', COST_TIER_10_50], ['claude-opus-4-7', COST_TIER_30_150], ]