mirror of
https://github.com/NanmiCoder/claude-code-haha.git
synced 2026-10-10 11:53:10 +08:00
fix(agent-teams): keep long-running teams recoverable (#1426)
Long Agent Teams runs lost members for good: a truncated provider stream ended a member's turn with nobody to wake it, the desktop Stop button and every lead restart killed all members and marked the plan interrupted, mail sent to a stopped member landed in an inbox nothing read, and a lead kept inside one long turn never saw member reports. Aligned with the official CLI 2.1.284 and verified with DeepSeek Flash through a fault-injecting proxy. Stream recovery - Re-send a stream that breaks before any tool ran (proxy truncation, transport errors), with the existing retry budget and backoff; the desktop drops the discarded attempt's tool cards and todo update. Desktop team runtime (teamPlanRuntime) - The server supervises members: a stopped member restarts from its own transcript when messaged; transient failures continue automatically (15s/45s/2m/5m/10m) and only exhausted retries reach the lead; ready dependent tasks wake their owner; a crash-loop guard ignores user stops. - Stop pauses the team instead of ending it; the lead's next user message is followed by a notice listing the stopped members and their open tasks. Lead restarts (model/permission switch, crash) keep members; server restarts re-own the team. Teams end on /clear or session delete. - Approving a plan no longer races a concurrent plan read into "Launch ownership was lost". Mailbox and messaging - Atomic inbox writes, identity-based read marking, read history files, idle notifications with result/failureReason, and write failures reported instead of "Message sent". External builds keep the official between-turn delivery to the lead. - SendMessage resumes non-running in-process teammates from their transcript, notes restarting desktop members, queues mail for members of a plan awaiting approval, and rejects unknown names. CLI in-process teammates - Compaction uses the teammate's own controller and real history and no longer kills it on error; failed turns are classified and continued; the turn-end mailbox drains as one batch; one durable transcript per teammate. Lead behaviour - An unmet /goal ends the lead turn while members work, so member reports arrive; WaitSessions on own team members returns immediately. Desktop UI - Member states for stopped, auto-retrying and failed, with reason, countdown and recovery hint in all five locales. Tests and tooling - Regression tests for every behaviour above; module mocks in four test files are restored after use so the single-process coverage run is not polluted; the desktop smoke asserts the new Stop semantics.
This commit is contained in:
@@ -401,4 +401,49 @@ describe('AgentTeamsCanvas', () => {
|
||||
const lead = screen.queryByTestId('agent-teams-canvas-member-model-team-lead@canvas-team')
|
||||
expect(lead).toBeNull()
|
||||
})
|
||||
|
||||
it('labels stopped, retrying and failed members and keeps their failure on hover', () => {
|
||||
const current = snapshot('current')
|
||||
const recovering: TeamWorkbenchSnapshot = {
|
||||
...current,
|
||||
team: {
|
||||
...current.team,
|
||||
members: current.team.members.map(member => (
|
||||
member.agentId === 'builder@canvas-team'
|
||||
? {
|
||||
...member,
|
||||
status: 'idle' as const,
|
||||
activity: 'idle' as const,
|
||||
lastError: 'API Error: 529 overloaded',
|
||||
autoRetry: { attempt: 2, max: 5, nextAt: Date.parse('2026-08-12T02:30:45.000Z') },
|
||||
}
|
||||
: member.agentId === 'reviewer@canvas-team'
|
||||
? { ...member, status: 'idle' as const, activity: 'stopped' as const }
|
||||
: member.agentId === 'qa@canvas-team'
|
||||
? { ...member, status: 'error' as const, activity: 'idle' as const, lastError: 'Credit balance is too low' }
|
||||
: member
|
||||
)),
|
||||
},
|
||||
}
|
||||
|
||||
render(<AgentTeamsCanvas {...props({
|
||||
snapshots: [recovering],
|
||||
selectedIndex: 0,
|
||||
snapshot: recovering,
|
||||
previousSnapshot: undefined,
|
||||
activeMessageId: null,
|
||||
})} />)
|
||||
|
||||
const builder = screen.getByTestId('agent-teams-canvas-member-builder@canvas-team')
|
||||
expect(builder.getAttribute('data-member-state')).toBe('retrying')
|
||||
expect(screen.getByText('Auto-retry 2/5').getAttribute('title')).toBe('API Error: 529 overloaded')
|
||||
|
||||
const reviewer = screen.getByTestId('agent-teams-canvas-member-reviewer@canvas-team')
|
||||
expect(reviewer.getAttribute('data-member-state')).toBe('stopped')
|
||||
expect(reviewer.textContent).toContain('Stopped')
|
||||
|
||||
const qa = screen.getByTestId('agent-teams-canvas-member-qa@canvas-team')
|
||||
expect(qa.getAttribute('data-member-state')).toBe('error')
|
||||
expect(screen.getByText('Error').getAttribute('title')).toBe('Credit balance is too low')
|
||||
})
|
||||
})
|
||||
|
||||
@@ -117,11 +117,24 @@ function memberStatusColor(state: MemberWorkState): string {
|
||||
return 'var(--color-text-secondary)'
|
||||
}
|
||||
|
||||
function memberDotColor(state: MemberWorkState, accent: string): string {
|
||||
if (state === 'working') return accent
|
||||
// The raw warning accent is too light for label text, but reads as a fill.
|
||||
if (state === 'retrying') return 'var(--color-warning)'
|
||||
return memberStatusColor(state)
|
||||
}
|
||||
|
||||
function taskStateLabel(state: WorkbenchTaskState, t: TranslationFn): string {
|
||||
return t(`agentTeams.task.${state}` as TranslationKey)
|
||||
}
|
||||
|
||||
function memberStateLabel(state: MemberWorkState, t: TranslationFn): string {
|
||||
function memberStateLabel(state: MemberWorkState, member: TeamMember, t: TranslationFn): string {
|
||||
if (state === 'retrying') {
|
||||
return t('agentTeams.member.retrying', {
|
||||
attempt: member.autoRetry?.attempt ?? '?',
|
||||
max: member.autoRetry?.max ?? '?',
|
||||
})
|
||||
}
|
||||
return t(`agentTeams.member.${state}` as TranslationKey)
|
||||
}
|
||||
|
||||
@@ -466,10 +479,10 @@ function MemberNode({
|
||||
: waitingDependency
|
||||
? t('agentTeams.member.waitingForDependency', { task: waitingDependency })
|
||||
: t('agentTeams.member.waitingForTask')
|
||||
: memberStateLabel(state, t)
|
||||
: memberStateLabel(state, member, t)
|
||||
const characterClass = state === 'working'
|
||||
? 'agent-teams-character-working'
|
||||
: state === 'idle'
|
||||
: state === 'idle' || state === 'retrying'
|
||||
? 'agent-teams-character-idle'
|
||||
: state === 'exited'
|
||||
? 'agent-teams-character-archived'
|
||||
@@ -547,10 +560,15 @@ function MemberNode({
|
||||
{!isLead ? (
|
||||
<span
|
||||
className="h-1.5 w-1.5 shrink-0 rounded-full"
|
||||
style={{ backgroundColor: state === 'working' ? accent : memberStatusColor(state) }}
|
||||
style={{ backgroundColor: memberDotColor(state, accent) }}
|
||||
/>
|
||||
) : null}
|
||||
<span className="truncate">{stateLabel}</span>
|
||||
<span
|
||||
className="truncate"
|
||||
title={state === 'stopped' || state === 'retrying' || state === 'error' ? member.lastError : undefined}
|
||||
>
|
||||
{stateLabel}
|
||||
</span>
|
||||
{isLead ? (
|
||||
<span className="shrink-0 text-[var(--color-text-tertiary)]">
|
||||
· {t('agentTeams.member.inbox', { count: position.inbox })}
|
||||
|
||||
@@ -238,4 +238,141 @@ describe('AgentTeamsMemberInspector', () => {
|
||||
expect(cell.textContent).toBe('Unknown')
|
||||
expect(cell.textContent).not.toContain('undefined')
|
||||
})
|
||||
|
||||
describe('recovery states', () => {
|
||||
const now = Date.parse('2026-08-08T07:00:00.000Z')
|
||||
|
||||
function renderMember(member: TeamMember, clock?: number) {
|
||||
const base = snapshot('2026-08-08T07:00:00.000Z', 'in_progress')
|
||||
const frame: TeamWorkbenchSnapshot = {
|
||||
...base,
|
||||
team: { ...base.team, members: [member, reviewer] },
|
||||
}
|
||||
return render(
|
||||
<AgentTeamsMemberInspector
|
||||
snapshots={[frame]}
|
||||
selectedIndex={0}
|
||||
snapshot={frame}
|
||||
member={member}
|
||||
isLead={false}
|
||||
leadIsStreaming={false}
|
||||
now={clock}
|
||||
onBack={vi.fn()}
|
||||
onClose={vi.fn()}
|
||||
onOpenExecution={vi.fn()}
|
||||
/>,
|
||||
)
|
||||
}
|
||||
|
||||
it('says a stopped member resumes from its saved conversation when messaged', () => {
|
||||
const onOpenExecution = vi.fn()
|
||||
const frame = snapshot('2026-08-08T07:00:00.000Z', 'in_progress')
|
||||
const stopped: TeamMember = { ...builder, status: 'idle', activity: 'stopped' }
|
||||
render(
|
||||
<AgentTeamsMemberInspector
|
||||
snapshots={[frame]}
|
||||
selectedIndex={0}
|
||||
snapshot={frame}
|
||||
member={stopped}
|
||||
isLead={false}
|
||||
leadIsStreaming={false}
|
||||
onBack={vi.fn()}
|
||||
onClose={vi.fn()}
|
||||
onOpenExecution={onOpenExecution}
|
||||
/>,
|
||||
)
|
||||
|
||||
expect(screen.getByText('Stopped')).toBeTruthy()
|
||||
const notice = screen.getByTestId('agent-teams-member-recovery')
|
||||
expect(notice.getAttribute('data-member-state')).toBe('stopped')
|
||||
expect(screen.getByTestId('agent-teams-member-recovery-hint').textContent)
|
||||
.toBe('Send a message to resume it from its saved conversation')
|
||||
expect(screen.queryByTestId('agent-teams-member-last-error')).toBeNull()
|
||||
|
||||
// The way back is the member's own conversation, where the composer lives.
|
||||
const execution = screen.getByRole('button', { name: 'View builder execution' }) as HTMLButtonElement
|
||||
expect(execution.disabled).toBe(false)
|
||||
fireEvent.click(execution)
|
||||
expect(onOpenExecution).toHaveBeenCalledOnce()
|
||||
})
|
||||
|
||||
it('counts down to the next automatic retry and names the failure behind it', () => {
|
||||
const retrying: TeamMember = {
|
||||
...builder,
|
||||
status: 'idle',
|
||||
activity: 'idle',
|
||||
lastError: 'API Error: 529 overloaded',
|
||||
autoRetry: { attempt: 2, max: 5, nextAt: now + 45_000 },
|
||||
}
|
||||
const view = renderMember(retrying, now)
|
||||
|
||||
expect(screen.getByText('Auto-retry 2/5')).toBeTruthy()
|
||||
expect(screen.getByTestId('agent-teams-member-recovery').getAttribute('data-member-state'))
|
||||
.toBe('retrying')
|
||||
expect(screen.getByTestId('agent-teams-member-last-error').textContent)
|
||||
.toBe('API Error: 529 overloaded')
|
||||
expect(screen.getByTestId('agent-teams-member-recovery-hint').textContent)
|
||||
.toBe('Retrying automatically in 45s')
|
||||
|
||||
view.rerender(
|
||||
<AgentTeamsMemberInspector
|
||||
snapshots={[]}
|
||||
selectedIndex={0}
|
||||
snapshot={snapshot('2026-08-08T07:00:00.000Z', 'in_progress')}
|
||||
member={{ ...retrying, autoRetry: { attempt: 3, max: 5, nextAt: now + 125_000 } }}
|
||||
isLead={false}
|
||||
leadIsStreaming={false}
|
||||
now={now}
|
||||
onBack={vi.fn()}
|
||||
onClose={vi.fn()}
|
||||
onOpenExecution={vi.fn()}
|
||||
/>,
|
||||
)
|
||||
expect(screen.getByTestId('agent-teams-member-recovery-hint').textContent)
|
||||
.toBe('Retrying automatically in about 2 min')
|
||||
view.unmount()
|
||||
|
||||
// Past its schedule, the retry is due rather than counting negative.
|
||||
renderMember(retrying, now + 46_000)
|
||||
expect(screen.getByTestId('agent-teams-member-recovery-hint').textContent)
|
||||
.toBe('Retrying automatically now')
|
||||
})
|
||||
|
||||
it('shows no countdown while replaying, where the present clock means nothing', () => {
|
||||
renderMember({
|
||||
...builder,
|
||||
status: 'idle',
|
||||
activity: 'idle',
|
||||
lastError: 'API Error: 529 overloaded',
|
||||
autoRetry: { attempt: 1, max: 5, nextAt: now + 15_000 },
|
||||
})
|
||||
|
||||
expect(screen.getByText('Auto-retry 1/5')).toBeTruthy()
|
||||
expect(screen.getByTestId('agent-teams-member-last-error').textContent)
|
||||
.toBe('API Error: 529 overloaded')
|
||||
expect(screen.queryByTestId('agent-teams-member-recovery-hint')).toBeNull()
|
||||
})
|
||||
|
||||
it('keeps a long failure reason on one line with the full text on hover', () => {
|
||||
const reason = 'API Error: 400 {"type":"error","error":{"type":"invalid_request_error","message":"prompt is too long: 215000 tokens > 200000 maximum"}}'
|
||||
renderMember({ ...builder, status: 'error', activity: 'idle', lastError: reason })
|
||||
|
||||
expect(screen.getByText('Error')).toBeTruthy()
|
||||
expect(screen.getByTestId('agent-teams-member-recovery').getAttribute('data-member-state'))
|
||||
.toBe('error')
|
||||
const lastError = screen.getByTestId('agent-teams-member-last-error')
|
||||
expect(lastError.textContent).toBe(reason)
|
||||
expect(lastError.getAttribute('title')).toBe(reason)
|
||||
expect(lastError.className).toContain('truncate')
|
||||
expect(screen.getByTestId('agent-teams-member-recovery-hint').textContent)
|
||||
.toBe('Once the problem is fixed, send a message to let it continue')
|
||||
})
|
||||
|
||||
it('drops the failure notice while a failed member works its way back', () => {
|
||||
renderMember({ ...builder, status: 'error', activity: 'active', lastError: 'Credit balance is too low' })
|
||||
|
||||
expect(screen.getByText('Executing #7')).toBeTruthy()
|
||||
expect(screen.queryByTestId('agent-teams-member-recovery')).toBeNull()
|
||||
})
|
||||
})
|
||||
})
|
||||
|
||||
@@ -23,6 +23,7 @@ import { IconButton } from '@/components/ui/IconButton'
|
||||
import { useTranslation, type TranslationKey } from '@/i18n'
|
||||
import type {
|
||||
TeamMember,
|
||||
TeamMemberAutoRetry,
|
||||
TeamWorkbenchMessage,
|
||||
TeamWorkbenchSnapshot,
|
||||
TeamWorkbenchTask,
|
||||
@@ -35,6 +36,11 @@ export type AgentTeamsMemberInspectorProps = {
|
||||
member: TeamMember
|
||||
isLead: boolean
|
||||
leadIsStreaming: boolean
|
||||
/**
|
||||
* The live clock, for the time left until an automatic retry. Omitted while
|
||||
* replaying, where a countdown from the present would describe nothing.
|
||||
*/
|
||||
now?: number
|
||||
onBack: () => void
|
||||
onClose: () => void
|
||||
onOpenExecution: () => void
|
||||
@@ -190,10 +196,25 @@ function taskTone(state: WorkbenchTaskState): Tone {
|
||||
function memberTone(state: MemberWorkState): Tone {
|
||||
if (state === 'working') return 'brand'
|
||||
if (state === 'error') return 'danger'
|
||||
if (state === 'retrying') return 'warning'
|
||||
if (state === 'exited' || state === 'stopped') return 'neutral'
|
||||
return 'info'
|
||||
}
|
||||
|
||||
/** Container and foreground pairs, the same ones `Badge` uses for these tones. */
|
||||
function recoveryNoticeClasses(state: MemberWorkState): string {
|
||||
if (state === 'error') return 'bg-[var(--color-error-container)] text-[var(--color-on-error-container)]'
|
||||
if (state === 'retrying') return 'bg-[var(--color-warning-container)] text-[var(--color-on-warning-container)]'
|
||||
return 'bg-[var(--color-surface-container)] text-[var(--color-text-secondary)]'
|
||||
}
|
||||
|
||||
function retryCountdown(autoRetry: TeamMemberAutoRetry, now: number, t: TranslationFn): string {
|
||||
const seconds = Math.ceil((autoRetry.nextAt - now) / 1000)
|
||||
if (seconds <= 0) return t('agentTeams.member.retryNow')
|
||||
if (seconds < 60) return t('agentTeams.member.retryInSeconds', { n: seconds })
|
||||
return t('agentTeams.member.retryInMinutes', { n: Math.round(seconds / 60) })
|
||||
}
|
||||
|
||||
function leadStatusLabel(snapshot: TeamWorkbenchSnapshot, t: TranslationFn): string {
|
||||
if (snapshot.deletedAt) return t('agentTeams.lead.archived')
|
||||
if (snapshot.tasks.length === 0) return t('agentTeams.lead.forming')
|
||||
@@ -283,6 +304,7 @@ export function AgentTeamsMemberInspector({
|
||||
member,
|
||||
isLead,
|
||||
leadIsStreaming,
|
||||
now,
|
||||
onBack,
|
||||
onClose,
|
||||
onOpenExecution,
|
||||
@@ -324,7 +346,27 @@ export function AgentTeamsMemberInspector({
|
||||
? t('agentTeams.member.waitingForDependency', { task: waitingDependency })
|
||||
: workState === 'idle'
|
||||
? t('agentTeams.member.waitingForTask')
|
||||
: t(`agentTeams.member.${workState}` as TranslationKey)
|
||||
: workState === 'retrying'
|
||||
? t('agentTeams.member.retrying', {
|
||||
attempt: member.autoRetry?.attempt ?? '?',
|
||||
max: member.autoRetry?.max ?? '?',
|
||||
})
|
||||
: t(`agentTeams.member.${workState}` as TranslationKey)
|
||||
// Stopped, retrying and failed members all come back on their own or through
|
||||
// a message; say why they paused and what brings them back.
|
||||
const awaitsRecovery = !isLead && (
|
||||
workState === 'stopped' || workState === 'retrying' || workState === 'error'
|
||||
)
|
||||
const failureReason = awaitsRecovery ? member.lastError : undefined
|
||||
const recoveryHint = !awaitsRecovery
|
||||
? undefined
|
||||
: workState === 'stopped'
|
||||
? t('agentTeams.member.stoppedHint')
|
||||
: workState === 'error'
|
||||
? t('agentTeams.member.errorHint')
|
||||
: member.autoRetry && now !== undefined
|
||||
? retryCountdown(member.autoRetry, now, t)
|
||||
: undefined
|
||||
|
||||
return (
|
||||
<section
|
||||
@@ -424,6 +466,32 @@ export function AgentTeamsMemberInspector({
|
||||
<dd className="mt-0.5 truncate font-extrabold" data-testid="agent-teams-member-provider">{member.providerName || member.providerId || t('teamPlan.official')}</dd>
|
||||
</div>}
|
||||
</dl>
|
||||
|
||||
{failureReason || recoveryHint ? (
|
||||
<div
|
||||
data-testid="agent-teams-member-recovery"
|
||||
data-member-state={workState}
|
||||
className={`mt-3 min-w-0 rounded-[var(--radius-md)] px-2.5 py-2 text-[11px] leading-[1.45] ${recoveryNoticeClasses(workState)}`}
|
||||
>
|
||||
{failureReason ? (
|
||||
<p className="flex min-w-0 items-baseline gap-1.5">
|
||||
<span className="shrink-0 font-semibold">{t('agentTeams.inspector.failureReason')}</span>
|
||||
<span
|
||||
data-testid="agent-teams-member-last-error"
|
||||
className="min-w-0 truncate font-mono"
|
||||
title={failureReason}
|
||||
>
|
||||
{failureReason}
|
||||
</span>
|
||||
</p>
|
||||
) : null}
|
||||
{recoveryHint ? (
|
||||
<p data-testid="agent-teams-member-recovery-hint" className={failureReason ? 'mt-0.5' : undefined}>
|
||||
{recoveryHint}
|
||||
</p>
|
||||
) : null}
|
||||
</div>
|
||||
) : null}
|
||||
</header>
|
||||
|
||||
<div className="min-h-0 flex-1 overflow-y-auto overscroll-contain">
|
||||
|
||||
@@ -518,6 +518,7 @@ export function AgentTeamsWorkbench({ sessionId }: { sessionId: string }) {
|
||||
member={selectedMember}
|
||||
isLead={selectedMemberIsLead}
|
||||
leadIsStreaming={leadIsStreaming}
|
||||
now={followingLive ? now : undefined}
|
||||
onBack={() => setSelectedMemberId(null)}
|
||||
onClose={closeCommunication}
|
||||
onOpenExecution={openSelectedExecution}
|
||||
|
||||
@@ -2,6 +2,7 @@ import { describe, expect, it } from 'vitest'
|
||||
import type { TeamMember, TeamWorkbenchSnapshot, TeamWorkbenchTask } from '../../types/team'
|
||||
import {
|
||||
formatWorkbenchMessageTime,
|
||||
getMemberWorkState,
|
||||
getWorkbenchPhase,
|
||||
getWorkbenchProgress,
|
||||
getWorkbenchTaskState,
|
||||
@@ -543,4 +544,93 @@ describe('Agent Teams member model display', () => {
|
||||
|
||||
expect(merged?.team.members[0]?.model).toBe('claude-haiku-4-5')
|
||||
})
|
||||
|
||||
it('drops a failure an earlier frame recorded once a later frame clears it', () => {
|
||||
// The roster records a recovered member by leaving the failure out, so
|
||||
// replaying the earlier frame underneath must not resurrect it.
|
||||
const failed = memberSnapshot('2026-08-08T07:00:00.000Z', [{
|
||||
...worker,
|
||||
status: 'idle',
|
||||
activity: 'idle',
|
||||
lastError: 'API Error: 529 overloaded',
|
||||
autoRetry: { attempt: 1, max: 5, nextAt: Date.parse('2026-08-08T07:00:15.000Z') },
|
||||
}])
|
||||
const recovered = memberSnapshot('2026-08-08T07:05:00.000Z', [{
|
||||
...worker,
|
||||
status: 'idle',
|
||||
activity: 'idle',
|
||||
}])
|
||||
|
||||
const replayed = snapshotWithHistoricalMembers([failed, recovered], 1)!.team.members[0]!
|
||||
expect(replayed.lastError).toBeUndefined()
|
||||
expect(replayed.autoRetry).toBeUndefined()
|
||||
expect(getMemberWorkState(replayed)).toBe('idle')
|
||||
|
||||
const earlier = snapshotWithHistoricalMembers([failed, recovered], 0)!.team.members[0]!
|
||||
expect(getMemberWorkState(earlier)).toBe('retrying')
|
||||
})
|
||||
})
|
||||
|
||||
describe('getMemberWorkState', () => {
|
||||
const retry = { attempt: 2, max: 5, nextAt: Date.parse('2026-08-08T07:00:45.000Z') }
|
||||
|
||||
function member(overrides: Partial<TeamMember>): TeamMember {
|
||||
return {
|
||||
agentId: 'builder@team-a',
|
||||
name: 'builder',
|
||||
role: 'Frontend',
|
||||
status: 'running',
|
||||
...overrides,
|
||||
}
|
||||
}
|
||||
|
||||
it('keeps stopped, retrying and failed members on the team rather than idle or gone', () => {
|
||||
expect(getMemberWorkState(member({ status: 'idle', activity: 'stopped' }))).toBe('stopped')
|
||||
expect(getMemberWorkState(member({
|
||||
status: 'idle',
|
||||
activity: 'idle',
|
||||
lastError: 'Provider stream ended without message_stop',
|
||||
autoRetry: retry,
|
||||
}))).toBe('retrying')
|
||||
expect(getMemberWorkState(member({
|
||||
status: 'error',
|
||||
activity: 'idle',
|
||||
lastError: 'Credit balance is too low',
|
||||
}))).toBe('error')
|
||||
expect(getMemberWorkState(member({ status: 'completed', activity: 'exited' }))).toBe('exited')
|
||||
expect(getMemberWorkState(member({ status: 'idle', activity: 'exited' }))).toBe('exited')
|
||||
})
|
||||
|
||||
it('lets a live turn and a gone process outrank the failure record', () => {
|
||||
// The runtime keeps the reason until the turn a message started succeeds,
|
||||
// so a failed member that is mid-turn again is recovering, not failed.
|
||||
expect(getMemberWorkState(member({
|
||||
status: 'error',
|
||||
activity: 'active',
|
||||
lastError: 'Credit balance is too low',
|
||||
}))).toBe('working')
|
||||
// An automatic continuation in flight still carries its schedule.
|
||||
expect(getMemberWorkState(member({ activity: 'active', autoRetry: retry }))).toBe('working')
|
||||
// A stopped process never runs the retry it had scheduled, and a message,
|
||||
// not the failure, is what brings it back.
|
||||
expect(getMemberWorkState(member({ status: 'idle', activity: 'stopped', autoRetry: retry }))).toBe('stopped')
|
||||
expect(getMemberWorkState(member({
|
||||
status: 'error',
|
||||
activity: 'stopped',
|
||||
lastError: 'Restart failed: spawn ENOENT',
|
||||
}))).toBe('stopped')
|
||||
// Leaving the team stays final whatever else was recorded.
|
||||
expect(getMemberWorkState(member({
|
||||
status: 'completed',
|
||||
activity: 'stopped',
|
||||
lastError: 'Restart failed',
|
||||
autoRetry: retry,
|
||||
}))).toBe('exited')
|
||||
})
|
||||
|
||||
it('reads the lead from its own session, not from member turn markers', () => {
|
||||
const lead = member({ agentId: 'lead@team-a', activity: 'stopped', autoRetry: retry })
|
||||
expect(getMemberWorkState(lead, { isLead: true, leadIsStreaming: true })).toBe('working')
|
||||
expect(getMemberWorkState(lead, { isLead: true, leadIsStreaming: false })).toBe('idle')
|
||||
})
|
||||
})
|
||||
|
||||
@@ -216,7 +216,14 @@ export function snapshotWithHistoricalMembers(
|
||||
}
|
||||
}
|
||||
const [currentMember] = remaining.splice(currentIndex, 1)
|
||||
const merged = { ...historicalMember, ...currentMember! }
|
||||
const merged = {
|
||||
...historicalMember,
|
||||
...currentMember!,
|
||||
// A frame records its failure state by omission once it clears, so an
|
||||
// earlier frame's failure must not outlive the recovery.
|
||||
lastError: currentMember!.lastError,
|
||||
autoRetry: currentMember!.autoRetry,
|
||||
}
|
||||
// A later frame can land with `model: undefined`, which would clobber a
|
||||
// model already learned from an earlier snapshot. Restore the historical
|
||||
// model when the incoming member has no model of its own.
|
||||
@@ -269,7 +276,7 @@ export function resolveTeamMemberIdentity(
|
||||
return { member, isLead }
|
||||
}
|
||||
|
||||
export type MemberWorkState = 'working' | 'idle' | 'stopped' | 'exited' | 'error'
|
||||
export type MemberWorkState = 'working' | 'idle' | 'retrying' | 'stopped' | 'exited' | 'error'
|
||||
|
||||
/**
|
||||
* What the member itself is doing, which is never what its tasks say. A task
|
||||
@@ -280,15 +287,27 @@ export type MemberWorkState = 'working' | 'idle' | 'stopped' | 'exited' | 'error
|
||||
*
|
||||
* The lead has no runner writing turn markers for it, so its caller supplies
|
||||
* whether its session is streaming.
|
||||
*
|
||||
* Only `exited` is final. A stopped, retrying or failed member is still on the
|
||||
* team, and a message is what brings it back.
|
||||
*/
|
||||
export function getMemberWorkState(
|
||||
member: TeamMember,
|
||||
options: { isLead?: boolean; leadIsStreaming?: boolean } = {},
|
||||
): MemberWorkState {
|
||||
if (member.status === 'completed' || member.activity === 'exited') return 'exited'
|
||||
if (member.status === 'error') return 'error'
|
||||
if (options.isLead) return options.leadIsStreaming ? 'working' : 'idle'
|
||||
if (options.isLead) {
|
||||
if (member.status === 'error') return 'error'
|
||||
return options.leadIsStreaming ? 'working' : 'idle'
|
||||
}
|
||||
// A live turn outranks the failure record: the runtime keeps the reason the
|
||||
// previous turn failed until the turn a message started succeeds.
|
||||
if (member.activity === 'active') return 'working'
|
||||
// With its process gone, a scheduled retry will never fire and the failure
|
||||
// only explains why it stopped; a message is what restarts it.
|
||||
if (member.activity === 'stopped') return 'stopped'
|
||||
if (member.autoRetry) return 'retrying'
|
||||
if (member.status === 'error') return 'error'
|
||||
if (member.activity === 'idle') return 'idle'
|
||||
// `unknown` means the backend records no turn markers and left no transcript
|
||||
// to date, so fall back to the coarser roster status.
|
||||
|
||||
@@ -3250,6 +3250,7 @@ Row 9, all 8 cells: continuing from straight down, turning left through lower-le
|
||||
'agentTeams.inspector.taskHistoryHint': 'One teammate executes multiple tasks in sequence',
|
||||
'agentTeams.inspector.sent': 'Sent',
|
||||
'agentTeams.inspector.received': 'Received',
|
||||
'agentTeams.inspector.failureReason': 'Reason',
|
||||
'agentTeams.model.label': 'Model',
|
||||
'agentTeams.model.inheritFromLead': 'Inherit from lead · {model}',
|
||||
'agentTeams.model.unknown': 'Unknown',
|
||||
@@ -3289,7 +3290,13 @@ Row 9, all 8 cells: continuing from straight down, turning left through lower-le
|
||||
'agentTeams.member.idle': 'Idle',
|
||||
'agentTeams.member.stopped': 'Stopped',
|
||||
'agentTeams.member.exited': 'Exited',
|
||||
'agentTeams.member.error': 'Blocked',
|
||||
'agentTeams.member.error': 'Error',
|
||||
'agentTeams.member.retrying': 'Auto-retry {attempt}/{max}',
|
||||
'agentTeams.member.retryInSeconds': 'Retrying automatically in {n}s',
|
||||
'agentTeams.member.retryInMinutes': 'Retrying automatically in about {n} min',
|
||||
'agentTeams.member.retryNow': 'Retrying automatically now',
|
||||
'agentTeams.member.stoppedHint': 'Send a message to resume it from its saved conversation',
|
||||
'agentTeams.member.errorHint': 'Once the problem is fixed, send a message to let it continue',
|
||||
'agentTeams.openMember': 'Open {name} details',
|
||||
'agentTeams.resizeCommunication': 'Resize communication panel',
|
||||
'agentTeams.memberTranscriptLoading': 'Loading member transcript...',
|
||||
|
||||
@@ -3251,6 +3251,7 @@ export const jp: Record<TranslationKey, string> = {
|
||||
'agentTeams.inspector.taskHistoryHint': '1 人のメンバーが複数タスクを順番に実行',
|
||||
'agentTeams.inspector.sent': '送信',
|
||||
'agentTeams.inspector.received': '受信',
|
||||
'agentTeams.inspector.failureReason': '原因',
|
||||
'agentTeams.model.label': 'モデル',
|
||||
'agentTeams.model.inheritFromLead': 'リードから継承 · {model}',
|
||||
'agentTeams.model.unknown': '不明',
|
||||
@@ -3290,7 +3291,13 @@ export const jp: Record<TranslationKey, string> = {
|
||||
'agentTeams.member.idle': '待機中',
|
||||
'agentTeams.member.stopped': '停止済み',
|
||||
'agentTeams.member.exited': '退出済み',
|
||||
'agentTeams.member.error': 'ブロック中',
|
||||
'agentTeams.member.error': 'エラー',
|
||||
'agentTeams.member.retrying': '自動再試行 {attempt}/{max}',
|
||||
'agentTeams.member.retryInSeconds': '{n} 秒後に自動で再試行',
|
||||
'agentTeams.member.retryInMinutes': '約 {n} 分後に自動で再試行',
|
||||
'agentTeams.member.retryNow': 'まもなく自動で再試行',
|
||||
'agentTeams.member.stoppedHint': 'メッセージを送ると、保存された会話から再開します',
|
||||
'agentTeams.member.errorHint': '問題を解決してからメッセージを送ると、作業を再開します',
|
||||
'agentTeams.openMember': '{name} の詳細を開く',
|
||||
'agentTeams.resizeCommunication': '通信パネルの幅を変更',
|
||||
'agentTeams.memberTranscriptLoading': 'メンバーの transcript を読み込み中...',
|
||||
|
||||
@@ -3253,6 +3253,7 @@ export const kr: Record<TranslationKey, string> = {
|
||||
'agentTeams.inspector.taskHistoryHint': '한 멤버가 여러 작업을 순차 실행',
|
||||
'agentTeams.inspector.sent': '보냄',
|
||||
'agentTeams.inspector.received': '받음',
|
||||
'agentTeams.inspector.failureReason': '원인',
|
||||
'agentTeams.model.label': '모델',
|
||||
'agentTeams.model.inheritFromLead': '리드로부터 상속 · {model}',
|
||||
'agentTeams.model.unknown': '알 수 없음',
|
||||
@@ -3292,7 +3293,13 @@ export const kr: Record<TranslationKey, string> = {
|
||||
'agentTeams.member.idle': '대기 중',
|
||||
'agentTeams.member.stopped': '중지됨',
|
||||
'agentTeams.member.exited': '종료됨',
|
||||
'agentTeams.member.error': '차단됨',
|
||||
'agentTeams.member.error': '오류',
|
||||
'agentTeams.member.retrying': '자동 재시도 {attempt}/{max}',
|
||||
'agentTeams.member.retryInSeconds': '{n}초 후 자동 재시도',
|
||||
'agentTeams.member.retryInMinutes': '약 {n}분 후 자동 재시도',
|
||||
'agentTeams.member.retryNow': '곧 자동 재시도',
|
||||
'agentTeams.member.stoppedHint': '메시지를 보내면 저장된 대화에서 다시 시작합니다',
|
||||
'agentTeams.member.errorHint': '문제를 해결한 뒤 메시지를 보내면 작업을 이어갑니다',
|
||||
'agentTeams.openMember': '{name} 세부 정보 열기',
|
||||
'agentTeams.resizeCommunication': '통신 패널 너비 조정',
|
||||
'agentTeams.memberTranscriptLoading': '멤버 transcript 불러오는 중...',
|
||||
|
||||
@@ -3250,6 +3250,7 @@ export const zh: Record<TranslationKey, string> = {
|
||||
'agentTeams.inspector.taskHistoryHint': '一位成員依序執行多個任務',
|
||||
'agentTeams.inspector.sent': '送出',
|
||||
'agentTeams.inspector.received': '收到',
|
||||
'agentTeams.inspector.failureReason': '原因',
|
||||
'agentTeams.model.label': '模型',
|
||||
'agentTeams.model.inheritFromLead': '繼承主代理 · {model}',
|
||||
'agentTeams.model.unknown': '未知',
|
||||
@@ -3289,7 +3290,13 @@ export const zh: Record<TranslationKey, string> = {
|
||||
'agentTeams.member.idle': '待命',
|
||||
'agentTeams.member.stopped': '已停止',
|
||||
'agentTeams.member.exited': '已退出',
|
||||
'agentTeams.member.error': '卡住',
|
||||
'agentTeams.member.error': '出錯',
|
||||
'agentTeams.member.retrying': '自動重試 {attempt}/{max}',
|
||||
'agentTeams.member.retryInSeconds': '{n} 秒後自動重試',
|
||||
'agentTeams.member.retryInMinutes': '約 {n} 分鐘後自動重試',
|
||||
'agentTeams.member.retryNow': '即將自動重試',
|
||||
'agentTeams.member.stoppedHint': '發訊息即可從已儲存的對話恢復',
|
||||
'agentTeams.member.errorHint': '解決問題後,發訊息即可讓它繼續',
|
||||
'agentTeams.openMember': '開啟 {name} 的詳細資料',
|
||||
'agentTeams.resizeCommunication': '調整通訊面板寬度',
|
||||
'agentTeams.memberTranscriptLoading': '正在載入成員 transcript...',
|
||||
|
||||
@@ -3249,6 +3249,7 @@ export const zh: Record<TranslationKey, string> = {
|
||||
'agentTeams.inspector.taskHistoryHint': '一个成员串行执行多个任务',
|
||||
'agentTeams.inspector.sent': '发出',
|
||||
'agentTeams.inspector.received': '收到',
|
||||
'agentTeams.inspector.failureReason': '原因',
|
||||
'agentTeams.model.label': '模型',
|
||||
'agentTeams.model.inheritFromLead': '继承主代理 · {model}',
|
||||
'agentTeams.model.unknown': '未知',
|
||||
@@ -3288,7 +3289,13 @@ export const zh: Record<TranslationKey, string> = {
|
||||
'agentTeams.member.idle': '待命',
|
||||
'agentTeams.member.stopped': '已停止',
|
||||
'agentTeams.member.exited': '已退出',
|
||||
'agentTeams.member.error': '卡住',
|
||||
'agentTeams.member.error': '出错',
|
||||
'agentTeams.member.retrying': '自动重试 {attempt}/{max}',
|
||||
'agentTeams.member.retryInSeconds': '{n} 秒后自动重试',
|
||||
'agentTeams.member.retryInMinutes': '约 {n} 分钟后自动重试',
|
||||
'agentTeams.member.retryNow': '即将自动重试',
|
||||
'agentTeams.member.stoppedHint': '发消息即可从保存的对话恢复',
|
||||
'agentTeams.member.errorHint': '解决问题后,发消息即可让它继续',
|
||||
'agentTeams.openMember': '打开 {name} 的详情',
|
||||
'agentTeams.resizeCommunication': '调整通讯面板宽度',
|
||||
'agentTeams.memberTranscriptLoading': '正在加载成员 transcript...',
|
||||
|
||||
@@ -1670,6 +1670,49 @@ describe('SubagentRunPage', () => {
|
||||
expect(getMemberTranscriptMock).toHaveBeenCalledTimes(1)
|
||||
})
|
||||
|
||||
it.each([
|
||||
['stopped', { status: 'idle' as const, activity: 'stopped' as const }],
|
||||
['retrying', {
|
||||
status: 'idle' as const,
|
||||
activity: 'idle' as const,
|
||||
lastError: 'API Error: 529 overloaded',
|
||||
autoRetry: { attempt: 1, max: 5, nextAt: Date.parse('2026-08-09T00:00:15.000Z') },
|
||||
}],
|
||||
['failed', {
|
||||
status: 'error' as const,
|
||||
activity: 'idle' as const,
|
||||
lastError: 'Credit balance is too low',
|
||||
}],
|
||||
])('keeps the composer for a %s member, since a message is what brings it back', async (_state, recovery) => {
|
||||
getMemberTranscriptMock.mockResolvedValue({ messages: [] })
|
||||
const member = {
|
||||
agentId: 'builder@review-team',
|
||||
name: 'builder',
|
||||
role: 'builder',
|
||||
...recovery,
|
||||
}
|
||||
const team = {
|
||||
name: 'review-team',
|
||||
leadSessionId: 'lead-session',
|
||||
members: [member],
|
||||
}
|
||||
useTeamStore.setState({ activeTeam: team })
|
||||
useTeamStore.getState().openMemberSession(member, team)
|
||||
|
||||
render(
|
||||
<TeamMemberRunPage
|
||||
tabId="team-member:builder@review-team"
|
||||
leadSessionId="lead-session"
|
||||
agentId={member.agentId}
|
||||
title="builder"
|
||||
/>,
|
||||
)
|
||||
|
||||
expect(await screen.findByTestId('team-member-conversation')).toBeInTheDocument()
|
||||
expect(screen.queryByTestId('team-member-readonly-note')).not.toBeInTheDocument()
|
||||
expect(screen.getByRole('textbox')).toBeInTheDocument()
|
||||
})
|
||||
|
||||
it('settles a completed member task without losing direct-message activity', async () => {
|
||||
const member = {
|
||||
agentId: 'ui-designer@review-team',
|
||||
|
||||
@@ -25,6 +25,7 @@ const {
|
||||
fetchSessionTasksMock,
|
||||
clearTasksMock,
|
||||
setTasksFromTodosMock,
|
||||
restoreTasksMock,
|
||||
markCompletedAndDismissedMock,
|
||||
resetCompletedTasksMock,
|
||||
refreshTasksMock,
|
||||
@@ -54,6 +55,7 @@ const {
|
||||
fetchSessionTasksMock: vi.fn(),
|
||||
clearTasksMock: vi.fn(),
|
||||
setTasksFromTodosMock: vi.fn(),
|
||||
restoreTasksMock: vi.fn(),
|
||||
markCompletedAndDismissedMock: vi.fn(),
|
||||
resetCompletedTasksMock: vi.fn(async () => {}),
|
||||
refreshTasksMock: vi.fn(),
|
||||
@@ -172,8 +174,11 @@ vi.mock('./cliTaskStore', () => ({
|
||||
fetchSessionTasks: fetchSessionTasksMock,
|
||||
tasks: cliTaskStoreSnapshot.tasks,
|
||||
sessionId: cliTaskStoreSnapshot.sessionId,
|
||||
completedAndDismissed: false,
|
||||
dismissedCompletionKey: null,
|
||||
clearTasks: clearTasksMock,
|
||||
setTasksFromTodos: setTasksFromTodosMock,
|
||||
restoreTasks: restoreTasksMock,
|
||||
markCompletedAndDismissed: markCompletedAndDismissedMock,
|
||||
resetCompletedTasks: resetCompletedTasksMock,
|
||||
refreshTasks: refreshTasksMock,
|
||||
@@ -541,6 +546,7 @@ describe('chatStore history mapping', () => {
|
||||
fetchSessionTasksMock.mockReset()
|
||||
clearTasksMock.mockReset()
|
||||
setTasksFromTodosMock.mockReset()
|
||||
restoreTasksMock.mockReset()
|
||||
markCompletedAndDismissedMock.mockReset()
|
||||
resetCompletedTasksMock.mockReset()
|
||||
refreshTasksMock.mockReset()
|
||||
@@ -2150,6 +2156,134 @@ describe('chatStore history mapping', () => {
|
||||
},
|
||||
)
|
||||
|
||||
it('drops the tool calls a retried stream attempt finished streaming but never ran', () => {
|
||||
const send = (message: ServerMessage) =>
|
||||
useChatStore.getState().handleServerMessage(TEST_SESSION_ID, message)
|
||||
const planBeforeAttempt = [{ id: '1', subject: 'Earlier plan', status: 'in_progress' }]
|
||||
cliTaskStoreSnapshot.sessionId = TEST_SESSION_ID
|
||||
cliTaskStoreSnapshot.tasks = planBeforeAttempt
|
||||
useChatStore.setState({
|
||||
sessions: {
|
||||
[TEST_SESSION_ID]: makeSession({
|
||||
chatState: 'thinking',
|
||||
messages: [
|
||||
{ id: 'user-1', type: 'user_text', content: 'Fix the build', timestamp: 1 },
|
||||
{
|
||||
id: 'earlier-read',
|
||||
type: 'tool_use',
|
||||
toolName: 'Read',
|
||||
toolUseId: 'toolu_earlier_read',
|
||||
input: { file_path: 'build.log' },
|
||||
timestamp: 2,
|
||||
isPending: false,
|
||||
},
|
||||
{
|
||||
id: 'earlier-read-result',
|
||||
type: 'tool_result',
|
||||
toolUseId: 'toolu_earlier_read',
|
||||
content: 'error TS2322',
|
||||
isError: false,
|
||||
timestamp: 3,
|
||||
},
|
||||
// From an earlier turn, before this attempt: never this retry's to drop.
|
||||
{
|
||||
id: 'earlier-unresolved',
|
||||
type: 'tool_use',
|
||||
toolName: 'Bash',
|
||||
toolUseId: 'toolu_earlier_unresolved',
|
||||
input: { command: 'ls' },
|
||||
timestamp: 4,
|
||||
isPending: false,
|
||||
},
|
||||
],
|
||||
}),
|
||||
},
|
||||
})
|
||||
|
||||
send({ type: 'status', state: 'thinking', attemptStart: true })
|
||||
// The server completes a call at its block stop, before message_stop.
|
||||
send({ type: 'content_start', blockType: 'tool_use', toolName: 'TodoWrite', toolUseId: 'toolu_ghost_todo' })
|
||||
send({
|
||||
type: 'tool_use_complete',
|
||||
toolName: 'TodoWrite',
|
||||
toolUseId: 'toolu_ghost_todo',
|
||||
input: { todos: [{ content: 'Plan from the discarded attempt', status: 'in_progress' }] },
|
||||
})
|
||||
send({ type: 'content_start', blockType: 'tool_use', toolName: 'Bash', toolUseId: 'toolu_ghost_bash' })
|
||||
send({
|
||||
type: 'tool_use_complete',
|
||||
toolName: 'Bash',
|
||||
toolUseId: 'toolu_ghost_bash',
|
||||
input: { command: 'npm run build' },
|
||||
})
|
||||
// A call that did run keeps its card, as does a sub-agent's call still
|
||||
// running in the same window: only the root request is being retried.
|
||||
send({ type: 'tool_use_complete', toolName: 'Grep', toolUseId: 'toolu_ran', input: { pattern: 'TS2322' } })
|
||||
send({ type: 'tool_result', toolUseId: 'toolu_ran', content: 'src/a.ts', isError: false })
|
||||
send({
|
||||
type: 'tool_use_complete',
|
||||
toolName: 'Read',
|
||||
toolUseId: 'toolu_agent/toolu_child',
|
||||
input: { file_path: 'src/a.ts' },
|
||||
parentToolUseId: 'toolu_agent',
|
||||
})
|
||||
send({ type: 'content_start', blockType: 'tool_use', toolName: 'Edit', toolUseId: 'toolu_partial' })
|
||||
|
||||
send({ type: 'streaming_fallback', cause: 'stream_retry' })
|
||||
|
||||
const session = useChatStore.getState().sessions[TEST_SESSION_ID]
|
||||
const toolCards = session?.messages
|
||||
.filter((message): message is Extract<UIMessage, { type: 'tool_use' }> => message.type === 'tool_use')
|
||||
.map((message) => message.toolUseId)
|
||||
expect(toolCards).toEqual([
|
||||
'toolu_earlier_read',
|
||||
'toolu_earlier_unresolved',
|
||||
'toolu_ran',
|
||||
'toolu_agent/toolu_child',
|
||||
])
|
||||
expect(session?.messages.filter((message) => message.type === 'tool_result')).toHaveLength(2)
|
||||
expect(session?.chatState).toBe('thinking')
|
||||
// The plan the discarded TodoWrite showed early is taken back.
|
||||
expect(setTasksFromTodosMock).toHaveBeenCalledTimes(1)
|
||||
expect(restoreTasksMock).toHaveBeenCalledWith({
|
||||
sessionId: TEST_SESSION_ID,
|
||||
tasks: planBeforeAttempt,
|
||||
completedAndDismissed: false,
|
||||
dismissedCompletionKey: null,
|
||||
})
|
||||
})
|
||||
|
||||
it('keeps a TodoWrite that ran when a later stream attempt is retried', () => {
|
||||
const send = (message: ServerMessage) =>
|
||||
useChatStore.getState().handleServerMessage(TEST_SESSION_ID, message)
|
||||
cliTaskStoreSnapshot.sessionId = TEST_SESSION_ID
|
||||
useChatStore.setState({
|
||||
sessions: {
|
||||
[TEST_SESSION_ID]: makeSession({
|
||||
chatState: 'thinking',
|
||||
messages: [{ id: 'user-1', type: 'user_text', content: 'Plan the work', timestamp: 1 }],
|
||||
}),
|
||||
},
|
||||
})
|
||||
|
||||
send({ type: 'status', state: 'thinking', attemptStart: true })
|
||||
send({
|
||||
type: 'tool_use_complete',
|
||||
toolName: 'TodoWrite',
|
||||
toolUseId: 'toolu_plan',
|
||||
input: { todos: [{ content: 'Committed plan', status: 'in_progress' }] },
|
||||
})
|
||||
send({ type: 'tool_result', toolUseId: 'toolu_plan', content: 'Todos have been modified', isError: false })
|
||||
send({ type: 'status', state: 'thinking', attemptStart: true })
|
||||
send({ type: 'streaming_fallback', cause: 'stream_retry' })
|
||||
|
||||
expect(restoreTasksMock).not.toHaveBeenCalled()
|
||||
expect(useChatStore.getState().sessions[TEST_SESSION_ID]?.messages
|
||||
.filter((message) => message.type === 'tool_use')
|
||||
.map((message) => message.type === 'tool_use' ? message.toolUseId : ''))
|
||||
.toEqual(['toolu_plan'])
|
||||
})
|
||||
|
||||
it('treats the first history load as cold when a live task arrived before it started', async () => {
|
||||
const sessionId = 'cold-live-before-load'
|
||||
let resolveHistory!: (value: { messages: MessageEntry[] }) => void
|
||||
|
||||
@@ -11,7 +11,7 @@ import { subagentsApi } from '../api/subagents'
|
||||
import { useTeamPlanStore } from './teamPlanStore'
|
||||
import { useTeamStore } from './teamStore'
|
||||
import { useSessionStore } from './sessionStore'
|
||||
import { useCLITaskStore } from './cliTaskStore'
|
||||
import { useCLITaskStore, type CLITaskSnapshot } from './cliTaskStore'
|
||||
import { useWorkflowStore } from './workflowStore'
|
||||
import { useSessionRuntimeStore } from './sessionRuntimeStore'
|
||||
import { useProviderStore } from './providerStore'
|
||||
@@ -472,6 +472,12 @@ type ChatStore = {
|
||||
const TASK_TOOL_NAMES = new Set(['TaskCreate', 'TaskUpdate', 'TaskGet', 'TaskList', 'TodoWrite'])
|
||||
const TASK_STOP_TOOL_NAMES = new Set(['TaskStop', 'KillShell'])
|
||||
const pendingTaskToolUseIdsBySession = new Map<string, Set<string>>()
|
||||
/**
|
||||
* What the task bar showed before a live TodoWrite from a stream attempt that
|
||||
* has not committed. A `stream_retry` re-sends the whole request, so a write
|
||||
* the discarded attempt applied never happened and is put back.
|
||||
*/
|
||||
const taskBarBeforeAttemptBySession = new Map<string, CLITaskSnapshot>()
|
||||
const pendingToolParentUseIdsBySession = new Map<string, Map<string, string>>()
|
||||
type OwnedTaskRouteRegistration = {
|
||||
count: number
|
||||
@@ -856,15 +862,30 @@ function consumePendingTaskToolUseId(sessionId: string, toolUseId: string): bool
|
||||
|
||||
function clearPendingTaskToolUseIds(sessionId: string): void {
|
||||
pendingTaskToolUseIdsBySession.delete(sessionId)
|
||||
taskBarBeforeAttemptBySession.delete(sessionId)
|
||||
}
|
||||
|
||||
function consumeAllPendingTaskToolUseIds(sessionId: string): boolean {
|
||||
const hasPendingTaskTools =
|
||||
(pendingTaskToolUseIdsBySession.get(sessionId)?.size ?? 0) > 0
|
||||
pendingTaskToolUseIdsBySession.delete(sessionId)
|
||||
taskBarBeforeAttemptBySession.delete(sessionId)
|
||||
return hasPendingTaskTools
|
||||
}
|
||||
|
||||
/** Remember the task bar once, before the attempt's first live TodoWrite changes it. */
|
||||
function rememberTaskBarBeforeAttempt(sessionId: string): void {
|
||||
if (taskBarBeforeAttemptBySession.has(sessionId)) return
|
||||
const taskStore = useCLITaskStore.getState()
|
||||
if (taskStore.sessionId !== sessionId) return
|
||||
taskBarBeforeAttemptBySession.set(sessionId, {
|
||||
sessionId,
|
||||
tasks: taskStore.tasks,
|
||||
completedAndDismissed: taskStore.completedAndDismissed,
|
||||
dismissedCompletionKey: taskStore.dismissedCompletionKey,
|
||||
})
|
||||
}
|
||||
|
||||
function rememberPendingToolParentUseId(
|
||||
sessionId: string,
|
||||
toolUseId: string | null | undefined,
|
||||
@@ -4800,6 +4821,9 @@ export const useChatStore = create<ChatStore>((setState, get) => {
|
||||
}
|
||||
|
||||
case 'status':
|
||||
// A new attempt only starts once the previous one committed or was
|
||||
// retried, so its task bar change is no longer undoable.
|
||||
if (msg.attemptStart) taskBarBeforeAttemptBySession.delete(sessionId)
|
||||
update((session) => {
|
||||
const pendingText = `${session.streamingText}${consumePendingDelta(sessionId)}`
|
||||
const hasPendingStreamText =
|
||||
@@ -4979,6 +5003,7 @@ export const useChatStore = create<ChatStore>((setState, get) => {
|
||||
|
||||
case 'streaming_fallback': {
|
||||
if (msg.cause === 'stream_retry') {
|
||||
const taskBarBeforeAttempt = taskBarBeforeAttemptBySession.get(sessionId)
|
||||
consumePendingDelta(sessionId)
|
||||
clearPendingToolInputDelta(sessionId)
|
||||
clearPendingTaskToolUseIds(sessionId)
|
||||
@@ -4991,12 +5016,22 @@ export const useChatStore = create<ChatStore>((setState, get) => {
|
||||
session.messages.length,
|
||||
),
|
||||
)
|
||||
const resolvedToolUseIds = new Set(session.messages.flatMap((message) => (
|
||||
message.type === 'tool_result' ? [message.toolUseId] : []
|
||||
)))
|
||||
const messages = [
|
||||
...session.messages.slice(0, startIndex),
|
||||
...session.messages.slice(startIndex).filter((message) =>
|
||||
message.type !== 'assistant_text' &&
|
||||
message.type !== 'thinking' &&
|
||||
!(message.type === 'tool_use' && message.isPending)),
|
||||
!(message.type === 'tool_use' && (
|
||||
message.isPending ||
|
||||
// The retry re-sends the whole request, so none of the
|
||||
// attempt's own calls ran, even one the server already
|
||||
// completed at its block stop. Only the root request is
|
||||
// retried here; a sub-agent's call may still be running.
|
||||
(!message.parentToolUseId && !resolvedToolUseIds.has(message.toolUseId))
|
||||
))),
|
||||
]
|
||||
return {
|
||||
messages,
|
||||
@@ -5015,6 +5050,7 @@ export const useChatStore = create<ChatStore>((setState, get) => {
|
||||
statusVerb: '',
|
||||
}
|
||||
})
|
||||
if (taskBarBeforeAttempt) useCLITaskStore.getState().restoreTasks(taskBarBeforeAttempt)
|
||||
ensureElapsedTimer()
|
||||
useTabStore.getState().updateTabStatus(sessionId, 'running')
|
||||
break
|
||||
@@ -5190,6 +5226,7 @@ export const useChatStore = create<ChatStore>((setState, get) => {
|
||||
}
|
||||
})
|
||||
if (!parentToolUseId && toolName === 'TodoWrite' && Array.isArray((msg.input as any)?.todos)) {
|
||||
rememberTaskBarBeforeAttempt(sessionId)
|
||||
useCLITaskStore.getState().setTasksFromTodos((msg.input as any).todos, sessionId)
|
||||
} else if (!parentToolUseId && TASK_TOOL_NAMES.has(toolName)) {
|
||||
const useId = msg.toolUseId || session?.activeToolUseId
|
||||
|
||||
@@ -317,6 +317,62 @@ describe('cliTaskStore', () => {
|
||||
])
|
||||
})
|
||||
|
||||
it('puts back the bar a live TodoWrite replaced, newer than any read in flight', async () => {
|
||||
const before = [makeTask('session-1', 'completed')]
|
||||
useCLITaskStore.setState({
|
||||
sessionId: 'session-1',
|
||||
tasks: before,
|
||||
expanded: true,
|
||||
completedAndDismissed: true,
|
||||
dismissedCompletionKey: 'session-1::done',
|
||||
})
|
||||
let resolveRead!: (value: { tasks: CLITask[] }) => void
|
||||
vi.mocked(cliTasksApi.getTasksForList).mockReturnValueOnce(new Promise((resolve) => {
|
||||
resolveRead = resolve
|
||||
}))
|
||||
const staleRead = useCLITaskStore.getState().refreshTasks('session-1')
|
||||
useCLITaskStore.getState().setTasksFromTodos([
|
||||
{ content: 'Plan from a discarded attempt', status: 'in_progress' },
|
||||
], 'session-1')
|
||||
|
||||
useCLITaskStore.getState().restoreTasks({
|
||||
sessionId: 'session-1',
|
||||
tasks: before,
|
||||
completedAndDismissed: true,
|
||||
dismissedCompletionKey: 'session-1::done',
|
||||
})
|
||||
resolveRead({ tasks: [makeTask('session-1', 'pending')] })
|
||||
await staleRead
|
||||
|
||||
expect(useCLITaskStore.getState()).toMatchObject({
|
||||
tasks: before,
|
||||
completedAndDismissed: true,
|
||||
dismissedCompletionKey: 'session-1::done',
|
||||
})
|
||||
})
|
||||
|
||||
it('does not restore a snapshot onto a bar that now tracks another session', () => {
|
||||
useCLITaskStore.setState({
|
||||
sessionId: 'session-2',
|
||||
tasks: [makeTask('session-2')],
|
||||
completedAndDismissed: false,
|
||||
dismissedCompletionKey: null,
|
||||
})
|
||||
|
||||
useCLITaskStore.getState().restoreTasks({
|
||||
sessionId: 'session-1',
|
||||
tasks: [makeTask('session-1', 'completed')],
|
||||
completedAndDismissed: true,
|
||||
dismissedCompletionKey: 'session-1::done',
|
||||
})
|
||||
|
||||
expect(useCLITaskStore.getState()).toMatchObject({
|
||||
sessionId: 'session-2',
|
||||
tasks: [{ taskListId: 'session-2' }],
|
||||
completedAndDismissed: false,
|
||||
})
|
||||
})
|
||||
|
||||
it('does not reset completed tasks for a different session', async () => {
|
||||
useCLITaskStore.setState({
|
||||
sessionId: 'session-1',
|
||||
|
||||
@@ -29,6 +29,8 @@ type CLITaskStore = {
|
||||
refreshTasks: (sessionId?: string) => Promise<void>
|
||||
/** Update tasks from TodoWrite V1 tool input (in-memory, no disk read needed) */
|
||||
setTasksFromTodos: (todos: TodoItem[], sessionId?: string) => void
|
||||
/** Put back what the bar showed, unless it has moved on to another session */
|
||||
restoreTasks: (snapshot: CLITaskSnapshot) => void
|
||||
/** Mark that completed tasks were already dismissed (conversation continued) */
|
||||
markCompletedAndDismissed: (sessionId?: string) => void
|
||||
/** Clear a completed task list locally and remotely so the next cycle starts clean */
|
||||
@@ -39,6 +41,11 @@ type CLITaskStore = {
|
||||
toggleExpanded: () => void
|
||||
}
|
||||
|
||||
/** What the task bar showed for one session, to undo a live update that did not happen. */
|
||||
export type CLITaskSnapshot = Pick<CLITaskStore, 'tasks' | 'completedAndDismissed' | 'dismissedCompletionKey'> & {
|
||||
sessionId: string
|
||||
}
|
||||
|
||||
let taskRequestSequence = 0
|
||||
let taskRequestGeneration = 0
|
||||
const latestAppliedTaskRequestBySession = new Map<string, number>()
|
||||
@@ -197,6 +204,17 @@ export const useCLITaskStore = create<CLITaskStore>((set, get) => ({
|
||||
}))
|
||||
},
|
||||
|
||||
restoreTasks: (snapshot) => {
|
||||
if (get().sessionId !== snapshot.sessionId) return
|
||||
// Like a live TodoWrite, this is newer than any read still in flight.
|
||||
invalidateTaskRequests()
|
||||
set({
|
||||
tasks: snapshot.tasks,
|
||||
completedAndDismissed: snapshot.completedAndDismissed,
|
||||
dismissedCompletionKey: snapshot.dismissedCompletionKey,
|
||||
})
|
||||
},
|
||||
|
||||
markCompletedAndDismissed: (targetSessionId) => {
|
||||
const sessionId = targetSessionId ?? get().sessionId
|
||||
if (!sessionId || get().sessionId !== sessionId) return
|
||||
|
||||
@@ -7,7 +7,7 @@ import {
|
||||
} from './teamStore'
|
||||
import { registerAgentRunSession, useChatStore } from './chatStore'
|
||||
import { useTabStore } from './tabStore'
|
||||
import type { UIMessage } from '../types/chat'
|
||||
import type { TeamMemberStatus, UIMessage } from '../types/chat'
|
||||
import type { TeamWorkbenchSnapshot } from '../types/team'
|
||||
|
||||
const {
|
||||
@@ -1942,6 +1942,265 @@ describe('teamStore workbench timeline', () => {
|
||||
})
|
||||
})
|
||||
|
||||
describe('teamStore member recovery states', () => {
|
||||
const WORKER_ID = 'worker@team-workbench'
|
||||
const retry = { attempt: 2, max: 5, nextAt: 1_790_000_045_000 }
|
||||
|
||||
/** A workbench whose worker carries server-shaped fields, e.g. `status: 'failed'`. */
|
||||
function workbenchWithWorker(version: string, worker: Record<string, unknown>): TeamWorkbenchSnapshot {
|
||||
const base = workbench(version, 'in_progress')
|
||||
return {
|
||||
...base,
|
||||
team: {
|
||||
...base.team,
|
||||
members: [
|
||||
base.team.members[0]!,
|
||||
{ ...base.team.members[1]!, ...worker } as TeamWorkbenchSnapshot['team']['members'][number],
|
||||
],
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
function latestWorker() {
|
||||
return useTeamStore.getState().workbenchesBySession['lead-session']?.snapshots.at(-1)
|
||||
?.team.members.find((member) => member.agentId === WORKER_ID)
|
||||
}
|
||||
|
||||
beforeEach(() => {
|
||||
vi.useRealTimers()
|
||||
getMemberTranscriptMock.mockReset()
|
||||
getWorkbenchForSessionMock.mockReset()
|
||||
getWorkbenchMock.mockReset()
|
||||
getTeamMock.mockReset()
|
||||
sendMemberMessageMock.mockReset()
|
||||
sendMemberMessageMock.mockResolvedValue({ ok: true })
|
||||
useTeamStore.getState().clearTeam()
|
||||
useChatStore.setState({ sessions: {} })
|
||||
useTabStore.setState({ tabs: [], activeTabId: null })
|
||||
})
|
||||
|
||||
afterEach(() => {
|
||||
useTeamStore.getState().clearTeam()
|
||||
})
|
||||
|
||||
it('reads stopped, failed and retrying members from the team roster', async () => {
|
||||
getTeamMock.mockResolvedValue({
|
||||
name: 'team-recovery',
|
||||
leadAgentId: 'lead@team-recovery',
|
||||
members: [
|
||||
{
|
||||
agentId: 'crashed@team-recovery',
|
||||
name: 'crashed',
|
||||
status: 'failed',
|
||||
activity: 'stopped',
|
||||
lastError: 'Restart failed: spawn ENOENT',
|
||||
},
|
||||
{
|
||||
agentId: 'retrying@team-recovery',
|
||||
name: 'retrying',
|
||||
status: 'idle',
|
||||
activity: 'idle',
|
||||
lastError: 'Provider stream ended without message_stop',
|
||||
autoRetry: retry,
|
||||
},
|
||||
{
|
||||
// A malformed record must not invent a failure or a schedule.
|
||||
agentId: 'garbled@team-recovery',
|
||||
name: 'garbled',
|
||||
status: 'idle',
|
||||
activity: 'idle',
|
||||
lastError: ' ',
|
||||
autoRetry: { attempt: '2', max: 5, nextAt: 1 },
|
||||
},
|
||||
],
|
||||
})
|
||||
|
||||
await useTeamStore.getState().fetchTeamDetail('team-recovery')
|
||||
|
||||
const [crashed, retrying, garbled] = useTeamStore.getState().activeTeam?.members ?? []
|
||||
expect(crashed).toMatchObject({
|
||||
status: 'error',
|
||||
activity: 'stopped',
|
||||
lastError: 'Restart failed: spawn ENOENT',
|
||||
})
|
||||
expect(crashed?.autoRetry).toBeUndefined()
|
||||
expect(retrying).toMatchObject({
|
||||
status: 'idle',
|
||||
activity: 'idle',
|
||||
lastError: 'Provider stream ended without message_stop',
|
||||
autoRetry: retry,
|
||||
})
|
||||
expect(garbled?.lastError).toBeUndefined()
|
||||
expect(garbled?.autoRetry).toBeUndefined()
|
||||
})
|
||||
|
||||
it('carries recovery state through live and archived workbench snapshots', async () => {
|
||||
getWorkbenchMock.mockResolvedValue(workbenchWithWorker('v1', {
|
||||
status: 'failed',
|
||||
activity: 'idle',
|
||||
lastError: 'Credit balance is too low',
|
||||
}))
|
||||
await useTeamStore.getState().fetchWorkbench('team-workbench')
|
||||
expect(latestWorker()).toMatchObject({ status: 'error', lastError: 'Credit balance is too low' })
|
||||
|
||||
getWorkbenchForSessionMock.mockResolvedValue({
|
||||
sessionId: 'lead-session',
|
||||
teamName: 'team-workbench',
|
||||
source: 'archive',
|
||||
snapshots: [workbenchWithWorker('v2', { status: 'idle', activity: 'stopped', autoRetry: retry })],
|
||||
})
|
||||
await useTeamStore.getState().fetchTeamForSession('lead-session', { force: true })
|
||||
|
||||
// A snapshot that no longer records the failure has recovered from it.
|
||||
expect(latestWorker()).toMatchObject({ status: 'idle', activity: 'stopped', autoRetry: retry })
|
||||
expect(latestWorker()?.lastError).toBeUndefined()
|
||||
})
|
||||
|
||||
it('applies watcher patches to every view, clearing on null and keeping what was omitted', () => {
|
||||
const snapshot = workbench('v1', 'in_progress')
|
||||
const memberSession = memberSessionId(WORKER_ID)
|
||||
useTeamStore.setState({
|
||||
activeTeam: snapshot.team,
|
||||
memberTeamBySession: { [memberSession]: snapshot.team },
|
||||
workbenchesBySession: {
|
||||
'lead-session': { teamName: 'team-workbench', snapshots: [snapshot], loading: false, error: null },
|
||||
},
|
||||
})
|
||||
const workerViews = () => {
|
||||
const state = useTeamStore.getState()
|
||||
return [
|
||||
state.activeTeam,
|
||||
state.memberTeamBySession[memberSession],
|
||||
state.workbenchesBySession['lead-session']?.snapshots.at(-1)?.team,
|
||||
].map((team) => team?.members.find((member) => member.agentId === WORKER_ID))
|
||||
}
|
||||
const update = (patch: Partial<TeamMemberStatus>) => useTeamStore.getState().handleTeamUpdate(
|
||||
'team-workbench',
|
||||
[{ agentId: WORKER_ID, role: 'worker', status: 'idle', ...patch }],
|
||||
)
|
||||
|
||||
update({
|
||||
status: 'error',
|
||||
activity: 'stopped',
|
||||
lastError: 'Restart failed: spawn ENOENT',
|
||||
autoRetry: null,
|
||||
})
|
||||
for (const worker of workerViews()) {
|
||||
expect(worker).toMatchObject({
|
||||
status: 'error',
|
||||
activity: 'stopped',
|
||||
lastError: 'Restart failed: spawn ENOENT',
|
||||
})
|
||||
expect(worker?.autoRetry).toBeUndefined()
|
||||
}
|
||||
|
||||
update({ activity: 'idle', lastError: 'API Error: 529 overloaded', autoRetry: retry })
|
||||
for (const worker of workerViews()) {
|
||||
expect(worker).toMatchObject({
|
||||
status: 'idle',
|
||||
activity: 'idle',
|
||||
lastError: 'API Error: 529 overloaded',
|
||||
autoRetry: retry,
|
||||
})
|
||||
}
|
||||
|
||||
// A server that predates the failure fields says nothing about them.
|
||||
update({ activity: 'active', status: 'running' })
|
||||
for (const worker of workerViews()) {
|
||||
expect(worker).toMatchObject({ lastError: 'API Error: 529 overloaded', autoRetry: retry })
|
||||
}
|
||||
|
||||
update({ activity: 'idle', lastError: null, autoRetry: null })
|
||||
for (const worker of workerViews()) {
|
||||
expect(worker?.lastError).toBeUndefined()
|
||||
expect(worker?.autoRetry).toBeUndefined()
|
||||
}
|
||||
})
|
||||
|
||||
it('shows a failed member busy while the turn that brings it back runs', async () => {
|
||||
getMemberTranscriptMock.mockResolvedValue({ messages: [] })
|
||||
getWorkbenchMock.mockResolvedValueOnce(workbenchWithWorker('v1', {
|
||||
status: 'failed',
|
||||
activity: 'idle',
|
||||
lastError: 'Credit balance is too low',
|
||||
}))
|
||||
await useTeamStore.getState().fetchWorkbench('team-workbench')
|
||||
const sessionId = memberSessionId(WORKER_ID)
|
||||
await useTeamStore.getState().refreshMemberSession(sessionId)
|
||||
expect(useChatStore.getState().sessions[sessionId]?.chatState).toBe('idle')
|
||||
|
||||
// The runtime keeps the failure on record until the new turn succeeds.
|
||||
getWorkbenchMock.mockResolvedValueOnce(workbenchWithWorker('v2', {
|
||||
status: 'failed',
|
||||
activity: 'active',
|
||||
lastError: 'Credit balance is too low',
|
||||
}))
|
||||
await useTeamStore.getState().fetchWorkbench('team-workbench')
|
||||
expect(useChatStore.getState().sessions[sessionId]?.chatState).toBe('thinking')
|
||||
|
||||
getWorkbenchMock.mockResolvedValueOnce(workbenchWithWorker('v3', { status: 'idle', activity: 'idle' }))
|
||||
await useTeamStore.getState().fetchWorkbench('team-workbench')
|
||||
expect(useChatStore.getState().sessions[sessionId]?.chatState).toBe('idle')
|
||||
})
|
||||
|
||||
it('waits for a failed member to answer the message sent to recover it', async () => {
|
||||
getMemberTranscriptMock.mockResolvedValue({ messages: [] })
|
||||
getWorkbenchMock.mockResolvedValue(workbenchWithWorker('v1', {
|
||||
status: 'failed',
|
||||
activity: 'idle',
|
||||
lastError: 'Credit balance is too low',
|
||||
}))
|
||||
await useTeamStore.getState().fetchWorkbench('team-workbench')
|
||||
const sessionId = memberSessionId(WORKER_ID)
|
||||
await useTeamStore.getState().refreshMemberSession(sessionId)
|
||||
expect(getMemberTranscriptMock).toHaveBeenCalledTimes(1)
|
||||
|
||||
await useTeamStore.getState().sendMessageToMember(sessionId, 'Credits topped up, please continue')
|
||||
|
||||
expect(sendMemberMessageMock).toHaveBeenCalledWith(
|
||||
'team-workbench',
|
||||
WORKER_ID,
|
||||
'Credits topped up, please continue',
|
||||
)
|
||||
// The conversation keeps following the transcript, and the failure still
|
||||
// on record is the one this message answers, so it does not end the wait.
|
||||
expect(getMemberTranscriptMock).toHaveBeenCalledTimes(2)
|
||||
expect(useChatStore.getState().sessions[sessionId]?.chatState).toBe('thinking')
|
||||
|
||||
// A different failure since then means the recovery turn failed as well.
|
||||
useTeamStore.getState().handleTeamUpdate('team-workbench', [{
|
||||
agentId: WORKER_ID,
|
||||
role: 'worker',
|
||||
status: 'error',
|
||||
activity: 'idle',
|
||||
lastError: 'API Error: 401 invalid x-api-key',
|
||||
autoRetry: null,
|
||||
}])
|
||||
expect(useChatStore.getState().sessions[sessionId]?.chatState).toBe('idle')
|
||||
})
|
||||
|
||||
it('stops waiting for a reply once the member fails after the message', async () => {
|
||||
getMemberTranscriptMock.mockResolvedValue({ messages: [] })
|
||||
getWorkbenchMock.mockResolvedValue(workbenchWithWorker('v1', { status: 'idle', activity: 'idle' }))
|
||||
await useTeamStore.getState().fetchWorkbench('team-workbench')
|
||||
const sessionId = memberSessionId(WORKER_ID)
|
||||
await useTeamStore.getState().refreshMemberSession(sessionId)
|
||||
|
||||
await useTeamStore.getState().sendMessageToMember(sessionId, 'Please review the API routes')
|
||||
expect(useChatStore.getState().sessions[sessionId]?.chatState).toBe('thinking')
|
||||
|
||||
useTeamStore.getState().handleTeamUpdate('team-workbench', [{
|
||||
agentId: WORKER_ID,
|
||||
role: 'worker',
|
||||
status: 'error',
|
||||
activity: 'idle',
|
||||
lastError: 'Credit balance is too low',
|
||||
autoRetry: null,
|
||||
}])
|
||||
expect(useChatStore.getState().sessions[sessionId]?.chatState).toBe('idle')
|
||||
})
|
||||
})
|
||||
|
||||
describe('teamStore execution provider snapshots', () => {
|
||||
it('preserves the actual provider and model when hydrating team detail', async () => {
|
||||
useTeamStore.getState().clearTeam()
|
||||
|
||||
@@ -6,6 +6,7 @@ import type {
|
||||
TeamDetail,
|
||||
TeamMember,
|
||||
TeamMemberActivity,
|
||||
TeamMemberAutoRetry,
|
||||
AgentColor,
|
||||
TeamWorkbenchSnapshot,
|
||||
TeamWorkbenchTimeline,
|
||||
@@ -62,6 +63,13 @@ type AwaitingMemberReply = {
|
||||
content: string
|
||||
sentAt: number
|
||||
baselineMessageKeys: Set<string>
|
||||
/**
|
||||
* The failure on record when the message went to a failed member (`null`
|
||||
* when it had no reason). The message answers that failure, which stays on
|
||||
* record until the runtime delivers the message, so only a different one
|
||||
* ends the wait.
|
||||
*/
|
||||
failureAtSend?: string | null
|
||||
}
|
||||
const awaitingMemberReplies = new Map<string, AwaitingMemberReply[]>()
|
||||
const teamLifecycleGenerations = new Map<string, number>()
|
||||
@@ -114,7 +122,8 @@ function createMemberSessionState() {
|
||||
}
|
||||
|
||||
function normalizeMemberStatus(status: string | undefined): TeamMember['status'] {
|
||||
if (status === 'running' || status === 'idle' || status === 'completed') {
|
||||
// The watcher reports a failed member as `error`, the REST roster as `failed`.
|
||||
if (status === 'running' || status === 'idle' || status === 'completed' || status === 'error') {
|
||||
return status
|
||||
}
|
||||
return status === 'failed' ? 'error' : 'idle'
|
||||
@@ -124,11 +133,26 @@ function normalizeMemberActivity(activity: unknown): TeamMemberActivity | undefi
|
||||
return activity === 'active' ||
|
||||
activity === 'idle' ||
|
||||
activity === 'exited' ||
|
||||
activity === 'stopped' ||
|
||||
activity === 'unknown'
|
||||
? activity
|
||||
: undefined
|
||||
}
|
||||
|
||||
function normalizeMemberLastError(value: unknown): string | undefined {
|
||||
return typeof value === 'string' && value.trim() ? value : undefined
|
||||
}
|
||||
|
||||
function normalizeMemberAutoRetry(value: unknown): TeamMemberAutoRetry | undefined {
|
||||
if (!value || typeof value !== 'object') return undefined
|
||||
const { attempt, max, nextAt } = value as Record<string, unknown>
|
||||
return typeof attempt === 'number' && Number.isFinite(attempt) &&
|
||||
typeof max === 'number' && Number.isFinite(max) &&
|
||||
typeof nextAt === 'number' && Number.isFinite(nextAt)
|
||||
? { attempt, max, nextAt }
|
||||
: undefined
|
||||
}
|
||||
|
||||
function toTeamMember(raw: Record<string, unknown>): TeamMember {
|
||||
return {
|
||||
agentId: (raw.agentId as string) || '',
|
||||
@@ -141,6 +165,9 @@ function toTeamMember(raw: Record<string, unknown>): TeamMember {
|
||||
'',
|
||||
status: normalizeMemberStatus(raw.status as string | undefined),
|
||||
activity: normalizeMemberActivity(raw.activity),
|
||||
// The roster omits both once they clear, so a fresh read always replaces them.
|
||||
lastError: normalizeMemberLastError(raw.lastError),
|
||||
autoRetry: normalizeMemberAutoRetry(raw.autoRetry),
|
||||
currentTask: raw.currentTask as string | undefined,
|
||||
color: raw.color as AgentColor | undefined,
|
||||
sessionId: raw.sessionId as string | undefined,
|
||||
@@ -243,6 +270,14 @@ function mergeTeamMemberStatuses(
|
||||
// The watcher omits whatever it could not determine from the roster
|
||||
// alone, so an absent field must not erase what a full team read knew.
|
||||
activity: normalizeMemberActivity(member.activity) ?? existing.activity,
|
||||
// The failure fields are always sent, as `null` once they clear; only a
|
||||
// server that predates them leaves them out.
|
||||
lastError: member.lastError === undefined
|
||||
? existing.lastError
|
||||
: normalizeMemberLastError(member.lastError),
|
||||
autoRetry: member.autoRetry === undefined
|
||||
? existing.autoRetry
|
||||
: normalizeMemberAutoRetry(member.autoRetry),
|
||||
currentTask: member.currentTask ?? existing.currentTask,
|
||||
color: existing.color ?? AGENT_COLORS[index % AGENT_COLORS.length]!,
|
||||
}
|
||||
@@ -470,12 +505,15 @@ function latestWorkbenchSnapshotForTeam(
|
||||
* This needs positive evidence, unlike the workbench figure: an unresolved
|
||||
* activity would leave a spinner running forever, and a member that really is
|
||||
* mid-turn still reads as busy through its unsettled replies.
|
||||
*
|
||||
* A failed member is not excluded: a member whose failure is still on record
|
||||
* while it is active is already working on the message that answers it.
|
||||
*/
|
||||
function memberIsWorking(
|
||||
member: TeamMember,
|
||||
snapshot: TeamWorkbenchSnapshot | undefined,
|
||||
): boolean {
|
||||
if (member.status === 'completed' || member.status === 'error' || snapshot?.deletedAt) {
|
||||
if (member.status === 'completed' || snapshot?.deletedAt) {
|
||||
return false
|
||||
}
|
||||
return member.activity === 'active'
|
||||
@@ -781,9 +819,8 @@ function syncMemberSessionMessages(
|
||||
requestedTaskUpdatedAt?: Map<string, number>,
|
||||
requestedStreamRevision?: number,
|
||||
) {
|
||||
const isTerminal = member.status === 'completed' ||
|
||||
member.status === 'error' ||
|
||||
Boolean(snapshot?.deletedAt)
|
||||
const isTerminal = member.status === 'completed' || Boolean(snapshot?.deletedAt)
|
||||
const failure = member.status === 'error' ? member.lastError ?? null : undefined
|
||||
const awaitingReplies = awaitingMemberReplies.get(sessionId) ?? []
|
||||
const unsettledReplies: AwaitingMemberReply[] = []
|
||||
if (!isTerminal) {
|
||||
@@ -794,7 +831,10 @@ function syncMemberSessionMessages(
|
||||
awaitingReplies[laterIndex]!.baselineMessageKeys.add(settlement.promptKey)
|
||||
}
|
||||
}
|
||||
if (!settlement.settled) unsettledReplies.push(reply)
|
||||
// A failure recorded after the message went out ends the wait: the turn
|
||||
// that would have answered it is over.
|
||||
const failedSince = failure !== undefined && failure !== reply.failureAtSend
|
||||
if (!settlement.settled && !failedSince) unsettledReplies.push(reply)
|
||||
})
|
||||
}
|
||||
if (unsettledReplies.length === 0) {
|
||||
@@ -802,10 +842,7 @@ function syncMemberSessionMessages(
|
||||
} else if (unsettledReplies.length !== awaitingReplies.length) {
|
||||
awaitingMemberReplies.set(sessionId, unsettledReplies)
|
||||
}
|
||||
const isActive = !isTerminal && (
|
||||
memberIsWorking(member, snapshot) ||
|
||||
unsettledReplies.length > 0
|
||||
)
|
||||
const isActive = memberIsWorking(member, snapshot) || unsettledReplies.length > 0
|
||||
useChatStore.getState().applyBoundedUpdate((state) => {
|
||||
const existing = state.sessions[sessionId]
|
||||
const nextState = existing ?? createMemberSessionState()
|
||||
@@ -1615,6 +1652,7 @@ export const useTeamStore = create<TeamStore>((set, get) => ({
|
||||
.filter((message) => message.id !== optimisticMessage?.id)
|
||||
.map(memberMessageKey),
|
||||
),
|
||||
...(member.status === 'error' ? { failureAtSend: member.lastError ?? null } : {}),
|
||||
}
|
||||
awaitingMemberReplies.set(sessionId, [
|
||||
...(awaitingMemberReplies.get(sessionId) ?? []),
|
||||
@@ -1629,11 +1667,12 @@ export const useTeamStore = create<TeamStore>((set, get) => ({
|
||||
? latestWorkbenchSnapshotForTeam(get().workbenchesBySession, currentTeam) ??
|
||||
get().memberSnapshotBySession[sessionId]
|
||||
: undefined
|
||||
// A message is what brings a failed member back, so its conversation
|
||||
// keeps following the transcript like any other member's.
|
||||
const sessionIsAvailable = Boolean(currentTeam && currentMember) &&
|
||||
memberIncarnationKey(currentTeam!, currentMember!) === incarnationKey &&
|
||||
currentMember?.agentId === member.agentId &&
|
||||
currentMember.status !== 'completed' &&
|
||||
currentMember.status !== 'error' &&
|
||||
!snapshot?.deletedAt
|
||||
if (!sessionIsAvailable) {
|
||||
removeAwaitingMemberReply(sessionId, awaitingReply)
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
import type { PermissionMode } from './settings'
|
||||
import type { RuntimeSelection } from './runtime'
|
||||
import type { TeamMemberActivity, TeamMemberAutoRetry } from './team'
|
||||
|
||||
// Source: src/server/ws/events.ts
|
||||
|
||||
@@ -223,8 +224,12 @@ export type TeamMemberStatus = {
|
||||
* Omitted when the watcher cannot tell from the roster alone, so a receiver
|
||||
* keeps whatever the last full team read established.
|
||||
*/
|
||||
activity?: 'active' | 'idle' | 'exited' | 'unknown'
|
||||
activity?: TeamMemberActivity
|
||||
currentTask?: string
|
||||
/** Why the last turn failed; `null` says the member has recovered. */
|
||||
lastError?: string | null
|
||||
/** A pending automatic continuation; `null` once it has settled. */
|
||||
autoRetry?: TeamMemberAutoRetry | null
|
||||
}
|
||||
|
||||
export type ComputerUseGrantFlags = {
|
||||
|
||||
@@ -13,8 +13,20 @@ export type TeamSummary = {
|
||||
* an `in_progress` task is evidence of neither: a teammate marks a task started
|
||||
* and can then end its turn, and an umbrella task stays open underneath every
|
||||
* turn it covers.
|
||||
*
|
||||
* `stopped` is a member whose process is gone (the user's Stop, a crash, a
|
||||
* closed lead). Unlike `exited`, it is still on the team: a message restarts it
|
||||
* from its saved conversation.
|
||||
*/
|
||||
export type TeamMemberActivity = 'active' | 'idle' | 'exited' | 'unknown'
|
||||
export type TeamMemberActivity = 'active' | 'idle' | 'exited' | 'stopped' | 'unknown'
|
||||
|
||||
/** An automatic continuation the server scheduled after a transient provider failure. */
|
||||
export type TeamMemberAutoRetry = {
|
||||
attempt: number
|
||||
max: number
|
||||
/** Epoch milliseconds. */
|
||||
nextAt: number
|
||||
}
|
||||
|
||||
export type TeamMember = {
|
||||
agentId: string
|
||||
@@ -22,6 +34,9 @@ export type TeamMember = {
|
||||
role: string
|
||||
status: 'running' | 'idle' | 'completed' | 'error'
|
||||
activity?: TeamMemberActivity
|
||||
/** Why the member's last turn failed. Kept while an automatic retry is pending. */
|
||||
lastError?: string
|
||||
autoRetry?: TeamMemberAutoRetry
|
||||
currentTask?: string
|
||||
color?: AgentColor
|
||||
sessionId?: string
|
||||
|
||||
@@ -91,6 +91,8 @@ Claude 每完成一轮并修改文件,对话里会出现一张「{n} 个文件
|
||||
|
||||
后台跑的子 Agent 的工具活动也会冒泡到这里,不用等它跑完才知道它在干什么。
|
||||
|
||||
团队成员遇到模型服务断流、限流、5xx 这类临时故障会自己重试,成员行显示「自动重试 2/5」,不用你或主控介入;重试用完,或者遇到需要你处理的错误(比如 API Key 失效),成员行显示「出错」并通知主控。按停止按钮会让整组成员一起停下,但进度不会丢:你下一次给主控发消息时,它会知道哪些成员停在了哪个任务上,再按你的意思决定要不要让它们继续;你也可以直接给某个成员发消息,它会从保存的对话接着干,显示「已停止」的成员也一样。切换模型或权限模式不会打断团队;重启应用后团队还在,给成员发消息即可继续。删除会话或执行 `/clear` 才会结束团队。
|
||||
|
||||
## 输入框能干的事
|
||||
|
||||

|
||||
|
||||
@@ -91,6 +91,8 @@ The first button on the right of the tab bar opens the Activity panel, which lis
|
||||
|
||||
Tool activity from background subagents bubbles up here too, so you don't have to wait for one to finish to see what it's doing.
|
||||
|
||||
Team members retry on their own when the model service drops a stream, rate-limits, or returns a 5xx; the member row shows "Auto-retry 2/5" and neither you nor the lead has to step in. When the retries run out, or the error needs you (an expired API key, say), the row shows "Error" and the lead is told. The Stop button halts the whole team without losing progress: your next message to the lead tells it which members stopped on which tasks, and it decides from your words whether they carry on. You can also message a member directly; it picks up from its saved conversation, and the same goes for a member marked "Stopped". Switching the model or permission mode doesn't interrupt the team, and after an app restart the team is still there — message a member to continue. Deleting the session or running `/clear` ends the team.
|
||||
|
||||
## What the composer can do
|
||||
|
||||

|
||||
|
||||
@@ -109,10 +109,15 @@ export async function runDesktopUiTeamPlanSmoke(options: {
|
||||
await browserStep(['wait', '--fn', '!document.querySelector("[data-testid=team-plan-open]") && !document.querySelector("[data-testid=team-plan-approve]")'])
|
||||
await browserStep(['wait', 'button[aria-label="Stop"]'])
|
||||
await browserStep(['click', 'button[aria-label="Stop"]'])
|
||||
await until(async () => (await getPlan()).state === 'interrupted', 'visible Stop to interrupt the running team')
|
||||
// Stop pauses an approved team: every worker stops and keeps its saved
|
||||
// conversation, and the plan stays running so a later message resumes it.
|
||||
await until(async () => {
|
||||
const team = JSON.parse(readFileSync(join(configDir, 'teams', TEAM_SMOKE_TEAM, 'config.json'), 'utf8'))
|
||||
return team.members.filter((member: { name: string }) => member.name !== 'team-lead').every((member: { isActive?: boolean }) => member.isActive === false)
|
||||
}, 'stopped workers to become inactive')
|
||||
return team.members
|
||||
.filter((member: { name: string }) => member.name !== 'team-lead')
|
||||
.every((member: { isActive?: boolean; terminated?: boolean }) => member.isActive === false && member.terminated === true)
|
||||
}, 'visible Stop to stop every running team worker')
|
||||
const paused = await getPlan()
|
||||
if (paused.state !== 'running') throw new Error(`Stop ended the approved team instead of pausing it (plan ${paused.state})`)
|
||||
await browserStep(['screenshot', join(artifactDir, 'team-review-stopped.png')], { allowFailure: true })
|
||||
}
|
||||
|
||||
@@ -12,6 +12,14 @@ const checks: Check[] = [
|
||||
title: 'Agent Teams plan sidecar compatibility and approval recovery',
|
||||
command: ['bun', 'test', './src/utils/swarm/teamPlanStore.test.ts', './src/server/services/teamPlanService.test.ts'],
|
||||
},
|
||||
{
|
||||
title: 'Agent Teams legacy mailbox migration to unread inbox plus history',
|
||||
command: ['bun', 'test', './src/utils/teammateMailbox.test.ts', '--test-name-pattern', 'legacy inbox files'],
|
||||
},
|
||||
{
|
||||
title: 'Agent Teams teammate resume from agent metadata written before the teammate fields',
|
||||
command: ['bun', 'test', './src/utils/swarm/inProcessRunner.resume.test.ts', '--test-name-pattern', 'metadata written before'],
|
||||
},
|
||||
{
|
||||
title: 'Session collaboration state migration and recovery',
|
||||
command: ['bun', 'test', './src/server/services/sessionCollaborationService.test.ts', '--test-name-pattern', 'migrat|recover'],
|
||||
|
||||
@@ -0,0 +1,138 @@
|
||||
import { afterEach, beforeEach, describe, expect, spyOn, test } from 'bun:test'
|
||||
import { mkdtemp, rm } from 'fs/promises'
|
||||
import { tmpdir } from 'os'
|
||||
import { join } from 'path'
|
||||
import * as mailbox from '../utils/teammateMailbox.js'
|
||||
import {
|
||||
createLeadMailboxPollState,
|
||||
type LeadMailboxPollStep,
|
||||
MAX_LEAD_MAILBOX_ACK_FAILURES,
|
||||
takeLeadMailboxBatch,
|
||||
} from './print.js'
|
||||
|
||||
const TEAM = 'print-lead-team'
|
||||
|
||||
let configDir: string
|
||||
let originalConfigDir: string | undefined
|
||||
|
||||
beforeEach(async () => {
|
||||
originalConfigDir = process.env.CLAUDE_CONFIG_DIR
|
||||
configDir = await mkdtemp(join(tmpdir(), 'cc-haha-print-lead-mailbox-'))
|
||||
process.env.CLAUDE_CONFIG_DIR = configDir
|
||||
})
|
||||
|
||||
afterEach(async () => {
|
||||
if (originalConfigDir === undefined) delete process.env.CLAUDE_CONFIG_DIR
|
||||
else process.env.CLAUDE_CONFIG_DIR = originalConfigDir
|
||||
await rm(configDir, { recursive: true, force: true })
|
||||
})
|
||||
|
||||
async function send(from: string, text: string, second: number): Promise<void> {
|
||||
const timestamp = new Date(Date.UTC(2026, 9, 3, 0, 0, second)).toISOString()
|
||||
expect(
|
||||
await mailbox.writeToMailbox('team-lead', { from, text, timestamp }, TEAM),
|
||||
).toBe(true)
|
||||
}
|
||||
|
||||
async function unreadTexts(): Promise<string[]> {
|
||||
return (await mailbox.readUnreadMessages('team-lead', TEAM)).map(m => m.text)
|
||||
}
|
||||
|
||||
function texts(step: LeadMailboxPollStep): string[] | undefined {
|
||||
return step.kind === 'batch' ? step.messages.map(m => m.text) : undefined
|
||||
}
|
||||
|
||||
describe('print-mode team lead mailbox poll', () => {
|
||||
test('acknowledges only the batch it read', async () => {
|
||||
await send('alice', 'report', 1)
|
||||
const readUnread = mailbox.readUnreadMessages
|
||||
const read = spyOn(mailbox, 'readUnreadMessages').mockImplementation(
|
||||
async (agentName, teamName) => {
|
||||
const batch = await readUnread(agentName, teamName)
|
||||
// A teammate fails while the lead is between its read and its ack.
|
||||
await send('bob', 'failed: provider stream ended', 2)
|
||||
return batch
|
||||
},
|
||||
)
|
||||
|
||||
try {
|
||||
const step = await takeLeadMailboxBatch(createLeadMailboxPollState(), {
|
||||
teamName: TEAM,
|
||||
hasActiveTeammates: true,
|
||||
})
|
||||
expect(texts(step)).toEqual(['report'])
|
||||
} finally {
|
||||
read.mockRestore()
|
||||
}
|
||||
|
||||
expect(await unreadTexts()).toEqual(['failed: provider stream ended'])
|
||||
})
|
||||
|
||||
test('keeps polling quietly while teammates work and nothing is unread', async () => {
|
||||
expect(
|
||||
await takeLeadMailboxBatch(createLeadMailboxPollState(), {
|
||||
teamName: TEAM,
|
||||
hasActiveTeammates: true,
|
||||
}),
|
||||
).toEqual({ kind: 'idle' })
|
||||
})
|
||||
|
||||
test('drains mail left behind by the last teammate before stopping', async () => {
|
||||
await send('worker', 'final report', 1)
|
||||
const state = createLeadMailboxPollState()
|
||||
const departed = { teamName: TEAM, hasActiveTeammates: false }
|
||||
|
||||
expect(texts(await takeLeadMailboxBatch(state, departed))).toEqual([
|
||||
'final report',
|
||||
])
|
||||
expect(await takeLeadMailboxBatch(state, departed)).toEqual({ kind: 'stop' })
|
||||
expect(await unreadTexts()).toEqual([])
|
||||
})
|
||||
|
||||
test('drains only once per departure, then waits for an active teammate', async () => {
|
||||
await send('worker', 'final report', 1)
|
||||
const state = createLeadMailboxPollState()
|
||||
const departed = { teamName: TEAM, hasActiveTeammates: false }
|
||||
|
||||
expect(texts(await takeLeadMailboxBatch(state, departed))).toEqual([
|
||||
'final report',
|
||||
])
|
||||
await send('worker', 'written during the drain turn', 2)
|
||||
expect(await takeLeadMailboxBatch(state, departed)).toEqual({ kind: 'stop' })
|
||||
expect(await unreadTexts()).toEqual(['written during the drain turn'])
|
||||
|
||||
expect(
|
||||
texts(
|
||||
await takeLeadMailboxBatch(state, {
|
||||
teamName: TEAM,
|
||||
hasActiveTeammates: true,
|
||||
}),
|
||||
),
|
||||
).toEqual(['written during the drain turn'])
|
||||
})
|
||||
|
||||
test('retries an unacknowledged batch before processing it anyway', async () => {
|
||||
await send('alice', 'report', 1)
|
||||
const claim = spyOn(mailbox, 'claimMailboxMessages').mockResolvedValue(
|
||||
undefined,
|
||||
)
|
||||
const state = createLeadMailboxPollState()
|
||||
const active = { teamName: TEAM, hasActiveTeammates: true }
|
||||
|
||||
try {
|
||||
for (let poll = 1; poll < MAX_LEAD_MAILBOX_ACK_FAILURES; poll++) {
|
||||
expect(await takeLeadMailboxBatch(state, active)).toEqual({
|
||||
kind: 'retry',
|
||||
})
|
||||
}
|
||||
expect(texts(await takeLeadMailboxBatch(state, active))).toEqual(['report'])
|
||||
} finally {
|
||||
claim.mockRestore()
|
||||
}
|
||||
|
||||
// Acknowledging works again: the failure streak resets.
|
||||
expect(texts(await takeLeadMailboxBatch(state, active))).toEqual(['report'])
|
||||
expect(state.consecutiveAckFailures).toBe(0)
|
||||
expect(await unreadTexts()).toEqual([])
|
||||
})
|
||||
})
|
||||
@@ -0,0 +1,149 @@
|
||||
import { afterAll, expect, mock, test } from 'bun:test'
|
||||
import { mkdtempSync, rmSync } from 'node:fs'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { join } from 'node:path'
|
||||
import { createSandboxedTestEnvironment } from '../../scripts/pr/test-environment.js'
|
||||
|
||||
const home = mkdtempSync(join(tmpdir(), 'lead-mailbox-drain-'))
|
||||
const originalEnv = { ...process.env }
|
||||
for (const key of Object.keys(process.env)) delete process.env[key]
|
||||
Object.assign(
|
||||
process.env,
|
||||
createSandboxedTestEnvironment(
|
||||
home,
|
||||
{ CLAUDE_CODE_SIMPLE: '1', CC_HAHA_AGENT_TEAMS_ENABLED: '1' },
|
||||
originalEnv,
|
||||
),
|
||||
)
|
||||
|
||||
const TEAM = 'drain-team'
|
||||
const LEAD_ID = `team-lead@${TEAM}`
|
||||
const WORKER_ID = `worker@${TEAM}`
|
||||
|
||||
// Each lead turn records its prompt instead of calling a model. The first turn
|
||||
// is where the last teammate goes away, as a shutdown would remove it.
|
||||
const prompts: string[] = []
|
||||
let onTurn: (prompt: string) => void = () => {}
|
||||
const queryEngine = await import('../QueryEngine.js')
|
||||
const originalQueryEngine = { ...queryEngine }
|
||||
mock.module('../QueryEngine.js', () => ({
|
||||
...queryEngine,
|
||||
ask: async function* (params: { prompt: unknown }) {
|
||||
const prompt =
|
||||
typeof params.prompt === 'string'
|
||||
? params.prompt
|
||||
: JSON.stringify(params.prompt)
|
||||
prompts.push(prompt)
|
||||
onTurn(prompt)
|
||||
},
|
||||
}))
|
||||
|
||||
const { __runHeadlessStreamingForTests } = await import('./print.js')
|
||||
const { StructuredIO } = await import('./structuredIO.js')
|
||||
const { Stream } = await import('../utils/stream.js')
|
||||
const { getDefaultAppState } = await import('../state/AppStateStore.js')
|
||||
const { clearCommandQueue } = await import('../utils/messageQueueManager.js')
|
||||
const { setIsInteractive, getIsInteractive } = await import(
|
||||
'../bootstrap/state.js'
|
||||
)
|
||||
const mailbox = await import('../utils/teammateMailbox.js')
|
||||
|
||||
const wasInteractive = getIsInteractive()
|
||||
setIsInteractive(false)
|
||||
afterAll(() => {
|
||||
// mock.restore does not undo mock.module; later files in the same run
|
||||
// must see the real QueryEngine.
|
||||
mock.module('../QueryEngine.js', () => originalQueryEngine)
|
||||
clearCommandQueue()
|
||||
setIsInteractive(wasInteractive)
|
||||
for (const key of Object.keys(process.env)) delete process.env[key]
|
||||
Object.assign(process.env, originalEnv)
|
||||
rmSync(home, { recursive: true, force: true })
|
||||
})
|
||||
|
||||
test('delivers mail the last teammate left behind, with its summary, before the lead stops polling', async () => {
|
||||
clearCommandQueue()
|
||||
expect(
|
||||
await mailbox.writeToMailbox(
|
||||
'team-lead',
|
||||
{
|
||||
from: 'worker',
|
||||
text: 'All 42 tests pass',
|
||||
summary: 'tests "green"',
|
||||
timestamp: new Date().toISOString(),
|
||||
},
|
||||
TEAM,
|
||||
),
|
||||
).toBe(true)
|
||||
|
||||
const defaults = getDefaultAppState()
|
||||
let state = {
|
||||
...defaults,
|
||||
teamContext: {
|
||||
teamName: TEAM,
|
||||
teamFilePath: join(home, 'teams', TEAM, 'config.json'),
|
||||
leadAgentId: LEAD_ID,
|
||||
teammates: {
|
||||
[LEAD_ID]: { name: 'team-lead' },
|
||||
[WORKER_ID]: { name: 'worker' },
|
||||
},
|
||||
},
|
||||
} as unknown as typeof defaults
|
||||
onTurn = prompt => {
|
||||
if (prompt.includes('start')) {
|
||||
const teamContext = state.teamContext!
|
||||
state = {
|
||||
...state,
|
||||
teamContext: {
|
||||
...teamContext,
|
||||
teammates: { [LEAD_ID]: teamContext.teammates[LEAD_ID]! },
|
||||
},
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const input = new Stream<string>()
|
||||
const output = __runHeadlessStreamingForTests(
|
||||
new StructuredIO(input),
|
||||
[],
|
||||
[],
|
||||
[],
|
||||
[],
|
||||
(() => undefined) as never,
|
||||
{},
|
||||
() => state,
|
||||
update => {
|
||||
state = update(state)
|
||||
},
|
||||
[],
|
||||
{ outputFormat: 'stream-json' } as never,
|
||||
)
|
||||
const drained = (async () => {
|
||||
for await (const _ of output) {
|
||||
// Keep the outbound queue flowing.
|
||||
}
|
||||
})()
|
||||
|
||||
try {
|
||||
input.enqueue(
|
||||
`${JSON.stringify({ type: 'user', message: { role: 'user', content: 'start' } })}\n`,
|
||||
)
|
||||
const deadline = Date.now() + 10_000
|
||||
while (prompts.length < 2 && Date.now() < deadline) await Bun.sleep(20)
|
||||
|
||||
expect(prompts).toHaveLength(2)
|
||||
expect(prompts[1]).toBe(
|
||||
'<teammate-message teammate_id="worker" summary="tests "green"">\n' +
|
||||
'All 42 tests pass\n' +
|
||||
'</teammate-message>',
|
||||
)
|
||||
expect(await mailbox.readUnreadMessages('team-lead', TEAM)).toEqual([])
|
||||
} finally {
|
||||
input.done()
|
||||
await drained
|
||||
clearCommandQueue()
|
||||
}
|
||||
|
||||
// One drain only: the lead then stops polling and the session closes.
|
||||
expect(prompts).toHaveLength(2)
|
||||
}, 30_000)
|
||||
+106
-24
@@ -165,7 +165,7 @@ import {
|
||||
DEFAULT_OUTPUT_STYLE_NAME,
|
||||
getAllOutputStyles,
|
||||
} from 'src/constants/outputStyles.js'
|
||||
import { TEAMMATE_MESSAGE_TAG, TICK_TAG } from 'src/constants/xml.js'
|
||||
import { TICK_TAG } from 'src/constants/xml.js'
|
||||
import {
|
||||
getSettings_DEPRECATED,
|
||||
getSettingsWithSources,
|
||||
@@ -358,8 +358,10 @@ import { TEAM_LEAD_NAME } from '../utils/swarm/constants.js'
|
||||
import { SHUTDOWN_TEAM_PROMPT } from '../utils/swarm/teamShutdownPrompt.js'
|
||||
import {
|
||||
readUnreadMessages,
|
||||
markMessagesAsRead,
|
||||
claimMailboxMessages,
|
||||
formatTeammateMessages,
|
||||
isShutdownApproved,
|
||||
type TeammateMessage,
|
||||
} from '../utils/teammateMailbox.js'
|
||||
import {
|
||||
partitionLeadMailboxMessages,
|
||||
@@ -424,6 +426,92 @@ export function createShutdownTeamPrompt(): QueuedCommand {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Polls in a row whose batch could not be acknowledged before the batch is
|
||||
* processed anyway: delivering twice beats never delivering.
|
||||
*/
|
||||
export const MAX_LEAD_MAILBOX_ACK_FAILURES = 5
|
||||
|
||||
export type LeadMailboxPollState = {
|
||||
/** The unread mail left once the last teammate went away was drained. */
|
||||
finalDrainDone: boolean
|
||||
/** Consecutive polls whose batch could not be acknowledged. */
|
||||
consecutiveAckFailures: number
|
||||
}
|
||||
|
||||
export function createLeadMailboxPollState(): LeadMailboxPollState {
|
||||
return { finalDrainDone: false, consecutiveAckFailures: 0 }
|
||||
}
|
||||
|
||||
export type LeadMailboxPollStep =
|
||||
| { kind: 'stop' }
|
||||
| { kind: 'idle' }
|
||||
| { kind: 'retry' }
|
||||
| { kind: 'batch'; messages: TeammateMessage[] }
|
||||
|
||||
/**
|
||||
* One poll of the team lead's inbox. The batch is claimed (acknowledged by
|
||||
* identity) before it is handed out, so mail that lands after the read stays
|
||||
* unread and a concurrent mid-turn attachment can never deliver the same
|
||||
* message again. When no teammate remains, unread mail gets one final drain
|
||||
* instead of being stranded.
|
||||
*/
|
||||
export async function takeLeadMailboxBatch(
|
||||
state: LeadMailboxPollState,
|
||||
options: { teamName: string | undefined; hasActiveTeammates: boolean },
|
||||
): Promise<LeadMailboxPollStep> {
|
||||
const { teamName, hasActiveTeammates } = options
|
||||
const unread = await readUnreadMessages(TEAM_LEAD_NAME, teamName)
|
||||
|
||||
if (!hasActiveTeammates) {
|
||||
if (
|
||||
unread.length === 0 ||
|
||||
state.finalDrainDone ||
|
||||
state.consecutiveAckFailures >= MAX_LEAD_MAILBOX_ACK_FAILURES
|
||||
) {
|
||||
if (unread.length === 0) {
|
||||
Object.assign(state, createLeadMailboxPollState())
|
||||
}
|
||||
return { kind: 'stop' }
|
||||
}
|
||||
state.finalDrainDone = true
|
||||
} else {
|
||||
state.finalDrainDone = false
|
||||
}
|
||||
|
||||
if (unread.length === 0) {
|
||||
state.consecutiveAckFailures = 0
|
||||
return { kind: 'idle' }
|
||||
}
|
||||
|
||||
const claimed = await claimMailboxMessages(TEAM_LEAD_NAME, teamName, unread)
|
||||
if (claimed) {
|
||||
state.consecutiveAckFailures = 0
|
||||
if (claimed.length === 0) {
|
||||
// Another consumer (a mid-turn attachment) took the whole batch.
|
||||
return hasActiveTeammates ? { kind: 'idle' } : { kind: 'stop' }
|
||||
}
|
||||
return { kind: 'batch', messages: claimed }
|
||||
}
|
||||
|
||||
state.consecutiveAckFailures += 1
|
||||
if (
|
||||
hasActiveTeammates &&
|
||||
state.consecutiveAckFailures < MAX_LEAD_MAILBOX_ACK_FAILURES
|
||||
) {
|
||||
logForDebugging(
|
||||
`[print.ts] Could not mark ${unread.length} inbox message(s) read (${state.consecutiveAckFailures}/${MAX_LEAD_MAILBOX_ACK_FAILURES}); retrying next poll`,
|
||||
{ level: 'warn' },
|
||||
)
|
||||
return { kind: 'retry' }
|
||||
}
|
||||
logForDebugging(
|
||||
`[print.ts] Could not mark ${unread.length} inbox message(s) read for ${state.consecutiveAckFailures} polls; processing the batch unmarked`,
|
||||
{ level: 'warn' },
|
||||
)
|
||||
return { kind: 'batch', messages: unread }
|
||||
}
|
||||
|
||||
// Track message UUIDs received during the current session runtime
|
||||
const MAX_RECEIVED_UUIDS = 10_000
|
||||
const receivedMessageUuids = new Set<UUID>()
|
||||
@@ -1114,6 +1202,7 @@ function runHeadlessStreaming(
|
||||
| undefined
|
||||
let inputClosed = false
|
||||
let shutdownPromptInjected = false
|
||||
const leadMailboxPoll = createLeadMailboxPollState()
|
||||
let heldBackResult: StdoutMessage | null = null
|
||||
const deferredAgentNotifications = new TaskNotificationFollowUpBatch()
|
||||
let abortController: AbortController | undefined
|
||||
@@ -2581,8 +2670,6 @@ function runHeadlessStreaming(
|
||||
const teamContext = currentAppState.teamContext
|
||||
|
||||
if (teamContext && isTeamLead(teamContext)) {
|
||||
const agentName = 'team-lead'
|
||||
|
||||
// Poll for messages while teammates are active
|
||||
// This is needed because teammates may send messages while we're waiting
|
||||
// Keep polling until the team is shut down
|
||||
@@ -2591,32 +2678,32 @@ function runHeadlessStreaming(
|
||||
while (true) {
|
||||
// Check if teammates are still active
|
||||
const refreshedState = getAppState()
|
||||
const hasActiveTeammates = hasTeammatesRequiringShutdown(refreshedState)
|
||||
const teamName = refreshedState.teamContext?.teamName
|
||||
// Claims exactly the batch it read before handing it out, and
|
||||
// drains leftover mail once after the last teammate is gone.
|
||||
const step = await takeLeadMailboxBatch(leadMailboxPoll, {
|
||||
teamName,
|
||||
hasActiveTeammates: hasTeammatesRequiringShutdown(refreshedState),
|
||||
})
|
||||
|
||||
if (!hasActiveTeammates) {
|
||||
if (step.kind === 'stop') {
|
||||
logForDebugging(
|
||||
'[print.ts] No more active teammates, stopping poll',
|
||||
)
|
||||
break
|
||||
}
|
||||
|
||||
const unread = await readUnreadMessages(
|
||||
agentName,
|
||||
refreshedState.teamContext?.teamName,
|
||||
)
|
||||
if (step.kind === 'retry') {
|
||||
await sleep(POLL_INTERVAL_MS)
|
||||
continue
|
||||
}
|
||||
|
||||
if (unread.length > 0) {
|
||||
if (step.kind === 'batch') {
|
||||
const unread = step.messages
|
||||
logForDebugging(
|
||||
`[print.ts] Team-lead found ${unread.length} unread messages`,
|
||||
)
|
||||
|
||||
// Mark as read immediately to avoid duplicate processing
|
||||
await markMessagesAsRead(
|
||||
agentName,
|
||||
refreshedState.teamContext?.teamName,
|
||||
)
|
||||
|
||||
const teamName = refreshedState.teamContext?.teamName
|
||||
const partitioned = partitionLeadMailboxMessages(unread)
|
||||
|
||||
if (
|
||||
@@ -2695,12 +2782,7 @@ function runHeadlessStreaming(
|
||||
|
||||
// Format remaining teammate chat the same way as useInboxPoller.
|
||||
// Permission requests stay out of the model context.
|
||||
const formatted = partitioned.remaining
|
||||
.map(
|
||||
(m: { from: string; text: string; color?: string }) =>
|
||||
`<${TEAMMATE_MESSAGE_TAG} teammate_id="${m.from}"${m.color ? ` color="${m.color}"` : ''}>\n${m.text}\n</${TEAMMATE_MESSAGE_TAG}>`,
|
||||
)
|
||||
.join('\n\n')
|
||||
const formatted = formatTeammateMessages(partitioned.remaining)
|
||||
|
||||
// Enqueue and process
|
||||
enqueue({
|
||||
|
||||
@@ -1119,6 +1119,16 @@ function PromptInput({
|
||||
clearBuffer();
|
||||
resetHistory();
|
||||
return;
|
||||
} else if (result.error === 'delivery_failed') {
|
||||
// Keep the draft: sending it to the lead model instead would be a
|
||||
// different action than the user asked for.
|
||||
addNotification({
|
||||
key: 'direct-message-failed',
|
||||
text: `Could not deliver to @${result.recipientName} — try again`,
|
||||
priority: 'immediate',
|
||||
timeoutMs: 5000
|
||||
});
|
||||
return;
|
||||
} else if (result.error === 'no_team_context') {
|
||||
// No team context - fall through to normal prompt submission
|
||||
} else {
|
||||
|
||||
@@ -0,0 +1,15 @@
|
||||
import { expect, test } from 'bun:test'
|
||||
import { formatTeammateMessages } from '../../utils/teammateMailbox.js'
|
||||
import { parseTeammateMessages } from './UserTeammateMessage.js'
|
||||
|
||||
test('teammate envelope attributes read back as the text the teammate sent', () => {
|
||||
const text = formatTeammateMessages([
|
||||
{ from: 'reviewer', color: 'blue', summary: 'Fix "auth" & <retry>', text: 'details' },
|
||||
{ from: 'tester', text: 'plain' },
|
||||
])
|
||||
|
||||
expect(parseTeammateMessages(text)).toEqual([
|
||||
{ teammateId: 'reviewer', color: 'blue', summary: 'Fix "auth" & <retry>', content: 'details' },
|
||||
{ teammateId: 'tester', color: undefined, summary: undefined, content: 'plain' },
|
||||
])
|
||||
})
|
||||
@@ -7,6 +7,7 @@ import { Ansi, Box, Text, type TextProps } from '../../ink.js';
|
||||
import { toInkColor } from '../../utils/ink.js';
|
||||
import { jsonParse } from '../../utils/slowOperations.js';
|
||||
import { isShutdownApproved } from '../../utils/teammateMailbox.js';
|
||||
import { unescapeXmlAttr } from '../../utils/xml.js';
|
||||
import { MessageResponse } from '../MessageResponse.js';
|
||||
import { tryRenderPlanApprovalMessage } from './PlanApprovalMessage.js';
|
||||
import { tryRenderShutdownMessage } from './ShutdownMessage.js';
|
||||
@@ -29,16 +30,17 @@ const TEAMMATE_MSG_REGEX = new RegExp(`<${TEAMMATE_MESSAGE_TAG}\\s+teammate_id="
|
||||
* <teammate-message teammate_id="alice" color="red" summary="Brief update">message content</teammate-message>
|
||||
* Supports multiple messages in a single text block.
|
||||
*/
|
||||
function parseTeammateMessages(text: string): ParsedMessage[] {
|
||||
export function parseTeammateMessages(text: string): ParsedMessage[] {
|
||||
const messages: ParsedMessage[] = [];
|
||||
// Use matchAll to find all matches (this is a RegExp method, not child_process)
|
||||
for (const match of text.matchAll(TEAMMATE_MSG_REGEX)) {
|
||||
if (match[1] && match[4]) {
|
||||
// Attribute values are XML-escaped by formatTeammateMessage.
|
||||
messages.push({
|
||||
teammateId: match[1],
|
||||
color: match[2],
|
||||
teammateId: unescapeXmlAttr(match[1]),
|
||||
color: match[2] === undefined ? undefined : unescapeXmlAttr(match[2]),
|
||||
// may be undefined
|
||||
summary: match[3],
|
||||
summary: match[3] === undefined ? undefined : unescapeXmlAttr(match[3]),
|
||||
// may be undefined
|
||||
content: match[4].trim()
|
||||
});
|
||||
|
||||
@@ -230,6 +230,7 @@ function looksLikeGoalStatusOutput(output: string): boolean {
|
||||
return (
|
||||
trimmed.startsWith('Goal set:') ||
|
||||
trimmed.startsWith('Goal continuing:') ||
|
||||
trimmed.startsWith('Goal waiting for teammates:') ||
|
||||
trimmed.startsWith('Goal cleared:') ||
|
||||
trimmed === 'Goal cleared.' ||
|
||||
trimmed === 'Goal marked complete.' ||
|
||||
|
||||
@@ -22,25 +22,54 @@ const backendDetectionModule = await import(
|
||||
'../utils/swarm/backends/detection.js'
|
||||
)
|
||||
const tasksModule = await import('../utils/tasks.js')
|
||||
const notifierModule = await import('../services/notifier.js')
|
||||
|
||||
// mock.module replaces exports process-wide and outlives this file, so every
|
||||
// module mocked below is put back afterwards from these untouched copies.
|
||||
// Otherwise later files in one `bun test` run (the coverage gate runs all of
|
||||
// src in a single process) see this file's fake mailbox, team file and tasks.
|
||||
const originalModules: Array<[string, Record<string, unknown>]> = [
|
||||
['usehooks-ts', { ...useHooksTsModule }],
|
||||
['../state/AppState.js', { ...appStateModule }],
|
||||
['../ink/useTerminalNotification.js', { ...terminalNotificationModule }],
|
||||
['../services/notifier.js', { ...notifierModule }],
|
||||
['../utils/teammate.js', { ...teammateModule }],
|
||||
['../utils/teammateContext.js', { ...teammateContextModule }],
|
||||
['../utils/swarm/backends/registry.js', { ...backendRegistryModule }],
|
||||
['../utils/swarm/backends/detection.js', { ...backendDetectionModule }],
|
||||
['../utils/tasks.js', { ...tasksModule }],
|
||||
['../utils/teammateMailbox.js', { ...mailboxModule }],
|
||||
]
|
||||
|
||||
let intervalCallback: (() => void) | undefined
|
||||
let state: AppState
|
||||
let unreadMessages: TeammateMessage[] = []
|
||||
let teamFileReadCount = 0
|
||||
let submitted: string[] = []
|
||||
let onSubmit: (formatted: string) => boolean = () => true
|
||||
// While set, every inbox read waits for it, like a slow disk.
|
||||
let readGate: Promise<void> | undefined
|
||||
|
||||
const markAllRead = mock(async () => {
|
||||
unreadMessages = unreadMessages.map(message => ({ ...message, read: true }))
|
||||
const readUnread = mock(async () => {
|
||||
await readGate
|
||||
return unreadMessages.filter(message => !message.read)
|
||||
})
|
||||
const markReadByPredicate = mock(
|
||||
const markReadByIdentity = mock(
|
||||
async (
|
||||
_agentName: string,
|
||||
predicate: (message: TeammateMessage) => boolean,
|
||||
_teamName: string | undefined,
|
||||
delivered: readonly TeammateMessage[],
|
||||
) => {
|
||||
const identities = new Set(
|
||||
delivered.map(mailboxModule.getMailboxMessageIdentity),
|
||||
)
|
||||
unreadMessages = unreadMessages.map(message =>
|
||||
!message.read && predicate(message)
|
||||
!message.read &&
|
||||
identities.has(mailboxModule.getMailboxMessageIdentity(message))
|
||||
? { ...message, read: true }
|
||||
: message,
|
||||
)
|
||||
return true
|
||||
},
|
||||
)
|
||||
const removeTeammate = mock(async () => true)
|
||||
@@ -137,11 +166,9 @@ mock.module('../utils/tasks.js', () => ({
|
||||
|
||||
mock.module('../utils/teammateMailbox.js', () => ({
|
||||
...mailboxModule,
|
||||
readUnreadMessages: async () =>
|
||||
unreadMessages.filter(message => !message.read),
|
||||
markMessagesAsRead: markAllRead,
|
||||
markMessagesAsReadByPredicate: markReadByPredicate,
|
||||
writeToMailbox: async () => {},
|
||||
readUnreadMessages: readUnread,
|
||||
markMessagesAsReadByIdentity: markReadByIdentity,
|
||||
writeToMailbox: async () => true,
|
||||
}))
|
||||
|
||||
const { useInboxPoller } = await import('./useInboxPoller.js')
|
||||
@@ -151,7 +178,10 @@ function Harness() {
|
||||
enabled: true,
|
||||
isLoading: false,
|
||||
focusedInputDialog: undefined,
|
||||
onSubmitMessage: () => true,
|
||||
onSubmitMessage: formatted => {
|
||||
submitted.push(formatted)
|
||||
return onSubmit(formatted)
|
||||
},
|
||||
})
|
||||
return null
|
||||
}
|
||||
@@ -159,8 +189,11 @@ function Harness() {
|
||||
beforeEach(() => {
|
||||
intervalCallback = undefined
|
||||
teamFileReadCount = 0
|
||||
markAllRead.mockClear()
|
||||
markReadByPredicate.mockClear()
|
||||
submitted = []
|
||||
onSubmit = () => true
|
||||
readGate = undefined
|
||||
readUnread.mockClear()
|
||||
markReadByIdentity.mockClear()
|
||||
removeTeammate.mockClear()
|
||||
killPane.mockClear()
|
||||
unreadMessages = [
|
||||
@@ -188,6 +221,9 @@ beforeEach(() => {
|
||||
afterAll(() => {
|
||||
// Bun's mock.restore does not undo mock.module export replacements.
|
||||
mock.module('../utils/swarm/teamHelpers.js', () => originalTeamHelpers)
|
||||
for (const [path, original] of originalModules) {
|
||||
mock.module(path, () => original)
|
||||
}
|
||||
mock.restore()
|
||||
})
|
||||
|
||||
@@ -257,6 +293,124 @@ describe('shutdown approval polling', () => {
|
||||
})
|
||||
})
|
||||
|
||||
describe('teammate message delivery', () => {
|
||||
test('acknowledges exactly the delivered batch', async () => {
|
||||
unreadMessages = [chat('alice', 'report')]
|
||||
onSubmit = () => {
|
||||
// A teammate writes while this poll is still delivering.
|
||||
unreadMessages = [...unreadMessages, chat('bob', 'late result')]
|
||||
return true
|
||||
}
|
||||
const output = new PassThrough()
|
||||
const app = render(<Harness />, {
|
||||
stdout: output,
|
||||
stderr: output,
|
||||
debug: false,
|
||||
exitOnCtrlC: false,
|
||||
})
|
||||
|
||||
try {
|
||||
await waitFor(() => markReadByIdentity.mock.calls.length === 1)
|
||||
|
||||
expect(submitted).toHaveLength(1)
|
||||
expect(submitted[0]).toContain('report')
|
||||
expect(markReadByIdentity.mock.calls[0]?.[2]).toEqual([
|
||||
chat('alice', 'report'),
|
||||
])
|
||||
expect(
|
||||
unreadMessages.filter(message => !message.read).map(m => m.text),
|
||||
).toEqual(['late result'])
|
||||
} finally {
|
||||
app.unmount()
|
||||
output.destroy()
|
||||
}
|
||||
})
|
||||
|
||||
test('delivers mail queued during a busy turn with the shared envelope', async () => {
|
||||
unreadMessages = []
|
||||
state = {
|
||||
...state,
|
||||
inbox: {
|
||||
messages: [
|
||||
{
|
||||
id: 'queued-1',
|
||||
from: 'worker',
|
||||
text: 'Finished the migration',
|
||||
timestamp: '2026-10-03T00:00:00.000Z',
|
||||
status: 'pending',
|
||||
summary: 'migration "done"',
|
||||
},
|
||||
],
|
||||
},
|
||||
} as AppState
|
||||
const output = new PassThrough()
|
||||
const app = render(<Harness />, {
|
||||
stdout: output,
|
||||
stderr: output,
|
||||
debug: false,
|
||||
exitOnCtrlC: false,
|
||||
})
|
||||
|
||||
try {
|
||||
await waitFor(() => submitted.length === 1)
|
||||
|
||||
expect(submitted[0]).toBe(
|
||||
'<teammate-message teammate_id="worker" summary="migration "done"">\n' +
|
||||
'Finished the migration\n' +
|
||||
'</teammate-message>',
|
||||
)
|
||||
expect(state.inbox.messages).toEqual([])
|
||||
} finally {
|
||||
app.unmount()
|
||||
output.destroy()
|
||||
}
|
||||
})
|
||||
|
||||
test('does not deliver a batch twice when polls overlap', async () => {
|
||||
unreadMessages = [chat('alice', 'report')]
|
||||
let openGate!: () => void
|
||||
readGate = new Promise<void>(resolve => {
|
||||
openGate = resolve
|
||||
})
|
||||
const output = new PassThrough()
|
||||
const app = render(<Harness />, {
|
||||
stdout: output,
|
||||
stderr: output,
|
||||
debug: false,
|
||||
exitOnCtrlC: false,
|
||||
})
|
||||
|
||||
try {
|
||||
await waitFor(() => readUnread.mock.calls.length === 1)
|
||||
// The next interval tick fires while the first poll is still reading.
|
||||
intervalCallback?.()
|
||||
intervalCallback?.()
|
||||
openGate()
|
||||
await waitFor(() => markReadByIdentity.mock.calls.length === 1)
|
||||
intervalCallback?.()
|
||||
await waitFor(() => readUnread.mock.calls.length === 2)
|
||||
await Bun.sleep(20)
|
||||
|
||||
expect(submitted).toHaveLength(1)
|
||||
expect(markReadByIdentity).toHaveBeenCalledTimes(1)
|
||||
} finally {
|
||||
openGate()
|
||||
app.unmount()
|
||||
output.destroy()
|
||||
}
|
||||
})
|
||||
})
|
||||
|
||||
function chat(from: string, text: string): TeammateMessage {
|
||||
return {
|
||||
id: `mailbox-${from}-${text}`,
|
||||
from,
|
||||
text,
|
||||
timestamp: '2026-10-03T00:00:00.000Z',
|
||||
read: false,
|
||||
}
|
||||
}
|
||||
|
||||
function shutdownApproval(
|
||||
from: string,
|
||||
paneId: string,
|
||||
|
||||
+33
-33
@@ -2,7 +2,6 @@ import { randomUUID } from 'crypto'
|
||||
import { useCallback, useEffect, useRef } from 'react'
|
||||
import { useInterval } from 'usehooks-ts'
|
||||
import type { ToolUseConfirm } from '../components/permissions/PermissionRequest.js'
|
||||
import { TEAMMATE_MESSAGE_TAG } from '../constants/xml.js'
|
||||
import { useTerminalNotification } from '../ink/useTerminalNotification.js'
|
||||
import { sendNotification } from '../services/notifier.js'
|
||||
import {
|
||||
@@ -50,6 +49,7 @@ import {
|
||||
} from '../utils/teammate.js'
|
||||
import { isInProcessTeammate } from '../utils/teammateContext.js'
|
||||
import {
|
||||
formatTeammateMessages,
|
||||
getTrustedShutdownApproval,
|
||||
isModeSetRequest,
|
||||
isPermissionRequest,
|
||||
@@ -62,8 +62,7 @@ import {
|
||||
isShutdownRequest,
|
||||
isTeamPermissionUpdate,
|
||||
isTrustedTeamLeaderMessage,
|
||||
markMessagesAsRead,
|
||||
markMessagesAsReadByPredicate,
|
||||
markMessagesAsReadByIdentity,
|
||||
readUnreadMessages,
|
||||
type TeammateMessage,
|
||||
writeToMailbox,
|
||||
@@ -139,8 +138,9 @@ export function useInboxPoller({
|
||||
const setAppState = useSetAppState()
|
||||
const inboxMessageCount = useAppState(s => s.inbox.messages.length)
|
||||
const terminal = useTerminalNotification()
|
||||
const pollInFlightRef = useRef(false)
|
||||
|
||||
const poll = useCallback(async () => {
|
||||
const pollInbox = useCallback(async () => {
|
||||
if (!enabled) return
|
||||
|
||||
// Use ref to avoid dependency on appState object (prevents infinite loop)
|
||||
@@ -203,16 +203,17 @@ export function useInboxPoller({
|
||||
|
||||
// Helper to mark messages as read in the inbox file.
|
||||
// Called after messages are successfully delivered or reliably queued.
|
||||
const markRead = () => {
|
||||
if (deferShutdownApprovals) {
|
||||
void markMessagesAsReadByPredicate(
|
||||
agentName,
|
||||
message => !isShutdownApproved(message.text),
|
||||
currentAppState.teamContext?.teamName,
|
||||
)
|
||||
return
|
||||
}
|
||||
void markMessagesAsRead(agentName, currentAppState.teamContext?.teamName)
|
||||
// Acknowledges exactly this poll's batch: anything that arrived after the
|
||||
// read stays unread for the next poll.
|
||||
const markRead = async () => {
|
||||
const processed = deferShutdownApprovals
|
||||
? unread.filter(message => !isShutdownApproved(message.text))
|
||||
: unread
|
||||
await markMessagesAsReadByIdentity(
|
||||
agentName,
|
||||
currentAppState.teamContext?.teamName,
|
||||
processed,
|
||||
)
|
||||
}
|
||||
|
||||
// Separate permission messages from regular teammate messages
|
||||
@@ -349,7 +350,7 @@ export function useInboxPoller({
|
||||
},
|
||||
}
|
||||
|
||||
// Deduplicate: if markMessagesAsRead failed on a prior poll,
|
||||
// Deduplicate: if acknowledging failed on a prior poll,
|
||||
// the same message will be re-read — skip if already queued.
|
||||
setToolUseConfirmQueue(queue => {
|
||||
if (queue.some(q => q.toolUseID === parsed.tool_use_id)) {
|
||||
@@ -828,21 +829,13 @@ export function useInboxPoller({
|
||||
if (regularMessages.length === 0) {
|
||||
// No regular messages, but we may have processed non-regular messages
|
||||
// (permissions, shutdown requests, etc.) above — mark those as read.
|
||||
markRead()
|
||||
await markRead()
|
||||
return
|
||||
}
|
||||
|
||||
// Format messages with XML wrapper for Claude (include color if available)
|
||||
// Transform plan approval requests to include instructions for Claude
|
||||
const formatted = regularMessages
|
||||
.map(m => {
|
||||
const colorAttr = m.color ? ` color="${m.color}"` : ''
|
||||
const summaryAttr = m.summary ? ` summary="${m.summary}"` : ''
|
||||
const messageContent = m.text
|
||||
|
||||
return `<${TEAMMATE_MESSAGE_TAG} teammate_id="${m.from}"${colorAttr}${summaryAttr}>\n${messageContent}\n</${TEAMMATE_MESSAGE_TAG}>`
|
||||
})
|
||||
.join('\n\n')
|
||||
const formatted = formatTeammateMessages(regularMessages)
|
||||
|
||||
// Helper to queue messages in AppState for later delivery
|
||||
const queueMessages = () => {
|
||||
@@ -886,7 +879,7 @@ export function useInboxPoller({
|
||||
// or reliably queued in AppState. This prevents permanent message loss
|
||||
// when the session is busy — if we crash before this point, the messages
|
||||
// will be re-read on the next poll cycle instead of being silently dropped.
|
||||
markRead()
|
||||
await markRead()
|
||||
}, [
|
||||
enabled,
|
||||
isLoading,
|
||||
@@ -897,6 +890,19 @@ export function useInboxPoller({
|
||||
store,
|
||||
])
|
||||
|
||||
// The interval fires every second whether or not the previous poll has
|
||||
// finished. A poll that is still delivering has not acknowledged its batch
|
||||
// yet, so an overlapping poll would read and deliver the same messages again.
|
||||
const poll = useCallback(async () => {
|
||||
if (pollInFlightRef.current) return
|
||||
pollInFlightRef.current = true
|
||||
try {
|
||||
await pollInbox()
|
||||
} finally {
|
||||
pollInFlightRef.current = false
|
||||
}
|
||||
}, [pollInbox])
|
||||
|
||||
// When session becomes idle, deliver any pending messages and clean up processed ones
|
||||
useEffect(() => {
|
||||
if (!enabled) return
|
||||
@@ -940,13 +946,7 @@ export function useInboxPoller({
|
||||
)
|
||||
|
||||
// Format messages with XML wrapper for Claude (include color if available)
|
||||
const formatted = pendingMessages
|
||||
.map(m => {
|
||||
const colorAttr = m.color ? ` color="${m.color}"` : ''
|
||||
const summaryAttr = m.summary ? ` summary="${m.summary}"` : ''
|
||||
return `<${TEAMMATE_MESSAGE_TAG} teammate_id="${m.from}"${colorAttr}${summaryAttr}>\n${m.text}\n</${TEAMMATE_MESSAGE_TAG}>`
|
||||
})
|
||||
.join('\n\n')
|
||||
const formatted = formatTeammateMessages(pendingMessages)
|
||||
|
||||
// Try to submit - only clear messages if successful
|
||||
const submitted = onSubmitTeammateMessage(formatted)
|
||||
|
||||
@@ -0,0 +1,90 @@
|
||||
import { afterAll, afterEach, beforeEach, expect, mock, test } from 'bun:test'
|
||||
import { mkdtemp, rm } from 'node:fs/promises'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { join } from 'node:path'
|
||||
|
||||
const GOAL_COMMAND = '<cc-haha-goal-hook>\nreviews/SUMMARY.md exists'
|
||||
let hookResults: Array<Record<string, unknown>> = []
|
||||
|
||||
const actualHooks = await import('../utils/hooks.js')
|
||||
const originalHooks = { ...actualHooks }
|
||||
mock.module('../utils/hooks.js', () => ({
|
||||
...actualHooks,
|
||||
executeStopHooks: async function* () {
|
||||
for (const result of hookResults) yield result
|
||||
},
|
||||
}))
|
||||
afterAll(() => {
|
||||
// mock.restore does not undo mock.module; later files in the same run
|
||||
// must see the real hooks.
|
||||
mock.module('../utils/hooks.js', () => originalHooks)
|
||||
mock.restore()
|
||||
})
|
||||
|
||||
const { handleStopHooks } = await import('./stopHooks.js')
|
||||
const { writeTeamFileAsync } = await import('../utils/swarm/teamHelpers.js')
|
||||
|
||||
let configDir: string
|
||||
const savedConfigDir = process.env.CLAUDE_CONFIG_DIR
|
||||
beforeEach(async () => {
|
||||
configDir = await mkdtemp(join(tmpdir(), 'goal-team-stop-'))
|
||||
process.env.CLAUDE_CONFIG_DIR = configDir
|
||||
hookResults = [
|
||||
{ preventContinuation: true, stopReason: 'goal' },
|
||||
{ blockingError: { blockingError: 'Prompt hook condition was not met: reviewers are still writing', command: GOAL_COMMAND } },
|
||||
]
|
||||
})
|
||||
afterEach(async () => {
|
||||
if (savedConfigDir === undefined) delete process.env.CLAUDE_CONFIG_DIR
|
||||
else process.env.CLAUDE_CONFIG_DIR = savedConfigDir
|
||||
await rm(configDir, { recursive: true, force: true })
|
||||
})
|
||||
|
||||
async function runLeadStopHooks(teamWorking: boolean) {
|
||||
await writeTeamFileAsync('review', {
|
||||
name: 'review', createdAt: 1, leadAgentId: 'team-lead@review', leadSessionId: 'lead-session',
|
||||
members: [
|
||||
{ agentId: 'team-lead@review', name: 'team-lead', joinedAt: 1, tmuxPaneId: '', cwd: configDir, subscriptions: [] },
|
||||
{ agentId: 'reader@review', name: 'reader', joinedAt: 1, tmuxPaneId: '', cwd: configDir, subscriptions: [], backendType: 'process', sessionId: 'worker', isActive: teamWorking },
|
||||
],
|
||||
})
|
||||
const appState = {
|
||||
toolPermissionContext: { mode: 'default' },
|
||||
tasks: {},
|
||||
teamContext: { teamName: 'review', leadAgentId: 'team-lead@review', isLeader: true, teammates: {} },
|
||||
}
|
||||
const toolUseContext = {
|
||||
getAppState: () => appState,
|
||||
setAppState: () => {},
|
||||
abortController: new AbortController(),
|
||||
options: { tools: [], mainLoopModel: 'fixture-model' },
|
||||
}
|
||||
const yielded: unknown[] = []
|
||||
const generator = handleStopHooks([], [], [] as never, {}, {}, toolUseContext as never, 'sdk')
|
||||
let next = await generator.next()
|
||||
while (!next.done) {
|
||||
yielded.push(next.value)
|
||||
next = await generator.next()
|
||||
}
|
||||
const text = JSON.stringify(yielded)
|
||||
return { result: next.value, text }
|
||||
}
|
||||
|
||||
test('an unmet goal ends the lead turn while its team is still working, to be re-checked when a member reports', async () => {
|
||||
const { result, text } = await runLeadStopHooks(true)
|
||||
expect(result).toEqual({ blockingErrors: [], preventContinuation: false })
|
||||
expect(text).toContain('Goal waiting for teammates: reviewers are still writing')
|
||||
})
|
||||
|
||||
test('an unmet goal keeps the lead working once its team is idle', async () => {
|
||||
const { result, text } = await runLeadStopHooks(false)
|
||||
expect(result.preventContinuation).toBe(false)
|
||||
expect(result.blockingErrors).toHaveLength(1)
|
||||
expect(text).toContain('Goal continuing: reviewers are still writing')
|
||||
})
|
||||
|
||||
test('a non-goal blocking hook still blocks even while the team works', async () => {
|
||||
hookResults = [{ blockingError: { blockingError: 'lint must pass', command: 'npm run lint' } }]
|
||||
const { result } = await runLeadStopHooks(true)
|
||||
expect(result.blockingErrors).toHaveLength(1)
|
||||
})
|
||||
+35
-2
@@ -44,6 +44,7 @@ import {
|
||||
import type { SystemPrompt } from '../utils/systemPromptType.js'
|
||||
import { getTaskListId, listTasks } from '../utils/tasks.js'
|
||||
import { getAgentName, getTeamName, isTeammate } from '../utils/teammate.js'
|
||||
import { hasTeamWorkInProgress } from '../utils/swarm/teamActivity.js'
|
||||
|
||||
/* eslint-disable @typescript-eslint/no-require-imports */
|
||||
const extractMemoriesModule = feature('EXTRACT_MEMORIES')
|
||||
@@ -78,19 +79,29 @@ export function shouldLetGoalPromptHookContinue(
|
||||
)
|
||||
}
|
||||
|
||||
export function formatGoalContinuationStatusOutput(reason: string): string {
|
||||
const normalizedReason = reason
|
||||
function normalizeGoalReason(reason: string): string {
|
||||
return reason
|
||||
.replace(/^Prompt hook condition was not met:\s*/i, '')
|
||||
.replace(/[<>&]/g, ' ')
|
||||
.replace(/\s+/g, ' ')
|
||||
.trim()
|
||||
.slice(0, 240)
|
||||
}
|
||||
|
||||
export function formatGoalContinuationStatusOutput(reason: string): string {
|
||||
const normalizedReason = normalizeGoalReason(reason)
|
||||
return normalizedReason
|
||||
? `Goal continuing: ${normalizedReason}`
|
||||
: 'Goal continuing: more work is required'
|
||||
}
|
||||
|
||||
export function formatGoalWaitingForTeamStatusOutput(reason: string): string {
|
||||
const normalizedReason = normalizeGoalReason(reason)
|
||||
return normalizedReason
|
||||
? `Goal waiting for teammates: ${normalizedReason}`
|
||||
: 'Goal waiting for teammates: more work is required'
|
||||
}
|
||||
|
||||
export async function* handleStopHooks(
|
||||
messagesForQuery: Message[],
|
||||
assistantMessages: AssistantMessage[],
|
||||
@@ -248,6 +259,7 @@ export async function* handleStopHooks(
|
||||
// Track whether any blockingError came from a goal hook, so we can
|
||||
// resolve the pending preventContinuation correctly after the loop.
|
||||
let goalBlockingErrorSeen = false
|
||||
const goalBlockingMessages = new Set<Message>()
|
||||
|
||||
for await (const result of generator) {
|
||||
if (result.message) {
|
||||
@@ -324,6 +336,7 @@ export async function* handleStopHooks(
|
||||
content: getStopHookMessage(result.blockingError),
|
||||
isMeta: true, // Hide from UI (shown in summary message instead)
|
||||
})
|
||||
if (isGoalHook) goalBlockingMessages.add(userMessage)
|
||||
blockingErrors.push(userMessage)
|
||||
yield userMessage
|
||||
hasOutput = true
|
||||
@@ -434,6 +447,26 @@ export async function* handleStopHooks(
|
||||
)
|
||||
}
|
||||
|
||||
// A team lead whose members are still working has nothing to do but wait
|
||||
// for their reports. Forcing it to continue makes it busy-poll with sleeps
|
||||
// and task-list reads (a full-context model call per loop) while teammate
|
||||
// messages can only reach it between turns. Let the turn end instead: a
|
||||
// member's report starts the next lead turn, and the goal is checked again
|
||||
// when that turn tries to stop.
|
||||
if (
|
||||
goalContinuationReason &&
|
||||
!preventedContinuation &&
|
||||
!toolUseContext.agentId &&
|
||||
blockingErrors.length > 0 &&
|
||||
blockingErrors.every(message => goalBlockingMessages.has(message)) &&
|
||||
hasTeamWorkInProgress(toolUseContext.getAppState())
|
||||
) {
|
||||
yield createCommandInputMessage(
|
||||
`<local-command-stdout>${formatGoalWaitingForTeamStatusOutput(goalContinuationReason)}</local-command-stdout>`,
|
||||
)
|
||||
return { blockingErrors: [], preventContinuation: false }
|
||||
}
|
||||
|
||||
if (goalContinuationReason) {
|
||||
yield createCommandInputMessage(
|
||||
`<local-command-stdout>${formatGoalContinuationStatusOutput(goalContinuationReason)}</local-command-stdout>`,
|
||||
|
||||
@@ -137,6 +137,8 @@ describe('TeamWatcher.extractMemberStatuses', () => {
|
||||
status: 'running',
|
||||
activity: 'active',
|
||||
currentTask: undefined,
|
||||
lastError: null,
|
||||
autoRetry: null,
|
||||
})
|
||||
expect(statuses[1]).toEqual({
|
||||
agentId: 'agent-worker',
|
||||
@@ -144,9 +146,26 @@ describe('TeamWatcher.extractMemberStatuses', () => {
|
||||
status: 'idle',
|
||||
activity: 'idle',
|
||||
currentTask: undefined,
|
||||
lastError: null,
|
||||
autoRetry: null,
|
||||
})
|
||||
})
|
||||
|
||||
it('reports a stopped process member, a failed turn, and a pending automatic retry', () => {
|
||||
const config = makeTeamConfig()
|
||||
const [lead, worker] = config.members as Array<Record<string, unknown>>
|
||||
Object.assign(worker!, { isActive: false, terminated: true })
|
||||
const stopped = watcher.extractMemberStatuses(config)[1]!
|
||||
// Stopped is restartable by a message, so it is neither idle nor exited.
|
||||
expect(stopped).toMatchObject({ status: 'idle', activity: 'stopped', lastError: null, autoRetry: null })
|
||||
|
||||
Object.assign(worker!, { terminated: false, lastError: 'API Error: 401 authentication_error' })
|
||||
expect(watcher.extractMemberStatuses(config)[1]).toMatchObject({ status: 'error', activity: 'idle', lastError: 'API Error: 401 authentication_error' })
|
||||
|
||||
Object.assign(lead!, { isActive: false, lastError: 'stream_truncated', autoRetry: { attempt: 1, max: 5, nextAt: 123 } })
|
||||
expect(watcher.extractMemberStatuses(config)[0]).toMatchObject({ status: 'idle', lastError: 'stream_truncated', autoRetry: { attempt: 1, max: 5, nextAt: 123 } })
|
||||
})
|
||||
|
||||
it('reports no activity for a member whose runner never recorded a turn', () => {
|
||||
const config = makeTeamConfig()
|
||||
delete (config.members[0] as Record<string, unknown>).isActive
|
||||
|
||||
@@ -1075,6 +1075,24 @@ describe('TeamService', () => {
|
||||
expect(worker.status).toBe('idle')
|
||||
})
|
||||
|
||||
it('reports stopped, failed and auto-retrying process members distinctly', async () => {
|
||||
const config = makeTeamConfig({ name: 'status-team' })
|
||||
const [lead, worker] = config.members as Array<Record<string, unknown>>
|
||||
Object.assign(worker!, { isActive: false, terminated: true })
|
||||
Object.assign(lead!, { isActive: false, lastError: 'stream_truncated', autoRetry: { attempt: 2, max: 5, nextAt: 99 } })
|
||||
await writeTeamConfig('status-team', config)
|
||||
let detail = await service.getTeam('status-team')
|
||||
// Stopped keeps its conversation and restarts on a message: not exited.
|
||||
expect(detail.members.find((m) => m.agentId === 'agent-worker')).toMatchObject({ status: 'idle', activity: 'stopped' })
|
||||
expect(detail.members.find((m) => m.agentId === 'agent-lead')).toMatchObject({ status: 'idle', lastError: 'stream_truncated', autoRetry: { attempt: 2, max: 5, nextAt: 99 } })
|
||||
|
||||
Object.assign(lead!, { autoRetry: undefined, lastError: 'API Error: 401 authentication_error' })
|
||||
await writeTeamConfig('status-team', config)
|
||||
detail = await service.getTeam('status-team')
|
||||
expect(detail.members.find((m) => m.agentId === 'agent-lead')).toMatchObject({ status: 'failed', lastError: 'API Error: 401 authentication_error' })
|
||||
expect(detail.members.find((m) => m.agentId === 'agent-lead')?.autoRetry).toBeUndefined()
|
||||
})
|
||||
|
||||
it('keeps delivered reports visible after the leader reads its mailbox', async () => {
|
||||
await writeTeamConfig('read-history-team', makeTeamConfig({ name: 'read-history-team' }))
|
||||
const report = { id: 'report-1', from: 'Worker Agent', text: 'Analysis complete', timestamp: '2026-09-27T15:25:36.411Z' }
|
||||
@@ -4366,6 +4384,21 @@ describe('TeamService', () => {
|
||||
})
|
||||
})
|
||||
|
||||
it('reports a member message that could not be written instead of claiming it was sent', async () => {
|
||||
await writeTeamConfig('blocked-mailbox-team', makeTeamConfig({ name: 'blocked-mailbox-team' }))
|
||||
// An inbox that cannot be read back can never take the message.
|
||||
await fs.mkdir(
|
||||
path.join(tmpDir, 'teams', 'blocked-mailbox-team', 'inboxes', 'Worker-Agent.json'),
|
||||
{ recursive: true },
|
||||
)
|
||||
|
||||
await expect(service.sendMemberMessage(
|
||||
'blocked-mailbox-team',
|
||||
'agent-worker',
|
||||
'Please review the latest diff',
|
||||
)).rejects.toMatchObject({ statusCode: 503, code: 'MAILBOX_UNAVAILABLE' })
|
||||
})
|
||||
|
||||
it('should send messages to inbox-discovered members', async () => {
|
||||
await writeTeamConfig('inbox-team', makeTeamConfig({ name: 'inbox-team' }))
|
||||
const inboxDir = path.join(tmpDir, 'teams', 'inbox-team', 'inboxes')
|
||||
|
||||
@@ -4376,7 +4376,9 @@ describe('WebSocket handler session isolation', () => {
|
||||
await new Promise((resolve) => setTimeout(resolve, 100))
|
||||
await flushMicrotasks(30)
|
||||
|
||||
expect(conversationService.stopSession).toHaveBeenCalledWith(sessionId)
|
||||
// A runtime restart replaces only the lead's process; approved team
|
||||
// members are independent processes and keep working.
|
||||
expect(conversationService.stopSession).toHaveBeenCalledWith(sessionId, { keepTeamWorkers: true })
|
||||
expect(conversationService.startSession).toHaveBeenCalledWith(
|
||||
sessionId,
|
||||
'/tmp',
|
||||
|
||||
@@ -29,6 +29,7 @@ import * as path from 'node:path'
|
||||
import { handleSideQuestionRoute } from './sideQuestions.js'
|
||||
import { sessionService } from '../services/sessionService.js'
|
||||
import { conversationService } from '../services/conversationService.js'
|
||||
import { endTeamsForParent } from '../services/teamPlanRuntime.js'
|
||||
import { ApiError, errorResponse } from '../middleware/errorHandler.js'
|
||||
import {
|
||||
closeSessionConnection,
|
||||
@@ -996,6 +997,8 @@ async function deleteSession(sessionId: string): Promise<Response> {
|
||||
closeSessionConnection(sessionId, 'session deleted')
|
||||
cleanupAdapterSessionMappings(sessionId)
|
||||
recentProjectsCache = null
|
||||
// A deleted lead ends its reviewed team for good; Stop and lead restarts only pause it.
|
||||
void endTeamsForParent(sessionId).catch(error => console.error(`[Sessions] Failed to end the deleted session's team: ${error}`))
|
||||
return Response.json({ ok: true })
|
||||
}
|
||||
|
||||
@@ -1018,6 +1021,7 @@ async function batchDeleteSessions(req: Request): Promise<Response> {
|
||||
for (const sessionId of result.successes) {
|
||||
closeSessionConnection(sessionId, 'session deleted')
|
||||
cleanupAdapterSessionMappings(sessionId)
|
||||
void endTeamsForParent(sessionId).catch(error => console.error(`[Sessions] Failed to end the deleted session's team: ${error}`))
|
||||
}
|
||||
if (result.successes.length > 0) {
|
||||
recentProjectsCache = null
|
||||
|
||||
@@ -3,7 +3,12 @@ import { join } from 'node:path'
|
||||
import { sanitizePath } from '../../../utils/sessionStoragePortable.js'
|
||||
// Deterministic worker protocol fixture. Never contacts anything except loopback.
|
||||
const args = process.argv.slice(2)
|
||||
const arg = (name: string) => args[args.indexOf(name) + 1]
|
||||
const arg = (name: string) => {
|
||||
const index = args.indexOf(name)
|
||||
return index === -1 ? undefined : args[index + 1]
|
||||
}
|
||||
// A restarted worker resumes its own transcript instead of starting a new one.
|
||||
const sessionId = arg('--session-id') ?? arg('--resume')
|
||||
const sdk = new WebSocket(arg('--sdk-url')!)
|
||||
sdk.onmessage = async event => {
|
||||
for (const line of String(event.data).trim().split('\n')) {
|
||||
@@ -13,15 +18,22 @@ sdk.onmessage = async event => {
|
||||
if (sdk.readyState !== WebSocket.OPEN) continue
|
||||
sdk.send(JSON.stringify({ type: 'control_response', response: { subtype: 'success', request_id: message.request_id, response: {} } }))
|
||||
} else if (message.type === 'user') {
|
||||
if (JSON.stringify(message.message?.content).includes('FIXTURE_SHUTDOWN')) { sdk.close(); setTimeout(() => process.exit(0), 5); continue }
|
||||
const content = JSON.stringify(message.message?.content)
|
||||
if (content.includes('FIXTURE_SHUTDOWN')) { sdk.close(); setTimeout(() => process.exit(0), 5); continue }
|
||||
const url = new URL(process.env.ANTHROPIC_BASE_URL!)
|
||||
if (url.hostname !== '127.0.0.1') throw new Error('Fixture refuses non-loopback upstream')
|
||||
const response = await fetch(url, { method: 'POST', headers: { 'content-type': 'application/json', 'x-api-key': process.env.ANTHROPIC_API_KEY ?? '', authorization: process.env.ANTHROPIC_AUTH_TOKEN ?? '' }, body: JSON.stringify({ model: arg('--model'), preset: JSON.parse(arg('--agents')!), presetType: process.env.CC_HAHA_TEAM_WORKER_PRESET_TYPE, presetSource: process.env.CC_HAHA_TEAM_WORKER_PRESET_SOURCE, omitClaudeMd: process.env.CC_HAHA_TEAM_WORKER_OMIT_CLAUDE_MD, subagentOverride: process.env.CLAUDE_CODE_SUBAGENT_MODEL }) })
|
||||
const response = await fetch(url, { method: 'POST', headers: { 'content-type': 'application/json', 'x-api-key': process.env.ANTHROPIC_API_KEY ?? '', authorization: process.env.ANTHROPIC_AUTH_TOKEN ?? '' }, body: JSON.stringify({ model: arg('--model'), preset: arg('--agents') ? JSON.parse(arg('--agents')!) : null, presetType: process.env.CC_HAHA_TEAM_WORKER_PRESET_TYPE, presetSource: process.env.CC_HAHA_TEAM_WORKER_PRESET_SOURCE, omitClaudeMd: process.env.CC_HAHA_TEAM_WORKER_OMIT_CLAUDE_MD, subagentOverride: process.env.CLAUDE_CODE_SUBAGENT_MODEL, resumed: arg('--resume') ?? null, prompt: message.message?.content }) })
|
||||
const text = await response.text()
|
||||
if (!args.includes('--no-session-persistence')) {
|
||||
const dir = join(process.env.CLAUDE_CONFIG_DIR!, 'projects', sanitizePath(process.cwd()))
|
||||
await mkdir(dir, { recursive: true })
|
||||
await appendFile(join(dir, `${arg('--session-id')}.jsonl`), JSON.stringify({ type: 'assistant', uuid: crypto.randomUUID(), timestamp: new Date().toISOString(), sessionId: arg('--session-id'), cwd: process.cwd(), teamName: arg('--team-name'), agentName: arg('--agent-name'), entrypoint: process.env.CC_HAHA_TRANSCRIPT_ENTRYPOINT, message: { role: 'assistant', content: [{ type: 'text', text }], model: arg('--model') } }) + '\n')
|
||||
await appendFile(join(dir, `${sessionId}.jsonl`), JSON.stringify({ type: 'assistant', uuid: crypto.randomUUID(), timestamp: new Date().toISOString(), sessionId, cwd: process.cwd(), teamName: arg('--team-name'), agentName: arg('--agent-name'), entrypoint: process.env.CC_HAHA_TRANSCRIPT_ENTRYPOINT, message: { role: 'assistant', content: [{ type: 'text', text }], model: arg('--model') } }) + '\n')
|
||||
}
|
||||
// The upstream fixture decides how the turn ends, like a real provider.
|
||||
if (text.startsWith('FIXTURE_CRASH')) { sdk.close(); setTimeout(() => process.exit(1), 5); continue }
|
||||
if (text.startsWith('FIXTURE_ERROR:')) {
|
||||
sdk.send(JSON.stringify({ type: 'result', subtype: 'success', is_error: true, result: text.slice('FIXTURE_ERROR:'.length) }))
|
||||
continue
|
||||
}
|
||||
sdk.send(JSON.stringify({ type: 'result', subtype: 'success', result: text }))
|
||||
}
|
||||
|
||||
@@ -205,6 +205,27 @@ test('stopping a parent stops its workers and cancels relayed pending approvals'
|
||||
expect(events).toContainEqual({ type: 'control_cancel_request', request_id: 'approval' })
|
||||
})
|
||||
|
||||
test('a lead process that exits on its own or restarts keeps its approved workers', async () => {
|
||||
const service = new ConversationService() as any
|
||||
const killed: string[] = []
|
||||
service.killProcess = (id: string) => killed.push(id)
|
||||
service.waitForProcessOutputDrain = async () => {}
|
||||
service.buildRuntimeExitMessage = () => 'CLI exited'
|
||||
const proc = { exited: Promise.resolve(1) }
|
||||
const results: any[] = []
|
||||
service.sessions.set('parent', { proc, outputCallbacks: [(message: any) => results.push(message)], pendingPermissionRequests: new Map(), sdkMessages: [], workDir: home, permissionMode: 'default' })
|
||||
service.sessions.set('child', { teamWorker: worker, pendingPermissionRequests: new Map() })
|
||||
await service.handleProcessExit('parent', proc, 1)
|
||||
expect(killed).toEqual([])
|
||||
expect(service.hasSession('child')).toBe(true)
|
||||
expect(results.at(-1)).toMatchObject({ type: 'result', is_error: true })
|
||||
|
||||
service.sessions.set('parent', { proc, outputCallbacks: [], pendingPermissionRequests: new Map() })
|
||||
service.stopSession('parent', { keepTeamWorkers: true })
|
||||
expect(killed).toEqual(['parent'])
|
||||
expect(service.hasSession('child')).toBe(true)
|
||||
})
|
||||
|
||||
test('approved roster launches, materializes canonical tasks, wakes an idle member, and cleans up', async () => {
|
||||
const { createHash } = await import('node:crypto')
|
||||
const { conversationService: runtimeService } = await import('./conversationService.js')
|
||||
|
||||
@@ -320,6 +320,23 @@ export type TeamWorkerStart = {
|
||||
systemPrompt: string
|
||||
tools?: string[]
|
||||
agentDefinition?: Record<string, unknown>
|
||||
/**
|
||||
* Restart a worker whose process stopped by resuming its own transcript.
|
||||
* A worker never inherits the parent's launch metadata, so resume is opt-in.
|
||||
*/
|
||||
resume?: boolean
|
||||
}
|
||||
|
||||
/**
|
||||
* Hooks for the server-owned Agent Teams runtime. Workers are independent
|
||||
* processes, so the lead's process lifecycle must not decide their fate on its
|
||||
* own: a lead restart keeps them, and Stop lets the runtime pause them first.
|
||||
*/
|
||||
export type TeamRuntimeListener = {
|
||||
/** Synchronously before Stop kills a lead's workers. */
|
||||
leadInterrupted?: (parentSessionId: string) => void
|
||||
/** After any CLI session finished starting. */
|
||||
sessionStarted?: (sessionId: string, info: { isTeamWorker: boolean }) => void
|
||||
}
|
||||
|
||||
export type SessionStartOptions = {
|
||||
@@ -352,6 +369,7 @@ export class ConversationStartupError extends Error {
|
||||
export class ConversationService {
|
||||
private sessions = new Map<string, SessionProcess>()
|
||||
private teamStopOperations = new Map<string, Promise<void>>()
|
||||
private teamRuntimeListeners = new Set<TeamRuntimeListener>()
|
||||
private deletedSessions = new Set<string>()
|
||||
private providerService = new ProviderService()
|
||||
private pendingPermissionModeChanges = new Map<string, Map<string, number>>()
|
||||
@@ -601,7 +619,9 @@ export class ConversationService {
|
||||
throw new ConversationStartupError('This temporary side chat has expired. Open a new side chat.', 'SESSION_DELETED')
|
||||
}
|
||||
const launchInfo = options?.teamWorker ? null : await sessionService.getSessionLaunchInfo(sessionId)
|
||||
const shouldResume = !!launchInfo && launchInfo.transcriptMessageCount > 0
|
||||
const shouldResume = options?.teamWorker
|
||||
? options.teamWorker.resume === true && !!await sessionService.findSessionFile(sessionId)
|
||||
: !!launchInfo && launchInfo.transcriptMessageCount > 0
|
||||
const shouldReplacePlaceholder =
|
||||
!side && !!launchInfo && launchInfo.transcriptMessageCount === 0
|
||||
const shouldCreateWorktree =
|
||||
@@ -866,6 +886,18 @@ export class ConversationService {
|
||||
}
|
||||
|
||||
console.log(`[ConversationService] CLI started successfully for ${sessionId}`)
|
||||
for (const listener of this.teamRuntimeListeners) {
|
||||
try {
|
||||
listener.sessionStarted?.(sessionId, { isTeamWorker: Boolean(options?.teamWorker) })
|
||||
} catch (error) {
|
||||
console.error('[ConversationService] Team runtime start hook failed', error)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
addTeamRuntimeListener(listener: TeamRuntimeListener): () => void {
|
||||
this.teamRuntimeListeners.add(listener)
|
||||
return () => this.teamRuntimeListeners.delete(listener)
|
||||
}
|
||||
|
||||
onOutput(sessionId: string, callback: (msg: any) => void): void {
|
||||
@@ -1151,12 +1183,23 @@ export class ConversationService {
|
||||
}
|
||||
|
||||
sendInterrupt(sessionId: string): boolean {
|
||||
// Stop ends all team activity immediately, but an approved team that is
|
||||
// already working is paused rather than destroyed: the runtime marks its
|
||||
// members as user-stopped before their processes die, and a later message
|
||||
// from the lead or the user resumes each member from its own transcript.
|
||||
for (const listener of this.teamRuntimeListeners) {
|
||||
try {
|
||||
listener.leadInterrupted?.(sessionId)
|
||||
} catch (error) {
|
||||
console.error('[ConversationService] Team runtime interrupt hook failed', error)
|
||||
}
|
||||
}
|
||||
for (const [childId, child] of this.sessions) {
|
||||
if (child.teamWorker?.parentSessionId === sessionId) this.stopSession(childId)
|
||||
}
|
||||
const stop = (this.teamStopOperations.get(sessionId) ?? Promise.resolve()).then(async () => {
|
||||
const runtime = await import('./teamPlanRuntime.js')
|
||||
await runtime.stopTeamPlanRuntimesForParent(sessionId)
|
||||
await runtime.stopTeamPlanRuntimesForParent(sessionId, { pauseReleased: true })
|
||||
}).catch(error => {
|
||||
console.error('[ConversationService] Failed to revoke interrupted team launch', error)
|
||||
}).finally(() => {
|
||||
@@ -1633,9 +1676,16 @@ export class ConversationService {
|
||||
pending.clear()
|
||||
}
|
||||
|
||||
stopSession(sessionId: string): void {
|
||||
for (const [childId, child] of this.sessions) {
|
||||
if (child.teamWorker?.parentSessionId === sessionId) this.stopSession(childId)
|
||||
/**
|
||||
* `keepTeamWorkers` is for restarting a lead's process (runtime or
|
||||
* permission change): approved team members are independent processes and
|
||||
* keep working while the lead's process is replaced.
|
||||
*/
|
||||
stopSession(sessionId: string, options?: { keepTeamWorkers?: boolean }): void {
|
||||
if (!options?.keepTeamWorkers) {
|
||||
for (const [childId, child] of this.sessions) {
|
||||
if (child.teamWorker?.parentSessionId === sessionId) this.stopSession(childId)
|
||||
}
|
||||
}
|
||||
const session = this.sessions.get(sessionId)
|
||||
if (!session) return
|
||||
@@ -1649,9 +1699,12 @@ export class ConversationService {
|
||||
async stopSessionAndWait(
|
||||
sessionId: string,
|
||||
timeoutMs = DESKTOP_CLI_GRACEFUL_SHUTDOWN_TIMEOUT_MS,
|
||||
options?: { keepTeamWorkers?: boolean },
|
||||
): Promise<void> {
|
||||
for (const [childId, child] of this.sessions) {
|
||||
if (child.teamWorker?.parentSessionId === sessionId) await this.stopSessionAndWait(childId, timeoutMs)
|
||||
if (!options?.keepTeamWorkers) {
|
||||
for (const [childId, child] of this.sessions) {
|
||||
if (child.teamWorker?.parentSessionId === sessionId) await this.stopSessionAndWait(childId, timeoutMs)
|
||||
}
|
||||
}
|
||||
const session = this.sessions.get(sessionId)
|
||||
if (!session) return
|
||||
@@ -1831,9 +1884,11 @@ export class ConversationService {
|
||||
|
||||
const activeSession = this.sessions.get(sessionId)
|
||||
if (activeSession?.proc === proc) {
|
||||
for (const [childId, child] of this.sessions) {
|
||||
if (child.teamWorker?.parentSessionId === sessionId) this.stopSession(childId)
|
||||
}
|
||||
// A lead process that dies on its own (crash, OOM, provider SDK fault)
|
||||
// must not take its approved team with it: members are independent
|
||||
// processes, their messages queue in the lead's mailbox, and the next
|
||||
// lead start receives the team snapshot again. Explicit stops (close,
|
||||
// delete, /clear) still stop workers through stopSession.
|
||||
this.cancelPendingControlRequests(
|
||||
activeSession,
|
||||
new Error('CLI session exited before the control request completed'),
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
import { join } from 'node:path'
|
||||
import { homedir } from 'node:os'
|
||||
import { listLeadTeamMemberIdentities } from '../../utils/swarm/teamHelpers.js'
|
||||
import { sessionService } from './sessionService.js'
|
||||
import { SearchService } from './searchService.js'
|
||||
import { conversationService } from './conversationService.js'
|
||||
@@ -93,6 +94,7 @@ export async function getSessionCollaborationService(): Promise<SessionCollabora
|
||||
}
|
||||
return titles
|
||||
},
|
||||
teamMemberIdentities: listLeadTeamMemberIdentities,
|
||||
async create(callerSessionId, input) {
|
||||
const workDir = input.workDir ?? await sessionService.getSessionWorkDir(callerSessionId)
|
||||
if (!workDir) throw new ApiError(409, 'The source session working directory is unavailable', 'SESSION_WORKSPACE_UNAVAILABLE')
|
||||
|
||||
@@ -387,6 +387,23 @@ describe('session collaboration', () => {
|
||||
expect(fresh).toMatchObject({ requestedTimeoutMs: 0, timeoutMs: 10_000 })
|
||||
}, 20_000)
|
||||
|
||||
test("a team lead waiting on its own members returns at once with guidance instead of blocking", async () => {
|
||||
deps.sessions.teamMemberIdentities = sessionId => sessionId === 'root' ? ['reviewer', 'reviewer@team', 'worker-session'] : []
|
||||
const startedAt = Date.now()
|
||||
const result = await service.wait(0, ['reviewer', 'worker-session'], 300_000, undefined, 'root')
|
||||
expect(Date.now() - startedAt).toBeLessThan(2_000)
|
||||
expect(result.guidance).toContain('Agent Team members')
|
||||
// Real collaboration targets still wait as before.
|
||||
const stillWaits = service.wait((await service.status()).revision, ['peer'], 1000, undefined, 'root')
|
||||
await service.send('root', 'peer', 'arrived', 'arrived-for-peer')
|
||||
expect((await stillWaits).guidance ?? '').not.toContain('Agent Team members')
|
||||
// Another session's member names are not this caller's team.
|
||||
deps.sessions.teamMemberIdentities = () => []
|
||||
const otherCaller = service.wait((await service.status()).revision, ['reviewer'], 1000, undefined, 'root')
|
||||
await service.send('root', 'reviewer', 'later', 'later')
|
||||
expect((await otherCaller).guidance ?? '').not.toContain('Agent Team members')
|
||||
})
|
||||
|
||||
test('wait cursor omits unchanged messages but returns a changed delivery receipt', async () => {
|
||||
await service.send('root', 'peer', 'message', 'stable')
|
||||
const first = await service.wait(0, ['peer'], 0)
|
||||
|
||||
@@ -45,6 +45,8 @@ export type SessionCollaborationDependencies = {
|
||||
create(callerSessionId: string, input: CollaborationCreateInput): Promise<{ sessionId: string; workDir?: string }>
|
||||
/** Optional display titles for snapshot members; keyed by session id. */
|
||||
titles?(sessionIds: string[]): Promise<Record<string, string>> | Record<string, string>
|
||||
/** Names and worker session ids of the Agent Team members this session leads. */
|
||||
teamMemberIdentities?(sessionId: string): Promise<string[]> | string[]
|
||||
}
|
||||
runtime: {
|
||||
getState?(sessionId: string): 'running' | 'blocked' | 'idle'
|
||||
@@ -62,6 +64,7 @@ export const COLLABORATION_READ_MAX_PAGES = 8
|
||||
export const COLLABORATION_WAIT_MIN_MS = 10_000
|
||||
export const COLLABORATION_WAIT_DEFAULT_MS = 30_000
|
||||
export const COLLABORATION_WAIT_MAX_MS = 300_000
|
||||
export const UNKNOWN_WAIT_TARGET_GUIDANCE = 'These are your Agent Team members, not collaboration sessions, so WaitSessions will never report on them. Their results, failures and questions arrive automatically as teammate messages. If you are waiting for teammates, end your turn instead of polling or sleeping.'
|
||||
export type CollaborationSnapshot = { revision: number; members: CollaborationMember[]; messages: CollaborationMessage[]; waitReason?: 'capacity_blocked'; guidance?: string; truncated?: boolean; omittedMessages?: number; omittedMembers?: number; requestedTimeoutMs?: number; timeoutMs?: number }
|
||||
|
||||
/** Mirrors the host's customTitle rule so the tool result carries the same name the session list shows. */
|
||||
@@ -392,6 +395,15 @@ export class SessionCollaborationService {
|
||||
const noteTimeout = (snapshot: CollaborationSnapshot): CollaborationSnapshot => requestedTimeoutMs === clampedTimeoutMs ? snapshot : { ...snapshot, requestedTimeoutMs, timeoutMs: clampedTimeoutMs, guidance: [`Requested timeout of ${requestedTimeoutMs}ms was clamped to ${clampedTimeoutMs}ms.`, snapshot.guidance].filter(Boolean).join('\n\n') }
|
||||
const inputVersion = callerSessionId ? this.userInputs.get(callerSessionId) ?? 0 : 0
|
||||
const current = await this.status(sessionIds)
|
||||
if (sessionIds?.length && callerSessionId && this.deps.sessions.teamMemberIdentities) {
|
||||
// A team lead that waits on its own members' names or worker session ids
|
||||
// would block for the full timeout: nothing here ever updates them,
|
||||
// because members report through teammate messages instead.
|
||||
const team = new Set(await Promise.resolve(this.deps.sessions.teamMemberIdentities(callerSessionId)).catch(() => []))
|
||||
if (sessionIds.every(id => team.has(id) && !this.store.members[id])) {
|
||||
return noteTimeout({ ...this.projectWait(current, afterRevision), guidance: UNKNOWN_WAIT_TARGET_GUIDANCE })
|
||||
}
|
||||
}
|
||||
const caller = callerSessionId ? this.store.members[callerSessionId] : undefined
|
||||
if (caller && caller.sessionId !== caller.rootSessionId) {
|
||||
const workers = Object.values(this.store.members).filter(member => member.rootSessionId === caller.rootSessionId && member.sessionId !== member.rootSessionId)
|
||||
|
||||
@@ -0,0 +1,447 @@
|
||||
import { afterEach, beforeEach, expect, spyOn, test } from 'bun:test'
|
||||
import { createHash } from 'node:crypto'
|
||||
import { mkdtemp, readFile, rm, writeFile } from 'node:fs/promises'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { join } from 'node:path'
|
||||
import { resetTerminalShellEnvironmentCacheForTests } from '../../utils/terminalShellEnvironment.js'
|
||||
|
||||
let home: string
|
||||
let originalEnv: NodeJS.ProcessEnv
|
||||
beforeEach(async () => {
|
||||
originalEnv = { ...process.env }
|
||||
home = await mkdtemp(join(tmpdir(), 'team-supervisor-'))
|
||||
process.env.HOME = home
|
||||
process.env.CLAUDE_CONFIG_DIR = home
|
||||
process.env.CC_HAHA_DISABLE_TERMINAL_SHELL_ENV = '1'
|
||||
process.env.CLAUDE_CLI_PATH = join(import.meta.dir, '__fixtures__', 'team-worker-cli.ts')
|
||||
resetTerminalShellEnvironmentCacheForTests()
|
||||
})
|
||||
afterEach(async () => {
|
||||
const runtime = await import('./teamPlanRuntime.js')
|
||||
runtime.setTeamRuntimeTimingForTests(null)
|
||||
for (const key of Object.keys(process.env)) if (!(key in originalEnv)) delete process.env[key]
|
||||
Object.assign(process.env, originalEnv)
|
||||
resetTerminalShellEnvironmentCacheForTests()
|
||||
await rm(home, { recursive: true, force: true })
|
||||
})
|
||||
|
||||
test('worker failures are classified as transient only for provider and transport hiccups', async () => {
|
||||
const { isTransientWorkerFailure } = await import('./teamPlanRuntime.js')
|
||||
for (const text of [
|
||||
'API Error: {"type":"error","error":{"type":"stream_truncated","message":"OpenAI Chat upstream stream ended without finish_reason"}}',
|
||||
'API Error: 529 {"type":"error","error":{"type":"overloaded_error"}}',
|
||||
'API Error: 502 Bad Gateway',
|
||||
'Request timed out',
|
||||
'API Error: Connection error.',
|
||||
'Repeated 529 Overloaded errors',
|
||||
'API Error: 429 Too Many Requests',
|
||||
]) expect(isTransientWorkerFailure(text)).toBe(true)
|
||||
for (const text of [
|
||||
'',
|
||||
'Prompt is too long',
|
||||
'API Error: 401 {"type":"error","error":{"type":"authentication_error"}}',
|
||||
'Credit balance is too low',
|
||||
'API Error: 400 {"type":"error","error":{"type":"invalid_request_error","message":"bad tool"}}',
|
||||
'Reached maximum turns',
|
||||
'Done: everything reviewed.',
|
||||
]) expect(isTransientWorkerFailure(text)).toBe(false)
|
||||
})
|
||||
|
||||
type Harness = Awaited<ReturnType<typeof startHarness>>
|
||||
|
||||
/**
|
||||
* A real ConversationService with fixture worker CLIs over a loopback SDK
|
||||
* bridge. `respond` decides how each upstream "model" call ends.
|
||||
*/
|
||||
async function startHarness(memberNames: string[]) {
|
||||
const { conversationService: service } = await import('./conversationService.js')
|
||||
const { ProviderService } = await import('./providerService.js')
|
||||
const { launchTeamPlanRuntime } = await import('./teamPlanRuntime.js')
|
||||
const { writeTeamFileAsync, getTeamDir } = await import('../../utils/swarm/teamHelpers.js')
|
||||
const { beginTaskListLifecycle, withTaskListLifecycleLock, getCanonicalTeamTaskListId } = await import('../../utils/tasks.js')
|
||||
const requests: Array<{ resumed: string | null; prompt: string }> = []
|
||||
const replies: string[] = []
|
||||
let held: Promise<void> | undefined
|
||||
const upstream = Bun.serve({ hostname: '127.0.0.1', port: 0, async fetch(request) {
|
||||
const body = await request.json() as { resumed?: string | null; prompt?: unknown }
|
||||
requests.push({ resumed: body.resumed ?? null, prompt: JSON.stringify(body.prompt ?? '') })
|
||||
await held
|
||||
return new Response(replies.shift() ?? 'fixture result')
|
||||
} })
|
||||
const bridge = Bun.serve<{ id: string }>({ hostname: '127.0.0.1', port: 0,
|
||||
fetch(request, server) {
|
||||
const url = new URL(request.url)
|
||||
const id = url.pathname.split('/').pop()!
|
||||
if (!service.authorizeSdkConnection(id, url.searchParams.get('token'))) return new Response('denied', { status: 401 })
|
||||
return server.upgrade(request, { data: { id } }) ? undefined : new Response('bad', { status: 400 })
|
||||
},
|
||||
websocket: { open(ws) { service.attachSdkConnection(ws.data.id, ws) }, message(ws, msg) { service.handleSdkPayload(ws.data.id, String(msg)) }, close(ws) { service.detachSdkConnection(ws.data.id, ws) } },
|
||||
})
|
||||
const oldPort = ProviderService.getServerPort()
|
||||
ProviderService.setServerPort(bridge.port!)
|
||||
const provider = await new ProviderService().addProvider({ presetId: 'custom', name: 'local', baseUrl: upstream.url.toString(), apiKey: 'fake', models: { main: 'cheap', sonnet: 'cheap', opus: 'capable', haiku: 'cheap' } })
|
||||
const parentId = `parent-${crypto.randomUUID()}`
|
||||
const teamName = `team-${crypto.randomUUID().slice(0, 8)}`
|
||||
const createdAt = Date.now()
|
||||
await writeTeamFileAsync(teamName, { name: teamName, createdAt, leadAgentId: `team-lead@${teamName}`, leadSessionId: parentId, members: [] } as any)
|
||||
const taskListId = getCanonicalTeamTaskListId(teamName)
|
||||
await withTaskListLifecycleLock(taskListId, () => beginTaskListLifecycle(taskListId, { teamName, createdAt, leadSessionId: parentId }))
|
||||
// The lead is a fixture process too; it only needs to answer control requests.
|
||||
await service.startSession(parentId, home, `ws://127.0.0.1:${bridge.port}/sdk/${parentId}?token=parent`, { providerId: provider.id, model: 'cheap', teamWorker: { parentSessionId: 'external', teamName: 'lead-fixture', memberId: 'lead', name: 'leader', systemPrompt: 'lead fixture' } })
|
||||
const members = memberNames.map(name => ({ id: name, name, agentType: 'research', prompt: `Work as ${name}`, runtime: { providerId: provider.id, modelId: 'cheap' }, agentSnapshot: { systemPrompt: 'Read only', tools: ['Read'] } }))
|
||||
const tasks = memberNames.map(name => ({ id: `task-${name}`, subject: `Task for ${name}`, ownerId: name, dependencies: [] }))
|
||||
const planId = crypto.randomUUID()
|
||||
const plan = { schemaVersion: 1, planId, sessionId: parentId, teamName, incarnationId: createHash('sha256').update(JSON.stringify([teamName, parentId, createdAt])).digest('hex'), revision: 2, state: 'launching', workDir: home, members, tasks, leaderRuntime: members[0]!.runtime, createdAt, updatedAt: createdAt, approvedSnapshot: { revision: 1, members, tasks, leaderRuntime: members[0]!.runtime, approvedAt: Date.now(), requestId: 'approve' } } as any
|
||||
await writeFile(join(getTeamDir(teamName), 'plan.json'), JSON.stringify(plan))
|
||||
const { memberIds } = await launchTeamPlanRuntime(plan)
|
||||
// The service records the running state the way TeamPlanService does after launch.
|
||||
await writeFile(join(getTeamDir(teamName), 'plan.json'), JSON.stringify({ ...plan, state: 'running', revision: 3, launch: { status: 'running', memberIds } }))
|
||||
const waitFor = async (condition: () => boolean | Promise<boolean>, label: string, timeoutMs = 8000) => {
|
||||
const end = Date.now() + timeoutMs
|
||||
while (!(await condition()) && Date.now() < end) await new Promise(resolve => setTimeout(resolve, 20))
|
||||
if (!(await condition())) throw new Error(`Timed out waiting for ${label}`)
|
||||
}
|
||||
await waitFor(() => requests.length >= memberNames.length, 'initial member turns')
|
||||
return {
|
||||
service, teamName, parentId, planId, memberIds, requests, replies, waitFor,
|
||||
/** Keeps model replies (and so member turns) pending until released. */
|
||||
holdReplies() {
|
||||
let release!: () => void
|
||||
held = new Promise(resolve => { release = resolve })
|
||||
return () => { held = undefined; release() }
|
||||
},
|
||||
async teamMember(name: string) {
|
||||
const { readTeamFile } = await import('../../utils/swarm/teamHelpers.js')
|
||||
return readTeamFile(teamName)?.members.find(member => member.name === name)
|
||||
},
|
||||
async leadNotifications() {
|
||||
const { readMailbox } = await import('../../utils/teammateMailbox.js')
|
||||
return (await readMailbox('team-lead', teamName)).map(message => {
|
||||
try { return { from: message.from, ...JSON.parse(message.text) } } catch { return { from: message.from, text: message.text } }
|
||||
})
|
||||
},
|
||||
async cleanup() {
|
||||
const runtime = await import('./teamPlanRuntime.js')
|
||||
await runtime.stopTeamPlanRuntime(planId)
|
||||
await service.stopAllSessionsAndWait(1000)
|
||||
ProviderService.setServerPort(oldPort)
|
||||
bridge.stop(true)
|
||||
upstream.stop(true)
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
async function messageMember(h: Harness, to: string, from: string, text: string) {
|
||||
const { writeToMailbox } = await import('../../utils/teammateMailbox.js')
|
||||
await writeToMailbox(to, { from, text, timestamp: new Date().toISOString() }, h.teamName)
|
||||
}
|
||||
|
||||
test('a member whose process died restarts from its own transcript when the lead messages it', async () => {
|
||||
const h = await startHarness(['reader'])
|
||||
try {
|
||||
const sessionId = h.memberIds.reader!
|
||||
h.replies.push('FIXTURE_CRASH')
|
||||
await messageMember(h, 'reader', 'team-lead', 'Check one more file')
|
||||
await h.waitFor(async () => (await h.teamMember('reader'))?.terminated === true, 'crash to be recorded')
|
||||
expect(h.service.hasSession(sessionId)).toBe(false)
|
||||
const crashNotice = (await h.leadNotifications()).find(item => item.type === 'idle_notification' && item.idleReason === 'failed')
|
||||
expect(crashNotice?.failureReason).toContain('restarts it from its saved conversation')
|
||||
|
||||
const before = h.requests.length
|
||||
await messageMember(h, 'reader', 'team-lead', 'Please continue')
|
||||
await h.waitFor(() => h.requests.length > before, 'restarted member turn')
|
||||
expect(h.requests.at(-1)?.resumed).toBe(sessionId)
|
||||
expect(h.requests.at(-1)?.prompt).toContain('Please continue')
|
||||
await h.waitFor(async () => (await h.teamMember('reader'))?.terminated === false, 'restart to be recorded')
|
||||
expect(h.service.hasSession(sessionId)).toBe(true)
|
||||
} finally {
|
||||
await h.cleanup()
|
||||
}
|
||||
}, 30_000)
|
||||
|
||||
test('transient turn failures are continued automatically and only exhausted retries reach the lead', async () => {
|
||||
const { setTeamRuntimeTimingForTests } = await import('./teamPlanRuntime.js')
|
||||
setTeamRuntimeTimingForTests({ autoContinueDelaysMs: [40, 40] })
|
||||
const h = await startHarness(['reader'])
|
||||
try {
|
||||
const truncated = 'FIXTURE_ERROR:API Error: {"type":"error","error":{"type":"stream_truncated","message":"OpenAI Chat upstream stream ended without finish_reason"}}'
|
||||
h.replies.push(truncated, truncated, truncated)
|
||||
const before = h.requests.length
|
||||
await messageMember(h, 'reader', 'team-lead', 'Review the runner')
|
||||
await h.waitFor(() => h.requests.length >= before + 3, 'two automatic continuations')
|
||||
const continuations = h.requests.slice(before + 1).filter(request => request.prompt.includes('[Team runtime notice]'))
|
||||
expect(continuations).toHaveLength(2)
|
||||
await h.waitFor(async () => (await h.leadNotifications()).some(item => item.idleReason === 'failed'), 'exhaustion notice')
|
||||
const failures = (await h.leadNotifications()).filter(item => item.idleReason === 'failed')
|
||||
expect(failures).toHaveLength(1)
|
||||
expect(failures[0].failureReason).toContain('automatic retries exhausted')
|
||||
// The member stayed alive throughout: a message still reaches it directly.
|
||||
expect(h.service.hasSession(h.memberIds.reader!)).toBe(true)
|
||||
} finally {
|
||||
await h.cleanup()
|
||||
}
|
||||
}, 30_000)
|
||||
|
||||
test('a successful turn reports its result to the lead and resets the retry budget', async () => {
|
||||
const { setTeamRuntimeTimingForTests } = await import('./teamPlanRuntime.js')
|
||||
setTeamRuntimeTimingForTests({ autoContinueDelaysMs: [40] })
|
||||
const h = await startHarness(['reader'])
|
||||
try {
|
||||
h.replies.push('FIXTURE_ERROR:API Error: 502 Bad Gateway', 'Found 3 issues in runner.ts')
|
||||
const before = h.requests.length
|
||||
await messageMember(h, 'reader', 'team-lead', 'Review the runner')
|
||||
await h.waitFor(async () => (await h.leadNotifications()).some(item => item.result === 'Found 3 issues in runner.ts'), 'result notification')
|
||||
expect(h.requests.length).toBe(before + 2)
|
||||
expect((await h.leadNotifications()).filter(item => item.idleReason === 'failed')).toHaveLength(0)
|
||||
const entry = await h.teamMember('reader') as any
|
||||
expect(entry.lastError).toBeUndefined()
|
||||
expect(entry.autoRetry).toBeUndefined()
|
||||
} finally {
|
||||
await h.cleanup()
|
||||
}
|
||||
}, 30_000)
|
||||
|
||||
test('a message to a failed member clears its failure as soon as the member starts on it', async () => {
|
||||
const h = await startHarness(['reader'])
|
||||
try {
|
||||
h.replies.push('FIXTURE_ERROR:API Error: 401 {"type":"error","error":{"type":"authentication_error","message":"invalid x-api-key"}}')
|
||||
await messageMember(h, 'reader', 'team-lead', 'Review the runner')
|
||||
await h.waitFor(async () => (await h.teamMember('reader') as any)?.lastError !== undefined, 'failure to be recorded')
|
||||
const release = h.holdReplies()
|
||||
try {
|
||||
await messageMember(h, 'reader', 'user', 'The key is fixed, please retry')
|
||||
await h.waitFor(async () => (await h.teamMember('reader'))?.isActive === true, 'recovery turn to start')
|
||||
const entry = await h.teamMember('reader') as any
|
||||
expect(entry.lastError).toBeUndefined()
|
||||
expect(entry.autoRetry).toBeUndefined()
|
||||
} finally {
|
||||
release()
|
||||
}
|
||||
await h.waitFor(async () => (await h.teamMember('reader'))?.isActive === false, 'recovery turn to finish')
|
||||
} finally {
|
||||
await h.cleanup()
|
||||
}
|
||||
}, 30_000)
|
||||
|
||||
test('an automatic continuation that cannot be sent reports the failure instead of a stale countdown', async () => {
|
||||
const { setTeamRuntimeTimingForTests } = await import('./teamPlanRuntime.js')
|
||||
setTeamRuntimeTimingForTests({ autoContinueDelaysMs: [40, 40] })
|
||||
const h = await startHarness(['reader'])
|
||||
try {
|
||||
h.replies.push('FIXTURE_ERROR:API Error: {"type":"error","error":{"type":"stream_truncated","message":"upstream stream ended without finish_reason"}}')
|
||||
const send = h.service.sendMessage.bind(h.service)
|
||||
const spy = spyOn(h.service, 'sendMessage').mockImplementation(async (...args: Parameters<typeof send>) =>
|
||||
String(args[1]).includes('[Team runtime notice]') ? false : send(...args))
|
||||
try {
|
||||
await messageMember(h, 'reader', 'team-lead', 'Review the runner')
|
||||
await h.waitFor(async () => (await h.leadNotifications()).some(item => item.idleReason === 'failed'), 'failure notice')
|
||||
} finally {
|
||||
spy.mockRestore()
|
||||
}
|
||||
const entry = await h.teamMember('reader') as any
|
||||
expect(entry.autoRetry).toBeUndefined()
|
||||
expect(entry.lastError).toContain('stream_truncated')
|
||||
expect(entry.isActive).toBe(false)
|
||||
const failure = (await h.leadNotifications()).find(item => item.idleReason === 'failed')
|
||||
expect(failure.failureReason).toContain('could not continue automatically')
|
||||
} finally {
|
||||
await h.cleanup()
|
||||
}
|
||||
}, 30_000)
|
||||
|
||||
test("the user's Stop pauses the team instead of ending it, and only an instruction resumes members", async () => {
|
||||
const h = await startHarness(['reader', 'writer'])
|
||||
try {
|
||||
const { isTeamPlanRuntimeActive } = await import('./teamPlanRuntime.js')
|
||||
const { getTeamDir } = await import('../../utils/swarm/teamHelpers.js')
|
||||
await h.waitFor(async () => (await h.leadNotifications()).length === 2, 'initial turn reports')
|
||||
const reportsBeforeStop = await h.leadNotifications()
|
||||
h.service.sendInterrupt(h.parentId)
|
||||
await h.service.waitForTeamWorkersStopped(h.parentId)
|
||||
expect(h.service.hasSession(h.memberIds.reader!)).toBe(false)
|
||||
expect(h.service.hasSession(h.memberIds.writer!)).toBe(false)
|
||||
expect(isTeamPlanRuntimeActive(h.planId)).toBe(true)
|
||||
expect(JSON.parse(await readFile(join(getTeamDir(h.teamName), 'plan.json'), 'utf8')).state).toBe('running')
|
||||
await h.waitFor(async () => (await h.teamMember('reader'))?.terminated === true && (await h.teamMember('writer'))?.terminated === true, 'members recorded as stopped')
|
||||
// Nothing new reaches the lead's mailbox: a delivered notice would start a
|
||||
// lead turn right after the user stopped everything.
|
||||
expect(await h.leadNotifications()).toEqual(reportsBeforeStop)
|
||||
|
||||
const before = h.requests.length
|
||||
await messageMember(h, 'writer', 'reader', 'peer note sent before the user stopped')
|
||||
await new Promise(resolve => setTimeout(resolve, 700))
|
||||
expect(h.requests.length).toBe(before)
|
||||
expect(h.service.hasSession(h.memberIds.writer!)).toBe(false)
|
||||
|
||||
await messageMember(h, 'writer', 'team-lead', 'The user wants you to continue')
|
||||
await h.waitFor(() => h.requests.length > before, 'resumed member turn')
|
||||
expect(h.requests.at(-1)?.resumed).toBe(h.memberIds.writer)
|
||||
expect(h.requests.at(-1)?.prompt).toContain('peer note sent before the user stopped')
|
||||
expect(h.requests.at(-1)?.prompt).toContain('The user wants you to continue')
|
||||
} finally {
|
||||
await h.cleanup()
|
||||
}
|
||||
}, 30_000)
|
||||
|
||||
test("after a Stop, the lead's next message tells it which members stopped and how to resume them", async () => {
|
||||
const h = await startHarness(['reader', 'writer'])
|
||||
try {
|
||||
const { deliverTeamPauseNotice } = await import('./teamPlanRuntime.js')
|
||||
await h.waitFor(async () => (await h.leadNotifications()).length === 2, 'initial turn reports')
|
||||
const sent: Array<[string, string]> = []
|
||||
const send = h.service.sendMessage.bind(h.service)
|
||||
const spy = spyOn(h.service, 'sendMessage').mockImplementation(async (...args: Parameters<typeof send>) => {
|
||||
sent.push([args[0], String(args[1])])
|
||||
return send(...args)
|
||||
})
|
||||
try {
|
||||
// Without a Stop there is nothing to tell.
|
||||
await deliverTeamPauseNotice(h.parentId)
|
||||
expect(sent).toEqual([])
|
||||
|
||||
h.service.sendInterrupt(h.parentId)
|
||||
await h.service.waitForTeamWorkersStopped(h.parentId)
|
||||
await deliverTeamPauseNotice(h.parentId)
|
||||
await deliverTeamPauseNotice(h.parentId)
|
||||
|
||||
const notices = sent.filter(([sessionId]) => sessionId === h.parentId).map(([, text]) => text)
|
||||
expect(notices).toHaveLength(1)
|
||||
expect(notices[0]).toContain('[Team runtime notice]')
|
||||
expect(notices[0]).toMatch(/- reader: #\d+ Task for reader \(pending\)/)
|
||||
expect(notices[0]).toMatch(/- writer: #\d+ Task for writer \(pending\)/)
|
||||
expect(notices[0]).toContain('SendMessage each member that still has work')
|
||||
} finally {
|
||||
spy.mockRestore()
|
||||
}
|
||||
} finally {
|
||||
await h.cleanup()
|
||||
}
|
||||
}, 30_000)
|
||||
|
||||
test('stopping and resuming the team repeatedly is never mistaken for a crash loop', async () => {
|
||||
const h = await startHarness(['reader'])
|
||||
try {
|
||||
await h.waitFor(async () => (await h.leadNotifications()).length === 1, 'initial turn report')
|
||||
// More Stop/resume cycles than the crash-loop guard allows restarts.
|
||||
for (let cycle = 1; cycle <= 4; cycle++) {
|
||||
h.service.sendInterrupt(h.parentId)
|
||||
await h.service.waitForTeamWorkersStopped(h.parentId)
|
||||
await h.waitFor(async () => (await h.teamMember('reader'))?.terminated === true, `stop ${cycle} to be recorded`)
|
||||
const before = h.requests.length
|
||||
await messageMember(h, 'reader', 'team-lead', `Continue after stop ${cycle}`)
|
||||
await h.waitFor(() => h.requests.length > before, `resume ${cycle}`)
|
||||
expect(h.requests.at(-1)?.prompt).toContain(`Continue after stop ${cycle}`)
|
||||
await h.waitFor(async () => (await h.teamMember('reader'))?.terminated === false, `resume ${cycle} to be recorded`)
|
||||
}
|
||||
expect((await h.leadNotifications()).filter(item => item.idleReason === 'failed')).toEqual([])
|
||||
expect((await h.teamMember('reader') as any).lastError).toBeUndefined()
|
||||
} finally {
|
||||
await h.cleanup()
|
||||
}
|
||||
}, 60_000)
|
||||
|
||||
test('replacing the lead process keeps members working and restores the roster on the next lead start', async () => {
|
||||
const h = await startHarness(['reader'])
|
||||
try {
|
||||
const control = spyOn(h.service, 'requestControl')
|
||||
h.service.stopSession(h.parentId, { keepTeamWorkers: true })
|
||||
expect(h.service.hasSession(h.parentId)).toBe(false)
|
||||
expect(h.service.hasSession(h.memberIds.reader!)).toBe(true)
|
||||
const { ProviderService } = await import('./providerService.js')
|
||||
await h.service.startSession(h.parentId, home, `ws://127.0.0.1:${ProviderService.getServerPort()}/sdk/${h.parentId}?token=parent-2`, { teamWorker: { parentSessionId: 'external', teamName: 'lead-fixture', memberId: 'lead', name: 'leader', systemPrompt: 'lead fixture' } })
|
||||
await h.waitFor(() => control.mock.calls.some(([sessionId, request]) => sessionId === h.parentId && (request as { subtype?: string }).subtype === 'team_runtime_snapshot'), 'team snapshot after lead restart')
|
||||
const before = h.requests.length
|
||||
await messageMember(h, 'reader', 'team-lead', 'Still there?')
|
||||
await h.waitFor(() => h.requests.length > before, 'member turn after lead restart')
|
||||
expect(h.requests.at(-1)?.resumed).toBeNull()
|
||||
control.mockRestore()
|
||||
} finally {
|
||||
await h.cleanup()
|
||||
}
|
||||
}, 30_000)
|
||||
|
||||
test('a team whose owner was lost is rehydrated dormant and its members restart on demand', async () => {
|
||||
const h = await startHarness(['reader'])
|
||||
try {
|
||||
const runtime = await import('./teamPlanRuntime.js')
|
||||
// Simulate a server restart: the processes and the in-memory owner are gone.
|
||||
await h.service.stopSessionAndWait(h.memberIds.reader!)
|
||||
runtime.forgetTeamPlanRuntimesForTests()
|
||||
expect(runtime.isTeamPlanRuntimeActive(h.planId)).toBe(false)
|
||||
expect(await runtime.rehydrateTeamPlanRuntimesForSession(h.parentId)).toBe(true)
|
||||
expect(runtime.isTeamPlanRuntimeActive(h.planId)).toBe(true)
|
||||
await h.waitFor(async () => (await h.teamMember('reader'))?.terminated === true, 'dormant member state')
|
||||
const before = h.requests.length
|
||||
await messageMember(h, 'reader', 'team-lead', 'Resume after restart')
|
||||
await h.waitFor(() => h.requests.length > before, 'restarted member turn')
|
||||
expect(h.requests.at(-1)?.resumed).toBe(h.memberIds.reader)
|
||||
} finally {
|
||||
await h.cleanup()
|
||||
}
|
||||
}, 30_000)
|
||||
|
||||
test('after a server restart the next lead start re-owns the team and restores its roster', async () => {
|
||||
const h = await startHarness(['reader'])
|
||||
try {
|
||||
const runtime = await import('./teamPlanRuntime.js')
|
||||
const { ProviderService } = await import('./providerService.js')
|
||||
await h.service.stopSessionAndWait(h.memberIds.reader!)
|
||||
h.service.stopSession(h.parentId, { keepTeamWorkers: true })
|
||||
runtime.forgetTeamPlanRuntimesForTests()
|
||||
const control = spyOn(h.service, 'requestControl')
|
||||
// A real lead is an ordinary session, never a team worker.
|
||||
await h.service.startSession(h.parentId, home, `ws://127.0.0.1:${ProviderService.getServerPort()}/sdk/${h.parentId}?token=parent-3`, {})
|
||||
await h.waitFor(() => runtime.isTeamPlanRuntimeActive(h.planId), 're-owned team', 20_000)
|
||||
await h.waitFor(() => control.mock.calls.some(([sessionId, request]) => sessionId === h.parentId && (request as { subtype?: string }).subtype === 'team_runtime_snapshot'), 'restored roster')
|
||||
control.mockRestore()
|
||||
const before = h.requests.length
|
||||
await messageMember(h, 'reader', 'team-lead', 'Pick up where you left off')
|
||||
await h.waitFor(() => h.requests.length > before, 'member restarted after server restart')
|
||||
expect(h.requests.at(-1)?.resumed).toBe(h.memberIds.reader)
|
||||
} finally {
|
||||
await h.cleanup()
|
||||
}
|
||||
}, 30_000)
|
||||
|
||||
test('ending a lead (/clear or deletion) stops its team and removes only its reviewed team', async () => {
|
||||
const h = await startHarness(['reader'])
|
||||
try {
|
||||
const runtime = await import('./teamPlanRuntime.js')
|
||||
const { getTeamDir, writeTeamFileAsync, readTeamFile } = await import('../../utils/swarm/teamHelpers.js')
|
||||
// Another session's team and an unreviewed (CLI) team of the same lead stay.
|
||||
await writeTeamFileAsync('other-team', { name: 'other-team', createdAt: 1, leadAgentId: 'team-lead@other-team', leadSessionId: 'someone-else', reviewRequired: true, members: [] })
|
||||
const { mutateTeamFileAsync } = await import('../../utils/swarm/teamHelpers.js')
|
||||
await mutateTeamFileAsync(h.teamName, team => ({ ...team, reviewRequired: true }))
|
||||
await runtime.endTeamsForParent(h.parentId)
|
||||
expect(runtime.isTeamPlanRuntimeActive(h.planId)).toBe(false)
|
||||
expect(h.service.hasSession(h.memberIds.reader!)).toBe(false)
|
||||
expect(readTeamFile(h.teamName)).toBeNull()
|
||||
await expect(readFile(join(getTeamDir(h.teamName), 'plan.json'), 'utf8')).rejects.toThrow()
|
||||
expect(readTeamFile('other-team')?.name).toBe('other-team')
|
||||
} finally {
|
||||
await h.cleanup()
|
||||
}
|
||||
}, 30_000)
|
||||
|
||||
test('a newly unblocked task wakes its idle owner once', async () => {
|
||||
const { setTeamRuntimeTimingForTests } = await import('./teamPlanRuntime.js')
|
||||
setTeamRuntimeTimingForTests({ unblockedTaskWakeDelayMs: 0 })
|
||||
const h = await startHarness(['reader'])
|
||||
try {
|
||||
const { createTask, updateTask, getCanonicalTeamTaskListId } = await import('../../utils/tasks.js')
|
||||
const taskListId = getCanonicalTeamTaskListId(h.teamName)
|
||||
const blocker = await createTask(taskListId, { subject: 'Blocker', description: '', status: 'in_progress', blocks: [], blockedBy: [] })
|
||||
const follow = await createTask(taskListId, { subject: 'Follow-up review', description: '', status: 'pending', owner: 'reader', blocks: [], blockedBy: [blocker] })
|
||||
await h.waitFor(async () => (await h.teamMember('reader'))?.isActive === false, 'member idle')
|
||||
const before = h.requests.length
|
||||
await new Promise(resolve => setTimeout(resolve, 400))
|
||||
expect(h.requests.length).toBe(before)
|
||||
await updateTask(taskListId, blocker, { status: 'completed' })
|
||||
await h.waitFor(() => h.requests.length > before, 'wake for the ready task')
|
||||
expect(h.requests.at(-1)?.prompt).toContain(`Task #${follow}`)
|
||||
await new Promise(resolve => setTimeout(resolve, 600))
|
||||
expect(h.requests.filter(request => request.prompt.includes(`Task #${follow}`))).toHaveLength(1)
|
||||
} finally {
|
||||
await h.cleanup()
|
||||
}
|
||||
}, 30_000)
|
||||
@@ -1,6 +1,7 @@
|
||||
import { createHash, randomUUID } from 'node:crypto'
|
||||
import { stat } from 'node:fs/promises'
|
||||
import { isValidTeamMemberName, type TeamPlanRecord, type TeamPlanMember } from '../../shared/teamPlan.js'
|
||||
import { readdir, readFile, stat } from 'node:fs/promises'
|
||||
import { join } from 'node:path'
|
||||
import { isValidTeamMemberName, teamPlanRecordSchema, type TeamPlanRecord, type TeamPlanMember } from '../../shared/teamPlan.js'
|
||||
import { conversationService } from './conversationService.js'
|
||||
import { ProviderService } from './providerService.js'
|
||||
import { CLAUDE_OFFICIAL_PROVIDER_ID } from '../types/provider.js'
|
||||
@@ -9,23 +10,85 @@ import { getModelReasoningCapabilityOverride, isModelReasoningEffort, normalizeM
|
||||
import { getPresetDefaultEnv, getPresetReasoningProviderKind } from './providerRuntimeEnv.js'
|
||||
import { validateTeamPlanPresetSource } from '../../utils/swarm/teamPlanPresetSource.js'
|
||||
import { readTeamPlan, findTeamPlanForSession, mutateTeamPlan } from '../../utils/swarm/teamPlanStore.js'
|
||||
import { readTeamFile, writeTeamFileAsync } from '../../utils/swarm/teamHelpers.js'
|
||||
import { createTask, listTasks, updateTask, withTaskListLifecycleLock, getCanonicalTeamTaskListId } from '../../utils/tasks.js'
|
||||
import { readUnreadMessages, markMessagesAsReadByPredicate, writeToMailbox, createIdleNotification } from '../../utils/teammateMailbox.js'
|
||||
import { cleanupTeamDirectories, getTeamDir, mutateTeamFileAsync, readTeamFile, writeTeamFileAsync, type TeamFile } from '../../utils/swarm/teamHelpers.js'
|
||||
import { getTeamsDir as getTeamsDirectory } from '../../utils/envUtils.js'
|
||||
import { TEAM_LEAD_NAME } from '../../utils/swarm/constants.js'
|
||||
import { createTask, listTasks, updateTask, withTaskListLifecycleLock, getCanonicalTeamTaskListId, type Task } from '../../utils/tasks.js'
|
||||
import { readUnreadMessages, markMessagesAsReadByPredicate, writeToMailbox, createIdleNotification, formatTeammateMessages, type TeammateMessage } from '../../utils/teammateMailbox.js'
|
||||
import { formatTeammateAutoContinuePrompt, isTransientTurnFailure, summarizeTurnFailure, TEAMMATE_AUTO_CONTINUE_DELAYS_MS } from '../../utils/swarm/turnFailure.js'
|
||||
|
||||
const providerService = new ProviderService()
|
||||
const launches = new Map<string, { parentId: string; plan: TeamPlanRecord; released: boolean; children: string[]; timer?: ReturnType<typeof setInterval>; stopped: boolean }>()
|
||||
const essentialTools = ['SendMessage', 'TaskCreate', 'TaskGet', 'TaskList', 'TaskUpdate']
|
||||
|
||||
const SUPERVISOR_INTERVAL_MS = 250
|
||||
const WORKER_AUTO_CONTINUE_DELAYS_MS = TEAMMATE_AUTO_CONTINUE_DELAYS_MS
|
||||
const WORKER_RESTART_WINDOW_MS = 10 * 60_000
|
||||
const WORKER_MAX_RESTARTS_IN_WINDOW = 3
|
||||
/** Let a member finish its own bookkeeping before a newly ready task wakes it. */
|
||||
const UNBLOCKED_TASK_WAKE_DELAY_MS = 3_000
|
||||
const timing = {
|
||||
autoContinueDelaysMs: WORKER_AUTO_CONTINUE_DELAYS_MS,
|
||||
unblockedTaskWakeDelayMs: UNBLOCKED_TASK_WAKE_DELAY_MS,
|
||||
}
|
||||
|
||||
/** Tests shorten the supervisor's real-time backoffs; `null` restores them. */
|
||||
export function setTeamRuntimeTimingForTests(overrides: Partial<typeof timing> | null): void {
|
||||
timing.autoContinueDelaysMs = overrides?.autoContinueDelaysMs ?? WORKER_AUTO_CONTINUE_DELAYS_MS
|
||||
timing.unblockedTaskWakeDelayMs = overrides?.unblockedTaskWakeDelayMs ?? UNBLOCKED_TASK_WAKE_DELAY_MS
|
||||
}
|
||||
const RESULT_TEXT_LIMIT = 4_000
|
||||
const FAILURE_REASON_LIMIT = 200
|
||||
|
||||
type WorkerRuntime = {
|
||||
member: TeamPlanMember
|
||||
sessionId: string
|
||||
released: boolean
|
||||
starting?: Promise<boolean>
|
||||
autoContinueAttempts: number
|
||||
autoContinueTimer?: ReturnType<typeof setTimeout>
|
||||
restartHistory: number[]
|
||||
/** Automatic restarts stopped after a crash loop; only an explicit message retries. */
|
||||
restartsExhausted: boolean
|
||||
/** The user's Stop ended this process; resuming it is not a crash restart. */
|
||||
stoppedByUser?: boolean
|
||||
wokenForTaskIds: Set<string>
|
||||
lastResultAt?: number
|
||||
}
|
||||
|
||||
type TeamLaunch = {
|
||||
parentId: string
|
||||
plan: TeamPlanRecord
|
||||
createdAt: number
|
||||
released: boolean
|
||||
/** True once every member has been released and the supervisor owns the team. */
|
||||
running: boolean
|
||||
children: string[]
|
||||
workers: Map<string, WorkerRuntime>
|
||||
timer?: ReturnType<typeof setInterval>
|
||||
supervising: boolean
|
||||
stopped: boolean
|
||||
/** Set by the user's Stop: hold automatic work until the lead or the user acts again. */
|
||||
pausedAt?: number
|
||||
/** The lead has not yet been told what the user's last Stop did to this team. */
|
||||
pauseNoticePending?: boolean
|
||||
permissionMode: string
|
||||
}
|
||||
|
||||
const launches = new Map<string, TeamLaunch>()
|
||||
|
||||
function incarnationOf(team: Pick<TeamFile, 'name' | 'leadSessionId' | 'createdAt'>): string {
|
||||
return createHash('sha256').update(JSON.stringify([team.name, team.leadSessionId || '', team.createdAt])).digest('hex')
|
||||
}
|
||||
|
||||
/** Validation is local: checking a proposal never invokes a model or discovers credentials. */
|
||||
export async function validateTeamPlanRuntime(plan: TeamPlanRecord): Promise<TeamPlanRecord> {
|
||||
if (!(await stat(plan.workDir)).isDirectory()) throw new Error('Team working directory is unavailable')
|
||||
const team = plan.teamName ? readTeamFile(plan.teamName) : null
|
||||
if (plan.teamName && (!team || team.leadSessionId !== plan.sessionId || createHash('sha256').update(JSON.stringify([team.name, team.leadSessionId || '', team.createdAt])).digest('hex') !== plan.incarnationId)) throw new Error('Team generation no longer exists')
|
||||
if (plan.teamName && (!team || team.leadSessionId !== plan.sessionId || incarnationOf(team) !== plan.incarnationId)) throw new Error('Team generation no longer exists')
|
||||
const members: TeamPlanMember[] = []
|
||||
for (const member of plan.members) {
|
||||
if (!isValidTeamMemberName(member.name)) throw new Error(`Invalid teammate name: ${member.name}`)
|
||||
if (team?.members.some(existing => existing.name === member.name)) throw new Error(`Member already exists: ${member.name}`)
|
||||
if (team?.members.some(existing => existing.name === member.name)) throw new Error(`Member already exists: ${member.name}. Message the existing member to give it more work; a stopped member restarts from its saved conversation.`)
|
||||
const snapshot = plan.agentCatalog?.[member.agentType]
|
||||
if (!snapshot || !snapshot.systemPrompt.trim()) throw new Error(`Agent preset is unavailable: ${member.agentType}`)
|
||||
validateTeamPlanPresetSource(snapshot)
|
||||
@@ -105,17 +168,405 @@ async function materializeTasks(plan: TeamPlanRecord): Promise<Record<string, st
|
||||
return mapping
|
||||
}
|
||||
|
||||
// ── Worker failure classification ───────────────────────────────────────────
|
||||
|
||||
/** A worker's CLI `result` text after a failed turn (see utils/swarm/turnFailure). */
|
||||
export const isTransientWorkerFailure = isTransientTurnFailure
|
||||
|
||||
function capText(text: string, limit: number): string {
|
||||
const clean = text.replace(/[\u0000-\u0008\u000b\u000c\u000e-\u001f]/g, '').trim()
|
||||
return clean.length > limit ? `${clean.slice(0, limit)}…` : clean
|
||||
}
|
||||
|
||||
function firstLine(text: string): string {
|
||||
return summarizeTurnFailure(text, FAILURE_REASON_LIMIT)
|
||||
}
|
||||
|
||||
// ── Team file updates ───────────────────────────────────────────────────────
|
||||
|
||||
type MemberEntry = TeamFile['members'][number]
|
||||
|
||||
async function updateWorkerEntry(launch: TeamLaunch, worker: WorkerRuntime, update: (entry: MemberEntry) => MemberEntry): Promise<void> {
|
||||
try {
|
||||
await mutateTeamFileAsync(launch.plan.teamName, team => {
|
||||
if (team.createdAt !== launch.createdAt) return
|
||||
const index = team.members.findIndex(entry => entry.sessionId === worker.sessionId)
|
||||
if (index === -1) return
|
||||
const members = [...team.members]
|
||||
members[index] = update(members[index]!)
|
||||
return { ...team, members }
|
||||
})
|
||||
} catch (error) {
|
||||
// The team may have been deleted between ticks; the supervisor notices on its next pass.
|
||||
console.warn(`[TeamPlanRuntime] cannot update member ${worker.member.name}:`, error instanceof Error ? error.message : error)
|
||||
}
|
||||
}
|
||||
|
||||
function withoutFailure(entry: MemberEntry): MemberEntry {
|
||||
const { lastError: _lastError, autoRetry: _autoRetry, ...rest } = entry as MemberEntry & { lastError?: string; autoRetry?: unknown }
|
||||
return rest
|
||||
}
|
||||
|
||||
async function notifyLead(launch: TeamLaunch, worker: WorkerRuntime, options: { idleReason: 'available' | 'failed'; result?: string; failureReason?: string }): Promise<void> {
|
||||
const notification = createIdleNotification(worker.member.name, {
|
||||
idleReason: options.idleReason,
|
||||
...(options.result ? { result: capText(options.result, RESULT_TEXT_LIMIT) } : {}),
|
||||
...(options.failureReason ? { failureReason: options.failureReason } : {}),
|
||||
})
|
||||
const write = () => writeToMailbox(TEAM_LEAD_NAME, { from: worker.member.name, timestamp: new Date().toISOString(), text: JSON.stringify(notification) }, launch.plan.teamName)
|
||||
// A lost failure notice is exactly how a member silently drops out of a long
|
||||
// run, so a contended mailbox gets one more attempt.
|
||||
if ((await write()) === false) {
|
||||
await new Promise(resolve => setTimeout(resolve, 1_000))
|
||||
if ((await write()) === false) console.error(`[TeamPlanRuntime] could not notify the lead about ${worker.member.name}`)
|
||||
}
|
||||
}
|
||||
|
||||
// ── Worker processes ────────────────────────────────────────────────────────
|
||||
|
||||
function workerSdkUrl(sessionId: string): string {
|
||||
const url = new URL(`ws://127.0.0.1:${ProviderService.getServerPort()}/sdk/${sessionId}`)
|
||||
url.searchParams.set('token', randomUUID())
|
||||
return url.toString()
|
||||
}
|
||||
|
||||
async function startWorkerProcess(launch: TeamLaunch, worker: WorkerRuntime, resume: boolean): Promise<void> {
|
||||
const { member } = worker
|
||||
const snapshot = member.agentSnapshot
|
||||
if (!snapshot) throw new Error(`Missing approved preset for ${member.name}`)
|
||||
const tools = snapshot.tools ? [...new Set([...snapshot.tools, ...essentialTools])] : undefined
|
||||
// A restarted member follows the lead's current permission mode, like a
|
||||
// freshly approved one would.
|
||||
if (resume && conversationService.hasSession(launch.parentId)) {
|
||||
launch.permissionMode = conversationService.getSessionPermissionMode(launch.parentId)
|
||||
}
|
||||
try {
|
||||
await conversationService.startSession(worker.sessionId, launch.plan.workDir, workerSdkUrl(worker.sessionId), {
|
||||
providerId: member.runtime.providerId === CLAUDE_OFFICIAL_PROVIDER_ID ? null : member.runtime.providerId,
|
||||
model: member.runtime.modelId,
|
||||
effort: member.runtime.effortLevel,
|
||||
permissionMode: launch.permissionMode,
|
||||
teamWorker: {
|
||||
parentSessionId: launch.parentId, teamName: launch.plan.teamName, memberId: member.id, name: member.name,
|
||||
systemPrompt: snapshot.systemPrompt, tools,
|
||||
agentDefinition: { ...snapshot, initialPrompt: undefined, effort: member.runtime.effortLevel },
|
||||
...(resume ? { resume: true } : {}),
|
||||
},
|
||||
})
|
||||
await conversationService.requestControl(worker.sessionId, { subtype: 'set_model', model: member.runtime.modelId }, 30_000)
|
||||
} catch (error) {
|
||||
await conversationService.stopSessionAndWait(worker.sessionId)
|
||||
throw error
|
||||
}
|
||||
conversationService.onOutput(worker.sessionId, message => {
|
||||
if (message?.type !== 'result') return
|
||||
void handleWorkerResult(launch, worker, message).catch(error => console.error(`[TeamPlanRuntime] cannot record ${member.name}'s turn`, error))
|
||||
})
|
||||
}
|
||||
|
||||
/**
|
||||
* Restart a member whose process is gone by resuming its own transcript, so a
|
||||
* crash, a lead restart or the user's Stop never costs the member its work.
|
||||
*/
|
||||
async function restartWorker(launch: TeamLaunch, worker: WorkerRuntime, reason: 'message' | 'user'): Promise<boolean> {
|
||||
if (worker.starting) return worker.starting
|
||||
const now = Date.now()
|
||||
worker.restartHistory = worker.restartHistory.filter(at => now - at < WORKER_RESTART_WINDOW_MS)
|
||||
// Stopping the team is the user's choice, not a crash: however often the
|
||||
// user stops and resumes, bringing a member back never trips the guard.
|
||||
const resumingFromStop = worker.stoppedByUser === true
|
||||
worker.stoppedByUser = false
|
||||
if (reason === 'message' && !resumingFromStop && (worker.restartsExhausted || worker.restartHistory.length >= WORKER_MAX_RESTARTS_IN_WINDOW)) {
|
||||
if (!worker.restartsExhausted) {
|
||||
worker.restartsExhausted = true
|
||||
await updateWorkerEntry(launch, worker, entry => ({ ...entry, isActive: false, terminated: true, lastError: 'Stopped restarting after repeated exits. A message from the user restarts it.' }))
|
||||
await notifyLead(launch, worker, { idleReason: 'failed', failureReason: `${worker.member.name} exited ${worker.restartHistory.length} times in ${WORKER_RESTART_WINDOW_MS / 60_000} minutes; automatic restarts are paused` })
|
||||
}
|
||||
return false
|
||||
}
|
||||
if (!resumingFromStop) worker.restartHistory.push(now)
|
||||
if (reason === 'user') worker.restartsExhausted = false
|
||||
const start = (async () => {
|
||||
try {
|
||||
await startWorkerProcess(launch, worker, true)
|
||||
if (launch.stopped) {
|
||||
await conversationService.stopSessionAndWait(worker.sessionId)
|
||||
return false
|
||||
}
|
||||
await updateWorkerEntry(launch, worker, entry => withoutFailure({ ...entry, terminated: false }))
|
||||
return true
|
||||
} catch (error) {
|
||||
const message = error instanceof Error ? error.message : String(error)
|
||||
console.error(`[TeamPlanRuntime] cannot restart ${worker.member.name}:`, message)
|
||||
await updateWorkerEntry(launch, worker, entry => ({ ...entry, isActive: false, terminated: true, lastError: firstLine(`Restart failed: ${message}`) }))
|
||||
return false
|
||||
} finally {
|
||||
worker.starting = undefined
|
||||
}
|
||||
})()
|
||||
worker.starting = start
|
||||
return start
|
||||
}
|
||||
|
||||
function cancelAutoContinue(worker: WorkerRuntime): void {
|
||||
if (worker.autoContinueTimer) clearTimeout(worker.autoContinueTimer)
|
||||
worker.autoContinueTimer = undefined
|
||||
}
|
||||
|
||||
function autoContinuePrompt(reason: string, attempt: number): string {
|
||||
return formatTeammateAutoContinuePrompt(reason, attempt, timing.autoContinueDelaysMs.length)
|
||||
}
|
||||
|
||||
async function autoContinueWorker(launch: TeamLaunch, worker: WorkerRuntime, reason: string, attempt: number): Promise<void> {
|
||||
worker.autoContinueTimer = undefined
|
||||
if (launch.stopped || launch.pausedAt || !conversationService.hasSession(worker.sessionId)) return
|
||||
// Queued mail resumes the member anyway, with the newer context.
|
||||
if ((await readUnreadMessages(worker.member.name, launch.plan.teamName)).length > 0) return
|
||||
const entry = readTeamFile(launch.plan.teamName)?.members.find(member => member.sessionId === worker.sessionId)
|
||||
if (entry?.isActive) return
|
||||
await updateWorkerEntry(launch, worker, current => ({ ...current, isActive: true }))
|
||||
const sent = await conversationService.sendMessage(worker.sessionId, autoContinuePrompt(reason, attempt))
|
||||
if (sent) return
|
||||
// No retry is scheduled any more: report the failure instead of leaving a
|
||||
// stale countdown that also keeps a /goal lead waiting on this member.
|
||||
await updateWorkerEntry(launch, worker, current => {
|
||||
const { autoRetry: _autoRetry, ...rest } = current as MemberEntry & { autoRetry?: unknown }
|
||||
return { ...rest, isActive: false, lastError: reason }
|
||||
})
|
||||
await notifyLead(launch, worker, { idleReason: 'failed', failureReason: `${reason} (could not continue automatically; message ${worker.member.name} to retry)` })
|
||||
}
|
||||
|
||||
async function handleWorkerResult(launch: TeamLaunch, worker: WorkerRuntime, message: { is_error?: boolean; result?: unknown }): Promise<void> {
|
||||
if (launch.stopped) return
|
||||
worker.lastResultAt = Date.now()
|
||||
const alive = conversationService.hasSession(worker.sessionId)
|
||||
const text = typeof message.result === 'string' ? message.result : ''
|
||||
const failed = !alive || message.is_error === true
|
||||
if (launch.pausedAt) {
|
||||
// The user's Stop already accounted for this turn; a burst of failure
|
||||
// notices would only make the lead restart members the user just halted.
|
||||
await updateWorkerEntry(launch, worker, entry => ({ ...entry, isActive: false, ...(alive ? {} : { terminated: true }) }))
|
||||
return
|
||||
}
|
||||
if (!failed) {
|
||||
worker.autoContinueAttempts = 0
|
||||
cancelAutoContinue(worker)
|
||||
await updateWorkerEntry(launch, worker, entry => withoutFailure({ ...entry, isActive: false }))
|
||||
await notifyLead(launch, worker, { idleReason: 'available', ...(text ? { result: text } : {}) })
|
||||
return
|
||||
}
|
||||
const reason = firstLine(text || 'The member process exited')
|
||||
if (alive && isTransientWorkerFailure(text) && worker.autoContinueAttempts < timing.autoContinueDelaysMs.length) {
|
||||
const attempt = ++worker.autoContinueAttempts
|
||||
const delay = timing.autoContinueDelaysMs[attempt - 1]!
|
||||
cancelAutoContinue(worker)
|
||||
worker.autoContinueTimer = setTimeout(() => {
|
||||
void autoContinueWorker(launch, worker, reason, attempt).catch(error => console.error(`[TeamPlanRuntime] cannot continue ${worker.member.name}`, error))
|
||||
}, delay)
|
||||
worker.autoContinueTimer.unref?.()
|
||||
await updateWorkerEntry(launch, worker, entry => ({ ...entry, isActive: false, lastError: reason, autoRetry: { attempt, max: timing.autoContinueDelaysMs.length, nextAt: Date.now() + delay } }))
|
||||
return
|
||||
}
|
||||
const exhausted = alive && isTransientWorkerFailure(text)
|
||||
await updateWorkerEntry(launch, worker, entry => {
|
||||
const { autoRetry: _autoRetry, ...rest } = entry as MemberEntry & { autoRetry?: unknown }
|
||||
return { ...rest, isActive: false, lastError: reason, ...(alive ? {} : { terminated: true }) }
|
||||
})
|
||||
await notifyLead(launch, worker, {
|
||||
idleReason: 'failed',
|
||||
failureReason: alive
|
||||
? exhausted ? `${reason} (automatic retries exhausted; message ${worker.member.name} to continue)` : reason
|
||||
: `${worker.member.name}'s process exited (${reason}). Messaging it restarts it from its saved conversation.`,
|
||||
})
|
||||
}
|
||||
|
||||
// ── Supervisor ──────────────────────────────────────────────────────────────
|
||||
|
||||
async function deliverToWorker(launch: TeamLaunch, worker: WorkerRuntime, messages: TeammateMessage[]): Promise<void> {
|
||||
cancelAutoContinue(worker)
|
||||
// A new instruction starts a fresh retry budget: the lead or the user has
|
||||
// looked at the failure and decided the member should go on.
|
||||
worker.autoContinueAttempts = 0
|
||||
// The message replaces any scheduled retry and answers the last failure;
|
||||
// the turn it starts records its own outcome.
|
||||
await updateWorkerEntry(launch, worker, entry => withoutFailure({ ...entry, isActive: true, terminated: false }))
|
||||
const accepted = await conversationService.sendMessage(worker.sessionId, formatTeammateMessages(messages))
|
||||
if (!accepted) {
|
||||
await updateWorkerEntry(launch, worker, entry => ({ ...entry, isActive: false }))
|
||||
return
|
||||
}
|
||||
const ids = new Set(messages.map(message => message.id).filter(Boolean))
|
||||
const legacy = new Set(messages.filter(message => !message.id).map(message => JSON.stringify([message.from, message.timestamp, message.text])))
|
||||
await markMessagesAsReadByPredicate(worker.member.name, message => message.id ? ids.has(message.id) : legacy.has(JSON.stringify([message.from, message.timestamp, message.text])), launch.plan.teamName)
|
||||
}
|
||||
|
||||
/**
|
||||
* Wake an idle member when a task it owns becomes ready (its dependencies just
|
||||
* completed). Process members have no poll loop of their own, so without this
|
||||
* a dependency chain stalls until the lead happens to notice.
|
||||
*/
|
||||
async function wakeForReadyTask(launch: TeamLaunch, worker: WorkerRuntime, entry: MemberEntry | undefined, tasks: Task[]): Promise<void> {
|
||||
if (!entry || entry.isActive || worker.autoContinueTimer || !conversationService.hasSession(worker.sessionId)) return
|
||||
if (worker.lastResultAt === undefined || Date.now() - worker.lastResultAt < timing.unblockedTaskWakeDelayMs) return
|
||||
const unresolved = new Set(tasks.filter(task => task.status !== 'completed').map(task => task.id))
|
||||
const ready = tasks.find(task => task.owner === worker.member.name && task.status === 'pending' && !worker.wokenForTaskIds.has(task.id)
|
||||
&& task.blockedBy.length > 0 && task.blockedBy.every(id => !unresolved.has(id)))
|
||||
if (!ready) return
|
||||
worker.wokenForTaskIds.add(ready.id)
|
||||
await deliverToWorker(launch, worker, [{
|
||||
from: 'task-list',
|
||||
text: `Task #${ready.id} assigned to you is now ready because its dependencies are complete: ${ready.subject}\nStart it now with TaskUpdate (status in_progress), and mark it completed when done.`,
|
||||
timestamp: new Date().toISOString(),
|
||||
read: false,
|
||||
}])
|
||||
}
|
||||
|
||||
async function superviseLaunch(launch: TeamLaunch): Promise<void> {
|
||||
const team = readTeamFile(launch.plan.teamName)
|
||||
if (!team || team.createdAt !== launch.createdAt) {
|
||||
await stopTeamPlanRuntime(launch.plan.planId)
|
||||
return
|
||||
}
|
||||
const leadAlive = conversationService.hasSession(launch.parentId)
|
||||
let tasks: Task[] | undefined
|
||||
for (const worker of launch.workers.values()) {
|
||||
// A restart in flight delivers on a later pass, once its process is ready;
|
||||
// it never holds up the rest of the team.
|
||||
if (!worker.released || launch.stopped || worker.starting) continue
|
||||
// A worker can be stopped without a result (its lead's session was closed
|
||||
// or reaped); keep the roster honest so the UI shows it as stopped.
|
||||
const entry = team.members.find(member => member.sessionId === worker.sessionId)
|
||||
if (entry && entry.terminated !== true && !conversationService.hasSession(worker.sessionId)) {
|
||||
await updateWorkerEntry(launch, worker, current => ({ ...current, isActive: false, terminated: true }))
|
||||
}
|
||||
const messages = await readUnreadMessages(worker.member.name, launch.plan.teamName)
|
||||
if (messages.length === 0) {
|
||||
if (!leadAlive || launch.pausedAt) continue
|
||||
tasks ??= await listTasks(getCanonicalTeamTaskListId(launch.plan.teamName)).catch(() => [])
|
||||
await wakeForReadyTask(launch, worker, entry, tasks)
|
||||
continue
|
||||
}
|
||||
if (launch.pausedAt) {
|
||||
// Only a new instruction from the lead or the user resumes a paused team.
|
||||
const resume = messages.some(message => (message.from === TEAM_LEAD_NAME || message.from === 'user') && Date.parse(message.timestamp) >= launch.pausedAt!)
|
||||
if (!resume) continue
|
||||
launch.pausedAt = undefined
|
||||
}
|
||||
if (!conversationService.hasSession(worker.sessionId)) {
|
||||
const fromUser = messages.some(message => message.from === 'user')
|
||||
// Without a lead nobody coordinates the restarted member; a direct user
|
||||
// message is the exception.
|
||||
if (!leadAlive && !fromUser) continue
|
||||
void restartWorker(launch, worker, fromUser ? 'user' : 'message')
|
||||
continue
|
||||
}
|
||||
await deliverToWorker(launch, worker, messages)
|
||||
}
|
||||
}
|
||||
|
||||
function startSupervisor(launch: TeamLaunch): void {
|
||||
if (launch.timer) return
|
||||
launch.timer = setInterval(() => {
|
||||
if (launch.supervising || launch.stopped) return
|
||||
launch.supervising = true
|
||||
void superviseLaunch(launch)
|
||||
.catch(error => { console.error('[TeamPlanRuntime] team supervision failed', error) })
|
||||
.finally(() => { launch.supervising = false })
|
||||
}, SUPERVISOR_INTERVAL_MS)
|
||||
launch.timer.unref?.()
|
||||
}
|
||||
|
||||
// ── Lifecycle hooks ─────────────────────────────────────────────────────────
|
||||
|
||||
async function sendTeamSnapshot(parentId: string, teamName: string, createdAt: number): Promise<void> {
|
||||
await conversationService.requestControl(parentId, { subtype: 'team_runtime_snapshot', team_name: teamName, created_at: createdAt })
|
||||
}
|
||||
|
||||
/** The user's Stop: every member stops now, nothing is lost, the next instruction resumes. */
|
||||
function pauseLaunchesForLead(parentSessionId: string): void {
|
||||
for (const launch of launches.values()) {
|
||||
if (launch.parentId !== parentSessionId || launch.stopped || !launch.running) continue
|
||||
launch.pausedAt = Date.now()
|
||||
launch.pauseNoticePending = true
|
||||
for (const worker of launch.workers.values()) {
|
||||
cancelAutoContinue(worker)
|
||||
worker.stoppedByUser = true
|
||||
}
|
||||
// No notice goes to the lead's mailbox: delivering it would start a lead
|
||||
// turn right after the user stopped everything. Messaging a stopped member
|
||||
// restarts it, so the lead needs no special knowledge to continue later.
|
||||
void mutateTeamFileAsync(launch.plan.teamName, team => {
|
||||
if (team.createdAt !== launch.createdAt) return
|
||||
const sessions = new Set([...launch.workers.values()].map(worker => worker.sessionId))
|
||||
return { ...team, members: team.members.map(entry => entry.sessionId && sessions.has(entry.sessionId) ? { ...entry, isActive: false, terminated: true } : entry) }
|
||||
}).catch(error => console.error('[TeamPlanRuntime] cannot record the paused team', error))
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The lead's first message from the user after a Stop: tell it what the Stop
|
||||
* did to its team. Nothing is said at the Stop itself, which would start a lead
|
||||
* turn the user just stopped, but without this a lead told to "continue" waits
|
||||
* for members that are no longer running. The lead decides from the user's
|
||||
* words whether the members go on.
|
||||
*/
|
||||
export async function deliverTeamPauseNotice(parentSessionId: string): Promise<void> {
|
||||
for (const launch of launches.values()) {
|
||||
if (launch.parentId !== parentSessionId || launch.stopped || !launch.pauseNoticePending) continue
|
||||
launch.pauseNoticePending = false
|
||||
const stopped = [...launch.workers.values()].filter(worker => worker.released && worker.stoppedByUser)
|
||||
if (stopped.length === 0) continue
|
||||
const tasks = await listTasks(getCanonicalTeamTaskListId(launch.plan.teamName)).catch(() => [] as Task[])
|
||||
const lines = stopped.map(worker => {
|
||||
const open = tasks.filter(task => task.owner === worker.member.name && task.status !== 'completed')
|
||||
return open.length > 0
|
||||
? `- ${worker.member.name}: ${open.map(task => `#${task.id} ${task.subject} (${task.status})`).join('; ')}`
|
||||
: `- ${worker.member.name}: no unfinished task`
|
||||
})
|
||||
await conversationService.sendMessage(parentSessionId, [
|
||||
`[Team runtime notice] The user's Stop also stopped your team ${launch.plan.teamName}. These members are stopped and keep their work so far:`,
|
||||
...lines,
|
||||
'A stopped member does nothing until it is messaged: if the work should go on, SendMessage each member that still has work, and it resumes from its saved conversation where it left off. Do not spawn replacements or redo their tasks. If the user wants the team to stay stopped, leave them.',
|
||||
].join('\n'))
|
||||
}
|
||||
}
|
||||
|
||||
let hooksInstalled = false
|
||||
function ensureRuntimeHooks(): void {
|
||||
if (hooksInstalled) return
|
||||
hooksInstalled = true
|
||||
conversationService.addTeamRuntimeListener({
|
||||
leadInterrupted: pauseLaunchesForLead,
|
||||
sessionStarted: (sessionId, { isTeamWorker }) => {
|
||||
// A lead that restarted (provider/permission change, crash, reopen, app
|
||||
// restart) lost its in-memory roster; give it back so it can keep
|
||||
// coordinating, re-owning the team first if the server restarted.
|
||||
const ownedLaunch = () => [...launches.values()].find(item => item.parentId === sessionId && item.running && !item.stopped)
|
||||
// Workers never lead a team; skip the disk scan a re-own needs.
|
||||
if (!ownedLaunch() && isTeamWorker) return
|
||||
void (async () => {
|
||||
if (!ownedLaunch()) await rehydrateTeamPlanRuntimesForSession(sessionId)
|
||||
const launch = ownedLaunch()
|
||||
if (launch) await sendTeamSnapshot(sessionId, launch.plan.teamName, launch.createdAt)
|
||||
})().catch(error => console.error('[TeamPlanRuntime] cannot restore the team after a lead start', error))
|
||||
},
|
||||
})
|
||||
}
|
||||
|
||||
// ── Public lifecycle ────────────────────────────────────────────────────────
|
||||
|
||||
export async function stopTeamPlanRuntime(planId: string): Promise<void> {
|
||||
const launch = launches.get(planId)
|
||||
if (!launch) return
|
||||
launch.stopped = true
|
||||
if (launch.timer) clearInterval(launch.timer)
|
||||
for (const worker of launch.workers.values()) cancelAutoContinue(worker)
|
||||
launches.delete(planId)
|
||||
await Promise.allSettled(launch.children.map(id => conversationService.stopSessionAndWait(id)))
|
||||
const plan = launch.plan
|
||||
await withTaskListLifecycleLock(getCanonicalTeamTaskListId(plan.teamName), async () => {
|
||||
const team = readTeamFile(plan.teamName)
|
||||
if (!team || createHash('sha256').update(JSON.stringify([team.name, team.leadSessionId || '', team.createdAt])).digest('hex') !== plan.incarnationId) return
|
||||
if (!team || incarnationOf(team) !== plan.incarnationId) return
|
||||
team.members = team.members.flatMap(member => {
|
||||
if (!member.sessionId || !launch.children.includes(member.sessionId)) return [member]
|
||||
return launch.released ? [{ ...member, isActive: false, terminated: true }] : []
|
||||
@@ -133,25 +584,27 @@ export async function stopTeamPlanRuntime(planId: string): Promise<void> {
|
||||
export async function launchTeamPlanRuntime(plan: TeamPlanRecord): Promise<{ memberIds: Record<string, string> }> {
|
||||
const approved = plan.approvedSnapshot
|
||||
if (!approved || approved.revision !== plan.revision - 1 || plan.state !== 'launching') throw new Error('Team plan is not approved for launch')
|
||||
ensureRuntimeHooks()
|
||||
await conversationService.waitForTeamWorkersStopped(plan.sessionId)
|
||||
const durable = await readTeamPlan(plan.teamName)
|
||||
if (!durable || durable.planId !== plan.planId || durable.incarnationId !== plan.incarnationId || durable.state !== 'launching' || durable.revision !== plan.revision || durable.approvedSnapshot?.requestId !== approved.requestId) throw new Error('Team launch authorization has been revoked')
|
||||
if (launches.has(plan.planId)) throw new Error('Team plan already has a launch in progress')
|
||||
if (!conversationService.hasSession(plan.sessionId)) throw new Error('Team leader must be connected before launching')
|
||||
const members = approved.members
|
||||
const launch = { parentId: plan.sessionId, plan, released: false, children: [] as string[], stopped: false, timer: undefined as ReturnType<typeof setInterval> | undefined }
|
||||
const launch: TeamLaunch = {
|
||||
parentId: plan.sessionId, plan, createdAt: 0, released: false, running: false, children: [], workers: new Map(),
|
||||
supervising: false, stopped: false, permissionMode: conversationService.getSessionPermissionMode(plan.sessionId),
|
||||
}
|
||||
launches.set(plan.planId, launch)
|
||||
const memberIds: Record<string, string> = {}
|
||||
const permissionMode = conversationService.getSessionPermissionMode(plan.sessionId)
|
||||
let taskMapping: Record<string, string> = {}
|
||||
let createdAt = 0
|
||||
let executionStarted = false
|
||||
try {
|
||||
await withTaskListLifecycleLock(getCanonicalTeamTaskListId(plan.teamName), async () => {
|
||||
const team = readTeamFile(plan.teamName)
|
||||
if (!team || team.leadSessionId !== plan.sessionId || createHash('sha256').update(JSON.stringify([team.name, team.leadSessionId || '', team.createdAt])).digest('hex') !== plan.incarnationId) throw new Error('Team generation no longer exists')
|
||||
if (!team || team.leadSessionId !== plan.sessionId || incarnationOf(team) !== plan.incarnationId) throw new Error('Team generation no longer exists')
|
||||
team.reviewRequired = true
|
||||
createdAt = team.createdAt
|
||||
launch.createdAt = team.createdAt
|
||||
await writeTeamFileAsync(plan.teamName, team)
|
||||
})
|
||||
taskMapping = await materializeTasks(plan)
|
||||
@@ -160,25 +613,9 @@ export async function launchTeamPlanRuntime(plan: TeamPlanRecord): Promise<{ mem
|
||||
const id = randomUUID()
|
||||
launch.children.push(id)
|
||||
memberIds[member.id] = id
|
||||
const url = new URL(`ws://127.0.0.1:${ProviderService.getServerPort()}/sdk/${id}`)
|
||||
url.searchParams.set('token', randomUUID())
|
||||
const snapshot = member.agentSnapshot
|
||||
if (!snapshot) throw new Error(`Missing approved preset for ${member.name}`)
|
||||
const tools = snapshot.tools ? [...new Set([...snapshot.tools, ...essentialTools])] : undefined
|
||||
try {
|
||||
await conversationService.startSession(id, plan.workDir, url.toString(), {
|
||||
providerId: member.runtime.providerId === CLAUDE_OFFICIAL_PROVIDER_ID ? null : member.runtime.providerId, model: member.runtime.modelId, effort: member.runtime.effortLevel, permissionMode,
|
||||
teamWorker: {
|
||||
parentSessionId: plan.sessionId, teamName: plan.teamName, memberId: member.id, name: member.name,
|
||||
systemPrompt: snapshot.systemPrompt, tools,
|
||||
agentDefinition: { ...snapshot, initialPrompt: undefined, effort: member.runtime.effortLevel },
|
||||
},
|
||||
})
|
||||
await conversationService.requestControl(id, { subtype: 'set_model', model: member.runtime.modelId }, 30_000)
|
||||
} catch (error) {
|
||||
await conversationService.stopSessionAndWait(id)
|
||||
throw error
|
||||
}
|
||||
const worker: WorkerRuntime = { member, sessionId: id, released: false, autoContinueAttempts: 0, restartHistory: [], restartsExhausted: false, wokenForTaskIds: new Set() }
|
||||
launch.workers.set(member.id, worker)
|
||||
await startWorkerProcess(launch, worker, false)
|
||||
return id
|
||||
}, async (member, id) => {
|
||||
if (launch.stopped || !conversationService.hasSession(plan.sessionId)) throw new Error('Team launch cancelled')
|
||||
@@ -186,7 +623,7 @@ export async function launchTeamPlanRuntime(plan: TeamPlanRecord): Promise<{ mem
|
||||
if (member === members[0]) {
|
||||
await withTaskListLifecycleLock(getCanonicalTeamTaskListId(plan.teamName), async () => {
|
||||
const team = readTeamFile(plan.teamName)
|
||||
if (!team || team.createdAt !== createdAt) throw new Error('Team generation changed during launch')
|
||||
if (!team || team.createdAt !== launch.createdAt) throw new Error('Team generation changed during launch')
|
||||
for (const entry of members) {
|
||||
if (team.members.some(old => old.name === entry.name)) throw new Error(`Member already exists: ${entry.name}`)
|
||||
team.members.push({ agentId: `${entry.name}@${plan.teamName}`, name: entry.name, agentType: entry.agentType,
|
||||
@@ -196,22 +633,8 @@ export async function launchTeamPlanRuntime(plan: TeamPlanRecord): Promise<{ mem
|
||||
}
|
||||
await writeTeamFileAsync(plan.teamName, team)
|
||||
})
|
||||
await conversationService.requestControl(plan.sessionId, { subtype: 'team_runtime_snapshot', team_name: plan.teamName, created_at: createdAt })
|
||||
await sendTeamSnapshot(plan.sessionId, plan.teamName, launch.createdAt)
|
||||
}
|
||||
conversationService.onOutput(id, message => {
|
||||
if (message?.type !== 'result') return
|
||||
void withTaskListLifecycleLock(getCanonicalTeamTaskListId(plan.teamName), async () => {
|
||||
const team = readTeamFile(plan.teamName)
|
||||
if (!team || team.createdAt !== createdAt) return
|
||||
const entry = team.members.find(entry => entry.sessionId === id)
|
||||
if (entry) { entry.isActive = false; if (!conversationService.hasSession(id)) entry.terminated = true; await writeTeamFileAsync(plan.teamName, team) }
|
||||
}).catch(error => console.error('[TeamPlanRuntime] cannot update idle member', error))
|
||||
const failed = !conversationService.hasSession(id) || message.is_error
|
||||
void writeToMailbox('team-lead', {
|
||||
from: member.name, timestamp: new Date().toISOString(),
|
||||
text: JSON.stringify(createIdleNotification(member.name, { idleReason: failed ? 'failed' : 'available', summary: typeof message.result === 'string' ? message.result.slice(0, 1000) : undefined })),
|
||||
}, plan.teamName)
|
||||
})
|
||||
const assigned = approved.tasks.filter(task => task.ownerId === member.id).map(task => ({ ...task, id: taskMapping[task.id], dependencies: task.dependencies.map(dep => taskMapping[dep]) }))
|
||||
await withTaskListLifecycleLock(getCanonicalTeamTaskListId(plan.teamName), async () => {
|
||||
const latest = await readTeamPlan(plan.teamName)
|
||||
@@ -220,40 +643,16 @@ export async function launchTeamPlanRuntime(plan: TeamPlanRecord): Promise<{ mem
|
||||
launch.released = true
|
||||
const sent = await conversationService.sendMessage(id, `${member.agentSnapshot?.initialPrompt ? member.agentSnapshot.initialPrompt + "\n" : ""}${member.prompt}\n\nApproved shared tasks (use TaskGet/TaskUpdate; respect dependencies):\n${JSON.stringify(assigned)}`, undefined, { canSend: () => !launch.stopped })
|
||||
if (!sent) throw new Error(`Failed to release ${member.name}`)
|
||||
const worker = launch.workers.get(member.id)
|
||||
if (worker) worker.released = true
|
||||
})
|
||||
}, id => conversationService.stopSessionAndWait(id))
|
||||
|
||||
// One consumer owns each worker inbox. SDK serializes user turns; a worker
|
||||
// One supervisor owns each worker inbox. SDK serializes user turns; a worker
|
||||
// remains connected while idle and wakes when another teammate writes.
|
||||
let polling = false
|
||||
launch.timer = setInterval(() => {
|
||||
if (polling || launch.stopped) return
|
||||
polling = true
|
||||
void (async () => {
|
||||
const currentTeam = readTeamFile(plan.teamName)
|
||||
if (!conversationService.hasSession(plan.sessionId) || !currentTeam || currentTeam.createdAt !== createdAt) { await stopTeamPlanRuntime(plan.planId); return }
|
||||
for (const member of members) {
|
||||
const id = memberIds[member.id]!
|
||||
if (!conversationService.hasSession(id)) continue
|
||||
const messages = await readUnreadMessages(member.name, plan.teamName)
|
||||
if (!messages.length) continue
|
||||
await withTaskListLifecycleLock(getCanonicalTeamTaskListId(plan.teamName), async () => {
|
||||
const team = readTeamFile(plan.teamName)
|
||||
if (!team || team.createdAt !== createdAt) throw new Error('Team generation has ended')
|
||||
const entry = team.members.find(entry => entry.sessionId === id)
|
||||
if (entry) { entry.isActive = true; await writeTeamFileAsync(plan.teamName, team) }
|
||||
})
|
||||
const accepted = await conversationService.sendMessage(id, messages.map(message => `<teammate-message teammate_id="${message.from}">\n${message.text}\n</teammate-message>`).join('\n'))
|
||||
if (accepted) {
|
||||
const ids = new Set(messages.map(message => message.id).filter(Boolean))
|
||||
const legacy = new Set(messages.filter(message => !message.id).map(message => JSON.stringify([message.from, message.timestamp, message.text])))
|
||||
await markMessagesAsReadByPredicate(member.name, message => message.id ? ids.has(message.id) : legacy.has(JSON.stringify([message.from, message.timestamp, message.text])), plan.teamName)
|
||||
}
|
||||
}
|
||||
})().catch(error => { console.error('[TeamPlanRuntime] mailbox delivery failed', error) }).finally(() => { polling = false })
|
||||
}, 250)
|
||||
launch.timer.unref()
|
||||
await conversationService.sendMessage(plan.sessionId, `Approved team ${plan.teamName} is running. Members and their shared tasks are ready. Continue coordinating the approved team; do not spawn these members again.`)
|
||||
launch.running = true
|
||||
startSupervisor(launch)
|
||||
await conversationService.sendMessage(plan.sessionId, `Approved team ${plan.teamName} is running. Members and their shared tasks are ready. Continue coordinating the approved team; do not spawn these members again. Members report back automatically when they finish or fail; you do not need to poll or wait in a loop.`)
|
||||
return { memberIds }
|
||||
} catch (error) {
|
||||
await stopTeamPlanRuntime(plan.planId)
|
||||
@@ -262,6 +661,66 @@ export async function launchTeamPlanRuntime(plan: TeamPlanRecord): Promise<{ mem
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Rebuild the supervisor for an approved team whose server-side owner was lost
|
||||
* (server or app restart). Members start dormant and restart from their own
|
||||
* transcripts as soon as the lead or the user messages them.
|
||||
*/
|
||||
export async function rehydrateTeamPlanRuntime(plan: TeamPlanRecord): Promise<boolean> {
|
||||
if (plan.state !== 'running' || !plan.approvedSnapshot || launches.has(plan.planId)) return launches.has(plan.planId)
|
||||
const team = readTeamFile(plan.teamName)
|
||||
if (!team || team.leadSessionId !== plan.sessionId || incarnationOf(team) !== plan.incarnationId) return false
|
||||
const memberIds = plan.launch?.memberIds ?? {}
|
||||
const workers = new Map<string, WorkerRuntime>()
|
||||
for (const member of plan.approvedSnapshot.members) {
|
||||
const sessionId = memberIds[member.id] ?? team.members.find(entry => entry.planMemberId === member.id && entry.backendType === 'process')?.sessionId
|
||||
if (!sessionId) continue
|
||||
const entry = team.members.find(candidate => candidate.sessionId === sessionId)
|
||||
if (!entry) continue
|
||||
workers.set(member.id, { member, sessionId, released: true, autoContinueAttempts: 0, restartHistory: [], restartsExhausted: false, wokenForTaskIds: new Set() })
|
||||
}
|
||||
if (workers.size === 0) return false
|
||||
ensureRuntimeHooks()
|
||||
const launch: TeamLaunch = {
|
||||
parentId: plan.sessionId, plan, createdAt: team.createdAt, released: true, running: true,
|
||||
children: [...workers.values()].map(worker => worker.sessionId), workers, supervising: false, stopped: false,
|
||||
permissionMode: conversationService.getSessionPermissionMode(plan.sessionId),
|
||||
}
|
||||
launches.set(plan.planId, launch)
|
||||
await mutateTeamFileAsync(plan.teamName, current => {
|
||||
if (current.createdAt !== team.createdAt) return
|
||||
const sessions = new Set(launch.children)
|
||||
return { ...current, members: current.members.map(entry => entry.sessionId && sessions.has(entry.sessionId) && !conversationService.hasSession(entry.sessionId) ? { ...entry, isActive: false, terminated: true } : entry) }
|
||||
}).catch(() => undefined)
|
||||
startSupervisor(launch)
|
||||
return true
|
||||
}
|
||||
|
||||
/** Restore every approved launch (current and archived parent plans) of a lead's team. */
|
||||
export async function rehydrateTeamPlanRuntimesForSession(sessionId: string): Promise<boolean> {
|
||||
const current = await findTeamPlanForSession(sessionId)
|
||||
if (!current) return false
|
||||
const plans: TeamPlanRecord[] = [current]
|
||||
try {
|
||||
for (const file of await readdir(getTeamDir(current.teamName))) {
|
||||
if (!/^plan-[a-f0-9-]{36}\.json$/i.test(file)) continue
|
||||
try {
|
||||
const archived = teamPlanRecordSchema.parse(JSON.parse(await readFile(join(getTeamDir(current.teamName), file), 'utf8')))
|
||||
if (archived.incarnationId === current.incarnationId && archived.sessionId === sessionId) plans.push(archived)
|
||||
} catch {
|
||||
// An unreadable archive cannot be restored; the live plan still can.
|
||||
}
|
||||
}
|
||||
} catch {
|
||||
// No team directory: nothing to restore.
|
||||
}
|
||||
let restored = false
|
||||
for (const plan of plans) {
|
||||
if (plan.state === 'running' && await rehydrateTeamPlanRuntime(plan)) restored = true
|
||||
}
|
||||
return restored
|
||||
}
|
||||
|
||||
export async function notifyTeamPlanLeader(plan: TeamPlanRecord, kind: 'approved' | 'returned' | 'cancelled'): Promise<void> {
|
||||
// Approval wakes the leader only after launch's readiness barrier and snapshot.
|
||||
if (kind === 'approved' || !conversationService.hasSession(plan.sessionId)) return
|
||||
@@ -274,16 +733,47 @@ export class TeamPlanExecutionInterruptedError extends Error {
|
||||
readonly executionStarted = true
|
||||
}
|
||||
|
||||
export function isTeamPlanRuntimeActive(planId: string): boolean {
|
||||
const launch = launches.get(planId)
|
||||
// A graceful member shutdown does not turn a successfully launched plan into
|
||||
// an interrupted plan. This probe checks host ownership, not current activity.
|
||||
return !!launch && !launch.stopped && conversationService.hasSession(launch.parentId)
|
||||
/** Tests simulate a server restart: owners vanish, processes and files stay. */
|
||||
export function forgetTeamPlanRuntimesForTests(): void {
|
||||
for (const launch of launches.values()) {
|
||||
if (launch.timer) clearInterval(launch.timer)
|
||||
for (const worker of launch.workers.values()) cancelAutoContinue(worker)
|
||||
}
|
||||
launches.clear()
|
||||
}
|
||||
|
||||
export async function stopTeamPlanRuntimesForParent(parentSessionId: string): Promise<void> {
|
||||
const stopOwned = () => Promise.all([...launches.entries()].filter(([, launch]) => launch.parentId === parentSessionId).map(([planId]) => stopTeamPlanRuntime(planId)))
|
||||
const hadReleasedWork = [...launches.values()].some(launch => launch.parentId === parentSessionId && launch.released)
|
||||
export function isTeamPlanRuntimeActive(planId: string): boolean {
|
||||
const launch = launches.get(planId)
|
||||
// A paused or lead-less team is still owned here: its members restart on
|
||||
// demand. Only an explicit stop or a deleted team ends ownership.
|
||||
return !!launch && !launch.stopped
|
||||
}
|
||||
|
||||
/** Whether a lead's approved team has a member mid-turn (keeps an unattended lead alive). */
|
||||
export function hasActiveTeamWorkForParent(parentSessionId: string): boolean {
|
||||
for (const launch of launches.values()) {
|
||||
if (launch.parentId !== parentSessionId || launch.stopped || launch.pausedAt) continue
|
||||
if (!launch.running) return true
|
||||
const team = readTeamFile(launch.plan.teamName)
|
||||
for (const worker of launch.workers.values()) {
|
||||
if (worker.autoContinueTimer) return true
|
||||
if (!conversationService.hasSession(worker.sessionId)) continue
|
||||
if (team?.members.find(entry => entry.sessionId === worker.sessionId)?.isActive) return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
/**
|
||||
* Stop the teams a lead owns. With `pauseReleased`, a team that is already
|
||||
* running is paused (the user's Stop) rather than ended; a launch that has not
|
||||
* finished starting is still revoked.
|
||||
*/
|
||||
export async function stopTeamPlanRuntimesForParent(parentSessionId: string, options: { pauseReleased?: boolean } = {}): Promise<void> {
|
||||
const owned = () => [...launches.entries()].filter(([, launch]) => launch.parentId === parentSessionId && !(options.pauseReleased && launch.running))
|
||||
const stopOwned = () => Promise.all(owned().map(([planId]) => stopTeamPlanRuntime(planId)))
|
||||
const hadReleasedWork = owned().some(([, launch]) => launch.released)
|
||||
const endedRunning = owned().filter(([, launch]) => launch.running).map(([, launch]) => launch.plan)
|
||||
await stopOwned()
|
||||
const plan = await findTeamPlanForSession(parentSessionId)
|
||||
if (plan?.state === 'launching') {
|
||||
@@ -296,5 +786,37 @@ export async function stopTeamPlanRuntimesForParent(parentSessionId: string): Pr
|
||||
console.error('[TeamPlanRuntime] stopped plan changed concurrently', error)
|
||||
})
|
||||
}
|
||||
// An ended (not paused) running team must not be rehydrated later.
|
||||
for (const ended of endedRunning) {
|
||||
const latest = await readTeamPlan(ended.teamName).catch(() => null)
|
||||
if (latest?.planId !== ended.planId || latest.state !== 'running') continue
|
||||
await mutateTeamPlan(latest.teamName, { ...latest, expectedRevision: latest.revision }, current => ({
|
||||
...current, state: 'interrupted', launch: { ...current.launch, status: 'failed', error: 'The team was stopped.' },
|
||||
})).catch(error => console.error('[TeamPlanRuntime] cannot record the stopped team', error))
|
||||
}
|
||||
await stopOwned()
|
||||
}
|
||||
|
||||
/**
|
||||
* End a lead's reviewed teams for good (/clear or deleting the session): stop
|
||||
* every launch and remove the team and task directories. The lead's CLI no
|
||||
* longer deletes them on exit because a desktop lead process restarts often.
|
||||
*/
|
||||
export async function endTeamsForParent(parentSessionId: string): Promise<void> {
|
||||
await stopTeamPlanRuntimesForParent(parentSessionId)
|
||||
let names: string[] = []
|
||||
try {
|
||||
names = await readdir(getTeamsDirectory())
|
||||
} catch {
|
||||
return
|
||||
}
|
||||
for (const name of names) {
|
||||
const team = readTeamFile(name)
|
||||
if (!team || team.leadSessionId !== parentSessionId || !team.reviewRequired) continue
|
||||
await cleanupTeamDirectories(name).catch(error => console.error(`[TeamPlanRuntime] cannot remove ended team ${name}`, error))
|
||||
}
|
||||
}
|
||||
|
||||
// Restarts of a lead (and of the server itself) must find the hooks in place
|
||||
// before any launch happens in this process.
|
||||
ensureRuntimeHooks()
|
||||
|
||||
@@ -1,9 +1,10 @@
|
||||
import { afterEach, beforeEach, expect, test } from 'bun:test'
|
||||
import { afterEach, beforeEach, expect, spyOn, test } from 'bun:test'
|
||||
import { mkdtemp, rm } from 'node:fs/promises'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { join } from 'node:path'
|
||||
import { TeamPlanService } from './teamPlanService.js'
|
||||
import { approveTeamPlan, ensureTeamDraft, readTeamPlan, replaceTeamPlan, stageMember, submitTeamPlan } from '../../utils/swarm/teamPlanStore.js'
|
||||
import * as teamPlanStore from '../../utils/swarm/teamPlanStore.js'
|
||||
import { writeTeamFileAsync } from '../../utils/swarm/teamHelpers.js'
|
||||
import type { TeamPlanRecord } from '../../shared/teamPlan.js'
|
||||
const saved = { home: process.env.HOME, config: process.env.CLAUDE_CONFIG_DIR }
|
||||
@@ -54,6 +55,29 @@ test('approval launches once and returns frozen selected runtime to executor', a
|
||||
release()
|
||||
expect((await settle('running')).launch?.memberIds).toEqual({ worker: 'child-1' })
|
||||
})
|
||||
test('a plan read while the approval is committing does not orphan the launch', async () => {
|
||||
let launches = 0
|
||||
const service = new TeamPlanService({ validate: async item => item, launch: async () => {
|
||||
launches++
|
||||
return { memberIds: { worker: 'child-1' } }
|
||||
}, stop: async () => {} })
|
||||
const plan = await ready()
|
||||
const commit = teamPlanStore.approveTeamPlan
|
||||
// The UI polls the plan while the approval runs: this read lands after the
|
||||
// plan already says `launching` but before the launch is registered.
|
||||
const approval = spyOn(teamPlanStore, 'approveTeamPlan').mockImplementation(async (...args: Parameters<typeof commit>) => {
|
||||
const result = await commit(...args)
|
||||
await service.getForSession('session')
|
||||
return result
|
||||
})
|
||||
try {
|
||||
expect((await service.approve('review', identity(plan))).state).toBe('launching')
|
||||
} finally {
|
||||
approval.mockRestore()
|
||||
}
|
||||
expect((await settle('running')).launch?.memberIds).toEqual({ worker: 'child-1' })
|
||||
expect(launches).toBe(1)
|
||||
})
|
||||
test('validation failure starts no process; launch failure requires explicit review retry', async () => {
|
||||
let calls = 0
|
||||
const plan = await ready()
|
||||
@@ -135,3 +159,27 @@ test('launch transitions preserve future persisted launch metadata', async () =>
|
||||
expect(interrupted?.launch?.futureLaunch).toEqual({ retained: true })
|
||||
expect(interrupted?.state).toBe('interrupted')
|
||||
})
|
||||
|
||||
test('a running team whose owner was lost is re-owned instead of being marked interrupted', async () => {
|
||||
const plan = await ready()
|
||||
const service = new TeamPlanService({ validate: async item => item, launch: async () => ({ memberIds: { worker: 'fixture-child' } }), stop: async () => {} })
|
||||
await service.approve('review', identity(plan))
|
||||
await settle('running')
|
||||
let owned = false
|
||||
const rehydrated: string[] = []
|
||||
const restarted = new TeamPlanService({
|
||||
validate: async item => item, launch: async () => ({ memberIds: {} }), stop: async () => {},
|
||||
isRunning: () => owned,
|
||||
rehydrate: async item => { rehydrated.push(item.planId); owned = true; return true },
|
||||
})
|
||||
expect((await restarted.getForSession('session'))?.state).toBe('running')
|
||||
expect(rehydrated).toHaveLength(1)
|
||||
expect((await readTeamPlan('review'))?.state).toBe('running')
|
||||
|
||||
owned = false
|
||||
const unrecoverable = new TeamPlanService({
|
||||
validate: async item => item, launch: async () => ({ memberIds: {} }), stop: async () => {},
|
||||
isRunning: () => false, rehydrate: async () => false,
|
||||
})
|
||||
expect((await unrecoverable.getForSession('session'))?.state).toBe('interrupted')
|
||||
})
|
||||
|
||||
@@ -16,10 +16,13 @@ export type TeamPlanRuntimeAdapter = {
|
||||
launch(plan: TeamPlanRecord): Promise<{ memberIds: Record<string, string> }>
|
||||
stop(planId: string): Promise<void>
|
||||
isRunning?(planId: string): boolean | Promise<boolean>
|
||||
/** Re-own an approved team whose owner was lost; its members restart on demand. */
|
||||
rehydrate?(plan: TeamPlanRecord): Promise<boolean>
|
||||
notifyLeader?(plan: TeamPlanRecord, kind: 'approved' | 'returned' | 'cancelled'): Promise<void>
|
||||
}
|
||||
const defaultRuntime: TeamPlanRuntimeAdapter = {
|
||||
isRunning: async id => (await import('./teamPlanRuntime.js')).isTeamPlanRuntimeActive(id),
|
||||
rehydrate: async plan => (await import('./teamPlanRuntime.js')).rehydrateTeamPlanRuntimesForSession(plan.sessionId),
|
||||
validate: async plan => (await import('./teamPlanRuntime.js')).validateTeamPlanRuntime(plan),
|
||||
launch: async plan => (await import('./teamPlanRuntime.js')).launchTeamPlanRuntime(plan),
|
||||
notifyLeader: async (plan, kind) => { await (await import('./teamPlanRuntime.js')).notifyTeamPlanLeader(plan, kind) },
|
||||
@@ -29,14 +32,19 @@ const defaultRuntime: TeamPlanRuntimeAdapter = {
|
||||
/** Only trusted HTTP/UI actions call approve; model tools import the draft store only. */
|
||||
export class TeamPlanService {
|
||||
private launches = new Map<string, Promise<void>>()
|
||||
/** Approvals whose plan may already read `launching` before its launch is registered. */
|
||||
private approving = new Set<string>()
|
||||
constructor(private runtime: TeamPlanRuntimeAdapter = defaultRuntime) {}
|
||||
async getForSession(sessionId: string): Promise<TeamPlanRecord | null> {
|
||||
const plan = await findTeamPlanForSession(sessionId)
|
||||
// A server restart lost ownership of an unfinished launch. Never silently replay it.
|
||||
if (plan?.state === 'launching' && !this.launches.has(plan.planId)) {
|
||||
if (plan?.state === 'launching' && !this.launches.has(plan.planId) && !this.approving.has(plan.planId)) {
|
||||
return mutateTeamPlan(plan.teamName, { ...plan, expectedRevision: plan.revision }, current => ({ ...current, state: 'interrupted', launch: { ...current.launch, status: 'failed', error: 'Launch ownership was lost. Work may have started and will not be replayed.' } }))
|
||||
}
|
||||
if (plan?.state === 'running' && this.runtime.isRunning && !await this.runtime.isRunning(plan.planId)) {
|
||||
// A server or app restart lost the in-memory owner, not the team: members
|
||||
// keep their transcripts and resume when messaged.
|
||||
if (this.runtime.rehydrate && await this.runtime.rehydrate(plan).catch(() => false) && await this.runtime.isRunning(plan.planId)) return plan
|
||||
return mutateTeamPlan(plan.teamName, { ...plan, expectedRevision: plan.revision }, current => ({ ...current, state: 'interrupted', launch: { ...current.launch, status: 'failed', error: 'The worker runtime was interrupted. Started work will not be replayed.' } }))
|
||||
}
|
||||
return plan
|
||||
@@ -88,9 +96,16 @@ export class TeamPlanService {
|
||||
try { validated = await this.runtime.validate(current) }
|
||||
catch (error) { throw new TeamPlanError(error instanceof Error ? error.message : 'Team configuration is unavailable', 400) }
|
||||
}
|
||||
const result = await approveTeamPlan(teamName, action, action.requestId, validated)
|
||||
if (result.committed) this.start(result.plan)
|
||||
return result.plan
|
||||
// The approval commits `launching` before this call can register the
|
||||
// launch; a plan read in between must not take it for an orphaned launch.
|
||||
this.approving.add(current.planId)
|
||||
try {
|
||||
const result = await approveTeamPlan(teamName, action, action.requestId, validated)
|
||||
if (result.committed) this.start(result.plan)
|
||||
return result.plan
|
||||
} finally {
|
||||
this.approving.delete(current.planId)
|
||||
}
|
||||
}
|
||||
async action(teamName: string, kind: 'return' | 'cancel' | 'retry', action: TeamPlanAction): Promise<TeamPlanRecord> {
|
||||
if (kind === 'retry') {
|
||||
|
||||
@@ -0,0 +1,124 @@
|
||||
import { afterEach, beforeEach, describe, expect, test } from 'bun:test'
|
||||
import * as fs from 'node:fs/promises'
|
||||
import * as os from 'node:os'
|
||||
import * as path from 'node:path'
|
||||
import {
|
||||
claimMailboxMessages,
|
||||
getInboxHistoryPath,
|
||||
getInboxPath,
|
||||
readUnreadMessages,
|
||||
writeToMailbox,
|
||||
} from '../../utils/teammateMailbox.js'
|
||||
import { TeamService } from './teamService.js'
|
||||
|
||||
const TEAM = 'history-team'
|
||||
|
||||
let configDir: string
|
||||
let originalConfigDir: string | undefined
|
||||
let service: TeamService
|
||||
|
||||
beforeEach(async () => {
|
||||
originalConfigDir = process.env.CLAUDE_CONFIG_DIR
|
||||
configDir = await fs.mkdtemp(path.join(os.tmpdir(), 'cc-haha-team-feed-'))
|
||||
process.env.CLAUDE_CONFIG_DIR = configDir
|
||||
const teamDir = path.join(configDir, 'teams', TEAM)
|
||||
await fs.mkdir(teamDir, { recursive: true })
|
||||
await fs.writeFile(
|
||||
path.join(teamDir, 'config.json'),
|
||||
JSON.stringify({
|
||||
name: TEAM,
|
||||
createdAt: 1700000000000,
|
||||
leadAgentId: `team-lead@${TEAM}`,
|
||||
members: ['team-lead', 'worker'].map(name => ({
|
||||
agentId: `${name}@${TEAM}`,
|
||||
name,
|
||||
joinedAt: 1700000000000,
|
||||
tmuxPaneId: '',
|
||||
cwd: configDir,
|
||||
isActive: true,
|
||||
})),
|
||||
}),
|
||||
)
|
||||
service = new TeamService()
|
||||
})
|
||||
|
||||
afterEach(async () => {
|
||||
if (originalConfigDir === undefined) delete process.env.CLAUDE_CONFIG_DIR
|
||||
else process.env.CLAUDE_CONFIG_DIR = originalConfigDir
|
||||
await fs.rm(configDir, { recursive: true, force: true })
|
||||
})
|
||||
|
||||
describe('Agent Teams communication feed', () => {
|
||||
test('keeps delivered messages after the CLI prunes them from the live inbox', async () => {
|
||||
expect(
|
||||
await writeToMailbox(
|
||||
'team-lead',
|
||||
{
|
||||
from: 'worker',
|
||||
text: 'Analysis complete',
|
||||
summary: 'done',
|
||||
timestamp: '2026-10-03T00:00:01.000Z',
|
||||
},
|
||||
TEAM,
|
||||
),
|
||||
).toBe(true)
|
||||
expect(
|
||||
await writeToMailbox(
|
||||
'worker',
|
||||
{
|
||||
from: 'team-lead',
|
||||
text: 'Next: add regression tests',
|
||||
timestamp: '2026-10-03T00:00:02.000Z',
|
||||
},
|
||||
TEAM,
|
||||
),
|
||||
).toBe(true)
|
||||
const before = await service.getWorkbench(TEAM)
|
||||
|
||||
await claimMailboxMessages(
|
||||
'team-lead',
|
||||
TEAM,
|
||||
await readUnreadMessages('team-lead', TEAM),
|
||||
)
|
||||
expect(await fs.readFile(getInboxPath('team-lead', TEAM), 'utf8')).toBe('[]')
|
||||
const after = await service.getWorkbench(TEAM)
|
||||
|
||||
expect(after.messages).toEqual(before.messages)
|
||||
expect(
|
||||
after.messages.map(message => [
|
||||
message.from,
|
||||
message.text,
|
||||
message.recipients,
|
||||
message.summary,
|
||||
]),
|
||||
).toEqual([
|
||||
['worker', 'Analysis complete', ['team-lead'], 'done'],
|
||||
['team-lead', 'Next: add regression tests', ['worker'], undefined],
|
||||
])
|
||||
})
|
||||
|
||||
test('reads a recipient known only from its history file', async () => {
|
||||
const historyPath = getInboxHistoryPath(getInboxPath('worker', TEAM))
|
||||
await fs.mkdir(path.dirname(historyPath), { recursive: true })
|
||||
await fs.writeFile(
|
||||
historyPath,
|
||||
`${JSON.stringify({
|
||||
id: 'mailbox-archived',
|
||||
from: 'team-lead',
|
||||
text: 'Archived instruction',
|
||||
timestamp: '2026-10-03T00:00:03.000Z',
|
||||
read: true,
|
||||
})}\n`,
|
||||
)
|
||||
|
||||
const snapshot = await service.getWorkbench(TEAM)
|
||||
|
||||
expect(snapshot.messages.map(message => [message.text, message.recipients])).toEqual([
|
||||
['Archived instruction', ['worker']],
|
||||
])
|
||||
expect(snapshot.team.members.map(member => member.name).sort()).toEqual([
|
||||
'team-lead',
|
||||
'worker',
|
||||
])
|
||||
})
|
||||
})
|
||||
@@ -15,7 +15,13 @@ import * as os from 'os'
|
||||
import * as crypto from 'node:crypto'
|
||||
import { ApiError } from '../middleware/errorHandler.js'
|
||||
import { readTeamTranscriptProjection } from './teamTranscriptProjection.js'
|
||||
import { writeToMailbox } from '../../utils/teammateMailbox.js'
|
||||
import { readMemberFailure } from '../../utils/swarm/turnFailure.js'
|
||||
import { unescapeXmlAttr } from '../../utils/xml.js'
|
||||
import {
|
||||
MAILBOX_HISTORY_SUFFIX,
|
||||
readMailboxHistoryAtPath,
|
||||
writeToMailbox,
|
||||
} from '../../utils/teammateMailbox.js'
|
||||
import {
|
||||
sessionService,
|
||||
type MessageEntry as SessionMessageEntry,
|
||||
@@ -44,7 +50,12 @@ import type { TaskInfo } from './taskService.js'
|
||||
* task started and can then finish its turn, and an umbrella task stays open
|
||||
* across every turn underneath it.
|
||||
*/
|
||||
export type TeamMemberActivity = 'active' | 'idle' | 'exited' | 'unknown'
|
||||
/**
|
||||
* `stopped`: a process-backed member whose process is not running (the user's
|
||||
* Stop, a crash, a closed lead). It keeps its conversation and a message
|
||||
* restarts it, unlike `exited`, which means it left the team.
|
||||
*/
|
||||
export type TeamMemberActivity = 'active' | 'idle' | 'exited' | 'stopped' | 'unknown'
|
||||
|
||||
export type TeamMember = {
|
||||
agentId: string
|
||||
@@ -62,6 +73,10 @@ export type TeamMember = {
|
||||
* readers fall back to `status` rather than assuming a member went quiet.
|
||||
*/
|
||||
activity?: TeamMemberActivity
|
||||
/** Why the last turn failed, when no automatic retry is pending. */
|
||||
lastError?: string
|
||||
/** An automatic continuation is scheduled after a transient provider failure. */
|
||||
autoRetry?: { attempt: number; max: number; nextAt: number }
|
||||
joinedAt: number
|
||||
cwd: string
|
||||
sessionId?: string
|
||||
@@ -1146,13 +1161,13 @@ export function projectTeamWorkbenchesFromTranscript(
|
||||
if (!body) continue
|
||||
team.messages.push({
|
||||
id: `${message.id}:teammate:${team.messages.length}`,
|
||||
from: match[1]!,
|
||||
from: unescapeXmlAttr(match[1]!),
|
||||
to: 'team-lead',
|
||||
recipients: ['team-lead'],
|
||||
kind: 'direct',
|
||||
text: body,
|
||||
timestamp,
|
||||
...(match[2] ? { color: match[2] } : {}),
|
||||
...(match[2] ? { color: unescapeXmlAttr(match[2]) } : {}),
|
||||
})
|
||||
team.updatedAt = timestamp
|
||||
}
|
||||
@@ -1421,6 +1436,9 @@ type TeamFileRaw = {
|
||||
sessionId?: string
|
||||
backendType?: string
|
||||
isActive?: boolean
|
||||
terminated?: boolean
|
||||
lastError?: string
|
||||
autoRetry?: { attempt: number; max: number; nextAt: number }
|
||||
mode?: string
|
||||
}>
|
||||
}
|
||||
@@ -1525,22 +1543,26 @@ export class TeamService {
|
||||
? await this.discoverSubagentLastWrites(config.leadSessionId)
|
||||
: new Map<string, number>()
|
||||
|
||||
const members: TeamMember[] = config.members.map((m) => ({
|
||||
agentId: m.agentId,
|
||||
name: m.name,
|
||||
agentType: m.agentType,
|
||||
...(m.model ? { model: m.model } : {}),
|
||||
...(m.providerId !== undefined ? { providerId: m.providerId } : {}),
|
||||
...(m.providerName !== undefined ? { providerName: m.providerName } : {}),
|
||||
...(m.effortLevel !== undefined ? { effortLevel: m.effortLevel } : {}),
|
||||
color: m.color,
|
||||
backendType: m.backendType,
|
||||
status: this.deriveStatus(m.isActive),
|
||||
activity: this.deriveActivity(m.isActive, lastWrites.get(m.name), now),
|
||||
joinedAt: m.joinedAt,
|
||||
cwd: m.cwd,
|
||||
sessionId: m.sessionId,
|
||||
}))
|
||||
const members: TeamMember[] = config.members.map((m) => {
|
||||
const failure = readMemberFailure(m as unknown as Record<string, unknown>)
|
||||
return {
|
||||
agentId: m.agentId,
|
||||
name: m.name,
|
||||
agentType: m.agentType,
|
||||
...(m.model ? { model: m.model } : {}),
|
||||
...(m.providerId !== undefined ? { providerId: m.providerId } : {}),
|
||||
...(m.providerName !== undefined ? { providerName: m.providerName } : {}),
|
||||
...(m.effortLevel !== undefined ? { effortLevel: m.effortLevel } : {}),
|
||||
color: m.color,
|
||||
backendType: m.backendType,
|
||||
status: failure.lastError && !failure.autoRetry ? 'failed' : this.deriveStatus(m.isActive),
|
||||
activity: m.terminated === true ? 'stopped' : this.deriveActivity(m.isActive, lastWrites.get(m.name), now),
|
||||
...failure,
|
||||
joinedAt: m.joinedAt,
|
||||
cwd: m.cwd,
|
||||
sessionId: m.sessionId,
|
||||
}
|
||||
})
|
||||
|
||||
// Discover members from inboxes/ that aren't in config.json (race condition fix)
|
||||
const inboxNames = await this.discoverInboxMembers(name)
|
||||
@@ -1794,7 +1816,7 @@ export class TeamService {
|
||||
)
|
||||
}
|
||||
|
||||
await writeToMailbox(
|
||||
const delivered = await writeToMailbox(
|
||||
recipientName,
|
||||
{
|
||||
from: 'user',
|
||||
@@ -1803,6 +1825,14 @@ export class TeamService {
|
||||
},
|
||||
teamName,
|
||||
)
|
||||
if (!delivered) {
|
||||
// The composer keeps the draft only if the request fails.
|
||||
throw new ApiError(
|
||||
503,
|
||||
`Could not deliver the message to ${recipientName}. Try again.`,
|
||||
'MAILBOX_UNAVAILABLE',
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
// ── Get Agent Teams workbench snapshot ──────────────────────────────────────
|
||||
@@ -2397,9 +2427,17 @@ export class TeamService {
|
||||
teamName: string,
|
||||
): Promise<TeamWorkbenchMessage[]> {
|
||||
const inboxDir = path.join(this.getTeamsDir(), teamName, 'inboxes')
|
||||
let files: string[]
|
||||
// Read entries leave {recipient}.json for {recipient}.history.jsonl, so a
|
||||
// recipient is known from either file.
|
||||
const recipients = new Set<string>()
|
||||
try {
|
||||
files = (await fs.readdir(inboxDir)).filter((file) => file.endsWith('.json'))
|
||||
for (const file of await fs.readdir(inboxDir)) {
|
||||
if (file.endsWith(MAILBOX_HISTORY_SUFFIX)) {
|
||||
recipients.add(file.slice(0, -MAILBOX_HISTORY_SUFFIX.length))
|
||||
} else if (file.endsWith('.json')) {
|
||||
recipients.add(file.slice(0, -'.json'.length))
|
||||
}
|
||||
}
|
||||
} catch {
|
||||
return []
|
||||
}
|
||||
@@ -2423,13 +2461,15 @@ export class TeamService {
|
||||
|
||||
const grouped = new Map<string, MessageAccumulator>()
|
||||
const occurrenceByRecipientAndContent = new Map<string, number>()
|
||||
for (const file of files.sort()) {
|
||||
const recipient = file.replace(/\.json$/, '')
|
||||
for (const recipient of [...recipients].sort()) {
|
||||
try {
|
||||
const parsed = JSON.parse(await fs.readFile(path.join(inboxDir, file), 'utf8'))
|
||||
if (!Array.isArray(parsed)) continue
|
||||
// Archived history plus the live inbox: delivered messages stay in
|
||||
// the feed after the CLI prunes them from the live file.
|
||||
const entries: InboxMessage[] = await readMailboxHistoryAtPath(
|
||||
path.join(inboxDir, `${recipient}.json`),
|
||||
)
|
||||
|
||||
for (const raw of parsed as InboxMessage[]) {
|
||||
for (const raw of entries) {
|
||||
if (
|
||||
typeof raw?.from !== 'string' ||
|
||||
typeof raw.text !== 'string' ||
|
||||
@@ -2469,8 +2509,8 @@ export class TeamService {
|
||||
}
|
||||
}
|
||||
} catch {
|
||||
// A mailbox can be between its truncate and rename while the CLI is
|
||||
// writing it. The watcher will invalidate the snapshot again.
|
||||
// The CLI may be replacing a mailbox file; the watcher will
|
||||
// invalidate the snapshot again.
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -11,6 +11,7 @@
|
||||
*/
|
||||
|
||||
import { readTeamPlan } from '../../utils/swarm/teamPlanStore.js'
|
||||
import { readMemberFailure } from '../../utils/swarm/turnFailure.js'
|
||||
import * as fs from 'fs'
|
||||
import * as path from 'path'
|
||||
import * as os from 'os'
|
||||
@@ -373,17 +374,27 @@ export class TeamWatcher {
|
||||
if (!Array.isArray(members)) return []
|
||||
|
||||
return members.map((m: Record<string, unknown>) => {
|
||||
const status = this.deriveStatus(m.isActive as boolean | undefined)
|
||||
const failure = readMemberFailure(m)
|
||||
const status = failure.lastError && !failure.autoRetry
|
||||
? 'error'
|
||||
: this.deriveStatus(m.isActive as boolean | undefined)
|
||||
return {
|
||||
agentId: (m.agentId as string) || '',
|
||||
role: (m.name as string) || (m.agentType as string) || 'member',
|
||||
status,
|
||||
// Only the runner's own turn markers are cheap enough to read on every
|
||||
// poll. Without one, say nothing so the last full team read stands.
|
||||
...(typeof m.isActive === 'boolean'
|
||||
? { activity: (m.isActive ? 'active' : 'idle') as const }
|
||||
: {}),
|
||||
// poll. Without one, say nothing so the last full team read stands. A
|
||||
// process member whose process is gone is stopped, not idle: messaging
|
||||
// it restarts it from its saved conversation.
|
||||
...(m.terminated === true
|
||||
? { activity: 'stopped' as const }
|
||||
: typeof m.isActive === 'boolean'
|
||||
? { activity: (m.isActive ? 'active' : 'idle') as const }
|
||||
: {}),
|
||||
currentTask: (m.currentTask as string) || undefined,
|
||||
// Always present so a recovered member clears what the last read set.
|
||||
lastError: failure.lastError ?? null,
|
||||
autoRetry: failure.autoRetry ?? null,
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
@@ -201,8 +201,15 @@ export type TeamMemberStatus = {
|
||||
* Omitted when the watcher cannot tell, so a receiver keeps whatever the last
|
||||
* full team read established rather than being told the member went quiet.
|
||||
*/
|
||||
activity?: 'active' | 'idle' | 'exited' | 'unknown'
|
||||
activity?: 'active' | 'idle' | 'exited' | 'stopped' | 'unknown'
|
||||
currentTask?: string
|
||||
/**
|
||||
* Why the member's last turn failed; `null` says it has recovered. Only the
|
||||
* process-backed (desktop) runtime records it.
|
||||
*/
|
||||
lastError?: string | null
|
||||
/** A pending automatic continuation after a transient provider failure; `null` once settled. */
|
||||
autoRetry?: { attempt: number; max: number; nextAt: number } | null
|
||||
}
|
||||
|
||||
export type ComputerUseGrantFlags = {
|
||||
|
||||
@@ -31,6 +31,7 @@ import {
|
||||
ConversationStartupError,
|
||||
conversationService,
|
||||
} from '../services/conversationService.js'
|
||||
import { deliverTeamPauseNotice, endTeamsForParent, hasActiveTeamWorkForParent } from '../services/teamPlanRuntime.js'
|
||||
import { computerUseApprovalService } from '../services/computerUseApprovalService.js'
|
||||
import {
|
||||
sessionService,
|
||||
@@ -429,7 +430,10 @@ function hasActiveCliRun(sessionId: string): boolean {
|
||||
function hasActiveSessionWork(sessionId: string): boolean {
|
||||
return hasPendingOrActiveUserTurn(sessionId) ||
|
||||
hasActiveCliRun(sessionId) ||
|
||||
hasActiveBackgroundTasks(sessionId)
|
||||
hasActiveBackgroundTasks(sessionId) ||
|
||||
// An idle lead still owns a working team; reaping it after the renderer
|
||||
// disconnects (sleep, reload) would strand every member mid-task.
|
||||
hasActiveTeamWorkForParent(sessionId)
|
||||
}
|
||||
|
||||
export function getSessionChatActivityState(sessionId: string): SessionChatActivityState {
|
||||
@@ -1076,6 +1080,12 @@ async function handleUserMessage(
|
||||
userMessageSent = true
|
||||
activeTurn.messageSent = true
|
||||
if (!collaboration) emitSessionTurnEvent({ type: 'input-committed', sessionId })
|
||||
// After the user's own words, so the lead weighs them first.
|
||||
if (!collaboration) {
|
||||
void deliverTeamPauseNotice(sessionId).catch(error =>
|
||||
console.error('[WS] cannot tell the lead about its stopped team', error),
|
||||
)
|
||||
}
|
||||
} finally {
|
||||
if (!activeTurn.messageSent) await admission?.release()
|
||||
}
|
||||
@@ -1399,6 +1409,8 @@ async function performDesktopClearCommand(
|
||||
if (activeTitleState) activeTitleState.activeTurn = undefined
|
||||
const pendingStartup = sessionStartupPromises.get(sessionId)
|
||||
conversationService.stopSession(sessionId)
|
||||
// Clearing the lead's context ends its reviewed team for good.
|
||||
void endTeamsForParent(sessionId).catch(error => console.error(`[WS] Failed to end the cleared session's team: ${error}`))
|
||||
pendingInterruptedTurnResults.delete(sessionId)
|
||||
// Clearing replaces the transcript, so do not enqueue terminal bookends that
|
||||
// could finish after the replacement write and repopulate the cleared file.
|
||||
@@ -2030,7 +2042,9 @@ async function restartSessionWithPermissionMode(
|
||||
const workDir = conversationService.getSessionWorkDir(sessionId)
|
||||
markActiveAgentsStopping(sessionId)
|
||||
runtimeExitStoppedSessions.add(sessionId)
|
||||
conversationService.stopSession(sessionId)
|
||||
// Approved team members are independent processes; replacing the lead's
|
||||
// process must not end their work.
|
||||
conversationService.stopSession(sessionId, { keepTeamWorkers: true })
|
||||
await emitAuthoritativeStoppedForActiveAgents(sessionId)
|
||||
await emitStoppedForNonAgentTasksAfterRuntimeExit(sessionId)
|
||||
|
||||
@@ -2151,7 +2165,9 @@ async function restartSessionWithRuntimeConfig(
|
||||
const workDir = await resolveRuntimeRestartWorkDir(sessionId)
|
||||
markActiveAgentsStopping(sessionId)
|
||||
runtimeExitStoppedSessions.add(sessionId)
|
||||
conversationService.stopSession(sessionId)
|
||||
// Approved team members are independent processes; replacing the lead's
|
||||
// process must not end their work.
|
||||
conversationService.stopSession(sessionId, { keepTeamWorkers: true })
|
||||
await emitAuthoritativeStoppedForActiveAgents(sessionId)
|
||||
await emitStoppedForNonAgentTasksAfterRuntimeExit(sessionId)
|
||||
|
||||
|
||||
+84
-87
@@ -227,7 +227,10 @@ import {
|
||||
stopSessionActivity,
|
||||
} from "../../utils/sessionActivity.js";
|
||||
import { isOpenAIPolicyError } from "../openaiAuth/policyError.js"
|
||||
import { shouldTriggerNonStreamingFallbackForEmptyStream } from "./streamFallback.js";
|
||||
import {
|
||||
shouldTriggerNonStreamingFallbackForEmptyStream,
|
||||
StreamEndedEarlyError,
|
||||
} from "./streamFallback.js";
|
||||
import { StreamAssistantCommitBuffer } from "./streamAssistantCommitBuffer.js";
|
||||
import {
|
||||
commitOutputLimitResponse,
|
||||
@@ -284,11 +287,11 @@ import { withStreamRetry } from "./streamRetry.js";
|
||||
import {
|
||||
CannotRetryError,
|
||||
FallbackTriggeredError,
|
||||
getStreamRetryKind,
|
||||
is529Error,
|
||||
isRetryableStreamError,
|
||||
isRetryableStreamTransportError,
|
||||
RetriableStreamError,
|
||||
type RetryContext,
|
||||
shouldRetryStreamAfterTransportDisconnect,
|
||||
withRetry,
|
||||
} from "./withRetry.js";
|
||||
|
||||
@@ -825,6 +828,7 @@ export async function queryModelWithoutStreaming({
|
||||
),
|
||||
options.model,
|
||||
messages,
|
||||
{ signal },
|
||||
);
|
||||
})) {
|
||||
if (message.type === "assistant") {
|
||||
@@ -875,6 +879,7 @@ export async function* queryModelWithStreaming({
|
||||
),
|
||||
options.model,
|
||||
messages,
|
||||
{ signal },
|
||||
);
|
||||
});
|
||||
}
|
||||
@@ -1989,7 +1994,6 @@ async function* queryModel(
|
||||
const completedBlockIndexes = new Set<number>();
|
||||
const completedToolUseIds = new Set<string>();
|
||||
const handledStopReasons = new Set<BetaStopReason>();
|
||||
let incompleteStream = false;
|
||||
let didFallBackToNonStreaming = false;
|
||||
let fallbackMessage: AssistantMessage | undefined;
|
||||
let maxOutputTokens = 0;
|
||||
@@ -2794,7 +2798,8 @@ async function* queryModel(
|
||||
// That's a legitimate empty response, not an incomplete stream. However,
|
||||
// stop_reason=tool_use with no completed tool block is incomplete: some
|
||||
// OpenAI-compatible streams send only finish_reason=tool_calls, and we
|
||||
// need the non-streaming fallback to recover the full tool call.
|
||||
// need the non-streaming fallback (or, without it, a re-sent stream) to
|
||||
// recover the full tool call.
|
||||
if (
|
||||
shouldTriggerNonStreamingFallbackForEmptyStream({
|
||||
hasMessageStart: partialMessage !== undefined,
|
||||
@@ -2804,10 +2809,10 @@ async function* queryModel(
|
||||
) {
|
||||
logForDebugging(
|
||||
!partialMessage
|
||||
? "Stream completed without receiving message_start event - triggering non-streaming fallback"
|
||||
? "Stream completed without receiving message_start event"
|
||||
: stopReason === "tool_use"
|
||||
? "Stream completed with tool_use stop but no completed tool block - triggering non-streaming fallback"
|
||||
: "Stream completed with message_start but no content blocks completed - triggering non-streaming fallback",
|
||||
? "Stream completed with tool_use stop but no completed tool block"
|
||||
: "Stream completed with message_start but no content blocks completed",
|
||||
{ level: "error" },
|
||||
);
|
||||
logEvent("tengu_stream_no_events", {
|
||||
@@ -2816,7 +2821,7 @@ async function* queryModel(
|
||||
request_id: (streamRequestId ??
|
||||
"unknown") as AnalyticsMetadata_I_VERIFIED_THIS_IS_NOT_CODE_OR_FILEPATHS,
|
||||
});
|
||||
throw new Error("Stream ended without receiving any events");
|
||||
throw new StreamEndedEarlyError("no_events");
|
||||
}
|
||||
|
||||
// A clean socket EOF is not a successful Anthropic response. Explicit
|
||||
@@ -2827,8 +2832,7 @@ async function* queryModel(
|
||||
(!streamWatchdogState.snapshot().messageStopReceived || stopReason === null ||
|
||||
contentBlocks.some((block, index) => block && !completedBlockIndexes.has(index)))
|
||||
) {
|
||||
incompleteStream = true;
|
||||
throw new Error("Provider stream ended before completing the response");
|
||||
throw new StreamEndedEarlyError("incomplete");
|
||||
}
|
||||
|
||||
// No tool boundary was crossed, so completed thinking/text blocks were
|
||||
@@ -2899,12 +2903,58 @@ async function* queryModel(
|
||||
// A safety rejection is terminal, including for non-streaming fallback.
|
||||
if (isOpenAIPolicyError(streamingError)) throw streamingError
|
||||
|
||||
// Never replay a response after server-side work began or a local tool
|
||||
// completed. Keep displayable blocks, but discard uncommitted local tools.
|
||||
// Completed partial text on a clean EOF is also surfaced with an error,
|
||||
// rather than silently accepted or replaced by a second response.
|
||||
if (incompleteStream || assistantCommitBuffer.hasCrossedSideEffectBoundary() ||
|
||||
streamWatchdogState.snapshot().serverToolUseStarted) {
|
||||
// When the flag is enabled, skip the non-streaming fallback and let the
|
||||
// error propagate to withRetry. The mid-stream fallback causes double tool
|
||||
// execution when streaming tool execution is active: the partial stream
|
||||
// starts a tool, then the non-streaming retry produces the same tool_use
|
||||
// and runs it again. See inc-4258.
|
||||
const disableFallback =
|
||||
isEnvTruthy(process.env.CLAUDE_CODE_DISABLE_NONSTREAMING_FALLBACK) ||
|
||||
getFeatureValue_CACHED_MAY_BE_STALE(
|
||||
"tengu_disable_streaming_to_non_streaming_fallback",
|
||||
false,
|
||||
);
|
||||
|
||||
// The attempt can be discarded and re-sent as a new stream only while
|
||||
// nothing from it reached the consumer and no server-side tool work
|
||||
// began. A completed local tool_use does not close that window: the
|
||||
// commit buffer holds it until the response completes, and query.ts only
|
||||
// executes tools from yielded assistant messages, so it has not run.
|
||||
const canReplay =
|
||||
!assistantCommitBuffer.hasCommitted() &&
|
||||
!streamWatchdogState.snapshot().serverToolUseStarted;
|
||||
const streamEndedIncomplete =
|
||||
streamingError instanceof StreamEndedEarlyError &&
|
||||
streamingError.reason === "incomplete";
|
||||
const retryKind = getStreamRetryKind({
|
||||
error: streamingError,
|
||||
canReplay,
|
||||
streamIdleAborted,
|
||||
signalAborted: signal.aborted,
|
||||
// The fallback below never replays an attempt that completed a tool
|
||||
// block, so such a truncation can only recover as a new stream.
|
||||
nonStreamingFallbackAvailable:
|
||||
!disableFallback &&
|
||||
!assistantCommitBuffer.hasCrossedSideEffectBoundary(),
|
||||
transientRetryAllowed:
|
||||
newMessages.length === 0 ||
|
||||
canRetryOpenAICodexStreamWithBufferedContent(
|
||||
streamResponse as Response | undefined,
|
||||
!canReplay,
|
||||
),
|
||||
});
|
||||
|
||||
// Never replay a response through the non-streaming fallback after it
|
||||
// committed output, server-side work began or a local tool completed.
|
||||
// Keep displayable blocks, but discard uncommitted local tools. Partial
|
||||
// text from a clean EOF that cannot be re-streamed is also surfaced with
|
||||
// an error, rather than silently accepted or replaced by a second response.
|
||||
if (
|
||||
retryKind === null &&
|
||||
(!canReplay ||
|
||||
streamEndedIncomplete ||
|
||||
assistantCommitBuffer.hasCrossedSideEffectBoundary())
|
||||
) {
|
||||
for (const committedMessage of assistantCommitBuffer.flushWithoutToolUse()) {
|
||||
yield committedMessage;
|
||||
}
|
||||
@@ -2967,90 +3017,37 @@ async function* queryModel(
|
||||
}
|
||||
}
|
||||
|
||||
// A watchdog stall is recoverable while the failed attempt is still
|
||||
// side-effect-free. Completed thinking/text and partial local tool JSON
|
||||
// stay buffered, so re-establishing the stream cannot duplicate a tool.
|
||||
// A completed local tool block or any server-side tool activity closes
|
||||
// this retry boundary permanently for the attempt.
|
||||
if (
|
||||
streamIdleAborted &&
|
||||
streamingError instanceof StreamWatchdogTimeoutError &&
|
||||
streamingError.safeToRetryStream() &&
|
||||
!assistantCommitBuffer.hasCrossedSideEffectBoundary() &&
|
||||
!signal.aborted
|
||||
) {
|
||||
logForDebugging(
|
||||
`Watchdog timeout before content/tool output, will retry stream: ${errorMessage(
|
||||
streamingError,
|
||||
)}`,
|
||||
{ level: "warn" },
|
||||
);
|
||||
throw new RetriableStreamError(streamingError, assistantCommitBuffer.flush());
|
||||
}
|
||||
|
||||
// The socket under the stream died mid-response (stale pooled keep-alive
|
||||
// connection, proxy/NAT dropping a reused one, upstream edge reset). It
|
||||
// arrives as a bare transport error inside the SSE body, so withRetry
|
||||
// (stream creation only) and isRetryableStreamError (SSE error payloads)
|
||||
// both miss it, and with the non-streaming fallback disabled the turn
|
||||
// would die on a fault a plain re-send clears. Recover on the same
|
||||
// side-effect boundary the watchdog retry uses.
|
||||
if (
|
||||
shouldRetryStreamAfterTransportDisconnect({
|
||||
error: streamingError,
|
||||
hasCrossedSideEffectBoundary:
|
||||
assistantCommitBuffer.hasCrossedSideEffectBoundary(),
|
||||
streamIdleAborted,
|
||||
signalAborted: signal.aborted,
|
||||
})
|
||||
) {
|
||||
// A recoverable failure — an idle stall, a dead socket, a truncated or
|
||||
// cleanly closed stream, an upstream api_error/overloaded_error — arrives
|
||||
// inside the SSE body, so withRetry (stream creation only) misses it, and
|
||||
// with the non-streaming fallback disabled the turn would die on a fault
|
||||
// a plain re-send clears. withStreamRetry re-sends it with backoff.
|
||||
if (retryKind !== null) {
|
||||
// Nothing arrived at all, so the connection was already dead when the
|
||||
// request went out — the pool is serving closed sockets. Stop reusing
|
||||
// it so the retry opens a fresh one. A disconnect after message_start
|
||||
// is a live connection that broke later; that pool stays trusted.
|
||||
if (partialMessage === undefined) {
|
||||
if (
|
||||
partialMessage === undefined &&
|
||||
isRetryableStreamTransportError(streamingError)
|
||||
) {
|
||||
disableKeepAlive();
|
||||
}
|
||||
logForDebugging(
|
||||
`Mid-stream transport disconnect before any tool output, will retry stream: ${errorMessage(
|
||||
`Mid-stream ${retryKind} failure before any output was committed, will retry stream: ${errorMessage(
|
||||
streamingError,
|
||||
)}`,
|
||||
{ level: "warn" },
|
||||
);
|
||||
throw new RetriableStreamError(streamingError, assistantCommitBuffer.flush());
|
||||
}
|
||||
|
||||
if (
|
||||
(newMessages.length === 0 ||
|
||||
canRetryOpenAICodexStreamWithBufferedContent(
|
||||
streamResponse as Response | undefined,
|
||||
assistantCommitBuffer.hasCrossedSideEffectBoundary(),
|
||||
)) &&
|
||||
!streamIdleAborted &&
|
||||
!signal.aborted &&
|
||||
isRetryableStreamError(streamingError)
|
||||
) {
|
||||
logForDebugging(
|
||||
`Transient mid-stream error before any output, will retry stream: ${errorMessage(
|
||||
streamingError,
|
||||
)}`,
|
||||
{ level: "warn" },
|
||||
// Completed text/thinking ride along for a retries-exhausted error; an
|
||||
// uncommitted local tool_use is dropped with the attempt.
|
||||
throw new RetriableStreamError(
|
||||
streamingError,
|
||||
assistantCommitBuffer.flushWithoutToolUse(),
|
||||
retryKind,
|
||||
);
|
||||
throw new RetriableStreamError(streamingError, assistantCommitBuffer.flush());
|
||||
}
|
||||
|
||||
// When the flag is enabled, skip the non-streaming fallback and let the
|
||||
// error propagate to withRetry. The mid-stream fallback causes double tool
|
||||
// execution when streaming tool execution is active: the partial stream
|
||||
// starts a tool, then the non-streaming retry produces the same tool_use
|
||||
// and runs it again. See inc-4258.
|
||||
const disableFallback =
|
||||
isEnvTruthy(process.env.CLAUDE_CODE_DISABLE_NONSTREAMING_FALLBACK) ||
|
||||
getFeatureValue_CACHED_MAY_BE_STALE(
|
||||
"tengu_disable_streaming_to_non_streaming_fallback",
|
||||
false,
|
||||
);
|
||||
|
||||
if (disableFallback) {
|
||||
logForDebugging(
|
||||
`Error streaming (non-streaming fallback disabled): ${errorMessage(streamingError)}`,
|
||||
|
||||
@@ -0,0 +1,363 @@
|
||||
import { afterAll, beforeAll, expect, test } from 'bun:test'
|
||||
import { mkdtemp, rm } from 'node:fs/promises'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { join } from 'node:path'
|
||||
import { createSandboxedTestEnvironment } from '../../../scripts/pr/test-environment.js'
|
||||
import { openaiChatStreamToAnthropic } from '../../server/proxy/streaming/openaiChatStreamToAnthropic.js'
|
||||
import type { AssistantMessage } from '../../types/message.js'
|
||||
|
||||
// Mid-stream failures through the real request path: SDK SSE parsing, the
|
||||
// provider proxy's OpenAI Chat converter, queryModel's retry classification
|
||||
// and withStreamRetry. One transport retry keeps each case to a single real
|
||||
// backoff wait; growth and bounds are covered in streamRetry.test.ts.
|
||||
|
||||
const originalEnvironment = { ...process.env }
|
||||
let sandbox: string
|
||||
let server: ReturnType<typeof Bun.serve>
|
||||
let responses: Array<() => Response> = []
|
||||
let requests: Array<{ stream: unknown }> = []
|
||||
let queryModelWithStreaming: typeof import('./claude.js').queryModelWithStreaming
|
||||
let createUserMessage: typeof import('../../utils/messages.js').createUserMessage
|
||||
let asSystemPrompt: typeof import('../../utils/systemPromptType.js').asSystemPrompt
|
||||
let getEmptyToolPermissionContext: typeof import('../../Tool.js').getEmptyToolPermissionContext
|
||||
const globals = globalThis as typeof globalThis & { MACRO?: { BUILD_TIME: string } }
|
||||
const originalMacro = globals.MACRO
|
||||
|
||||
beforeAll(async () => {
|
||||
sandbox = await mkdtemp(join(tmpdir(), 'claude-stream-recovery-'))
|
||||
for (const key of Object.keys(process.env)) delete process.env[key]
|
||||
Object.assign(process.env, createSandboxedTestEnvironment(sandbox, {
|
||||
NODE_ENV: 'production',
|
||||
CLAUDE_CODE_SIMPLE: '1',
|
||||
ANTHROPIC_API_KEY: 'offline-fixture-key',
|
||||
CLAUDE_CODE_MAX_RETRIES: '1',
|
||||
}, originalEnvironment))
|
||||
server = Bun.serve({ hostname: '127.0.0.1', port: 0, async fetch(request) {
|
||||
const body = await request.json() as { stream?: unknown }
|
||||
requests.push({ stream: body.stream })
|
||||
const next = responses.shift()
|
||||
return next ? next() : new Response('unexpected request', { status: 500 })
|
||||
} })
|
||||
process.env.ANTHROPIC_BASE_URL = `http://127.0.0.1:${server.port}`
|
||||
globals.MACRO = { BUILD_TIME: '' }
|
||||
;({ queryModelWithStreaming } = await import('./claude.js'))
|
||||
;({ createUserMessage } = await import('../../utils/messages.js'))
|
||||
;({ asSystemPrompt } = await import('../../utils/systemPromptType.js'))
|
||||
;({ getEmptyToolPermissionContext } = await import('../../Tool.js'))
|
||||
const { enableConfigs } = await import('../../utils/config.js')
|
||||
enableConfigs()
|
||||
})
|
||||
|
||||
afterAll(async () => {
|
||||
server?.stop(true)
|
||||
for (const key of Object.keys(process.env)) delete process.env[key]
|
||||
Object.assign(process.env, originalEnvironment)
|
||||
if (originalMacro === undefined) delete globals.MACRO
|
||||
else globals.MACRO = originalMacro
|
||||
await rm(sandbox, { recursive: true, force: true })
|
||||
})
|
||||
|
||||
const encoder = new TextEncoder()
|
||||
|
||||
function sse(events: Array<Record<string, unknown>>): string {
|
||||
return events.map(event => `event: ${event.type}\ndata: ${JSON.stringify(event)}\n\n`).join('')
|
||||
}
|
||||
|
||||
function anthropicStream(events: Array<Record<string, unknown>>, holdOpen = false): () => Response {
|
||||
return () => new Response(holdOpen
|
||||
? new ReadableStream({ start(controller) { controller.enqueue(encoder.encode(sse(events))) } })
|
||||
: sse(events), { headers: { 'content-type': 'text/event-stream' } })
|
||||
}
|
||||
|
||||
/** An OpenAI Chat upstream piped through the provider proxy's converter. */
|
||||
function proxiedChatStream(chunks: Array<Record<string, unknown>>, upstreamFailure?: Error): () => Response {
|
||||
return () => {
|
||||
let index = 0
|
||||
const upstream = new ReadableStream<Uint8Array>({
|
||||
pull(controller) {
|
||||
if (index < chunks.length) {
|
||||
controller.enqueue(encoder.encode(`data: ${JSON.stringify(chunks[index++])}\n\n`))
|
||||
} else if (upstreamFailure) {
|
||||
controller.error(upstreamFailure)
|
||||
} else {
|
||||
controller.close()
|
||||
}
|
||||
},
|
||||
})
|
||||
return new Response(openaiChatStreamToAnthropic(upstream, 'fixture-model'), {
|
||||
headers: { 'content-type': 'text/event-stream' },
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
const messageStart = { type: 'message_start', message: {
|
||||
id: 'msg_recovery', type: 'message', role: 'assistant', model: 'fixture-model', content: [],
|
||||
stop_reason: null, stop_sequence: null, usage: { input_tokens: 7, output_tokens: 0 },
|
||||
} }
|
||||
|
||||
function textBlock(text: string, index = 0) {
|
||||
return [
|
||||
{ type: 'content_block_start', index, content_block: { type: 'text', text: '' } },
|
||||
{ type: 'content_block_delta', index, delta: { type: 'text_delta', text } },
|
||||
{ type: 'content_block_stop', index },
|
||||
]
|
||||
}
|
||||
|
||||
function toolBlock(id: string, index = 0, blockType = 'tool_use') {
|
||||
return [
|
||||
{ type: 'content_block_start', index, content_block: { type: blockType, id, name: blockType === 'tool_use' ? 'Bash' : 'web_search', input: {} } },
|
||||
{ type: 'content_block_delta', index, delta: { type: 'input_json_delta', partial_json: '{"command":"echo fixture"}' } },
|
||||
{ type: 'content_block_stop', index },
|
||||
]
|
||||
}
|
||||
|
||||
function terminal(reason: string) {
|
||||
return [
|
||||
{ type: 'message_delta', delta: { stop_reason: reason, stop_sequence: null }, usage: { output_tokens: 5 } },
|
||||
{ type: 'message_stop' },
|
||||
]
|
||||
}
|
||||
|
||||
function errorEvent(type: string, message = 'fixture failure') {
|
||||
return { type: 'error', error: { type, message } }
|
||||
}
|
||||
|
||||
const recovered = anthropicStream([messageStart, ...textBlock('recovered'), ...terminal('end_turn')])
|
||||
|
||||
function chatText(content: string) {
|
||||
return { id: 'chatcmpl-recovery', model: 'fixture-model', choices: [{ index: 0, delta: { content }, finish_reason: null }] }
|
||||
}
|
||||
|
||||
async function run(queued: Array<() => Response>, env: Record<string, string | undefined> = {}) {
|
||||
const saved = Object.fromEntries(Object.keys(env).map(key => [key, process.env[key]]))
|
||||
for (const [key, value] of Object.entries(env)) {
|
||||
if (value === undefined) delete process.env[key]
|
||||
else process.env[key] = value
|
||||
}
|
||||
process.env.CLAUDE_CODE_DISABLE_NONSTREAMING_FALLBACK ??= '1'
|
||||
responses = [...queued]
|
||||
requests = []
|
||||
// biome-ignore lint/suspicious/noExplicitAny: collects heterogeneous stream messages
|
||||
const out: any[] = []
|
||||
try {
|
||||
for await (const message of queryModelWithStreaming({
|
||||
messages: [createUserMessage({ content: 'fixture' })],
|
||||
systemPrompt: asSystemPrompt([]), thinkingConfig: { type: 'disabled' }, tools: [],
|
||||
signal: new AbortController().signal,
|
||||
options: { model: 'fixture-model', querySource: 'insights', agents: [], isNonInteractiveSession: true,
|
||||
hasAppendSystemPrompt: false, mcpTools: [], enablePromptCaching: false,
|
||||
getToolPermissionContext: async () => getEmptyToolPermissionContext() },
|
||||
})) {
|
||||
out.push(message)
|
||||
}
|
||||
} finally {
|
||||
for (const [key, value] of Object.entries(saved)) {
|
||||
if (value === undefined) delete process.env[key]
|
||||
else process.env[key] = value
|
||||
}
|
||||
delete process.env.CLAUDE_CODE_DISABLE_NONSTREAMING_FALLBACK
|
||||
}
|
||||
const assistants: AssistantMessage[] = out.filter(message => message.type === 'assistant')
|
||||
return {
|
||||
out,
|
||||
assistants,
|
||||
replies: assistants.filter(message => !message.isApiErrorMessage),
|
||||
errors: assistants.filter(message => message.isApiErrorMessage),
|
||||
}
|
||||
}
|
||||
|
||||
function contentOf(messages: AssistantMessage[]): string {
|
||||
return JSON.stringify(messages.map(message => message.message.content))
|
||||
}
|
||||
|
||||
// biome-ignore lint/suspicious/noExplicitAny: inspects heterogeneous stream messages
|
||||
function expectRetriedOnce(out: any[]) {
|
||||
expect(out.filter(message => message.type === 'system' && message.subtype === 'streaming_fallback')
|
||||
.map(message => message.cause)).toEqual(['stream_retry'])
|
||||
const statuses = out.filter(message => message.type === 'system' && message.subtype === 'api_error')
|
||||
expect(statuses.map(message => [message.retryAttempt, message.maxRetries])).toEqual([[1, 1]])
|
||||
expect(statuses[0].retryInMs).toBeGreaterThanOrEqual(500)
|
||||
}
|
||||
|
||||
test('a proxy stream_truncated event is re-sent and only the retry reaches the transcript', async () => {
|
||||
const { out, replies, errors } = await run([
|
||||
proxiedChatStream([chatText('partial-attempt')]),
|
||||
recovered,
|
||||
])
|
||||
|
||||
expect(requests).toHaveLength(2)
|
||||
expectRetriedOnce(out)
|
||||
expect(errors).toHaveLength(0)
|
||||
expect(replies).toHaveLength(1)
|
||||
expect(replies[0]?.message.content).toEqual([{ type: 'text', text: 'recovered' }])
|
||||
expect(contentOf(replies)).not.toContain('partial-attempt')
|
||||
}, 15_000)
|
||||
|
||||
test('a proxy stream_error event is re-sent', async () => {
|
||||
const { out, replies, errors } = await run([
|
||||
proxiedChatStream([chatText('partial-attempt')], Object.assign(new Error('upstream reset'), { code: 'ECONNRESET' })),
|
||||
recovered,
|
||||
])
|
||||
|
||||
expect(requests).toHaveLength(2)
|
||||
expectRetriedOnce(out)
|
||||
expect(errors).toHaveLength(0)
|
||||
expect(contentOf(replies)).toBe(JSON.stringify([[{ type: 'text', text: 'recovered' }]]))
|
||||
}, 15_000)
|
||||
|
||||
test('a clean EOF before message_stop is re-sent without committing the first attempt', async () => {
|
||||
const { out, replies, errors } = await run([
|
||||
anthropicStream([messageStart, ...textBlock('partial-attempt')]),
|
||||
recovered,
|
||||
])
|
||||
|
||||
expect(requests).toHaveLength(2)
|
||||
expectRetriedOnce(out)
|
||||
expect(errors).toHaveLength(0)
|
||||
expect(contentOf(replies)).toBe(JSON.stringify([[{ type: 'text', text: 'recovered' }]]))
|
||||
}, 15_000)
|
||||
|
||||
test('an empty stream is re-sent when the non-streaming fallback is disabled', async () => {
|
||||
const { out, replies, errors } = await run([anthropicStream([]), recovered])
|
||||
|
||||
expect(requests).toHaveLength(2)
|
||||
expectRetriedOnce(out)
|
||||
expect(errors).toHaveLength(0)
|
||||
expect(replies).toHaveLength(1)
|
||||
}, 15_000)
|
||||
|
||||
test('a completed but uncommitted tool call is re-sent and released only once', async () => {
|
||||
const { replies, errors } = await run([
|
||||
anthropicStream([messageStart, ...toolBlock('toolu_first')]),
|
||||
anthropicStream([messageStart, ...toolBlock('toolu_second'), ...terminal('tool_use')]),
|
||||
])
|
||||
|
||||
expect(requests).toHaveLength(2)
|
||||
expect(errors).toHaveLength(0)
|
||||
const toolIds = replies.flatMap(message => message.message.content)
|
||||
.flatMap(block => block.type === 'tool_use' ? [block.id] : [])
|
||||
expect(toolIds).toEqual(['toolu_second'])
|
||||
}, 15_000)
|
||||
|
||||
test('a proxied tool call cut off after finish_reason is re-sent, never released twice', async () => {
|
||||
const toolCall = {
|
||||
id: 'chatcmpl-tool', model: 'fixture-model', choices: [{ index: 0, delta: { tool_calls: [{
|
||||
index: 0, id: 'call_first', type: 'function',
|
||||
function: { name: 'Bash', arguments: '{"command":"echo fixture"}' },
|
||||
}] }, finish_reason: null }],
|
||||
}
|
||||
const finished = { id: 'chatcmpl-tool', model: 'fixture-model', choices: [{ index: 0, delta: {}, finish_reason: 'tool_calls' }] }
|
||||
const { out, replies, errors } = await run([
|
||||
// The converter has already closed the tool block when the upstream dies.
|
||||
proxiedChatStream([toolCall, finished], Object.assign(new Error('upstream reset'), { code: 'ECONNRESET' })),
|
||||
anthropicStream([messageStart, ...toolBlock('toolu_second'), ...terminal('tool_use')]),
|
||||
])
|
||||
|
||||
expect(out.some(message => message.type === 'stream_event'
|
||||
&& message.event.type === 'content_block_stop')).toBe(true)
|
||||
expect(requests).toHaveLength(2)
|
||||
expect(errors).toHaveLength(0)
|
||||
const toolIds = replies.flatMap(message => message.message.content)
|
||||
.flatMap(block => block.type === 'tool_use' ? [block.id] : [])
|
||||
expect(toolIds).toEqual(['toolu_second'])
|
||||
}, 15_000)
|
||||
|
||||
test('an idle stall after a completed tool block is re-sent on the watchdog budget', async () => {
|
||||
const { out, replies, errors } = await run([
|
||||
anthropicStream([messageStart, ...toolBlock('toolu_first')], true),
|
||||
anthropicStream([messageStart, ...toolBlock('toolu_second'), ...terminal('tool_use')]),
|
||||
], {
|
||||
CLAUDE_ENABLE_STREAM_WATCHDOG: '1',
|
||||
CLAUDE_STREAM_IDLE_TIMEOUT_MS: '300',
|
||||
CLAUDE_STREAM_FIRST_TOKEN_TIMEOUT_MS: '300',
|
||||
})
|
||||
|
||||
expect(requests).toHaveLength(2)
|
||||
expect(errors).toHaveLength(0)
|
||||
const statuses = out.filter(message => message.type === 'system' && message.subtype === 'api_error')
|
||||
expect(statuses.map(message => message.maxRetries)).toEqual([2])
|
||||
const toolIds = replies.flatMap(message => message.message.content)
|
||||
.flatMap(block => block.type === 'tool_use' ? [block.id] : [])
|
||||
expect(toolIds).toEqual(['toolu_second'])
|
||||
}, 15_000)
|
||||
|
||||
test('retries are bounded, then the original error surfaces with the last attempt\'s text once', async () => {
|
||||
const truncatedAfterText = anthropicStream([
|
||||
messageStart,
|
||||
...textBlock('kept-once'),
|
||||
errorEvent('stream_truncated', 'OpenAI Chat upstream stream ended without finish_reason'),
|
||||
])
|
||||
const { replies, errors } = await run([truncatedAfterText, truncatedAfterText])
|
||||
|
||||
expect(requests).toHaveLength(2)
|
||||
expect(contentOf(replies)).toBe(JSON.stringify([[{ type: 'text', text: 'kept-once' }]]))
|
||||
expect(errors).toHaveLength(1)
|
||||
expect(contentOf(errors)).toContain('ended without finish_reason')
|
||||
}, 15_000)
|
||||
|
||||
for (const type of ['invalid_request_error', 'authentication_error', 'permission_error', 'billing_error']) {
|
||||
test(`a mid-stream ${type} is not retried`, async () => {
|
||||
const { errors } = await run([
|
||||
anthropicStream([messageStart, errorEvent(type)]),
|
||||
recovered,
|
||||
])
|
||||
|
||||
expect(requests).toHaveLength(1)
|
||||
expect(errors).toHaveLength(1)
|
||||
})
|
||||
}
|
||||
|
||||
test('server-side tool activity is never replayed', async () => {
|
||||
const { errors } = await run([
|
||||
anthropicStream([messageStart, ...toolBlock('srvtoolu_first', 0, 'server_tool_use').slice(0, 2),
|
||||
errorEvent('stream_truncated')]),
|
||||
recovered,
|
||||
])
|
||||
|
||||
expect(requests).toHaveLength(1)
|
||||
expect(errors).toHaveLength(1)
|
||||
})
|
||||
|
||||
test('an attempt that already committed output is never replayed', async () => {
|
||||
const { assistants } = await run([
|
||||
anthropicStream([
|
||||
messageStart,
|
||||
...textBlock('committed'),
|
||||
{ type: 'message_delta', delta: { stop_reason: 'refusal', stop_sequence: null }, usage: { output_tokens: 5 } },
|
||||
errorEvent('stream_truncated'),
|
||||
]),
|
||||
recovered,
|
||||
])
|
||||
|
||||
expect(requests).toHaveLength(1)
|
||||
expect(contentOf(assistants).match(/committed/g)).toHaveLength(1)
|
||||
expect(contentOf(assistants)).not.toContain('recovered')
|
||||
})
|
||||
|
||||
test('with the non-streaming fallback available, a proxy truncation still goes to it', async () => {
|
||||
const fallbackReply = () => Response.json({
|
||||
id: 'msg_fallback', type: 'message', role: 'assistant', model: 'fixture-model',
|
||||
content: [{ type: 'text', text: 'fallback' }], stop_reason: 'end_turn', stop_sequence: null,
|
||||
usage: { input_tokens: 7, output_tokens: 1 },
|
||||
})
|
||||
const { out, replies } = await run([
|
||||
proxiedChatStream([chatText('partial-attempt')]),
|
||||
fallbackReply,
|
||||
], { CLAUDE_CODE_DISABLE_NONSTREAMING_FALLBACK: '0' })
|
||||
|
||||
expect(requests.map(request => request.stream === true)).toEqual([true, false])
|
||||
expect(out.some(message => message.subtype === 'streaming_fallback' && message.cause === 'stream_retry')).toBe(false)
|
||||
expect(contentOf(replies)).toBe(JSON.stringify([[{ type: 'text', text: 'fallback' }]]))
|
||||
}, 15_000)
|
||||
|
||||
test('with the non-streaming fallback available, a response cut off mid-way is re-streamed', async () => {
|
||||
const { out, replies, errors } = await run([
|
||||
anthropicStream([messageStart, ...textBlock('partial-attempt')]),
|
||||
recovered,
|
||||
], { CLAUDE_CODE_DISABLE_NONSTREAMING_FALLBACK: '0' })
|
||||
|
||||
expect(requests.map(request => request.stream === true)).toEqual([true, true])
|
||||
expectRetriedOnce(out)
|
||||
expect(errors).toHaveLength(0)
|
||||
expect(contentOf(replies)).toBe(JSON.stringify([[{ type: 'text', text: 'recovered' }]]))
|
||||
}, 15_000)
|
||||
@@ -54,4 +54,32 @@ describe('StreamAssistantCommitBuffer', () => {
|
||||
expect(buffer.flushWithoutToolUse()).toEqual([thinking])
|
||||
expect(buffer.flush()).toEqual([])
|
||||
})
|
||||
|
||||
test('a deferred tool crosses the boundary without committing, so the attempt stays replayable', () => {
|
||||
const buffer = new StreamAssistantCommitBuffer<ReturnType<typeof assistant>>({
|
||||
deferToolUseCommit: true,
|
||||
})
|
||||
|
||||
buffer.add(assistant('text'), 'text')
|
||||
buffer.add(assistant('tool'), 'tool_use')
|
||||
buffer.add(assistant('after-tool'), 'text')
|
||||
expect(buffer.hasCrossedSideEffectBoundary()).toBe(true)
|
||||
expect(buffer.hasCommitted()).toBe(false)
|
||||
|
||||
expect(buffer.flush().map(message => message.uuid)).toEqual(['text', 'tool', 'after-tool'])
|
||||
expect(buffer.hasCommitted()).toBe(true)
|
||||
})
|
||||
|
||||
test('commits as soon as anything is handed out', () => {
|
||||
const serverTool = new StreamAssistantCommitBuffer<ReturnType<typeof assistant>>({
|
||||
deferToolUseCommit: true,
|
||||
})
|
||||
serverTool.add(assistant('server-tool'), 'server_tool_use')
|
||||
expect(serverTool.hasCommitted()).toBe(true)
|
||||
|
||||
const empty = new StreamAssistantCommitBuffer<ReturnType<typeof assistant>>()
|
||||
expect(empty.flush()).toEqual([])
|
||||
expect(empty.flushWithoutToolUse()).toEqual([])
|
||||
expect(empty.hasCommitted()).toBe(false)
|
||||
})
|
||||
})
|
||||
|
||||
@@ -2,10 +2,16 @@
|
||||
* Holds completed, side-effect-free assistant blocks until a stream either
|
||||
* finishes or crosses a tool boundary. A watchdog retry can then discard the
|
||||
* failed attempt without leaving orphan thinking/text in the transcript.
|
||||
*
|
||||
* With deferToolUseCommit, a completed local tool_use crosses the boundary but
|
||||
* stays held too, so it is never handed to the tool executor before the
|
||||
* response completes. Whether an attempt can still be discarded and re-sent is
|
||||
* therefore hasCommitted(), not hasCrossedSideEffectBoundary().
|
||||
*/
|
||||
export class StreamAssistantCommitBuffer<T> {
|
||||
private pending: Array<{ value: T; blockType: string }> = []
|
||||
private crossedSideEffectBoundary = false
|
||||
private committed = false
|
||||
|
||||
constructor(
|
||||
private readonly options: { deferToolUseCommit?: boolean } = {},
|
||||
@@ -13,7 +19,7 @@ export class StreamAssistantCommitBuffer<T> {
|
||||
|
||||
add(value: T, blockType: string): T[] {
|
||||
if (this.crossedSideEffectBoundary) {
|
||||
if (!this.options.deferToolUseCommit) return [value]
|
||||
if (!this.options.deferToolUseCommit) return this.commit([value])
|
||||
this.pending.push({ value, blockType })
|
||||
return []
|
||||
}
|
||||
@@ -39,7 +45,7 @@ export class StreamAssistantCommitBuffer<T> {
|
||||
.filter(entry => entry.blockType !== 'tool_use')
|
||||
.map(entry => entry.value)
|
||||
this.pending = []
|
||||
return values
|
||||
return this.commit(values)
|
||||
}
|
||||
|
||||
hasPendingToolUse(): boolean {
|
||||
@@ -50,9 +56,19 @@ export class StreamAssistantCommitBuffer<T> {
|
||||
return this.crossedSideEffectBoundary
|
||||
}
|
||||
|
||||
/** True once any block was handed out to be yielded to the consumer. */
|
||||
hasCommitted(): boolean {
|
||||
return this.committed
|
||||
}
|
||||
|
||||
private drain(): T[] {
|
||||
const values = this.pending.map(entry => entry.value)
|
||||
this.pending = []
|
||||
return this.commit(values)
|
||||
}
|
||||
|
||||
private commit(values: T[]): T[] {
|
||||
if (values.length > 0) this.committed = true
|
||||
return values
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,5 +1,20 @@
|
||||
import { describe, expect, test } from 'bun:test'
|
||||
import { shouldTriggerNonStreamingFallbackForEmptyStream } from './streamFallback.js'
|
||||
import {
|
||||
shouldTriggerNonStreamingFallbackForEmptyStream,
|
||||
StreamEndedEarlyError,
|
||||
} from './streamFallback.js'
|
||||
|
||||
describe('StreamEndedEarlyError', () => {
|
||||
test('keeps the user-visible wording that error matchers key on', () => {
|
||||
expect(new StreamEndedEarlyError('no_events').message).toBe(
|
||||
'Stream ended without receiving any events',
|
||||
)
|
||||
expect(new StreamEndedEarlyError('incomplete').message).toBe(
|
||||
'Provider stream ended before completing the response',
|
||||
)
|
||||
expect(new StreamEndedEarlyError('incomplete')).toBeInstanceOf(Error)
|
||||
})
|
||||
})
|
||||
|
||||
describe('stream fallback policy', () => {
|
||||
test('falls back when a stream never produced message_start', () => {
|
||||
|
||||
@@ -11,3 +11,22 @@ export function shouldTriggerNonStreamingFallbackForEmptyStream({
|
||||
if (assistantMessageCount > 0) return false
|
||||
return stopReason === null || stopReason === 'tool_use'
|
||||
}
|
||||
|
||||
export type StreamEndedEarlyReason = 'no_events' | 'incomplete'
|
||||
|
||||
/**
|
||||
* A 200 SSE body that closed cleanly before the response completed: nothing
|
||||
* usable arrived (`no_events`), or it stopped short of message_stop and a
|
||||
* stop_reason, possibly with blocks still open (`incomplete`). The provider
|
||||
* never ruled on the request, so — like a reset socket — a re-send can clear it.
|
||||
*/
|
||||
export class StreamEndedEarlyError extends Error {
|
||||
constructor(readonly reason: StreamEndedEarlyReason) {
|
||||
super(
|
||||
reason === 'no_events'
|
||||
? 'Stream ended without receiving any events'
|
||||
: 'Provider stream ended before completing the response',
|
||||
)
|
||||
this.name = 'StreamEndedEarlyError'
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,9 +1,14 @@
|
||||
import { afterAll, beforeAll, describe, expect, test } from 'bun:test'
|
||||
import { afterAll, beforeAll, describe, expect, spyOn, test } from 'bun:test'
|
||||
import { APIError } from '@anthropic-ai/sdk'
|
||||
import { withStreamRetry } from './streamRetry.js'
|
||||
import { RetriableStreamError } from './withRetry.js'
|
||||
|
||||
const RETRY_ENV = 'CLAUDE_STREAM_TRANSIENT_RETRY_MAX'
|
||||
const API_RETRY_ENV = 'CLAUDE_CODE_MAX_RETRIES'
|
||||
|
||||
// Backoff waits are covered by recordingSleep below; the remaining tests only
|
||||
// care about which attempts run.
|
||||
const noSleep = async () => {}
|
||||
|
||||
// getAssistantMessageFromError() (invoked when retries are exhausted) consults
|
||||
// isClaudeAISubscriber(), which throws if no auth is configured. We only assert
|
||||
@@ -58,7 +63,7 @@ describe('withStreamRetry', () => {
|
||||
yield { type: 'assistant', message: { content: [] }, uuid: 'ok' }
|
||||
})()
|
||||
|
||||
const out = await collect(withStreamRetry(attempt, 'test-model', []))
|
||||
const out = await collect(withStreamRetry(attempt, 'test-model', [], { sleep: noSleep }))
|
||||
|
||||
expect(calls).toBe(2)
|
||||
expect(out).toContainEqual(expect.objectContaining({
|
||||
@@ -84,7 +89,7 @@ describe('withStreamRetry', () => {
|
||||
throw retriableError()
|
||||
})()
|
||||
|
||||
const out = await collect(withStreamRetry(attempt, 'test-model', []))
|
||||
const out = await collect(withStreamRetry(attempt, 'test-model', [], { sleep: noSleep }))
|
||||
|
||||
expect(calls).toBe(3) // 1 initial attempt + 2 retries
|
||||
expect(out.filter(
|
||||
@@ -122,7 +127,7 @@ describe('withStreamRetry', () => {
|
||||
yield { type: 'assistant', message: { content: [] }, uuid: 'recovered' }
|
||||
})()
|
||||
|
||||
const recovered = await collect(withStreamRetry(recovers, 'test-model', []))
|
||||
const recovered = await collect(withStreamRetry(recovers, 'test-model', [], { sleep: noSleep }))
|
||||
expect(calls).toBe(2)
|
||||
expect(recovered.at(-1)?.uuid).toBe('recovered')
|
||||
expect(recovered.some(m => m.isApiErrorMessage)).toBe(false)
|
||||
@@ -133,7 +138,7 @@ describe('withStreamRetry', () => {
|
||||
throw socketReset()
|
||||
})()
|
||||
|
||||
const failed = await collect(withStreamRetry(persists, 'test-model', []))
|
||||
const failed = await collect(withStreamRetry(persists, 'test-model', [], { sleep: noSleep }))
|
||||
const last = failed.at(-1)
|
||||
expect(last?.type).toBe('assistant')
|
||||
expect(last?.isApiErrorMessage).toBe(true)
|
||||
@@ -150,7 +155,7 @@ describe('withStreamRetry', () => {
|
||||
})()
|
||||
|
||||
await expect(
|
||||
collect(withStreamRetry(attempt, 'test-model', [])),
|
||||
collect(withStreamRetry(attempt, 'test-model', [], { sleep: noSleep })),
|
||||
).rejects.toThrow('fatal')
|
||||
expect(calls).toBe(1)
|
||||
})
|
||||
@@ -165,7 +170,7 @@ describe('withStreamRetry', () => {
|
||||
throw retriableError()
|
||||
})()
|
||||
|
||||
const out = await collect(withStreamRetry(attempt, 'test-model', []))
|
||||
const out = await collect(withStreamRetry(attempt, 'test-model', [], { sleep: noSleep }))
|
||||
|
||||
expect(calls).toBe(1)
|
||||
expect(out.at(-1)?.type).toBe('assistant')
|
||||
@@ -205,7 +210,7 @@ describe('withStreamRetry', () => {
|
||||
)
|
||||
})()
|
||||
|
||||
const out = await collect(withStreamRetry(attempt, 'test-model', []))
|
||||
const out = await collect(withStreamRetry(attempt, 'test-model', [], { sleep: noSleep }))
|
||||
const partials = out.filter(
|
||||
message =>
|
||||
message.type === 'assistant' &&
|
||||
@@ -228,10 +233,244 @@ describe('withStreamRetry', () => {
|
||||
yield { type: 'assistant', message: { content: [] }, uuid: 'clean' }
|
||||
})()
|
||||
|
||||
const out = await collect(withStreamRetry(attempt, 'test-model', []))
|
||||
const out = await collect(withStreamRetry(attempt, 'test-model', [], { sleep: noSleep }))
|
||||
|
||||
expect(calls).toBe(1)
|
||||
expect(out).toHaveLength(1)
|
||||
expect(out[0].uuid).toBe('clean')
|
||||
})
|
||||
})
|
||||
|
||||
/** The SSE error event the provider proxy writes when the upstream is cut off. */
|
||||
function proxyTruncation(): RetriableStreamError {
|
||||
const body = {
|
||||
type: 'error',
|
||||
error: {
|
||||
type: 'stream_truncated',
|
||||
message: 'OpenAI Chat upstream stream ended without finish_reason',
|
||||
},
|
||||
}
|
||||
return new RetriableStreamError(
|
||||
new APIError(undefined, body, undefined, undefined),
|
||||
[],
|
||||
'transport',
|
||||
)
|
||||
}
|
||||
|
||||
function withEnv(values: Record<string, string | undefined>) {
|
||||
const saved = Object.fromEntries(Object.keys(values).map(key => [key, process.env[key]]))
|
||||
for (const [key, value] of Object.entries(values)) {
|
||||
if (value === undefined) delete process.env[key]
|
||||
else process.env[key] = value
|
||||
}
|
||||
return () => {
|
||||
for (const [key, value] of Object.entries(saved)) {
|
||||
if (value === undefined) delete process.env[key]
|
||||
else process.env[key] = value
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// biome-ignore lint/suspicious/noExplicitAny: test harness inspects heterogeneous stream messages
|
||||
function retryStatuses(out: any[]): any[] {
|
||||
return out.filter(m => m.type === 'system' && m.subtype === 'api_error')
|
||||
}
|
||||
|
||||
describe('withStreamRetry backoff and budgets', () => {
|
||||
test('backs off between re-sends with growing, capped delays and reports each wait', async () => {
|
||||
const restore = withEnv({ [RETRY_ENV]: undefined, [API_RETRY_ENV]: undefined })
|
||||
const random = spyOn(Math, 'random').mockReturnValue(0)
|
||||
try {
|
||||
const delays: number[] = []
|
||||
let calls = 0
|
||||
const attempt = () =>
|
||||
// biome-ignore lint/suspicious/noExplicitAny: mock stream messages
|
||||
(async function* (): AsyncGenerator<any, void> {
|
||||
calls++
|
||||
if (calls <= 7) throw proxyTruncation()
|
||||
yield { type: 'assistant', message: { content: [] }, uuid: 'recovered' }
|
||||
})()
|
||||
|
||||
const out = await collect(withStreamRetry(attempt, 'test-model', [], {
|
||||
sleep: async ms => { delays.push(ms) },
|
||||
}))
|
||||
|
||||
expect(calls).toBe(8)
|
||||
expect(delays).toEqual([500, 1000, 2000, 4000, 8000, 16000, 32000])
|
||||
expect(retryStatuses(out).map(m => [m.retryAttempt, m.maxRetries, m.retryInMs])).toEqual(
|
||||
delays.map((delay, index) => [index + 1, 10, delay]),
|
||||
)
|
||||
expect(out.at(-1)?.uuid).toBe('recovered')
|
||||
expect(out.some(m => m.isApiErrorMessage)).toBe(false)
|
||||
} finally {
|
||||
random.mockRestore()
|
||||
restore()
|
||||
}
|
||||
})
|
||||
|
||||
test('jitter stays within a quarter of the base delay', async () => {
|
||||
const restore = withEnv({ [RETRY_ENV]: undefined, [API_RETRY_ENV]: '3' })
|
||||
try {
|
||||
const delays: number[] = []
|
||||
const attempt = () =>
|
||||
// biome-ignore lint/suspicious/noExplicitAny: mock stream messages
|
||||
(async function* (): AsyncGenerator<any, void> {
|
||||
throw proxyTruncation()
|
||||
})()
|
||||
|
||||
await collect(withStreamRetry(attempt, 'test-model', [], {
|
||||
sleep: async ms => { delays.push(ms) },
|
||||
}))
|
||||
|
||||
expect(delays).toHaveLength(3)
|
||||
delays.forEach((delay, index) => {
|
||||
const base = 500 * 2 ** index
|
||||
expect(delay).toBeGreaterThanOrEqual(base)
|
||||
expect(delay).toBeLessThanOrEqual(base * 1.25)
|
||||
})
|
||||
} finally {
|
||||
restore()
|
||||
}
|
||||
})
|
||||
|
||||
test('transport failures draw on the API retry budget, then surface the original error', async () => {
|
||||
for (const [apiRetries, expectedCalls] of [[undefined, 11], ['3', 4]] as const) {
|
||||
const restore = withEnv({ [RETRY_ENV]: undefined, [API_RETRY_ENV]: apiRetries })
|
||||
try {
|
||||
let calls = 0
|
||||
let sleeps = 0
|
||||
const attempt = () =>
|
||||
// biome-ignore lint/suspicious/noExplicitAny: mock stream messages
|
||||
(async function* (): AsyncGenerator<any, void> {
|
||||
calls++
|
||||
throw proxyTruncation()
|
||||
})()
|
||||
|
||||
const out = await collect(withStreamRetry(attempt, 'test-model', [], {
|
||||
sleep: async () => { sleeps++ },
|
||||
}))
|
||||
|
||||
expect(calls).toBe(expectedCalls)
|
||||
expect(sleeps).toBe(expectedCalls - 1)
|
||||
const last = out.at(-1)
|
||||
expect(last?.type).toBe('assistant')
|
||||
expect(last?.isApiErrorMessage).toBe(true)
|
||||
expect(JSON.stringify(last?.message.content)).toContain('ended without finish_reason')
|
||||
} finally {
|
||||
restore()
|
||||
}
|
||||
}
|
||||
})
|
||||
|
||||
test('stalls and upstream-reported errors keep their small budget', async () => {
|
||||
const restore = withEnv({ [RETRY_ENV]: undefined, [API_RETRY_ENV]: undefined })
|
||||
try {
|
||||
let calls = 0
|
||||
const attempt = () =>
|
||||
// biome-ignore lint/suspicious/noExplicitAny: mock stream messages
|
||||
(async function* (): AsyncGenerator<any, void> {
|
||||
calls++
|
||||
throw retriableError()
|
||||
})()
|
||||
|
||||
const out = await collect(withStreamRetry(attempt, 'test-model', [], { sleep: noSleep }))
|
||||
|
||||
expect(calls).toBe(3)
|
||||
expect(out.at(-1)?.isApiErrorMessage).toBe(true)
|
||||
} finally {
|
||||
restore()
|
||||
}
|
||||
})
|
||||
|
||||
test('discards the failed attempt and reports the retry before waiting', async () => {
|
||||
const restore = withEnv({ [RETRY_ENV]: undefined, [API_RETRY_ENV]: undefined })
|
||||
try {
|
||||
// biome-ignore lint/suspicious/noExplicitAny: mock stream messages
|
||||
const out: any[] = []
|
||||
let yieldedBeforeWait = -1
|
||||
let calls = 0
|
||||
const attempt = () =>
|
||||
// biome-ignore lint/suspicious/noExplicitAny: mock stream messages
|
||||
(async function* (): AsyncGenerator<any, void> {
|
||||
calls++
|
||||
if (calls === 1) {
|
||||
yield { type: 'stream_event', event: { type: 'content_block_delta' } }
|
||||
throw proxyTruncation()
|
||||
}
|
||||
yield { type: 'assistant', message: { content: [] }, uuid: 'second' }
|
||||
})()
|
||||
|
||||
for await (const message of withStreamRetry(attempt, 'test-model', [], {
|
||||
sleep: async () => { yieldedBeforeWait = out.length },
|
||||
})) {
|
||||
out.push(message)
|
||||
}
|
||||
|
||||
expect(out.map(m => m.subtype ?? m.type)).toEqual([
|
||||
'stream_event',
|
||||
'streaming_fallback',
|
||||
'api_error',
|
||||
'assistant',
|
||||
])
|
||||
expect(out[1].cause).toBe('stream_retry')
|
||||
expect(out[2].retryAttempt).toBe(1)
|
||||
expect(yieldedBeforeWait).toBe(3)
|
||||
} finally {
|
||||
restore()
|
||||
}
|
||||
})
|
||||
|
||||
test('an interrupt during the backoff stops without another attempt or a provider error', async () => {
|
||||
const restore = withEnv({ [RETRY_ENV]: undefined, [API_RETRY_ENV]: undefined })
|
||||
try {
|
||||
const controller = new AbortController()
|
||||
let calls = 0
|
||||
const attempt = () =>
|
||||
// biome-ignore lint/suspicious/noExplicitAny: mock stream messages
|
||||
(async function* (): AsyncGenerator<any, void> {
|
||||
calls++
|
||||
throw proxyTruncation()
|
||||
})()
|
||||
|
||||
const out = await collect(withStreamRetry(attempt, 'test-model', [], {
|
||||
signal: controller.signal,
|
||||
sleep: async () => { controller.abort() },
|
||||
}))
|
||||
|
||||
expect(calls).toBe(1)
|
||||
expect(out.some(m => m.type === 'assistant')).toBe(false)
|
||||
expect(out.at(-1)?.subtype).toBe('api_error')
|
||||
} finally {
|
||||
restore()
|
||||
}
|
||||
})
|
||||
|
||||
test('never replays an attempt that already committed assistant output', async () => {
|
||||
const restore = withEnv({ [RETRY_ENV]: undefined, [API_RETRY_ENV]: undefined })
|
||||
try {
|
||||
let calls = 0
|
||||
const toolUse = {
|
||||
type: 'assistant',
|
||||
message: { content: [{ type: 'tool_use', id: 'toolu_committed', name: 'Bash', input: {} }] },
|
||||
uuid: 'committed-tool',
|
||||
}
|
||||
const attempt = () =>
|
||||
// biome-ignore lint/suspicious/noExplicitAny: mock stream messages
|
||||
(async function* (): AsyncGenerator<any, void> {
|
||||
calls++
|
||||
// Once a tool_use reaches the consumer it may already be running.
|
||||
yield toolUse
|
||||
throw proxyTruncation()
|
||||
})()
|
||||
|
||||
const out = await collect(withStreamRetry(attempt, 'test-model', [], { sleep: noSleep }))
|
||||
|
||||
expect(calls).toBe(1)
|
||||
expect(out.filter(m => m.uuid === 'committed-tool')).toHaveLength(1)
|
||||
expect(out.some(m => m.subtype === 'streaming_fallback')).toBe(false)
|
||||
expect(out.at(-1)?.isApiErrorMessage).toBe(true)
|
||||
} finally {
|
||||
restore()
|
||||
}
|
||||
})
|
||||
})
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
import { APIConnectionError, APIError } from "@anthropic-ai/sdk";
|
||||
import type {
|
||||
AssistantMessage,
|
||||
Message,
|
||||
@@ -7,14 +8,19 @@ import type {
|
||||
} from "../../types/message.js";
|
||||
import { logForDebugging } from "../../utils/debug.js";
|
||||
import { errorMessage } from "../../utils/errors.js";
|
||||
import { createSystemStreamingFallbackMessage } from "../../utils/messages.js";
|
||||
import {
|
||||
createSystemAPIErrorMessage,
|
||||
createSystemStreamingFallbackMessage,
|
||||
} from "../../utils/messages.js";
|
||||
import { sleep } from "../../utils/sleep.js";
|
||||
import {
|
||||
type AnalyticsMetadata_I_VERIFIED_THIS_IS_NOT_CODE_OR_FILEPATHS,
|
||||
logEvent,
|
||||
} from "../analytics/index.js";
|
||||
import { getAssistantMessageFromError } from "./errors.js";
|
||||
import {
|
||||
getMaxStreamTransientRetries,
|
||||
getMaxStreamRetries,
|
||||
getRetryDelay,
|
||||
RetriableStreamError,
|
||||
} from "./withRetry.js";
|
||||
|
||||
@@ -25,52 +31,78 @@ type StreamQueryMessage =
|
||||
| SystemAPIErrorMessage
|
||||
| SystemStreamingFallbackMessage;
|
||||
|
||||
type StreamRetryOptions = {
|
||||
/** Cancels the backoff wait; the caller then handles the interrupt. */
|
||||
signal?: AbortSignal;
|
||||
/** Injectable for tests. Must resolve (not reject) when `signal` aborts. */
|
||||
sleep?: (ms: number, signal?: AbortSignal) => Promise<void>;
|
||||
};
|
||||
|
||||
/**
|
||||
* Wrap a single streaming query attempt with transient mid-stream retries.
|
||||
* Wrap a single streaming query attempt with mid-stream retries.
|
||||
*
|
||||
* Mid-stream transient errors — a malformed tool_call a local provider rejects,
|
||||
* or an upstream api_error/overloaded_error SSE event — arrive inside the 200
|
||||
* SSE body, so they never reach withRetry (which only guards stream creation).
|
||||
* queryModel() detects them via isRetryableStreamError and throws
|
||||
* RetriableStreamError; here we catch it and re-run the whole attempt by
|
||||
* re-invoking the generator factory. Re-running queryModel() is a clean re-send:
|
||||
* every per-request value is a fresh local, so there is nothing to reset by hand.
|
||||
* Mid-stream failures — a socket reset, a proxy-reported truncation, a clean
|
||||
* EOF before message_stop, a watchdog stall, or an upstream api_error /
|
||||
* overloaded_error SSE event — arrive inside the 200 SSE body, so they never
|
||||
* reach withRetry (which only guards stream creation). queryModel() classifies
|
||||
* them (getStreamRetryKind) and throws RetriableStreamError; here we catch it
|
||||
* and re-run the whole attempt by re-invoking the generator factory. Re-running
|
||||
* queryModel() is a clean re-send: every per-request value is a fresh local, so
|
||||
* there is nothing to reset by hand. Each attempt arms its own watchdog timers,
|
||||
* so the backoff never counts against an attempt's idle or max-duration budget.
|
||||
*
|
||||
* Re-sends back off like withRetry (getRetryDelay) within the budget of the
|
||||
* failure's kind (getMaxStreamRetries), then surface the original error.
|
||||
*
|
||||
* Safe against the double-tool-execution hazard (#766 / inc-4258): queryModel
|
||||
* buffers completed thinking/text and only throws RetriableStreamError before a
|
||||
* local tool block completes or server-side tool activity starts.
|
||||
* holds completed blocks — local tool_use included — until the response
|
||||
* completes, and only throws RetriableStreamError while nothing was committed
|
||||
* and no server-side tool work began. As a backstop, an attempt that already
|
||||
* yielded an assistant message is never replayed.
|
||||
*
|
||||
* A failed attempt may already have yielded raw stream_event partials. Before
|
||||
* retrying, emit a bounded recovery signal so streaming consumers discard that
|
||||
* attempt instead of appending the next attempt to stale text/tool JSON.
|
||||
* attempt instead of appending the next attempt to stale text/tool JSON, then a
|
||||
* retry status for the backoff wait.
|
||||
*/
|
||||
export async function* withStreamRetry(
|
||||
attempt: () => AsyncGenerator<StreamQueryMessage, void>,
|
||||
model: string,
|
||||
messages: Message[],
|
||||
{ signal, sleep: wait = sleep }: StreamRetryOptions = {},
|
||||
): AsyncGenerator<StreamQueryMessage, void> {
|
||||
const maxRetries = getMaxStreamTransientRetries();
|
||||
for (let i = 0; ; i++) {
|
||||
for (let retries = 0; ; retries++) {
|
||||
let committedOutput = false;
|
||||
try {
|
||||
yield* attempt();
|
||||
for await (const message of attempt()) {
|
||||
if (message.type === "assistant") committedOutput = true;
|
||||
yield message;
|
||||
}
|
||||
return;
|
||||
} catch (error) {
|
||||
if (!(error instanceof RetriableStreamError)) {
|
||||
throw error;
|
||||
}
|
||||
if (i >= maxRetries) {
|
||||
// Retries exhausted — surface the original error as an assistant
|
||||
// message, matching queryModel's normal terminal-error behavior.
|
||||
const maxRetries = getMaxStreamRetries(error.kind);
|
||||
if (committedOutput || retries >= maxRetries) {
|
||||
// Surface the original error as an assistant message, matching
|
||||
// queryModel's normal terminal-error behavior.
|
||||
logForDebugging(
|
||||
`Transient mid-stream error: retries exhausted after ${maxRetries} attempt(s): ${errorMessage(
|
||||
error.originalError,
|
||||
)}`,
|
||||
committedOutput
|
||||
? `Mid-stream ${error.kind} error after the attempt committed output; not replaying: ${errorMessage(
|
||||
error.originalError,
|
||||
)}`
|
||||
: `Mid-stream ${error.kind} error: retries exhausted after ${maxRetries} attempt(s): ${errorMessage(
|
||||
error.originalError,
|
||||
)}`,
|
||||
{ level: "error" },
|
||||
);
|
||||
logEvent("tengu_stream_transient_retry_exhausted", {
|
||||
attempts: maxRetries,
|
||||
attempts: retries,
|
||||
model:
|
||||
model as AnalyticsMetadata_I_VERIFIED_THIS_IS_NOT_CODE_OR_FILEPATHS,
|
||||
kind:
|
||||
error.kind as AnalyticsMetadata_I_VERIFIED_THIS_IS_NOT_CODE_OR_FILEPATHS,
|
||||
});
|
||||
for (const bufferedMessage of error.bufferedMessages) {
|
||||
yield bufferedMessage;
|
||||
@@ -80,20 +112,44 @@ export async function* withStreamRetry(
|
||||
});
|
||||
return;
|
||||
}
|
||||
const retryAttempt = retries + 1;
|
||||
const delayMs = getRetryDelay(retryAttempt);
|
||||
logForDebugging(
|
||||
`Transient mid-stream error, retrying (attempt ${i + 1}/${maxRetries}): ${errorMessage(
|
||||
`Mid-stream ${error.kind} error, retrying in ${Math.round(delayMs)}ms (attempt ${retryAttempt}/${maxRetries}): ${errorMessage(
|
||||
error.originalError,
|
||||
)}`,
|
||||
{ level: "warn" },
|
||||
);
|
||||
logEvent("tengu_stream_transient_retry", {
|
||||
attempt: i + 1,
|
||||
attempt: retryAttempt,
|
||||
delayMs,
|
||||
model:
|
||||
model as AnalyticsMetadata_I_VERIFIED_THIS_IS_NOT_CODE_OR_FILEPATHS,
|
||||
kind:
|
||||
error.kind as AnalyticsMetadata_I_VERIFIED_THIS_IS_NOT_CODE_OR_FILEPATHS,
|
||||
});
|
||||
// Raw deltas from the failed attempt may already be visible. Consumers
|
||||
// use this bounded retry signal to discard only that in-flight attempt.
|
||||
yield createSystemStreamingFallbackMessage("stream_retry");
|
||||
yield createSystemAPIErrorMessage(
|
||||
toRetryStatusError(error.originalError),
|
||||
delayMs,
|
||||
retryAttempt,
|
||||
maxRetries,
|
||||
);
|
||||
await wait(delayMs, signal);
|
||||
// An interrupt during the wait is the user's, not a stream fault: stop
|
||||
// quietly and let the caller handle it, as queryModel does for aborts.
|
||||
if (signal?.aborted) return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/** Retry status messages carry an APIError; transport faults are bare Errors. */
|
||||
function toRetryStatusError(error: unknown): APIError {
|
||||
if (error instanceof APIError) return error;
|
||||
return new APIConnectionError({
|
||||
message: errorMessage(error),
|
||||
cause: error instanceof Error ? error : undefined,
|
||||
});
|
||||
}
|
||||
|
||||
@@ -0,0 +1,183 @@
|
||||
import { expect, test } from 'bun:test'
|
||||
import { mkdtemp, readFile, rm } from 'node:fs/promises'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { join } from 'node:path'
|
||||
import { fileURLToPath } from 'node:url'
|
||||
import { createSandboxedTestEnvironment } from '../../../scripts/pr/test-environment.js'
|
||||
import type { Tool, ToolUseContext } from '../../Tool.js'
|
||||
import type { QueryParams } from '../../query.js'
|
||||
|
||||
// The first response dies after its tool call already completed in the stream.
|
||||
// The re-sent request must run that tool exactly once, through the real query
|
||||
// loop and tool executor, never from the failed attempt.
|
||||
const scenarios = ['chat-error-after-finish', 'anthropic-eof-after-tool', 'chat-truncated-mid-tool'] as const
|
||||
type Scenario = typeof scenarios[number]
|
||||
const resultPrefix = 'STREAM_RETRY_TOOL_RESULT:'
|
||||
const childScenario = process.env.CC_HAHA_STREAM_RETRY_TOOL_SCENARIO
|
||||
|
||||
// Loading the real query graph in a shared Bun test process can cache runtime
|
||||
// modules before later mock.module tests install their fixtures, so every
|
||||
// production import runs in a fresh child.
|
||||
async function runScenario(root: string, scenario: Scenario) {
|
||||
;(globalThis as typeof globalThis & { MACRO?: { BUILD_TIME: string } }).MACRO = { BUILD_TIME: '' }
|
||||
const { randomUUID } = await import('node:crypto')
|
||||
const { writeFile } = await import('node:fs/promises')
|
||||
const { z } = await import('zod')
|
||||
const { openaiChatStreamToAnthropic } = await import('../../server/proxy/streaming/openaiChatStreamToAnthropic.js')
|
||||
const bootstrap = await import('../../bootstrap/state.js')
|
||||
bootstrap.setCwdState(root)
|
||||
bootstrap.setOriginalCwd(root)
|
||||
bootstrap.setProjectRoot(root)
|
||||
process.chdir(root)
|
||||
const { query } = await import('../../query.js')
|
||||
const { queryModelWithStreaming: callModel } = await import('./claude.js')
|
||||
const { getDefaultAppState } = await import('../../state/AppStateStore.js')
|
||||
const { createUserMessage } = await import('../../utils/messages.js')
|
||||
const { asSystemPrompt } = await import('../../utils/systemPromptType.js')
|
||||
const { enableConfigs } = await import('../../utils/config.js')
|
||||
enableConfigs()
|
||||
|
||||
let executions = 0
|
||||
let requests = 0
|
||||
const committedToolIds: string[] = []
|
||||
const target = join(root, `${scenario}.txt`)
|
||||
const args = JSON.stringify({ file_path: target, content: 'written exactly once' })
|
||||
const encoder = new TextEncoder()
|
||||
|
||||
const proxied = (lines: unknown[], failure?: Error) => {
|
||||
let index = 0
|
||||
const upstream = new ReadableStream<Uint8Array>({
|
||||
pull(controller) {
|
||||
if (index < lines.length) controller.enqueue(encoder.encode(`data: ${JSON.stringify(lines[index++])}\n\n`))
|
||||
else if (failure) controller.error(failure)
|
||||
else controller.close()
|
||||
},
|
||||
})
|
||||
return openaiChatStreamToAnthropic(upstream, 'fixture-model')
|
||||
}
|
||||
const chatToolCall = { id: 'chatcmpl-fixture', model: 'fixture-model', choices: [{ index: 0, delta: { tool_calls: [{
|
||||
index: 0, id: 'call_fixture', type: 'function', function: { name: 'FixtureWrite', arguments: args },
|
||||
}] }, finish_reason: null }] }
|
||||
const chatFinish = { id: 'chatcmpl-fixture', model: 'fixture-model', choices: [{ index: 0, delta: {}, finish_reason: 'tool_calls' }] }
|
||||
const chatDone = { id: 'chatcmpl-fixture', model: 'fixture-model', choices: [{ index: 0, delta: { content: 'complete' }, finish_reason: 'stop' }] }
|
||||
const event = (type: string, fields: Record<string, unknown>) =>
|
||||
`event: ${type}\ndata: ${JSON.stringify({ type, ...fields })}\n\n`
|
||||
const anthropicToolWithoutStop = event('message_start', { message: {
|
||||
id: 'msg_fixture', type: 'message', role: 'assistant', model: 'fixture-model',
|
||||
content: [], stop_reason: null, stop_sequence: null, usage: { input_tokens: 10, output_tokens: 0 },
|
||||
} })
|
||||
+ event('content_block_start', { index: 0, content_block: { type: 'tool_use', id: 'call_fixture', name: 'FixtureWrite', input: {} } })
|
||||
+ event('content_block_delta', { index: 0, delta: { type: 'input_json_delta', partial_json: args } })
|
||||
+ event('content_block_stop', { index: 0 })
|
||||
+ event('message_delta', { delta: { stop_reason: 'tool_use', stop_sequence: null }, usage: { output_tokens: 5 } })
|
||||
|
||||
const server = Bun.serve({ hostname: '127.0.0.1', port: 0, async fetch(request) {
|
||||
await request.json()
|
||||
requests++
|
||||
let body: ReadableStream<Uint8Array> | string
|
||||
if (requests === 1) {
|
||||
const reset = Object.assign(new Error('upstream reset'), { code: 'ECONNRESET' })
|
||||
body = scenario === 'chat-error-after-finish'
|
||||
? proxied([chatToolCall, chatFinish], reset)
|
||||
: scenario === 'chat-truncated-mid-tool'
|
||||
? proxied([chatToolCall])
|
||||
: anthropicToolWithoutStop
|
||||
} else if (requests === 2) {
|
||||
body = proxied([chatToolCall, chatFinish])
|
||||
} else {
|
||||
body = proxied([chatDone])
|
||||
}
|
||||
return new Response(body, { headers: { 'content-type': 'text/event-stream' } })
|
||||
} })
|
||||
process.env.ANTHROPIC_BASE_URL = `http://127.0.0.1:${server.port}`
|
||||
|
||||
const input = z.object({ file_path: z.string(), content: z.string() })
|
||||
const tool = {
|
||||
name: 'FixtureWrite', inputSchema: input,
|
||||
prompt: async () => 'Write a fixture file', maxResultSizeChars: 1000, isConcurrencySafe: () => false, isReadOnly: () => false,
|
||||
isEnabled: () => true, userFacingName: () => 'fixture write', description: async () => 'Write a fixture file',
|
||||
call: async (value: z.infer<typeof input>) => {
|
||||
executions++
|
||||
await writeFile(value.file_path, value.content, { flag: 'a' })
|
||||
return { data: 'written' }
|
||||
},
|
||||
mapToolResultToToolResultBlockParam: (data: string, id: string) => ({ type: 'tool_result', tool_use_id: id, content: data }),
|
||||
} as unknown as Tool
|
||||
let state = getDefaultAppState()
|
||||
const toolUseContext = {
|
||||
options: { commands: [], debug: false, mainLoopModel: 'fixture-model', tools: [tool], verbose: false,
|
||||
thinkingConfig: { type: 'disabled' }, mcpClients: [], mcpResources: {}, isNonInteractiveSession: true,
|
||||
agentDefinitions: { activeAgents: [], allAgents: [] } },
|
||||
abortController: new AbortController(), readFileState: new Map(),
|
||||
getAppState: () => state, setAppState: (update: (value: typeof state) => typeof state) => { state = update(state) },
|
||||
setInProgressToolUseIDs: () => {}, setResponseLength: () => {},
|
||||
updateFileHistoryState: () => {}, updateAttributionState: () => {}, messages: [],
|
||||
} as unknown as ToolUseContext
|
||||
const params: QueryParams = {
|
||||
messages: [createUserMessage({ content: 'Run the fixture write once' })],
|
||||
systemPrompt: asSystemPrompt([]), userContext: {}, systemContext: {},
|
||||
canUseTool: async (_tool, value) => ({ behavior: 'allow', updatedInput: value }),
|
||||
toolUseContext, querySource: 'sdk', maxTurns: 3,
|
||||
deps: { callModel, microcompact: async messages => ({ messages }), autocompact: async () => ({}), uuid: randomUUID },
|
||||
}
|
||||
try {
|
||||
for await (const message of query(params)) {
|
||||
if (message.type !== 'assistant') continue
|
||||
for (const block of message.message.content) {
|
||||
if (block.type === 'tool_use') committedToolIds.push(block.id)
|
||||
}
|
||||
}
|
||||
return { executions, requests, committedToolIds }
|
||||
} finally {
|
||||
toolUseContext.abortController.abort()
|
||||
server.stop(true)
|
||||
}
|
||||
}
|
||||
|
||||
for (const scenario of scenarios) {
|
||||
if (childScenario && childScenario !== scenario) continue
|
||||
test(`a re-sent stream runs its tool exactly once: ${scenario}`, async () => {
|
||||
if (childScenario) {
|
||||
const result = await runScenario(process.env.HOME!, scenario)
|
||||
console.log(resultPrefix + JSON.stringify(result))
|
||||
return
|
||||
}
|
||||
const root = await mkdtemp(join(tmpdir(), 'stream-retry-tool-'))
|
||||
const child = Bun.spawn([process.execPath, '--no-env-file', 'test', fileURLToPath(import.meta.url)], {
|
||||
cwd: root,
|
||||
env: createSandboxedTestEnvironment(root, {
|
||||
CC_HAHA_STREAM_RETRY_TOOL_SCENARIO: scenario,
|
||||
NODE_ENV: 'production',
|
||||
CLAUDE_CODE_SIMPLE: '1',
|
||||
CLAUDE_CODE_DISABLE_AUTO_MEMORY: '1',
|
||||
CLAUDE_CODE_ENABLE_PROMPT_SUGGESTION: '0',
|
||||
CLAUDE_CODE_DISABLE_NONSTREAMING_FALLBACK: '1',
|
||||
CLAUDE_CODE_MAX_RETRIES: '1',
|
||||
CLAUDE_CODE_DISABLE_NONESSENTIAL_TRAFFIC: '1',
|
||||
ANTHROPIC_API_KEY: 'loopback-fixture-key',
|
||||
ANTHROPIC_MODEL: 'fixture-model',
|
||||
}),
|
||||
stdin: 'ignore', stdout: 'pipe', stderr: 'pipe',
|
||||
})
|
||||
const timeout = setTimeout(() => child.kill('SIGKILL'), 15_000)
|
||||
try {
|
||||
const [exitCode, stdout, stderr] = await Promise.all([
|
||||
child.exited, new Response(child.stdout).text(), new Response(child.stderr).text(),
|
||||
])
|
||||
expect(exitCode, stderr).toBe(0)
|
||||
const resultLine = stdout.split('\n').find(line => line.startsWith(resultPrefix))
|
||||
expect(resultLine, stdout + stderr).toBeDefined()
|
||||
const result = JSON.parse(resultLine!.slice(resultPrefix.length))
|
||||
// Request 1 dies, request 2 is the re-send, request 3 carries the tool result.
|
||||
expect(result.requests).toBe(3)
|
||||
expect(result.committedToolIds).toEqual(['call_fixture'])
|
||||
expect(result.executions).toBe(1)
|
||||
expect(await readFile(join(root, `${scenario}.txt`), 'utf8')).toBe('written exactly once')
|
||||
} finally {
|
||||
clearTimeout(timeout)
|
||||
child.kill()
|
||||
await child.exited
|
||||
await rm(root, { recursive: true, force: true })
|
||||
}
|
||||
}, 20_000)
|
||||
}
|
||||
@@ -79,7 +79,7 @@ describe('stream watchdog state', () => {
|
||||
expect(error.message).not.toContain('no chunks received')
|
||||
})
|
||||
|
||||
test('retries partial local tool input but stops after the tool block completes', () => {
|
||||
test('retries a local tool block whether partial or completed, since it is held until the response completes', () => {
|
||||
const state = createStreamWatchdogState()
|
||||
state.recordEvent({ type: 'message_start' })
|
||||
state.recordEvent({
|
||||
@@ -101,7 +101,8 @@ describe('stream watchdog state', () => {
|
||||
const error = state.createTimeoutError('idle', 240_000)
|
||||
|
||||
expect(error.phase).toBe('mid_stream')
|
||||
expect(error.safeToRetryStream()).toBe(false)
|
||||
expect(error.streamSnapshot.localToolUseCompleted).toBe(true)
|
||||
expect(error.safeToRetryStream()).toBe(true)
|
||||
})
|
||||
|
||||
test('never retries after server-side tool activity begins', () => {
|
||||
|
||||
@@ -160,10 +160,15 @@ export class StreamWatchdogTimeoutError extends Error {
|
||||
this.phase = streamSnapshot.phase
|
||||
}
|
||||
|
||||
/**
|
||||
* Only an idle stall is re-sent: max_duration and tool_input_duration
|
||||
* already spent their budget, and server-side tool work cannot be undone.
|
||||
* A completed local tool_use does not block it — the commit buffer holds
|
||||
* that block until the response completes, so the tool has not run.
|
||||
*/
|
||||
safeToRetryStream(): boolean {
|
||||
return (
|
||||
this.reason === 'idle' &&
|
||||
!this.streamSnapshot.localToolUseCompleted &&
|
||||
!this.streamSnapshot.serverToolUseStarted
|
||||
)
|
||||
}
|
||||
|
||||
@@ -3,15 +3,19 @@ import type Anthropic from '@anthropic-ai/sdk'
|
||||
import { APIConnectionError, APIError, APIUserAbortError } from '@anthropic-ai/sdk'
|
||||
import { clearFastModeCooldown, getFastModeRuntimeState } from '../../utils/fastMode.js'
|
||||
import { _resetKeepAliveForTesting, getProxyFetchOptions } from '../../utils/proxy.js'
|
||||
import { StreamEndedEarlyError } from './streamFallback.js'
|
||||
import { createStreamWatchdogState } from './streamWatchdog.js'
|
||||
import {
|
||||
BASE_DELAY_MS,
|
||||
getRetryDelay,
|
||||
CannotRetryError,
|
||||
getMaxStreamRetries,
|
||||
getMaxStreamTransientRetries,
|
||||
getStreamRetryKind,
|
||||
isProxyStreamTransportError,
|
||||
isRetryableStreamError,
|
||||
isRetryableStreamTransportError,
|
||||
RetriableStreamError,
|
||||
shouldRetryStreamAfterTransportDisconnect,
|
||||
withRetry,
|
||||
} from './withRetry.js'
|
||||
|
||||
@@ -312,55 +316,184 @@ describe('isRetryableStreamTransportError', () => {
|
||||
})
|
||||
})
|
||||
|
||||
describe('shouldRetryStreamAfterTransportDisconnect', () => {
|
||||
/** An SSE `error` event exactly as the SDK raises it mid-stream (no status). */
|
||||
function sseErrorEvent(type: string, message = 'fixture'): APIError {
|
||||
const body = { type: 'error', error: { type, message } }
|
||||
return new APIError(undefined, body, undefined, undefined)
|
||||
}
|
||||
|
||||
describe('isProxyStreamTransportError', () => {
|
||||
test('matches the truncation and transport errors the provider proxy emits', () => {
|
||||
expect(isProxyStreamTransportError(sseErrorEvent(
|
||||
'stream_truncated',
|
||||
'OpenAI Chat upstream stream ended without finish_reason',
|
||||
))).toBe(true)
|
||||
expect(isProxyStreamTransportError(sseErrorEvent('stream_error', 'socket hang up'))).toBe(true)
|
||||
})
|
||||
|
||||
test('does not match provider verdicts on the request', () => {
|
||||
for (const type of [
|
||||
'api_error',
|
||||
'overloaded_error',
|
||||
'invalid_request_error',
|
||||
'authentication_error',
|
||||
'permission_error',
|
||||
'billing_error',
|
||||
'rate_limit_error',
|
||||
'not_found_error',
|
||||
]) {
|
||||
expect(isProxyStreamTransportError(sseErrorEvent(type))).toBe(false)
|
||||
}
|
||||
})
|
||||
|
||||
test('only applies to status-less stream events and never to policy rejections', () => {
|
||||
const body = { type: 'error', error: { type: 'stream_error', message: 'x' } }
|
||||
expect(isProxyStreamTransportError(new APIError(502, body, undefined, undefined))).toBe(false)
|
||||
expect(isProxyStreamTransportError(new Error('{"type":"stream_truncated"}'))).toBe(false)
|
||||
expect(isProxyStreamTransportError(new APIError(undefined, {
|
||||
type: 'error', error: { type: 'stream_error', code: 'cyber_policy', message: 'Rejected' },
|
||||
}, undefined, undefined))).toBe(false)
|
||||
})
|
||||
})
|
||||
|
||||
describe('getStreamRetryKind', () => {
|
||||
const disconnect = Object.assign(new Error('socket closed'), {
|
||||
code: 'ECONNRESET',
|
||||
})
|
||||
const clean = {
|
||||
error: disconnect,
|
||||
hasCrossedSideEffectBoundary: false,
|
||||
const replayable = {
|
||||
error: disconnect as unknown,
|
||||
canReplay: true,
|
||||
streamIdleAborted: false,
|
||||
signalAborted: false,
|
||||
nonStreamingFallbackAvailable: false,
|
||||
transientRetryAllowed: true,
|
||||
}
|
||||
|
||||
test('recovers a disconnect while the attempt is still side-effect-free', () => {
|
||||
expect(shouldRetryStreamAfterTransportDisconnect(clean)).toBe(true)
|
||||
test('re-sends a dead socket as a transport failure', () => {
|
||||
expect(getStreamRetryKind(replayable)).toBe('transport')
|
||||
expect(getStreamRetryKind({ ...replayable, nonStreamingFallbackAvailable: true })).toBe('transport')
|
||||
})
|
||||
|
||||
test('refuses once a tool block completed — a re-send would run it twice', () => {
|
||||
expect(
|
||||
shouldRetryStreamAfterTransportDisconnect({
|
||||
...clean,
|
||||
hasCrossedSideEffectBoundary: true,
|
||||
}),
|
||||
).toBe(false)
|
||||
test('re-sends proxy truncations, clean EOF and empty streams when no fallback takes them', () => {
|
||||
for (const error of [
|
||||
sseErrorEvent('stream_truncated'),
|
||||
sseErrorEvent('stream_error'),
|
||||
new StreamEndedEarlyError('incomplete'),
|
||||
new StreamEndedEarlyError('no_events'),
|
||||
]) {
|
||||
expect(getStreamRetryKind({ ...replayable, error })).toBe('transport')
|
||||
}
|
||||
})
|
||||
|
||||
test('leaves watchdog aborts to their own retry path', () => {
|
||||
expect(
|
||||
shouldRetryStreamAfterTransportDisconnect({
|
||||
...clean,
|
||||
streamIdleAborted: true,
|
||||
}),
|
||||
).toBe(false)
|
||||
test('leaves truncations the non-streaming fallback recovers to it, except a response cut off mid-way', () => {
|
||||
const withFallback = { ...replayable, nonStreamingFallbackAvailable: true }
|
||||
expect(getStreamRetryKind({ ...withFallback, error: sseErrorEvent('stream_truncated') })).toBeNull()
|
||||
expect(getStreamRetryKind({ ...withFallback, error: sseErrorEvent('stream_error') })).toBeNull()
|
||||
expect(getStreamRetryKind({ ...withFallback, error: new StreamEndedEarlyError('no_events') })).toBeNull()
|
||||
expect(getStreamRetryKind({ ...withFallback, error: new StreamEndedEarlyError('incomplete') })).toBe('transport')
|
||||
})
|
||||
|
||||
test('never fights a user abort', () => {
|
||||
expect(
|
||||
shouldRetryStreamAfterTransportDisconnect({
|
||||
...clean,
|
||||
signalAborted: true,
|
||||
}),
|
||||
).toBe(false)
|
||||
test('refuses once the attempt committed output or started server-side work', () => {
|
||||
for (const error of [
|
||||
disconnect,
|
||||
sseErrorEvent('stream_truncated'),
|
||||
new StreamEndedEarlyError('incomplete'),
|
||||
sseErrorEvent('api_error'),
|
||||
]) {
|
||||
expect(getStreamRetryKind({ ...replayable, error, canReplay: false })).toBeNull()
|
||||
}
|
||||
})
|
||||
|
||||
test('ignores non-transport stream errors', () => {
|
||||
expect(
|
||||
shouldRetryStreamAfterTransportDisconnect({
|
||||
...clean,
|
||||
error: new Error('Stream ended without receiving any events'),
|
||||
}),
|
||||
).toBe(false)
|
||||
test('keeps upstream-reported errors on the transient budget, only where allowed', () => {
|
||||
expect(getStreamRetryKind({ ...replayable, error: sseErrorEvent('api_error') })).toBe('transient')
|
||||
expect(getStreamRetryKind({ ...replayable, error: sseErrorEvent('overloaded_error') })).toBe('transient')
|
||||
expect(getStreamRetryKind({
|
||||
...replayable,
|
||||
error: sseErrorEvent('api_error'),
|
||||
transientRetryAllowed: false,
|
||||
})).toBeNull()
|
||||
})
|
||||
|
||||
test('keeps provider rejections and unknown faults non-retryable', () => {
|
||||
for (const error of [
|
||||
sseErrorEvent('invalid_request_error', 'prompt is too long'),
|
||||
sseErrorEvent('authentication_error'),
|
||||
sseErrorEvent('permission_error'),
|
||||
sseErrorEvent('billing_error'),
|
||||
new APIError(400, { type: 'error', error: { type: 'invalid_request_error', message: 'bad' } }, undefined, undefined),
|
||||
new RangeError('Content block not found'),
|
||||
new Error('boom'),
|
||||
]) {
|
||||
expect(getStreamRetryKind({ ...replayable, error })).toBeNull()
|
||||
}
|
||||
})
|
||||
|
||||
test('retries only an idle watchdog stall, on its own budget', () => {
|
||||
const state = createStreamWatchdogState()
|
||||
state.recordEvent({ type: 'message_start' })
|
||||
state.recordEvent({ type: 'content_block_start', index: 0, content_block: { type: 'tool_use' } })
|
||||
state.recordEvent({ type: 'content_block_stop', index: 0 })
|
||||
const stalled = { ...replayable, streamIdleAborted: true }
|
||||
expect(getStreamRetryKind({ ...stalled, error: state.createTimeoutError('idle', 240_000) })).toBe('watchdog')
|
||||
expect(getStreamRetryKind({ ...stalled, error: state.createTimeoutError('max_duration', 600_000) })).toBeNull()
|
||||
expect(getStreamRetryKind({ ...stalled, error: state.createTimeoutError('tool_input_duration', 120_000) })).toBeNull()
|
||||
// A watchdog abort is never mistaken for the socket fault it causes.
|
||||
expect(getStreamRetryKind({ ...stalled, error: disconnect })).toBeNull()
|
||||
})
|
||||
|
||||
test('never fights a user abort or a policy rejection', () => {
|
||||
expect(getStreamRetryKind({ ...replayable, signalAborted: true })).toBeNull()
|
||||
const policy = Object.assign(new Error('Disconnected'), {
|
||||
code: 'ECONNRESET',
|
||||
cause: { error: { code: 'cyber_policy' } },
|
||||
})
|
||||
expect(getStreamRetryKind({ ...replayable, error: policy })).toBeNull()
|
||||
})
|
||||
})
|
||||
|
||||
describe('getMaxStreamRetries', () => {
|
||||
const TRANSIENT_ENV = 'CLAUDE_STREAM_TRANSIENT_RETRY_MAX'
|
||||
const API_ENV = 'CLAUDE_CODE_MAX_RETRIES'
|
||||
|
||||
function withEnv(values: Record<string, string | undefined>, run: () => void) {
|
||||
const saved = { [TRANSIENT_ENV]: process.env[TRANSIENT_ENV], [API_ENV]: process.env[API_ENV] }
|
||||
try {
|
||||
for (const [key, value] of Object.entries(values)) {
|
||||
if (value === undefined) delete process.env[key]
|
||||
else process.env[key] = value
|
||||
}
|
||||
run()
|
||||
} finally {
|
||||
for (const [key, value] of Object.entries(saved)) {
|
||||
if (value === undefined) delete process.env[key]
|
||||
else process.env[key] = value
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
test('gives transport failures the API retry budget instead of the transient one', () => {
|
||||
withEnv({ [TRANSIENT_ENV]: undefined, [API_ENV]: undefined }, () => {
|
||||
expect(getMaxStreamRetries('transport')).toBe(10)
|
||||
expect(getMaxStreamRetries('watchdog')).toBe(2)
|
||||
expect(getMaxStreamRetries('transient')).toBe(2)
|
||||
})
|
||||
withEnv({ [TRANSIENT_ENV]: undefined, [API_ENV]: '4' }, () => {
|
||||
expect(getMaxStreamRetries('transport')).toBe(4)
|
||||
})
|
||||
withEnv({ [TRANSIENT_ENV]: undefined, [API_ENV]: 'abc' }, () => {
|
||||
expect(getMaxStreamRetries('transport')).toBe(10)
|
||||
})
|
||||
})
|
||||
|
||||
test('an explicit transient override still caps transport retries, so 0 disables them', () => {
|
||||
withEnv({ [TRANSIENT_ENV]: '0', [API_ENV]: undefined }, () => {
|
||||
expect(getMaxStreamRetries('transport')).toBe(0)
|
||||
expect(getMaxStreamRetries('transient')).toBe(0)
|
||||
})
|
||||
withEnv({ [TRANSIENT_ENV]: '3', [API_ENV]: '1' }, () => {
|
||||
expect(getMaxStreamRetries('transport')).toBe(1)
|
||||
expect(getMaxStreamRetries('watchdog')).toBe(3)
|
||||
})
|
||||
})
|
||||
})
|
||||
|
||||
@@ -400,6 +533,8 @@ describe('RetriableStreamError', () => {
|
||||
expect(wrapped.originalError).toBe(original)
|
||||
expect(wrapped.name).toBe('RetriableStreamError')
|
||||
expect(wrapped.message).toContain('boom')
|
||||
expect(wrapped.kind).toBe('transient')
|
||||
expect(new RetriableStreamError(original, [], 'transport').kind).toBe('transport')
|
||||
})
|
||||
})
|
||||
|
||||
|
||||
+121
-37
@@ -51,6 +51,8 @@ import {
|
||||
REPEATED_529_ERROR_MESSAGE,
|
||||
} from './errors.js'
|
||||
import { extractConnectionErrorDetails } from './errorUtils.js'
|
||||
import { StreamEndedEarlyError } from './streamFallback.js'
|
||||
import { StreamWatchdogTimeoutError } from './streamWatchdog.js'
|
||||
import { isOpenAIPolicyError } from '../openaiAuth/policyError.js'
|
||||
|
||||
const abortError = () => new APIUserAbortError()
|
||||
@@ -174,12 +176,23 @@ export class FallbackTriggeredError extends Error {
|
||||
}
|
||||
|
||||
/**
|
||||
* Raised inside the streaming path when a transient, server-side error arrives
|
||||
* Which budget a mid-stream re-send draws from (see getMaxStreamRetries):
|
||||
* - transport: the connection under the stream died or the body ended before
|
||||
* the response did (reset socket, proxy-reported truncation, clean EOF).
|
||||
* Nothing ruled on the request, so it gets the stream-creation budget.
|
||||
* - watchdog: an idle stall. The attempt already burned its idle budget.
|
||||
* - transient: an error the upstream itself reported inside the stream
|
||||
* (api_error/overloaded_error).
|
||||
*/
|
||||
export type StreamRetryKind = 'transport' | 'watchdog' | 'transient'
|
||||
|
||||
/**
|
||||
* Raised inside the streaming path when a recoverable failure arrives
|
||||
* mid-stream (inside the 200 SSE body) and therefore bypasses withRetry — which
|
||||
* only wraps stream *creation*, not stream *consumption*. Caught by
|
||||
* withStreamRetry() in claude.ts, which re-establishes the stream and retries
|
||||
* the whole request. Carries the original SDK error so the retries-exhausted
|
||||
* path can surface a faithful API-error message.
|
||||
* withStreamRetry() in streamRetry.ts, which re-establishes the stream and
|
||||
* retries the whole request. Carries the original SDK error so the
|
||||
* retries-exhausted path can surface a faithful API-error message.
|
||||
*/
|
||||
export class RetriableStreamError extends Error {
|
||||
public readonly bufferedMessages: readonly AssistantMessage[]
|
||||
@@ -187,6 +200,7 @@ export class RetriableStreamError extends Error {
|
||||
constructor(
|
||||
public readonly originalError: unknown,
|
||||
bufferedMessages: readonly AssistantMessage[] = [],
|
||||
public readonly kind: StreamRetryKind = 'transient',
|
||||
) {
|
||||
super(errorMessage(originalError))
|
||||
this.name = 'RetriableStreamError'
|
||||
@@ -252,8 +266,7 @@ const STREAM_TRANSPORT_DISCONNECT_CODES = new Set([
|
||||
* socket connection was closed unexpectedly", which is not an APIError at all.
|
||||
* Hence the check walks the cause chain for a code instead of keying on type.
|
||||
*
|
||||
* Retrying is only safe behind a side-effect guard; see
|
||||
* shouldRetryStreamAfterTransportDisconnect.
|
||||
* Retrying is only safe behind a replay guard; see getStreamRetryKind.
|
||||
*/
|
||||
export function isRetryableStreamTransportError(error: unknown): boolean {
|
||||
if (isOpenAIPolicyError(error)) return false
|
||||
@@ -262,43 +275,114 @@ export function isRetryableStreamTransportError(error: unknown): boolean {
|
||||
}
|
||||
|
||||
/**
|
||||
* Decide whether a mid-stream transport disconnect can be recovered by
|
||||
* re-establishing the stream.
|
||||
*
|
||||
* SSE has no resume, so recovery is a full re-send — legitimate only while the
|
||||
* failed attempt is still side-effect-free. That is exactly the boundary
|
||||
* StreamAssistantCommitBuffer tracks: completed thinking/text stay buffered and
|
||||
* are discarded with the attempt, but once a tool_use block completes or
|
||||
* server-side tool activity starts, a re-send could duplicate a tool call
|
||||
* (#766 / inc-4258), so the disconnect must surface to the user instead.
|
||||
*
|
||||
* A watchdog abort is excluded because it has its own retry path with a
|
||||
* narrower safety check, and a user abort is not a fault to recover from.
|
||||
* Error types the local provider proxy (src/server/proxy) writes into a 200
|
||||
* SSE body when it loses the upstream response: `stream_error` when reading or
|
||||
* converting the upstream body failed, `stream_truncated` when the upstream
|
||||
* closed without a terminal finish_reason. Unlike api_error or
|
||||
* invalid_request_error they report the connection, not a provider verdict.
|
||||
*/
|
||||
export function shouldRetryStreamAfterTransportDisconnect(input: {
|
||||
error: unknown
|
||||
hasCrossedSideEffectBoundary: boolean
|
||||
streamIdleAborted: boolean
|
||||
signalAborted: boolean
|
||||
}): boolean {
|
||||
return (
|
||||
!input.hasCrossedSideEffectBoundary &&
|
||||
!input.streamIdleAborted &&
|
||||
!input.signalAborted &&
|
||||
isRetryableStreamTransportError(input.error)
|
||||
)
|
||||
const PROXY_STREAM_TRANSPORT_ERROR_TYPES: ReadonlySet<string> = new Set([
|
||||
'stream_error',
|
||||
'stream_truncated',
|
||||
])
|
||||
|
||||
export function isProxyStreamTransportError(error: unknown): boolean {
|
||||
if (isOpenAIPolicyError(error)) return false
|
||||
// The SDK raises an SSE `error` event as a status-less APIError whose
|
||||
// `error` is the parsed event: { type: 'error', error: { type, message } }.
|
||||
if (!(error instanceof APIError) || error.status !== undefined) return false
|
||||
const event = error.error as { error?: { type?: unknown } } | undefined
|
||||
const type = event?.error?.type
|
||||
return typeof type === 'string' && PROXY_STREAM_TRANSPORT_ERROR_TYPES.has(type)
|
||||
}
|
||||
|
||||
/**
|
||||
* Max times withStreamRetry() re-establishes a stream after a transient
|
||||
* mid-stream error (see RetriableStreamError). Small by default — a malformed
|
||||
* tool_call or a one-off blip usually clears on the first retry; this is not a
|
||||
* capacity backoff loop. Override with CLAUDE_STREAM_TRANSIENT_RETRY_MAX;
|
||||
* values are capped at 5 so a bad environment value cannot make it unbounded.
|
||||
* Decide whether a failed streaming attempt can be discarded and re-sent as a
|
||||
* new stream, and which budget that draws from. null leaves the failure to the
|
||||
* non-streaming fallback or the terminal error path.
|
||||
*
|
||||
* SSE has no resume, so recovery is a full re-send — legitimate only while
|
||||
* `canReplay` holds: nothing from the attempt was committed to the consumer and
|
||||
* no server-side tool work began. Local tool_use blocks stay in
|
||||
* StreamAssistantCommitBuffer until the response completes, so a completed but
|
||||
* uncommitted tool call has not run and a re-send cannot duplicate it
|
||||
* (#766 / inc-4258).
|
||||
*
|
||||
* Truncations the non-streaming fallback already recovers keep going there
|
||||
* when it is available. A response that ended mid-way never did: the fallback
|
||||
* would silently replace output the user already saw, while a stream retry
|
||||
* first retracts it.
|
||||
*/
|
||||
export function getStreamRetryKind(input: {
|
||||
error: unknown
|
||||
canReplay: boolean
|
||||
streamIdleAborted: boolean
|
||||
signalAborted: boolean
|
||||
nonStreamingFallbackAvailable: boolean
|
||||
/** Whether an upstream-reported error may re-send this attempt. */
|
||||
transientRetryAllowed: boolean
|
||||
}): StreamRetryKind | null {
|
||||
const { error } = input
|
||||
if (!input.canReplay || input.signalAborted || isOpenAIPolicyError(error)) {
|
||||
return null
|
||||
}
|
||||
if (input.streamIdleAborted) {
|
||||
return error instanceof StreamWatchdogTimeoutError &&
|
||||
error.safeToRetryStream()
|
||||
? 'watchdog'
|
||||
: null
|
||||
}
|
||||
if (isRetryableStreamTransportError(error)) return 'transport'
|
||||
if (
|
||||
error instanceof StreamEndedEarlyError ||
|
||||
isProxyStreamTransportError(error)
|
||||
) {
|
||||
const fallbackRecovers =
|
||||
input.nonStreamingFallbackAvailable &&
|
||||
!(error instanceof StreamEndedEarlyError && error.reason === 'incomplete')
|
||||
return fallbackRecovers ? null : 'transport'
|
||||
}
|
||||
return input.transientRetryAllowed && isRetryableStreamError(error)
|
||||
? 'transient'
|
||||
: null
|
||||
}
|
||||
|
||||
const DEFAULT_MAX_STREAM_TRANSIENT_RETRIES = 2
|
||||
|
||||
function getStreamTransientRetryOverride(): number | undefined {
|
||||
const raw = parseInt(process.env.CLAUDE_STREAM_TRANSIENT_RETRY_MAX || '', 10)
|
||||
return Number.isFinite(raw) && raw >= 0 ? Math.min(raw, 5) : undefined
|
||||
}
|
||||
|
||||
/**
|
||||
* Max times withStreamRetry() re-establishes a stream after a watchdog stall
|
||||
* or an upstream-reported transient error. Small by default — a malformed
|
||||
* tool_call or a one-off blip usually clears on the first retry, and a stalled
|
||||
* attempt already burned its idle budget. Override with
|
||||
* CLAUDE_STREAM_TRANSIENT_RETRY_MAX; values are capped at 5 so a bad
|
||||
* environment value cannot make it unbounded.
|
||||
*/
|
||||
export function getMaxStreamTransientRetries(): number {
|
||||
const raw = parseInt(process.env.CLAUDE_STREAM_TRANSIENT_RETRY_MAX || '', 10)
|
||||
return Number.isFinite(raw) && raw >= 0 ? Math.min(raw, 5) : 2
|
||||
return getStreamTransientRetryOverride() ?? DEFAULT_MAX_STREAM_TRANSIENT_RETRIES
|
||||
}
|
||||
|
||||
/**
|
||||
* Retry budget for one kind of mid-stream failure. A transport failure is a
|
||||
* request that never got an answer, so it gets the same budget as a stream
|
||||
* that failed to open (CLAUDE_CODE_MAX_RETRIES, default 10); otherwise an
|
||||
* unattended run drops out on the first flaky upstream. An explicit
|
||||
* CLAUDE_STREAM_TRANSIENT_RETRY_MAX also caps it, so 0 still disables every
|
||||
* mid-stream retry.
|
||||
*/
|
||||
export function getMaxStreamRetries(kind: StreamRetryKind): number {
|
||||
if (kind !== 'transport') return getMaxStreamTransientRetries()
|
||||
const configured = getDefaultMaxRetries()
|
||||
const apiRetries =
|
||||
Number.isFinite(configured) && configured >= 0
|
||||
? configured
|
||||
: DEFAULT_MAX_RETRIES
|
||||
const override = getStreamTransientRetryOverride()
|
||||
return override === undefined ? apiRetries : Math.min(override, apiRetries)
|
||||
}
|
||||
|
||||
export async function* withRetry<T>(
|
||||
|
||||
@@ -17,6 +17,9 @@ export type TeammateIdentity = {
|
||||
color?: string
|
||||
planModeRequired: boolean
|
||||
parentSessionId: string // Leader's session ID
|
||||
/** Agent id of the teammate's one durable transcript, kept across turns
|
||||
* and resumes (agentId above is the name@team address). */
|
||||
resumableAgentId?: string
|
||||
}
|
||||
|
||||
export type InProcessTeammateTaskState = TaskStateBase & {
|
||||
|
||||
@@ -9,6 +9,7 @@ import {
|
||||
resolvePersistedAgentType,
|
||||
runAgent,
|
||||
selectInitialTranscriptMessages,
|
||||
selectUnrecordedTranscriptMessages,
|
||||
resolveSubagentEffortValue,
|
||||
resolveSubagentThinkingConfig,
|
||||
} from './runAgent.js'
|
||||
@@ -44,6 +45,41 @@ describe('subagent runtime configuration', () => {
|
||||
})
|
||||
})
|
||||
|
||||
test('appends only what follows the last recorded message of a long-lived transcript', () => {
|
||||
const recorded = createUserMessage({ content: 'old prompt' })
|
||||
const recordedAnswer = createUserMessage({ content: 'old tool result' })
|
||||
const notRecorded = createUserMessage({ content: 'kept in memory only' })
|
||||
const next = createUserMessage({ content: 'new prompt' })
|
||||
|
||||
expect(selectUnrecordedTranscriptMessages(
|
||||
[recorded, notRecorded, recordedAnswer, next],
|
||||
new Set([recorded.uuid, recordedAnswer.uuid]),
|
||||
)).toEqual({
|
||||
messages: [next],
|
||||
startingParentUuid: recordedAnswer.uuid,
|
||||
})
|
||||
// Progress entries are recorded but never chained to
|
||||
const recordedProgress = {
|
||||
type: 'progress',
|
||||
uuid: '00000000-0000-4000-8000-000000000001',
|
||||
} as unknown as Parameters<typeof selectUnrecordedTranscriptMessages>[0][number]
|
||||
expect(selectUnrecordedTranscriptMessages(
|
||||
[recorded, recordedProgress, next],
|
||||
new Set([recorded.uuid, recordedProgress.uuid]),
|
||||
)).toEqual({
|
||||
messages: [next],
|
||||
startingParentUuid: recorded.uuid,
|
||||
})
|
||||
// Nothing recorded yet: a new teammate, or history replaced by compaction
|
||||
expect(selectUnrecordedTranscriptMessages(
|
||||
[notRecorded, next],
|
||||
new Set(),
|
||||
)).toEqual({
|
||||
messages: [notRecorded, next],
|
||||
startingParentUuid: undefined,
|
||||
})
|
||||
})
|
||||
|
||||
test('inherits the parent thinking configuration for regular and fork agents', () => {
|
||||
const disabled = { type: 'disabled' } as const
|
||||
const enabled = { type: 'enabled', budgetTokens: 4096 } as const
|
||||
|
||||
@@ -0,0 +1,188 @@
|
||||
import { afterEach, beforeEach, describe, expect, mock, spyOn, test } from 'bun:test'
|
||||
import type { UUID } from 'crypto'
|
||||
import { mkdtemp, readFile, rm } from 'fs/promises'
|
||||
import { tmpdir } from 'os'
|
||||
import { join } from 'path'
|
||||
import { switchSession } from '../../bootstrap/state.js'
|
||||
import * as queryModule from '../../query.js'
|
||||
import { getDefaultAppState } from '../../state/AppStateStore.js'
|
||||
import type { ToolUseContext } from '../../Tool.js'
|
||||
import { asAgentId, type SessionId } from '../../types/ids.js'
|
||||
import type { Message } from '../../types/message.js'
|
||||
import { createFileStateCacheWithSizeLimit } from '../../utils/fileStateCache.js'
|
||||
import {
|
||||
createAssistantMessage,
|
||||
createUserMessage,
|
||||
} from '../../utils/messages.js'
|
||||
import {
|
||||
flushSessionStorage,
|
||||
getAgentTranscript,
|
||||
getAgentTranscriptPath,
|
||||
readAgentMetadata,
|
||||
resetProjectForTesting,
|
||||
} from '../../utils/sessionStorage.js'
|
||||
import { asSystemPrompt } from '../../utils/systemPromptType.js'
|
||||
import { runAgent } from './runAgent.js'
|
||||
|
||||
const AGENT_ID = asAgentId('aworker-00112233445566aa')
|
||||
const savedEnv = {
|
||||
configDir: process.env.CLAUDE_CONFIG_DIR,
|
||||
persistence: process.env.TEST_ENABLE_SESSION_PERSISTENCE,
|
||||
userType: process.env.USER_TYPE,
|
||||
}
|
||||
let configDir: string
|
||||
|
||||
function restoreEnv(name: string, value: string | undefined): void {
|
||||
if (value === undefined) delete process.env[name]
|
||||
else process.env[name] = value
|
||||
}
|
||||
|
||||
beforeEach(async () => {
|
||||
configDir = await mkdtemp(join(tmpdir(), 'cc-haha-agent-transcript-'))
|
||||
process.env.CLAUDE_CONFIG_DIR = configDir
|
||||
process.env.TEST_ENABLE_SESSION_PERSISTENCE = '1'
|
||||
process.env.USER_TYPE = 'external'
|
||||
switchSession('7a1c4c5e-1d2b-4f3a-9b8c-0d1e2f3a4b5c' as SessionId)
|
||||
resetProjectForTesting()
|
||||
})
|
||||
|
||||
afterEach(async () => {
|
||||
mock.restore()
|
||||
resetProjectForTesting()
|
||||
restoreEnv('CLAUDE_CONFIG_DIR', savedEnv.configDir)
|
||||
restoreEnv('TEST_ENABLE_SESSION_PERSISTENCE', savedEnv.persistence)
|
||||
restoreEnv('USER_TYPE', savedEnv.userType)
|
||||
await rm(configDir, { recursive: true, force: true })
|
||||
})
|
||||
|
||||
function parentContext(): ToolUseContext {
|
||||
const agentDefinition = {
|
||||
agentType: 'worker',
|
||||
whenToUse: 'Teammate',
|
||||
rawSystemPrompt: 'Work.',
|
||||
getSystemPrompt: () => 'Work.',
|
||||
source: 'projectSettings',
|
||||
} as const
|
||||
const state = getDefaultAppState()
|
||||
return {
|
||||
options: {
|
||||
commands: [],
|
||||
debug: false,
|
||||
mainLoopModel: 'sonnet',
|
||||
tools: [],
|
||||
verbose: false,
|
||||
thinkingConfig: { type: 'disabled' as const },
|
||||
mcpClients: [],
|
||||
mcpResources: {},
|
||||
isNonInteractiveSession: true,
|
||||
agentDefinitions: {
|
||||
activeAgents: [agentDefinition],
|
||||
allAgents: [agentDefinition],
|
||||
},
|
||||
},
|
||||
abortController: new AbortController(),
|
||||
readFileState: createFileStateCacheWithSizeLimit(),
|
||||
getAppState: () => state,
|
||||
setAppState: () => {},
|
||||
setResponseLength: () => {},
|
||||
messages: [],
|
||||
} as unknown as ToolUseContext
|
||||
}
|
||||
|
||||
async function runTurn(params: {
|
||||
prompt: Message
|
||||
context?: Message[]
|
||||
answer: Message
|
||||
recordedUuids: Set<UUID>
|
||||
}): Promise<Message[]> {
|
||||
spyOn(queryModule, 'query').mockImplementation(async function* () {
|
||||
yield params.answer
|
||||
} as never)
|
||||
const yielded: Message[] = []
|
||||
for await (const message of runAgent({
|
||||
agentDefinition: {
|
||||
agentType: 'worker',
|
||||
whenToUse: 'Teammate',
|
||||
getSystemPrompt: () => 'Work.',
|
||||
source: 'projectSettings',
|
||||
} as never,
|
||||
promptMessages: [params.prompt],
|
||||
toolUseContext: parentContext(),
|
||||
canUseTool: (async () => ({ behavior: 'allow' })) as never,
|
||||
isAsync: true,
|
||||
forkContextMessages: params.context,
|
||||
querySource: 'agent:custom',
|
||||
override: {
|
||||
userContext: {},
|
||||
systemContext: {},
|
||||
systemPrompt: asSystemPrompt([]),
|
||||
agentId: AGENT_ID,
|
||||
},
|
||||
availableTools: [],
|
||||
recordedUuids: params.recordedUuids,
|
||||
extraMetadata: {
|
||||
taskKind: 'in_process_teammate',
|
||||
teamName: 'team',
|
||||
name: 'worker',
|
||||
color: 'blue',
|
||||
planModeRequired: false,
|
||||
permissionMode: 'default',
|
||||
},
|
||||
})) {
|
||||
yielded.push(message)
|
||||
}
|
||||
return yielded
|
||||
}
|
||||
|
||||
describe('a teammate transcript kept across runs', () => {
|
||||
test('appends each turn once and keeps one parent chain', async () => {
|
||||
const recordedUuids = new Set<UUID>()
|
||||
const firstPrompt = createUserMessage({ content: 'Investigate the failing build' })
|
||||
const firstAnswer = createAssistantMessage({ content: 'The lockfile is stale.' })
|
||||
await runTurn({ prompt: firstPrompt, answer: firstAnswer, recordedUuids })
|
||||
|
||||
const secondPrompt = createUserMessage({ content: 'Regenerate it' })
|
||||
const secondAnswer = createAssistantMessage({ content: 'Regenerated.' })
|
||||
await runTurn({
|
||||
prompt: secondPrompt,
|
||||
// The in-process runner passes the whole conversation back every turn
|
||||
context: [firstPrompt, firstAnswer],
|
||||
answer: secondAnswer,
|
||||
recordedUuids,
|
||||
})
|
||||
await flushSessionStorage()
|
||||
|
||||
const lines = (await readFile(getAgentTranscriptPath(AGENT_ID), 'utf-8'))
|
||||
.split('\n')
|
||||
.filter(line => line.trim())
|
||||
.map(line => JSON.parse(line) as { uuid?: string })
|
||||
const conversation = [firstPrompt, firstAnswer, secondPrompt, secondAnswer]
|
||||
for (const message of conversation) {
|
||||
expect(lines.filter(line => line.uuid === message.uuid)).toHaveLength(1)
|
||||
}
|
||||
expect(
|
||||
(await getAgentTranscript(AGENT_ID))?.messages.map(message => message.uuid),
|
||||
).toEqual(conversation.map(message => message.uuid))
|
||||
expect([...recordedUuids].sort()).toEqual(
|
||||
conversation.map(message => message.uuid).sort(),
|
||||
)
|
||||
})
|
||||
|
||||
test('records the team identity in the metadata sidecar', async () => {
|
||||
await runTurn({
|
||||
prompt: createUserMessage({ content: 'Start' }),
|
||||
answer: createAssistantMessage({ content: 'Started.' }),
|
||||
recordedUuids: new Set(),
|
||||
})
|
||||
|
||||
expect(await readAgentMetadata(AGENT_ID)).toEqual({
|
||||
agentType: 'worker',
|
||||
taskKind: 'in_process_teammate',
|
||||
teamName: 'team',
|
||||
name: 'worker',
|
||||
color: 'blue',
|
||||
planModeRequired: false,
|
||||
permissionMode: 'default',
|
||||
})
|
||||
})
|
||||
})
|
||||
@@ -59,6 +59,7 @@ import { createUserMessage } from '../../utils/messages.js'
|
||||
import { getAgentModel } from '../../utils/model/agent.js'
|
||||
import type { ModelAlias } from '../../utils/model/aliases.js'
|
||||
import {
|
||||
isChainParticipant,
|
||||
recordSidechainTranscript,
|
||||
writeAgentMetadata,
|
||||
type AgentMetadata,
|
||||
@@ -126,6 +127,35 @@ export function selectInitialTranscriptMessages(
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The initial messages an agent run still has to append to a transcript that
|
||||
* already holds `recordedUuids`: everything after the last recorded message,
|
||||
* chained to the last recorded message that can be a parent.
|
||||
*/
|
||||
export function selectUnrecordedTranscriptMessages(
|
||||
messages: Message[],
|
||||
recordedUuids: ReadonlySet<UUID>,
|
||||
): {
|
||||
messages: Message[]
|
||||
startingParentUuid: UUID | null | undefined
|
||||
} {
|
||||
const lastRecorded = messages.findLastIndex(message =>
|
||||
recordedUuids.has(message.uuid),
|
||||
)
|
||||
if (lastRecorded === -1) {
|
||||
return { messages, startingParentUuid: undefined }
|
||||
}
|
||||
const parent = messages
|
||||
.slice(0, lastRecorded + 1)
|
||||
.findLast(
|
||||
message => recordedUuids.has(message.uuid) && isChainParticipant(message),
|
||||
)
|
||||
return {
|
||||
messages: messages.slice(lastRecorded + 1),
|
||||
startingParentUuid: parent?.uuid ?? null,
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Initialize agent-specific MCP servers
|
||||
* Agents can define their own MCP servers in their frontmatter that are additive
|
||||
@@ -314,6 +344,8 @@ export async function* runAgent({
|
||||
streamTargetAgentId,
|
||||
persistedAgentType,
|
||||
alreadyPersistedMessageCount,
|
||||
recordedUuids,
|
||||
extraMetadata,
|
||||
workflow,
|
||||
onQueryProgress,
|
||||
}: {
|
||||
@@ -384,6 +416,14 @@ export async function* runAgent({
|
||||
* transcript. Resume sends them to the model for context but must only
|
||||
* append the new continuation messages to disk. */
|
||||
alreadyPersistedMessageCount?: number
|
||||
/** Uuids of the messages already in this agent's transcript, for an agent
|
||||
* that keeps one transcript across many runs (an in-process teammate).
|
||||
* Only messages after the last recorded one are appended, and every message
|
||||
* this run records is added. Takes precedence over
|
||||
* alreadyPersistedMessageCount. */
|
||||
recordedUuids?: Set<UUID>
|
||||
/** Additional fields for this run's metadata sidecar. */
|
||||
extraMetadata?: Partial<AgentMetadata>
|
||||
/** Optional subdirectory under subagents/ to group this agent's transcript
|
||||
* with related ones (e.g. workflows/<runId> for workflow subagents). */
|
||||
/** Set when this agent is one step of a dynamic workflow run. */
|
||||
@@ -827,10 +867,12 @@ export async function* runAgent({
|
||||
// Record initial messages before the query loop starts, plus the agentType
|
||||
// so resume can route correctly when subagent_type is omitted. Both writes
|
||||
// are fire-and-forget — persistence failure shouldn't block the agent.
|
||||
const initialTranscriptWrite = selectInitialTranscriptMessages(
|
||||
initialMessages,
|
||||
alreadyPersistedMessageCount,
|
||||
)
|
||||
const initialTranscriptWrite = recordedUuids
|
||||
? selectUnrecordedTranscriptMessages(initialMessages, recordedUuids)
|
||||
: selectInitialTranscriptMessages(
|
||||
initialMessages,
|
||||
alreadyPersistedMessageCount,
|
||||
)
|
||||
void recordSidechainTranscript(
|
||||
initialTranscriptWrite.messages,
|
||||
agentId,
|
||||
@@ -838,6 +880,9 @@ export async function* runAgent({
|
||||
).catch(_err =>
|
||||
logForDebugging(`Failed to record sidechain transcript: ${_err}`),
|
||||
)
|
||||
for (const message of initialTranscriptWrite.messages) {
|
||||
recordedUuids?.add(message.uuid)
|
||||
}
|
||||
void writeAgentMetadata(agentId, {
|
||||
agentType: resolvePersistedAgentType(
|
||||
persistedAgentType,
|
||||
@@ -849,6 +894,7 @@ export async function* runAgent({
|
||||
...(spawningToolUseId && { toolUseId: spawningToolUseId }),
|
||||
...(ownerAgentId && { ownerAgentId }),
|
||||
...(workflow && { workflow }),
|
||||
...extraMetadata,
|
||||
}).catch(_err => logForDebugging(`Failed to write agent metadata: ${_err}`))
|
||||
|
||||
// Track the last recorded message UUID for parent chain continuity
|
||||
@@ -920,6 +966,7 @@ export async function* runAgent({
|
||||
).catch(err =>
|
||||
logForDebugging(`Failed to record sidechain transcript: ${err}`),
|
||||
)
|
||||
recordedUuids?.add(message.uuid)
|
||||
if (message.type !== 'progress') {
|
||||
lastRecordedUuid = message.uuid
|
||||
}
|
||||
|
||||
@@ -0,0 +1,247 @@
|
||||
import { afterEach, beforeEach, describe, expect, spyOn, test } from 'bun:test'
|
||||
import { mkdir, mkdtemp, rm, writeFile } from 'fs/promises'
|
||||
import { tmpdir } from 'os'
|
||||
import { join } from 'path'
|
||||
import type { AppState } from '../../state/AppState.js'
|
||||
import type { ToolUseContext } from '../../Tool.js'
|
||||
import * as gracefulShutdownModule from '../../utils/gracefulShutdown.js'
|
||||
import * as lockfile from '../../utils/lockfile.js'
|
||||
import {
|
||||
clearDynamicTeamContext,
|
||||
setDynamicTeamContext,
|
||||
} from '../../utils/teammate.js'
|
||||
import * as mailbox from '../../utils/teammateMailbox.js'
|
||||
import { SendMessageTool } from './SendMessageTool.js'
|
||||
|
||||
const TEAM = 'send-team'
|
||||
|
||||
let configDir: string
|
||||
let originalConfigDir: string | undefined
|
||||
|
||||
beforeEach(async () => {
|
||||
originalConfigDir = process.env.CLAUDE_CONFIG_DIR
|
||||
configDir = await mkdtemp(join(tmpdir(), 'cc-haha-send-message-'))
|
||||
process.env.CLAUDE_CONFIG_DIR = configDir
|
||||
})
|
||||
|
||||
afterEach(async () => {
|
||||
clearDynamicTeamContext()
|
||||
if (originalConfigDir === undefined) delete process.env.CLAUDE_CONFIG_DIR
|
||||
else process.env.CLAUDE_CONFIG_DIR = originalConfigDir
|
||||
await rm(configDir, { recursive: true, force: true })
|
||||
})
|
||||
|
||||
function toolContext(): ToolUseContext {
|
||||
const appState = {
|
||||
teamContext: {
|
||||
teamName: TEAM,
|
||||
leadAgentId: `team-lead@${TEAM}`,
|
||||
teammates: {
|
||||
[`team-lead@${TEAM}`]: { name: 'team-lead' },
|
||||
[`worker@${TEAM}`]: { name: 'worker' },
|
||||
[`reviewer@${TEAM}`]: { name: 'reviewer' },
|
||||
},
|
||||
},
|
||||
agentNameRegistry: new Map(),
|
||||
tasks: {},
|
||||
toolPermissionContext: { mode: 'default' },
|
||||
} as unknown as AppState
|
||||
return {
|
||||
getAppState: () => appState,
|
||||
setAppState: () => {},
|
||||
abortController: new AbortController(),
|
||||
} as unknown as ToolUseContext
|
||||
}
|
||||
|
||||
async function call(input: unknown): Promise<{ success: boolean } & Record<string, unknown>> {
|
||||
const result = await SendMessageTool.call(
|
||||
input as never,
|
||||
toolContext(),
|
||||
undefined as never,
|
||||
undefined as never,
|
||||
)
|
||||
return result.data as { success: boolean } & Record<string, unknown>
|
||||
}
|
||||
|
||||
describe('SendMessage when a mailbox write fails', () => {
|
||||
test('reports a held inbox lock instead of claiming the message was sent', async () => {
|
||||
expect(
|
||||
await mailbox.writeToMailbox(
|
||||
'worker',
|
||||
{ from: 'team-lead', text: 'seed', timestamp: new Date().toISOString() },
|
||||
TEAM,
|
||||
),
|
||||
).toBe(true)
|
||||
const inboxPath = mailbox.getInboxPath('worker', TEAM)
|
||||
const release = await lockfile.lock(inboxPath, {
|
||||
lockfilePath: `${inboxPath}.lock`,
|
||||
})
|
||||
|
||||
let data: Awaited<ReturnType<typeof call>>
|
||||
try {
|
||||
data = await call({
|
||||
to: 'worker',
|
||||
summary: 'rerun tests',
|
||||
message: 'Please rerun the tests',
|
||||
})
|
||||
} finally {
|
||||
await release()
|
||||
}
|
||||
|
||||
expect(data).toEqual({
|
||||
success: false,
|
||||
message: "Failed to write to worker's inbox — nothing was sent. Try again.",
|
||||
})
|
||||
expect((await mailbox.readMailbox('worker', TEAM)).map(m => m.text)).toEqual([
|
||||
'seed',
|
||||
])
|
||||
}, 15_000)
|
||||
|
||||
test('names the broadcast recipients that did not receive the message', async () => {
|
||||
const teamDir = join(configDir, 'teams', TEAM)
|
||||
await mkdir(teamDir, { recursive: true })
|
||||
await writeFile(
|
||||
join(teamDir, 'config.json'),
|
||||
JSON.stringify({
|
||||
name: TEAM,
|
||||
createdAt: 1,
|
||||
leadAgentId: `team-lead@${TEAM}`,
|
||||
members: ['team-lead', 'worker', 'reviewer'].map(name => ({
|
||||
agentId: `${name}@${TEAM}`,
|
||||
name,
|
||||
joinedAt: 1,
|
||||
tmuxPaneId: '',
|
||||
cwd: configDir,
|
||||
subscriptions: [],
|
||||
})),
|
||||
}),
|
||||
)
|
||||
const write = spyOn(mailbox, 'writeToMailbox').mockImplementation(
|
||||
async recipient => recipient !== 'reviewer',
|
||||
)
|
||||
|
||||
try {
|
||||
const data = await call({
|
||||
to: '*',
|
||||
summary: 'status',
|
||||
message: 'Status check',
|
||||
})
|
||||
|
||||
expect(data).toMatchObject({
|
||||
success: false,
|
||||
recipients: ['worker'],
|
||||
})
|
||||
expect(data.message).toContain('1 of 2 teammate(s): worker')
|
||||
expect(data.message).toContain('inbox of reviewer')
|
||||
} finally {
|
||||
write.mockRestore()
|
||||
}
|
||||
})
|
||||
|
||||
test('reports every structured protocol message that could not be written', async () => {
|
||||
const write = spyOn(mailbox, 'writeToMailbox').mockResolvedValue(false)
|
||||
|
||||
try {
|
||||
expect(
|
||||
await call({
|
||||
to: 'worker',
|
||||
message: { type: 'shutdown_request', reason: 'work is done' },
|
||||
}),
|
||||
).toMatchObject({
|
||||
success: false,
|
||||
message:
|
||||
"Failed to write the shutdown request to worker's inbox — nothing was sent. Try again.",
|
||||
target: 'worker',
|
||||
})
|
||||
expect(
|
||||
await call({
|
||||
to: 'worker',
|
||||
message: {
|
||||
type: 'plan_approval_response',
|
||||
request_id: 'plan-1',
|
||||
approve: true,
|
||||
},
|
||||
}),
|
||||
).toEqual({
|
||||
success: false,
|
||||
message:
|
||||
"Failed to write the plan approval to worker's inbox — nothing was sent. Try again.",
|
||||
request_id: 'plan-1',
|
||||
})
|
||||
expect(
|
||||
await call({
|
||||
to: 'worker',
|
||||
message: {
|
||||
type: 'plan_approval_response',
|
||||
request_id: 'plan-2',
|
||||
approve: false,
|
||||
feedback: 'split the migration',
|
||||
},
|
||||
}),
|
||||
).toEqual({
|
||||
success: false,
|
||||
message:
|
||||
"Failed to write the plan rejection to worker's inbox — nothing was sent. Try again.",
|
||||
request_id: 'plan-2',
|
||||
})
|
||||
|
||||
setDynamicTeamContext({
|
||||
agentId: `worker@${TEAM}`,
|
||||
agentName: 'worker',
|
||||
teamName: TEAM,
|
||||
planModeRequired: false,
|
||||
})
|
||||
expect(
|
||||
await call({
|
||||
to: 'team-lead',
|
||||
message: {
|
||||
type: 'shutdown_response',
|
||||
request_id: 'shutdown-2',
|
||||
approve: false,
|
||||
reason: 'still migrating',
|
||||
},
|
||||
}),
|
||||
).toEqual({
|
||||
success: false,
|
||||
message:
|
||||
"Failed to write the shutdown rejection to team-lead's inbox — nothing was sent. Try again.",
|
||||
request_id: 'shutdown-2',
|
||||
})
|
||||
} finally {
|
||||
write.mockRestore()
|
||||
}
|
||||
})
|
||||
|
||||
test('keeps a teammate running when its shutdown approval cannot reach the lead', async () => {
|
||||
setDynamicTeamContext({
|
||||
agentId: `worker@${TEAM}`,
|
||||
agentName: 'worker',
|
||||
teamName: TEAM,
|
||||
planModeRequired: false,
|
||||
})
|
||||
const write = spyOn(mailbox, 'writeToMailbox').mockResolvedValue(false)
|
||||
const shutdown = spyOn(
|
||||
gracefulShutdownModule,
|
||||
'gracefulShutdown',
|
||||
).mockResolvedValue(undefined as never)
|
||||
|
||||
try {
|
||||
const data = await call({
|
||||
to: 'team-lead',
|
||||
message: {
|
||||
type: 'shutdown_response',
|
||||
request_id: 'shutdown-1',
|
||||
approve: true,
|
||||
},
|
||||
})
|
||||
await new Promise<void>(resolve => setImmediate(resolve))
|
||||
|
||||
expect(data).toMatchObject({ success: false, request_id: 'shutdown-1' })
|
||||
expect(data.message).toContain('worker is still running')
|
||||
expect(shutdown).not.toHaveBeenCalled()
|
||||
} finally {
|
||||
write.mockRestore()
|
||||
shutdown.mockRestore()
|
||||
}
|
||||
})
|
||||
})
|
||||
@@ -0,0 +1,360 @@
|
||||
import { afterEach, beforeEach, describe, expect, mock, spyOn, test } from 'bun:test'
|
||||
import { mkdtemp, rm, writeFile } from 'fs/promises'
|
||||
import { tmpdir } from 'os'
|
||||
import { join } from 'path'
|
||||
import {
|
||||
resetStateForTests,
|
||||
setIsInteractive,
|
||||
switchSession,
|
||||
} from '../../bootstrap/state.js'
|
||||
import * as promptsModule from '../../constants/prompts.js'
|
||||
import type { AppState } from '../../state/AppState.js'
|
||||
import type { ToolUseContext } from '../../Tool.js'
|
||||
import type { InProcessTeammateTaskState } from '../../tasks/InProcessTeammateTask/types.js'
|
||||
import { asAgentId, type SessionId } from '../../types/ids.js'
|
||||
import type { Message } from '../../types/message.js'
|
||||
import { createFileStateCacheWithSizeLimit } from '../../utils/fileStateCache.js'
|
||||
import {
|
||||
createAssistantMessage,
|
||||
createUserMessage,
|
||||
} from '../../utils/messages.js'
|
||||
import { drainSdkEvents } from '../../utils/sdkEventQueue.js'
|
||||
import {
|
||||
flushSessionStorage,
|
||||
recordSidechainTranscript,
|
||||
resetProjectForTesting,
|
||||
} from '../../utils/sessionStorage.js'
|
||||
import { teamPlanRecordSchema } from '../../shared/teamPlan.js'
|
||||
import {
|
||||
getTeamDir,
|
||||
readTeamFile,
|
||||
removeMemberByAgentId,
|
||||
type TeamFile,
|
||||
writeTeamFileAsync,
|
||||
} from '../../utils/swarm/teamHelpers.js'
|
||||
import { readMailbox } from '../../utils/teammateMailbox.js'
|
||||
import * as runAgentModule from '../AgentTool/runAgent.js'
|
||||
import { spawnTeammate } from '../shared/spawnMultiAgent.js'
|
||||
import { SendMessageTool } from './SendMessageTool.js'
|
||||
|
||||
const TEAM = 'msg-team'
|
||||
type RunAgentInput = Parameters<typeof runAgentModule.runAgent>[0]
|
||||
|
||||
const savedEnv = {
|
||||
configDir: process.env.CLAUDE_CONFIG_DIR,
|
||||
persistence: process.env.TEST_ENABLE_SESSION_PERSISTENCE,
|
||||
userType: process.env.USER_TYPE,
|
||||
}
|
||||
let configDir: string
|
||||
|
||||
function restoreEnv(name: string, value: string | undefined): void {
|
||||
if (value === undefined) delete process.env[name]
|
||||
else process.env[name] = value
|
||||
}
|
||||
|
||||
async function writeTeam(members: TeamFile['members']): Promise<void> {
|
||||
await writeTeamFileAsync(TEAM, {
|
||||
name: TEAM,
|
||||
createdAt: Date.now(),
|
||||
leadAgentId: `team-lead@${TEAM}`,
|
||||
members,
|
||||
})
|
||||
}
|
||||
|
||||
/** A desktop team plan whose members have no roster entry yet. */
|
||||
async function writeTeamPlan(state: string, memberNames: string[]): Promise<void> {
|
||||
const runtime = { providerId: 'claude-official', modelId: 'claude-sonnet-4-6' }
|
||||
const plan = teamPlanRecordSchema.parse({
|
||||
schemaVersion: 1,
|
||||
planId: 'plan-1',
|
||||
sessionId: 'lead-session',
|
||||
teamName: TEAM,
|
||||
incarnationId: 'incarnation-1',
|
||||
revision: 1,
|
||||
state,
|
||||
workDir: configDir,
|
||||
leaderRuntime: runtime,
|
||||
members: memberNames.map(name => ({
|
||||
id: name,
|
||||
name,
|
||||
agentType: 'general-purpose',
|
||||
prompt: `Work as ${name}`,
|
||||
runtime,
|
||||
})),
|
||||
tasks: [],
|
||||
createdAt: Date.now(),
|
||||
updatedAt: Date.now(),
|
||||
})
|
||||
await writeFile(join(getTeamDir(TEAM), 'plan.json'), JSON.stringify(plan))
|
||||
}
|
||||
|
||||
function member(
|
||||
name: string,
|
||||
extra: Partial<TeamFile['members'][number]> = {},
|
||||
): TeamFile['members'][number] {
|
||||
return {
|
||||
agentId: `${name}@${TEAM}`,
|
||||
name,
|
||||
joinedAt: Date.now(),
|
||||
tmuxPaneId: '',
|
||||
cwd: configDir,
|
||||
subscriptions: [],
|
||||
...extra,
|
||||
}
|
||||
}
|
||||
|
||||
beforeEach(async () => {
|
||||
configDir = await mkdtemp(join(tmpdir(), 'cc-haha-send-teammates-'))
|
||||
process.env.CLAUDE_CONFIG_DIR = configDir
|
||||
process.env.TEST_ENABLE_SESSION_PERSISTENCE = '1'
|
||||
process.env.USER_TYPE = 'external'
|
||||
resetStateForTests()
|
||||
setIsInteractive(false)
|
||||
switchSession('3c9d5a1e-6b7f-4c2d-8e9f-0a1b2c3d4e5f' as SessionId)
|
||||
resetProjectForTesting()
|
||||
spyOn(promptsModule, 'getSystemPrompt').mockResolvedValue(['System prompt'])
|
||||
})
|
||||
|
||||
afterEach(async () => {
|
||||
mock.restore()
|
||||
drainSdkEvents()
|
||||
resetProjectForTesting()
|
||||
resetStateForTests()
|
||||
restoreEnv('CLAUDE_CONFIG_DIR', savedEnv.configDir)
|
||||
restoreEnv('TEST_ENABLE_SESSION_PERSISTENCE', savedEnv.persistence)
|
||||
restoreEnv('USER_TYPE', savedEnv.userType)
|
||||
await rm(configDir, { recursive: true, force: true })
|
||||
})
|
||||
|
||||
function leadContext() {
|
||||
let state = {
|
||||
mainLoopModel: 'claude-sonnet-4-6',
|
||||
toolPermissionContext: { mode: 'default' },
|
||||
teamContext: {
|
||||
teamName: TEAM,
|
||||
teamFilePath: '',
|
||||
leadAgentId: `team-lead@${TEAM}`,
|
||||
teammates: {},
|
||||
},
|
||||
agentNameRegistry: new Map(),
|
||||
tasks: {},
|
||||
} as unknown as AppState
|
||||
const context = {
|
||||
getAppState: () => state,
|
||||
setAppState: (updater: (previous: AppState) => AppState) => {
|
||||
state = updater(state)
|
||||
},
|
||||
options: {
|
||||
agentDefinitions: { activeAgents: [], allAgents: [] },
|
||||
tools: [],
|
||||
mainLoopModel: 'claude-sonnet-4-6',
|
||||
mcpClients: [],
|
||||
thinkingConfig: { type: 'disabled' },
|
||||
},
|
||||
abortController: new AbortController(),
|
||||
readFileState: createFileStateCacheWithSizeLimit(10),
|
||||
messages: [],
|
||||
} as unknown as ToolUseContext
|
||||
const teammates = () =>
|
||||
Object.values(state.tasks).filter(
|
||||
(task): task is InProcessTeammateTaskState =>
|
||||
task.type === 'in_process_teammate',
|
||||
)
|
||||
return { context, teammates }
|
||||
}
|
||||
|
||||
async function waitFor(predicate: () => boolean): Promise<void> {
|
||||
for (let attempt = 0; attempt < 300; attempt++) {
|
||||
if (predicate()) return
|
||||
await new Promise(resolve => setTimeout(resolve, 10))
|
||||
}
|
||||
throw new Error('condition not reached')
|
||||
}
|
||||
|
||||
async function send(
|
||||
context: ToolUseContext,
|
||||
to: string,
|
||||
message: string,
|
||||
): Promise<{ success: boolean; message: string }> {
|
||||
const result = await SendMessageTool.call(
|
||||
{ to, summary: 'next step', message } as never,
|
||||
context,
|
||||
undefined as never,
|
||||
undefined as never,
|
||||
)
|
||||
return result.data as { success: boolean; message: string }
|
||||
}
|
||||
|
||||
/** Spawns `worker` as an in-process teammate whose runner then stops. */
|
||||
async function spawnAndStopWorker(lead: ReturnType<typeof leadContext>) {
|
||||
const calls: RunAgentInput[] = []
|
||||
spyOn(runAgentModule, 'runAgent').mockImplementation(async function* (
|
||||
input: RunAgentInput,
|
||||
) {
|
||||
calls.push(input)
|
||||
// Every run ends its teammate after one turn
|
||||
for (const task of lead.teammates()) task.abortController?.abort()
|
||||
yield* []
|
||||
})
|
||||
await spawnTeammate(
|
||||
{ name: 'worker', prompt: 'Fix the parser', team_name: TEAM },
|
||||
lead.context,
|
||||
)
|
||||
await waitFor(() => calls.length === 1 && lead.teammates().length === 0)
|
||||
return calls
|
||||
}
|
||||
|
||||
describe('SendMessage to an in-process teammate that is not running', () => {
|
||||
test('resumes it from its transcript with the message as its next prompt', async () => {
|
||||
await writeTeam([])
|
||||
const lead = leadContext()
|
||||
const calls = await spawnAndStopWorker(lead)
|
||||
const transcriptId = calls[0]!.override?.agentId
|
||||
// What the teammate's runs recorded before it stopped
|
||||
const earlier: Message[] = [
|
||||
createUserMessage({ content: 'Fix the parser' }),
|
||||
createAssistantMessage({ content: 'Fixed the parser.' }),
|
||||
]
|
||||
if (transcriptId) {
|
||||
await recordSidechainTranscript(earlier, transcriptId)
|
||||
await flushSessionStorage()
|
||||
}
|
||||
// It also left the roster, as a teammate that approved shutdown does
|
||||
await removeMemberByAgentId(TEAM, `worker@${TEAM}`)
|
||||
|
||||
const result = await send(lead.context, 'worker', 'Please update the docs too')
|
||||
|
||||
expect(result).toEqual({
|
||||
success: true,
|
||||
message:
|
||||
'Teammate "worker" was not running; resumed it as an in-process teammate with 2 prior messages and your message as its next prompt.',
|
||||
})
|
||||
expect(transcriptId).toBeString()
|
||||
await waitFor(() => calls.length === 2)
|
||||
const resumed = calls[1]!
|
||||
expect(resumed.override?.agentId).toBe(asAgentId(transcriptId!))
|
||||
expect(resumed.forkContextMessages?.map(message => message.uuid)).toEqual(
|
||||
earlier.map(message => message.uuid),
|
||||
)
|
||||
const prompt = resumed.promptMessages[0]!.message.content as string
|
||||
expect(prompt).toContain('Please update the docs too')
|
||||
expect(prompt).toContain('teammate_id="team-lead"')
|
||||
expect(readTeamFile(TEAM)?.members).toContainEqual(
|
||||
expect.objectContaining({ agentId: `worker@${TEAM}`, backendType: 'in-process' }),
|
||||
)
|
||||
// Delivered as the prompt, not left in an inbox nobody reads
|
||||
expect((await readMailbox('worker', TEAM)).filter(m => !m.read)).toEqual([])
|
||||
await waitFor(() => lead.teammates().length === 0)
|
||||
})
|
||||
|
||||
test('resumes only once when two messages race, and keeps the second for its next turn', async () => {
|
||||
await writeTeam([])
|
||||
const lead = leadContext()
|
||||
const calls = await spawnAndStopWorker(lead)
|
||||
// Keep the resumed teammate alive so the second message has a live target
|
||||
let releaseResumed: (() => void) | undefined
|
||||
spyOn(runAgentModule, 'runAgent').mockImplementation(async function* (
|
||||
input: RunAgentInput,
|
||||
) {
|
||||
calls.push(input)
|
||||
await new Promise<void>(resolve => {
|
||||
releaseResumed = resolve
|
||||
})
|
||||
for (const task of lead.teammates()) task.abortController?.abort()
|
||||
yield* []
|
||||
})
|
||||
|
||||
const results = await Promise.all([
|
||||
send(lead.context, 'worker', 'First follow-up'),
|
||||
send(lead.context, 'worker', 'Second follow-up'),
|
||||
])
|
||||
|
||||
const resumedResults = results.filter(result =>
|
||||
result.message.includes('was not running; resumed it'),
|
||||
)
|
||||
expect(resumedResults).toHaveLength(1)
|
||||
expect(results.every(result => result.success)).toBe(true)
|
||||
expect(lead.teammates().filter(task => task.status === 'running')).toHaveLength(1)
|
||||
const queued = (await readMailbox('worker', TEAM)).filter(m => !m.read)
|
||||
expect(queued).toHaveLength(1)
|
||||
await waitFor(() => releaseResumed !== undefined)
|
||||
releaseResumed!()
|
||||
await waitFor(() => lead.teammates().length === 0)
|
||||
})
|
||||
})
|
||||
|
||||
describe('SendMessage to other recipients', () => {
|
||||
test('rejects a name that is not on the team instead of writing an orphan inbox', async () => {
|
||||
await writeTeam([member('worker', { backendType: 'tmux' })])
|
||||
const lead = leadContext()
|
||||
|
||||
expect(await send(lead.context, 'ghost', 'Anyone there?')).toEqual({
|
||||
success: false,
|
||||
message:
|
||||
"No teammate named 'ghost' is currently on team 'msg-team'. Spawn one with the Agent tool (name: 'ghost') first.",
|
||||
})
|
||||
expect(await readMailbox('ghost', TEAM)).toEqual([])
|
||||
})
|
||||
|
||||
test('queues a message for a member whose team plan awaits approval', async () => {
|
||||
await writeTeam([])
|
||||
await writeTeamPlan('review_pending', ['reviewer'])
|
||||
const lead = leadContext()
|
||||
|
||||
expect(await send(lead.context, 'reviewer', 'Start with the parser')).toMatchObject({
|
||||
success: true,
|
||||
message:
|
||||
'reviewer has not started yet; it reads your message once the user approves the team plan.',
|
||||
})
|
||||
expect((await readMailbox('reviewer', TEAM)).map(m => m.text)).toEqual([
|
||||
'Start with the parser',
|
||||
])
|
||||
})
|
||||
|
||||
test('rejects a member of a team plan that was cancelled', async () => {
|
||||
await writeTeam([])
|
||||
await writeTeamPlan('cancelled', ['reviewer'])
|
||||
const lead = leadContext()
|
||||
|
||||
expect(await send(lead.context, 'reviewer', 'Start with the parser')).toMatchObject({
|
||||
success: false,
|
||||
})
|
||||
expect(await readMailbox('reviewer', TEAM)).toEqual([])
|
||||
})
|
||||
|
||||
test('tells the lead a stopped desktop member is restarting to read the message', async () => {
|
||||
await writeTeam([member('builder', { backendType: 'process', terminated: true })])
|
||||
const lead = leadContext()
|
||||
|
||||
const result = await send(lead.context, 'builder', 'Rebuild with the fix')
|
||||
|
||||
expect(result).toMatchObject({
|
||||
success: true,
|
||||
message:
|
||||
'builder was stopped; it is restarting from its saved conversation to read your message.',
|
||||
})
|
||||
expect((await readMailbox('builder', TEAM)).map(m => m.text)).toEqual([
|
||||
'Rebuild with the fix',
|
||||
])
|
||||
// The server restarts process members; they never run in-process here
|
||||
expect(lead.teammates()).toEqual([])
|
||||
})
|
||||
|
||||
test('writes to a running desktop member and a pane teammate as before', async () => {
|
||||
await writeTeam([
|
||||
member('builder', { backendType: 'process' }),
|
||||
member('painter', { backendType: 'tmux' }),
|
||||
])
|
||||
const lead = leadContext()
|
||||
|
||||
expect(await send(lead.context, 'builder', 'Status?')).toMatchObject({
|
||||
success: true,
|
||||
message: "Message sent to builder's inbox",
|
||||
})
|
||||
expect(await send(lead.context, 'painter', 'Status?')).toMatchObject({
|
||||
success: true,
|
||||
message: "Message sent to painter's inbox",
|
||||
})
|
||||
expect(lead.teammates()).toEqual([])
|
||||
})
|
||||
})
|
||||
@@ -11,7 +11,7 @@ import {
|
||||
} from '../../tasks/LocalAgentTask/LocalAgentTask.js'
|
||||
import { isMainSessionTask } from '../../tasks/LocalMainSessionTask.js'
|
||||
import { toAgentId } from '../../types/ids.js'
|
||||
import { generateRequestId } from '../../utils/agentId.js'
|
||||
import { formatAgentId, generateRequestId } from '../../utils/agentId.js'
|
||||
import { isAgentSwarmsEnabled } from '../../utils/agentSwarmsEnabled.js'
|
||||
import { logForDebugging } from '../../utils/debug.js'
|
||||
import { errorMessage } from '../../utils/errors.js'
|
||||
@@ -23,12 +23,18 @@ import { semanticBoolean } from '../../utils/semanticBoolean.js'
|
||||
import { jsonStringify } from '../../utils/slowOperations.js'
|
||||
import type { BackendType } from '../../utils/swarm/backends/types.js'
|
||||
import { TEAM_LEAD_NAME } from '../../utils/swarm/constants.js'
|
||||
import {
|
||||
hasResumableInProcessTeammate,
|
||||
resumeInProcessTeammate,
|
||||
} from '../../utils/swarm/inProcessRunner.js'
|
||||
import { readTeamFileAsync } from '../../utils/swarm/teamHelpers.js'
|
||||
import { readTeamPlan } from '../../utils/swarm/teamPlanStore.js'
|
||||
import {
|
||||
getAgentId,
|
||||
getAgentName,
|
||||
getTeammateColor,
|
||||
getTeamName,
|
||||
isInProcessTeammate,
|
||||
isTeamLead,
|
||||
isTeammate,
|
||||
} from '../../utils/teammate.js'
|
||||
@@ -160,7 +166,7 @@ async function handleMessage(
|
||||
getAgentName() || (isTeammate() ? 'teammate' : TEAM_LEAD_NAME)
|
||||
const senderColor = getTeammateColor()
|
||||
|
||||
await writeToMailbox(
|
||||
const written = await writeToMailbox(
|
||||
recipientName,
|
||||
{
|
||||
from: senderName,
|
||||
@@ -171,6 +177,14 @@ async function handleMessage(
|
||||
},
|
||||
teamName,
|
||||
)
|
||||
if (!written) {
|
||||
return {
|
||||
data: {
|
||||
success: false,
|
||||
message: `Failed to write to ${recipientName}'s inbox — nothing was sent. Try again.`,
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
const recipientColor = findTeammateColor(appState, recipientName)
|
||||
|
||||
@@ -190,6 +204,163 @@ async function handleMessage(
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Delivers a plain message to one teammate by name.
|
||||
*
|
||||
* Mail for an in-process teammate whose runner is gone -- it failed,
|
||||
* finished, was stopped or evicted -- would sit in an inbox nothing reads, so
|
||||
* the teammate is resumed from its transcript with the message as its next
|
||||
* prompt. A desktop process member needs only its inbox: the server restarts
|
||||
* a stopped member when mail arrives. A name that is on neither the team's
|
||||
* roster nor this process's teammates is an error, not an orphan inbox.
|
||||
*/
|
||||
async function handleTeammateMessage(
|
||||
recipientName: string,
|
||||
content: string,
|
||||
summary: string | undefined,
|
||||
context: ToolUseContext,
|
||||
): Promise<{ data: MessageOutput }> {
|
||||
const appState = context.getAppState()
|
||||
const teamName = getTeamName(appState.teamContext)
|
||||
if (!teamName || recipientName === TEAM_LEAD_NAME) {
|
||||
return handleMessage(recipientName, content, summary, context)
|
||||
}
|
||||
|
||||
const task = findTeammateTaskByAgentId(
|
||||
formatAgentId(recipientName, teamName),
|
||||
appState.tasks,
|
||||
)
|
||||
if (task?.status === 'running') {
|
||||
return handleMessage(recipientName, content, summary, context)
|
||||
}
|
||||
|
||||
const teamFile = await readTeamFileAsync(teamName)
|
||||
const member = teamFile?.members?.find(entry => entry.name === recipientName)
|
||||
if (member?.backendType === 'process') {
|
||||
const sent = await handleMessage(recipientName, content, summary, context)
|
||||
if (!sent.data.success || member.terminated !== true) return sent
|
||||
return {
|
||||
data: {
|
||||
...sent.data,
|
||||
message: `${recipientName} was stopped; it is restarting from its saved conversation to read your message.`,
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
// In-process teammates live in the lead's process: a pane teammate or a
|
||||
// desktop worker cannot host one.
|
||||
const hostsInProcessTeammates = !isTeammate() || isInProcessTeammate()
|
||||
const isStoppedInProcessTeammate =
|
||||
task !== undefined ||
|
||||
member?.backendType === 'in-process' ||
|
||||
hasResumableInProcessTeammate(recipientName, teamName, teamFile?.createdAt)
|
||||
if (hostsInProcessTeammates && isStoppedInProcessTeammate) {
|
||||
return resumeTeammateWithMessage(
|
||||
recipientName,
|
||||
teamName,
|
||||
content,
|
||||
summary,
|
||||
context,
|
||||
)
|
||||
}
|
||||
|
||||
const onRoster =
|
||||
member !== undefined ||
|
||||
Object.values(appState.teamContext?.teammates ?? {}).some(
|
||||
teammate => teammate.name === recipientName,
|
||||
)
|
||||
// Without a readable team file the name cannot be checked
|
||||
if (onRoster || !teamFile) {
|
||||
return handleMessage(recipientName, content, summary, context)
|
||||
}
|
||||
// A member of a desktop team plan awaiting the user's approval has no
|
||||
// roster entry yet; the server delivers its inbox once the plan launches.
|
||||
if (await isPendingTeamPlanMember(teamName, recipientName)) {
|
||||
const queued = await handleMessage(recipientName, content, summary, context)
|
||||
if (!queued.data.success) return queued
|
||||
return {
|
||||
data: {
|
||||
...queued.data,
|
||||
message: `${recipientName} has not started yet; it reads your message once the user approves the team plan.`,
|
||||
},
|
||||
}
|
||||
}
|
||||
return {
|
||||
data: {
|
||||
success: false,
|
||||
message: isTeammate()
|
||||
? `No teammate named '${recipientName}' is currently on team '${teamName}'. Message the lead to spawn one.`
|
||||
: `No teammate named '${recipientName}' is currently on team '${teamName}'. Spawn one with the Agent tool (name: '${recipientName}') first.`,
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
const PENDING_TEAM_PLAN_STATES = new Set(['draft', 'review_pending', 'launching'])
|
||||
|
||||
async function isPendingTeamPlanMember(
|
||||
teamName: string,
|
||||
name: string,
|
||||
): Promise<boolean> {
|
||||
const plan = await readTeamPlan(teamName).catch(() => null)
|
||||
return (
|
||||
!!plan &&
|
||||
PENDING_TEAM_PLAN_STATES.has(plan.state) &&
|
||||
plan.members.some(member => member.name === name)
|
||||
)
|
||||
}
|
||||
|
||||
async function resumeTeammateWithMessage(
|
||||
recipientName: string,
|
||||
teamName: string,
|
||||
content: string,
|
||||
summary: string | undefined,
|
||||
context: ToolUseContext,
|
||||
): Promise<{ data: MessageOutput }> {
|
||||
const senderName =
|
||||
getAgentName() || (isTeammate() ? 'teammate' : TEAM_LEAD_NAME)
|
||||
try {
|
||||
const outcome = await resumeInProcessTeammate({
|
||||
agentName: recipientName,
|
||||
teamName,
|
||||
prompt: content,
|
||||
from: senderName,
|
||||
summary,
|
||||
context,
|
||||
})
|
||||
if (outcome.kind === 'already_running') {
|
||||
const queued = await handleMessage(
|
||||
recipientName,
|
||||
content,
|
||||
summary,
|
||||
context,
|
||||
)
|
||||
if (!queued.data.success) return queued
|
||||
return {
|
||||
data: {
|
||||
...queued.data,
|
||||
message: `Teammate "${recipientName}" is already running; queued your message for its next turn.`,
|
||||
},
|
||||
}
|
||||
}
|
||||
return {
|
||||
data: {
|
||||
success: true,
|
||||
message:
|
||||
outcome.resumedMessageCount > 0
|
||||
? `Teammate "${recipientName}" was not running; resumed it as an in-process teammate with ${outcome.resumedMessageCount} prior messages and your message as its next prompt.`
|
||||
: `Teammate "${recipientName}" was not running; resumed it as an in-process teammate (no prior transcript) with your message as its next prompt.`,
|
||||
},
|
||||
}
|
||||
} catch (error) {
|
||||
return {
|
||||
data: {
|
||||
success: false,
|
||||
message: `Failed to resume teammate "${recipientName}" — nothing was sent: ${errorMessage(error)}`,
|
||||
},
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
async function handleBroadcast(
|
||||
content: string,
|
||||
summary: string | undefined,
|
||||
@@ -239,8 +410,9 @@ async function handleBroadcast(
|
||||
}
|
||||
}
|
||||
|
||||
const failedRecipients: string[] = []
|
||||
for (const recipientName of recipients) {
|
||||
await writeToMailbox(
|
||||
const written = await writeToMailbox(
|
||||
recipientName,
|
||||
{
|
||||
id: messageId,
|
||||
@@ -252,6 +424,23 @@ async function handleBroadcast(
|
||||
},
|
||||
teamName,
|
||||
)
|
||||
if (!written) failedRecipients.push(recipientName)
|
||||
}
|
||||
|
||||
if (failedRecipients.length > 0) {
|
||||
const delivered = recipients.filter(
|
||||
name => !failedRecipients.includes(name),
|
||||
)
|
||||
return {
|
||||
data: {
|
||||
success: false,
|
||||
message:
|
||||
delivered.length === 0
|
||||
? `Failed to write to the inbox of ${failedRecipients.join(', ')} — nothing was sent. Try again.`
|
||||
: `Message broadcast to ${delivered.length} of ${recipients.length} teammate(s): ${delivered.join(', ')}. Failed to write to the inbox of ${failedRecipients.join(', ')} — they did not receive it. Message them directly instead of broadcasting again.`,
|
||||
recipients: delivered,
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
return {
|
||||
@@ -286,7 +475,7 @@ async function handleShutdownRequest(
|
||||
reason,
|
||||
})
|
||||
|
||||
await writeToMailbox(
|
||||
const written = await writeToMailbox(
|
||||
targetName,
|
||||
{
|
||||
from: senderName,
|
||||
@@ -296,6 +485,16 @@ async function handleShutdownRequest(
|
||||
},
|
||||
teamName,
|
||||
)
|
||||
if (!written) {
|
||||
return {
|
||||
data: {
|
||||
success: false,
|
||||
message: `Failed to write the shutdown request to ${targetName}'s inbox — nothing was sent. Try again.`,
|
||||
request_id: requestId,
|
||||
target: targetName,
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
return {
|
||||
data: {
|
||||
@@ -339,7 +538,7 @@ async function handleShutdownApproval(
|
||||
backendType: ownBackendType,
|
||||
})
|
||||
|
||||
await writeToMailbox(
|
||||
const written = await writeToMailbox(
|
||||
TEAM_LEAD_NAME,
|
||||
{
|
||||
from: agentName,
|
||||
@@ -349,6 +548,17 @@ async function handleShutdownApproval(
|
||||
},
|
||||
teamName,
|
||||
)
|
||||
// Exiting without the approval reaching the lead would leave this member
|
||||
// on its roster indefinitely; stay up so the approval can be sent again.
|
||||
if (!written) {
|
||||
return {
|
||||
data: {
|
||||
success: false,
|
||||
message: `Failed to write the shutdown approval to ${TEAM_LEAD_NAME}'s inbox — nothing was sent and ${agentName} is still running. Try again.`,
|
||||
request_id: requestId,
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
if (ownBackendType === 'in-process') {
|
||||
logForDebugging(
|
||||
@@ -416,7 +626,7 @@ async function handleShutdownRejection(
|
||||
reason,
|
||||
})
|
||||
|
||||
await writeToMailbox(
|
||||
const written = await writeToMailbox(
|
||||
TEAM_LEAD_NAME,
|
||||
{
|
||||
from: agentName,
|
||||
@@ -426,6 +636,15 @@ async function handleShutdownRejection(
|
||||
},
|
||||
teamName,
|
||||
)
|
||||
if (!written) {
|
||||
return {
|
||||
data: {
|
||||
success: false,
|
||||
message: `Failed to write the shutdown rejection to ${TEAM_LEAD_NAME}'s inbox — nothing was sent. Try again.`,
|
||||
request_id: requestId,
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
return {
|
||||
data: {
|
||||
@@ -461,7 +680,7 @@ async function handlePlanApproval(
|
||||
permissionMode: modeToInherit,
|
||||
}
|
||||
|
||||
await writeToMailbox(
|
||||
const written = await writeToMailbox(
|
||||
recipientName,
|
||||
{
|
||||
from: TEAM_LEAD_NAME,
|
||||
@@ -470,6 +689,15 @@ async function handlePlanApproval(
|
||||
},
|
||||
teamName,
|
||||
)
|
||||
if (!written) {
|
||||
return {
|
||||
data: {
|
||||
success: false,
|
||||
message: `Failed to write the plan approval to ${recipientName}'s inbox — nothing was sent. Try again.`,
|
||||
request_id: requestId,
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
return {
|
||||
data: {
|
||||
@@ -503,7 +731,7 @@ async function handlePlanRejection(
|
||||
timestamp: new Date().toISOString(),
|
||||
}
|
||||
|
||||
await writeToMailbox(
|
||||
const written = await writeToMailbox(
|
||||
recipientName,
|
||||
{
|
||||
from: TEAM_LEAD_NAME,
|
||||
@@ -512,6 +740,15 @@ async function handlePlanRejection(
|
||||
},
|
||||
teamName,
|
||||
)
|
||||
if (!written) {
|
||||
return {
|
||||
data: {
|
||||
success: false,
|
||||
message: `Failed to write the plan rejection to ${recipientName}'s inbox — nothing was sent. Try again.`,
|
||||
request_id: requestId,
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
return {
|
||||
data: {
|
||||
@@ -890,7 +1127,12 @@ export const SendMessageTool: Tool<InputSchema, SendMessageToolOutput> =
|
||||
if (input.to === '*') {
|
||||
return handleBroadcast(input.message, input.summary, context)
|
||||
}
|
||||
return handleMessage(input.to, input.message, input.summary, context)
|
||||
return handleTeammateMessage(
|
||||
input.to,
|
||||
input.message,
|
||||
input.summary,
|
||||
context,
|
||||
)
|
||||
}
|
||||
|
||||
if (input.to === '*') {
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
import { afterEach, beforeEach, describe, expect, test } from 'bun:test'
|
||||
import { mkdtemp, rm, writeFile } from 'node:fs/promises'
|
||||
import { mkdir, mkdtemp, rm, writeFile } from 'node:fs/promises'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { createHash } from 'node:crypto'
|
||||
import { join } from 'node:path'
|
||||
@@ -156,6 +156,17 @@ describe('whole-team planning tools', () => {
|
||||
expect(resolveProposedTeamPlan(explicit, context).members[0]?.runtime.effortLevel).toBeUndefined()
|
||||
})
|
||||
|
||||
test('a suggested runtime naming an unknown provider falls back to the leader runtime', async () => {
|
||||
const proposal = (providerId: string) => ({ members: [{ id: 'w', name: 'w', prompt: 'Work', runtime: { providerId, modelId: 'claude-sonnet-5-5' } }], tasks: [] })
|
||||
// The model cannot see provider ids; a guess must not fail approval or
|
||||
// move the member to a different, billed provider.
|
||||
expect(resolveProposedTeamPlan(proposal('anthropic'), context).members[0]?.runtime).toEqual({ providerId: 'fake-provider', modelId: 'fake-model' })
|
||||
expect(resolveProposedTeamPlan(proposal('claude-official'), context).members[0]?.runtime.providerId).toBe('fake-provider')
|
||||
await mkdir(join(directory, 'cc-haha'), { recursive: true })
|
||||
await writeFile(join(directory, 'cc-haha', 'providers.json'), JSON.stringify({ schemaVersion: 6, activeId: null, providers: [{ id: 'configured-provider', presetId: 'custom', name: 'Configured', apiKey: 'fake', baseUrl: 'http://127.0.0.1:1', apiFormat: 'anthropic', models: { main: 'm', haiku: 'm', sonnet: 'm', opus: 'm' } }] }))
|
||||
expect(resolveProposedTeamPlan(proposal('configured-provider'), context).members[0]?.runtime).toEqual({ providerId: 'configured-provider', modelId: 'claude-sonnet-5-5' })
|
||||
})
|
||||
|
||||
test('inline MCP credentials never enter the durable catalog or a proposed member', async () => {
|
||||
const definitions = context.options.agentDefinitions.activeAgents
|
||||
definitions.push({ agentType: 'inline-mcp', source: 'flagSettings', whenToUse: 'Fixture', getSystemPrompt: () => 'Fixture prompt', mcpServers: [{ private: { command: 'fixture', env: { API_KEY: 'do-not-persist-fixture-token' } } }] } as typeof definitions[number])
|
||||
|
||||
@@ -1,4 +1,7 @@
|
||||
import { readFileSync } from 'node:fs'
|
||||
import { join } from 'node:path'
|
||||
import { z } from 'zod/v4'
|
||||
import { getClaudeConfigHomeDir } from '../../utils/envUtils.js'
|
||||
import { snapshotTeamPlanPresetSource } from '../../utils/swarm/teamPlanPresetSource.js'
|
||||
import type { ToolUseContext } from '../../Tool.js'
|
||||
import type { TeamPlanMember, TeamPlanRecord, TeamPlanRuntime, TeamPlanTask } from '../../shared/teamPlan.js'
|
||||
@@ -57,6 +60,25 @@ export function snapshotTeamAgents(context: ToolUseContext) {
|
||||
}]))
|
||||
}
|
||||
|
||||
/**
|
||||
* The model cannot see provider ids, so a suggested runtime is only kept when
|
||||
* it names the leader's provider or one the user actually configured. A
|
||||
* guessed id ("anthropic") would otherwise fail approval, and a guessed
|
||||
* first-party id could silently move members onto a different, billed provider.
|
||||
*/
|
||||
function isConfiguredProviderSuggestion(providerId: string, leaderRuntime: TeamPlanRuntime): boolean {
|
||||
if (providerId === leaderRuntime.providerId) return true
|
||||
// Only the saved ids matter here; parsing the file directly keeps this tool
|
||||
// free of the server's provider modules (importing them is an init cycle).
|
||||
try {
|
||||
const index = JSON.parse(readFileSync(join(getClaudeConfigHomeDir(), 'cc-haha', 'providers.json'), 'utf8')) as { providers?: unknown }
|
||||
return Array.isArray(index.providers) && index.providers.some(provider =>
|
||||
!!provider && typeof provider === 'object' && (provider as { id?: unknown }).id === providerId)
|
||||
} catch {
|
||||
return false
|
||||
}
|
||||
}
|
||||
|
||||
export function resolveProposedTeamPlan(plan: ProposedTeamPlan, context: ToolUseContext) {
|
||||
const state = context.getAppState()
|
||||
const leaderRuntime = getTeamLeaderRuntime(context.options.mainLoopModel ?? state.mainLoopModelForSession ?? state.mainLoopModel ?? '')
|
||||
@@ -66,11 +88,14 @@ export function resolveProposedTeamPlan(plan: ProposedTeamPlan, context: ToolUse
|
||||
const agentSnapshot = agentCatalog[agentType]
|
||||
if (!agentSnapshot) throw new Error(`Agent preset '${agentType}' is unavailable. Choose a listed Agent preset.`)
|
||||
if (agentSnapshot.configurationError) throw new Error(agentSnapshot.configurationError)
|
||||
const runtime: TeamPlanRuntime = member.runtime ?? {
|
||||
const suggested = member.runtime && isConfiguredProviderSuggestion(member.runtime.providerId, leaderRuntime)
|
||||
? member.runtime
|
||||
: undefined
|
||||
const runtime: TeamPlanRuntime = suggested ?? {
|
||||
...leaderRuntime,
|
||||
modelId: resolveTeammateModel(undefined, leaderRuntime.modelId, agentSnapshot.model, true),
|
||||
}
|
||||
if (!member.runtime) {
|
||||
if (!suggested) {
|
||||
if (agentSnapshot.effortLevel !== undefined) runtime.effortLevel = agentSnapshot.effortLevel
|
||||
else if (runtime.modelId !== leaderRuntime.modelId) delete runtime.effortLevel
|
||||
}
|
||||
|
||||
@@ -43,6 +43,22 @@ const originalTeamHelpers = { ...teamHelpersModule }
|
||||
const mutateTeamFileAsyncActual = teamHelpersModule.mutateTeamFileAsync
|
||||
const readTeamFileAsyncActual = teamHelpersModule.readTeamFileAsync
|
||||
|
||||
// mock.module replaces exports process-wide and outlives this file, so every
|
||||
// module mocked below is put back afterwards from these untouched copies.
|
||||
// Otherwise later files in one `bun test` run (the coverage gate runs all of
|
||||
// src in a single process) get a runner, mailbox and backends without their
|
||||
// real exports.
|
||||
const originalModules: Array<[string, Record<string, unknown>]> = [
|
||||
['../../utils/swarm/backends/registry.js', { ...(await import('../../utils/swarm/backends/registry.js')) }],
|
||||
['../../utils/swarm/backends/detection.js', { ...(await import('../../utils/swarm/backends/detection.js')) }],
|
||||
['../../utils/swarm/teammateLayoutManager.js', { ...(await import('../../utils/swarm/teammateLayoutManager.js')) }],
|
||||
['../../utils/swarm/inProcessRunner.js', { ...(await import('../../utils/swarm/inProcessRunner.js')) }],
|
||||
['../../utils/swarm/spawnInProcess.js', { ...(await import('../../utils/swarm/spawnInProcess.js')) }],
|
||||
['../../utils/task/framework.js', { ...taskFrameworkModule }],
|
||||
['../../utils/teammateMailbox.js', { ...(await import('../../utils/teammateMailbox.js')) }],
|
||||
['../../utils/execFileNoThrow.js', { ...execFileNoThrowModule }],
|
||||
]
|
||||
|
||||
mock.module('../../utils/swarm/backends/registry.js', () => ({
|
||||
detectAndGetBackend: async () => ({
|
||||
backend: { type: 'tmux' },
|
||||
@@ -126,6 +142,9 @@ beforeEach(() => {
|
||||
|
||||
afterAll(() => {
|
||||
mock.module('../../utils/swarm/teamHelpers.js', () => originalTeamHelpers)
|
||||
for (const [path, original] of originalModules) {
|
||||
mock.module(path, () => original)
|
||||
}
|
||||
if (originalSubagentModel === undefined) {
|
||||
delete process.env.CLAUDE_CODE_SUBAGENT_MODEL
|
||||
} else {
|
||||
|
||||
@@ -0,0 +1,138 @@
|
||||
import { afterEach, beforeEach, describe, expect, mock, spyOn, test } from 'bun:test'
|
||||
import { mkdtemp, rm } from 'fs/promises'
|
||||
import { tmpdir } from 'os'
|
||||
import { join } from 'path'
|
||||
import { resetStateForTests, setIsInteractive } from '../../bootstrap/state.js'
|
||||
import * as promptsModule from '../../constants/prompts.js'
|
||||
import type { AppState } from '../../state/AppState.js'
|
||||
import type { ToolUseContext } from '../../Tool.js'
|
||||
import type { InProcessTeammateTaskState } from '../../tasks/InProcessTeammateTask/types.js'
|
||||
import { drainSdkEvents } from '../../utils/sdkEventQueue.js'
|
||||
import { readTeamFile, writeTeamFileAsync } from '../../utils/swarm/teamHelpers.js'
|
||||
import { readMailbox, writeToMailbox } from '../../utils/teammateMailbox.js'
|
||||
import * as runAgentModule from '../AgentTool/runAgent.js'
|
||||
import { spawnTeammate } from './spawnMultiAgent.js'
|
||||
|
||||
const TEAM = 'spawn-team'
|
||||
let configDir: string
|
||||
let originalConfigDir: string | undefined
|
||||
|
||||
beforeEach(async () => {
|
||||
originalConfigDir = process.env.CLAUDE_CONFIG_DIR
|
||||
configDir = await mkdtemp(join(tmpdir(), 'cc-haha-spawn-in-process-'))
|
||||
process.env.CLAUDE_CONFIG_DIR = configDir
|
||||
resetStateForTests()
|
||||
// A non-interactive session always spawns in-process teammates
|
||||
setIsInteractive(false)
|
||||
await writeTeamFileAsync(TEAM, {
|
||||
name: TEAM,
|
||||
createdAt: Date.now(),
|
||||
leadAgentId: `team-lead@${TEAM}`,
|
||||
members: [],
|
||||
})
|
||||
})
|
||||
|
||||
afterEach(async () => {
|
||||
mock.restore()
|
||||
drainSdkEvents()
|
||||
resetStateForTests()
|
||||
if (originalConfigDir === undefined) delete process.env.CLAUDE_CONFIG_DIR
|
||||
else process.env.CLAUDE_CONFIG_DIR = originalConfigDir
|
||||
await rm(configDir, { recursive: true, force: true })
|
||||
})
|
||||
|
||||
function leadContext() {
|
||||
let state = {
|
||||
mainLoopModel: 'claude-sonnet-4-6',
|
||||
toolPermissionContext: { mode: 'default' },
|
||||
teamContext: {
|
||||
teamName: TEAM,
|
||||
teamFilePath: '',
|
||||
leadAgentId: `team-lead@${TEAM}`,
|
||||
teammates: {},
|
||||
},
|
||||
tasks: {},
|
||||
} as unknown as AppState
|
||||
const context = {
|
||||
getAppState: () => state,
|
||||
setAppState: (updater: (previous: AppState) => AppState) => {
|
||||
state = updater(state)
|
||||
},
|
||||
options: {
|
||||
agentDefinitions: { activeAgents: [], allAgents: [] },
|
||||
tools: [],
|
||||
mainLoopModel: 'claude-sonnet-4-6',
|
||||
mcpClients: [],
|
||||
thinkingConfig: { type: 'disabled' },
|
||||
},
|
||||
abortController: new AbortController(),
|
||||
messages: [],
|
||||
} as unknown as ToolUseContext
|
||||
const teammates = () =>
|
||||
Object.values(state.tasks).filter(
|
||||
(task): task is InProcessTeammateTaskState =>
|
||||
task.type === 'in_process_teammate',
|
||||
)
|
||||
return { context, teammates }
|
||||
}
|
||||
|
||||
async function waitFor(predicate: () => boolean | Promise<boolean>): Promise<void> {
|
||||
for (let attempt = 0; attempt < 300; attempt++) {
|
||||
if (await predicate()) return
|
||||
await new Promise(resolve => setTimeout(resolve, 10))
|
||||
}
|
||||
throw new Error('condition not reached')
|
||||
}
|
||||
|
||||
describe('spawning in-process teammates', () => {
|
||||
test('concurrent spawns with one name reserve distinct names on the roster', async () => {
|
||||
spyOn(promptsModule, 'getSystemPrompt').mockResolvedValue(['System prompt'])
|
||||
const lead = leadContext()
|
||||
spyOn(runAgentModule, 'runAgent').mockImplementation(async function* (
|
||||
input: Parameters<typeof runAgentModule.runAgent>[0],
|
||||
) {
|
||||
// Each teammate ends after its first turn
|
||||
for (const task of lead.teammates()) task.abortController?.abort()
|
||||
yield* []
|
||||
void input
|
||||
})
|
||||
|
||||
const [first, second] = await Promise.all([
|
||||
spawnTeammate({ name: 'worker', prompt: 'Fix the parser', team_name: TEAM }, lead.context),
|
||||
spawnTeammate({ name: 'worker', prompt: 'Write the docs', team_name: TEAM }, lead.context),
|
||||
])
|
||||
|
||||
expect(new Set([first.data.name, second.data.name])).toEqual(
|
||||
new Set(['worker', 'worker-2']),
|
||||
)
|
||||
expect(readTeamFile(TEAM)?.members.map(member => member.agentId).sort()).toEqual([
|
||||
`worker-2@${TEAM}`,
|
||||
`worker@${TEAM}`,
|
||||
])
|
||||
await waitFor(() => lead.teammates().every(task => task.status !== 'running'))
|
||||
})
|
||||
|
||||
test('a new teammate does not inherit mail left for an earlier one with its name', async () => {
|
||||
spyOn(promptsModule, 'getSystemPrompt').mockResolvedValue(['System prompt'])
|
||||
await writeToMailbox(
|
||||
'worker',
|
||||
{ from: 'reviewer', text: 'Stale note for the previous worker', timestamp: new Date().toISOString() },
|
||||
TEAM,
|
||||
)
|
||||
const lead = leadContext()
|
||||
let unreadAtFirstTurn: string[] | undefined
|
||||
spyOn(runAgentModule, 'runAgent').mockImplementation(async function* () {
|
||||
unreadAtFirstTurn = (await readMailbox('worker', TEAM))
|
||||
.filter(message => !message.read)
|
||||
.map(message => message.text)
|
||||
for (const task of lead.teammates()) task.abortController?.abort()
|
||||
yield* []
|
||||
})
|
||||
|
||||
await spawnTeammate({ name: 'worker', prompt: 'Fix the parser', team_name: TEAM }, lead.context)
|
||||
await waitFor(() => unreadAtFirstTurn !== undefined)
|
||||
|
||||
expect(unreadAtFirstTurn).toEqual([])
|
||||
await waitFor(() => lead.teammates().every(task => task.status !== 'running'))
|
||||
})
|
||||
})
|
||||
@@ -54,6 +54,7 @@ import { buildInheritedEnvVars } from '../../utils/swarm/spawnUtils.js'
|
||||
import {
|
||||
mutateTeamFileAsync,
|
||||
readTeamFileAsync,
|
||||
removeMemberByAgentId,
|
||||
sanitizeAgentName,
|
||||
sanitizeName,
|
||||
} from '../../utils/swarm/teamHelpers.js'
|
||||
@@ -290,7 +291,14 @@ export async function generateUniqueTeammateName(
|
||||
return baseName
|
||||
}
|
||||
|
||||
const existingNames = new Set(teamFile.members.map(m => m.name.toLowerCase()))
|
||||
return pickUniqueTeammateName(baseName, teamFile.members)
|
||||
}
|
||||
|
||||
function pickUniqueTeammateName(
|
||||
baseName: string,
|
||||
members: ReadonlyArray<{ name: string }>,
|
||||
): string {
|
||||
const existingNames = new Set(members.map(m => m.name.toLowerCase()))
|
||||
|
||||
// If the base name doesn't exist, use it as-is
|
||||
if (!existingNames.has(baseName.toLowerCase())) {
|
||||
@@ -889,18 +897,6 @@ async function handleSpawnInProcess(
|
||||
)
|
||||
}
|
||||
|
||||
// Generate unique name if duplicate exists in team
|
||||
const uniqueName = await generateUniqueTeammateName(name, teamName)
|
||||
|
||||
// Sanitize the name to prevent @ in agent IDs
|
||||
const sanitizedName = sanitizeAgentName(uniqueName)
|
||||
|
||||
// Generate deterministic agent ID from name and team
|
||||
const teammateId = formatAgentId(sanitizedName, teamName)
|
||||
|
||||
// Assign a unique color to this teammate
|
||||
const teammateColor = assignTeammateColor(teammateId)
|
||||
|
||||
// Look up custom agent definition if agent_type is provided
|
||||
let agentDefinition:
|
||||
| CustomAgentDefinition
|
||||
@@ -919,6 +915,36 @@ async function handleSpawnInProcess(
|
||||
)
|
||||
}
|
||||
|
||||
// Reserve a unique name and put the member on the roster under the
|
||||
// team-file lock before the teammate starts: concurrent spawns can never
|
||||
// pick the same name, and a running teammate is always on the roster.
|
||||
// Sanitize first (no @ in agent IDs) so uniqueness covers the name used.
|
||||
const baseName = sanitizeAgentName(name)
|
||||
let sanitizedName = baseName
|
||||
let teammateId = formatAgentId(baseName, teamName)
|
||||
let teammateColor: ReturnType<typeof assignTeammateColor> | undefined
|
||||
await mutateTeamFileAsync(teamName, (teamFile) => {
|
||||
sanitizedName = pickUniqueTeammateName(baseName, teamFile.members)
|
||||
// Generate deterministic agent ID from name and team
|
||||
teammateId = formatAgentId(sanitizedName, teamName)
|
||||
// Assign a unique color to this teammate
|
||||
teammateColor = assignTeammateColor(teammateId)
|
||||
teamFile.members.push({
|
||||
agentId: teammateId,
|
||||
name: sanitizedName,
|
||||
agentType: agent_type,
|
||||
model,
|
||||
prompt,
|
||||
color: teammateColor,
|
||||
planModeRequired: plan_mode_required,
|
||||
joinedAt: Date.now(),
|
||||
tmuxPaneId: 'in-process',
|
||||
cwd: getCwd(),
|
||||
subscriptions: [],
|
||||
backendType: 'in-process',
|
||||
})
|
||||
})
|
||||
|
||||
// Spawn in-process teammate
|
||||
const config: InProcessSpawnConfig = {
|
||||
name: sanitizedName,
|
||||
@@ -932,6 +958,12 @@ async function handleSpawnInProcess(
|
||||
const result = await spawnInProcessTeammate(config, context)
|
||||
|
||||
if (!result.success) {
|
||||
// Give the reserved name back
|
||||
await removeMemberByAgentId(teamName, teammateId).catch(error =>
|
||||
logForDebugging(
|
||||
`[handleSpawnInProcess] could not release ${teammateId}: ${errorMessage(error)}`,
|
||||
),
|
||||
)
|
||||
throw new Error(result.error ?? 'Failed to spawn in-process teammate')
|
||||
}
|
||||
|
||||
@@ -943,7 +975,7 @@ async function handleSpawnInProcess(
|
||||
// Start the agent execution loop (fire-and-forget)
|
||||
if (result.taskId && result.teammateContext && result.abortController) {
|
||||
startInProcessTeammate({
|
||||
identity: {
|
||||
identity: result.identity ?? {
|
||||
agentId: teammateId,
|
||||
agentName: sanitizedName,
|
||||
teamName,
|
||||
@@ -1018,23 +1050,6 @@ async function handleSpawnInProcess(
|
||||
}
|
||||
})
|
||||
|
||||
await mutateTeamFileAsync(teamName, (teamFile) => {
|
||||
teamFile.members.push({
|
||||
agentId: teammateId,
|
||||
name: sanitizedName,
|
||||
agentType: agent_type,
|
||||
model,
|
||||
prompt,
|
||||
color: teammateColor,
|
||||
planModeRequired: plan_mode_required,
|
||||
joinedAt: Date.now(),
|
||||
tmuxPaneId: 'in-process',
|
||||
cwd: getCwd(),
|
||||
subscriptions: [],
|
||||
backendType: 'in-process',
|
||||
})
|
||||
})
|
||||
|
||||
// Note: Do NOT send the prompt via mailbox for in-process teammates.
|
||||
// In-process teammates receive the prompt directly via startInProcessTeammate().
|
||||
// The mailbox is only needed for tmux-based teammates which poll for their initial message.
|
||||
|
||||
@@ -0,0 +1,227 @@
|
||||
import { afterEach, beforeEach, describe, expect, spyOn, test } from 'bun:test'
|
||||
import { mkdtemp, rm } from 'fs/promises'
|
||||
import { tmpdir } from 'os'
|
||||
import { join } from 'path'
|
||||
import { getIsInteractive, setIsInteractive } from '../bootstrap/state.js'
|
||||
import type { AppState } from '../state/AppState.js'
|
||||
import type { ToolUseContext } from '../Tool.js'
|
||||
import { getTeammateMailboxAttachments } from './attachments.js'
|
||||
import { clearDynamicTeamContext, setDynamicTeamContext } from './teammate.js'
|
||||
import {
|
||||
createTeammateContext,
|
||||
runWithTeammateContext,
|
||||
} from './teammateContext.js'
|
||||
import * as mailbox from './teammateMailbox.js'
|
||||
|
||||
const TEAM = 'attach-team'
|
||||
const LEAD_ID = `team-lead@${TEAM}`
|
||||
const WORKER_ID = `worker@${TEAM}`
|
||||
|
||||
const ENV_KEYS = [
|
||||
'CLAUDE_CONFIG_DIR',
|
||||
'USER_TYPE',
|
||||
'CC_HAHA_AGENT_TEAMS_ENABLED',
|
||||
'CC_HAHA_TEAM_WORKER',
|
||||
] as const
|
||||
let savedEnv: Partial<Record<(typeof ENV_KEYS)[number], string>>
|
||||
let savedInteractive: boolean
|
||||
let configDir: string
|
||||
|
||||
beforeEach(async () => {
|
||||
savedEnv = Object.fromEntries(ENV_KEYS.map(key => [key, process.env[key]]))
|
||||
savedInteractive = getIsInteractive()
|
||||
configDir = await mkdtemp(join(tmpdir(), 'cc-haha-mailbox-attachments-'))
|
||||
process.env.CLAUDE_CONFIG_DIR = configDir
|
||||
process.env.CC_HAHA_AGENT_TEAMS_ENABLED = '1'
|
||||
delete process.env.USER_TYPE
|
||||
delete process.env.CC_HAHA_TEAM_WORKER
|
||||
// The desktop lead runs in print mode.
|
||||
setIsInteractive(false)
|
||||
})
|
||||
|
||||
afterEach(async () => {
|
||||
clearDynamicTeamContext()
|
||||
setIsInteractive(savedInteractive)
|
||||
for (const key of ENV_KEYS) {
|
||||
const value = savedEnv[key]
|
||||
if (value === undefined) delete process.env[key]
|
||||
else process.env[key] = value
|
||||
}
|
||||
await rm(configDir, { recursive: true, force: true })
|
||||
})
|
||||
|
||||
function leadState(inbox: AppState['inbox']['messages'] = []): AppState {
|
||||
return {
|
||||
teamContext: {
|
||||
teamName: TEAM,
|
||||
leadAgentId: LEAD_ID,
|
||||
teammates: {
|
||||
[LEAD_ID]: { name: 'team-lead' },
|
||||
[WORKER_ID]: { name: 'worker' },
|
||||
},
|
||||
},
|
||||
inbox: { messages: inbox },
|
||||
tasks: {},
|
||||
} as unknown as AppState
|
||||
}
|
||||
|
||||
function contextFor(state: AppState, agentId?: string): ToolUseContext {
|
||||
let current = state
|
||||
return {
|
||||
agentId,
|
||||
getAppState: () => current,
|
||||
setAppState: (update: (previous: AppState) => AppState) => {
|
||||
current = update(current)
|
||||
},
|
||||
} as unknown as ToolUseContext
|
||||
}
|
||||
|
||||
async function send(
|
||||
recipient: string,
|
||||
from: string,
|
||||
text: string,
|
||||
second: number,
|
||||
): Promise<void> {
|
||||
const timestamp = new Date(Date.UTC(2026, 9, 3, 0, 0, second)).toISOString()
|
||||
expect(
|
||||
await mailbox.writeToMailbox(recipient, { from, text, timestamp }, TEAM),
|
||||
).toBe(true)
|
||||
}
|
||||
|
||||
async function unreadTexts(recipient: string): Promise<string[]> {
|
||||
return (await mailbox.readUnreadMessages(recipient, TEAM)).map(m => m.text)
|
||||
}
|
||||
|
||||
function attachedTexts(attachments: unknown[]): string[] {
|
||||
const [attachment] = attachments as Array<{
|
||||
type: string
|
||||
messages: Array<{ text: string }>
|
||||
}>
|
||||
expect(attachment?.type).toBe('teammate_mailbox')
|
||||
return attachment!.messages.map(message => message.text)
|
||||
}
|
||||
|
||||
function joinAsWorker(): void {
|
||||
setDynamicTeamContext({
|
||||
agentId: WORKER_ID,
|
||||
agentName: 'worker',
|
||||
teamName: TEAM,
|
||||
planModeRequired: false,
|
||||
})
|
||||
}
|
||||
|
||||
describe('external builds deliver teammate mail between turns only', () => {
|
||||
test("the lead's mail waits for its next turn, where the chat shows it", async () => {
|
||||
await send('team-lead', 'worker', 'Analysis complete', 1)
|
||||
|
||||
expect(await getTeammateMailboxAttachments(contextFor(leadState()))).toEqual([])
|
||||
expect(await unreadTexts('team-lead')).toEqual(['Analysis complete'])
|
||||
})
|
||||
|
||||
test('no teammate consumes any inbox mid-turn', async () => {
|
||||
await send('worker', 'team-lead', 'next task', 1)
|
||||
await send('team-lead', 'worker', 'for the lead', 2)
|
||||
|
||||
// A desktop process worker leaves both inboxes to the server and the lead.
|
||||
process.env.CC_HAHA_TEAM_WORKER = '1'
|
||||
expect(await getTeammateMailboxAttachments(contextFor(leadState()))).toEqual([])
|
||||
delete process.env.CC_HAHA_TEAM_WORKER
|
||||
|
||||
joinAsWorker()
|
||||
expect(await getTeammateMailboxAttachments(contextFor(leadState()))).toEqual([])
|
||||
clearDynamicTeamContext()
|
||||
|
||||
const teammate = createTeammateContext({
|
||||
agentId: WORKER_ID,
|
||||
agentName: 'worker',
|
||||
teamName: TEAM,
|
||||
planModeRequired: false,
|
||||
parentSessionId: 'lead-session',
|
||||
abortController: new AbortController(),
|
||||
})
|
||||
expect(
|
||||
await runWithTeammateContext(teammate, () =>
|
||||
getTeammateMailboxAttachments(contextFor(leadState(), WORKER_ID)),
|
||||
),
|
||||
).toEqual([])
|
||||
|
||||
expect(await unreadTexts('worker')).toEqual(['next task'])
|
||||
expect(await unreadTexts('team-lead')).toEqual(['for the lead'])
|
||||
})
|
||||
})
|
||||
|
||||
describe('ant builds deliver mail mid-turn', () => {
|
||||
beforeEach(() => {
|
||||
process.env.USER_TYPE = 'ant'
|
||||
})
|
||||
|
||||
test('attaches teammate mail and leaves protocol messages for the poller', async () => {
|
||||
const idle = JSON.stringify(
|
||||
mailbox.createIdleNotification('worker', {
|
||||
idleReason: 'failed',
|
||||
failureReason: 'provider stream ended',
|
||||
}),
|
||||
)
|
||||
const permissionRequest = JSON.stringify(
|
||||
mailbox.createPermissionRequestMessage({
|
||||
request_id: 'perm-1',
|
||||
agent_id: 'worker',
|
||||
tool_name: 'Bash',
|
||||
tool_use_id: 'tool-1',
|
||||
description: 'run tests',
|
||||
input: { command: 'bun test' },
|
||||
}),
|
||||
)
|
||||
await send('team-lead', 'worker', 'Analysis complete', 1)
|
||||
await send('team-lead', 'worker', idle, 2)
|
||||
await send('team-lead', 'worker', permissionRequest, 3)
|
||||
|
||||
const attachments = await getTeammateMailboxAttachments(
|
||||
contextFor(leadState()),
|
||||
)
|
||||
|
||||
expect(attachedTexts(attachments)).toEqual(['Analysis complete', idle])
|
||||
expect(await unreadTexts('team-lead')).toEqual([permissionRequest])
|
||||
expect(
|
||||
(await mailbox.readMailboxHistory('team-lead', TEAM))
|
||||
.filter(message => message.read)
|
||||
.map(message => message.text),
|
||||
).toEqual(['Analysis complete', idle])
|
||||
})
|
||||
|
||||
test('acknowledges only what it attached', async () => {
|
||||
await send('team-lead', 'worker', 'first result', 1)
|
||||
const readUnread = mailbox.readUnreadMessages
|
||||
const read = spyOn(mailbox, 'readUnreadMessages').mockImplementation(
|
||||
async (agentName, teamName) => {
|
||||
const batch = await readUnread(agentName, teamName)
|
||||
await send('team-lead', 'reviewer', 'arrived after the read', 2)
|
||||
return batch
|
||||
},
|
||||
)
|
||||
|
||||
try {
|
||||
expect(
|
||||
attachedTexts(
|
||||
await getTeammateMailboxAttachments(contextFor(leadState())),
|
||||
),
|
||||
).toEqual(['first result'])
|
||||
} finally {
|
||||
read.mockRestore()
|
||||
}
|
||||
|
||||
expect(await unreadTexts('team-lead')).toEqual(['arrived after the read'])
|
||||
})
|
||||
|
||||
test('a teammate receives its own mail', async () => {
|
||||
joinAsWorker()
|
||||
await send('worker', 'team-lead', 'next task', 1)
|
||||
|
||||
expect(
|
||||
attachedTexts(
|
||||
await getTeammateMailboxAttachments(contextFor(leadState())),
|
||||
),
|
||||
).toEqual(['next task'])
|
||||
expect(await unreadTexts('worker')).toEqual([])
|
||||
})
|
||||
})
|
||||
+21
-17
@@ -233,7 +233,7 @@ import { getAutoMemPath, isAutoMemoryEnabled } from '../memdir/paths.js'
|
||||
import { getAgentMemoryDir } from '../tools/AgentTool/agentMemory.js'
|
||||
import {
|
||||
readUnreadMessages,
|
||||
markMessagesAsReadByPredicate,
|
||||
claimMailboxMessages,
|
||||
isShutdownApproved,
|
||||
isStructuredProtocolMessage,
|
||||
isIdleNotification,
|
||||
@@ -3663,8 +3663,13 @@ async function getAsyncHookResponseAttachments(): Promise<Attachment[]> {
|
||||
*
|
||||
* Messages from AppState.inbox are delivered mid-turn as attachments,
|
||||
* allowing teammates to receive messages without waiting for the turn to end.
|
||||
*
|
||||
* Like the official external CLI, external builds never deliver mail mid-turn:
|
||||
* the lead receives teammate messages between turns (the print-mode poll or
|
||||
* useInboxPoller) as ordinary prompts, which the desktop chat shows. A
|
||||
* mid-turn attachment is only in the model context and the transcript.
|
||||
*/
|
||||
async function getTeammateMailboxAttachments(
|
||||
export async function getTeammateMailboxAttachments(
|
||||
toolUseContext: ToolUseContext,
|
||||
): Promise<Attachment[]> {
|
||||
if (!isAgentSwarmsEnabled()) {
|
||||
@@ -3723,11 +3728,21 @@ async function getTeammateMailboxAttachments(
|
||||
// messages as read, and if attachments wins, protocol messages get bundled as raw
|
||||
// LLM context text instead of being routed to their UI handlers.
|
||||
const allUnreadMessages = await readUnreadMessages(agentName, teamName)
|
||||
const unreadMessages = allUnreadMessages.filter(
|
||||
const unreadCandidates = allUnreadMessages.filter(
|
||||
m => !isStructuredProtocolMessage(m.text),
|
||||
)
|
||||
// Claim before attaching: only the messages this call marked read are
|
||||
// attached, so the print-mode lead poll (or a concurrent attachment pass)
|
||||
// can never deliver one of them a second time, and mail that arrives after
|
||||
// the read stays unread. A failed claim attaches nothing; the messages
|
||||
// remain unread for the next pass.
|
||||
const unreadMessages =
|
||||
unreadCandidates.length > 0
|
||||
? ((await claimMailboxMessages(agentName, teamName, unreadCandidates)) ??
|
||||
[])
|
||||
: []
|
||||
logForDebugging(
|
||||
`[MailboxBridge] Found ${allUnreadMessages.length} unread message(s) for "${agentName}" (${allUnreadMessages.length - unreadMessages.length} structured protocol messages filtered out)`,
|
||||
`[MailboxBridge] Found ${allUnreadMessages.length} unread message(s) for "${agentName}" (${allUnreadMessages.length - unreadCandidates.length} structured protocol messages filtered out, ${unreadMessages.length} claimed)`,
|
||||
)
|
||||
|
||||
// Also check AppState.inbox for pending messages (queued mid-turn by useInboxPoller)
|
||||
@@ -3810,6 +3825,8 @@ async function getTeammateMailboxAttachments(
|
||||
|
||||
// Build the attachment BEFORE marking messages as processed
|
||||
// This prevents message loss if any operation below fails
|
||||
// (mailbox messages were already claimed above; structured protocol
|
||||
// messages stay unread for useInboxPoller to handle).
|
||||
const attachment: Attachment[] = [
|
||||
{
|
||||
type: 'teammate_mailbox',
|
||||
@@ -3817,19 +3834,6 @@ async function getTeammateMailboxAttachments(
|
||||
},
|
||||
]
|
||||
|
||||
// Mark only non-structured mailbox messages as read after attachment is built.
|
||||
// Structured protocol messages stay unread for useInboxPoller to handle.
|
||||
if (unreadMessages.length > 0) {
|
||||
await markMessagesAsReadByPredicate(
|
||||
agentName,
|
||||
m => !isStructuredProtocolMessage(m.text),
|
||||
teamName,
|
||||
)
|
||||
logForDebugging(
|
||||
`[MailboxBridge] marked ${unreadMessages.length} non-structured message(s) as read for agent="${agentName}" team="${teamName || 'default'}"`,
|
||||
)
|
||||
}
|
||||
|
||||
// Process shutdown_approved messages - remove teammates from team file
|
||||
// This mirrors what useInboxPoller does in interactive mode (lines 546-606)
|
||||
// In -p mode, useInboxPoller doesn't run, so we must handle this here
|
||||
|
||||
@@ -0,0 +1,24 @@
|
||||
import { expect, test } from 'bun:test'
|
||||
import { parseDirectMemberMessage, sendDirectMemberMessage } from './directMemberMessage.js'
|
||||
|
||||
const teamContext = {
|
||||
teamName: 'review',
|
||||
leadAgentId: 'team-lead@review',
|
||||
teammates: { 'reader@review': { name: 'reader' } },
|
||||
} as never
|
||||
|
||||
test('a direct member message is reported as sent only once it is persisted', async () => {
|
||||
const parsed = parseDirectMemberMessage('@reader please re-check runner.ts')
|
||||
expect(parsed).toEqual({ recipientName: 'reader', message: 'please re-check runner.ts' })
|
||||
|
||||
const written: unknown[] = []
|
||||
expect(await sendDirectMemberMessage('reader', 'hi', teamContext, async (...args) => { written.push(args); return true }))
|
||||
.toEqual({ success: true, recipientName: 'reader' })
|
||||
expect(written).toHaveLength(1)
|
||||
|
||||
// A contended or unreadable inbox must not look like a delivered message.
|
||||
expect(await sendDirectMemberMessage('reader', 'hi', teamContext, async () => false))
|
||||
.toEqual({ success: false, error: 'delivery_failed', recipientName: 'reader' })
|
||||
expect(await sendDirectMemberMessage('stranger', 'hi', teamContext, async () => true))
|
||||
.toEqual({ success: false, error: 'unknown_recipient', recipientName: 'stranger' })
|
||||
})
|
||||
@@ -23,15 +23,16 @@ export type DirectMessageResult =
|
||||
| { success: true; recipientName: string }
|
||||
| {
|
||||
success: false
|
||||
error: 'no_team_context' | 'unknown_recipient'
|
||||
error: 'no_team_context' | 'unknown_recipient' | 'delivery_failed'
|
||||
recipientName?: string
|
||||
}
|
||||
|
||||
/** Resolves true once the message is persisted in the member's inbox. */
|
||||
type WriteToMailboxFn = (
|
||||
recipientName: string,
|
||||
message: { from: string; text: string; timestamp: string },
|
||||
teamName: string,
|
||||
) => Promise<void>
|
||||
) => Promise<boolean>
|
||||
|
||||
/**
|
||||
* Send a direct message to a team member, bypassing the model.
|
||||
@@ -55,7 +56,7 @@ export async function sendDirectMemberMessage(
|
||||
return { success: false, error: 'unknown_recipient', recipientName }
|
||||
}
|
||||
|
||||
await writeToMailbox(
|
||||
const delivered = await writeToMailbox(
|
||||
recipientName,
|
||||
{
|
||||
from: 'user',
|
||||
@@ -64,6 +65,9 @@ export async function sendDirectMemberMessage(
|
||||
},
|
||||
teamContext.teamName,
|
||||
)
|
||||
if (!delivered) {
|
||||
return { success: false, error: 'delivery_failed', recipientName }
|
||||
}
|
||||
|
||||
return { success: true, recipientName }
|
||||
}
|
||||
|
||||
@@ -522,3 +522,19 @@ describe('replaceMediaWithPlaceholders', () => {
|
||||
expect(replaceMediaWithPlaceholders(blocks)).toBe(blocks)
|
||||
})
|
||||
})
|
||||
|
||||
describe('stream retry display', () => {
|
||||
test("a mid-stream re-send clears the failed attempt's streaming tool calls", async () => {
|
||||
const { createSystemStreamingFallbackMessage, handleMessageFromStream } = await import('./messages.js')
|
||||
let toolUses = [{ index: 0, contentBlock: { type: 'tool_use', id: 'stale', name: 'Write', input: {} }, unparsedToolInput: '{"file_path":' }] as never[]
|
||||
const delivered: unknown[] = []
|
||||
const update = (f: (current: never[]) => never[]) => { toolUses = f(toolUses) }
|
||||
handleMessageFromStream(createSystemStreamingFallbackMessage('stream_retry'), message => delivered.push(message), () => {}, () => {}, update)
|
||||
expect(toolUses).toEqual([])
|
||||
expect(delivered).toHaveLength(1)
|
||||
|
||||
toolUses = [{ index: 0 }] as never[]
|
||||
handleMessageFromStream(createUserMessage({ content: 'unrelated' }), () => {}, () => {}, () => {}, update)
|
||||
expect(toolUses).toHaveLength(1)
|
||||
})
|
||||
})
|
||||
|
||||
+10
-1
@@ -3256,6 +3256,15 @@ export function handleMessageFromStream(
|
||||
}))
|
||||
}
|
||||
}
|
||||
// A mid-stream re-send discards the failed attempt; its half-streamed tool
|
||||
// calls never ran and must not linger until the retry's message_stop.
|
||||
if (
|
||||
message.type === 'system' &&
|
||||
message.subtype === 'streaming_fallback' &&
|
||||
message.cause === 'stream_retry'
|
||||
) {
|
||||
onStreamingToolUses(() => [])
|
||||
}
|
||||
// Clear streaming text NOW so the render can switch displayedMessages
|
||||
// from deferredMessages to messages in the same batch, making the
|
||||
// transition from streaming text → final message atomic (no gap, no duplication).
|
||||
@@ -4941,7 +4950,7 @@ export function createSystemStreamingFallbackMessage(
|
||||
subtype: 'streaming_fallback',
|
||||
level: 'info',
|
||||
content: cause === 'stream_retry'
|
||||
? 'Provider stream stalled before a tool side effect; retrying safely'
|
||||
? 'Provider stream was interrupted before any tool ran; retrying safely'
|
||||
: `Streaming request failed (${cause.replace(/_/g, ' ')}); retrying in non-streaming mode`,
|
||||
cause,
|
||||
timestamp: new Date().toISOString(),
|
||||
|
||||
@@ -81,6 +81,7 @@ import { logError } from './log.js'
|
||||
import { extractTag, isCompactBoundaryMessage } from './messages.js'
|
||||
import type { ModelAlias } from './model/aliases.js'
|
||||
import { sanitizePath } from './path.js'
|
||||
import type { PermissionMode } from './permissions/PermissionMode.js'
|
||||
import {
|
||||
extractJsonStringField,
|
||||
extractLastJsonStringField,
|
||||
@@ -297,6 +298,22 @@ export type AgentMetadata = {
|
||||
/** The runtime's sequential agent number within the run. */
|
||||
agentIndex: number
|
||||
}
|
||||
/**
|
||||
* Set on an in-process teammate's transcript. A teammate keeps this one
|
||||
* transcript for its whole life, and SendMessage resumes a teammate that is
|
||||
* no longer running from it. The fields below restore its team identity.
|
||||
* All optional — older metadata files and ordinary subagents lack them.
|
||||
*/
|
||||
taskKind?: 'in_process_teammate'
|
||||
teamName?: string
|
||||
/** The teammate's name within its team. */
|
||||
name?: string
|
||||
color?: string
|
||||
planModeRequired?: boolean
|
||||
/** agentType of the custom or plugin Agent the teammate was spawned as. */
|
||||
customAgentType?: string
|
||||
/** Permission mode of the teammate's latest turn. */
|
||||
permissionMode?: PermissionMode
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -0,0 +1,44 @@
|
||||
import { afterEach, expect, spyOn, test } from 'bun:test'
|
||||
import type { ToolUseContext } from '../../../Tool.js'
|
||||
import * as mailbox from '../../teammateMailbox.js'
|
||||
import { InProcessBackend } from './InProcessBackend.js'
|
||||
|
||||
let restore: (() => void) | undefined
|
||||
afterEach(() => {
|
||||
restore?.()
|
||||
restore = undefined
|
||||
})
|
||||
|
||||
test('a message or shutdown request the inbox did not take is reported instead of assumed sent', async () => {
|
||||
const write = spyOn(mailbox, 'writeToMailbox').mockResolvedValue(false)
|
||||
restore = () => write.mockRestore()
|
||||
const backend = new InProcessBackend()
|
||||
|
||||
await expect(
|
||||
backend.sendMessage('worker@review', {
|
||||
from: 'team-lead',
|
||||
text: 'Please re-check the parser',
|
||||
timestamp: new Date().toISOString(),
|
||||
}),
|
||||
).rejects.toThrow("Failed to write to worker's inbox")
|
||||
|
||||
const task = {
|
||||
id: 'task-1',
|
||||
type: 'in_process_teammate',
|
||||
status: 'running',
|
||||
shutdownRequested: false,
|
||||
identity: { agentId: 'worker@review', agentName: 'worker', teamName: 'review' },
|
||||
}
|
||||
let shutdownMarked = false
|
||||
backend.setContext({
|
||||
getAppState: () => ({ tasks: { [task.id]: task } }),
|
||||
setAppState: () => {
|
||||
shutdownMarked = true
|
||||
},
|
||||
} as unknown as ToolUseContext)
|
||||
|
||||
expect(await backend.terminate('worker@review', 'Review complete')).toBe(false)
|
||||
// Nothing was delivered, so the teammate is not marked as shutting down.
|
||||
expect(shutdownMarked).toBe(false)
|
||||
expect(write).toHaveBeenCalledTimes(2)
|
||||
})
|
||||
@@ -105,7 +105,7 @@ export class InProcessBackend implements TeammateExecutor {
|
||||
// Start the agent loop in the background (fire-and-forget)
|
||||
// The prompt is passed through the task state and config
|
||||
startInProcessTeammate({
|
||||
identity: {
|
||||
identity: result.identity ?? {
|
||||
agentId: result.agentId,
|
||||
agentName: config.name,
|
||||
teamName: config.teamName,
|
||||
@@ -165,7 +165,7 @@ export class InProcessBackend implements TeammateExecutor {
|
||||
const { agentName, teamName } = parsed
|
||||
|
||||
// Write to file-based mailbox
|
||||
await writeToMailbox(
|
||||
const written = await writeToMailbox(
|
||||
agentName,
|
||||
{
|
||||
text: message.text,
|
||||
@@ -175,6 +175,9 @@ export class InProcessBackend implements TeammateExecutor {
|
||||
},
|
||||
teamName,
|
||||
)
|
||||
if (!written) {
|
||||
throw new Error(`Failed to write to ${agentName}'s inbox`)
|
||||
}
|
||||
|
||||
logForDebugging(`[InProcessBackend] sendMessage() completed for ${agentId}`)
|
||||
}
|
||||
@@ -232,7 +235,7 @@ export class InProcessBackend implements TeammateExecutor {
|
||||
|
||||
// Send to teammate's mailbox
|
||||
const teammateAgentName = task.identity.agentName
|
||||
await writeToMailbox(
|
||||
const written = await writeToMailbox(
|
||||
teammateAgentName,
|
||||
{
|
||||
from: 'team-lead',
|
||||
@@ -241,6 +244,12 @@ export class InProcessBackend implements TeammateExecutor {
|
||||
},
|
||||
task.identity.teamName,
|
||||
)
|
||||
if (!written) {
|
||||
logForDebugging(
|
||||
`[InProcessBackend] terminate() could not deliver the shutdown request to ${agentId}`,
|
||||
)
|
||||
return false
|
||||
}
|
||||
|
||||
// Mark the task as shutdown requested
|
||||
requestTeammateShutdown(task.id, this.context.setAppState)
|
||||
|
||||
@@ -0,0 +1,182 @@
|
||||
import { afterEach, beforeEach, describe, expect, test } from 'bun:test'
|
||||
import { mkdtemp, rm } from 'fs/promises'
|
||||
import { tmpdir } from 'os'
|
||||
import { join } from 'path'
|
||||
import { ERROR_MESSAGE_USER_ABORT } from '../../services/compact/compact.js'
|
||||
import type { Message } from '../../types/message.js'
|
||||
import {
|
||||
createAssistantAPIErrorMessage,
|
||||
createAssistantMessage,
|
||||
createUserMessage,
|
||||
INTERRUPT_MESSAGE,
|
||||
} from '../messages.js'
|
||||
import { jsonStringify } from '../slowOperations.js'
|
||||
import { readMailbox, writeToMailbox } from '../teammateMailbox.js'
|
||||
import {
|
||||
classifyTeammateTurnFailure,
|
||||
getTeammateTurnResult,
|
||||
takeTeammateMailbox,
|
||||
} from './inProcessRunner.js'
|
||||
|
||||
function sendMessageCall(id: string, to: string, text = 'Report'): Message {
|
||||
return createAssistantMessage({
|
||||
content: [
|
||||
{ type: 'text', text },
|
||||
{
|
||||
type: 'tool_use',
|
||||
id,
|
||||
name: 'SendMessage',
|
||||
input: { to, summary: 'report', message: 'All done' },
|
||||
},
|
||||
] as never,
|
||||
})
|
||||
}
|
||||
|
||||
function toolResult(id: string, body: unknown, isError = false): Message {
|
||||
return createUserMessage({
|
||||
content: [
|
||||
{
|
||||
type: 'tool_result',
|
||||
tool_use_id: id,
|
||||
is_error: isError,
|
||||
content: [{ type: 'text', text: jsonStringify(body) }],
|
||||
},
|
||||
] as never,
|
||||
})
|
||||
}
|
||||
|
||||
describe('teammate turn result', () => {
|
||||
test('is the newest assistant text since the turn prompt', () => {
|
||||
expect(getTeammateTurnResult([
|
||||
createUserMessage({ content: 'Earlier prompt' }),
|
||||
createAssistantMessage({ content: 'Earlier answer' }),
|
||||
createUserMessage({ content: 'Current prompt' }),
|
||||
createAssistantMessage({ content: 'Looking at it' }),
|
||||
toolResult('read-1', { ok: true }),
|
||||
createAssistantMessage({ content: 'Fixed and verified.' }),
|
||||
])).toBe('Fixed and verified.')
|
||||
})
|
||||
|
||||
test('is absent when the turn produced no text of its own', () => {
|
||||
expect(getTeammateTurnResult([
|
||||
createUserMessage({ content: 'Earlier prompt' }),
|
||||
createAssistantMessage({ content: 'Earlier answer' }),
|
||||
createUserMessage({ content: 'Current prompt' }),
|
||||
createAssistantAPIErrorMessage({ content: 'API Error: 529 Overloaded' }),
|
||||
])).toBeUndefined()
|
||||
})
|
||||
|
||||
test('is suppressed when the turn already reported to the lead', () => {
|
||||
expect(getTeammateTurnResult([
|
||||
createUserMessage({ content: 'Prompt' }),
|
||||
sendMessageCall('send-1', 'team-lead', 'Reporting now'),
|
||||
toolResult('send-1', { success: true, message: 'sent' }),
|
||||
])).toBeUndefined()
|
||||
})
|
||||
|
||||
test('keeps the text when the report to the lead did not go through', () => {
|
||||
expect(getTeammateTurnResult([
|
||||
createUserMessage({ content: 'Prompt' }),
|
||||
sendMessageCall('send-1', 'team-lead', 'Reporting now'),
|
||||
toolResult('send-1', { success: false, message: 'Failed to write' }),
|
||||
])).toBe('Reporting now')
|
||||
})
|
||||
|
||||
test('keeps text written after the report, and ignores peer messages', () => {
|
||||
expect(getTeammateTurnResult([
|
||||
createUserMessage({ content: 'Prompt' }),
|
||||
sendMessageCall('send-1', 'team-lead'),
|
||||
toolResult('send-1', { success: true }),
|
||||
createAssistantMessage({ content: 'One more finding.' }),
|
||||
])).toBe('One more finding.')
|
||||
expect(getTeammateTurnResult([
|
||||
createUserMessage({ content: 'Prompt' }),
|
||||
sendMessageCall('send-2', 'reviewer', 'Asked the reviewer'),
|
||||
toolResult('send-2', { success: true }),
|
||||
])).toBe('Asked the reviewer')
|
||||
})
|
||||
})
|
||||
|
||||
describe('teammate turn failure', () => {
|
||||
test('classifies the API error that ended a turn', () => {
|
||||
expect(classifyTeammateTurnFailure([
|
||||
createAssistantMessage({ content: 'Working' }),
|
||||
createAssistantAPIErrorMessage({ content: 'API Error: 529 Overloaded\nretry later' }),
|
||||
])).toEqual({ reason: 'API Error: 529 Overloaded', isTransient: true })
|
||||
expect(classifyTeammateTurnFailure([
|
||||
createAssistantAPIErrorMessage({ content: 'API Error: 401 Invalid API key' }),
|
||||
])).toEqual({ reason: 'API Error: 401 Invalid API key', isTransient: false })
|
||||
})
|
||||
|
||||
test('ignores turns that ended normally or were cancelled', () => {
|
||||
expect(classifyTeammateTurnFailure([
|
||||
createAssistantAPIErrorMessage({ content: 'API Error: 529 Overloaded' }),
|
||||
createAssistantMessage({ content: 'Recovered' }),
|
||||
])).toBeUndefined()
|
||||
expect(classifyTeammateTurnFailure([
|
||||
createAssistantAPIErrorMessage({ content: INTERRUPT_MESSAGE }),
|
||||
])).toBeUndefined()
|
||||
expect(classifyTeammateTurnFailure([
|
||||
createAssistantAPIErrorMessage({ content: ERROR_MESSAGE_USER_ABORT }),
|
||||
])).toBeUndefined()
|
||||
})
|
||||
})
|
||||
|
||||
describe('teammate mailbox drain', () => {
|
||||
const TEAM = 'drain-team'
|
||||
let configDir: string
|
||||
let originalConfigDir: string | undefined
|
||||
|
||||
beforeEach(async () => {
|
||||
originalConfigDir = process.env.CLAUDE_CONFIG_DIR
|
||||
configDir = await mkdtemp(join(tmpdir(), 'cc-haha-runner-drain-'))
|
||||
process.env.CLAUDE_CONFIG_DIR = configDir
|
||||
})
|
||||
|
||||
afterEach(async () => {
|
||||
if (originalConfigDir === undefined) delete process.env.CLAUDE_CONFIG_DIR
|
||||
else process.env.CLAUDE_CONFIG_DIR = originalConfigDir
|
||||
await rm(configDir, { recursive: true, force: true })
|
||||
})
|
||||
|
||||
async function mail(from: string, text: string): Promise<void> {
|
||||
expect(await writeToMailbox('worker', { from, text, timestamp: new Date().toISOString() }, TEAM)).toBe(true)
|
||||
}
|
||||
|
||||
test('hands over a shutdown request alone and first', async () => {
|
||||
await mail('reviewer', 'Peer chatter')
|
||||
await mail('team-lead', jsonStringify({
|
||||
type: 'shutdown_request',
|
||||
requestId: 'shutdown-1',
|
||||
from: 'team-lead',
|
||||
timestamp: new Date().toISOString(),
|
||||
}))
|
||||
|
||||
const first = await takeTeammateMailbox({ agentName: 'worker', teamName: TEAM })
|
||||
expect(first).toMatchObject({ type: 'shutdown_request', from: 'team-lead' })
|
||||
// The rest waits for the model's decision
|
||||
const second = await takeTeammateMailbox({ agentName: 'worker', teamName: TEAM })
|
||||
expect(second).toEqual({
|
||||
type: 'new_messages',
|
||||
messages: [expect.objectContaining({ from: 'reviewer', text: 'Peer chatter' })],
|
||||
})
|
||||
expect(await takeTeammateMailbox({ agentName: 'worker', teamName: TEAM })).toBeNull()
|
||||
})
|
||||
|
||||
test('delivers nothing twice and leaves an empty inbox empty', async () => {
|
||||
expect(await takeTeammateMailbox({ agentName: 'worker', teamName: TEAM })).toBeNull()
|
||||
await mail('team-lead', 'One')
|
||||
await mail('reviewer', 'Two')
|
||||
|
||||
const delivery = await takeTeammateMailbox({ agentName: 'worker', teamName: TEAM })
|
||||
expect(delivery).toEqual({
|
||||
type: 'new_messages',
|
||||
messages: [
|
||||
expect.objectContaining({ text: 'One' }),
|
||||
expect.objectContaining({ text: 'Two' }),
|
||||
],
|
||||
})
|
||||
expect(await takeTeammateMailbox({ agentName: 'worker', teamName: TEAM })).toBeNull()
|
||||
expect((await readMailbox('worker', TEAM)).filter(message => !message.read)).toEqual([])
|
||||
})
|
||||
})
|
||||
@@ -0,0 +1,241 @@
|
||||
import { afterEach, beforeEach, describe, expect, mock, spyOn, test } from 'bun:test'
|
||||
import { mkdtemp, rm } from 'fs/promises'
|
||||
import { tmpdir } from 'os'
|
||||
import { join } from 'path'
|
||||
import {
|
||||
resetStateForTests,
|
||||
setIsInteractive,
|
||||
switchSession,
|
||||
} from '../../bootstrap/state.js'
|
||||
import * as promptsModule from '../../constants/prompts.js'
|
||||
import type { AppState } from '../../state/AppState.js'
|
||||
import type { ToolUseContext } from '../../Tool.js'
|
||||
import type { InProcessTeammateTaskState } from '../../tasks/InProcessTeammateTask/types.js'
|
||||
import type { CustomAgentDefinition } from '../../tools/AgentTool/loadAgentsDir.js'
|
||||
import * as runAgentModule from '../../tools/AgentTool/runAgent.js'
|
||||
import { spawnTeammate } from '../../tools/shared/spawnMultiAgent.js'
|
||||
import { asAgentId, type SessionId } from '../../types/ids.js'
|
||||
import { createFileStateCacheWithSizeLimit } from '../fileStateCache.js'
|
||||
import { drainSdkEvents } from '../sdkEventQueue.js'
|
||||
import {
|
||||
type AgentMetadata,
|
||||
readAgentMetadata,
|
||||
resetProjectForTesting,
|
||||
writeAgentMetadata,
|
||||
} from '../sessionStorage.js'
|
||||
import {
|
||||
hasResumableInProcessTeammate,
|
||||
resumeInProcessTeammate,
|
||||
} from './inProcessRunner.js'
|
||||
import { readTeamFile, type TeamFile, writeTeamFileAsync } from './teamHelpers.js'
|
||||
|
||||
const TEAM = 'resume-team'
|
||||
type RunAgentInput = Parameters<typeof runAgentModule.runAgent>[0]
|
||||
|
||||
const savedEnv = {
|
||||
configDir: process.env.CLAUDE_CONFIG_DIR,
|
||||
persistence: process.env.TEST_ENABLE_SESSION_PERSISTENCE,
|
||||
userType: process.env.USER_TYPE,
|
||||
}
|
||||
let configDir: string
|
||||
|
||||
function restoreEnv(name: string, value: string | undefined): void {
|
||||
if (value === undefined) delete process.env[name]
|
||||
else process.env[name] = value
|
||||
}
|
||||
|
||||
async function writeTeam(
|
||||
members: TeamFile['members'],
|
||||
createdAt = Date.now(),
|
||||
): Promise<void> {
|
||||
await writeTeamFileAsync(TEAM, {
|
||||
name: TEAM,
|
||||
createdAt,
|
||||
leadAgentId: `team-lead@${TEAM}`,
|
||||
members,
|
||||
})
|
||||
}
|
||||
|
||||
const reviewer: CustomAgentDefinition = {
|
||||
agentType: 'deep-reviewer',
|
||||
whenToUse: 'Review deeply',
|
||||
rawSystemPrompt: 'Review carefully.',
|
||||
getSystemPrompt: () => 'Review carefully.',
|
||||
source: 'projectSettings',
|
||||
tools: ['Read'],
|
||||
}
|
||||
|
||||
beforeEach(async () => {
|
||||
configDir = await mkdtemp(join(tmpdir(), 'cc-haha-runner-resume-'))
|
||||
process.env.CLAUDE_CONFIG_DIR = configDir
|
||||
process.env.TEST_ENABLE_SESSION_PERSISTENCE = '1'
|
||||
process.env.USER_TYPE = 'external'
|
||||
resetStateForTests()
|
||||
setIsInteractive(false)
|
||||
switchSession('9e8d7c6b-5a49-4382-9716-a5b4c3d2e1f0' as SessionId)
|
||||
resetProjectForTesting()
|
||||
spyOn(promptsModule, 'getSystemPrompt').mockResolvedValue(['System prompt'])
|
||||
})
|
||||
|
||||
afterEach(async () => {
|
||||
mock.restore()
|
||||
drainSdkEvents()
|
||||
resetProjectForTesting()
|
||||
resetStateForTests()
|
||||
restoreEnv('CLAUDE_CONFIG_DIR', savedEnv.configDir)
|
||||
restoreEnv('TEST_ENABLE_SESSION_PERSISTENCE', savedEnv.persistence)
|
||||
restoreEnv('USER_TYPE', savedEnv.userType)
|
||||
await rm(configDir, { recursive: true, force: true })
|
||||
})
|
||||
|
||||
function leadContext() {
|
||||
let state = {
|
||||
mainLoopModel: 'claude-sonnet-4-6',
|
||||
toolPermissionContext: { mode: 'default' },
|
||||
teamContext: {
|
||||
teamName: TEAM,
|
||||
teamFilePath: '',
|
||||
leadAgentId: `team-lead@${TEAM}`,
|
||||
teammates: {},
|
||||
},
|
||||
agentNameRegistry: new Map(),
|
||||
tasks: {},
|
||||
} as unknown as AppState
|
||||
const context = {
|
||||
getAppState: () => state,
|
||||
setAppState: (updater: (previous: AppState) => AppState) => {
|
||||
state = updater(state)
|
||||
},
|
||||
options: {
|
||||
agentDefinitions: { activeAgents: [reviewer], allAgents: [reviewer] },
|
||||
tools: [],
|
||||
mainLoopModel: 'claude-sonnet-4-6',
|
||||
mcpClients: [],
|
||||
thinkingConfig: { type: 'disabled' },
|
||||
},
|
||||
abortController: new AbortController(),
|
||||
readFileState: createFileStateCacheWithSizeLimit(10),
|
||||
messages: [],
|
||||
} as unknown as ToolUseContext
|
||||
const teammates = () =>
|
||||
Object.values(state.tasks).filter(
|
||||
(task): task is InProcessTeammateTaskState =>
|
||||
task.type === 'in_process_teammate',
|
||||
)
|
||||
// Every teammate run ends after one turn
|
||||
const calls: RunAgentInput[] = []
|
||||
spyOn(runAgentModule, 'runAgent').mockImplementation(async function* (
|
||||
input: RunAgentInput,
|
||||
) {
|
||||
calls.push(input)
|
||||
for (const task of teammates()) task.abortController?.abort()
|
||||
yield* []
|
||||
})
|
||||
return { context, teammates, calls }
|
||||
}
|
||||
|
||||
async function waitFor(predicate: () => boolean): Promise<void> {
|
||||
for (let attempt = 0; attempt < 300; attempt++) {
|
||||
if (predicate()) return
|
||||
await new Promise(resolve => setTimeout(resolve, 10))
|
||||
}
|
||||
throw new Error('condition not reached')
|
||||
}
|
||||
|
||||
async function resumeWorker(lead: ReturnType<typeof leadContext>, name = 'worker') {
|
||||
const before = lead.calls.length
|
||||
const outcome = await resumeInProcessTeammate({
|
||||
agentName: name,
|
||||
teamName: TEAM,
|
||||
prompt: 'Carry on',
|
||||
from: 'team-lead',
|
||||
context: lead.context,
|
||||
})
|
||||
await waitFor(() => lead.calls.length === before + 1 && lead.teammates().length === 0)
|
||||
return { outcome, run: lead.calls.at(-1)! }
|
||||
}
|
||||
|
||||
describe('resuming an in-process teammate', () => {
|
||||
test('reads metadata written before the teammate fields existed', async () => {
|
||||
await writeTeam([])
|
||||
const lead = leadContext()
|
||||
await spawnTeammate({ name: 'worker', prompt: 'Start', team_name: TEAM }, lead.context)
|
||||
await waitFor(() => lead.calls.length === 1 && lead.teammates().length === 0)
|
||||
const transcriptId = asAgentId(lead.calls[0]!.override!.agentId!)
|
||||
|
||||
// A sidecar from an older version: no taskKind, no permission mode
|
||||
await writeAgentMetadata(transcriptId, { agentType: 'worker', description: 'Start' })
|
||||
expect(await readAgentMetadata(transcriptId)).toEqual({ agentType: 'worker', description: 'Start' })
|
||||
const fromOld = await resumeWorker(lead)
|
||||
expect(fromOld.outcome).toMatchObject({ kind: 'resumed' })
|
||||
expect(fromOld.run.override?.agentId).toBe(transcriptId)
|
||||
expect(fromOld.run.agentDefinition.permissionMode).toBe('default')
|
||||
|
||||
const teammateMetadata: AgentMetadata = {
|
||||
agentType: 'worker',
|
||||
taskKind: 'in_process_teammate',
|
||||
teamName: TEAM,
|
||||
name: 'worker',
|
||||
permissionMode: 'acceptEdits',
|
||||
}
|
||||
await writeAgentMetadata(transcriptId, teammateMetadata)
|
||||
expect((await resumeWorker(lead)).run.agentDefinition.permissionMode).toBe('acceptEdits')
|
||||
|
||||
// A file on disk never grants bypassPermissions
|
||||
await writeAgentMetadata(transcriptId, { ...teammateMetadata, permissionMode: 'bypassPermissions' })
|
||||
expect((await resumeWorker(lead)).run.agentDefinition.permissionMode).toBe('default')
|
||||
})
|
||||
|
||||
test('restarts a roster member this process never ran without earlier conversation', async () => {
|
||||
await writeTeam([
|
||||
{
|
||||
agentId: `scout@${TEAM}`,
|
||||
name: 'scout',
|
||||
agentType: 'deep-reviewer',
|
||||
model: 'gpt-5.6-luna',
|
||||
color: 'green',
|
||||
joinedAt: Date.now(),
|
||||
tmuxPaneId: 'in-process',
|
||||
cwd: configDir,
|
||||
subscriptions: [],
|
||||
backendType: 'in-process',
|
||||
},
|
||||
])
|
||||
const lead = leadContext()
|
||||
|
||||
const { outcome, run } = await resumeWorker(lead, 'scout')
|
||||
|
||||
expect(outcome).toMatchObject({ kind: 'resumed', resumedMessageCount: 0 })
|
||||
expect(run.forkContextMessages).toBeUndefined()
|
||||
expect(run.model).toBe('gpt-5.6-luna')
|
||||
expect(run.agentDefinition.tools).toContain('Read')
|
||||
expect(run.extraMetadata).toMatchObject({
|
||||
color: 'green',
|
||||
customAgentType: 'deep-reviewer',
|
||||
})
|
||||
})
|
||||
|
||||
test('does not bring back a teammate of an earlier team with the same name', async () => {
|
||||
await writeTeam([])
|
||||
const lead = leadContext()
|
||||
await spawnTeammate({ name: 'worker', prompt: 'Start', team_name: TEAM }, lead.context)
|
||||
await waitFor(() => lead.calls.length === 1 && lead.teammates().length === 0)
|
||||
const original = readTeamFile(TEAM)!
|
||||
expect(hasResumableInProcessTeammate('worker', TEAM, original.createdAt)).toBe(true)
|
||||
|
||||
// The team is deleted and created again under the same name
|
||||
await writeTeam([], original.createdAt + 1)
|
||||
|
||||
expect(hasResumableInProcessTeammate('worker', TEAM, original.createdAt + 1)).toBe(false)
|
||||
await expect(
|
||||
resumeInProcessTeammate({
|
||||
agentName: 'worker',
|
||||
teamName: TEAM,
|
||||
prompt: 'Carry on',
|
||||
from: 'team-lead',
|
||||
context: lead.context,
|
||||
}),
|
||||
).rejects.toThrow('not an in-process teammate')
|
||||
expect(lead.teammates()).toEqual([])
|
||||
})
|
||||
})
|
||||
@@ -0,0 +1,585 @@
|
||||
import { afterEach, beforeEach, describe, expect, mock, spyOn, test } from 'bun:test'
|
||||
import { mkdtemp, rm } from 'fs/promises'
|
||||
import { tmpdir } from 'os'
|
||||
import { join } from 'path'
|
||||
import { resetStateForTests, setIsInteractive } from '../../bootstrap/state.js'
|
||||
import * as promptsModule from '../../constants/prompts.js'
|
||||
import * as autoCompactModule from '../../services/compact/autoCompact.js'
|
||||
import * as compactModule from '../../services/compact/compact.js'
|
||||
import type { AppState } from '../../state/AppState.js'
|
||||
import type { ToolUseContext } from '../../Tool.js'
|
||||
import type {
|
||||
InProcessTeammateTaskState,
|
||||
TeammateIdentity,
|
||||
} from '../../tasks/InProcessTeammateTask/types.js'
|
||||
import * as runAgentModule from '../../tools/AgentTool/runAgent.js'
|
||||
import type { Message } from '../../types/message.js'
|
||||
import { AbortError } from '../errors.js'
|
||||
import { createFileStateCacheWithSizeLimit } from '../fileStateCache.js'
|
||||
import {
|
||||
createAssistantAPIErrorMessage,
|
||||
createAssistantMessage,
|
||||
createCompactBoundaryMessage,
|
||||
createUserMessage,
|
||||
} from '../messages.js'
|
||||
import { drainSdkEvents } from '../sdkEventQueue.js'
|
||||
import { jsonStringify } from '../slowOperations.js'
|
||||
import { createTeammateContext } from '../teammateContext.js'
|
||||
import {
|
||||
type IdleNotificationMessage,
|
||||
isIdleNotification,
|
||||
readMailbox,
|
||||
writeToMailbox,
|
||||
} from '../teammateMailbox.js'
|
||||
import { runInProcessTeammate } from './inProcessRunner.js'
|
||||
import { writeTeamFileAsync } from './teamHelpers.js'
|
||||
|
||||
const TEAM = 'runtime-team'
|
||||
const NAME = 'worker'
|
||||
const TASK_ID = 'in-process-worker'
|
||||
const TRANSCRIPT_ID = 'aworker-0123456789abcdef'
|
||||
|
||||
type RunAgentInput = Parameters<typeof runAgentModule.runAgent>[0]
|
||||
type Turn = (input: RunAgentInput) => Promise<Message[]> | Message[]
|
||||
|
||||
let configDir: string
|
||||
let originalConfigDir: string | undefined
|
||||
|
||||
beforeEach(async () => {
|
||||
originalConfigDir = process.env.CLAUDE_CONFIG_DIR
|
||||
configDir = await mkdtemp(join(tmpdir(), 'cc-haha-runner-runtime-'))
|
||||
process.env.CLAUDE_CONFIG_DIR = configDir
|
||||
resetStateForTests()
|
||||
setIsInteractive(false)
|
||||
drainSdkEvents()
|
||||
await writeTeamFileAsync(TEAM, {
|
||||
name: TEAM,
|
||||
createdAt: Date.now(),
|
||||
leadAgentId: `team-lead@${TEAM}`,
|
||||
leadSessionId: 'leader-session',
|
||||
members: [
|
||||
{
|
||||
agentId: `${NAME}@${TEAM}`,
|
||||
name: NAME,
|
||||
joinedAt: Date.now(),
|
||||
tmuxPaneId: 'in-process',
|
||||
cwd: configDir,
|
||||
subscriptions: [],
|
||||
backendType: 'in-process',
|
||||
},
|
||||
],
|
||||
})
|
||||
})
|
||||
|
||||
afterEach(async () => {
|
||||
mock.restore()
|
||||
drainSdkEvents()
|
||||
resetStateForTests()
|
||||
if (originalConfigDir === undefined) delete process.env.CLAUDE_CONFIG_DIR
|
||||
else process.env.CLAUDE_CONFIG_DIR = originalConfigDir
|
||||
await rm(configDir, { recursive: true, force: true })
|
||||
})
|
||||
|
||||
function text(value: string): Message {
|
||||
return createAssistantMessage({ content: value })
|
||||
}
|
||||
|
||||
function apiError(value: string): Message {
|
||||
return createAssistantAPIErrorMessage({ content: value })
|
||||
}
|
||||
|
||||
function promptOf(input: RunAgentInput): string {
|
||||
return input.promptMessages[0]!.message.content as string
|
||||
}
|
||||
|
||||
async function leadNotifications(): Promise<IdleNotificationMessage[]> {
|
||||
return (await readMailbox('team-lead', TEAM)).flatMap(message => {
|
||||
const idle = isIdleNotification(message.text)
|
||||
return idle ? [idle] : []
|
||||
})
|
||||
}
|
||||
|
||||
async function mailWorker(from: string, text: string): Promise<void> {
|
||||
expect(
|
||||
await writeToMailbox(
|
||||
NAME,
|
||||
{ from, text, timestamp: new Date().toISOString() },
|
||||
TEAM,
|
||||
),
|
||||
).toBe(true)
|
||||
}
|
||||
|
||||
/**
|
||||
* Drives runInProcessTeammate with scripted runAgent turns. A call past the
|
||||
* script stops the teammate; a safety timer stops a runner that never gets
|
||||
* there, so a regression fails its assertions instead of hanging.
|
||||
*/
|
||||
function createFixture(turns: Turn[], options: { leaderAbort?: AbortController } = {}) {
|
||||
const abortController = new AbortController()
|
||||
const identity: TeammateIdentity = {
|
||||
agentId: `${NAME}@${TEAM}`,
|
||||
agentName: NAME,
|
||||
teamName: TEAM,
|
||||
planModeRequired: false,
|
||||
parentSessionId: 'leader-session',
|
||||
resumableAgentId: TRANSCRIPT_ID,
|
||||
}
|
||||
const task: InProcessTeammateTaskState = {
|
||||
id: TASK_ID,
|
||||
type: 'in_process_teammate',
|
||||
status: 'running',
|
||||
description: 'worker',
|
||||
toolUseId: 'spawn-tool',
|
||||
startTime: Date.now(),
|
||||
outputFile: join(configDir, 'worker.output'),
|
||||
outputOffset: 0,
|
||||
notified: false,
|
||||
identity,
|
||||
prompt: 'Do the work',
|
||||
abortController,
|
||||
awaitingPlanApproval: false,
|
||||
permissionMode: 'default',
|
||||
isIdle: false,
|
||||
shutdownRequested: false,
|
||||
lastReportedToolCount: 0,
|
||||
lastReportedTokenCount: 0,
|
||||
pendingUserMessages: [],
|
||||
messages: [],
|
||||
}
|
||||
let state = { tasks: { [TASK_ID]: task } } as unknown as AppState
|
||||
const idleTransitions: boolean[] = []
|
||||
const setAppState = (updater: (prev: AppState) => AppState) => {
|
||||
const before = (state.tasks[TASK_ID] as InProcessTeammateTaskState | undefined)?.isIdle
|
||||
state = updater(state)
|
||||
const after = (state.tasks[TASK_ID] as InProcessTeammateTaskState | undefined)?.isIdle
|
||||
if (after !== undefined && after !== before) idleTransitions.push(after)
|
||||
}
|
||||
const toolUseContext = {
|
||||
options: { tools: [], mainLoopModel: 'test-model', mcpClients: [] },
|
||||
abortController: options.leaderAbort ?? new AbortController(),
|
||||
readFileState: createFileStateCacheWithSizeLimit(10),
|
||||
getAppState: () => state,
|
||||
setAppState,
|
||||
} as unknown as ToolUseContext
|
||||
|
||||
const calls: RunAgentInput[] = []
|
||||
spyOn(runAgentModule, 'runAgent').mockImplementation(async function* (
|
||||
input: RunAgentInput,
|
||||
) {
|
||||
calls.push(input)
|
||||
const turn = turns[calls.length - 1]
|
||||
if (!turn) {
|
||||
abortController.abort()
|
||||
return
|
||||
}
|
||||
for (const message of await turn(input)) yield message
|
||||
})
|
||||
|
||||
return {
|
||||
abortController,
|
||||
identity,
|
||||
calls,
|
||||
idleTransitions,
|
||||
getState: () => state,
|
||||
/** Input typed into the teammate's view: picked up at its next idle poll */
|
||||
queueInput(value: string) {
|
||||
setAppState(prev => {
|
||||
const current = prev.tasks[TASK_ID] as InProcessTeammateTaskState
|
||||
return {
|
||||
...prev,
|
||||
tasks: {
|
||||
...prev.tasks,
|
||||
[TASK_ID]: {
|
||||
...current,
|
||||
pendingUserMessages: [...current.pendingUserMessages, value],
|
||||
},
|
||||
},
|
||||
}
|
||||
})
|
||||
},
|
||||
async run(overrides: Partial<Parameters<typeof runInProcessTeammate>[0]> = {}) {
|
||||
const safety = setTimeout(() => abortController.abort(), 3_000)
|
||||
try {
|
||||
return await runInProcessTeammate({
|
||||
identity,
|
||||
taskId: TASK_ID,
|
||||
prompt: 'Do the work',
|
||||
teammateContext: createTeammateContext({ ...identity, abortController }),
|
||||
toolUseContext,
|
||||
abortController,
|
||||
systemPrompt: 'Teammate prompt',
|
||||
systemPromptMode: 'replace',
|
||||
autoContinueDelaysMs: [10, 10],
|
||||
...overrides,
|
||||
})
|
||||
} finally {
|
||||
clearTimeout(safety)
|
||||
}
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
describe('in-process teammate compaction', () => {
|
||||
function fakeCompaction() {
|
||||
const boundary = createCompactBoundaryMessage('auto', 100)
|
||||
const summary = createUserMessage({
|
||||
content: 'Summary of the earlier work',
|
||||
isCompactSummary: true,
|
||||
})
|
||||
return {
|
||||
boundary,
|
||||
result: {
|
||||
boundaryMarker: boundary,
|
||||
summaryMessages: [summary],
|
||||
attachments: [],
|
||||
hookResults: [],
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
test('summarizes the real conversation under the teammate own controller after the lead turn was interrupted', async () => {
|
||||
const leaderAbort = new AbortController()
|
||||
leaderAbort.abort()
|
||||
spyOn(autoCompactModule, 'getAutoCompactThreshold').mockReturnValue(0)
|
||||
const { boundary, result } = fakeCompaction()
|
||||
let compactArgs: Parameters<typeof compactModule.compactConversation> | undefined
|
||||
spyOn(compactModule, 'compactConversation').mockImplementation(
|
||||
async (...args: Parameters<typeof compactModule.compactConversation>) => {
|
||||
compactArgs = [args[0], args[1], { ...args[2], forkContextMessages: [...args[2].forkContextMessages] }, ...args.slice(3)] as never
|
||||
return result as never
|
||||
},
|
||||
)
|
||||
const firstAnswer = text('Found the failing test in parser.ts')
|
||||
let secondTurnContext: Message[] | undefined
|
||||
const fixture = createFixture(
|
||||
[
|
||||
() => {
|
||||
fixture.queueInput('Now fix it')
|
||||
return [firstAnswer]
|
||||
},
|
||||
input => {
|
||||
secondTurnContext = input.forkContextMessages
|
||||
fixture.abortController.abort()
|
||||
return []
|
||||
},
|
||||
],
|
||||
{ leaderAbort },
|
||||
)
|
||||
|
||||
const outcome = await fixture.run()
|
||||
|
||||
expect(outcome.success).toBe(true)
|
||||
expect(compactArgs).toBeDefined()
|
||||
const [, context, cacheSafeParams] = compactArgs!
|
||||
// Never the lead's (already aborted) per-turn controller
|
||||
expect(context.abortController).toBe(fixture.abortController)
|
||||
expect(context.agentId).toBe(TRANSCRIPT_ID)
|
||||
// The summary fork must see the conversation, not an empty one
|
||||
expect(cacheSafeParams.forkContextMessages.map(m => m.uuid)).toContain(firstAnswer.uuid)
|
||||
expect(secondTurnContext?.[0]?.uuid).toBe(boundary.uuid)
|
||||
expect(await leadNotifications()).not.toContainEqual(
|
||||
expect.objectContaining({ idleReason: 'failed' }),
|
||||
)
|
||||
})
|
||||
|
||||
test('keeps the teammate working on the full history when compaction fails', async () => {
|
||||
spyOn(autoCompactModule, 'getAutoCompactThreshold').mockReturnValue(0)
|
||||
spyOn(compactModule, 'compactConversation').mockRejectedValue(
|
||||
new Error('API Error: 500 {"type":"api_error"}'),
|
||||
)
|
||||
const firstAnswer = text('First pass done')
|
||||
let secondTurnContext: Message[] | undefined
|
||||
const fixture = createFixture([
|
||||
() => {
|
||||
fixture.queueInput('Continue')
|
||||
return [firstAnswer]
|
||||
},
|
||||
input => {
|
||||
secondTurnContext = input.forkContextMessages
|
||||
fixture.abortController.abort()
|
||||
return []
|
||||
},
|
||||
])
|
||||
|
||||
const outcome = await fixture.run()
|
||||
|
||||
expect(outcome.success).toBe(true)
|
||||
expect(fixture.calls).toHaveLength(2)
|
||||
expect(secondTurnContext?.map(m => m.uuid)).toContain(firstAnswer.uuid)
|
||||
expect(await leadNotifications()).not.toContainEqual(
|
||||
expect.objectContaining({ idleReason: 'failed' }),
|
||||
)
|
||||
})
|
||||
|
||||
test('exits cleanly when the teammate is stopped during compaction', async () => {
|
||||
spyOn(autoCompactModule, 'getAutoCompactThreshold').mockReturnValue(0)
|
||||
const fixture = createFixture([
|
||||
() => {
|
||||
fixture.queueInput('Continue')
|
||||
return [text('First pass done')]
|
||||
},
|
||||
])
|
||||
spyOn(compactModule, 'compactConversation').mockImplementation(async () => {
|
||||
fixture.abortController.abort()
|
||||
throw new Error('Request was aborted.')
|
||||
})
|
||||
|
||||
const outcome = await fixture.run()
|
||||
|
||||
expect(outcome.success).toBe(true)
|
||||
expect(fixture.calls).toHaveLength(1)
|
||||
expect(drainSdkEvents()).toContainEqual(
|
||||
expect.objectContaining({ subtype: 'task_notification', task_id: TASK_ID, status: 'completed' }),
|
||||
)
|
||||
expect(await leadNotifications()).not.toContainEqual(
|
||||
expect.objectContaining({ idleReason: 'failed' }),
|
||||
)
|
||||
})
|
||||
})
|
||||
|
||||
describe('in-process teammate turn failures', () => {
|
||||
test('reports a non-transient API error to the lead as a failed turn', async () => {
|
||||
const fixture = createFixture([
|
||||
() => {
|
||||
fixture.queueInput('Try again later')
|
||||
return [apiError('API Error: 401 Invalid API key · Please run /login')]
|
||||
},
|
||||
])
|
||||
|
||||
await fixture.run()
|
||||
|
||||
expect(await leadNotifications()).toEqual([
|
||||
expect.objectContaining({
|
||||
idleReason: 'failed',
|
||||
failureReason: 'API Error: 401 Invalid API key · Please run /login',
|
||||
}),
|
||||
])
|
||||
})
|
||||
|
||||
test('continues after a transient failure without telling the lead, then reports the result', async () => {
|
||||
const fixture = createFixture([
|
||||
() => [apiError('API Error: 529 Overloaded')],
|
||||
() => {
|
||||
fixture.queueInput('Thanks')
|
||||
return [text('Recovered and finished the migration.')]
|
||||
},
|
||||
])
|
||||
|
||||
await fixture.run()
|
||||
|
||||
expect(fixture.calls.length).toBeGreaterThanOrEqual(2)
|
||||
expect(promptOf(fixture.calls[1]!)).toContain(
|
||||
'automatic retry 1 of 2',
|
||||
)
|
||||
expect(promptOf(fixture.calls[1]!)).toContain('API Error: 529 Overloaded')
|
||||
expect(await leadNotifications()).toEqual([
|
||||
expect.objectContaining({
|
||||
idleReason: 'available',
|
||||
result: 'Recovered and finished the migration.',
|
||||
}),
|
||||
])
|
||||
})
|
||||
|
||||
test('tells the lead only once the automatic retries are exhausted', async () => {
|
||||
const fixture = createFixture([
|
||||
() => [apiError('API Error: 529 Overloaded')],
|
||||
() => {
|
||||
fixture.queueInput('Status?')
|
||||
return [apiError('API Error: 529 Overloaded')]
|
||||
},
|
||||
])
|
||||
|
||||
await fixture.run({ autoContinueDelaysMs: [10] })
|
||||
|
||||
expect(fixture.calls.length).toBeGreaterThanOrEqual(2)
|
||||
expect(promptOf(fixture.calls[1]!)).toContain('automatic retry 1 of 1')
|
||||
expect(await leadNotifications()).toEqual([
|
||||
expect.objectContaining({
|
||||
idleReason: 'failed',
|
||||
failureReason:
|
||||
'API Error: 529 Overloaded (automatic retries exhausted; message worker to continue)',
|
||||
}),
|
||||
])
|
||||
})
|
||||
|
||||
test('new mail cuts the retry backoff short and is delivered instead of the retry', async () => {
|
||||
const fixture = createFixture([
|
||||
async () => {
|
||||
await mailWorker('team-lead', 'Switch to the docs task instead')
|
||||
return [apiError('API Error: 529 Overloaded')]
|
||||
},
|
||||
() => {
|
||||
fixture.abortController.abort()
|
||||
return []
|
||||
},
|
||||
])
|
||||
const started = Date.now()
|
||||
|
||||
await fixture.run({ autoContinueDelaysMs: [60_000] })
|
||||
|
||||
expect(Date.now() - started).toBeLessThan(2_500)
|
||||
expect(fixture.calls).toHaveLength(2)
|
||||
expect(promptOf(fixture.calls[1]!)).toContain('Switch to the docs task instead')
|
||||
expect(promptOf(fixture.calls[1]!)).not.toContain('automatic retry')
|
||||
// The lead hears nothing about a failure that is being handled
|
||||
expect(await leadNotifications()).toEqual([])
|
||||
})
|
||||
|
||||
test('treats an AbortError after the user stopped the turn as an interruption', async () => {
|
||||
const fixture = createFixture([
|
||||
input => {
|
||||
fixture.queueInput('Carry on')
|
||||
input.override?.abortController?.abort()
|
||||
throw new AbortError()
|
||||
},
|
||||
])
|
||||
|
||||
const outcome = await fixture.run()
|
||||
|
||||
expect(outcome.success).toBe(true)
|
||||
// The teammate survived and took the next prompt
|
||||
expect(fixture.calls).toHaveLength(2)
|
||||
expect(await leadNotifications()).toEqual([
|
||||
expect.objectContaining({ idleReason: 'interrupted' }),
|
||||
])
|
||||
})
|
||||
|
||||
test('settles the task when its system prompt cannot be built', async () => {
|
||||
spyOn(promptsModule, 'getSystemPrompt').mockRejectedValue(
|
||||
new Error('prompt build failed'),
|
||||
)
|
||||
const fixture = createFixture([])
|
||||
|
||||
const outcome = await fixture.run({ systemPrompt: undefined, systemPromptMode: undefined })
|
||||
|
||||
expect(outcome).toMatchObject({ success: false, error: 'prompt build failed' })
|
||||
expect(fixture.calls).toHaveLength(0)
|
||||
expect(drainSdkEvents()).toContainEqual(
|
||||
expect.objectContaining({ subtype: 'task_notification', task_id: TASK_ID, status: 'failed' }),
|
||||
)
|
||||
expect(await leadNotifications()).toEqual([
|
||||
expect.objectContaining({ idleReason: 'failed', failureReason: 'prompt build failed' }),
|
||||
])
|
||||
})
|
||||
})
|
||||
|
||||
describe('in-process teammate results and mail', () => {
|
||||
test('sends the final response of a successful turn with its idle notification', async () => {
|
||||
const fixture = createFixture([
|
||||
() => {
|
||||
fixture.queueInput('Next')
|
||||
return [text('The parser bug is fixed; all 42 tests pass.')]
|
||||
},
|
||||
])
|
||||
|
||||
await fixture.run()
|
||||
|
||||
expect(await leadNotifications()).toEqual([
|
||||
expect.objectContaining({
|
||||
idleReason: 'available',
|
||||
result: 'The parser bug is fixed; all 42 tests pass.',
|
||||
}),
|
||||
])
|
||||
})
|
||||
|
||||
test('takes mail that arrived during a turn as one batch without going idle in between', async () => {
|
||||
let notificationsAtSecondTurn: IdleNotificationMessage[] = []
|
||||
let idleTransitionsAtSecondTurn: boolean[] = []
|
||||
const fixture = createFixture([
|
||||
async () => {
|
||||
await mailWorker('reviewer', 'Found a bug in the lexer')
|
||||
await mailWorker('team-lead', 'Also update the docs')
|
||||
return [text('Turn one done')]
|
||||
},
|
||||
async () => {
|
||||
notificationsAtSecondTurn = await leadNotifications()
|
||||
idleTransitionsAtSecondTurn = [...fixture.idleTransitions]
|
||||
fixture.abortController.abort()
|
||||
return []
|
||||
},
|
||||
])
|
||||
|
||||
await fixture.run()
|
||||
|
||||
const secondPrompt = promptOf(fixture.calls[1]!)
|
||||
expect(secondPrompt).toContain('Found a bug in the lexer')
|
||||
expect(secondPrompt).toContain('Also update the docs')
|
||||
expect(secondPrompt.indexOf('Found a bug in the lexer')).toBeLessThan(
|
||||
secondPrompt.indexOf('Also update the docs'),
|
||||
)
|
||||
// A result-only frame, not an idle flap
|
||||
expect(notificationsAtSecondTurn).toEqual([
|
||||
expect.objectContaining({ result: 'Turn one done' }),
|
||||
])
|
||||
expect(notificationsAtSecondTurn[0]?.idleReason).toBeUndefined()
|
||||
expect(idleTransitionsAtSecondTurn).not.toContain(true)
|
||||
expect((await readMailbox(NAME, TEAM)).filter(m => !m.read)).toEqual([])
|
||||
})
|
||||
|
||||
test('never hands protocol frames to the model as prose', async () => {
|
||||
const fixture = createFixture([
|
||||
async () => {
|
||||
await mailWorker(
|
||||
'team-lead',
|
||||
jsonStringify({ type: 'permission_response', request_id: 'perm-1', subtype: 'success' }),
|
||||
)
|
||||
await mailWorker(
|
||||
'reviewer',
|
||||
jsonStringify({ type: 'mode_set_request', mode: 'bypassPermissions', from: 'reviewer' }),
|
||||
)
|
||||
await mailWorker(
|
||||
'team-lead',
|
||||
jsonStringify({ type: 'plan_approval_response', requestId: 'plan-1', approved: true, timestamp: new Date().toISOString() }),
|
||||
)
|
||||
await mailWorker('reviewer', 'Plain question about the API')
|
||||
return [text('Turn one done')]
|
||||
},
|
||||
() => {
|
||||
fixture.abortController.abort()
|
||||
return []
|
||||
},
|
||||
])
|
||||
|
||||
await fixture.run()
|
||||
|
||||
const secondPrompt = promptOf(fixture.calls[1]!)
|
||||
expect(secondPrompt).toContain('Plain question about the API')
|
||||
// The plan verdict is still delivered, as before
|
||||
expect(secondPrompt).toContain('plan_approval_response')
|
||||
expect(secondPrompt).not.toContain('permission_response')
|
||||
expect(secondPrompt).not.toContain('mode_set_request')
|
||||
expect((await readMailbox(NAME, TEAM)).filter(m => !m.read)).toEqual([])
|
||||
})
|
||||
})
|
||||
|
||||
describe('in-process teammate transcript', () => {
|
||||
test('runs every turn under one durable transcript id with team metadata', async () => {
|
||||
const fixture = createFixture([
|
||||
() => {
|
||||
fixture.queueInput('Second task')
|
||||
return [text('First done')]
|
||||
},
|
||||
() => {
|
||||
fixture.abortController.abort()
|
||||
return []
|
||||
},
|
||||
])
|
||||
|
||||
await fixture.run()
|
||||
|
||||
expect(fixture.calls).toHaveLength(2)
|
||||
const [first, second] = fixture.calls
|
||||
expect(first!.override?.agentId).toBe(TRANSCRIPT_ID)
|
||||
expect(second!.override?.agentId).toBe(TRANSCRIPT_ID)
|
||||
// runAgent appends to the same transcript instead of re-recording it
|
||||
expect(first!.recordedUuids).toBeInstanceOf(Set)
|
||||
expect(second!.recordedUuids).toBe(first!.recordedUuids)
|
||||
expect(first!.extraMetadata).toEqual({
|
||||
taskKind: 'in_process_teammate',
|
||||
teamName: TEAM,
|
||||
name: NAME,
|
||||
planModeRequired: false,
|
||||
permissionMode: 'default',
|
||||
})
|
||||
})
|
||||
})
|
||||
+1099
-309
File diff suppressed because it is too large
Load Diff
@@ -7,7 +7,8 @@ import * as sdk from '../sdkEventQueue.js'
|
||||
import * as diskOutput from '../task/diskOutput.js'
|
||||
import * as framework from '../task/framework.js'
|
||||
import * as tracing from '../telemetry/perfettoTracing.js'
|
||||
import { killInProcessTeammate } from './spawnInProcess.js'
|
||||
import { readMailbox, writeToMailbox } from '../teammateMailbox.js'
|
||||
import { killInProcessTeammate, spawnInProcessTeammate } from './spawnInProcess.js'
|
||||
import * as teamHelpers from './teamHelpers.js'
|
||||
|
||||
async function withKillFixture(run: (fixture: {
|
||||
@@ -97,3 +98,57 @@ for (const failureCode of [undefined, 'ELOCKED', 'EPERM']) {
|
||||
})
|
||||
})
|
||||
}
|
||||
|
||||
async function withSpawnFixture(run: (fixture: {
|
||||
spawn: (resumableAgentId?: string) => ReturnType<typeof spawnInProcessTeammate>
|
||||
}) => Promise<void>) {
|
||||
const directory = await mkdtemp(join(tmpdir(), 'cc-haha-spawn-teammate-'))
|
||||
const originalConfig = process.env.CLAUDE_CONFIG_DIR
|
||||
process.env.CLAUDE_CONFIG_DIR = directory
|
||||
let state = { tasks: {} } as unknown as AppState
|
||||
const spawned: AbortController[] = []
|
||||
try {
|
||||
await run({
|
||||
spawn: async resumableAgentId => {
|
||||
const result = await spawnInProcessTeammate(
|
||||
{ name: 'worker', teamName: 'spawn-team', prompt: 'Work', planModeRequired: false, resumableAgentId },
|
||||
{ setAppState: update => { state = update(state) } },
|
||||
)
|
||||
if (result.abortController) spawned.push(result.abortController)
|
||||
return result
|
||||
},
|
||||
})
|
||||
} finally {
|
||||
for (const controller of spawned) controller.abort()
|
||||
if (originalConfig === undefined) delete process.env.CLAUDE_CONFIG_DIR
|
||||
else process.env.CLAUDE_CONFIG_DIR = originalConfig
|
||||
await rm(directory, { recursive: true, force: true })
|
||||
}
|
||||
}
|
||||
|
||||
async function unreadFor(name: string): Promise<string[]> {
|
||||
return (await readMailbox(name, 'spawn-team')).filter(m => !m.read).map(m => m.text)
|
||||
}
|
||||
|
||||
test('a fresh teammate gets a durable transcript id and none of the mail left for an earlier one', async () => {
|
||||
await withSpawnFixture(async fixture => {
|
||||
await writeToMailbox('worker', { from: 'reviewer', text: 'Stale note for the old worker', timestamp: new Date().toISOString() }, 'spawn-team')
|
||||
|
||||
const result = await fixture.spawn()
|
||||
|
||||
expect(result.success).toBe(true)
|
||||
expect(result.identity?.resumableAgentId).toMatch(/^a[0-9a-f]{16}$/)
|
||||
expect(await unreadFor('worker')).toEqual([])
|
||||
})
|
||||
})
|
||||
|
||||
test('a resumed teammate keeps its transcript id and the mail that arrived while it was stopped', async () => {
|
||||
await withSpawnFixture(async fixture => {
|
||||
await writeToMailbox('worker', { from: 'team-lead', text: 'Waiting for you', timestamp: new Date().toISOString() }, 'spawn-team')
|
||||
|
||||
const result = await fixture.spawn('aworker-0123456789abcdef')
|
||||
|
||||
expect(result.identity?.resumableAgentId).toBe('aworker-0123456789abcdef')
|
||||
expect(await unreadFor('worker')).toEqual(['Waiting for you'])
|
||||
})
|
||||
})
|
||||
|
||||
@@ -27,6 +27,7 @@ import { createAbortController } from '../abortController.js'
|
||||
import { formatAgentId } from '../agentId.js'
|
||||
import { registerCleanup } from '../cleanupRegistry.js'
|
||||
import { logForDebugging } from '../debug.js'
|
||||
import type { PermissionMode } from '../permissions/PermissionMode.js'
|
||||
import { emitTaskTerminatedSdk } from '../sdkEventQueue.js'
|
||||
import { evictTaskOutput } from '../task/diskOutput.js'
|
||||
import {
|
||||
@@ -35,11 +36,13 @@ import {
|
||||
STOPPED_DISPLAY_MS,
|
||||
} from '../task/framework.js'
|
||||
import { createTeammateContext } from '../teammateContext.js'
|
||||
import { clearMailbox } from '../teammateMailbox.js'
|
||||
import {
|
||||
isPerfettoTracingEnabled,
|
||||
registerAgent as registerPerfettoAgent,
|
||||
unregisterAgent as unregisterPerfettoAgent,
|
||||
} from '../telemetry/perfettoTracing.js'
|
||||
import { createAgentId } from '../uuid.js'
|
||||
import { removeMemberByAgentId } from './teamHelpers.js'
|
||||
import { isTeamReviewRequired } from './teamPlanPolicy.js'
|
||||
import { readTeamPlan } from './teamPlanStore.js'
|
||||
@@ -71,6 +74,11 @@ export type InProcessSpawnConfig = {
|
||||
planModeRequired: boolean
|
||||
/** Optional model override for this teammate */
|
||||
model?: string
|
||||
/** Set when resuming a teammate: the agent id of its existing transcript.
|
||||
* A fresh spawn gets a new one and starts with an empty inbox. */
|
||||
resumableAgentId?: string
|
||||
/** Permission mode to resume with; defaults from planModeRequired. */
|
||||
permissionMode?: PermissionMode
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -81,6 +89,8 @@ export type InProcessSpawnOutput = {
|
||||
success: boolean
|
||||
/** Full agent ID (format: "name@team") */
|
||||
agentId: string
|
||||
/** Identity registered on the task, including its durable transcript id */
|
||||
identity?: TeammateIdentity
|
||||
/** Task ID for tracking in AppState */
|
||||
taskId?: string
|
||||
/** AbortController for this teammate (linked to parent) */
|
||||
@@ -122,6 +132,12 @@ export async function spawnInProcessTeammate(
|
||||
)
|
||||
|
||||
try {
|
||||
// A new teammate must not read mail left for an earlier one with this
|
||||
// name; a resumed teammate keeps whatever arrived while it was stopped.
|
||||
if (config.resumableAgentId === undefined) {
|
||||
await clearMailbox(name, teamName)
|
||||
}
|
||||
|
||||
// Create independent AbortController for this teammate
|
||||
// Teammates should not be aborted when the leader's query is interrupted
|
||||
const abortController = createAbortController()
|
||||
@@ -137,6 +153,7 @@ export async function spawnInProcessTeammate(
|
||||
color,
|
||||
planModeRequired,
|
||||
parentSessionId,
|
||||
resumableAgentId: config.resumableAgentId ?? createAgentId(),
|
||||
}
|
||||
|
||||
// Create teammate context for AsyncLocalStorage
|
||||
@@ -175,7 +192,8 @@ export async function spawnInProcessTeammate(
|
||||
awaitingPlanApproval: false,
|
||||
spinnerVerb: sample(getSpinnerVerbs()),
|
||||
pastTenseVerb: sample(TURN_COMPLETION_VERBS),
|
||||
permissionMode: planModeRequired ? 'plan' : 'default',
|
||||
permissionMode:
|
||||
config.permissionMode ?? (planModeRequired ? 'plan' : 'default'),
|
||||
isIdle: false,
|
||||
shutdownRequested: false,
|
||||
lastReportedToolCount: 0,
|
||||
@@ -202,6 +220,7 @@ export async function spawnInProcessTeammate(
|
||||
return {
|
||||
success: true,
|
||||
agentId,
|
||||
identity,
|
||||
taskId,
|
||||
abortController,
|
||||
teammateContext,
|
||||
|
||||
@@ -0,0 +1,24 @@
|
||||
import type { AppState } from '../../state/AppState.js'
|
||||
import { hasWorkingInProcessTeammates, isTeamLead } from '../teammate.js'
|
||||
import { TEAM_LEAD_NAME } from './constants.js'
|
||||
import { readTeamFile } from './teamHelpers.js'
|
||||
|
||||
/**
|
||||
* Whether a team lead still has members doing work it is waiting for: an
|
||||
* in-process teammate mid-turn, or a desktop process member that is mid-turn
|
||||
* or scheduled for an automatic retry (the server records both in the team
|
||||
* file). Their results reach the lead as teammate messages, which start a new
|
||||
* lead turn when the current one has ended.
|
||||
*/
|
||||
export function hasTeamWorkInProgress(appState: AppState): boolean {
|
||||
const team = appState.teamContext
|
||||
if (!team || !isTeamLead(team)) return false
|
||||
if (hasWorkingInProcessTeammates(appState)) return true
|
||||
const teamFile = readTeamFile(team.teamName)
|
||||
return !!teamFile?.members.some(member =>
|
||||
member.name !== TEAM_LEAD_NAME &&
|
||||
member.backendType === 'process' &&
|
||||
member.terminated !== true &&
|
||||
(member.isActive === true || member.autoRetry !== undefined),
|
||||
)
|
||||
}
|
||||
@@ -6,6 +6,7 @@ import { join } from 'path'
|
||||
import {
|
||||
addHiddenPaneId,
|
||||
getTeamFilePath,
|
||||
listLeadTeamMemberIdentities,
|
||||
mutateTeamFileAsync,
|
||||
readTeamFile,
|
||||
removeHiddenPaneId,
|
||||
@@ -311,3 +312,62 @@ test('setMemberActive preserves concurrent updates to different members', async
|
||||
await rm(configDir, { recursive: true, force: true })
|
||||
}
|
||||
})
|
||||
|
||||
test('a desktop-hosted lead exit keeps its reviewed team; a terminal lead exit still cleans up', async () => {
|
||||
const originalConfigDir = process.env.CLAUDE_CONFIG_DIR
|
||||
const originalReview = process.env.CC_HAHA_TEAM_REVIEW_REQUIRED
|
||||
const configDir = await mkdtemp(join(tmpdir(), 'cc-haha-team-session-cleanup-'))
|
||||
process.env.CLAUDE_CONFIG_DIR = configDir
|
||||
const { cleanupSessionTeams, registerTeamForSessionCleanup, unregisterTeamForSessionCleanup, getTeamDir } = await import('./teamHelpers.js')
|
||||
const { beginTaskListLifecycle, getCanonicalTeamTaskListId } = await import('../tasks.js')
|
||||
const teamName = 'desktop-owned'
|
||||
try {
|
||||
const createdAt = Date.now()
|
||||
const lifecycle = await beginTaskListLifecycle(getCanonicalTeamTaskListId(teamName), { teamName, createdAt, leadSessionId: 'lead-session' })
|
||||
await writeTeamFileAsync(teamName, { name: teamName, createdAt, leadAgentId: `team-lead@${teamName}`, leadSessionId: 'lead-session', reviewRequired: true, members: [] })
|
||||
registerTeamForSessionCleanup(teamName, lifecycle)
|
||||
|
||||
// The desktop replaces the lead process on provider/permission changes,
|
||||
// crashes and reopen; the server owns the team's end of life.
|
||||
process.env.CC_HAHA_TEAM_REVIEW_REQUIRED = '1'
|
||||
await cleanupSessionTeams()
|
||||
expect(readTeamFile(teamName)?.name).toBe(teamName)
|
||||
|
||||
delete process.env.CC_HAHA_TEAM_REVIEW_REQUIRED
|
||||
await cleanupSessionTeams()
|
||||
expect(readTeamFile(teamName)).toBeNull()
|
||||
await expect(fs.access(getTeamDir(teamName))).rejects.toThrow()
|
||||
} finally {
|
||||
unregisterTeamForSessionCleanup(teamName)
|
||||
if (originalConfigDir === undefined) delete process.env.CLAUDE_CONFIG_DIR
|
||||
else process.env.CLAUDE_CONFIG_DIR = originalConfigDir
|
||||
if (originalReview === undefined) delete process.env.CC_HAHA_TEAM_REVIEW_REQUIRED
|
||||
else process.env.CC_HAHA_TEAM_REVIEW_REQUIRED = originalReview
|
||||
await rm(configDir, { recursive: true, force: true })
|
||||
}
|
||||
})
|
||||
|
||||
test("lists every address of a lead's own team members and nothing from other teams", async () => {
|
||||
await withTeamFixture(async (teamName, initial) => {
|
||||
await writeTeamFileAsync(teamName, {
|
||||
...initial,
|
||||
leadSessionId: 'lead-session',
|
||||
members: [
|
||||
{ ...initial.members[0]!, name: 'team-lead', agentId: `team-lead@${teamName}` },
|
||||
{ ...initial.members[1]!, sessionId: 'worker-session' },
|
||||
],
|
||||
})
|
||||
await writeTeamFileAsync('other-team', {
|
||||
...initial,
|
||||
name: 'other-team',
|
||||
leadSessionId: 'another-lead',
|
||||
})
|
||||
|
||||
expect(await listLeadTeamMemberIdentities('lead-session')).toEqual([
|
||||
'worker-1',
|
||||
`worker-1@${teamName}`,
|
||||
'worker-session',
|
||||
])
|
||||
expect(await listLeadTeamMemberIdentities('nobody')).toEqual([])
|
||||
})
|
||||
})
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
import { randomUUID } from 'crypto'
|
||||
import { readFileSync } from 'fs'
|
||||
import { access, mkdir, readFile, rename, rm, writeFile } from 'fs/promises'
|
||||
import { access, mkdir, readdir, readFile, rename, rm, writeFile } from 'fs/promises'
|
||||
import { join } from 'path'
|
||||
import { z } from 'zod/v4'
|
||||
import { getSessionCreatedTeams } from '../../bootstrap/state.js'
|
||||
@@ -28,6 +28,7 @@ import {
|
||||
import { getAgentName, getTeamName, isTeammate } from '../teammate.js'
|
||||
import { type BackendType, isPaneBackend } from './backends/types.js'
|
||||
import { TEAM_LEAD_NAME } from './constants.js'
|
||||
import { isTeamReviewRequired } from './teamPlanPolicy.js'
|
||||
|
||||
export const inputSchema = lazySchema(() =>
|
||||
z.strictObject({
|
||||
@@ -105,6 +106,10 @@ export type TeamFile = {
|
||||
backendType?: BackendType
|
||||
isActive?: boolean // false when idle, undefined/true when active
|
||||
mode?: PermissionMode // Current permission mode for this teammate
|
||||
/** First line of the error that ended the member's last turn, if it failed. */
|
||||
lastError?: string
|
||||
/** Pending automatic continuation after a transient provider failure. */
|
||||
autoRetry?: { attempt: number; max: number; nextAt: number }
|
||||
}>
|
||||
}
|
||||
|
||||
@@ -169,6 +174,33 @@ export function readTeamFile(teamName: string): TeamFile | null {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Every name, agent id and session id by which a member of a team led by
|
||||
* `leadSessionId` can be addressed, so a lead's tools can tell its own
|
||||
* members apart from other sessions.
|
||||
*/
|
||||
export async function listLeadTeamMemberIdentities(
|
||||
leadSessionId: string,
|
||||
): Promise<string[]> {
|
||||
let teamNames: string[]
|
||||
try {
|
||||
teamNames = await readdir(getTeamsDir())
|
||||
} catch {
|
||||
return []
|
||||
}
|
||||
const identities: string[] = []
|
||||
for (const teamName of teamNames) {
|
||||
const team = readTeamFile(teamName)
|
||||
if (team?.leadSessionId !== leadSessionId) continue
|
||||
for (const member of team.members) {
|
||||
if (member.name === TEAM_LEAD_NAME) continue
|
||||
identities.push(member.name, member.agentId)
|
||||
if (member.sessionId) identities.push(member.sessionId)
|
||||
}
|
||||
}
|
||||
return identities
|
||||
}
|
||||
|
||||
/**
|
||||
* Reads a team file by name (async — for tool handlers and other async contexts)
|
||||
*/
|
||||
@@ -713,6 +745,16 @@ export function unregisterTeamForSessionCleanup(
|
||||
export async function cleanupSessionTeams(): Promise<void> {
|
||||
const sessionCreatedTeams = getSessionCreatedTeams()
|
||||
if (sessionCreatedTeams.size === 0) return
|
||||
// A desktop-hosted lead process is replaced on every provider/permission
|
||||
// change, crash or reopen while its session lives on. Its reviewed team
|
||||
// belongs to the desktop server, which ends it on /clear, session deletion
|
||||
// or TeamDelete; deleting it here destroyed running teams on every restart.
|
||||
if (isTeamReviewRequired()) {
|
||||
logForDebugging(
|
||||
`cleanupSessionTeams: keeping ${sessionCreatedTeams.size} desktop-owned team dir(s)`,
|
||||
)
|
||||
return
|
||||
}
|
||||
const teams = Array.from(sessionCreatedTeams.entries())
|
||||
logForDebugging(
|
||||
`cleanupSessionTeams: removing ${teams.length} orphan team dir(s): ${teams.map(([name]) => name).join(', ')}`,
|
||||
|
||||
@@ -134,7 +134,7 @@ export async function stageMember(teamName: string, member: TeamPlanMember): Pro
|
||||
let plan = await readTeamPlan(teamName)
|
||||
if (!plan) throw new TeamPlanError('Create the team draft first')
|
||||
await requireCurrent(teamName, { ...plan, expectedRevision: plan.revision })
|
||||
if (await isTeamExecutionApproved(teamName, member.name)) throw new TeamPlanError('A running team already has this member')
|
||||
if (await isTeamExecutionApproved(teamName, member.name)) throw new TeamPlanError(`A running team already has member ${member.name}. Give it more work with SendMessage instead; a stopped member restarts from its saved conversation when messaged.`)
|
||||
if (plan.state === 'running') {
|
||||
await writeFile(join(getTeamDir(teamName), `plan-${plan.planId}.json`), JSON.stringify(plan, null, 2))
|
||||
plan = { ...plan, parentPlanId: plan.planId, planId: randomUUID(), revision: 0, state: 'draft', members: [], tasks: [], approvedSnapshot: undefined, launch: undefined }
|
||||
|
||||
@@ -0,0 +1,41 @@
|
||||
import { expect, test } from 'bun:test'
|
||||
import { formatTeammateAutoContinuePrompt, isTransientTurnFailure, summarizeTurnFailure } from './turnFailure.js'
|
||||
|
||||
test('provider and transport hiccups are transient; configuration and content errors are not', () => {
|
||||
for (const text of [
|
||||
'API Error: {"type":"error","error":{"type":"stream_truncated","message":"OpenAI Chat upstream stream ended without finish_reason"}}',
|
||||
'API Error: {"type":"error","error":{"type":"stream_error","message":"upstream reset"}}',
|
||||
'Provider stream ended before message_stop',
|
||||
'API Error: 503 Service Unavailable',
|
||||
'Request timed out',
|
||||
'API Error: Connection error.',
|
||||
'Repeated 529 Overloaded errors',
|
||||
'API Error: 429 rate_limit_error',
|
||||
'Upstream stream idle timeout after 240000ms',
|
||||
]) expect(isTransientTurnFailure(text)).toBe(true)
|
||||
for (const text of [
|
||||
'',
|
||||
' ',
|
||||
'Prompt is too long',
|
||||
'API Error: 401 {"type":"error","error":{"type":"authentication_error"}}',
|
||||
'API Error: 403 forbidden',
|
||||
'Credit balance is too low',
|
||||
'API Error: 400 {"type":"error","error":{"type":"invalid_request_error","message":"model_not_found"}}',
|
||||
'Reached maximum turns (3)',
|
||||
'[Request interrupted by user]',
|
||||
'All findings written to reviews/runner.md',
|
||||
]) expect(isTransientTurnFailure(text)).toBe(false)
|
||||
})
|
||||
|
||||
test('failure summaries keep one clean, bounded line', () => {
|
||||
expect(summarizeTurnFailure('\n API Error: 502\nstack trace line')).toBe('API Error: 502')
|
||||
expect(summarizeTurnFailure('bad\u0007 char')).toBe('bad char')
|
||||
expect(summarizeTurnFailure('x'.repeat(300), 10)).toBe(`${'x'.repeat(10)}…`)
|
||||
})
|
||||
|
||||
test('the continuation prompt tells the member where it is in the retry budget', () => {
|
||||
const prompt = formatTeammateAutoContinuePrompt('API Error: 502', 2, 5)
|
||||
expect(prompt).toContain('API Error: 502')
|
||||
expect(prompt).toContain('automatic retry 2 of 5')
|
||||
expect(prompt).toContain('Do not redo steps that already finished')
|
||||
})
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user