feat(workflows): dynamic workflow orchestration end to end

A workflow is a JS script the model writes in the moment and hands to the
Workflow tool, which runs it in a locked-down `node:vm` and orchestrates
subagents through `agent()`/`parallel()`/`pipeline()`/`phase()`. Saving one
as a `/name` command is the secondary path; the inline script is the point.

Runtime: cross-realm value marshalling so a script cannot reach the host
`Function`, determinism guards on `Date.now()`/`Math.random()` (they would
make a resume replay diverge), a FIFO concurrency gate, and a journal that
lets an interrupted run resume from its longest unchanged prefix.

Desktop: the run shows up as a `workflow` section in the existing activity
panel — phases as headings, their agents beneath. A workflow agent is an
ordinary subagent run by the same runner, so its row opens the existing
subagent page rather than a parallel viewer of its own; that needed a
`by-agent` lookup, because these agents have no parent `Agent` tool call to
key off. Finished runs are rebuilt from the per-agent sidecars when a
session is reopened, since the live progress stream does not outlive the
process.
This commit is contained in:
程序员阿江(Relakkes)
2026-08-10 00:17:03 +08:00
parent 0f3a517085
commit 3ce7c67822
100 changed files with 10676 additions and 296 deletions
+4
View File
@@ -18,6 +18,8 @@
"@opentelemetry/sdk-metrics": "^2.6.1",
"@opentelemetry/sdk-trace-base": "^2.6.1",
"@opentelemetry/semantic-conventions": "^1.40.0",
"acorn": "^8.16.0",
"acorn-walk": "^8.3.4",
"ajv": "^8.18.0",
"asciichart": "^1.5.25",
"auto-bind": "^5.0.1",
@@ -370,6 +372,8 @@
"acorn": ["acorn@8.16.0", "https://registry.npmmirror.com/acorn/-/acorn-8.16.0.tgz", { "bin": { "acorn": "bin/acorn" } }, "sha512-UVJyE9MttOsBQIDKw1skb9nAwQuR5wuGD3+82K6JgJlm/Y+KI92oNsMNGZCYdDsVtRHSak0pcV5Dno5+4jh9sw=="],
"acorn-walk": ["acorn-walk@8.3.5", "https://registry.npmmirror.com/acorn-walk/-/acorn-walk-8.3.5.tgz", { "dependencies": { "acorn": "^8.11.0" } }, "sha512-HEHNfbars9v4pgpW6SO1KSPkfoS0xVOM/9UzkJltjlsHZmJasxg8aXkuZa7SMf8vKGIBhpUsPluQSqhJFCqebw=="],
"agent-base": ["agent-base@8.0.0", "https://registry.npmmirror.com/agent-base/-/agent-base-8.0.0.tgz", {}, "sha512-QT8i0hCz6C/KQ+KTAbSNwCHDGdmUJl2tp2ZpNlGSWCfhUNVbYG2WLE3MdZGBAgXPV4GAvjGMxo+C1hroyxmZEg=="],
"ajv": ["ajv@8.18.0", "https://registry.npmmirror.com/ajv/-/ajv-8.18.0.tgz", { "dependencies": { "fast-deep-equal": "^3.1.3", "fast-uri": "^3.0.1", "json-schema-traverse": "^1.0.0", "require-from-string": "^2.0.2" } }, "sha512-PlXPeEWMXMZ7sPYOHqmDyCJzcfNrUr3fGNKtezX14ykXOEIvyK81d+qydx89KY5O71FKMPaQ2vBfBFI5NHR63A=="],
+29
View File
@@ -35,7 +35,36 @@ export type SubagentRunResponse = {
canSendMessage?: boolean
}
/**
* Marks a subagent addressed by agent id rather than by the `Agent` tool call
* that spawned it.
*
* Workflow agents are spawned by the workflow runtime, so no such tool call
* exists. Carrying the distinction in the identifier means the tab id, the
* page, and the return path all stay exactly as they are for every other
* subagent — only the fetch differs.
*/
export const AGENT_ID_REF_PREFIX = 'agent:'
export function isAgentIdRef(ref: string): boolean {
return ref.startsWith(AGENT_ID_REF_PREFIX)
}
export function toAgentIdRef(agentId: string): string {
return `${AGENT_ID_REF_PREFIX}${agentId}`
}
export function readAgentIdRef(ref: string): string {
return ref.slice(AGENT_ID_REF_PREFIX.length)
}
export const subagentsApi = {
getRunByAgent(sessionId: string, agentId: string) {
return api.get<SubagentRunResponse>(
`/api/sessions/${encodeURIComponent(sessionId)}/subagents/by-agent/${encodeURIComponent(agentId)}`,
)
},
getRunByTool(sessionId: string, toolUseId: string, taskId?: string) {
const query = taskId ? `?taskId=${encodeURIComponent(taskId)}` : ''
return api.get<SubagentRunResponse>(
+71
View File
@@ -0,0 +1,71 @@
import { api } from './client'
import type {
ReconstructedWorkflowRun,
WorkflowDefinition,
WorkflowRunDetail,
WorkflowRunSummary,
} from '../types/workflow'
export type WorkflowValidateResult = {
ok: boolean
error?: string
name?: string
description?: string
phases?: { title: string; detail?: string; model?: string }[]
}
export const workflowsApi = {
list(cwd?: string) {
const query = cwd ? `?cwd=${encodeURIComponent(cwd)}` : ''
return api.get<{ workflows: WorkflowDefinition[] }>(`/api/workflows${query}`)
},
get(name: string, cwd?: string) {
const query = cwd ? `?cwd=${encodeURIComponent(cwd)}` : ''
return api.get<WorkflowDefinition>(
`/api/workflows/${encodeURIComponent(name)}${query}`,
)
},
/** Finished runs for one session, rebuilt from what the CLI left on disk. */
sessionRuns(sessionId: string) {
return api.get<{ runs: ReconstructedWorkflowRun[] }>(
`/api/workflows/session-runs/${encodeURIComponent(sessionId)}`,
)
},
listRuns(options?: { sessionId?: string; limit?: number }) {
const params = new URLSearchParams()
if (options?.sessionId) params.set('sessionId', options.sessionId)
if (options?.limit) params.set('limit', String(options.limit))
const query = params.toString()
return api.get<{ runs: WorkflowRunSummary[] }>(
`/api/workflows/runs${query ? `?${query}` : ''}`,
)
},
getRun(sessionId: string, runId: string) {
return api.get<WorkflowRunDetail>(
`/api/workflows/runs/${encodeURIComponent(sessionId)}/${encodeURIComponent(runId)}`,
)
},
validate(script: string) {
return api.post<WorkflowValidateResult>('/api/workflows/validate', { script })
},
save(script: string, scope: 'user' | 'project', cwd?: string) {
return api.post<{ ok: true; name: string; filePath: string }>(
'/api/workflows/save',
{ script, scope, cwd },
)
},
remove(name: string, scope: 'user' | 'project', cwd?: string) {
const params = new URLSearchParams({ scope })
if (cwd) params.set('cwd', cwd)
return api.delete<{ ok: true }>(
`/api/workflows/${encodeURIComponent(name)}?${params.toString()}`,
)
},
}
@@ -34,6 +34,7 @@ vi.mock('../../i18n', () => ({
'session.activity.details.usage': 'Usage',
'session.activity.section.tasks': 'Tasks',
'session.activity.section.team': 'Team',
'session.activity.section.workflow': 'Workflow',
'session.activity.section.backgroundTasks': 'Background Tasks',
'session.activity.section.subagents': 'SubAgents',
'session.activity.section.sources': 'Sources',
@@ -75,6 +76,7 @@ function model(overrides: Partial<SessionActivityModel> = {}): SessionActivityMo
badgeCount: 1,
sections: {
output: { id: 'output', title: 'Output', emptyLabel: 'No output', rows: [] },
workflow: { id: 'workflow', title: 'Workflow', emptyLabel: 'No workflow running', rows: [] },
tasks: {
id: 'tasks',
title: 'Tasks',
@@ -844,6 +846,73 @@ describe('SessionActivityPanel', () => {
expect(onOpenSubagent).not.toHaveBeenCalled()
})
it('renders a workflow as phase headers with their agents, each opening the subagent page', () => {
const onOpenSubagent = vi.fn()
render(
<SessionActivityPanel
model={model({
sections: {
...model().sections,
tasks: { id: 'tasks', title: 'Tasks', emptyLabel: 'No tasks', rows: [] },
workflow: {
id: 'workflow',
title: 'Workflow',
emptyLabel: 'No workflow running',
rows: [
{
id: 'w1-phase-1',
section: 'workflow',
label: 'Survey',
status: 'completed',
groupProgress: { done: 2, total: 2 },
openable: false,
},
{
id: 'w1-agent-1',
section: 'workflow',
label: 'survey response.js',
status: 'completed',
group: 'Survey',
toolUseId: 'agent:a11',
openable: true,
},
{
id: 'w1-agent-4',
section: 'workflow',
label: 'check response #2',
status: 'pending',
group: 'Cross-check',
openable: false,
},
],
},
},
})}
open
onClose={vi.fn()}
onOpenSubagent={onOpenSubagent}
/>,
)
// The phase is a heading, not something you can open.
const header = screen.getByTestId('workflow-phase-header')
expect(header).toHaveTextContent('Survey')
expect(header).toHaveTextContent('2/2')
// Its agent opens the ordinary subagent page, addressed by agent id
// because a workflow agent has no parent Agent tool call.
fireEvent.click(screen.getByRole('button', { name: /open run survey response\.js/i }))
expect(onOpenSubagent).toHaveBeenCalledWith(
expect.objectContaining({ toolUseId: 'agent:a11', title: 'survey response.js' }),
)
// A queued agent has no transcript yet, so it must not offer to open one.
expect(
screen.queryByRole('button', { name: /open run check response #2/i }),
).not.toBeInTheDocument()
expect(screen.getByText('check response #2')).toBeInTheDocument()
})
it('does not render when closed', () => {
render(<SessionActivityPanel model={model()} open={false} onClose={vi.fn()} onOpenSubagent={vi.fn()} />)
@@ -1,5 +1,5 @@
import { useCallback, useEffect, useMemo, useRef, useState } from 'react'
import { Check, ChevronRight, Circle, FileText, LoaderCircle, Square, Terminal, Users, X } from 'lucide-react'
import { Check, ChevronRight, Circle, FileText, LoaderCircle, Square, Terminal, Users, X, Zap } from 'lucide-react'
import { Badge, StatusDot, type Tone } from '@/components/ui/Badge'
import { Button } from '@/components/ui/Button'
import { IconButton } from '@/components/ui/IconButton'
@@ -72,6 +72,8 @@ function getSectionTitle(sectionId: ActivitySectionId, t: TranslationFn): string
return t('session.activity.section.tasks')
case 'team':
return t('session.activity.section.team')
case 'workflow':
return t('session.activity.section.workflow')
case 'backgroundTasks':
return t('session.activity.section.backgroundTasks')
case 'subagents':
@@ -92,6 +94,8 @@ function getSectionRowsClassName(sectionId: ActivitySectionId, rowCount: number)
return base
case 'team':
return base
case 'workflow':
return base
case 'backgroundTasks':
return base
case 'subagents':
@@ -201,6 +205,8 @@ function getRowIcon(row: ActivityRow) {
switch (row.section) {
case 'team':
return Users
case 'workflow':
return Zap
case 'backgroundTasks':
return Terminal
case 'subagents':
@@ -255,7 +261,9 @@ function ActivityRowIcon({
sessionId: string
status?: ActivityRow['status']
}) {
if (row.section === 'subagents') {
// Workflow agents get the same mascot as any other subagent — they are the
// same thing, and giving them a different glyph would imply otherwise.
if (row.section === 'subagents' || (row.section === 'workflow' && row.group)) {
return <AgentMascot seed={`${sessionId}:${row.toolUseId ?? row.taskId ?? row.id}`} status={status} />
}
@@ -328,6 +336,41 @@ function BackgroundTaskStopButton({
)
}
/**
* A phase heading inside the workflow section.
*
* Deliberately not a card: the section is already a bordered list, and boxing
* each phase inside it turned three stages into three nested frames. A rule
* plus the settled count carries the grouping on its own.
*/
function WorkflowPhaseHeader({
label,
status,
done,
total,
}: {
label: string
status: ActivityRow['status']
done: number
total: number
}) {
return (
<div
data-testid="workflow-phase-header"
data-status={status}
className="flex items-center gap-2 px-2 pb-1 pt-2.5 first:pt-1"
>
<span className="truncate text-[11px] font-semibold uppercase tracking-wide text-[var(--color-text-secondary)]">
{label}
</span>
<span className="h-px flex-1 bg-[var(--color-border)]" aria-hidden="true" />
<span className="shrink-0 font-mono text-[11px] tabular-nums text-[var(--color-text-tertiary)]">
{done}/{total}
</span>
</div>
)
}
function ActivityRowView({
row,
sessionId,
@@ -348,6 +391,19 @@ function ActivityRowView({
selected?: boolean
}) {
const t = useTranslation()
// A workflow phase is a heading over its agents, not a row you can open.
// Rendering it as one made a fan-out read as a flat list where the stage
// boundaries were invisible.
if (row.groupProgress) {
return (
<WorkflowPhaseHeader
label={row.label}
status={row.status}
done={row.groupProgress.done}
total={row.groupProgress.total}
/>
)
}
const isTask = row.section === 'tasks'
const isStoppingSubagent = row.section === 'subagents' && row.status === 'running' && stoppingBackgroundTask
const displayStatus: ActivityRow['status'] = isStoppingSubagent ? 'pending' : row.status
@@ -430,7 +486,13 @@ function ActivityRowView({
)
}
if (row.section === 'subagents' && row.openable && row.toolUseId) {
// A workflow agent is an ordinary subagent, so it opens the same page by the
// same handler — there is nothing workflow-specific to render for one.
if (
(row.section === 'subagents' || row.section === 'workflow') &&
row.openable &&
row.toolUseId
) {
const openButton = (
<button
type="button"
@@ -441,7 +503,7 @@ function ActivityRowView({
toolUseId: row.toolUseId!,
title: row.label,
})}
className={`${interactiveRowClassName} ${stopButton ? 'flex-1' : 'w-full'}`}
className={`${interactiveRowClassName} ${stopButton ? 'flex-1' : 'w-full'} ${row.group ? 'pl-3' : ''}`}
>
{content}
</button>
@@ -1720,3 +1720,110 @@ describe('buildSessionActivityModel', () => {
expect(model.badgeCount).toBe(1)
})
})
describe('workflow section', () => {
const AGENTS = [
{ type: 'workflow_agent', index: 1, label: 'survey response.js', state: 'done', phaseIndex: 1, phaseTitle: 'Survey', agentId: 'a11', tokens: 24_100 },
{ type: 'workflow_agent', index: 2, label: 'survey request.js', state: 'done', phaseIndex: 1, phaseTitle: 'Survey', agentId: 'a12' },
{ type: 'workflow_agent', index: 3, label: 'check response #1', state: 'progress', phaseIndex: 2, phaseTitle: 'Cross-check', agentId: 'a13' },
// Queued: accepted by the runtime but never given a slot, so no transcript.
{ type: 'workflow_agent', index: 4, label: 'check response #2', state: 'start', phaseIndex: 2, phaseTitle: 'Cross-check' },
]
function run(overrides: Record<string, unknown> = {}) {
return {
taskId: 'w1',
sessionId: 'session-1',
workflowName: 'route-survey',
status: 'running',
startedAt: 0,
updatedAt: 0,
agentCount: 4,
totalTokens: 0,
toolCalls: 0,
progress: [
{ type: 'workflow_phase', index: 1, title: 'Survey' },
{ type: 'workflow_phase', index: 2, title: 'Cross-check' },
...AGENTS,
],
...overrides,
} as never
}
function build() {
return buildSessionActivityModel({
sessionId: 'session-1',
tasks: [],
completedAndDismissed: false,
backgroundTasks: [],
agentNotifications: [],
workflowRuns: [run()],
})
}
it('lays each phase out as a header followed by its agents', () => {
const rows = build().sections.workflow.rows
expect(rows.map((row) => [row.label, row.groupProgress ? 'phase' : row.group])).toEqual([
['Survey', 'phase'],
['survey response.js', 'Survey'],
['survey request.js', 'Survey'],
['Cross-check', 'phase'],
['check response #1', 'Cross-check'],
['check response #2', 'Cross-check'],
])
})
it('counts settled agents on the phase header', () => {
const headers = build().sections.workflow.rows.filter((row) => row.groupProgress)
expect(headers[0]!.groupProgress).toEqual({ done: 2, total: 2 })
expect(headers[0]!.status).toBe('completed')
expect(headers[1]!.groupProgress).toEqual({ done: 0, total: 2 })
expect(headers[1]!.status).toBe('running')
})
it('opens each agent through the ordinary subagent route', () => {
// A workflow agent is a subagent run by the same runner, so the row carries
// the reference the existing page opens with rather than anything bespoke.
const rows = build().sections.workflow.rows
const running = rows.find((row) => row.label === 'check response #1')!
expect(running.openable).toBe(true)
expect(running.toolUseId).toBe('agent:a13')
// Queued agents have no transcript yet — offering to open one would 404.
const queued = rows.find((row) => row.label === 'check response #2')!
expect(queued.openable).toBe(false)
expect(queued.toolUseId).toBeUndefined()
})
it('labels an unphased group with the run name instead of "Phase 0"', () => {
// Runs recorded before phases were persisted come back ungrouped. The
// workflow name identifies them; a bare index does not.
const model = buildSessionActivityModel({
sessionId: 'session-1',
tasks: [],
completedAndDismissed: false,
backgroundTasks: [],
agentNotifications: [],
workflowRuns: [run({
workflowName: 'review-last-month',
progress: [
{ type: 'workflow_agent', index: 1, label: 'review:security', state: 'done', phaseIndex: 0, agentId: 'a1' },
],
})],
})
const header = model.sections.workflow.rows.find((row) => row.groupProgress)!
expect(header.label).toBe('review-last-month')
})
it('badges only the agents, never the phase headers', () => {
// One running plus one queued agent. The Cross-check header is also
// "running", but counting it would double-count the very agents beneath
// it — the badge is a count of work, not of headings.
expect(build().badgeCount).toBe(2)
})
it('shows the workflow above the individual subagents it spawned', () => {
const order = getVisibleActivitySections(build()).map((section) => section.id)
expect(order).toEqual(['workflow'])
})
})
@@ -3,10 +3,12 @@ import type { TaskSummaryItem, UIMessage } from '../../types/chat'
import type { CLITask, TaskStatus } from '../../types/cliTask'
import type { TeamMember } from '../../types/team'
import { createBackgroundTaskDismissKey } from '../../lib/backgroundTasks'
import { toAgentIdRef } from '../../api/subagents'
import type { WorkflowAgentEvent, WorkflowRun } from '../../types/workflow'
export type ActivityStatus = TaskStatus | BackgroundAgentTask['status'] | TeamMember['status']
export type ActivitySectionId = 'output' | 'tasks' | 'team' | 'backgroundTasks' | 'subagents' | 'sources'
export type ActivitySectionId = 'output' | 'tasks' | 'team' | 'workflow' | 'backgroundTasks' | 'subagents' | 'sources'
export type ActivityRow = {
id: string
@@ -16,6 +18,14 @@ export type ActivityRow = {
description?: string
summary?: string
toolUseId?: string
/**
* Phase this row belongs to, for sections that group. A workflow is phases
* of N agents, and the grouping is the only thing that makes a fan-out
* readable — twelve flat rows say nothing about which stage they belong to.
*/
group?: string
/** Set on a group's header row; agents under it carry `group` instead. */
groupProgress?: { done: number; total: number }
taskId?: string
taskType?: BackgroundAgentTask['taskType']
workflowName?: string
@@ -55,6 +65,8 @@ export type BuildSessionActivityModelInput = {
dismissedBackgroundTaskKeys?: Set<string>
agentNotifications: AgentTaskNotification[]
teamMembers?: TeamMember[]
/** Live workflow runs for this session, newest first. */
workflowRuns?: WorkflowRun[]
}
/**
@@ -64,6 +76,9 @@ export type BuildSessionActivityModelInput = {
*/
export const VISIBLE_ACTIVITY_SECTION_ORDER = [
'tasks',
// A running workflow is the turn's whole shape, so it sits above the
// individual agents it spawned rather than among them.
'workflow',
'subagents',
'team',
'backgroundTasks',
@@ -76,6 +91,7 @@ const SECTION_META: Record<ActivitySectionId, Pick<ActivitySection, 'title' | 'e
output: { title: 'Output', emptyLabel: 'No output' },
tasks: { title: 'Tasks', emptyLabel: 'No tasks' },
team: { title: 'Team', emptyLabel: 'No team members' },
workflow: { title: 'Workflow', emptyLabel: 'No workflow running' },
backgroundTasks: { title: 'Background Tasks', emptyLabel: 'No background tasks' },
subagents: { title: 'SubAgents', emptyLabel: 'No SubAgents' },
sources: { title: 'Sources', emptyLabel: 'No sources' },
@@ -86,6 +102,7 @@ function createEmptySections(): Record<ActivitySectionId, ActivitySection> {
output: createSection('output'),
tasks: createSection('tasks'),
team: createSection('team'),
workflow: createSection('workflow'),
backgroundTasks: createSection('backgroundTasks'),
subagents: createSection('subagents'),
sources: createSection('sources'),
@@ -764,6 +781,96 @@ function mergeNotificationRow(existing: ActivityRow | undefined, notification: A
}
}
/**
* Flatten a workflow run into phase headers each followed by its agents.
*
* The agents are ordinary subagents, so every row carries the reference the
* existing subagent page opens with — there is nothing workflow-specific to
* render for one of them. An agent that has not been given a concurrency slot
* yet has no transcript to open, so it is listed but not openable.
*/
function buildWorkflowRows(run: WorkflowRun): ActivityRow[] {
const phaseTitles = new Map<number, string>()
const agentsByPhase = new Map<number, WorkflowAgentEvent[]>()
for (const event of run.progress) {
if (event.type === 'workflow_phase') {
if (!phaseTitles.has(event.index)) phaseTitles.set(event.index, event.title)
if (!agentsByPhase.has(event.index)) agentsByPhase.set(event.index, [])
continue
}
const phaseIndex = event.phaseIndex ?? 0
if (!phaseTitles.has(phaseIndex)) {
phaseTitles.set(phaseIndex, event.phaseTitle ?? '')
}
const bucket = agentsByPhase.get(phaseIndex) ?? []
bucket.push(event)
agentsByPhase.set(phaseIndex, bucket)
}
const rows: ActivityRow[] = []
for (const [phaseIndex, title] of [...phaseTitles.entries()].sort(([a], [b]) => a - b)) {
const agents = (agentsByPhase.get(phaseIndex) ?? [])
.slice()
.sort((a, b) => a.index - b.index)
if (agents.length === 0 && !title) continue
// Agents emitted before any `phase()` call — and every agent of a run
// recorded before phases were persisted — have no title. The run's name
// says more about them than "Phase 0" does, and it also tells two runs in
// the same session apart.
const groupLabel = title || run.workflowName
const done = agents.filter(
agent => agent.state === 'done' || agent.state === 'error',
).length
rows.push({
id: `${run.taskId}-phase-${phaseIndex}`,
section: 'workflow',
label: groupLabel,
status: workflowPhaseStatus(agents),
groupProgress: { done, total: agents.length },
workflowName: run.workflowName,
openable: false,
})
for (const agent of agents) {
rows.push({
id: `${run.taskId}-agent-${agent.index}`,
section: 'workflow',
label: agent.label,
status: workflowAgentStatus(agent),
group: groupLabel,
summary: agent.resultPreview,
toolUseId: agent.agentId ? toAgentIdRef(agent.agentId) : undefined,
taskType: 'local_agent',
workflowName: run.workflowName,
usage: agent.tokens ? { totalTokens: agent.tokens } : undefined,
openable: Boolean(agent.agentId),
})
}
}
return rows
}
function workflowAgentStatus(agent: WorkflowAgentEvent): ActivityStatus {
if (agent.state === 'done') return 'completed'
if (agent.state === 'error') return 'failed'
if (agent.state === 'progress') return 'running'
return 'pending'
}
function workflowPhaseStatus(agents: WorkflowAgentEvent[]): ActivityStatus {
if (agents.length === 0) return 'pending'
if (agents.some(agent => agent.state === 'progress')) return 'running'
if (agents.every(agent => agent.state === 'done' || agent.state === 'error')) {
return agents.some(agent => agent.state === 'error') ? 'failed' : 'completed'
}
return agents.some(agent => agent.state === 'done' || agent.state === 'error')
? 'running'
: 'pending'
}
function buildOutputRow(key: string, outputFile: string): ActivityRow {
return {
id: `output-${key}`,
@@ -788,6 +895,15 @@ export function buildSessionActivityModel(input: BuildSessionActivityModelInput)
}
}
for (const run of input.workflowRuns ?? []) {
sections.workflow.rows.push(...buildWorkflowRows(run))
}
for (const row of sections.workflow.rows) {
if (isBadgeStatus(row.status) && !row.groupProgress) {
badgeCount += 1
}
}
for (const member of input.teamMembers ?? []) {
sections.team.rows.push(buildTeamRow(member))
}
+39
View File
@@ -2327,6 +2327,7 @@ Row 9, all 8 cells: continuing from straight down, turning left through lower-le
'session.activity.section.tasks': 'Tasks',
'session.activity.tasksProgress': 'Task progress {completed}/{total}',
'session.activity.section.team': 'Team',
'session.activity.section.workflow': 'Workflow',
'session.activity.section.backgroundTasks': 'Background Tasks',
'session.activity.section.subagents': 'SubAgents',
'session.activity.section.sources': 'Sources',
@@ -2821,6 +2822,44 @@ Row 9, all 8 cells: continuing from straight down, turning left through lower-le
'browser.selection.navigationBody': 'Navigating will discard the {count} selected elements and revert their previews.',
'browser.selection.navigationContinue': 'Discard and continue',
'browser.selection.navigationDiscarded': 'Cleared {count} selections after the page changed.',
'workflows.status.running': 'Running',
'workflows.status.completed': 'Completed',
'workflows.status.failed': 'Failed',
'workflows.status.stopped': 'Stopped',
'workflows.report.title': 'Dynamic workflow run',
'workflows.report.progressLabel': 'Progress of workflow {name}',
'workflows.report.waiting': 'Waiting for the first agent…',
'workflows.report.result': 'Result',
'workflows.ungroupedPhase': 'Ungrouped',
'workflows.meta.agents': '{count} agents',
'workflows.meta.tokens': '{tokens} tokens',
'workflows.meta.tools': '{count} tool calls',
'workflows.phase.counts': '{done}/{total} done',
'workflows.phase.failed': '{count} failed',
'workflows.agent.cached': 'cached',
'workflows.dock.running': '{count} running',
'workflows.dock.expand': 'Show workflow details',
'workflows.dock.collapse': 'Hide workflow details',
'workflows.launcher.title': 'Run a workflow',
'workflows.launcher.hint': 'A workflow orchestrates many subagents from a saved script. It runs in the background and reports once.',
'workflows.launcher.empty': 'No saved workflows. Save one from a run, or drop a script in .claude/workflows/.',
'workflows.launcher.run': 'Run',
'workflows.launcher.argsLabel': 'Input passed to the script as args',
'workflows.workbench.title': 'Dynamic workflow workbench',
'workflows.workbench.close': 'Close workflow workbench',
'workflows.workbench.open': 'Open workflow workbench',
'workflows.workbench.closeDetail': 'Close agent details',
'workflows.workbench.counts': '{done}/{total} agents settled',
'workflows.workbench.runningCount': '{count} running',
'workflows.workbench.queued': '{count} queued',
'workflows.workbench.prompt': 'Prompt',
'workflows.workbench.detailTitle': 'Agent',
'workflows.history.title': 'Workflow history',
'workflows.history.empty': 'No past runs on this machine yet.',
'workflows.history.pick': 'Pick a run to see its script and per-agent results.',
'workflows.history.agents': '{count} agents recorded',
'workflows.history.script': 'Script',
'workflows.launcher.argsPlaceholder': 'e.g. a question, or a list of paths',
} as const
export type TranslationKey = keyof typeof en
+39
View File
@@ -2329,6 +2329,7 @@ export const jp: Record<TranslationKey, string> = {
'session.activity.section.tasks': 'タスク',
'session.activity.tasksProgress': 'タスク進捗 {completed}/{total}',
'session.activity.section.team': 'チーム',
'session.activity.section.workflow': 'ワークフロー',
'session.activity.section.backgroundTasks': 'バックグラウンドタスク',
'session.activity.section.subagents': 'SubAgent',
'session.activity.section.sources': 'ソース',
@@ -2823,4 +2824,42 @@ export const jp: Record<TranslationKey, string> = {
'browser.selection.navigationBody': '移動すると、選択した {count} 件が破棄され、ページ上のプレビュー変更も元に戻ります。',
'browser.selection.navigationContinue': '破棄して続行',
'browser.selection.navigationDiscarded': 'ページが変わったため、{count} 件の選択をクリアしました。',
'workflows.status.running': '実行中',
'workflows.status.completed': '完了',
'workflows.status.failed': '失敗',
'workflows.status.stopped': '停止',
'workflows.report.title': '動的ワークフローの実行',
'workflows.report.progressLabel': 'ワークフロー {name} の進捗',
'workflows.report.waiting': '最初のエージェントを待機中…',
'workflows.report.result': '結果',
'workflows.ungroupedPhase': '未分類',
'workflows.meta.agents': 'エージェント {count} 件',
'workflows.meta.tokens': '{tokens} トークン',
'workflows.meta.tools': 'ツール呼び出し {count} 回',
'workflows.phase.counts': '{done}/{total} 完了',
'workflows.phase.failed': '{count} 件失敗',
'workflows.agent.cached': 'キャッシュ',
'workflows.dock.running': '{count} 件実行中',
'workflows.dock.expand': 'ワークフローの詳細を表示',
'workflows.dock.collapse': 'ワークフローの詳細を隠す',
'workflows.launcher.title': 'ワークフローを実行',
'workflows.launcher.hint': 'ワークフローは保存済みスクリプトから多数のサブエージェントを編成し、バックグラウンドで実行して最後にまとめて報告します。',
'workflows.launcher.empty': '保存済みのワークフローがありません。実行から保存するか、.claude/workflows/ にスクリプトを置いてください。',
'workflows.launcher.run': '実行',
'workflows.launcher.argsLabel': 'args としてスクリプトに渡す入力',
'workflows.workbench.title': '動的ワークフロー ワークベンチ',
'workflows.workbench.close': 'ワークフロー ワークベンチを閉じる',
'workflows.workbench.open': 'ワークフロー ワークベンチを開く',
'workflows.workbench.closeDetail': 'エージェント詳細を閉じる',
'workflows.workbench.counts': 'エージェント {done}/{total} 件が終了',
'workflows.workbench.runningCount': '{count} 件実行中',
'workflows.workbench.queued': '{count} 件待機中',
'workflows.workbench.prompt': 'プロンプト',
'workflows.workbench.detailTitle': 'エージェント',
'workflows.history.title': 'ワークフロー履歴',
'workflows.history.empty': 'このマシンにはまだ実行履歴がありません。',
'workflows.history.pick': '実行を選ぶと、スクリプトと各エージェントの結果を表示します。',
'workflows.history.agents': '{count} 個のエージェントを記録',
'workflows.history.script': 'スクリプト',
'workflows.launcher.argsPlaceholder': '例: 質問、またはパスのリスト',
}
+39
View File
@@ -2329,6 +2329,7 @@ export const kr: Record<TranslationKey, string> = {
'session.activity.section.tasks': '작업',
'session.activity.tasksProgress': '작업 진행률 {completed}/{total}',
'session.activity.section.team': '팀',
'session.activity.section.workflow': '워크플로',
'session.activity.section.backgroundTasks': '백그라운드 작업',
'session.activity.section.subagents': 'SubAgent',
'session.activity.section.sources': '소스',
@@ -2823,4 +2824,42 @@ export const kr: Record<TranslationKey, string> = {
'browser.selection.navigationBody': '이동하면 선택한 요소 {count}개가 삭제되고 페이지의 미리보기 수정도 되돌아갑니다.',
'browser.selection.navigationContinue': '삭제하고 계속',
'browser.selection.navigationDiscarded': '페이지가 변경되어 선택 {count}개를 지웠습니다.',
'workflows.status.running': '실행 중',
'workflows.status.completed': '완료',
'workflows.status.failed': '실패',
'workflows.status.stopped': '중지됨',
'workflows.report.title': '동적 워크플로 실행',
'workflows.report.progressLabel': '워크플로 {name} 진행률',
'workflows.report.waiting': '첫 에이전트를 기다리는 중…',
'workflows.report.result': '결과',
'workflows.ungroupedPhase': '미분류',
'workflows.meta.agents': '에이전트 {count}개',
'workflows.meta.tokens': '{tokens} 토큰',
'workflows.meta.tools': '도구 호출 {count}회',
'workflows.phase.counts': '{done}/{total} 완료',
'workflows.phase.failed': '{count}개 실패',
'workflows.agent.cached': '캐시',
'workflows.dock.running': '{count}개 실행 중',
'workflows.dock.expand': '워크플로 세부 정보 펼치기',
'workflows.dock.collapse': '워크플로 세부 정보 접기',
'workflows.launcher.title': '워크플로 실행',
'workflows.launcher.hint': '워크플로는 저장된 스크립트로 여러 서브에이전트를 편성해 백그라운드에서 실행하고 끝나면 한 번에 보고합니다.',
'workflows.launcher.empty': '저장된 워크플로가 없습니다. 실행에서 저장하거나 .claude/workflows/ 에 스크립트를 넣으세요.',
'workflows.launcher.run': '실행',
'workflows.launcher.argsLabel': 'args 로 스크립트에 전달할 입력',
'workflows.workbench.title': '동적 워크플로 워크벤치',
'workflows.workbench.close': '워크플로 워크벤치 닫기',
'workflows.workbench.open': '워크플로 워크벤치 열기',
'workflows.workbench.closeDetail': '에이전트 세부 정보 닫기',
'workflows.workbench.counts': '에이전트 {done}/{total}개 종료',
'workflows.workbench.runningCount': '{count}개 실행 중',
'workflows.workbench.queued': '{count}개 대기 중',
'workflows.workbench.prompt': '프롬프트',
'workflows.workbench.detailTitle': '에이전트',
'workflows.history.title': '워크플로 기록',
'workflows.history.empty': '이 컴퓨터에는 아직 실행 기록이 없습니다.',
'workflows.history.pick': '실행을 선택하면 스크립트와 에이전트별 결과를 볼 수 있습니다.',
'workflows.history.agents': '{count}개 에이전트 기록됨',
'workflows.history.script': '스크립트',
'workflows.launcher.argsPlaceholder': '예: 질문 또는 경로 목록',
}
+39
View File
@@ -2328,6 +2328,7 @@ export const zh: Record<TranslationKey, string> = {
'session.activity.section.tasks': '任務',
'session.activity.tasksProgress': '任務進度 {completed}/{total}',
'session.activity.section.team': '團隊',
'session.activity.section.workflow': '工作流程',
'session.activity.section.backgroundTasks': '後台任務',
'session.activity.section.subagents': 'SubAgent',
'session.activity.section.sources': '來源',
@@ -2822,4 +2823,42 @@ export const zh: Record<TranslationKey, string> = {
'browser.selection.navigationBody': '繼續導覽會捨棄已選的 {count} 個元素,並還原頁面中的預覽修改。',
'browser.selection.navigationContinue': '捨棄並繼續',
'browser.selection.navigationDiscarded': '頁面已變更,已清除 {count} 個選取。',
'workflows.status.running': '執行中',
'workflows.status.completed': '已完成',
'workflows.status.failed': '失敗',
'workflows.status.stopped': '已停止',
'workflows.report.title': '動態工作流執行',
'workflows.report.progressLabel': '工作流 {name} 的進度',
'workflows.report.waiting': '等待第一個 agent…',
'workflows.report.result': '結果',
'workflows.ungroupedPhase': '未分組',
'workflows.meta.agents': '{count} 個 agent',
'workflows.meta.tokens': '{tokens} tokens',
'workflows.meta.tools': '{count} 次工具呼叫',
'workflows.phase.counts': '{done}/{total} 已完成',
'workflows.phase.failed': '{count} 個失敗',
'workflows.agent.cached': '快取',
'workflows.dock.running': '{count} 個執行中',
'workflows.dock.expand': '展開工作流詳情',
'workflows.dock.collapse': '收合工作流詳情',
'workflows.launcher.title': '執行工作流',
'workflows.launcher.hint': '工作流透過已儲存的腳本編排多個子 agent,在背景執行並在結束時統一回報。',
'workflows.launcher.empty': '還沒有已儲存的工作流。可以從某次執行儲存,或把腳本放進 .claude/workflows/。',
'workflows.launcher.run': '執行',
'workflows.launcher.argsLabel': '作為 args 傳給腳本的輸入',
'workflows.workbench.title': '動態工作流工作台',
'workflows.workbench.close': '關閉工作流工作台',
'workflows.workbench.open': '開啟工作流工作台',
'workflows.workbench.closeDetail': '關閉 agent 詳情',
'workflows.workbench.counts': '{done}/{total} 個 agent 已結束',
'workflows.workbench.runningCount': '{count} 個執行中',
'workflows.workbench.queued': '{count} 個排隊中',
'workflows.workbench.prompt': '提示詞',
'workflows.workbench.detailTitle': 'Agent',
'workflows.history.title': '工作流程歷史',
'workflows.history.empty': '本機還沒有歷史執行記錄。',
'workflows.history.pick': '選擇一次執行,查看它的腳本和各 agent 的結果。',
'workflows.history.agents': '已記錄 {count} 個 agent',
'workflows.history.script': '腳本',
'workflows.launcher.argsPlaceholder': '例如一個問題,或一組路徑',
}
+39
View File
@@ -2328,6 +2328,7 @@ export const zh: Record<TranslationKey, string> = {
'session.activity.section.tasks': '任务',
'session.activity.tasksProgress': '任务进度 {completed}/{total}',
'session.activity.section.team': '团队',
'session.activity.section.workflow': '工作流',
'session.activity.section.backgroundTasks': '后台任务',
'session.activity.section.subagents': 'SubAgent',
'session.activity.section.sources': '来源',
@@ -2822,4 +2823,42 @@ export const zh: Record<TranslationKey, string> = {
'browser.selection.navigationBody': '继续导航会丢弃已选的 {count} 个元素,并还原页面中的预览修改。',
'browser.selection.navigationContinue': '丢弃并继续',
'browser.selection.navigationDiscarded': '页面已变化,已清空 {count} 个选择。',
'workflows.status.running': '运行中',
'workflows.status.completed': '已完成',
'workflows.status.failed': '失败',
'workflows.status.stopped': '已停止',
'workflows.report.title': '动态工作流运行',
'workflows.report.progressLabel': '工作流 {name} 的进度',
'workflows.report.waiting': '等待第一个 agent…',
'workflows.report.result': '结果',
'workflows.ungroupedPhase': '未分组',
'workflows.meta.agents': '{count} 个 agent',
'workflows.meta.tokens': '{tokens} tokens',
'workflows.meta.tools': '{count} 次工具调用',
'workflows.phase.counts': '{done}/{total} 已完成',
'workflows.phase.failed': '{count} 个失败',
'workflows.agent.cached': '缓存',
'workflows.dock.running': '{count} 个运行中',
'workflows.dock.expand': '展开工作流详情',
'workflows.dock.collapse': '收起工作流详情',
'workflows.launcher.title': '运行工作流',
'workflows.launcher.hint': '工作流通过已保存的脚本编排多个子 agent,在后台运行并在结束时统一汇报。',
'workflows.launcher.empty': '还没有已保存的工作流。可以从某次运行保存,或把脚本放进 .claude/workflows/。',
'workflows.launcher.run': '运行',
'workflows.launcher.argsLabel': '作为 args 传给脚本的输入',
'workflows.workbench.title': '动态工作流工作台',
'workflows.workbench.close': '关闭工作流工作台',
'workflows.workbench.open': '打开工作流工作台',
'workflows.workbench.closeDetail': '关闭 agent 详情',
'workflows.workbench.counts': '{done}/{total} 个 agent 已结束',
'workflows.workbench.runningCount': '{count} 个运行中',
'workflows.workbench.queued': '{count} 个排队中',
'workflows.workbench.prompt': '提示词',
'workflows.workbench.detailTitle': 'Agent',
'workflows.history.title': '工作流历史',
'workflows.history.empty': '本机还没有历史运行记录。',
'workflows.history.pick': '选择一次运行,查看它的脚本和各 agent 的结果。',
'workflows.history.agents': '已记录 {count} 个 agent',
'workflows.history.script': '脚本',
'workflows.launcher.argsPlaceholder': '例如一个问题,或一组路径',
}
+11
View File
@@ -41,6 +41,7 @@ import { AgentTeamsReport } from '../components/agentTeams/AgentTeamsReport'
import { AgentTeamsStrip } from '../components/agentTeams/AgentTeamsSummary'
import { SessionActivityPanel } from '../components/activity/SessionActivityPanel'
import { buildSessionActivityModel, hasVisibleSessionActivity } from '../components/activity/sessionActivityModel'
import { runsForSession, useWorkflowStore } from '../stores/workflowStore'
import { TerminalSettings } from './TerminalSettings'
import type { SessionListItem } from '../types/session'
import type { ActiveGoalState, TokenUsage } from '../types/chat'
@@ -519,6 +520,14 @@ export function ActiveSession() {
[dismissedBackgroundTaskKeyList],
)
const agentTaskNotifications = sessionState?.agentTaskNotifications ?? EMPTY_AGENT_TASK_NOTIFICATIONS
// Subscribe to the stable `runs` record and derive the per-session list here:
// a selector that filtered would allocate a new array on every store read,
// which zustand compares by identity and would re-render forever.
const allWorkflowRuns = useWorkflowStore(state => state.runs)
const workflowRuns = useMemo(
() => (activeTabId ? runsForSession({ runs: allWorkflowRuns }, activeTabId) : []),
[allWorkflowRuns, activeTabId],
)
const activeGoal = sessionState?.activeGoal ?? null
const isEmpty =
messages.length === 0 &&
@@ -570,6 +579,7 @@ export function ActiveSession() {
backgroundTasks,
dismissedBackgroundTaskKeys,
agentNotifications: Object.values(agentTaskNotifications),
workflowRuns,
})
}, [
activeTabId,
@@ -581,6 +591,7 @@ export function ActiveSession() {
dismissedBackgroundTaskKeys,
messages,
trackedTaskSessionId,
workflowRuns,
])
const hasVisibleActivity = activityModel ? hasVisibleSessionActivity(activityModel) : false
const hasAutoOpenActivity = activityModel ? activityModel.badgeCount > 0 : false
+25 -1
View File
@@ -5,9 +5,13 @@ import type { SubagentRunResponse } from '../api/subagents'
import { useSettingsStore } from '../stores/settingsStore'
import { setComposerText } from '../components/chat/composerTestUtils'
vi.mock('../api/subagents', () => ({
vi.mock('../api/subagents', async (importOriginal) => ({
// Keep the real ref helpers: they decide whether the page fetches by tool
// call or by agent id, so stubbing them would test a fiction.
...(await importOriginal<typeof import('../api/subagents')>()),
subagentsApi: {
getRunByTool: vi.fn(),
getRunByAgent: vi.fn(),
sendMessage: vi.fn(),
},
}))
@@ -70,6 +74,7 @@ describe('SubagentRunPage', () => {
cleanup()
vi.useRealTimers()
vi.mocked(subagentsApi.getRunByTool).mockReset()
vi.mocked(subagentsApi.getRunByAgent).mockReset()
vi.mocked(subagentsApi.sendMessage).mockReset()
})
@@ -86,6 +91,25 @@ describe('SubagentRunPage', () => {
expect(useTabStore.getState().tabs.map((tab) => tab.sessionId)).toEqual(['session-1'])
})
it('fetches a workflow agent by agent id, not by tool call', async () => {
// Workflow agents are spawned by the workflow runtime, so no parent Agent
// tool call exists to look them up by. Same page, same rendering — only
// the lookup differs.
vi.mocked(subagentsApi.getRunByAgent).mockResolvedValue(subagentRun())
render(
<SubagentRunPage
sourceSessionId="session-1"
toolUseId="agent:wfagent1"
title="survey response.js"
/>,
)
expect(await screen.findByText('survey response.js')).toBeInTheDocument()
expect(subagentsApi.getRunByAgent).toHaveBeenCalledWith('session-1', 'wfagent1')
expect(subagentsApi.getRunByTool).not.toHaveBeenCalled()
})
it('renders SubAgent run details', async () => {
vi.mocked(subagentsApi.getRunByTool).mockResolvedValue(subagentRun({
outputFile: '/tmp/result.md',
+7 -1
View File
@@ -1,6 +1,8 @@
import { useCallback, useEffect, useRef, useState } from 'react'
import { ArrowLeft, RefreshCw } from 'lucide-react'
import {
isAgentIdRef,
readAgentIdRef,
subagentsApi,
type SubagentRunResponse,
type SubagentRunStatus,
@@ -68,7 +70,11 @@ export function SubagentRunPage({
setError(null)
if (options?.resetData) setData(null)
try {
const nextData = await subagentsApi.getRunByTool(sourceSessionId, toolUseId, resolvedTaskId)
// A workflow agent has no parent Agent tool call, so it is addressed by
// agent id. Everything downstream of this line is identical.
const nextData = isAgentIdRef(toolUseId)
? await subagentsApi.getRunByAgent(sourceSessionId, readAgentIdRef(toolUseId))
: await subagentsApi.getRunByTool(sourceSessionId, toolUseId, resolvedTaskId)
if (requestIdRef.current !== requestId) return
setData(nextData)
} catch (err) {
+12
View File
@@ -5,6 +5,7 @@ import { subagentsApi } from '../api/subagents'
import { useTeamStore } from './teamStore'
import { useSessionStore } from './sessionStore'
import { useCLITaskStore } from './cliTaskStore'
import { useWorkflowStore } from './workflowStore'
import { useSessionRuntimeStore } from './sessionRuntimeStore'
import { useTabStore } from './tabStore'
import { randomSpinnerVerb } from '../config/spinnerVerbs'
@@ -1796,6 +1797,11 @@ export const useChatStore = create<ChatStore>((set, get) => ({
const existingLoad = historyLoadsInFlight.get(sessionId)
if (existingLoad) return existingLoad
// Workflow runs are rebuilt from disk alongside the transcript. Without
// this, reopening a session that ran a workflow showed no trace of it —
// the progress stream that populates the panel is live-only.
void useWorkflowStore.getState().hydrateSession(sessionId)
const requestedMutationEpoch = get().sessions[sessionId]?.historyMutationEpoch ?? 0
let load!: Promise<void>
load = (async () => {
@@ -3276,6 +3282,7 @@ export const useChatStore = create<ChatStore>((set, get) => ({
clearPendingTaskToolUseIds(sessionId)
clearPendingToolParentUseIds(sessionId)
useCLITaskStore.getState().clearTasks(sessionId)
useWorkflowStore.getState().clearSession(sessionId)
useSessionStore.getState().updateSessionTitle(sessionId, 'New Session')
useSessionStore.getState().updateSessionMessageCount(sessionId, 0)
useTabStore.getState().updateTabTitle(sessionId, 'New Session')
@@ -3355,6 +3362,10 @@ export const useChatStore = create<ChatStore>((set, get) => ({
}
}
if ((msg.subtype === 'task_started' || msg.subtype === 'task_progress') && msg.data && typeof msg.data === 'object') {
// Workflow runs also flow through the generic background-task path
// below; the workflow store keeps the phase/agent detail that the
// generic task shape has nowhere to put.
useWorkflowStore.getState().handleTaskEvent(sessionId, msg.subtype, msg.data)
const taskEvent = normalizeBackgroundAgentTaskEvent(msg.data, msg.subtype)
if (taskEvent) {
const now = Date.now()
@@ -3393,6 +3404,7 @@ export const useChatStore = create<ChatStore>((set, get) => ({
}
}
if (msg.subtype === 'task_notification' && msg.data && typeof msg.data === 'object') {
useWorkflowStore.getState().handleTaskEvent(sessionId, 'task_notification', msg.data)
const data = msg.data as Record<string, unknown>
const taskEvent = normalizeBackgroundAgentTaskEvent(data, 'task_notification')
const toolUseId =
+356
View File
@@ -0,0 +1,356 @@
import { beforeEach, describe, expect, it, vi } from 'vitest'
const sessionRunsMock = vi.hoisted(() => vi.fn())
vi.mock('../api/workflows', async importOriginal => {
const actual = await importOriginal<typeof import('../api/workflows')>()
return {
...actual,
workflowsApi: { ...actual.workflowsApi, sessionRuns: sessionRunsMock },
}
})
import {
activeRunForSession,
groupRunPhases,
runCompletion,
runsForSession,
useWorkflowStore,
} from './workflowStore'
const SESSION = 'session-1'
const TASK = 'w1234abcd'
function taskStarted(overrides: Record<string, unknown> = {}) {
return {
task_id: TASK,
task_type: 'local_workflow',
workflow_name: 'audit-routes',
description: 'Audit every route handler',
...overrides,
}
}
function progress(
rows: Array<Record<string, unknown>>,
usage?: { total_tokens?: number; tool_uses?: number },
) {
return {
task_id: TASK,
workflow_progress: rows,
...(usage ? { usage } : {}),
}
}
function agentRow(
index: number,
state: string,
extra: Record<string, unknown> = {},
) {
return {
type: 'workflow_agent',
index,
label: `agent ${index}`,
state,
...extra,
}
}
describe('workflowStore', () => {
beforeEach(() => {
useWorkflowStore.setState({
runs: {},
definitions: [],
definitionsLoading: false,
definitionsError: null,
history: [],
historyLoading: false,
openRunId: null,
})
})
const store = () => useWorkflowStore.getState()
it('creates a run from task_started and tracks it for the session', () => {
store().handleTaskEvent(SESSION, 'task_started', taskStarted())
const runs = runsForSession(useWorkflowStore.getState(), SESSION)
expect(runs).toHaveLength(1)
expect(runs[0]).toMatchObject({
taskId: TASK,
workflowName: 'audit-routes',
description: 'Audit every route handler',
status: 'running',
})
expect(activeRunForSession(useWorkflowStore.getState(), SESSION)?.taskId).toBe(
TASK,
)
})
it('still ignores a bare notification for a task it never tracked', () => {
store().handleTaskEvent(SESSION, 'task_notification', {
task_id: 'b0000009',
status: 'completed',
})
expect(runsForSession(useWorkflowStore.getState(), SESSION)).toHaveLength(0)
})
it('ignores task events that are not workflows', () => {
store().handleTaskEvent(SESSION, 'task_started', {
task_id: 'b0000001',
task_type: 'local_bash',
description: 'npm test',
})
expect(runsForSession(useWorkflowStore.getState(), SESSION)).toHaveLength(0)
})
it('replaces an agent row in place instead of appending a new one', () => {
store().handleTaskEvent(SESSION, 'task_started', taskStarted())
store().handleTaskEvent(
SESSION,
'task_progress',
progress([agentRow(1, 'progress', { tokens: 100 })]),
)
store().handleTaskEvent(
SESSION,
'task_progress',
progress([agentRow(1, 'done', { tokens: 400 })]),
)
const run = useWorkflowStore.getState().runs[TASK]!
const agents = run.progress.filter(row => row.type === 'workflow_agent')
expect(agents).toHaveLength(1)
expect(agents[0]).toMatchObject({ index: 1, state: 'done', tokens: 400 })
expect(run.agentCount).toBe(1)
})
it('keeps the object identity stable when an event changes nothing', () => {
store().handleTaskEvent(SESSION, 'task_started', taskStarted())
const before = useWorkflowStore.getState().runs[TASK]
store().handleTaskEvent(SESSION, 'task_progress', progress([]))
expect(useWorkflowStore.getState().runs[TASK]).toBe(before)
})
it('carries usage totals from the event', () => {
store().handleTaskEvent(SESSION, 'task_started', taskStarted())
store().handleTaskEvent(
SESSION,
'task_progress',
progress([agentRow(1, 'done')], { total_tokens: 1234, tool_uses: 7 }),
)
expect(useWorkflowStore.getState().runs[TASK]).toMatchObject({
totalTokens: 1234,
toolCalls: 7,
})
})
it('settles the run on task_notification and records the result', () => {
store().handleTaskEvent(SESSION, 'task_started', taskStarted())
// The CLI's terminal notification carries no task_type / workflow_name /
// workflow_progress — only the task id. Dropping it here is what left the
// dock reading "1 running" after a finished run.
store().handleTaskEvent(SESSION, 'task_notification', {
task_id: TASK,
status: 'completed',
result: '["ALPHA","BETA"]',
summary: 'Dynamic workflow "route-survey" completed',
})
const run = useWorkflowStore.getState().runs[TASK]!
expect(run.status).toBe('completed')
expect(run.result).toBe('["ALPHA","BETA"]')
expect(run.endedAt).toBeGreaterThan(0)
expect(activeRunForSession(useWorkflowStore.getState(), SESSION)).toBeNull()
})
it('a late progress event does not revive a settled run', () => {
store().handleTaskEvent(SESSION, 'task_started', taskStarted())
store().handleTaskEvent(SESSION, 'task_notification', {
task_id: TASK,
status: 'failed',
error: 'boom',
})
store().handleTaskEvent(
SESSION,
'task_progress',
progress([agentRow(2, 'progress')]),
)
const run = useWorkflowStore.getState().runs[TASK]!
expect(run.status).toBe('failed')
expect(run.error).toBe('boom')
})
it('surfaces a failure reason that only arrived in the summary', () => {
// Seen on a real failed run: the CLI's terminal event has no `error`
// field, so the reason only ever comes through `summary`. Reading it as
// the description put it in a truncated one-liner and left the error
// banner empty — a red "failed" badge with no explanation.
store().handleTaskEvent(SESSION, 'task_started', taskStarted())
store().handleTaskEvent(SESSION, 'task_notification', {
task_id: TASK,
status: 'failed',
summary:
'Dynamic workflow "audit-routes" failed: deliberate failure after the probe settled',
})
const run = useWorkflowStore.getState().runs[TASK]!
expect(run.error).toContain('deliberate failure after the probe settled')
// The run's own description must survive the notification.
expect(run.description).toBe('Audit every route handler')
})
it('keeps the description when a run completes', () => {
store().handleTaskEvent(SESSION, 'task_started', taskStarted())
store().handleTaskEvent(SESSION, 'task_notification', {
task_id: TASK,
status: 'completed',
summary: 'Dynamic workflow "audit-routes" completed',
result: '{}',
})
const run = useWorkflowStore.getState().runs[TASK]!
expect(run.description).toBe('Audit every route handler')
// Boilerplate completion text is not an error.
expect(run.error).toBeUndefined()
})
it('maps a killed CLI status onto stopped', () => {
store().handleTaskEvent(SESSION, 'task_started', taskStarted())
store().handleTaskEvent(SESSION, 'task_notification', {
task_id: TASK,
status: 'killed',
})
expect(useWorkflowStore.getState().runs[TASK]?.status).toBe('stopped')
})
it('clearSession drops only that session and closes an open run', () => {
store().handleTaskEvent(SESSION, 'task_started', taskStarted())
store().handleTaskEvent('other', 'task_started', taskStarted({ task_id: 'w9' }))
store().openRun(TASK)
store().clearSession(SESSION)
expect(runsForSession(useWorkflowStore.getState(), SESSION)).toHaveLength(0)
expect(runsForSession(useWorkflowStore.getState(), 'other')).toHaveLength(1)
expect(useWorkflowStore.getState().openRunId).toBeNull()
})
it('groups agents under the phase they reported', () => {
store().handleTaskEvent(SESSION, 'task_started', taskStarted())
store().handleTaskEvent(
SESSION,
'task_progress',
progress([
{ type: 'workflow_phase', index: 1, title: 'Review', kind: 'meta' },
{ type: 'workflow_phase', index: 2, title: 'Verify', kind: 'script' },
agentRow(1, 'done', { phaseIndex: 1, phaseTitle: 'Review' }),
agentRow(2, 'progress', { phaseIndex: 2, phaseTitle: 'Verify' }),
agentRow(3, 'done'),
]),
)
const run = useWorkflowStore.getState().runs[TASK]!
const { phases, ungrouped } = groupRunPhases(run)
expect(phases.map(phase => phase.title)).toEqual(['Review', 'Verify'])
expect(phases[0]?.agents.map(agent => agent.index)).toEqual([1])
expect(phases[1]?.agents.map(agent => agent.index)).toEqual([2])
expect(ungrouped.map(agent => agent.index)).toEqual([3])
})
it('reports completion as the settled fraction of agents', () => {
store().handleTaskEvent(SESSION, 'task_started', taskStarted())
store().handleTaskEvent(
SESSION,
'task_progress',
progress([
agentRow(1, 'done'),
agentRow(2, 'error'),
agentRow(3, 'progress'),
agentRow(4, 'start'),
]),
)
expect(runCompletion(useWorkflowStore.getState().runs[TASK]!)).toBe(0.5)
})
it('recognises a workflow from workflow_progress even without task_type', () => {
store().handleTaskEvent(
SESSION,
'task_progress',
progress([agentRow(1, 'progress')]),
)
expect(runsForSession(useWorkflowStore.getState(), SESSION)).toHaveLength(1)
})
})
describe('hydrateSession', () => {
const RECONSTRUCTED = {
runId: 'wf_06ee51bf-6b1',
workflowName: 'review-last-month',
startedAt: 1_700_000_000_000,
agents: [
{ agentId: 'a1', label: 'review:imagegen', phaseIndex: 1, phaseTitle: '维度审查', agentIndex: 1 },
{ agentId: 'a2', label: 'review:security', phaseIndex: 1, phaseTitle: '维度审查', agentIndex: 2 },
{ agentId: 'a3', label: 'verify:security', phaseIndex: 2, phaseTitle: '对抗验证', agentIndex: 3 },
],
}
beforeEach(() => {
sessionRunsMock.mockReset()
useWorkflowStore.setState({ runs: {}, openRunId: null })
})
it('rebuilds a finished run so a reopened session still shows it', async () => {
// The whole point: the live progress stream is gone by now.
sessionRunsMock.mockResolvedValue({ runs: [RECONSTRUCTED] })
await useWorkflowStore.getState().hydrateSession(SESSION)
const runs = runsForSession(useWorkflowStore.getState(), SESSION)
expect(runs).toHaveLength(1)
const { phases } = groupRunPhases(runs[0]!)
expect(phases.map(phase => phase.title)).toEqual(['维度审查', '对抗验证'])
expect(phases[0]!.agents.map(agent => agent.label)).toEqual([
'review:imagegen',
'review:security',
])
// Every reconstructed agent has a transcript, so it is openable.
expect(phases[0]!.agents.every(agent => agent.agentId)).toBe(true)
})
it('leaves an unrecorded phase title empty rather than inventing one', () => {
// The renderer falls back to the run name; filling in "Phase 0" here made
// that impossible and showed a meaningless heading.
sessionRunsMock.mockResolvedValue({
runs: [{
...RECONSTRUCTED,
agents: [{ agentId: 'a1', label: 'review:security', phaseIndex: 0, agentIndex: 1 }],
}],
})
return useWorkflowStore.getState().hydrateSession(SESSION).then(() => {
const run = runsForSession(useWorkflowStore.getState(), SESSION)[0]!
const phase = run.progress.find((row) => row.type === 'workflow_phase')!
expect(phase.title).toBe('')
})
})
it('never overwrites a run that is still streaming live', async () => {
useWorkflowStore.getState().handleTaskEvent(SESSION, 'task_started', {
task_id: TASK,
task_type: 'local_workflow',
workflow_name: 'review-last-month',
workflow_run_id: RECONSTRUCTED.runId,
})
useWorkflowStore.getState().handleTaskEvent(
SESSION,
'task_progress',
progress([agentRow(1, 'progress')]),
)
sessionRunsMock.mockResolvedValue({ runs: [RECONSTRUCTED] })
await useWorkflowStore.getState().hydrateSession(SESSION)
const runs = runsForSession(useWorkflowStore.getState(), SESSION)
// One run, still the live one — a reconstruction would have marked its
// in-flight agent as done.
expect(runs).toHaveLength(1)
expect(runs[0]!.status).toBe('running')
expect(runs[0]!.taskId).toBe(TASK)
})
it('stays quiet when the server cannot rebuild anything', async () => {
sessionRunsMock.mockRejectedValue(new Error('offline'))
await useWorkflowStore.getState().hydrateSession(SESSION)
expect(runsForSession(useWorkflowStore.getState(), SESSION)).toHaveLength(0)
})
})
+508
View File
@@ -0,0 +1,508 @@
import { create } from 'zustand'
import { workflowsApi } from '../api/workflows'
import type {
ReconstructedWorkflowRun,
WorkflowAgentEvent,
WorkflowDefinition,
WorkflowPhaseGroup,
WorkflowProgressEvent,
WorkflowRun,
WorkflowRunStatus,
WorkflowRunSummary,
} from '../types/workflow'
/** Cap kept in sync with the CLI's own progress-row budget. */
const MAX_PROGRESS_ROWS = 500
type WorkflowStore = {
/** Live and just-finished runs, keyed by CLI task id. */
runs: Record<string, WorkflowRun>
definitions: WorkflowDefinition[]
definitionsLoading: boolean
definitionsError: string | null
history: WorkflowRunSummary[]
historyLoading: boolean
/** Task id of the run the workflow panel is showing, if any. */
openRunId: string | null
handleTaskEvent(
sessionId: string,
subtype: 'task_started' | 'task_progress' | 'task_notification',
data: unknown,
): void
clearSession(sessionId: string): void
hydrateSession(sessionId: string): Promise<void>
openRun(taskId: string | null): void
loadDefinitions(cwd?: string): Promise<void>
loadHistory(sessionId?: string): Promise<void>
}
export const useWorkflowStore = create<WorkflowStore>(set => ({
runs: {},
definitions: [],
definitionsLoading: false,
definitionsError: null,
history: [],
historyLoading: false,
openRunId: null,
handleTaskEvent(sessionId, subtype, data) {
// A run we already track is identified by its task id alone. The terminal
// `task_notification` the CLI emits carries no task_type, workflow_name or
// workflow_progress (print.ts builds it from the notification XML), so
// re-deriving "is this a workflow" from the payload would drop it and the
// run would sit at "running" forever.
const known = Boolean(
typeof data === 'object' &&
data !== null &&
typeof (data as { task_id?: unknown }).task_id === 'string' &&
useWorkflowStore.getState().runs[(data as { task_id: string }).task_id],
)
const event = parseWorkflowTaskEvent(sessionId, subtype, data, known)
if (!event) return
set(state => {
const existing = state.runs[event.taskId]
const merged = mergeRun(existing, event)
if (merged === existing) return state
return { runs: { ...state.runs, [event.taskId]: merged } }
})
},
/**
* Load a session's finished runs from disk.
*
* Live progress only exists while a run is happening, so a session reopened
* later showed no workflow at all. This fills that in from the per-agent
* sidecars the CLI persisted. A run already being tracked live is left
* alone — the live stream is strictly better than the reconstruction.
*/
async hydrateSession(sessionId) {
try {
const { runs } = await workflowsApi.sessionRuns(sessionId)
if (runs.length === 0) return
set(state => {
const liveRunIds = new Set(
Object.values(state.runs)
.filter(run => run.sessionId === sessionId)
.map(run => run.runId)
.filter((runId): runId is string => Boolean(runId)),
)
const next = { ...state.runs }
let added = false
for (const run of runs) {
if (liveRunIds.has(run.runId) || next[run.runId]) continue
added = true
next[run.runId] = reconstructedToRun(sessionId, run)
}
return added ? { runs: next } : state
})
} catch {
// History is an enhancement; a session that cannot load it still works.
}
},
clearSession(sessionId) {
set(state => {
const kept: Record<string, WorkflowRun> = {}
let removed = false
for (const [taskId, run] of Object.entries(state.runs)) {
if (run.sessionId === sessionId) {
removed = true
continue
}
kept[taskId] = run
}
if (!removed) return state
const openRunId =
state.openRunId && kept[state.openRunId] ? state.openRunId : null
return { runs: kept, openRunId }
})
},
openRun(taskId) {
set({ openRunId: taskId })
},
async loadDefinitions(cwd) {
set({ definitionsLoading: true, definitionsError: null })
try {
const { workflows } = await workflowsApi.list(cwd)
set({ definitions: workflows, definitionsLoading: false })
} catch (error) {
set({
definitionsLoading: false,
definitionsError:
error instanceof Error ? error.message : 'Failed to load workflows',
})
}
},
async loadHistory(sessionId) {
set({ historyLoading: true })
try {
const { runs } = await workflowsApi.listRuns({ sessionId, limit: 50 })
set({ history: runs, historyLoading: false })
} catch {
set({ historyLoading: false })
}
},
}))
/**
* Turn a disk-reconstructed run into the shape the panel renders.
*
* Every agent listed produced a transcript, so `done` is a fact rather than an
* assumption. Per-agent token counts and the run's own outcome are not
* persisted, so they are left empty instead of being invented — the panel
* shows neither for a historical run.
*/
function reconstructedToRun(
sessionId: string,
run: ReconstructedWorkflowRun,
): WorkflowRun {
const phases = new Map<number, string | undefined>()
const progress: WorkflowProgressEvent[] = []
for (const agent of run.agents) {
if (!phases.has(agent.phaseIndex)) {
phases.set(agent.phaseIndex, agent.phaseTitle)
}
}
for (const [index, title] of [...phases.entries()].sort(([a], [b]) => a - b)) {
progress.push({
type: 'workflow_phase',
index,
// Left empty rather than invented when the run never recorded a phase
// title, so the renderer can fall back to something meaningful. Filling
// in "Phase 0" here hid that the title was simply unknown.
title: title ?? '',
})
}
for (const agent of run.agents) {
progress.push({
type: 'workflow_agent',
index: agent.agentIndex,
label: agent.label,
state: 'done',
phaseIndex: agent.phaseIndex,
...(agent.phaseTitle ? { phaseTitle: agent.phaseTitle } : {}),
agentId: agent.agentId,
})
}
return {
// Keyed by run id: a reconstructed run has no CLI task id, and the run id
// is what keeps it from colliding with a live run for the same workflow.
taskId: run.runId,
sessionId,
runId: run.runId,
workflowName: run.workflowName,
status: 'completed',
startedAt: run.startedAt,
updatedAt: run.startedAt,
endedAt: run.startedAt,
agentCount: run.agents.length,
totalTokens: 0,
toolCalls: 0,
progress,
}
}
// ── Selectors ────────────────────────────────────────────────────────────────
export function runsForSession(
state: Pick<WorkflowStore, 'runs'>,
sessionId: string,
): WorkflowRun[] {
return Object.values(state.runs)
.filter(run => run.sessionId === sessionId)
.sort((a, b) => b.startedAt - a.startedAt)
}
export function activeRunForSession(
state: Pick<WorkflowStore, 'runs'>,
sessionId: string,
): WorkflowRun | null {
return (
runsForSession(state, sessionId).find(run => run.status === 'running') ?? null
)
}
/**
* Fold a run's flat progress rows into phases.
*
* Agents emitted before any `phase()` call carry no `phaseIndex`; they are
* grouped under a synthetic phase rather than dropped, so a script that never
* calls `phase()` still shows all of its agents.
*/
export function groupRunPhases(run: WorkflowRun): {
phases: WorkflowPhaseGroup[]
ungrouped: WorkflowAgentEvent[]
} {
const byIndex = new Map<number, WorkflowPhaseGroup>()
const ungrouped: WorkflowAgentEvent[] = []
for (const event of run.progress) {
if (event.type === 'workflow_phase') {
if (!byIndex.has(event.index)) {
byIndex.set(event.index, {
index: event.index,
title: event.title,
agents: [],
})
}
continue
}
if (event.phaseIndex === undefined) {
ungrouped.push(event)
continue
}
const group = byIndex.get(event.phaseIndex) ?? {
index: event.phaseIndex,
title: event.phaseTitle ?? `Phase ${event.phaseIndex}`,
agents: [],
}
group.agents.push(event)
byIndex.set(event.phaseIndex, group)
}
return {
phases: [...byIndex.values()].sort((a, b) => a.index - b.index),
ungrouped,
}
}
/** Fraction of a run's agents that have settled, for the progress bar. */
export function runCompletion(run: WorkflowRun): number {
const agents = run.progress.filter(
(event): event is WorkflowAgentEvent => event.type === 'workflow_agent',
)
if (agents.length === 0) return run.status === 'running' ? 0 : 1
const settled = agents.filter(
agent => agent.state === 'done' || agent.state === 'error',
).length
return settled / agents.length
}
// ── Event parsing ────────────────────────────────────────────────────────────
type ParsedEvent = {
taskId: string
sessionId: string
runId?: string
workflowName?: string
description?: string
status?: WorkflowRunStatus
progress: WorkflowProgressEvent[]
totalTokens?: number
toolCalls?: number
result?: string
error?: string
}
/**
* Read a CLI task event, keeping only the ones that belong to a workflow.
*
* `task_progress` does not carry `task_type`, and the terminal
* `task_notification` carries neither that nor `workflow_progress`. A run is
* therefore recognised either by its workflow markers or — once `task_started`
* has registered it — by its task id alone.
*/
export function parseWorkflowTaskEvent(
sessionId: string,
subtype: 'task_started' | 'task_progress' | 'task_notification',
data: unknown,
/** True when a run with this task id is already tracked. */
known = false,
): ParsedEvent | null {
if (typeof data !== 'object' || data === null) return null
const payload = data as Record<string, unknown>
const taskId = readString(payload.task_id)
if (!taskId) return null
const taskType = readString(payload.task_type)
const workflowName = readString(payload.workflow_name)
const progress = readProgress(payload.workflow_progress)
const isWorkflow =
known ||
taskType === 'local_workflow' ||
Boolean(workflowName) ||
progress.length > 0
if (!isWorkflow) return null
const usage =
typeof payload.usage === 'object' && payload.usage !== null
? (payload.usage as Record<string, unknown>)
: undefined
const terminal = subtype === 'task_notification'
const status = terminal
? normalizeStatus(readString(payload.status))
: 'running'
const summary = readString(payload.summary)
return {
taskId,
sessionId,
runId: readString(payload.workflow_run_id),
workflowName,
// A terminal notification's `summary` describes the outcome, not the run.
// Letting it through as the description replaced "Survey Express
// request/response helpers…" with `Dynamic workflow "…" failed: …` in a
// single truncated line.
description: terminal ? undefined : (summary ?? readString(payload.description)),
status,
progress,
totalTokens: readNumber(usage?.total_tokens),
toolCalls: readNumber(usage?.tool_uses),
result: readString(payload.result),
// The CLI's terminal event carries no `error` field — the reason a run
// failed only ever arrives inside `summary`. Without this a failed run
// showed a red badge and no explanation anywhere.
error:
readString(payload.error) ??
(status === 'failed' || status === 'stopped' ? summary : undefined),
}
}
/**
* Apply one event to a run.
*
* Agent and phase rows are keyed by `type:index` and replaced in place, so a
* run that emits thousands of token-count updates for twenty agents keeps
* twenty rows. Returning the previous object unchanged when nothing moved is
* what stops zustand subscribers from re-rendering on every heartbeat.
*/
function mergeRun(
existing: WorkflowRun | undefined,
event: ParsedEvent,
): WorkflowRun {
const now = Date.now()
const base: WorkflowRun = existing ?? {
taskId: event.taskId,
sessionId: event.sessionId,
runId: event.runId,
workflowName: event.workflowName ?? 'workflow',
description: event.description,
status: 'running',
startedAt: now,
updatedAt: now,
agentCount: 0,
totalTokens: 0,
toolCalls: 0,
progress: [],
}
// A terminal run must not be revived by a late progress event.
if (base.status !== 'running' && event.status === 'running') return base
let progress = base.progress
if (event.progress.length > 0) {
const next = [...base.progress]
const indexByKey = new Map<string, number>()
for (let i = 0; i < next.length; i++) {
const row = next[i]!
indexByKey.set(`${row.type}:${row.index}`, i)
}
for (const row of event.progress) {
const key = `${row.type}:${row.index}`
const at = indexByKey.get(key)
if (at === undefined) {
indexByKey.set(key, next.length)
next.push(row)
} else {
next[at] = row
}
}
progress = next.length > MAX_PROGRESS_ROWS
? next.slice(next.length - MAX_PROGRESS_ROWS)
: next
}
const agentCount = progress.reduce(
(max, row) => (row.type === 'workflow_agent' ? Math.max(max, row.index) : max),
0,
)
const status = event.status ?? base.status
const isTerminal = status !== 'running'
const merged: WorkflowRun = {
...base,
runId: event.runId ?? base.runId,
workflowName: event.workflowName ?? base.workflowName,
description: event.description ?? base.description,
status,
updatedAt: now,
endedAt: isTerminal ? (base.endedAt ?? now) : undefined,
agentCount: Math.max(base.agentCount, agentCount),
totalTokens: event.totalTokens ?? base.totalTokens,
toolCalls: event.toolCalls ?? base.toolCalls,
progress,
result: event.result ?? base.result,
error: event.error ?? base.error,
}
return hasChanged(base, merged) ? merged : base
}
function hasChanged(before: WorkflowRun, after: WorkflowRun): boolean {
return (
before.progress !== after.progress ||
before.status !== after.status ||
before.totalTokens !== after.totalTokens ||
before.toolCalls !== after.toolCalls ||
before.agentCount !== after.agentCount ||
before.result !== after.result ||
before.error !== after.error ||
before.workflowName !== after.workflowName ||
before.description !== after.description ||
before.runId !== after.runId
)
}
function readProgress(value: unknown): WorkflowProgressEvent[] {
if (!Array.isArray(value)) return []
const rows: WorkflowProgressEvent[] = []
for (const entry of value) {
if (typeof entry !== 'object' || entry === null) continue
const row = entry as Record<string, unknown>
const index = readNumber(row.index)
if (index === undefined) continue
if (row.type === 'workflow_phase' && typeof row.title === 'string') {
rows.push({
type: 'workflow_phase',
index,
title: row.title,
kind: row.kind === 'meta' || row.kind === 'script' ? row.kind : undefined,
})
continue
}
if (row.type === 'workflow_agent' && typeof row.label === 'string') {
rows.push({ ...(row as unknown as WorkflowAgentEvent), index })
}
}
return rows
}
function normalizeStatus(value: string | undefined): WorkflowRunStatus {
switch (value) {
case 'completed':
return 'completed'
case 'failed':
return 'failed'
case 'killed':
case 'stopped':
return 'stopped'
default:
return 'running'
}
}
function readString(value: unknown): string | undefined {
return typeof value === 'string' && value.trim() !== '' ? value : undefined
}
function readNumber(value: unknown): number | undefined {
return typeof value === 'number' && Number.isFinite(value) ? value : undefined
}
+132
View File
@@ -0,0 +1,132 @@
/**
* Desktop-side mirror of the CLI's dynamic-workflow progress protocol.
*
* These shapes arrive over the session WebSocket inside
* `system_notification` → `task_started` / `task_progress` / `task_notification`
* payloads. They are declared here rather than imported from `src/` because the
* desktop bundle must not pull in the CLI runtime.
*/
export type WorkflowAgentRunState = 'start' | 'progress' | 'done' | 'error'
export type WorkflowAgentEvent = {
type: 'workflow_agent'
index: number
label: string
state: WorkflowAgentRunState
phaseIndex?: number
phaseTitle?: string
agentType?: string
isolation?: 'worktree' | 'remote'
model?: string
agentId?: string
queuedAt?: number
startedAt?: number
lastProgressAt?: number
durationMs?: number
tokens?: number
toolCalls?: number
cached?: boolean
skipped?: boolean
blocked?: boolean
error?: string
promptPreview?: string
resultPreview?: string
lastToolName?: string
}
export type WorkflowPhaseEvent = {
type: 'workflow_phase'
index: number
title: string
kind?: 'meta' | 'script'
}
export type WorkflowProgressEvent = WorkflowAgentEvent | WorkflowPhaseEvent
export type WorkflowRunStatus = 'running' | 'completed' | 'failed' | 'stopped'
/** One workflow run as the desktop knows it, live or just finished. */
export type WorkflowRun = {
taskId: string
sessionId: string
runId?: string
workflowName: string
description?: string
status: WorkflowRunStatus
startedAt: number
updatedAt: number
endedAt?: number
agentCount: number
totalTokens: number
toolCalls: number
/** Keyed `workflow_agent:<index>` / `workflow_phase:<index>`, newest value wins. */
progress: WorkflowProgressEvent[]
result?: string
error?: string
}
export type WorkflowPhaseMeta = {
title: string
detail?: string
model?: string
}
export type WorkflowDefinition = {
name: string
description: string
whenToUse?: string
source: 'built-in' | 'userSettings' | 'projectSettings' | 'plugin' | 'inline'
phases?: WorkflowPhaseMeta[]
filePath?: string
script?: string
}
export type WorkflowRunSummary = {
runId: string
sessionId: string
workflowName: string
scriptPath: string
startedAt: number
completedAgents: number
}
export type WorkflowRunDetail = WorkflowRunSummary & {
script: string
description?: string
phases?: WorkflowPhaseMeta[]
agents: Array<{ key: string; agentId: string; result: unknown }>
}
/** A phase with the agents that reported under it. */
export type WorkflowPhaseGroup = {
index: number
title: string
agents: WorkflowAgentEvent[]
}
export function isWorkflowAgentEvent(
event: WorkflowProgressEvent,
): event is WorkflowAgentEvent {
return event.type === 'workflow_agent'
}
/**
* A finished run rebuilt from disk, as the server reconstructs it.
*
* Carries only what outlives the process: which agents ran, under which
* phase. There is no live state here — every agent listed has a transcript,
* which is the fact that it ran.
*/
export type ReconstructedWorkflowRun = {
runId: string
workflowName: string
startedAt: number
agents: Array<{
agentId: string
label: string
phaseIndex: number
phaseTitle?: string
agentIndex: number
}>
}
+2
View File
@@ -59,6 +59,8 @@
"@opentelemetry/sdk-metrics": "^2.6.1",
"@opentelemetry/sdk-trace-base": "^2.6.1",
"@opentelemetry/semantic-conventions": "^1.40.0",
"acorn": "^8.16.0",
"acorn-walk": "^8.3.4",
"ajv": "^8.18.0",
"asciichart": "^1.5.25",
"auto-bind": "^5.0.1",
+3
View File
@@ -15,6 +15,9 @@ export type TaskType =
export type TaskStatus =
| 'pending'
| 'running'
// Only dynamic workflows pause: the run is stopped but its journal is intact,
// so resuming replays completed agents instead of re-running them.
| 'paused'
| 'completed'
| 'failed'
| 'killed'
+6 -10
View File
@@ -84,11 +84,9 @@ const voiceCommand = feature('VOICE_MODE')
const forceSnip = feature('HISTORY_SNIP')
? require('./commands/force-snip.js').default
: null
const workflowsCmd = feature('WORKFLOW_SCRIPTS')
? (
require('./commands/workflows/index.js') as typeof import('./commands/workflows/index.js')
).default
: null
const workflowsCmd = (
require('./commands/workflows/index.js') as typeof import('./commands/workflows/index.js')
).default
const webCmd = feature('CCR_REMOTE_SETUP')
? (
require('./commands/remote-setup/index.js') as typeof import('./commands/remote-setup/index.js')
@@ -400,11 +398,9 @@ async function getSkills(cwd: string): Promise<{
}
/* eslint-disable @typescript-eslint/no-require-imports */
const getWorkflowCommands = feature('WORKFLOW_SCRIPTS')
? (
require('./tools/WorkflowTool/createWorkflowCommand.js') as typeof import('./tools/WorkflowTool/createWorkflowCommand.js')
).getWorkflowCommands
: null
const getWorkflowCommands = (
require('./tools/WorkflowTool/createWorkflowCommand.js') as typeof import('./tools/WorkflowTool/createWorkflowCommand.js')
).getWorkflowCommands
/* eslint-enable @typescript-eslint/no-require-imports */
/**
File diff suppressed because one or more lines are too long
+12 -34
View File
@@ -1,34 +1,12 @@
// @generated stub from scan-missing-imports
// 该文件自动生成,对应 ant-internal 的 feature() gated 模块。
// 所有外部 build 的代码路径在 DCE 后都不会真的执行这里的代码,这只是
// bun build resolver 的占位符。
const __target = function noop() {}
const __handler: ProxyHandler<any> = {
get(_t, prop) {
if (prop === '__esModule') return true
if (prop === 'default') return new Proxy(__target, __handler)
if (prop === Symbol.toPrimitive) return () => undefined
if (prop === Symbol.iterator) return function* () {}
if (prop === Symbol.asyncIterator) return async function* () {}
if (prop === 'then') return undefined
return new Proxy(__target, __handler)
},
apply() {
return new Proxy(__target, __handler)
},
construct() {
return new Proxy(__target, __handler)
},
}
const stub: any = new Proxy(__target, __handler)
export default stub
export const __stubMissing = true
// 兼容常见的命名导出 —— 没列在这里的也会通过 default Proxy 兜底
export const createCachedMCState = stub
export const isCachedMicrocompactEnabled = stub
export const isModelSupportedForCacheEditing = stub
export const getCachedMCConfig = stub
export const markToolsSentToAPI = stub
export const resetCachedMCState = stub
export const checkProtectedNamespace = stub
export const getCoordinatorUserContext = stub
import type { Command } from '../../commands.js'
import { areWorkflowsEnabled } from '../../utils/workflows/enabled.js'
const workflows = {
type: 'local-jsx',
name: 'workflows',
description: 'Watch and manage dynamic workflow runs',
isEnabled: () => areWorkflowsEnabled(),
load: () => import('./workflows.js'),
} satisfies Command
export default workflows
+200
View File
@@ -0,0 +1,200 @@
import figures from 'figures'
import React, { useCallback, useMemo, useState } from 'react'
import type { LocalJSXCommandContext } from '../../commands.js'
import { Byline } from '../../components/design-system/Byline.js'
import { KeyboardShortcutHint } from '../../components/design-system/KeyboardShortcutHint.js'
import {
getTaskStatusColor,
getTaskStatusIcon,
} from '../../components/tasks/taskStatusUtils.js'
import { WorkflowDetailDialog } from '../../components/tasks/WorkflowDetailDialog.js'
import type { KeyboardEvent } from '../../ink/events/keyboard-event.js'
import { Box, Text } from '../../ink.js'
import {
buildResumePrompt,
killWorkflowTask,
pauseWorkflowTask,
retryWorkflowAgent,
skipWorkflowAgent,
type LocalWorkflowTaskState,
} from '../../tasks/LocalWorkflowTask/LocalWorkflowTask.js'
import type { LocalJSXCommandOnDone } from '../../types/command.js'
import type { DeepImmutable } from '../../types/utils.js'
import { saveWorkflowScript } from '../../utils/workflows/save.js'
/**
* `/workflows` — list this session's dynamic workflow runs and open one.
*
* Separate from `/tasks` on purpose: a workflow's interesting state is its
* phase/agent tree, which the generic background-task list cannot show, and
* during a large run the workflow rows would bury every other task.
*/
export async function call(
onDone: LocalJSXCommandOnDone,
context: LocalJSXCommandContext,
): Promise<React.ReactNode> {
return <WorkflowsDialog toolUseContext={context} onDone={onDone} />
}
function WorkflowsDialog({
toolUseContext,
onDone,
}: {
toolUseContext: LocalJSXCommandContext
onDone: LocalJSXCommandOnDone
}): React.ReactNode {
const appState = toolUseContext.getAppState()
const setAppState = toolUseContext.setAppState
const [selectedId, setSelectedId] = useState<string | null>(null)
const [cursor, setCursor] = useState(0)
const [saveScope, setSaveScope] = useState<'user' | 'project' | null>(null)
const runs = useMemo(() => {
const all = Object.values(appState.tasks ?? {}).filter(
(task): task is DeepImmutable<LocalWorkflowTaskState> =>
task.type === 'local_workflow',
)
// Newest first: during a long session the run you just started is the one
// you came here to watch.
return [...all].sort((a, b) => b.startTime - a.startTime)
}, [appState.tasks])
const selected = selectedId
? (runs.find(run => run.id === selectedId) ?? null)
: null
const close = useCallback(() => onDone(), [onDone])
const handleKeyDown = useCallback(
(event: KeyboardEvent) => {
if (selected) return
const key = event.key
if (key === 'up') {
event.preventDefault()
setCursor(prev => Math.max(0, prev - 1))
return
}
if (key === 'down') {
event.preventDefault()
setCursor(prev => Math.min(Math.max(0, runs.length - 1), prev + 1))
return
}
if (key === 'return' || key === 'right') {
event.preventDefault()
const run = runs[cursor]
if (run) setSelectedId(run.id)
return
}
if (key === 'x') {
event.preventDefault()
const run = runs[cursor]
if (run && run.status === 'running') killWorkflowTask(run.id, setAppState)
return
}
if (key === 'escape' || key === 'left' || key === ' ') {
event.preventDefault()
close()
}
},
[close, cursor, runs, selected, setAppState],
)
if (selected) {
return (
<WorkflowDetailDialog
workflow={selected}
onDone={close}
onBack={() => setSelectedId(null)}
onKill={
selected.status === 'running'
? () => killWorkflowTask(selected.id, setAppState)
: undefined
}
onPause={
selected.status === 'running'
? () => {
if (pauseWorkflowTask(selected.id, setAppState)) {
onDone(buildResumePrompt(selected as LocalWorkflowTaskState), {
display: 'system',
})
}
}
: undefined
}
onSkipAgent={
selected.status === 'running'
? key => skipWorkflowAgent(selected.id, key, setAppState)
: undefined
}
onRetryAgent={
selected.status === 'running'
? key => retryWorkflowAgent(selected.id, key, setAppState)
: undefined
}
ultracode={appState.ultracode === true}
saveScope={saveScope}
onToggleSaveScope={() =>
setSaveScope(prev =>
prev === null ? 'project' : prev === 'project' ? 'user' : 'project',
)
}
onSave={async () => {
const scope = saveScope ?? 'project'
const result = await saveWorkflowScript({
script: selected.script,
scope,
})
onDone(
'error' in result
? `Could not save workflow: ${result.error}`
: `Saved /${result.name} to ${result.filePath}`,
{ display: 'system' },
)
}}
/>
)
}
return (
<Box flexDirection="column" tabIndex={0} autoFocus onKeyDown={handleKeyDown}>
<Box
flexDirection="column"
borderStyle="round"
borderColor="permission"
paddingX={1}
>
<Text bold>Dynamic workflows</Text>
{runs.length === 0 ? (
<Text dimColor>
No workflow runs in this session. Ask Claude to &quot;use a
workflow&quot;, or run a saved one with /&lt;name&gt;.
</Text>
) : (
runs.map((run, index) => {
const isSelected = index === cursor
return (
<Box key={run.id}>
<Text color={isSelected ? 'permission' : undefined}>
{`${isSelected ? figures.pointer : ' '} `}
</Text>
<Text color={getTaskStatusColor(run.status)}>
{getTaskStatusIcon(run.status)}
</Text>
<Text>{` ${run.workflowName ?? 'workflow'}`}</Text>
<Text dimColor>
{` ${run.agentCount} agents · ${run.totalTokens} tok · ${run.status}`}
</Text>
</Box>
)
})
)}
</Box>
<Byline>
<KeyboardShortcutHint shortcut="↑/↓" action="select" />
<KeyboardShortcutHint shortcut="Enter" action="open" />
<KeyboardShortcutHint shortcut="x" action="stop" />
<KeyboardShortcutHint shortcut="Esc" action="close" />
</Byline>
</Box>
)
}
+69 -1
View File
@@ -16,6 +16,8 @@ import { companionReservedColumns } from '../../buddy/CompanionSprite.js';
import { findBuddyTriggerPositions, useBuddyNotification } from '../../buddy/useBuddyNotification.js';
import { FastModePicker } from '../../commands/fast/fast.js';
import { isUltrareviewEnabled } from '../../commands/review/ultrareviewEnabled.js';
import { isWorkflowKeywordTriggerEnabled } from '../../utils/workflows/enabled.js';
import { findKeywordRanges } from '../../utils/workflows/keyword.js';
import { getNativeCSIuTerminalDisplayName } from '../../commands/terminalSetup/terminalSetup.js';
import { type Command, hasCommand } from '../../commands.js';
import { useIsModalOverlayActive } from '../../context/overlayContext.js';
@@ -518,6 +520,24 @@ function PromptInput({
const ultraplanLaunching = useAppState(s => s.ultraplanLaunching);
const ultraplanTriggers = useMemo(() => feature('ULTRAPLAN') && !ultraplanSessionUrl && !ultraplanLaunching ? findUltraplanTriggerPositions(displayedValue) : [], [displayedValue, ultraplanSessionUrl, ultraplanLaunching]);
const ultrareviewTriggers = useMemo(() => isUltrareviewEnabled() ? findUltrareviewTriggerPositions(displayedValue) : [], [displayedValue]);
// `ultracode` opts this turn into multi-agent orchestration. The highlight is
// the only warning the user gets before a one-line prompt becomes a run that
// spawns dozens of agents, so it must be visible before submit.
const [ultracodeKeywordDismissed, setUltracodeKeywordDismissed] = useState(false);
const ultracodeTriggers = useMemo(() => isWorkflowKeywordTriggerEnabled() && !ultracodeKeywordDismissed ? findKeywordRanges(displayedValue) : [], [displayedValue, ultracodeKeywordDismissed]);
const hasUltracodeKeyword = useMemo(() => isWorkflowKeywordTriggerEnabled() ? findKeywordRanges(displayedValue).length > 0 : false, [displayedValue]);
// A fresh prompt is opted in again — the dismissal is per-prompt, not sticky.
// The AppState mirror is what actually suppresses the reminder; the local
// flag only drives the highlight, so both have to move together.
useEffect(() => {
if (!hasUltracodeKeyword && ultracodeKeywordDismissed) {
setUltracodeKeywordDismissed(false);
setAppState(prev => prev.suppressWorkflowKeyword ? {
...prev,
suppressWorkflowKeyword: false
} : prev);
}
}, [hasUltracodeKeyword, ultracodeKeywordDismissed, setAppState]);
const btwTriggers = useMemo(() => findBtwTriggerPositions(displayedValue), [displayedValue]);
const buddyTriggers = useMemo(() => findBuddyTriggerPositions(displayedValue), [displayedValue]);
const slashCommandTriggers = useMemo(() => {
@@ -722,6 +742,19 @@ function PromptInput({
}
}
// Same rainbow treatment for the ultracode keyword
for (const trigger of ultracodeTriggers) {
for (let i = trigger.start; i < trigger.end; i++) {
highlights.push({
start: i,
end: i + 1,
color: getRainbowColor(i - trigger.start),
shimmerColor: getRainbowColor(i - trigger.start, true),
priority: 10
});
}
}
// Rainbow for /buddy
for (const trigger of buddyTriggers) {
for (let i = trigger.start; i < trigger.end; i++) {
@@ -735,7 +768,7 @@ function PromptInput({
}
}
return highlights;
}, [isSearchingHistory, historyQuery, historyMatch, historyFailedMatch, cursorOffset, btwTriggers, imageRefPositions, memberMentionHighlights, slashCommandTriggers, tokenBudgetTriggers, slackChannelTriggers, displayedValue, voiceInterimRange, thinkTriggers, ultraplanTriggers, ultrareviewTriggers, buddyTriggers]);
}, [isSearchingHistory, historyQuery, historyMatch, historyFailedMatch, cursorOffset, btwTriggers, imageRefPositions, memberMentionHighlights, slashCommandTriggers, tokenBudgetTriggers, slackChannelTriggers, displayedValue, voiceInterimRange, thinkTriggers, ultraplanTriggers, ultrareviewTriggers, ultracodeTriggers, buddyTriggers]);
const {
addNotification,
removeNotification
@@ -766,6 +799,41 @@ function PromptInput({
removeNotification('ultraplan-active');
}
}, [addNotification, removeNotification, ultraplanTriggers.length]);
useEffect(() => {
if (ultracodeTriggers.length) {
addNotification({
key: 'ultracode-active',
text: 'Ultracode: this prompt runs as a workflow · opt+w to undo',
priority: 'immediate',
timeoutMs: 5000
});
} else {
removeNotification('ultracode-active');
}
}, [addNotification, removeNotification, ultracodeTriggers.length]);
useKeybinding('chat:workflowKeywordToggle', () => {
if (!hasUltracodeKeyword) return;
setUltracodeKeywordDismissed(prev => {
const next = !prev;
setAppState(state => ({
...state,
suppressWorkflowKeyword: next
}));
if (next) {
addNotification({
key: 'workflow-keyword-ignored',
text: 'Ultracode keyword ignored for this prompt · opt+w to undo',
priority: 'immediate',
timeoutMs: 5000
});
} else {
removeNotification('workflow-keyword-ignored');
}
return next;
});
}, {
context: 'Chat'
});
useEffect(() => {
if (isUltrareviewEnabled() && ultrareviewTriggers.length) {
addNotification({
+53
View File
@@ -36,6 +36,7 @@ import { useIsInsideModal } from '../../context/modalContext.js';
import { SearchBox } from '../SearchBox.js';
import { isSupportedTerminal, hasAccessToIDEExtensionDiffFeature } from '../../utils/ide.js';
import { getInitialSettings, getSettingsForSource, updateSettingsForSource } from '../../utils/settings/settings.js';
import { getWorkflowSizeGuidelineFromSettings, WORKFLOW_SIZE_GUIDELINES } from '../../utils/workflows/enabled.js';
import { getUserMsgOptIn, setUserMsgOptIn } from '../../bootstrap/state.js';
import { DEFAULT_OUTPUT_STYLE_NAME } from 'src/constants/outputStyles.js';
import { isEnvTruthy, isRunningOnHomespace } from 'src/utils/envUtils.js';
@@ -431,6 +432,58 @@ export function Config({
enabled: enabled_3
});
}
}] : []), {
id: 'workflows',
label: 'Dynamic workflows',
value: settingsData?.disableWorkflows === true ? false : settingsData?.enableWorkflows ?? true,
type: 'boolean' as const,
onChange(workflowsEnabled: boolean) {
// Writes `enableWorkflows` and clears `disableWorkflows`: the two keys
// are separate so managed settings can hard-disable the feature without
// stomping the user's own toggle underneath.
updateSettingsForSource('userSettings', {
enableWorkflows: workflowsEnabled ? undefined : false,
disableWorkflows: undefined
});
setSettingsData(getInitialSettings());
logEvent('tengu_workflows_setting_changed', {
enabled: workflowsEnabled
});
}
}, {
id: 'workflowKeywordTriggerEnabled',
label: 'Ultracode keyword trigger',
value: settingsData?.workflowKeywordTriggerEnabled ?? true,
type: 'boolean' as const,
onChange(keywordEnabled: boolean) {
updateSettingsForSource('userSettings', {
workflowKeywordTriggerEnabled: keywordEnabled ? undefined : false
});
setSettingsData(getInitialSettings());
logEvent('tengu_workflow_keyword_trigger_setting_changed', {
enabled: keywordEnabled
});
}
}, ...(getWorkflowSizeGuidelineFromSettings() === undefined ? [{
id: 'workflowSizeGuideline',
label: 'Dynamic workflow size',
value: globalConfig.workflowSizeGuideline ?? 'medium (default)',
options: [...WORKFLOW_SIZE_GUIDELINES],
type: 'enum' as const,
onChange(size: string) {
const guideline = ((WORKFLOW_SIZE_GUIDELINES as readonly string[]).includes(size) ? size : 'unrestricted') as (typeof WORKFLOW_SIZE_GUIDELINES)[number];
saveGlobalConfig(currentWf => currentWf.workflowSizeGuideline === guideline ? currentWf : {
...currentWf,
workflowSizeGuideline: guideline
});
setGlobalConfig({
...getGlobalConfig(),
workflowSizeGuideline: guideline
});
logEvent('tengu_workflow_size_guideline_changed', {
size: guideline as AnalyticsMetadata_I_VERIFIED_THIS_IS_NOT_CODE_OR_FILEPATHS
});
}
}] : []), {
id: 'verbose',
label: 'Verbose output',
@@ -35,6 +35,10 @@ const NULL_RENDERING_TYPES = [
'companion_intro',
'token_usage',
'ultrathink_effort',
'workflow_keyword_request',
'ultra_effort_enter',
'ultra_effort_exit',
'workflow_size_guideline_change',
'max_turns_reached',
'task_reminder',
'auto_mode',
@@ -35,8 +35,8 @@ import { WebFetchPermissionRequest } from './WebFetchPermissionRequest/WebFetchP
/* eslint-disable @typescript-eslint/no-require-imports */
const ReviewArtifactTool = feature('REVIEW_ARTIFACT') ? (require('../../tools/ReviewArtifactTool/ReviewArtifactTool.js') as typeof import('../../tools/ReviewArtifactTool/ReviewArtifactTool.js')).ReviewArtifactTool : null;
const ReviewArtifactPermissionRequest = feature('REVIEW_ARTIFACT') ? (require('./ReviewArtifactPermissionRequest/ReviewArtifactPermissionRequest.js') as typeof import('./ReviewArtifactPermissionRequest/ReviewArtifactPermissionRequest.js')).ReviewArtifactPermissionRequest : null;
const WorkflowTool = feature('WORKFLOW_SCRIPTS') ? (require('../../tools/WorkflowTool/WorkflowTool.js') as typeof import('../../tools/WorkflowTool/WorkflowTool.js')).WorkflowTool : null;
const WorkflowPermissionRequest = feature('WORKFLOW_SCRIPTS') ? (require('../../tools/WorkflowTool/WorkflowPermissionRequest.js') as typeof import('../../tools/WorkflowTool/WorkflowPermissionRequest.js')).WorkflowPermissionRequest : null;
const WorkflowTool = (require('../../tools/WorkflowTool/WorkflowTool.js') as typeof import('../../tools/WorkflowTool/WorkflowTool.js')).WorkflowTool;
const WorkflowPermissionRequest = (require('../../tools/WorkflowTool/WorkflowPermissionRequest.js') as typeof import('../../tools/WorkflowTool/WorkflowPermissionRequest.js')).WorkflowPermissionRequest;
const MonitorTool = feature('MONITOR_TOOL') ? (require('../../tools/MonitorTool/MonitorTool.js') as typeof import('../../tools/MonitorTool/MonitorTool.js')).MonitorTool : null;
const MonitorPermissionRequest = feature('MONITOR_TOOL') ? (require('./MonitorPermissionRequest/MonitorPermissionRequest.js') as typeof import('./MonitorPermissionRequest/MonitorPermissionRequest.js')).MonitorPermissionRequest : null;
import type { ContentBlockParam } from '@anthropic-ai/sdk/resources/messages.mjs';
@@ -69,7 +69,7 @@ function permissionComponentForTool(tool: Tool): React.ComponentType<PermissionR
case AskUserQuestionTool:
return AskUserQuestionPermissionRequest;
case WorkflowTool:
return WorkflowPermissionRequest ?? FallbackPermissionRequest;
return WorkflowPermissionRequest;
case MonitorTool:
return MonitorPermissionRequest ?? FallbackPermissionRequest;
case GlobTool:
@@ -102,12 +102,11 @@ type ListItem = {
status: 'running';
};
// WORKFLOW_SCRIPTS is ant-only (build_flags.yaml). Static imports would leak
// ~1.3K lines into external builds. Gate with feature() + require so the
// bundler can dead-code-eliminate the branch.
// Lazy require, not a static import: the workflow detail dialog and the task
// module both reach back into the tools/tasks graph this file is part of.
/* eslint-disable @typescript-eslint/no-require-imports */
const WorkflowDetailDialog = feature('WORKFLOW_SCRIPTS') ? (require('./WorkflowDetailDialog.js') as typeof import('./WorkflowDetailDialog.js')).WorkflowDetailDialog : null;
const workflowTaskModule = feature('WORKFLOW_SCRIPTS') ? require('src/tasks/LocalWorkflowTask/LocalWorkflowTask.js') as typeof import('src/tasks/LocalWorkflowTask/LocalWorkflowTask.js') : null;
const WorkflowDetailDialog = (require('./WorkflowDetailDialog.js') as typeof import('./WorkflowDetailDialog.js')).WorkflowDetailDialog;
const workflowTaskModule = require('src/tasks/LocalWorkflowTask/LocalWorkflowTask.js') as typeof import('src/tasks/LocalWorkflowTask/LocalWorkflowTask.js');
const killWorkflowTask = workflowTaskModule?.killWorkflowTask ?? null;
const skipWorkflowAgent = workflowTaskModule?.skipWorkflowAgent ?? null;
const retryWorkflowAgent = workflowTaskModule?.retryWorkflowAgent ?? null;
@@ -1,34 +0,0 @@
// @generated stub from scan-missing-imports
// 该文件自动生成,对应 ant-internal 的 feature() gated 模块。
// 所有外部 build 的代码路径在 DCE 后都不会真的执行这里的代码,这只是
// bun build resolver 的占位符。
const __target = function noop() {}
const __handler: ProxyHandler<any> = {
get(_t, prop) {
if (prop === '__esModule') return true
if (prop === 'default') return new Proxy(__target, __handler)
if (prop === Symbol.toPrimitive) return () => undefined
if (prop === Symbol.iterator) return function* () {}
if (prop === Symbol.asyncIterator) return async function* () {}
if (prop === 'then') return undefined
return new Proxy(__target, __handler)
},
apply() {
return new Proxy(__target, __handler)
},
construct() {
return new Proxy(__target, __handler)
},
}
const stub: any = new Proxy(__target, __handler)
export default stub
export const __stubMissing = true
// 兼容常见的命名导出 —— 没列在这里的也会通过 default Proxy 兜底
export const createCachedMCState = stub
export const isCachedMicrocompactEnabled = stub
export const isModelSupportedForCacheEditing = stub
export const getCachedMCConfig = stub
export const markToolsSentToAPI = stub
export const resetCachedMCState = stub
export const checkProtectedNamespace = stub
export const getCoordinatorUserContext = stub
@@ -0,0 +1,572 @@
import figures from 'figures'
import React, { useCallback, useMemo, useState } from 'react'
import { useElapsedTime } from '../../hooks/useElapsedTime.js'
import type { KeyboardEvent } from '../../ink/events/keyboard-event.js'
import { Box, Text } from '../../ink.js'
import type { LocalWorkflowTaskState } from '../../tasks/LocalWorkflowTask/LocalWorkflowTask.js'
import type { DeepImmutable } from '../../types/utils.js'
import { getLargeWorkflowWarning } from '../../utils/workflows/enabled.js'
import type {
WorkflowAgentEvent,
WorkflowProgressEvent,
} from '../../utils/workflows/types.js'
import { Byline } from '../design-system/Byline.js'
import { KeyboardShortcutHint } from '../design-system/KeyboardShortcutHint.js'
import { getTaskStatusColor, getTaskStatusIcon } from './taskStatusUtils.js'
type Props = {
workflow: DeepImmutable<LocalWorkflowTaskState>
onDone: () => void
onBack?: () => void
onKill?: () => void
onPause?: () => void
onSkipAgent?: (agentKey: string) => void
onRetryAgent?: (agentKey: string) => void
onSave?: () => void
/** `null` until the user presses Tab; then the destination they picked. */
saveScope?: 'user' | 'project' | null
onToggleSaveScope?: () => void
/** Suppresses the large-run advisory — ultracode already opts in to scale. */
ultracode?: boolean
}
type AgentFilter = 'all' | 'running' | 'done' | 'error'
const AGENT_FILTERS: AgentFilter[] = ['all', 'running', 'done', 'error']
/** Rows visible in the agent list before it scrolls. */
const VISIBLE_AGENTS = 12
/** Log lines shown at the run level. */
const VISIBLE_LOGS = 6
type PhaseGroup = {
index: number
title: string
agents: WorkflowAgentEvent[]
}
/**
* Live progress view for one dynamic workflow run.
*
* Three levels: the run (phases with totals), a phase (its agents), and one
* agent (its prompt, model, and result). The whole thing is derived from the
* task's `workflowProgress` rows — the runtime is the single source of truth
* and this component holds no state beyond where the cursor is.
*/
export function WorkflowDetailDialog({
workflow,
onDone,
onBack,
onKill,
onPause,
onSkipAgent,
onRetryAgent,
onSave,
saveScope,
onToggleSaveScope,
ultracode,
}: Props): React.ReactNode {
const elapsed = useElapsedTime(
workflow.startTime,
workflow.status === 'running',
1000,
0,
)
const [selectedPhase, setSelectedPhase] = useState<number | null>(null)
const [selectedAgentIndex, setSelectedAgentIndex] = useState<number | null>(
null,
)
const [cursor, setCursor] = useState(0)
const [filter, setFilter] = useState<AgentFilter>('all')
const { phases, orphanAgents, logs } = useMemo(
() => groupProgress(workflow.workflowProgress as WorkflowProgressEvent[]),
[workflow.workflowProgress],
)
const activePhase =
selectedPhase === null
? null
: (phases.find(phase => phase.index === selectedPhase) ??
(selectedPhase === 0
? { index: 0, title: 'Ungrouped', agents: orphanAgents }
: null))
const visibleAgents = useMemo(
() => (activePhase ? applyFilter(activePhase.agents, filter) : []),
[activePhase, filter],
)
const selectedAgent =
selectedAgentIndex === null
? null
: (visibleAgents.find(agent => agent.index === selectedAgentIndex) ?? null)
const rowCount =
activePhase === null
? phases.length + (orphanAgents.length > 0 ? 1 : 0)
: visibleAgents.length
const handleKeyDown = useCallback(
(event: KeyboardEvent) => {
const key = event.key
if (key === 'up' || key === 'k') {
event.preventDefault()
setCursor(prev => Math.max(0, prev - 1))
return
}
if (key === 'down' || key === 'j') {
event.preventDefault()
setCursor(prev => Math.min(Math.max(0, rowCount - 1), prev + 1))
return
}
if (key === 'return' || key === 'right') {
event.preventDefault()
if (selectedAgent) return
if (activePhase === null) {
const rows =
orphanAgents.length > 0
? [...phases, { index: 0, title: 'Ungrouped', agents: orphanAgents }]
: phases
const target = rows[cursor]
if (target) {
setSelectedPhase(target.index)
setCursor(0)
}
return
}
const target = visibleAgents[cursor]
if (target) setSelectedAgentIndex(target.index)
return
}
if (key === 'escape' || key === 'left') {
event.preventDefault()
if (selectedAgentIndex !== null) {
setSelectedAgentIndex(null)
return
}
if (selectedPhase !== null) {
setSelectedPhase(null)
setCursor(0)
return
}
if (onBack) onBack()
else onDone()
return
}
if (key === 'f' && activePhase !== null) {
event.preventDefault()
setFilter(prev => {
const next = AGENT_FILTERS[(AGENT_FILTERS.indexOf(prev) + 1) % AGENT_FILTERS.length]
return next ?? 'all'
})
setCursor(0)
return
}
if (key === 'x') {
event.preventDefault()
if (selectedAgent && onSkipAgent) {
onSkipAgent(agentKey(workflow.workflowRunId, selectedAgent.index))
return
}
if (workflow.status === 'running' && onKill) onKill()
return
}
if (key === 'r' && selectedAgent && onRetryAgent) {
event.preventDefault()
onRetryAgent(agentKey(workflow.workflowRunId, selectedAgent.index))
return
}
if (key === 'p' && workflow.status === 'running' && onPause) {
event.preventDefault()
onPause()
return
}
if (key === 'tab' && onToggleSaveScope) {
event.preventDefault()
onToggleSaveScope()
return
}
if (key === 's' && onSave) {
event.preventDefault()
onSave()
return
}
if (key === ' ') {
event.preventDefault()
onDone()
}
},
[
activePhase,
cursor,
onBack,
onDone,
onKill,
onPause,
onRetryAgent,
onSave,
onSkipAgent,
onToggleSaveScope,
orphanAgents,
phases,
rowCount,
selectedAgent,
selectedAgentIndex,
selectedPhase,
visibleAgents,
workflow.status,
workflow.workflowRunId,
],
)
const statusColor = getTaskStatusColor(workflow.status)
const statusIcon = getTaskStatusIcon(workflow.status)
const started = phases
.flatMap(phase => phase.agents)
.concat(orphanAgents)
.filter(agent => agent.state !== 'start').length
const largeWarning =
workflow.status === 'running'
? getLargeWorkflowWarning({
scheduledAgents: workflow.agentCount,
startedAgents: started,
totalTokens: workflow.totalTokens,
ultracodeActive: ultracode === true,
})
: undefined
return (
<Box flexDirection="column" tabIndex={0} autoFocus onKeyDown={handleKeyDown}>
<Box flexDirection="column" borderStyle="round" borderColor="permission" paddingX={1}>
<Box>
<Text bold>{workflow.workflowName ?? 'Dynamic workflow'}</Text>
<Text dimColor>{` ${workflow.workflowRunId}`}</Text>
</Box>
{workflow.summary ? <Text dimColor>{workflow.summary}</Text> : null}
<Box marginTop={1}>
<Text color={statusColor}>{`${statusIcon} ${workflow.status}`}</Text>
<Text dimColor>
{` · ${workflow.agentCount} agents · ${formatTokens(workflow.totalTokens)} tok · ${workflow.totalToolCalls} tools · ${formatElapsed(elapsed)}`}
</Text>
</Box>
{workflow.error ? (
<Box marginTop={1}>
<Text color="error">{workflow.error}</Text>
</Box>
) : null}
{largeWarning ? (
<Box marginTop={1}>
<Text color="warning">
{`⚠ Large workflow — ${describeLargeWarning(largeWarning)}. Press x to stop.`}
</Text>
</Box>
) : null}
<Box flexDirection="column" marginTop={1}>
{selectedAgent ? (
<AgentDetail agent={selectedAgent} />
) : activePhase ? (
<AgentList
phase={activePhase}
agents={visibleAgents}
cursor={cursor}
filter={filter}
/>
) : (
<PhaseList
phases={phases}
orphanAgents={orphanAgents}
cursor={cursor}
/>
)}
</Box>
{!activePhase && logs.length > 0 ? (
<Box flexDirection="column" marginTop={1}>
<Text dimColor>Log</Text>
{logs.slice(-VISIBLE_LOGS).map((line, index) => (
<Text key={`${index}-${line.slice(0, 12)}`} dimColor>
{` ${line}`}
</Text>
))}
</Box>
) : null}
</Box>
<Byline>
<KeyboardShortcutHint shortcut="↑/↓" action="select" />
<KeyboardShortcutHint shortcut="Enter/→" action="open" />
<KeyboardShortcutHint shortcut="←/Esc" action="back" />
{activePhase ? <KeyboardShortcutHint shortcut="f" action={`filter: ${filter}`} /> : null}
{workflow.status === 'running' && onPause ? (
<KeyboardShortcutHint shortcut="p" action="pause" />
) : null}
{selectedAgent && onSkipAgent ? (
<KeyboardShortcutHint shortcut="x" action="skip agent" />
) : workflow.status === 'running' && onKill ? (
<KeyboardShortcutHint shortcut="x" action="stop" />
) : null}
{selectedAgent && onRetryAgent ? (
<KeyboardShortcutHint shortcut="r" action="restart" />
) : null}
{onSave ? (
<KeyboardShortcutHint
shortcut="s"
action={`save to ${saveScope ?? 'project'}`}
/>
) : null}
{onToggleSaveScope ? (
<KeyboardShortcutHint shortcut="Tab" action="save location" />
) : null}
</Byline>
</Box>
)
}
function PhaseList({
phases,
orphanAgents,
cursor,
}: {
phases: PhaseGroup[]
orphanAgents: WorkflowAgentEvent[]
cursor: number
}): React.ReactNode {
const rows =
orphanAgents.length > 0
? [...phases, { index: 0, title: 'Ungrouped', agents: orphanAgents }]
: phases
if (rows.length === 0) {
return <Text dimColor>Waiting for the first agent…</Text>
}
return (
<Box flexDirection="column">
{rows.map((phase, rowIndex) => {
const totals = summarize(phase.agents)
const selected = rowIndex === cursor
return (
<Box key={`${phase.index}-${phase.title}`}>
<Text color={selected ? 'permission' : undefined}>
{`${selected ? figures.pointer : ' '} ${phase.title}`}
</Text>
<Text dimColor>
{` ${totals.done}/${phase.agents.length} done`}
{totals.errors > 0 ? ` ${totals.errors} failed` : ''}
{` · ${formatTokens(totals.tokens)} tok`}
</Text>
</Box>
)
})}
</Box>
)
}
function AgentList({
phase,
agents,
cursor,
filter,
}: {
phase: PhaseGroup
agents: WorkflowAgentEvent[]
cursor: number
filter: AgentFilter
}): React.ReactNode {
if (agents.length === 0) {
return <Text dimColor>{`No ${filter} agents in ${phase.title}.`}</Text>
}
const start = Math.max(0, Math.min(cursor - VISIBLE_AGENTS + 1, agents.length - VISIBLE_AGENTS))
const window = agents.slice(Math.max(0, start), Math.max(0, start) + VISIBLE_AGENTS)
return (
<Box flexDirection="column">
<Text dimColor>{`${phase.title} — ${agents.length} agent(s), filter: ${filter}`}</Text>
{window.map(agent => {
const selected = agent.index === agents[cursor]?.index
return (
<Box key={agent.index}>
<Text color={selected ? 'permission' : undefined}>
{`${selected ? figures.pointer : ' '} ${agentIcon(agent)} ${agent.label}`}
</Text>
<Text dimColor>
{agent.cached ? ' cached' : ''}
{agent.tokens ? ` ${formatTokens(agent.tokens)} tok` : ''}
{agent.toolCalls ? ` ${agent.toolCalls} tools` : ''}
{agent.error ? ` ${agent.error}` : ''}
</Text>
</Box>
)
})}
</Box>
)
}
function AgentDetail({ agent }: { agent: WorkflowAgentEvent }): React.ReactNode {
return (
<Box flexDirection="column">
<Text bold>{`${agentIcon(agent)} ${agent.label}`}</Text>
<Text dimColor>
{[
agent.model ? `model ${agent.model}` : null,
agent.agentType ? `agent ${agent.agentType}` : null,
agent.isolation ? `isolation ${agent.isolation}` : null,
agent.agentId ? `id ${agent.agentId}` : null,
]
.filter(Boolean)
.join(' · ')}
</Text>
<Text dimColor>
{`${formatTokens(agent.tokens ?? 0)} tok · ${agent.toolCalls ?? 0} tools`}
{agent.durationMs ? ` · ${Math.round(agent.durationMs / 1000)}s` : ''}
</Text>
{agent.promptPreview ? (
<Box flexDirection="column" marginTop={1}>
<Text dimColor>Prompt</Text>
<Text>{agent.promptPreview}</Text>
</Box>
) : null}
{agent.resultPreview ? (
<Box flexDirection="column" marginTop={1}>
<Text dimColor>Result</Text>
<Text>{agent.resultPreview}</Text>
</Box>
) : null}
{agent.error ? (
<Box marginTop={1}>
<Text color="error">{agent.error}</Text>
</Box>
) : null}
</Box>
)
}
/**
* Fold the flat progress stream into phases.
*
* Agents that ran before any `phase()` call have no `phaseIndex`; they are
* collected separately rather than dropped, so a script that never calls
* `phase()` still shows all its work.
*/
function groupProgress(events: readonly WorkflowProgressEvent[]): {
phases: PhaseGroup[]
orphanAgents: WorkflowAgentEvent[]
logs: string[]
} {
const phaseByIndex = new Map<number, PhaseGroup>()
const orphanAgents: WorkflowAgentEvent[] = []
const logs: string[] = []
for (const event of events) {
if (event.type === 'workflow_phase') {
if (!phaseByIndex.has(event.index)) {
phaseByIndex.set(event.index, {
index: event.index,
title: event.title,
agents: [],
})
}
continue
}
if (event.type === 'workflow_log') {
logs.push(event.message)
continue
}
if (event.phaseIndex === undefined) {
orphanAgents.push(event)
continue
}
const phase = phaseByIndex.get(event.phaseIndex) ?? {
index: event.phaseIndex,
title: event.phaseTitle ?? `Phase ${event.phaseIndex}`,
agents: [],
}
phase.agents.push(event)
phaseByIndex.set(event.phaseIndex, phase)
}
return {
phases: [...phaseByIndex.values()].sort((a, b) => a.index - b.index),
orphanAgents,
logs,
}
}
function applyFilter(
agents: WorkflowAgentEvent[],
filter: AgentFilter,
): WorkflowAgentEvent[] {
switch (filter) {
case 'running':
return agents.filter(
agent => agent.state === 'start' || agent.state === 'progress',
)
case 'done':
return agents.filter(agent => agent.state === 'done')
case 'error':
return agents.filter(agent => agent.state === 'error')
default:
return agents
}
}
function summarize(agents: WorkflowAgentEvent[]): {
done: number
errors: number
tokens: number
} {
let done = 0
let errors = 0
let tokens = 0
for (const agent of agents) {
if (agent.state === 'done') done++
if (agent.state === 'error') errors++
tokens += agent.tokens ?? 0
}
return { done, errors, tokens }
}
function agentIcon(agent: WorkflowAgentEvent): string {
switch (agent.state) {
case 'done':
return figures.tick
case 'error':
return figures.cross
case 'progress':
return figures.play
default:
return figures.bullet
}
}
function describeLargeWarning(
warning: NonNullable<ReturnType<typeof getLargeWorkflowWarning>>,
): string {
const parts: string[] = []
if (warning.axis !== 'tokens') {
parts.push(
`${warning.scheduledAgents} agents scheduled (over ${warning.agentCap}${warning.capFromGuideline ? ', from your size guideline' : ''})`,
)
}
if (warning.axis !== 'agents') {
parts.push(
`projected ${formatTokens(warning.projectedTokens)} tokens (over ${formatTokens(warning.tokenCap)})`,
)
}
return parts.join(' and ')
}
/** Agent controller key, mirroring the harness's `${runId}-${index}`. */
function agentKey(runId: string, index: number): string {
return `${runId}-${index}`
}
function formatTokens(tokens: number): string {
if (tokens >= 1_000_000) return `${(tokens / 1_000_000).toFixed(1)}M`
if (tokens >= 1_000) return `${(tokens / 1_000).toFixed(1)}k`
return String(tokens)
}
function formatElapsed(seconds: number): string {
if (seconds < 60) return `${seconds}s`
const minutes = Math.floor(seconds / 60)
return `${minutes}m ${seconds % 60}s`
}
+1 -1
View File
@@ -42,7 +42,7 @@ export const ALL_AGENT_DISALLOWED_TOOLS = new Set([
ASK_USER_QUESTION_TOOL_NAME,
TASK_STOP_TOOL_NAME,
// Prevent recursive workflow execution inside subagents.
...(feature('WORKFLOW_SCRIPTS') ? [WORKFLOW_TOOL_NAME] : []),
WORKFLOW_TOOL_NAME,
])
export const CUSTOM_AGENT_DISALLOWED_TOOLS = new Set([
+1
View File
@@ -70,6 +70,7 @@ export const DEFAULT_BINDINGS: KeybindingBlock[] = [
'meta+p': 'chat:modelPicker',
'meta+o': 'chat:fastMode',
'meta+t': 'chat:thinkingToggle',
'meta+w': 'chat:workflowKeywordToggle',
enter: 'chat:submit',
up: 'history:previous',
down: 'history:next',
+1
View File
@@ -84,6 +84,7 @@ export const KEYBINDING_ACTIONS = [
'chat:modelPicker',
'chat:fastMode',
'chat:thinkingToggle',
'chat:workflowKeywordToggle',
'chat:submit',
'chat:newline',
'chat:undo',
+16 -3
View File
@@ -52,6 +52,7 @@ import { getSubscriptionType, isClaudeAISubscriber, prefetchAwsCredentialsAndBed
import { checkHasTrustDialogAccepted, getGlobalConfig, getRemoteControlAtStartup, isAutoUpdaterDisabled, saveGlobalConfig } from './utils/config.js';
import { seedEarlyInput, stopCapturingEarlyInput } from './utils/earlyInput.js';
import { getInitialEffortSetting, parseEffortValue } from './utils/effort.js';
import { ULTRACODE_EFFORT_ARG, ULTRACODE_EFFORT_LEVEL } from './utils/workflows/ultracode.js';
import { getInitialFastModeSetting, isFastModeEnabled, prefetchFastModeStatus, resolveFastModeStatusFromCache } from './utils/fastMode.js';
import { applyConfigEnvironmentVariables } from './utils/managedEnv.js';
import { createSystemMessage, createUserMessage } from './utils/messages.js';
@@ -991,10 +992,16 @@ async function run(): Promise<CommanderCommand> {
return Number.isFinite(n) ? n : undefined;
}).hideHelp()).option('--from-pr [value]', 'Resume a session linked to a PR by PR number/URL, or open interactive picker with optional search term', value => value || true).option('--no-session-persistence', 'Disable session persistence - sessions will not be saved to disk and cannot be resumed (only works with --print)').addOption(new Option('--resume-session-at <message id>', 'When resuming, only messages up to and including the assistant message with <message.id> (use with --resume in print mode)').argParser(String).hideHelp()).addOption(new Option('--rewind-files <user-message-id>', 'Restore files to state at the specified user message and exit (requires --resume)').hideHelp())
// @[MODEL LAUNCH]: Update the example model ID in the --model help text.
.option('--model <model>', `Model for the current session. Provide an alias for the latest model (e.g. 'sonnet' or 'opus') or a model's full name (e.g. 'claude-sonnet-4-6').`).addOption(new Option('--effort <level>', `Effort level for the current session (low, medium, high, xhigh, max, or an integer)`).argParser((rawValue: string) => {
.option('--model <model>', `Model for the current session. Provide an alias for the latest model (e.g. 'sonnet' or 'opus') or a model's full name (e.g. 'claude-sonnet-4-6').`).addOption(new Option('--effort <level>', `Effort level for the current session (low, medium, high, xhigh, max, ultracode, or an integer)`).argParser((rawValue: string) => {
// `ultracode` is passed through as-is: it is not an effort level but a
// request for xhigh plus standing workflow orchestration, resolved later
// where the session's model and workflow availability are known.
if (rawValue.toLowerCase() === ULTRACODE_EFFORT_ARG) {
return ULTRACODE_EFFORT_ARG;
}
const value = parseEffortValue(rawValue);
if (value === undefined) {
throw new InvalidArgumentError('It must be one of: low, medium, high, xhigh, max, or an integer');
throw new InvalidArgumentError('It must be one of: low, medium, high, xhigh, max, ultracode, or an integer');
}
return value;
})).option('--agent <agent>', `Agent for the current session. Overrides the 'agent' setting.`).option('--betas <betas...>', 'Beta headers to include in API requests (API key users only)').option('--fallback-model <model>', 'Enable automatic fallback to specified model when default model is overloaded (only works with --print)').addOption(new Option('--workload <tag>', 'Workload tag for billing-header attribution (cc_workload). Process-scoped; set by SDK daemon callers that spawn subprocesses for cron work. (only works with --print)').hideHelp()).option('--settings <file-or-json>', 'Path to a settings JSON file or a JSON string to load additional settings from').option('--add-dir <directories...>', 'Additional directories to allow tool access to').option('--ide', 'Automatically connect to IDE on startup if exactly one valid IDE is available', () => true).option('--strict-mcp-config', 'Only use MCP servers from --mcp-config, ignoring all other MCP configurations', () => true).option('--session-id <uuid>', 'Use a specific session ID for the conversation (must be a valid UUID)').option('-n, --name <name>', 'Set a display name for this session (shown in /resume and terminal title)').option('--agents <json>', 'JSON object defining custom agents (e.g. \'{"reviewer": {"description": "Reviews code", "prompt": "You are a code reviewer"}}\')').option('--setting-sources <sources>', 'Comma-separated list of setting sources to load (user, project, local).')
@@ -2132,7 +2139,11 @@ async function run(): Promise<CommanderCommand> {
// An explicit CLI effort remains authoritative. Otherwise a selected
// main-thread agent may provide its own effort before falling back to the
// session setting.
const effectiveEffort = parseEffortValue(options.effort) ?? mainThreadAgentDefinition?.effort ?? getInitialEffortSetting();
// `--effort ultracode` is not a sixth level: it resolves to xhigh and
// raises a separate session flag, so every model-capability check keeps
// reasoning about the five real levels.
const ultracodeRequested = String(options.effort ?? '').toLowerCase() === ULTRACODE_EFFORT_ARG;
const effectiveEffort = ultracodeRequested ? ULTRACODE_EFFORT_LEVEL : parseEffortValue(options.effort) ?? mainThreadAgentDefinition?.effort ?? getInitialEffortSetting();
// Compute resolved model for hooks (use user-specified model at launch)
setInitialMainLoopModel(getUserSpecifiedModelSetting() || null);
@@ -2655,6 +2666,7 @@ async function run(): Promise<CommanderCommand> {
},
toolPermissionContext,
effortValue: effectiveEffort,
ultracode: ultracodeRequested,
...(isFastModeEnabled() && {
fastMode: getInitialFastModeSetting(effectiveModel ?? null)
}),
@@ -3083,6 +3095,7 @@ async function run(): Promise<CommanderCommand> {
})
} : null,
effortValue: effectiveEffort,
ultracode: ultracodeRequested,
activeOverlays: new Set<string>(),
fastMode: getInitialFastModeSetting(resolvedInitialModel),
...(isAdvisorEnabled() && advisorModel && {
+250
View File
@@ -0,0 +1,250 @@
import { afterEach, beforeEach, describe, expect, it } from 'bun:test'
import * as fs from 'node:fs/promises'
import * as os from 'node:os'
import * as path from 'node:path'
import { resetSettingsCache } from '../../utils/settings/settingsCache.js'
import { handleWorkflowsApi } from '../api/workflows.js'
let tmpHome: string
let projectDir: string
let originalConfigDir: string | undefined
function request(
urlStr: string,
init?: { method?: string; body?: unknown },
): { req: Request; url: URL; segments: string[] } {
const url = new URL(urlStr, 'http://localhost:3456')
const req = new Request(url.toString(), {
method: init?.method ?? 'GET',
...(init?.body !== undefined
? {
body: JSON.stringify(init.body),
headers: { 'content-type': 'application/json' },
}
: {}),
})
return { req, url, segments: url.pathname.split('/').filter(Boolean) }
}
function call(
urlStr: string,
init?: { method?: string; body?: unknown },
): Promise<Response> {
const { req, url, segments } = request(urlStr, init)
return handleWorkflowsApi(req, url, segments)
}
const VALID_SCRIPT = [
"export const meta = { name: 'my-audit', description: 'Audit the routes', phases: [{ title: 'Scan' }] }",
"const out = await agent('scan')",
'return out',
].join('\n')
describe('Workflows API', () => {
beforeEach(async () => {
tmpHome = await fs.mkdtemp(path.join(os.tmpdir(), 'wf-api-'))
projectDir = path.join(tmpHome, 'project')
await fs.mkdir(path.join(projectDir, '.git'), { recursive: true })
originalConfigDir = process.env.CLAUDE_CONFIG_DIR
process.env.CLAUDE_CONFIG_DIR = path.join(tmpHome, 'claude')
resetSettingsCache()
})
afterEach(async () => {
if (originalConfigDir === undefined) delete process.env.CLAUDE_CONFIG_DIR
else process.env.CLAUDE_CONFIG_DIR = originalConfigDir
resetSettingsCache()
await fs.rm(tmpHome, { recursive: true, force: true })
})
it('lists the bundled workflow', async () => {
const response = await call(`/api/workflows?cwd=${encodeURIComponent(projectDir)}`)
expect(response.status).toBe(200)
const body = (await response.json()) as {
workflows: Array<{ name: string; source: string; phases?: unknown[] }>
}
const deepResearch = body.workflows.find(w => w.name === 'deep-research')
expect(deepResearch?.source).toBe('built-in')
expect(deepResearch?.phases?.length).toBeGreaterThan(0)
// The list stays small: no script bodies.
expect(body.workflows.every(w => !('script' in w) || w.script === undefined)).toBe(true)
})
it('returns the script on the detail endpoint', async () => {
const response = await call(
`/api/workflows/deep-research?cwd=${encodeURIComponent(projectDir)}`,
)
const body = (await response.json()) as { script?: string }
expect(body.script).toContain('export const meta')
})
it('validates a script without running it', async () => {
const ok = await (
await call('/api/workflows/validate', {
method: 'POST',
body: { script: VALID_SCRIPT },
})
).json()
expect(ok).toMatchObject({ ok: true, name: 'my-audit' })
const nondeterministic = await (
await call('/api/workflows/validate', {
method: 'POST',
body: {
script:
"export const meta = { name: 'x', description: 'y' }\nconst t = Date.now()\nreturn t\n",
},
})
).json()
expect(nondeterministic).toMatchObject({ ok: true })
expect((nondeterministic as { warnings: string[] }).warnings[0]).toContain(
'Date.now()',
)
const bad = await (
await call('/api/workflows/validate', {
method: 'POST',
body: { script: 'const x = 1\n' },
})
).json()
expect(bad).toMatchObject({ ok: false })
expect((bad as { error: string }).error).toContain('FIRST statement')
})
it('saves a workflow and then lists and deletes it', async () => {
const saved = await (
await call('/api/workflows/save', {
method: 'POST',
body: { script: VALID_SCRIPT, scope: 'project', cwd: projectDir },
})
).json()
expect(saved).toMatchObject({ ok: true, name: 'my-audit' })
expect((saved as { filePath: string }).filePath).toBe(
path.join(projectDir, '.claude', 'workflows', 'my-audit.js'),
)
const listed = (await (
await call(`/api/workflows?cwd=${encodeURIComponent(projectDir)}`)
).json()) as { workflows: Array<{ name: string; source: string }> }
expect(
listed.workflows.find(w => w.name === 'my-audit')?.source,
).toBe('projectSettings')
const deleted = await call(
`/api/workflows/my-audit?scope=project&cwd=${encodeURIComponent(projectDir)}`,
{ method: 'DELETE' },
)
expect(deleted.status).toBe(200)
const after = (await (
await call(`/api/workflows?cwd=${encodeURIComponent(projectDir)}`)
).json()) as { workflows: Array<{ name: string }> }
expect(after.workflows.find(w => w.name === 'my-audit')).toBeUndefined()
})
it('rejects saving a script that does not compile', async () => {
const response = await call('/api/workflows/save', {
method: 'POST',
body: {
script:
"export const meta = { name: 'broken', description: 'x' }\nconst a: string = 1\n",
scope: 'user',
},
})
expect(response.status).toBe(400)
})
it('reconstructs a past run from its script and journal', async () => {
const sessionId = '11111111-2222-3333-4444-555555555555'
const sessionDir = path.join(
tmpHome,
'claude',
'projects',
'-tmp-project',
sessionId,
)
await fs.mkdir(path.join(sessionDir, 'workflows'), { recursive: true })
await fs.writeFile(
path.join(sessionDir, 'workflows', 'my-audit.wf_abc12345-def.js'),
VALID_SCRIPT,
'utf8',
)
const journalDir = path.join(
sessionDir,
'subagents',
'workflows',
'wf_abc12345-def',
)
await fs.mkdir(journalDir, { recursive: true })
await fs.writeFile(
path.join(journalDir, 'journal.jsonl'),
[
JSON.stringify({ type: 'started', key: '|scan|null', agentId: 'a1' }),
JSON.stringify({
type: 'result',
key: '|scan|null',
agentId: 'a1',
result: 'found nothing',
}),
'not json at all',
].join('\n'),
'utf8',
)
const runs = (await (await call('/api/workflows/runs')).json()) as {
runs: Array<{ runId: string; workflowName: string; completedAgents: number }>
}
expect(runs.runs).toHaveLength(1)
expect(runs.runs[0]).toMatchObject({
runId: 'wf_abc12345-def',
workflowName: 'my-audit',
completedAgents: 1,
})
const detail = (await (
await call(`/api/workflows/runs/${sessionId}/wf_abc12345-def`)
).json()) as {
script: string
description?: string
agents: Array<{ agentId: string; result: unknown; key: string }>
}
expect(detail.script).toBe(VALID_SCRIPT)
expect(detail.description).toBe('Audit the routes')
expect(detail.agents).toHaveLength(1)
expect(detail.agents[0]?.result).toBe('found nothing')
// The raw journal key chains every prior prompt — the API must not leak it.
expect(detail.agents[0]?.key).not.toContain('scan')
})
it('404s an unknown run', async () => {
const response = await call('/api/workflows/runs/nope/wf_000000-aaa')
expect(response.status).toBe(404)
})
it('refuses to write through a symlinked target', async () => {
const workflowsDir = path.join(tmpHome, 'claude', 'workflows')
await fs.mkdir(workflowsDir, { recursive: true })
const outside = path.join(tmpHome, 'outside.js')
await fs.writeFile(outside, '// pre-existing\n', 'utf8')
await fs.symlink(outside, path.join(workflowsDir, 'my-audit.js'))
const response = await call('/api/workflows/save', {
method: 'POST',
body: { script: VALID_SCRIPT, scope: 'user' },
})
expect(response.status).toBe(400)
expect(await fs.readFile(outside, 'utf8')).toBe('// pre-existing\n')
})
it('403s every route when workflows are disabled', async () => {
await fs.mkdir(path.join(tmpHome, 'claude'), { recursive: true })
await fs.writeFile(
path.join(tmpHome, 'claude', 'settings.json'),
JSON.stringify({ disableWorkflows: true }),
'utf8',
)
resetSettingsCache()
expect((await call('/api/workflows')).status).toBe(403)
expect((await call('/api/workflows/runs')).status).toBe(403)
})
})
+26 -1
View File
@@ -51,7 +51,7 @@ import {
import { registerChangedFileAccessRoot, registerFilesystemAccessRoot } from '../services/filesystemAccessRoots.js'
import { findGitRoot } from '../../utils/git.js'
import { traceCaptureService, trimTraceCallPreviews } from '../services/traceCaptureService.js'
import { getSubagentRunByTool } from '../services/subagentRunService.js'
import { getSubagentRunByAgentId, getSubagentRunByTool } from '../services/subagentRunService.js'
import { isValidPermissionMode } from '../services/settingsService.js'
import { handleWorkspaceSearchRoute } from './workspaceSearch.js'
import { localIndexCoordinator } from '../services/localIndex/coordinator.js'
@@ -227,6 +227,31 @@ export async function handleSessionsApi(
}
if (subResource === 'subagents') {
// Workflow agents have no parent `Agent` tool call to key off, so they
// are addressed by agent id instead. Same response shape, same page.
if (segments[4] === 'by-agent' && segments[5] && segments.length === 6) {
if (req.method !== 'GET') {
return Response.json(
{ error: 'METHOD_NOT_ALLOWED', message: `Method ${req.method} not allowed` },
{ status: 405 },
)
}
let agentId: string
try {
agentId = decodeURIComponent(segments[5])
} catch {
return Response.json(
{ error: 'NOT_FOUND', message: 'SubAgent route not found' },
{ status: 404 },
)
}
const byAgent = await getSubagentRunByAgentId(sessionId, agentId)
if (!byAgent) {
throw ApiError.notFound(`SubAgent run not found: ${agentId}`)
}
return Response.json(byAgent)
}
const isRunRoute = segments[4] === 'by-tool' && Boolean(segments[5])
const isRunRead = isRunRoute && segments.length === 6 && req.method === 'GET'
const isRunMessage = isRunRoute && segments.length === 7 &&
+127
View File
@@ -0,0 +1,127 @@
/**
* Dynamic workflows REST API
*
* GET /api/workflows — list runnable workflow definitions
* GET /api/workflows/runs — list past runs (?sessionId=&limit=)
* GET /api/workflows/runs/:sessionId/:runId — one run: script, phases, agent results
* GET /api/workflows/session-runs/:sessionId — finished runs rebuilt from disk
* POST /api/workflows/validate — parse + compile a script without running it
* POST /api/workflows/save — save a script as a /name command
* GET /api/workflows/:name — one definition, including its script
* DELETE /api/workflows/:name — delete a saved workflow (?scope=user|project)
*
* Starting a run is deliberately not here: a run belongs to a conversation
* turn, so the desktop sends the prompt (or `/name`) over the session
* WebSocket and watches `task_progress` events for live phase/agent state.
*/
import { ApiError, errorResponse } from '../middleware/errorHandler.js'
import {
workflowService,
type WorkflowSaveScope,
} from '../services/workflowService.js'
export async function handleWorkflowsApi(
req: Request,
url: URL,
segments: string[],
): Promise<Response> {
try {
const method = req.method
const first = segments[2] ? decodeURIComponent(segments[2]) : undefined
if (method === 'GET' && !first) {
const cwd = url.searchParams.get('cwd') ?? undefined
const workflows = await workflowService.listDefinitions(cwd)
return Response.json({ workflows })
}
if (method === 'GET' && first === 'runs' && !segments[3]) {
const rawLimit = url.searchParams.get('limit')
const limit = rawLimit ? Number.parseInt(rawLimit, 10) : undefined
const runs = await workflowService.listRuns({
sessionId: url.searchParams.get('sessionId') ?? undefined,
limit: Number.isSafeInteger(limit) && limit! > 0 ? limit : undefined,
})
return Response.json({ runs })
}
// Everything this session ever ran, rebuilt from disk. The desktop calls
// this when a session is opened so a finished run is still visible.
if (method === 'GET' && first === 'session-runs' && segments[3]) {
const runs = await workflowService.reconstructSessionRuns(
decodeURIComponent(segments[3]),
)
return Response.json({ runs })
}
if (method === 'GET' && first === 'runs' && segments[3] && segments[4]) {
const run = await workflowService.getRun(
decodeURIComponent(segments[3]),
decodeURIComponent(segments[4]),
)
return Response.json(run)
}
if (method === 'POST' && first === 'validate') {
const body = await readJson<{ script?: string }>(req)
if (typeof body.script !== 'string') {
throw ApiError.badRequest('`script` is required')
}
return Response.json(workflowService.validate(body.script))
}
if (method === 'POST' && first === 'save') {
const body = await readJson<{
script?: string
scope?: string
cwd?: string
}>(req)
if (typeof body.script !== 'string') {
throw ApiError.badRequest('`script` is required')
}
const saved = await workflowService.saveDefinition({
script: body.script,
scope: parseScope(body.scope),
cwd: body.cwd,
})
return Response.json({ ok: true, ...saved })
}
if (method === 'GET' && first) {
const cwd = url.searchParams.get('cwd') ?? undefined
return Response.json(await workflowService.getDefinition(first, cwd))
}
if (method === 'DELETE' && first) {
await workflowService.deleteDefinition(
first,
parseScope(url.searchParams.get('scope')),
url.searchParams.get('cwd') ?? undefined,
)
return Response.json({ ok: true })
}
throw new ApiError(
405,
`Method ${method} not allowed on /api/workflows${first ? `/${first}` : ''}`,
'METHOD_NOT_ALLOWED',
)
} catch (error) {
return errorResponse(error)
}
}
async function readJson<T>(req: Request): Promise<T> {
try {
return (await req.json()) as T
} catch {
throw ApiError.badRequest('Invalid JSON body')
}
}
function parseScope(value: string | null | undefined): WorkflowSaveScope {
if (value === 'project') return 'project'
if (value === 'user' || value == null || value === '') return 'user'
throw ApiError.badRequest(`Invalid scope: ${value}`)
}
+4
View File
@@ -29,6 +29,7 @@ import { handleOpenTargetsApi } from './api/open-targets.js'
import { handleMemoryApi } from './api/memory.js'
import { handleDesktopUiApi } from './api/desktop-ui.js'
import { handleTracesApi } from './api/traces.js'
import { handleWorkflowsApi } from './api/workflows.js'
export async function handleApiRequest(req: Request, url: URL): Promise<Response> {
const path = url.pathname
@@ -76,6 +77,9 @@ export async function handleApiRequest(req: Request, url: URL): Promise<Response
case 'teams':
return handleTeamsApi(req, url, segments)
case 'workflows':
return handleWorkflowsApi(req, url, segments)
case 'providers':
return handleProvidersApi(req, url, segments)
@@ -3,6 +3,7 @@ import * as fs from 'node:fs/promises'
import * as os from 'node:os'
import * as path from 'node:path'
import {
getSubagentRunByAgentId,
getSubagentRunByTool,
mergeTeammateTranscriptFragments,
resolveSubagentRunFromMessages,
@@ -737,3 +738,57 @@ describe('getSubagentRunByTool', () => {
await expect(getSubagentRunByTool(sessionId, 'tool-1')).resolves.toBeNull()
})
})
describe('getSubagentRunByAgentId', () => {
afterEach(async () => {
if (tmpDir) {
await fs.rm(tmpDir, { recursive: true, force: true })
tmpDir = null
}
delete process.env.CLAUDE_CONFIG_DIR
})
it('reads a run that has no parent Agent tool call', async () => {
await setupTmpConfigDir()
const sessionId = 'aaaaaaaa-bbbb-cccc-dddd-ffffffffffff'
const projectDir = '-tmp-workflow-agent'
const agentId = 'wfagent1'
// A workflow agent is spawned by the workflow runtime, so the parent
// session has no Agent tool_use to key off — only the transcript exists.
await writeSessionFile(projectDir, sessionId, [])
await writeSubagentTranscriptFile(projectDir, sessionId, agentId, [
{
type: 'user',
message: { role: 'user', content: 'Survey lib/response.js' },
uuid: 'wf-user',
timestamp: '2026-01-01T00:00:05.000Z',
},
{
type: 'assistant',
message: {
role: 'assistant',
content: [{ type: 'text', text: 'res.send(403) sends a JSON body' }],
usage: { input_tokens: 11, output_tokens: 7 },
},
uuid: 'wf-assistant',
timestamp: '2026-01-01T00:00:06.000Z',
},
])
const result = await getSubagentRunByAgentId(sessionId, agentId)
expect(result).toMatchObject({ sessionId, agentId, source: 'subagent-jsonl' })
expect(result?.messages.length).toBeGreaterThan(0)
// A workflow agent answers once into its script; there is no inbox.
expect(result?.canSendMessage).toBe(false)
})
it('returns null when no transcript exists for that agent', async () => {
await setupTmpConfigDir()
const sessionId = 'aaaaaaaa-bbbb-cccc-dddd-000000000000'
await writeSessionFile('-tmp-workflow-missing', sessionId, [])
expect(await getSubagentRunByAgentId(sessionId, 'nope')).toBeNull()
})
})
+49
View File
@@ -340,6 +340,55 @@ async function resolveTranscript(
return { agentId: null, messages: [] }
}
/**
* Read a subagent run straight from its transcript.
*
* {@link getSubagentRunByTool} starts from an `Agent` tool call in the parent
* conversation, which is how an agent the assistant dispatched is found. A
* workflow's agents are spawned by the workflow runtime instead, so no such
* tool call exists and that lookup can never resolve them — but they are
* ordinary subagents written by the same runner to the same place, so their
* `agent-<id>.jsonl` is all that is needed.
*/
export async function getSubagentRunByAgentId(
sessionId: string,
agentId: string,
): Promise<SubagentRunResponse | null> {
const normalized = normalizeAgentIdHint(agentId)
if (!normalized) return null
const transcript = await resolveTranscript(sessionId, [normalized])
if (!transcript.agentId) return null
const messages = transcript.messages
const truncated = truncateSubagentMessages(messages)
const usage = usageFromTranscriptMessages(messages)
const updatedAt = latestTimestamp(
...messages.map(message =>
isRecord(message) && typeof message.timestamp === 'string'
? message.timestamp
: undefined,
),
)
return {
sessionId,
// The caller addressed this run by agent id; echoing it keeps the response
// self-describing without inventing a tool call that never happened.
toolUseId: normalized,
agentId: transcript.agentId,
status: 'completed',
...(usage ? { usage } : {}),
messages: truncated.messages,
truncated: truncated.truncated,
...(updatedAt ? { updatedAt } : {}),
source: 'subagent-jsonl',
// A workflow agent answers once into its script and is gone; there is no
// inbox a follow-up could reach.
canSendMessage: false,
}
}
export async function getSubagentRunByTool(
sessionId: string,
toolUseId: string,
+500
View File
@@ -0,0 +1,500 @@
/**
* Dynamic workflow service for the local API.
*
* Two things live here: the *definitions* a user can run (bundled, personal,
* project) and the *runs* the CLI has already executed. Runs are reconstructed
* from the artifacts the runtime writes under the session directory — the
* script, the resume journal, and the task output file — so the desktop can
* show a run's history without the CLI process still being alive. Live
* progress arrives separately over the WebSocket as `task_progress` events.
*/
import { createHash } from 'crypto'
import * as fs from 'fs/promises'
import * as path from 'path'
import { getClaudeConfigHomeDir } from '../../utils/envUtils.js'
import { loadWorkflows } from '../../utils/workflows/discovery.js'
import { areWorkflowsEnabled } from '../../utils/workflows/enabled.js'
import {
parseWorkflowScript,
usesBannedNondeterminism,
} from '../../utils/workflows/meta.js'
import { compileWorkflowScript } from '../../utils/workflows/compile.js'
import type {
WorkflowDefinition,
WorkflowPhaseMeta,
WorkflowProgressEvent,
} from '../../utils/workflows/types.js'
import { ApiError } from '../middleware/errorHandler.js'
export type WorkflowDefinitionSummary = {
name: string
description: string
whenToUse?: string
source: WorkflowDefinition['source']
phases?: WorkflowPhaseMeta[]
filePath?: string
/** Present only on the detail endpoint — the list stays small. */
script?: string
}
export type WorkflowRunSummary = {
runId: string
sessionId: string
workflowName: string
scriptPath: string
startedAt: number
/** Number of agents with a recorded result in the journal. */
completedAgents: number
status: 'completed' | 'failed' | 'unknown'
}
export type WorkflowRunDetail = WorkflowRunSummary & {
script: string
description?: string
phases?: WorkflowPhaseMeta[]
agents: Array<{ key: string; agentId: string; result: unknown }>
progress?: WorkflowProgressEvent[]
logs?: string[]
result?: unknown
error?: string
totalTokens?: number
totalToolCalls?: number
}
/** A finished run rebuilt from disk, in the shape the desktop's panel needs. */
export type ReconstructedRun = {
runId: string
workflowName: string
startedAt: number
agents: Array<{
agentId: string
label: string
phaseIndex: number
phaseTitle?: string
agentIndex: number
}>
}
export type WorkflowSaveScope = 'user' | 'project'
const RUN_SCRIPT_PATTERN = /^(.+)\.(wf_[a-z0-9-]{6,})\.js$/
class WorkflowService {
private configDir(): string {
return path.resolve(getClaudeConfigHomeDir())
}
private projectsDir(): string {
return path.join(this.configDir(), 'projects')
}
async listDefinitions(cwd?: string): Promise<WorkflowDefinitionSummary[]> {
this.assertEnabled()
const workflows = await loadWorkflows(cwd)
return workflows.map(workflow => ({
name: workflow.name,
description: workflow.description,
whenToUse: workflow.whenToUse,
source: workflow.source,
phases: workflow.phases,
filePath: workflow.filePath,
}))
}
async getDefinition(
name: string,
cwd?: string,
): Promise<WorkflowDefinitionSummary> {
this.assertEnabled()
const workflows = await loadWorkflows(cwd)
const workflow = workflows.find(entry => entry.name === name)
if (!workflow) throw ApiError.notFound(`Unknown workflow: ${name}`)
return {
name: workflow.name,
description: workflow.description,
whenToUse: workflow.whenToUse,
source: workflow.source,
phases: workflow.phases,
filePath: workflow.filePath,
script: workflow.script,
}
}
/**
* Parse and compile a script without running it.
*
* The desktop calls this before sending a run so a malformed script is
* reported in the editor instead of coming back as a failed turn.
*/
validate(script: string): {
ok: boolean
error?: string
warnings?: string[]
name?: string
description?: string
phases?: WorkflowPhaseMeta[]
} {
this.assertEnabled()
const parsed = parseWorkflowScript(script)
if ('error' in parsed) return { ok: false, error: parsed.error }
const compiled = compileWorkflowScript(parsed.scriptBody)
if (!compiled.ok) return { ok: false, error: compiled.error }
// A script can compile and still be guaranteed to throw on its first
// Date.now()/Math.random(). Saying so here is the difference between a
// squiggle in the editor and a failed run.
const warnings = usesBannedNondeterminism(parsed.scriptBody)
? [
'Date.now(), new Date() and Math.random() throw at run time — they would make a resume replay diverge.',
]
: undefined
return {
ok: true,
...(warnings ? { warnings } : {}),
name: parsed.meta.name,
description: parsed.meta.description,
phases: parsed.meta.phases,
}
}
/**
* Save a script as a reusable `/name` command.
*
* Refuses to write through a symlink: the target directory is user-owned
* config, and following a link would place the file somewhere the caller
* did not choose.
*/
async saveDefinition(params: {
script: string
scope: WorkflowSaveScope
cwd?: string
}): Promise<{ name: string; filePath: string }> {
this.assertEnabled()
const parsed = parseWorkflowScript(params.script)
if ('error' in parsed) throw ApiError.badRequest(parsed.error)
const compiled = compileWorkflowScript(parsed.scriptBody)
if (!compiled.ok) throw ApiError.badRequest(compiled.error)
const dir =
params.scope === 'project'
? path.join(params.cwd ?? process.cwd(), '.claude', 'workflows')
: path.join(this.configDir(), 'workflows')
const filePath = path.join(dir, `${parsed.meta.name}.js`)
await this.assertNotSymlink(filePath)
await fs.mkdir(dir, { recursive: true })
await fs.writeFile(filePath, params.script, 'utf8')
return { name: parsed.meta.name, filePath }
}
async deleteDefinition(
name: string,
scope: WorkflowSaveScope,
cwd?: string,
): Promise<void> {
this.assertEnabled()
if (!/^[a-zA-Z0-9][a-zA-Z0-9_-]*$/.test(name)) {
throw ApiError.badRequest(`Invalid workflow name: ${name}`)
}
const dir =
scope === 'project'
? path.join(cwd ?? process.cwd(), '.claude', 'workflows')
: path.join(this.configDir(), 'workflows')
const filePath = path.join(dir, `${name}.js`)
await this.assertNotSymlink(filePath)
try {
await fs.unlink(filePath)
} catch (error) {
if ((error as NodeJS.ErrnoException).code === 'ENOENT') {
throw ApiError.notFound(`No saved workflow at ${filePath}`)
}
throw error
}
}
/**
* Rebuild a session's workflow runs from what the CLI left on disk.
*
* The live progress stream only exists while a run is happening, so
* reopening a finished session had nothing to show. Each agent's sidecar
* metadata records the run and phase it belonged to and is written before
* the agent starts, which makes it the durable record of a run's shape.
* Runs from before that field existed still list their agents, ungrouped.
*/
async reconstructSessionRuns(sessionId: string): Promise<ReconstructedRun[]> {
this.assertEnabled()
const runs = await this.listRuns({ sessionId })
if (runs.length === 0) return []
const byRunId = new Map<string, ReconstructedRun>()
for (const run of runs) {
byRunId.set(run.runId, {
runId: run.runId,
workflowName: run.workflowName,
startedAt: run.startedAt,
agents: [],
})
}
for (const { dir } of await this.findRunDirs(sessionId)) {
// `dir` is the session's `workflows/` script directory; the agent
// sidecars are its sibling `subagents/`.
const subagentsDir = path.join(path.dirname(dir), 'subagents')
await this.collectRunAgents(subagentsDir, byRunId)
// Runs from before agents recorded their phase wrote their sidecars
// into `subagents/workflows/<runId>/` instead. The directory name is
// the only provenance they have, so those agents are recovered without
// grouping rather than being dropped entirely.
const legacyRoot = path.join(subagentsDir, 'workflows')
let legacyRunIds: string[]
try {
legacyRunIds = await fs.readdir(legacyRoot)
} catch {
continue
}
for (const legacyRunId of legacyRunIds) {
if (!byRunId.has(legacyRunId)) continue
await this.collectRunAgents(
path.join(legacyRoot, legacyRunId),
byRunId,
legacyRunId,
)
}
}
for (const run of byRunId.values()) {
run.agents.sort((a, b) => a.agentIndex - b.agentIndex)
}
// Only runs we could actually rebuild agents for are worth returning; an
// empty shell would render as a phantom section.
return [...byRunId.values()]
.filter(run => run.agents.length > 0)
.sort((a, b) => b.startedAt - a.startedAt)
}
/**
* Add every workflow agent sidecar in `dir` to the run it belongs to.
*
* `fallbackRunId` covers legacy layouts where provenance came from the
* directory rather than the metadata; those agents land in phase 0 because
* their phase was never recorded anywhere.
*/
private async collectRunAgents(
dir: string,
byRunId: Map<string, ReconstructedRun>,
fallbackRunId?: string,
): Promise<void> {
let entries: string[]
try {
entries = await fs.readdir(dir)
} catch {
return
}
let fallbackIndex = 0
for (const entry of entries) {
if (!entry.endsWith('.meta.json')) continue
let meta: {
agentType?: string
description?: string
workflow?: {
runId: string
name: string
phaseIndex: number
phaseTitle?: string
agentIndex: number
}
}
try {
meta = JSON.parse(await fs.readFile(path.join(dir, entry), 'utf8'))
} catch {
continue
}
const runId = meta.workflow?.runId ?? fallbackRunId
if (!runId) continue
const run = byRunId.get(runId)
if (!run) continue
const agentId = entry.replace(/^agent-/, '').replace(/\.meta\.json$/, '')
if (run.agents.some(agent => agent.agentId === agentId)) continue
fallbackIndex += 1
run.agents.push({
agentId,
label:
meta.description ??
`agent ${meta.workflow?.agentIndex ?? fallbackIndex}`,
phaseIndex: meta.workflow?.phaseIndex ?? 0,
...(meta.workflow?.phaseTitle
? { phaseTitle: meta.workflow.phaseTitle }
: {}),
agentIndex: meta.workflow?.agentIndex ?? fallbackIndex,
})
}
}
/** Every run whose script the CLI persisted, newest first. */
async listRuns(options?: {
sessionId?: string
limit?: number
}): Promise<WorkflowRunSummary[]> {
this.assertEnabled()
const runs: WorkflowRunSummary[] = []
for (const { sessionId, dir } of await this.findRunDirs(options?.sessionId)) {
let entries: string[]
try {
entries = await fs.readdir(dir)
} catch {
continue
}
for (const entry of entries) {
const match = RUN_SCRIPT_PATTERN.exec(entry)
if (!match) continue
const [, workflowName, runId] = match
const scriptPath = path.join(dir, entry)
let startedAt = 0
try {
startedAt = (await fs.stat(scriptPath)).mtimeMs
} catch {
continue
}
const journal = await this.readJournal(sessionId, runId!)
runs.push({
runId: runId!,
sessionId,
workflowName: workflowName!,
scriptPath,
startedAt,
completedAgents: journal.length,
status: 'unknown',
})
}
}
runs.sort((a, b) => b.startedAt - a.startedAt)
return options?.limit ? runs.slice(0, options.limit) : runs
}
async getRun(sessionId: string, runId: string): Promise<WorkflowRunDetail> {
this.assertEnabled()
const runs = await this.listRuns({ sessionId })
const summary = runs.find(run => run.runId === runId)
if (!summary) throw ApiError.notFound(`Unknown workflow run: ${runId}`)
const script = await fs.readFile(summary.scriptPath, 'utf8')
const parsed = parseWorkflowScript(script)
const agents = await this.readJournal(sessionId, runId)
return {
...summary,
script,
description: 'error' in parsed ? undefined : parsed.meta.description,
phases: 'error' in parsed ? undefined : parsed.meta.phases,
agents,
}
}
/** `<projects>/<project>/<sessionId>/workflows` directories to scan. */
private async findRunDirs(
sessionId?: string,
): Promise<Array<{ sessionId: string; dir: string }>> {
const projectsDir = this.projectsDir()
let projects: string[]
try {
projects = await fs.readdir(projectsDir)
} catch {
return []
}
const found: Array<{ sessionId: string; dir: string }> = []
for (const project of projects) {
const projectPath = path.join(projectsDir, project)
let sessions: string[]
try {
sessions = await fs.readdir(projectPath)
} catch {
continue
}
for (const entry of sessions) {
if (sessionId && entry !== sessionId) continue
const dir = path.join(projectPath, entry, 'workflows')
try {
if (!(await fs.stat(dir)).isDirectory()) continue
} catch {
continue
}
found.push({ sessionId: entry, dir })
}
}
return found
}
private async readJournal(
sessionId: string,
runId: string,
): Promise<Array<{ key: string; agentId: string; result: unknown }>> {
const dirs = await this.findRunDirs(sessionId)
for (const { dir } of dirs) {
const journalPath = path.join(
path.dirname(dir),
'subagents',
'workflows',
runId,
'journal.jsonl',
)
let raw: string
try {
raw = await fs.readFile(journalPath, 'utf8')
} catch {
continue
}
const results: Array<{ key: string; agentId: string; result: unknown }> = []
for (const line of raw.split('\n')) {
if (line.trim() === '') continue
try {
const entry = JSON.parse(line) as {
type: string
key: string
agentId: string
result?: unknown
}
if (entry.type !== 'result') continue
results.push({
// The raw key chains every prior prompt; hash it so the API stays
// small and does not leak the whole prompt history.
key: createHash('sha256').update(entry.key).digest('hex').slice(0, 12),
agentId: entry.agentId,
result: entry.result,
})
} catch {
continue
}
}
return results
}
return []
}
private async assertNotSymlink(filePath: string): Promise<void> {
try {
const stats = await fs.lstat(filePath)
if (stats.isSymbolicLink()) {
throw ApiError.badRequest(
`Refusing to write through a symlink: ${filePath}`,
)
}
} catch (error) {
if ((error as NodeJS.ErrnoException).code === 'ENOENT') return
throw error
}
}
private assertEnabled(): void {
if (!areWorkflowsEnabled()) {
throw new ApiError(
403,
'Dynamic workflows are disabled (`disableWorkflows`).',
'WORKFLOWS_DISABLED',
)
}
}
}
export const workflowService = new WorkflowService()
+10
View File
@@ -425,6 +425,14 @@ export type AppState = DeepImmutable<{
advisorModel?: string
// Effort value
effortValue?: EffortValue
// Ultracode: xhigh effort plus standing dynamic-workflow orchestration.
// Session-scoped and never persisted — /effort ultracode sets it, a new
// session starts without it.
ultracode?: boolean
// Set by opt+w in the composer: the user typed `ultracode` but does not want
// this turn orchestrated. Cleared as soon as the keyword leaves the input,
// so it never leaks into the next prompt.
suppressWorkflowKeyword?: boolean
// Set synchronously in launchUltraplan before the detached flow starts.
// Prevents duplicate launches during the ~5s window before
// ultraplanSessionUrl is set by teleportToRemote. Cleared by launchDetached
@@ -563,6 +571,8 @@ export function getDefaultAppState(): AppState {
authVersion: 0,
initialMessage: null,
effortValue: undefined,
ultracode: false,
suppressWorkflowKeyword: false,
activeOverlays: new Set<string>(),
fastMode: false,
}
+2 -4
View File
@@ -3,12 +3,10 @@ import type { Task, TaskType } from './Task.js'
import { DreamTask } from './tasks/DreamTask/DreamTask.js'
import { LocalAgentTask } from './tasks/LocalAgentTask/LocalAgentTask.js'
import { LocalShellTask } from './tasks/LocalShellTask/LocalShellTask.js'
import { LocalWorkflowTask } from './tasks/LocalWorkflowTask/LocalWorkflowTask.js'
import { RemoteAgentTask } from './tasks/RemoteAgentTask/RemoteAgentTask.js'
/* eslint-disable @typescript-eslint/no-require-imports */
const LocalWorkflowTask: Task | null = feature('WORKFLOW_SCRIPTS')
? require('./tasks/LocalWorkflowTask/LocalWorkflowTask.js').LocalWorkflowTask
: null
const MonitorMcpTask: Task | null = feature('MONITOR_TOOL')
? require('./tasks/MonitorMcpTask/MonitorMcpTask.js').MonitorMcpTask
: null
@@ -25,8 +23,8 @@ export function getAllTasks(): Task[] {
LocalAgentTask,
RemoteAgentTask,
DreamTask,
LocalWorkflowTask,
]
if (LocalWorkflowTask) tasks.push(LocalWorkflowTask)
if (MonitorMcpTask) tasks.push(MonitorMcpTask)
return tasks
}
@@ -0,0 +1,190 @@
import { describe, expect, test } from 'bun:test'
import type { AppState } from '../../state/AppState.js'
import type { SetAppState } from '../../Task.js'
import { WORKFLOW_MAX_PROGRESS_ROWS } from '../../utils/workflows/constants.js'
import type { WorkflowProgressEvent } from '../../utils/workflows/types.js'
import {
buildResumePrompt,
buildWorkflowNotification,
registerWorkflowTask,
updateWorkflowProgressBatch,
type LocalWorkflowTaskState,
} from './LocalWorkflowTask.js'
/** A minimal AppState stand-in: these helpers only read and write `tasks`. */
function makeStore(task: LocalWorkflowTaskState): {
setAppState: SetAppState
get: () => LocalWorkflowTaskState
} {
let state = { tasks: { [task.id]: task } } as unknown as AppState
return {
setAppState: updater => {
state = updater(state)
},
get: () => state.tasks[task.id] as LocalWorkflowTaskState,
}
}
function makeTask(): LocalWorkflowTaskState {
return registerWorkflowTask({
taskId: 'w0000001',
script: "export const meta = { name: 'demo', description: 'Demo' }\n",
scriptPath: '/tmp/demo.wf_abc12345-def.js',
summary: 'Demo',
workflowName: 'demo',
workflowRunId: 'wf_abc12345-def',
})
}
function agent(
index: number,
state: 'start' | 'progress' | 'done' | 'error',
extra: Partial<Extract<WorkflowProgressEvent, { type: 'workflow_agent' }>> = {},
): WorkflowProgressEvent {
return { type: 'workflow_agent', index, label: `a${index}`, state, ...extra }
}
describe('updateWorkflowProgressBatch', () => {
test('replaces an agent row in place and recomputes the totals', () => {
const store = makeStore(makeTask())
updateWorkflowProgressBatch(
'w0000001',
[agent(1, 'progress', { tokens: 10, toolCalls: 1 })],
store.setAppState,
)
updateWorkflowProgressBatch(
'w0000001',
[
agent(1, 'done', { tokens: 40, toolCalls: 3 }),
agent(2, 'progress', { tokens: 5, toolCalls: 0 }),
],
store.setAppState,
)
const task = store.get()
expect(task.workflowProgress.filter(r => r.type === 'workflow_agent')).toHaveLength(2)
expect(task.agentCount).toBe(2)
expect(task.totalTokens).toBe(45)
expect(task.totalToolCalls).toBe(3)
expect(task.progressVersion).toBe(3)
})
test('phase rows are keyed separately from agent rows with the same index', () => {
const store = makeStore(makeTask())
updateWorkflowProgressBatch(
'w0000001',
[
{ type: 'workflow_phase', index: 1, title: 'Scan', kind: 'meta' },
agent(1, 'done'),
],
store.setAppState,
)
const rows = store.get().workflowProgress
expect(rows).toHaveLength(2)
expect(rows[0]?.type).toBe('workflow_phase')
expect(rows[1]?.type).toBe('workflow_agent')
})
test('logs are trimmed before agent rows when the buffer overflows', () => {
const store = makeStore(makeTask())
const logs: WorkflowProgressEvent[] = Array.from(
{ length: WORKFLOW_MAX_PROGRESS_ROWS * 2 + 10 },
(_unused, i) => ({ type: 'workflow_log', message: `line ${i}` }),
)
updateWorkflowProgressBatch(
'w0000001',
[agent(1, 'done', { tokens: 7 }), ...logs],
store.setAppState,
)
const rows = store.get().workflowProgress
expect(rows.length).toBeLessThanOrEqual(WORKFLOW_MAX_PROGRESS_ROWS + 1)
expect(rows.some(r => r.type === 'workflow_agent')).toBe(true)
// Oldest logs go first, so the newest line must survive.
const kept = rows.filter(r => r.type === 'workflow_log')
expect(kept.at(-1)).toEqual({
type: 'workflow_log',
message: `line ${logs.length - 1}`,
})
expect(store.get().totalTokens).toBe(7)
})
test('a settled task ignores late progress', () => {
const task = makeTask()
const store = makeStore({ ...task, status: 'completed' })
updateWorkflowProgressBatch('w0000001', [agent(1, 'done')], store.setAppState)
expect(store.get().workflowProgress).toHaveLength(0)
})
test('an empty batch does not bump the version', () => {
const store = makeStore(makeTask())
updateWorkflowProgressBatch('w0000001', [], store.setAppState)
expect(store.get().progressVersion).toBe(0)
})
})
describe('buildWorkflowNotification', () => {
const base = {
taskId: 'w0000001',
summary: 'Demo',
agentCount: 3,
totalTokens: 1200,
totalToolCalls: 9,
durationMs: 4500,
transcriptDir: '/tmp/run',
scriptPath: '/tmp/demo.wf_abc12345-def.js',
workflowRunId: 'wf_abc12345-def',
}
test('a completed run points at the journal and the replay call', () => {
const message = buildWorkflowNotification({
...base,
status: 'completed',
result: ['ALPHA'],
})
expect(message).toContain('<status>completed</status>')
expect(message).toContain('/tmp/run/journal.jsonl')
expect(message).toContain(
"Workflow({scriptPath: '/tmp/demo.wf_abc12345-def.js', resumeFromRunId: 'wf_abc12345-def'})",
)
expect(message).toContain('<result>["ALPHA"]</result>')
expect(message).toContain(
'<usage><agent_count>3</agent_count><subagent_tokens>1200</subagent_tokens><tool_uses>9</tool_uses><duration_ms>4500</duration_ms></usage>',
)
})
test('a failed run offers recovery instead of the journal note', () => {
const message = buildWorkflowNotification({
...base,
status: 'failed',
error: 'agent exploded',
failures: ['parallel[1] failed: boom'],
})
expect(message).toContain('<recovery>')
expect(message).not.toContain('journal.jsonl')
expect(message).toContain('<failures>\nparallel[1] failed: boom\n</failures>')
expect(message).toContain('failed: agent exploded')
})
test('args are threaded into the resume call so a replay reruns the same input', () => {
const message = buildWorkflowNotification({
...base,
status: 'failed',
error: 'nope',
args: ['a.ts', 'b.ts'],
})
expect(message).toContain('args: ["a.ts","b.ts"]')
})
})
describe('buildResumePrompt', () => {
test('names the script path and run id', () => {
const prompt = buildResumePrompt({
...makeTask(),
args: { q: 1 },
})
expect(prompt).toContain("scriptPath: '/tmp/demo.wf_abc12345-def.js'")
expect(prompt).toContain("resumeFromRunId: 'wf_abc12345-def'")
expect(prompt).toContain('args: {"q":1}')
})
})
+505 -32
View File
@@ -1,34 +1,507 @@
// @generated stub from scan-missing-imports
// 该文件自动生成,对应 ant-internal 的 feature() gated 模块。
// 所有外部 build 的代码路径在 DCE 后都不会真的执行这里的代码,这只是
// bun build resolver 的占位符。
const __target = function noop() {}
const __handler: ProxyHandler<any> = {
get(_t, prop) {
if (prop === '__esModule') return true
if (prop === 'default') return new Proxy(__target, __handler)
if (prop === Symbol.toPrimitive) return () => undefined
if (prop === Symbol.iterator) return function* () {}
if (prop === Symbol.asyncIterator) return async function* () {}
if (prop === 'then') return undefined
return new Proxy(__target, __handler)
},
apply() {
return new Proxy(__target, __handler)
},
construct() {
return new Proxy(__target, __handler)
import { writeFile } from 'fs/promises'
import {
OUTPUT_FILE_TAG,
STATUS_TAG,
SUMMARY_TAG,
TASK_ID_TAG,
TASK_NOTIFICATION_TAG,
TASK_TYPE_TAG,
TOOL_USE_ID_TAG,
} from '../../constants/xml.js'
import type { SetAppState, Task, TaskStateBase } from '../../Task.js'
import { createTaskStateBase } from '../../Task.js'
import { createAbortController } from '../../utils/abortController.js'
import { logForDebugging } from '../../utils/debug.js'
import { enqueuePendingNotification } from '../../utils/messageQueueManager.js'
import {
evictTaskOutput,
getTaskOutputPath,
} from '../../utils/task/diskOutput.js'
import { PANEL_GRACE_MS, updateTaskState } from '../../utils/task/framework.js'
import { WORKFLOW_MAX_PROGRESS_ROWS } from '../../utils/workflows/constants.js'
import {
isDurableWorkflowEvent,
type WorkflowPhaseMeta,
type WorkflowProgressEvent,
} from '../../utils/workflows/types.js'
export type LocalWorkflowTaskState = TaskStateBase & {
type: 'local_workflow'
/** Full script text as executed — the same bytes written to `scriptPath`. */
script: string
scriptPath?: string
/** Kept under `prompt` too so generic task consumers can show something. */
prompt: string
args?: unknown
summary?: string
workflowName?: string
title?: string
phases?: WorkflowPhaseMeta[]
defaultModel?: string
workflowRunId: string
ownerAgentId?: string
workflowProgress: WorkflowProgressEvent[]
/** Bumped on every applied batch so views can diff cheaply. */
progressVersion: number
agentCount: number
totalTokens: number
totalToolCalls: number
logs: string[]
result?: unknown
error?: string
abortController?: AbortController
/** Per-agent controllers, so a single agent can be skipped or restarted. */
agentControllers?: Map<string, AbortController>
evictAfter?: number
}
export function isLocalWorkflowTask(
task: unknown,
): task is LocalWorkflowTaskState {
return (
typeof task === 'object' &&
task !== null &&
'type' in task &&
task.type === 'local_workflow'
)
}
export function registerWorkflowTask(params: {
taskId: string
script: string
scriptPath?: string
args?: unknown
summary?: string
workflowName?: string
title?: string
phases?: WorkflowPhaseMeta[]
defaultModel?: string
workflowRunId: string
ownerAgentId?: string
toolUseId?: string
startTime?: number
}): LocalWorkflowTaskState {
const base = createTaskStateBase(
params.taskId,
'local_workflow',
params.summary ?? 'Dynamic workflow',
params.toolUseId,
)
return {
...base,
...(params.startTime !== undefined ? { startTime: params.startTime } : {}),
type: 'local_workflow',
status: 'running',
script: params.script,
scriptPath: params.scriptPath,
args: params.args,
prompt: params.script,
summary: params.summary,
workflowName: params.workflowName,
title: params.title,
phases: params.phases,
defaultModel: params.defaultModel,
workflowRunId: params.workflowRunId,
ownerAgentId: params.ownerAgentId,
workflowProgress: [],
progressVersion: 0,
agentCount: 0,
totalTokens: 0,
totalToolCalls: 0,
logs: [],
abortController: createAbortController(),
agentControllers: new Map(),
}
}
/**
* Fold a batch of progress events into the task's state.
*
* Agent and phase rows are keyed by `type:index` and overwritten in place, so
* a run that emits hundreds of updates for the same twenty agents keeps twenty
* rows. Only logs accumulate, and they are the first thing trimmed.
*/
export function updateWorkflowProgressBatch(
taskId: string,
events: WorkflowProgressEvent[],
setAppState: SetAppState,
): void {
if (events.length === 0) return
updateTaskState<LocalWorkflowTaskState>(taskId, setAppState, task => {
if (task.status !== 'running') return task
const rows = [...task.workflowProgress]
const rowIndexByKey = new Map<string, number>()
for (let i = 0; i < rows.length; i++) {
const row = rows[i]!
if (row.type === 'workflow_agent' || row.type === 'workflow_phase') {
rowIndexByKey.set(`${row.type}:${row.index}`, i)
}
}
let agentCount = task.agentCount
let appendedLogs = false
for (const event of events) {
if (event.type === 'workflow_log') {
rows.push(event)
appendedLogs = true
continue
}
const key = `${event.type}:${event.index}`
const existing = rowIndexByKey.get(key)
if (existing !== undefined) rows[existing] = event
else {
rowIndexByKey.set(key, rows.length)
rows.push(event)
}
if (event.type === 'workflow_agent') {
agentCount = Math.max(agentCount, event.index)
}
}
const trimmed =
appendedLogs && rows.length > WORKFLOW_MAX_PROGRESS_ROWS * 2
? dropOldestLogs(rows, rows.length - WORKFLOW_MAX_PROGRESS_ROWS)
: rows
let totalTokens = 0
let totalToolCalls = 0
for (const row of trimmed) {
if (row.type !== 'workflow_agent') continue
totalTokens += row.tokens ?? 0
totalToolCalls += row.toolCalls ?? 0
}
return {
...task,
workflowProgress: trimmed,
progressVersion: task.progressVersion + events.length,
agentCount,
totalTokens,
totalToolCalls,
}
})
}
function dropOldestLogs(
rows: WorkflowProgressEvent[],
toDrop: number,
): WorkflowProgressEvent[] {
let remaining = toDrop
const kept: WorkflowProgressEvent[] = []
for (const row of rows) {
if (remaining > 0 && row.type === 'workflow_log') {
remaining--
continue
}
kept.push(row)
}
return kept
}
function settleWorkflowTask(
taskId: string,
setAppState: SetAppState,
status: 'completed' | 'failed' | 'killed' | 'paused',
patch: Partial<LocalWorkflowTaskState>,
): LocalWorkflowTaskState | null {
let settled: LocalWorkflowTaskState | null = null
updateTaskState<LocalWorkflowTaskState>(taskId, setAppState, task => {
if (task.status !== 'running') return task
settled = task
task.abortController?.abort()
const endTime = Date.now()
return {
...task,
...patch,
status,
endTime,
...(status !== 'paused' ? { evictAfter: endTime + PANEL_GRACE_MS } : {}),
abortController: undefined,
agentControllers: undefined,
}
})
return settled
}
export function completeWorkflowTask(
taskId: string,
result: unknown,
agentCount: number,
logs: string[],
setAppState: SetAppState,
): void {
const settled = settleWorkflowTask(taskId, setAppState, 'completed', {
result,
agentCount,
logs,
})
if (!settled) return
void writeWorkflowOutput(settled, { result, agentCount, logs })
}
export function failWorkflowTask(
taskId: string,
error: string,
agentCount: number,
logs: string[],
setAppState: SetAppState,
): void {
const settled = settleWorkflowTask(taskId, setAppState, 'failed', {
error,
agentCount,
logs,
})
if (!settled) return
void evictTaskOutput(taskId)
void writeWorkflowOutput(settled, { error, agentCount, logs })
}
/** Stop the run but keep it resumable: the journal on disk is still valid. */
export function pauseWorkflowTask(
taskId: string,
setAppState: SetAppState,
): boolean {
return (
settleWorkflowTask(taskId, setAppState, 'paused', { notified: true }) !==
null
)
}
export function killWorkflowTask(
taskId: string,
setAppState: SetAppState,
): boolean {
const settled = settleWorkflowTask(taskId, setAppState, 'killed', {
notified: true,
})
if (settled) void evictTaskOutput(taskId)
return settled !== null
}
function abortWorkflowAgent(
taskId: string,
agentKey: string,
reason: 'user-skip' | 'user-retry',
setAppState: SetAppState,
): boolean {
let aborted = false
updateTaskState<LocalWorkflowTaskState>(taskId, setAppState, task => {
if (task.status !== 'running') return task
const controller = task.agentControllers?.get(agentKey)
if (controller && !controller.signal.aborted) {
controller.abort(new DOMException(reason, 'AbortError'))
aborted = true
}
return task
})
return aborted
}
export function skipWorkflowAgent(
taskId: string,
agentKey: string,
setAppState: SetAppState,
): boolean {
return abortWorkflowAgent(taskId, agentKey, 'user-skip', setAppState)
}
export function retryWorkflowAgent(
taskId: string,
agentKey: string,
setAppState: SetAppState,
): boolean {
return abortWorkflowAgent(taskId, agentKey, 'user-retry', setAppState)
}
/** The `Workflow(...)` call that picks a stopped run back up. */
export function buildResumePrompt(task: LocalWorkflowTaskState): string {
const argsPart =
task.args !== undefined ? `, args: ${JSON.stringify(task.args)}` : ''
return (
`Resume the paused workflow by calling: Workflow({scriptPath: '${task.scriptPath}', ` +
`resumeFromRunId: '${task.workflowRunId}'${argsPart}}) — completed agents return cached results.`
)
}
export function enqueueWorkflowNotification(params: {
taskId: string
summary: string
status: 'completed' | 'failed' | 'killed'
result?: unknown
error?: string
failures?: string[]
agentCount: number
totalTokens: number
totalToolCalls: number
durationMs: number
toolUseId?: string
transcriptDir?: string
scriptPath?: string
workflowRunId?: string
args?: unknown
setAppState: SetAppState
}): void {
let shouldEnqueue = false
updateTaskState<LocalWorkflowTaskState>(
params.taskId,
params.setAppState,
task => {
if (task.notified) return task
shouldEnqueue = true
return { ...task, notified: true }
},
)
if (!shouldEnqueue) return
const message = buildWorkflowNotification(params)
enqueuePendingNotification({ value: message, mode: 'task-notification' })
}
/**
* The `<task-notification>` the model reads when a run settles.
*
* The recovery and journal hints matter: without the exact `Workflow({...})`
* call and the journal path, a model looking at an empty result has no way to
* tell "the agents returned nothing" from "post-processing dropped it".
*/
export function buildWorkflowNotification(params: {
taskId: string
summary: string
status: 'completed' | 'failed' | 'killed'
result?: unknown
error?: string
failures?: string[]
agentCount: number
totalTokens: number
totalToolCalls: number
durationMs: number
toolUseId?: string
transcriptDir?: string
scriptPath?: string
workflowRunId?: string
args?: unknown
}): string {
const {
taskId,
summary,
status,
result,
error,
failures,
agentCount,
totalTokens,
totalToolCalls,
durationMs,
toolUseId,
transcriptDir,
scriptPath,
workflowRunId,
args,
} = params
const headline =
status === 'completed'
? `Dynamic workflow "${summary}" completed`
: status === 'failed'
? `Dynamic workflow "${summary}" failed: ${error || 'Unknown error'}`
: `Dynamic workflow "${summary}" was stopped`
const argsPart = args !== undefined ? `, args: ${JSON.stringify(args)}` : ''
const resumeCall =
scriptPath && workflowRunId
? `Workflow({scriptPath: '${scriptPath}', resumeFromRunId: '${workflowRunId}'${argsPart}})`
: undefined
const sections: string[] = []
if (status !== 'completed') {
const recovery: string[] = []
if (resumeCall) {
recovery.push(`To resume after editing the script, call: ${resumeCall}`)
}
if (transcriptDir) recovery.push(`Agent transcripts: ${transcriptDir}`)
if (recovery.length > 0) {
sections.push(`\n<recovery>\n${recovery.join('\n')}\n</recovery>`)
}
} else if (transcriptDir) {
const notes = [
`Per-agent results: ${transcriptDir}/journal.jsonl — one {"type":"result",...} line per completed agent with its full return value.`,
'If the result above is empty or unexpected, Read this file BEFORE diagnosing — do not assume agents returned non-empty results.',
]
if (resumeCall) {
notes.push(
`To re-run with edited post-processing: ${resumeCall} — agents whose (prompt, opts) are unchanged replay from cache.`,
)
}
sections.push(`\n<transcripts>\n${notes.join('\n')}\n</transcripts>`)
}
if (failures && failures.length > 0) {
sections.push(`\n<failures>\n${failures.join('\n')}\n</failures>`)
}
const resultSection =
result === undefined
? ''
: `\n<result>${safeJson(result)}</result>`
const toolUseIdLine = toolUseId
? `\n<${TOOL_USE_ID_TAG}>${toolUseId}</${TOOL_USE_ID_TAG}>`
: ''
return `<${TASK_NOTIFICATION_TAG}>
<${TASK_ID_TAG}>${taskId}</${TASK_ID_TAG}>${toolUseIdLine}
<${TASK_TYPE_TAG}>local_workflow</${TASK_TYPE_TAG}>
<${OUTPUT_FILE_TAG}>${getTaskOutputPath(taskId)}</${OUTPUT_FILE_TAG}>
<${STATUS_TAG}>${status}</${STATUS_TAG}>
<${SUMMARY_TAG}>${headline}</${SUMMARY_TAG}>${resultSection}${sections.join('')}
<usage><agent_count>${agentCount}</agent_count><subagent_tokens>${totalTokens}</subagent_tokens><tool_uses>${totalToolCalls}</tool_uses><duration_ms>${durationMs}</duration_ms></usage>
</${TASK_NOTIFICATION_TAG}>`
}
async function writeWorkflowOutput(
task: LocalWorkflowTaskState,
extra: { result?: unknown; error?: string; agentCount: number; logs: string[] },
): Promise<void> {
try {
await writeFile(
task.outputFile,
JSON.stringify(
{
summary: task.summary,
workflowName: task.workflowName,
workflowRunId: task.workflowRunId,
agentCount: extra.agentCount,
logs: extra.logs,
result: extra.result,
error: extra.error,
workflowProgress: task.workflowProgress.filter(isDurableWorkflowEvent),
totalTokens: task.totalTokens,
totalToolCalls: task.totalToolCalls,
},
null,
2,
),
'utf8',
)
} catch (error) {
logForDebugging(
`Failed to write workflow output for ${task.id}: ${error instanceof Error ? error.message : String(error)}`,
)
}
}
function safeJson(value: unknown): string {
if (typeof value === 'string') return value
try {
return JSON.stringify(value) ?? 'null'
} catch {
return '[unserializable result]'
}
}
export const LocalWorkflowTask: Task = {
name: 'LocalWorkflowTask',
type: 'local_workflow',
async kill(taskId, setAppState) {
killWorkflowTask(taskId, setAppState)
},
}
const stub: any = new Proxy(__target, __handler)
export default stub
export const __stubMissing = true
// 兼容常见的命名导出 —— 没列在这里的也会通过 default Proxy 兜底
export const createCachedMCState = stub
export const isCachedMicrocompactEnabled = stub
export const isModelSupportedForCacheEditing = stub
export const getCachedMCConfig = stub
export const markToolsSentToAPI = stub
export const resetCachedMCState = stub
export const checkProtectedNamespace = stub
export const getCoordinatorUserContext = stub
+8 -7
View File
@@ -128,13 +128,14 @@ const SnipTool = feature('HISTORY_SNIP')
const ListPeersTool = feature('UDS_INBOX')
? require('./tools/ListPeersTool/ListPeersTool.js').ListPeersTool
: null
const WorkflowTool = feature('WORKFLOW_SCRIPTS')
? (() => {
require('./tools/WorkflowTool/bundled/index.js').initBundledWorkflows()
return require('./tools/WorkflowTool/WorkflowTool.js').WorkflowTool
})()
: null
/* eslint-enable custom-rules/no-process-env-top-level, @typescript-eslint/no-require-imports */
// Lazy require: WorkflowTool -> launchWorkflow -> tools.js (assembleToolPool)
// is a cycle, so it cannot be a top-level import.
/* eslint-disable @typescript-eslint/no-require-imports */
const getWorkflowTool = () =>
require('./tools/WorkflowTool/WorkflowTool.js')
.WorkflowTool as typeof import('./tools/WorkflowTool/WorkflowTool.js').WorkflowTool
/* eslint-enable @typescript-eslint/no-require-imports */
import type { ToolPermissionContext } from './Tool.js'
import { getDenyRuleForTool } from './utils/permissions/permissions.js'
import { hasEmbeddedSearchTools } from './utils/embeddedTools.js'
@@ -236,7 +237,7 @@ export function getAllBaseTools(): Tools {
: []),
...(VerifyPlanExecutionTool ? [VerifyPlanExecutionTool] : []),
...(process.env.USER_TYPE === 'ant' && REPLTool ? [REPLTool] : []),
...(WorkflowTool ? [WorkflowTool] : []),
getWorkflowTool(),
...(SleepTool ? [SleepTool] : []),
...cronTools,
...(RemoteTriggerTool ? [RemoteTriggerTool] : []),
+5 -11
View File
@@ -59,10 +59,9 @@ import { createUserMessage } from '../../utils/messages.js'
import { getAgentModel } from '../../utils/model/agent.js'
import type { ModelAlias } from '../../utils/model/aliases.js'
import {
clearAgentTranscriptSubdir,
recordSidechainTranscript,
setAgentTranscriptSubdir,
writeAgentMetadata,
type AgentMetadata,
} from '../../utils/sessionStorage.js'
import {
isRestrictedToPluginOnly,
@@ -311,7 +310,7 @@ export async function* runAgent({
spawningToolUseId,
persistedAgentType,
alreadyPersistedMessageCount,
transcriptSubdir,
workflow,
onQueryProgress,
}: {
agentDefinition: AgentDefinition
@@ -379,7 +378,8 @@ export async function* runAgent({
alreadyPersistedMessageCount?: number
/** Optional subdirectory under subagents/ to group this agent's transcript
* with related ones (e.g. workflows/<runId> for workflow subagents). */
transcriptSubdir?: string
/** Set when this agent is one step of a dynamic workflow run. */
workflow?: AgentMetadata['workflow']
/** Optional callback fired on every message yielded by query() — including
* stream_event deltas that runAgent otherwise drops. Use to detect liveness
* during long single-block streams (e.g. thinking) where no assistant
@@ -405,12 +405,6 @@ export async function* runAgent({
const agentId = override?.agentId ? override.agentId : createAgentId()
// Route this agent's transcript into a grouping subdirectory if requested
// (e.g. workflow subagents write to subagents/workflows/<runId>/).
if (transcriptSubdir) {
setAgentTranscriptSubdir(agentId, transcriptSubdir)
}
// Register agent in Perfetto trace for hierarchy visualization
if (isPerfettoTracingEnabled()) {
const parentId = toolUseContext.agentId ?? getSessionId()
@@ -826,6 +820,7 @@ export async function* runAgent({
...(worktreePath && { worktreePath }),
...(description && { description }),
...(spawningToolUseId && { toolUseId: spawningToolUseId }),
...(workflow && { workflow }),
}).catch(_err => logForDebugging(`Failed to write agent metadata: ${_err}`))
// Track the last recorded message UUID for parent chain continuity
@@ -918,7 +913,6 @@ export async function* runAgent({
// Release perfetto agent registry entry
unregisterPerfettoAgent(agentId)
// Release transcript subdir mapping
clearAgentTranscriptSubdir(agentId)
// Release this agent's todos entry. Without this, every subagent that
// called TodoWrite leaves a key in AppState.todos forever (even after all
// items complete, the value is [] but the key stays). Whale sessions
@@ -1,34 +0,0 @@
// @generated stub from scan-missing-imports
// 该文件自动生成,对应 ant-internal 的 feature() gated 模块。
// 所有外部 build 的代码路径在 DCE 后都不会真的执行这里的代码,这只是
// bun build resolver 的占位符。
const __target = function noop() {}
const __handler: ProxyHandler<any> = {
get(_t, prop) {
if (prop === '__esModule') return true
if (prop === 'default') return new Proxy(__target, __handler)
if (prop === Symbol.toPrimitive) return () => undefined
if (prop === Symbol.iterator) return function* () {}
if (prop === Symbol.asyncIterator) return async function* () {}
if (prop === 'then') return undefined
return new Proxy(__target, __handler)
},
apply() {
return new Proxy(__target, __handler)
},
construct() {
return new Proxy(__target, __handler)
},
}
const stub: any = new Proxy(__target, __handler)
export default stub
export const __stubMissing = true
// 兼容常见的命名导出 —— 没列在这里的也会通过 default Proxy 兜底
export const createCachedMCState = stub
export const isCachedMicrocompactEnabled = stub
export const isModelSupportedForCacheEditing = stub
export const getCachedMCConfig = stub
export const markToolsSentToAPI = stub
export const resetCachedMCState = stub
export const checkProtectedNamespace = stub
export const getCoordinatorUserContext = stub
@@ -0,0 +1,214 @@
import React, { useCallback, useMemo, useState } from 'react'
import { getOriginalCwd } from '../../bootstrap/state.js'
import { PermissionDialog } from '../../components/permissions/PermissionDialog.js'
import {
PermissionPrompt,
type PermissionPromptOption,
type ToolAnalyticsContext,
} from '../../components/permissions/PermissionPrompt.js'
import type { PermissionRequestProps } from '../../components/permissions/PermissionRequest.js'
import { PermissionRuleExplanation } from '../../components/permissions/PermissionRuleExplanation.js'
import {
type UnaryEvent,
usePermissionRequestLogging,
} from '../../components/permissions/hooks.js'
import { Box, Text } from '../../ink.js'
import { sanitizeToolNameForAnalytics } from '../../services/analytics/metadata.js'
import { shouldShowAlwaysAllowOptions } from '../../utils/permissions/permissionsLoader.js'
import { recordWorkflowAutoModeConsent } from '../../utils/workflows/autoModeConsent.js'
import { parseWorkflowScript } from '../../utils/workflows/meta.js'
import type { WorkflowMeta } from '../../utils/workflows/types.js'
import { WORKFLOW_TOOL_NAME } from './constants.js'
type WorkflowOptionValue = 'yes' | 'yes-always' | 'view' | 'no'
/**
* Approval dialog shown before a dynamic workflow starts.
*
* The point of the dialog is the phase list: it is the only preview the user
* gets of how many agents are about to run and what they will do, and it is
* the last decision point — once the run starts, its subagents' file edits are
* auto-approved.
*/
export function WorkflowPermissionRequest(
props: PermissionRequestProps,
): React.ReactNode {
const { toolUseConfirm, onDone, onReject, workerBadge } = props
const unaryEvent = useMemo<UnaryEvent>(
() => ({ completion_type: 'tool_use_single', language_name: 'none' }),
[],
)
usePermissionRequestLogging(toolUseConfirm, unaryEvent)
const input = toolUseConfirm.input as {
script?: string
name?: string
scriptPath?: string
args?: unknown
}
const script = typeof input.script === 'string' ? input.script : undefined
const meta = useMemo(() => readMeta(input.script), [input.script])
const workflowName = meta?.name ?? input.name ?? 'workflow'
const description = meta?.description
const phases = meta?.phases ?? []
const originalCwd = getOriginalCwd()
const showAlwaysAllow = shouldShowAlwaysAllowOptions() && Boolean(input.name)
const [showScript, setShowScript] = useState(false)
const options = useMemo<PermissionPromptOption<WorkflowOptionValue>[]>(() => {
const built: PermissionPromptOption<WorkflowOptionValue>[] = [
{ label: 'Yes, run it', value: 'yes', feedbackConfig: { type: 'accept' } },
]
if (showAlwaysAllow) {
built.push({
label: (
<Text>
Yes, and don&apos;t ask again for <Text bold>{workflowName}</Text> in{' '}
<Text bold>{originalCwd}</Text>
</Text>
),
value: 'yes-always',
})
}
if (script && !showScript) {
built.push({ label: 'View raw script', value: 'view' })
}
built.push({ label: 'No', value: 'no', feedbackConfig: { type: 'reject' } })
return built
}, [showAlwaysAllow, workflowName, originalCwd, script, showScript])
const toolAnalyticsContext = useMemo<ToolAnalyticsContext>(
() => ({
toolName: sanitizeToolNameForAnalytics(toolUseConfirm.tool.name),
isMcp: toolUseConfirm.tool.isMcp ?? false,
}),
[toolUseConfirm.tool.name, toolUseConfirm.tool.isMcp],
)
const isAutoMode =
toolUseConfirm.toolUseContext.getAppState().toolPermissionContext.mode ===
'auto'
const handleSelect = useCallback(
(value: WorkflowOptionValue, feedback?: string) => {
// In auto mode a Yes of either kind is the one-time consent — after this
// the launch prompt stops appearing.
if (isAutoMode && (value === 'yes' || value === 'yes-always')) {
recordWorkflowAutoModeConsent()
}
switch (value) {
case 'yes':
toolUseConfirm.onAllow(toolUseConfirm.input, [], feedback)
onDone()
break
case 'yes-always':
toolUseConfirm.onAllow(toolUseConfirm.input, [
{
type: 'addRules',
rules: [
{ toolName: WORKFLOW_TOOL_NAME, ruleContent: workflowName },
],
behavior: 'allow',
destination: 'localSettings',
},
])
onDone()
break
case 'view':
// Stay in the dialog: the whole point is to read the script and then
// decide, so this must not resolve the permission either way.
setShowScript(true)
break
case 'no':
toolUseConfirm.onReject(feedback)
onReject()
onDone()
break
}
},
[toolUseConfirm, onDone, onReject, workflowName, isAutoMode],
)
const handleCancel = useCallback(() => {
toolUseConfirm.onReject()
onReject()
onDone()
}, [toolUseConfirm, onDone, onReject])
return (
<PermissionDialog
title={`Run workflow "${workflowName}"?`}
workerBadge={workerBadge}
>
<Text>
A workflow spawns many subagents in the background. Their file edits are
auto-approved and the run can use a large number of tokens.
</Text>
<Box flexDirection="column" paddingX={2} paddingY={1}>
{description ? <Text dimColor>{description}</Text> : null}
{phases.length > 0 ? (
<Box flexDirection="column" marginTop={description ? 1 : 0}>
<Text dimColor>Phases:</Text>
{phases.map((phase, index) => (
<Text key={`${phase.title}-${index}`} dimColor>
{` ${index + 1}. ${phase.title}`}
{phase.detail ? ` — ${phase.detail}` : ''}
</Text>
))}
</Box>
) : null}
{input.scriptPath ? (
<Box marginTop={1}>
<Text dimColor>{`Script: ${input.scriptPath}`}</Text>
</Box>
) : null}
{showScript && script ? (
<Box flexDirection="column" marginTop={1}>
<Text dimColor>Raw script:</Text>
<Text>{clipScript(script)}</Text>
</Box>
) : null}
</Box>
<Box flexDirection="column">
<PermissionRuleExplanation
permissionResult={toolUseConfirm.permissionResult}
toolType="tool"
/>
<PermissionPrompt
options={options}
onSelect={handleSelect}
onCancel={handleCancel}
toolAnalyticsContext={toolAnalyticsContext}
/>
</Box>
</PermissionDialog>
)
}
const SCRIPT_PREVIEW_LINES = 60
/** Long scripts are clipped: the dialog must stay smaller than the terminal. */
function clipScript(script: string): string {
const lines = script.split('\n')
if (lines.length <= SCRIPT_PREVIEW_LINES) return script
const remaining = lines.length - SCRIPT_PREVIEW_LINES
return (
`${lines.slice(0, SCRIPT_PREVIEW_LINES).join('\n')}\n` +
`… ${remaining} more lines — the full script is persisted under the session directory`
)
}
/**
* Read the script's `meta` for the preview.
*
* Parsing can fail here — the tool has not validated the script yet — and a
* bad script should still reach the tool so the model sees the real parse
* error, not a silent refusal in the dialog.
*/
function readMeta(script: string | undefined): WorkflowMeta | undefined {
if (!script) return undefined
const parsed = parseWorkflowScript(script)
return 'error' in parsed ? undefined : parsed.meta
}
+182
View File
@@ -0,0 +1,182 @@
import { describe, expect, test } from 'bun:test'
import type { ToolUseContext } from '../../Tool.js'
import { getEmptyToolPermissionContext } from '../../Tool.js'
import type { AppState } from '../../state/AppState.js'
import type { PermissionMode } from '../../types/permissions.js'
import { resolveScriptForTesting, WorkflowTool } from './WorkflowTool.js'
function makeContext(options: {
mode?: PermissionMode
isNonInteractiveSession?: boolean
allowRules?: string[]
ultracode?: boolean
}): ToolUseContext {
const permissionContext = {
...getEmptyToolPermissionContext(),
mode: options.mode ?? 'default',
...(options.allowRules
? {
alwaysAllowRules: {
localSettings: options.allowRules.map(
content => `${WorkflowTool.name}(${content})`,
),
},
}
: {}),
}
return {
options: {
isNonInteractiveSession: options.isNonInteractiveSession ?? false,
},
getAppState: () =>
({
toolPermissionContext: permissionContext,
ultracode: options.ultracode ?? false,
}) as unknown as AppState,
} as unknown as ToolUseContext
}
describe('WorkflowTool.checkPermissions', () => {
test('asks before starting a run in default mode', async () => {
const result = await WorkflowTool.checkPermissions(
{ name: 'deep-research' },
makeContext({ mode: 'default' }),
)
expect(result.behavior).toBe('ask')
if (result.behavior !== 'ask') return
expect(result.message).toContain('spawn many subagents')
// The "don't ask again" option needs a rule to write.
expect(result.suggestions?.[0]).toMatchObject({
type: 'addRules',
behavior: 'allow',
destination: 'localSettings',
})
})
test('still asks in acceptEdits — a run auto-approves its agents edits', async () => {
const result = await WorkflowTool.checkPermissions(
{ name: 'deep-research' },
makeContext({ mode: 'acceptEdits' }),
)
expect(result.behavior).toBe('ask')
})
test('ultracode skips the launch prompt — it is a standing opt-in', async () => {
const result = await WorkflowTool.checkPermissions(
{ name: 'deep-research' },
makeContext({ mode: 'default', ultracode: true }),
)
expect(result.behavior).toBe('allow')
})
test('allows without prompting under bypassPermissions', async () => {
const result = await WorkflowTool.checkPermissions(
{ name: 'deep-research' },
makeContext({ mode: 'bypassPermissions' }),
)
expect(result.behavior).toBe('allow')
})
test('allows in a non-interactive session, where nobody can answer', async () => {
const result = await WorkflowTool.checkPermissions(
{ script: "export const meta = { name: 'x', description: 'y' }\n" },
makeContext({ isNonInteractiveSession: true }),
)
expect(result.behavior).toBe('allow')
})
test('an allow rule for one workflow does not cover another', async () => {
const context = makeContext({ allowRules: ['deep-research'] })
const allowed = await WorkflowTool.checkPermissions(
{ name: 'deep-research' },
context,
)
expect(allowed.behavior).toBe('allow')
const other = await WorkflowTool.checkPermissions(
{ name: 'something-else' },
context,
)
expect(other.behavior).toBe('ask')
})
test('an inline script has no rule to match, so it always asks', async () => {
const result = await WorkflowTool.checkPermissions(
{ script: "export const meta = { name: 'x', description: 'y' }\n" },
makeContext({ allowRules: ['x'] }),
)
expect(result.behavior).toBe('ask')
if (result.behavior !== 'ask') return
expect(result.suggestions).toBeUndefined()
})
})
describe('resolveScript', () => {
test('a named workflow carries no scriptPath, so the run gets a session copy', async () => {
const resolved = await resolveScriptForTesting({ name: 'deep-research' })
expect('script' in resolved && resolved.script).toContain('export const meta')
// Pointing the run at ~/.claude/workflows/<name>.js would make it write
// back over the user's file, and a later resume would replay whatever that
// file says rather than what actually ran.
expect(
(resolved as { scriptPath?: string }).scriptPath,
).toBeUndefined()
})
test('an explicit scriptPath is preserved so a resume reads the edited file', async () => {
const resolved = await resolveScriptForTesting({
scriptPath: '/definitely/not/here.js',
})
expect('error' in resolved && resolved.error).toContain('Failed to read')
})
})
describe('WorkflowTool.call', () => {
test('rejects a call with no script, scriptPath, or name', async () => {
await expect(
WorkflowTool.call(
{},
makeContext({ mode: 'bypassPermissions' }),
(() => {}) as never,
{} as never,
),
).rejects.toThrow('requires one of `script`, `scriptPath`, or `name`')
})
test('surfaces the parse error for a script without a leading meta export', async () => {
await expect(
WorkflowTool.call(
{ script: 'const x = 1\n' },
makeContext({ mode: 'bypassPermissions' }),
(() => {}) as never,
{} as never,
),
).rejects.toThrow('must be the FIRST statement')
})
test('points at plain JavaScript when the body has TypeScript syntax', async () => {
await expect(
WorkflowTool.call(
{
script:
"export const meta = { name: 'x', description: 'y' }\nconst files: string[] = []\n",
},
makeContext({ mode: 'bypassPermissions' }),
(() => {}) as never,
{} as never,
),
).rejects.toThrow('plain JavaScript')
})
test('reports a missing named workflow with the available list', async () => {
await expect(
WorkflowTool.call(
{ name: 'no-such-workflow' },
makeContext({ mode: 'bypassPermissions' }),
(() => {}) as never,
{} as never,
),
).rejects.toThrow("Unknown workflow 'no-such-workflow'")
})
})
+295 -30
View File
@@ -1,34 +1,299 @@
// @generated stub from scan-missing-imports
// 该文件自动生成,对应 ant-internal 的 feature() gated 模块。
// 所有外部 build 的代码路径在 DCE 后都不会真的执行这里的代码,这只是
// bun build resolver 的占位符。
const __target = function noop() {}
const __handler: ProxyHandler<any> = {
get(_t, prop) {
if (prop === '__esModule') return true
if (prop === 'default') return new Proxy(__target, __handler)
if (prop === Symbol.toPrimitive) return () => undefined
if (prop === Symbol.iterator) return function* () {}
if (prop === Symbol.asyncIterator) return async function* () {}
if (prop === 'then') return undefined
return new Proxy(__target, __handler)
import { readFile } from 'fs/promises'
import { z } from 'zod/v4'
import { buildTool, type ToolDef } from '../../Tool.js'
import { generateTaskId } from '../../Task.js'
import { lazySchema } from '../../utils/lazySchema.js'
import { getRuleByContentsForTool } from '../../utils/permissions/permissions.js'
import { hasAcceptedWorkflowsInAutoMode } from '../../utils/workflows/autoModeConsent.js'
import { jsonStringify } from '../../utils/slowOperations.js'
import { findWorkflowByName, loadWorkflows } from '../../utils/workflows/discovery.js'
import {
areWorkflowsEnabled,
describeWorkflowsDisabled,
getWorkflowsDisabledReason,
} from '../../utils/workflows/enabled.js'
import { WORKFLOW_SCRIPT_MAX_BYTES } from '../../utils/workflows/constants.js'
import { createWorkflowRunId } from '../../utils/workflows/paths.js'
import { prepareWorkflowScript } from '../../utils/workflows/runtime.js'
import { launchWorkflow } from './launchWorkflow.js'
import { WORKFLOW_TOOL_NAME } from './constants.js'
import { getWorkflowToolPrompt } from './prompt.js'
const inputSchema = lazySchema(() =>
z.strictObject({
script: z
.string()
.max(WORKFLOW_SCRIPT_MAX_BYTES)
.optional()
.describe(
'Self-contained workflow script. Must begin with `export const meta = { name, description, phases }` ' +
'(pure literal, no computed values) followed by the script body using agent()/parallel()/pipeline()/phase().',
),
scriptPath: z
.string()
.optional()
.describe(
'Path to a workflow script file on disk. Every Workflow invocation persists its script under the ' +
'session directory and returns the path in the tool result. Takes precedence over `script` and `name`.',
),
name: z
.string()
.optional()
.describe(
'Name of a predefined workflow (built-in or from .claude/workflows/).',
),
args: z
.unknown()
.optional()
.describe(
'Optional input value exposed to the script as the global `args`, verbatim. Pass arrays/objects as ' +
'actual JSON values, NOT as a JSON-encoded string.',
),
title: z.string().optional().describe('Ignored — set the title in `meta`.'),
description: z
.string()
.optional()
.describe('Ignored — set the description in `meta`.'),
resumeFromRunId: z
.string()
.regex(/^wf_[a-z0-9-]{6,}$/)
.optional()
.describe(
'Run ID of a prior Workflow invocation to resume from. Completed agent() calls with unchanged ' +
'(prompt, opts) return their cached results instantly; only edited or new calls re-run.',
),
}),
)
type InputSchema = ReturnType<typeof inputSchema>
// Mirrors the official WorkflowOutput in @anthropic-ai/claude-code's
// sdk-tools.d.ts. Optional fields are optional there too, so a transcript
// written before a field existed still replays without re-validation failing.
const outputSchema = lazySchema(() =>
z.object({
status: z.enum(['async_launched', 'remote_launched']),
taskId: z.string(),
taskType: z.enum(['local_workflow', 'remote_agent']).optional(),
workflowName: z.string().optional(),
runId: z.string().optional(),
summary: z.string().optional(),
transcriptDir: z.string().optional(),
scriptPath: z.string().optional(),
sessionUrl: z.string().optional(),
warning: z.string().optional(),
error: z.string().optional(),
}),
)
type OutputSchema = ReturnType<typeof outputSchema>
export type Output = z.infer<OutputSchema>
export const WorkflowTool = buildTool({
name: WORKFLOW_TOOL_NAME,
searchHint: 'orchestrate many subagents from a script',
maxResultSizeChars: 100_000,
userFacingName: () => 'Workflow',
get inputSchema(): InputSchema {
return inputSchema()
},
apply() {
return new Proxy(__target, __handler)
get outputSchema(): OutputSchema {
return outputSchema()
},
construct() {
return new Proxy(__target, __handler)
isEnabled() {
return areWorkflowsEnabled()
},
isConcurrencySafe() {
return false
},
isReadOnly() {
return false
},
isOpenWorld() {
return true
},
toAutoClassifierInput(input) {
return input.script ?? input.scriptPath ?? input.name ?? ''
},
/**
* A run spawns many agents whose file edits are auto-approved, so the launch
* itself is the only place the user gets to say no. `bypassPermissions` and
* non-interactive runs have nobody to ask; everywhere else prompts unless the
* user has allow-listed this workflow name.
*/
async checkPermissions(input, context) {
const appState = context?.getAppState()
const permissionContext = appState?.toolPermissionContext
if (
permissionContext?.mode === 'bypassPermissions' ||
context?.options.isNonInteractiveSession
) {
return { behavior: 'allow', updatedInput: input }
}
// Ultracode is a standing instruction to orchestrate every task; prompting
// per run would mean prompting every turn.
if (appState?.ultracode === true) {
return { behavior: 'allow', updatedInput: input }
}
// Auto mode asks once per machine, then remembers.
if (permissionContext?.mode === 'auto' && hasAcceptedWorkflowsInAutoMode()) {
return { behavior: 'allow', updatedInput: input }
}
const ruleContent = typeof input.name === 'string' ? input.name : undefined
if (ruleContent && permissionContext) {
const denied = getRuleByContentsForTool(
permissionContext,
WorkflowTool,
'deny',
).get(ruleContent)
if (denied) {
return {
behavior: 'deny',
message: `${WORKFLOW_TOOL_NAME} denied for workflow "${ruleContent}".`,
decisionReason: { type: 'rule', rule: denied },
}
}
const allowed = getRuleByContentsForTool(
permissionContext,
WorkflowTool,
'allow',
).get(ruleContent)
if (allowed) {
return {
behavior: 'allow',
updatedInput: input,
decisionReason: { type: 'rule', rule: allowed },
}
}
}
return {
behavior: 'ask',
message:
'Claude wants to run a dynamic workflow, which can spawn many subagents and use a large number of tokens.',
...(ruleContent
? {
suggestions: [
{
type: 'addRules' as const,
rules: [{ toolName: WORKFLOW_TOOL_NAME, ruleContent }],
behavior: 'allow' as const,
destination: 'localSettings' as const,
},
],
}
: {}),
}
},
async description(input) {
if (input.name) return `Run the ${input.name} workflow`
return 'Run a dynamic workflow'
},
async prompt() {
return getWorkflowToolPrompt()
},
mapToolResultToToolResultBlockParam(output, toolUseID) {
return {
tool_use_id: toolUseID,
type: 'tool_result',
content: jsonStringify(output),
}
},
async call(input, toolUseContext, canUseTool) {
const disabled = getWorkflowsDisabledReason()
if (disabled) throw new Error(describeWorkflowsDisabled(disabled))
const resolved = await resolveScript(input)
if ('error' in resolved) throw new Error(resolved.error)
const prepared = prepareWorkflowScript(resolved.script)
if (!prepared.ok) throw new Error(prepared.error)
const workflowRunId = input.resumeFromRunId ?? createWorkflowRunId()
const taskId = generateTaskId('local_workflow')
const launched = launchWorkflow({
taskId,
workflowRunId,
script: resolved.script,
scriptPath: resolved.scriptPath,
args: input.args,
meta: prepared.meta,
vmScript: prepared.vmScript,
toolUseContext,
canUseTool,
toolUseId: toolUseContext.toolUseId,
isResume: input.resumeFromRunId !== undefined,
})
return {
data: {
status: 'async_launched' as const,
taskId,
taskType: 'local_workflow' as const,
workflowName: prepared.meta.name,
runId: workflowRunId,
summary: prepared.meta.description,
transcriptDir: launched.transcriptDir,
scriptPath: launched.scriptPath,
},
}
},
} satisfies ToolDef<InputSchema, Output>)
/**
* Work out which script this call should run.
*
* `scriptPath` wins so an edited run can be relaunched byte-for-byte, then a
* saved `name`, then an inline `script`. Resolving by name here (rather than
* making the model paste the script back) is what lets `/deep-research` and
* saved workflows be one-line calls.
*/
export async function resolveScriptForTesting(input: {
script?: string
scriptPath?: string
name?: string
}): Promise<{ script: string; scriptPath?: string } | { error: string }> {
return resolveScript(input)
}
async function resolveScript(input: {
script?: string
scriptPath?: string
name?: string
}): Promise<{ script: string; scriptPath?: string } | { error: string }> {
if (input.scriptPath) {
try {
const script = await readFile(input.scriptPath, 'utf8')
return { script, scriptPath: input.scriptPath }
} catch (error) {
return {
error: `Failed to read workflow script file ${input.scriptPath}: ${
error instanceof Error ? error.message : String(error)
}`,
}
}
}
if (input.name) {
const workflow = await findWorkflowByName(input.name)
if (!workflow) {
const available = (await loadWorkflows())
.map(entry => entry.name)
.join(', ')
return {
error: `Unknown workflow '${input.name}'.${available ? ` Available: ${available}` : ''}`,
}
}
// Deliberately no scriptPath: the run gets its own session copy. Pointing
// it at the saved workflow would make the run write back over the user's
// file, and would resume from whatever that file says later rather than
// from what actually ran.
return { script: workflow.script }
}
if (input.script) return { script: input.script }
return {
error: 'Workflow requires one of `script`, `scriptPath`, or `name`.',
}
}
const stub: any = new Proxy(__target, __handler)
export default stub
export const __stubMissing = true
// 兼容常见的命名导出 —— 没列在这里的也会通过 default Proxy 兜底
export const createCachedMCState = stub
export const isCachedMicrocompactEnabled = stub
export const isModelSupportedForCacheEditing = stub
export const getCachedMCConfig = stub
export const markToolsSentToAPI = stub
export const resetCachedMCState = stub
export const checkProtectedNamespace = stub
export const getCoordinatorUserContext = stub
-34
View File
@@ -1,34 +0,0 @@
// @generated stub from scan-missing-imports
// 该文件自动生成,对应 ant-internal 的 feature() gated 模块。
// 所有外部 build 的代码路径在 DCE 后都不会真的执行这里的代码,这只是
// bun build resolver 的占位符。
const __target = function noop() {}
const __handler: ProxyHandler<any> = {
get(_t, prop) {
if (prop === '__esModule') return true
if (prop === 'default') return new Proxy(__target, __handler)
if (prop === Symbol.toPrimitive) return () => undefined
if (prop === Symbol.iterator) return function* () {}
if (prop === Symbol.asyncIterator) return async function* () {}
if (prop === 'then') return undefined
return new Proxy(__target, __handler)
},
apply() {
return new Proxy(__target, __handler)
},
construct() {
return new Proxy(__target, __handler)
},
}
const stub: any = new Proxy(__target, __handler)
export default stub
export const __stubMissing = true
// 兼容常见的命名导出 —— 没列在这里的也会通过 default Proxy 兜底
export const createCachedMCState = stub
export const isCachedMicrocompactEnabled = stub
export const isModelSupportedForCacheEditing = stub
export const getCachedMCConfig = stub
export const markToolsSentToAPI = stub
export const resetCachedMCState = stub
export const checkProtectedNamespace = stub
export const getCoordinatorUserContext = stub
+1 -1
View File
@@ -1 +1 @@
export const WORKFLOW_TOOL_NAME = 'workflow'
export const WORKFLOW_TOOL_NAME = 'Workflow'
+66 -33
View File
@@ -1,34 +1,67 @@
// @generated stub from scan-missing-imports
// 该文件自动生成,对应 ant-internal 的 feature() gated 模块。
// 所有外部 build 的代码路径在 DCE 后都不会真的执行这里的代码,这只是
// bun build resolver 的占位符。
const __target = function noop() {}
const __handler: ProxyHandler<any> = {
get(_t, prop) {
if (prop === '__esModule') return true
if (prop === 'default') return new Proxy(__target, __handler)
if (prop === Symbol.toPrimitive) return () => undefined
if (prop === Symbol.iterator) return function* () {}
if (prop === Symbol.asyncIterator) return async function* () {}
if (prop === 'then') return undefined
return new Proxy(__target, __handler)
},
apply() {
return new Proxy(__target, __handler)
},
construct() {
return new Proxy(__target, __handler)
},
import type { ContentBlockParam } from '@anthropic-ai/sdk/resources/index.mjs'
import type { Command } from '../../types/command.js'
import { loadWorkflows } from '../../utils/workflows/discovery.js'
import { areWorkflowsEnabled } from '../../utils/workflows/enabled.js'
import type { WorkflowDefinition } from '../../utils/workflows/types.js'
import { WORKFLOW_TOOL_NAME } from './constants.js'
/**
* Turn every discovered workflow into a `/<name>` command.
*
* The command is a prompt, not a direct tool call: whatever the user typed
* after the name has to become the `args` value, and only the model can decide
* whether "issues 1024, 1025" is a list of numbers or a sentence. The prompt
* pins everything else so the model's only job is that conversion.
*/
export async function getWorkflowCommands(cwd?: string): Promise<Command[]> {
if (!areWorkflowsEnabled()) return []
const workflows = await loadWorkflows(cwd)
return workflows.map(workflow => createWorkflowCommand(workflow))
}
export function createWorkflowCommand(workflow: WorkflowDefinition): Command {
return {
type: 'prompt',
name: workflow.name,
description: workflow.description,
progressMessage: `running the ${workflow.name} workflow`,
contentLength: workflow.description.length + 200,
argNames: ['args'],
source: workflowCommandSource(workflow),
isEnabled: () => areWorkflowsEnabled(),
userFacingName: () => workflow.name,
async getPromptForCommand(args: string): Promise<ContentBlockParam[]> {
const argsLine =
args.trim() === ''
? 'Pass no `args`.'
: `Convert the following invocation text into the \`args\` value and pass it: ${JSON.stringify(args.trim())}. ` +
'If it is a list of items, pass a real JSON array, not a string.'
return [
{
type: 'text',
text:
`Run the saved workflow "${workflow.name}" by calling ${WORKFLOW_TOOL_NAME} exactly once with ` +
`{ name: "${workflow.name}" }. ${argsLine}\n` +
'Do not write a new script and do not do the work yourself — the saved workflow already ' +
'contains the orchestration. After the tool returns, end your turn; the result arrives as a ' +
'task notification.',
},
]
},
} as Command
}
function workflowCommandSource(
workflow: WorkflowDefinition,
): 'builtin' | 'userSettings' | 'projectSettings' | 'plugin' {
switch (workflow.source) {
case 'built-in':
return 'builtin'
case 'plugin':
return 'plugin'
case 'projectSettings':
return 'projectSettings'
default:
return 'userSettings'
}
}
const stub: any = new Proxy(__target, __handler)
export default stub
export const __stubMissing = true
// 兼容常见的命名导出 —— 没列在这里的也会通过 default Proxy 兜底
export const createCachedMCState = stub
export const isCachedMicrocompactEnabled = stub
export const isModelSupportedForCacheEditing = stub
export const getCachedMCConfig = stub
export const markToolsSentToAPI = stub
export const resetCachedMCState = stub
export const checkProtectedNamespace = stub
export const getCoordinatorUserContext = stub
+365
View File
@@ -0,0 +1,365 @@
import { mkdir, writeFile } from 'fs/promises'
import { dirname } from 'path'
import type vm from 'vm'
import {
getCurrentTurnTokenBudget,
getTurnOutputTokens,
} from '../../bootstrap/state.js'
import type { CanUseToolFn } from '../../hooks/useCanUseTool.js'
import type { SetAppState } from '../../Task.js'
import type { ToolUseContext } from '../../Tool.js'
import {
completeWorkflowTask,
enqueueWorkflowNotification,
failWorkflowTask,
registerWorkflowTask,
updateWorkflowProgressBatch,
type LocalWorkflowTaskState,
} from '../../tasks/LocalWorkflowTask/LocalWorkflowTask.js'
import { logForDebugging } from '../../utils/debug.js'
import { registerTask, updateTaskState } from '../../utils/task/framework.js'
import { emitTaskProgress } from '../../utils/task/sdkProgress.js'
import {
WORKFLOW_PANEL_EMIT_INTERVAL_MS,
WORKFLOW_PROGRESS_BATCH_MS,
} from '../../utils/workflows/constants.js'
import { createWorkflowSharedCounters } from '../../utils/workflows/harness.js'
import { createRunJournal } from '../../utils/workflows/journal.js'
import {
getWorkflowTranscriptDir,
getWorkflowScriptPath,
} from '../../utils/workflows/paths.js'
import { executeWorkflowScript } from '../../utils/workflows/runtime.js'
import { createNestedWorkflowRunner } from './runNestedWorkflow.js'
import {
isDurableWorkflowEvent,
type WorkflowMeta,
type WorkflowProgressEvent,
} from '../../utils/workflows/types.js'
export type LaunchWorkflowParams = {
taskId: string
workflowRunId: string
script: string
scriptPath?: string
args?: unknown
meta: WorkflowMeta
vmScript: vm.Script
toolUseContext: ToolUseContext
canUseTool: CanUseToolFn
toolUseId?: string
isResume: boolean
runNestedWorkflow?: (nameOrRef: unknown, args: unknown) => Promise<unknown>
}
export type LaunchedWorkflow = {
task: LocalWorkflowTaskState
transcriptDir: string
scriptPath: string
}
/**
* Register the background task for a run and start executing its script.
*
* Returns as soon as the task exists so the tool call can hand the model a run
* id immediately; everything after that happens on the detached promise below.
*/
export function launchWorkflow(params: LaunchWorkflowParams): LaunchedWorkflow {
const {
taskId,
workflowRunId,
script,
args,
meta,
vmScript,
toolUseContext,
canUseTool,
toolUseId,
isResume,
} = params
const setAppState: SetAppState =
toolUseContext.setAppStateForTasks ?? toolUseContext.setAppState
const transcriptDir = getWorkflowTranscriptDir(workflowRunId)
// A caller-supplied path is read-only: it is the user's file, and a resume
// is supposed to pick up their edits, not overwrite them.
const callerSuppliedPath = params.scriptPath !== undefined
const scriptPath =
params.scriptPath ?? getWorkflowScriptPath(workflowRunId, meta.name)
// A resumed run replaces the settled task for the same run id, so the
// progress view shows one entry instead of a stack of dead ones.
if (isResume) {
const tasks = toolUseContext.getAppState().tasks
for (const [id, task] of Object.entries(tasks)) {
if (
task.type === 'local_workflow' &&
task.workflowRunId === workflowRunId &&
task.status !== 'running'
) {
setAppState(prev => {
const { [id]: _removed, ...rest } = prev.tasks
return { ...prev, tasks: rest }
})
}
}
}
const task = registerWorkflowTask({
taskId,
script,
scriptPath,
args,
summary: meta.description,
workflowName: meta.name,
title: meta.title,
phases: meta.phases,
defaultModel: toolUseContext.options.mainLoopModel,
workflowRunId,
ownerAgentId: toolUseContext.agentId,
toolUseId,
})
registerTask(task, setAppState)
if (!callerSuppliedPath) void persistScript(scriptPath, script)
const budgetTotal = getCurrentTurnTokenBudget()
const spentBeforeRun = getTurnOutputTokens()
const tokenBudget = {
total: budgetTotal,
getTurnSpent: () => getTurnOutputTokens() - spentBeforeRun,
}
const batcher = createProgressBatcher({
taskId,
toolUseId,
setAppState,
getTask: () =>
toolUseContext.getAppState().tasks[taskId] as
| LocalWorkflowTaskState
| undefined,
fallbackDescription: task.description,
startTime: task.startTime,
summary: meta.description,
})
void (async () => {
const journal = createRunJournal(workflowRunId)
const journalSnapshot = isResume ? await journal.load() : undefined
const shared = createWorkflowSharedCounters()
const nestedContext = {
...toolUseContext,
abortController: task.abortController ?? toolUseContext.abortController,
}
const onAgentController = (
agentKey: string,
controller: AbortController | undefined,
) => {
updateTaskState<LocalWorkflowTaskState>(taskId, setAppState, current => {
if (!current.agentControllers) return current
if (controller) current.agentControllers.set(agentKey, controller)
else current.agentControllers.delete(agentKey)
return current
})
}
const outcome = await executeWorkflowScript({
vmScript,
toolUseContext: nestedContext,
canUseTool,
runId: workflowRunId,
workflowName: meta.name,
args,
seedPhaseTitles: meta.phases?.map(phase => phase.title),
tokenBudget,
journal,
journalSnapshot,
shared,
onProgress: event => batcher.push(event),
onAgentController,
runNestedWorkflow:
params.runNestedWorkflow ??
createNestedWorkflowRunner({
toolUseContext: nestedContext,
canUseTool,
runId: workflowRunId,
tokenBudget,
journal,
shared,
onProgress: event => batcher.push(event),
onAgentController,
}),
})
batcher.flush()
const settled = toolUseContext.getAppState().tasks[taskId] as
| LocalWorkflowTaskState
| undefined
// A run the user stopped is already terminal; do not overwrite that.
if (settled && settled.status !== 'running') return
const totalTokens = settled?.totalTokens ?? 0
const totalToolCalls = settled?.totalToolCalls ?? 0
const status = outcome.error ? 'failed' : 'completed'
if (outcome.error) {
failWorkflowTask(
taskId,
outcome.error,
outcome.agentCount,
outcome.logs,
setAppState,
)
} else {
completeWorkflowTask(
taskId,
outcome.result,
outcome.agentCount,
outcome.logs,
setAppState,
)
}
enqueueWorkflowNotification({
taskId,
summary: meta.description,
status,
result: outcome.error ? undefined : outcome.result,
error: outcome.error,
failures: outcome.failures,
agentCount: outcome.agentCount,
totalTokens,
totalToolCalls,
durationMs: outcome.durationMs,
toolUseId,
transcriptDir,
scriptPath,
workflowRunId,
args,
setAppState,
})
})().catch(error => {
batcher.cancel()
const message = error instanceof Error ? error.message : String(error)
logForDebugging(`Workflow ${workflowRunId} crashed: ${message}`)
failWorkflowTask(taskId, message, 0, [], setAppState)
enqueueWorkflowNotification({
taskId,
summary: meta.description,
status: 'failed',
error: message,
agentCount: 0,
totalTokens: 0,
totalToolCalls: 0,
durationMs: Date.now() - task.startTime,
toolUseId,
transcriptDir,
scriptPath,
workflowRunId,
args,
setAppState,
})
})
return { task, transcriptDir, scriptPath }
}
/**
* Coalesce progress events before they touch AppState.
*
* A 40-agent run emits thousands of token-count updates; applying each one
* would re-render the whole task list per event. Batching on a short timer
* keeps the view live without making the UI the bottleneck.
*/
function createProgressBatcher(params: {
taskId: string
toolUseId?: string
setAppState: SetAppState
getTask: () => LocalWorkflowTaskState | undefined
fallbackDescription: string
startTime: number
summary?: string
}) {
let pending: WorkflowProgressEvent[] = []
let timer: ReturnType<typeof setTimeout> | undefined
let lastPanelEmit = 0
const drain = (): void => {
timer = undefined
if (pending.length === 0) return
const batch = pending
pending = []
updateWorkflowProgressBatch(params.taskId, batch, params.setAppState)
emitPanelProgress(batch)
}
const emitPanelProgress = (batch: WorkflowProgressEvent[]): void => {
const durable = batch.filter(isDurableWorkflowEvent)
if (durable.length === 0) return
const task = params.getTask()
if (task?.type !== 'local_workflow' || task.status !== 'running') return
const lastAgent = [...durable]
.reverse()
.find(event => event.type === 'workflow_agent')
// Token-count churn is throttled; anything else (an agent finishing, a new
// phase) goes out immediately so the panel is never visibly behind.
const onlyTokenChurn = durable.every(
event => event.type === 'workflow_agent' && event.state === 'progress',
)
const now = Date.now()
if (onlyTokenChurn && now - lastPanelEmit < WORKFLOW_PANEL_EMIT_INTERVAL_MS) {
return
}
lastPanelEmit = now
emitTaskProgress({
taskId: params.taskId,
toolUseId: params.toolUseId,
description:
lastAgent?.type === 'workflow_agent'
? lastAgent.phaseTitle
? `${lastAgent.phaseTitle}: ${lastAgent.label}`
: lastAgent.label
: params.fallbackDescription,
startTime: params.startTime,
totalTokens: task.totalTokens,
toolUses: task.totalToolCalls,
lastToolName:
lastAgent?.type === 'workflow_agent' ? lastAgent.label : undefined,
summary: params.summary,
workflowProgress: task.workflowProgress.filter(isDurableWorkflowEvent),
})
}
return {
push(event: WorkflowProgressEvent): void {
pending.push(event)
if (!timer) {
timer = setTimeout(drain, WORKFLOW_PROGRESS_BATCH_MS)
timer.unref?.()
}
},
flush(): void {
if (timer) clearTimeout(timer)
drain()
},
cancel(): void {
if (timer) clearTimeout(timer)
timer = undefined
pending = []
},
}
}
async function persistScript(path: string, script: string): Promise<void> {
try {
await mkdir(dirname(path), { recursive: true })
await writeFile(path, script, 'utf8')
} catch (error) {
logForDebugging(
`Failed to persist workflow script to ${path}: ${error instanceof Error ? error.message : String(error)}`,
)
}
}
+161
View File
@@ -0,0 +1,161 @@
import {
WORKFLOW_MAX_AGENTS,
WORKFLOW_MAX_FANOUT,
getWorkflowConcurrency,
} from '../../utils/workflows/constants.js'
import { describeWorkflowSizeGuideline } from '../../utils/workflows/enabled.js'
/**
* The Workflow tool description.
*
* This is the whole product surface for the model: everything it knows about
* when to reach for a workflow, how to shape one, and which patterns produce
* trustworthy results comes from here. Sections that read as over-explained
* (the pipeline-vs-barrier argument, the loop-until-dry example) are load
* bearing — without them models default to a single `parallel()` barrier per
* stage and lose most of the wall-clock advantage.
*/
export function getWorkflowToolPrompt(): string {
return `Execute a workflow script that orchestrates multiple subagents deterministically. Workflows run in the background — this tool returns immediately with a task ID, and a <task-notification> arrives when the workflow completes. Use /workflows to watch live progress.
A workflow structures work across many agents — to be comprehensive (decompose and cover in parallel), to be confident (independent perspectives and adversarial checks before committing), or to take on scale one context can't hold (migrations, audits, broad sweeps). The script is where you encode that structure: what fans out, what verifies, what synthesizes.
ONLY call this tool when the user has explicitly opted into multi-agent orchestration. Workflows can spawn dozens of agents and consume a large amount of tokens; the user must request that scale, not have it inferred. Explicit opt-in means one of:
- The user included the keyword "ultracode" in their prompt (you'll see a system-reminder confirming it).
- Ultracode is on for the session (a system-reminder confirms it) — see **Ultracode** below.
- The user directly asked you to run a workflow or use multi-agent orchestration in their own words ("use a workflow", "run a workflow", "fan out agents", "orchestrate this with subagents"). The ask must be in the user's words — a task that would merely benefit from a workflow does not count.
- The user invoked a skill or slash command whose instructions tell you to call Workflow.
- The user asked you to run a specific named or saved workflow.
For any other task — even one that would clearly benefit from parallelism — do NOT call this tool. Use the Agent tool for individual subagents, or briefly describe what a multi-agent workflow could do and how much it would roughly cost, and ask the user whether to run it.
**Ultracode.** When a system-reminder confirms ultracode is on, that opt-in is standing: author and run a workflow for every substantive task by default. The goal is the most exhaustive, correct answer you can produce — token cost is not a constraint. For multi-phase work (understand → design → implement → review), that often means several workflows in sequence — one per phase — so you stay in the loop between them. The quality patterns below (adversarial verify, multi-modal sweep, completeness critic, loop-until-dry) are the tools; pick what fits the task. Lean toward orchestrating with workflows and adversarially verifying your findings — unless the work is trivial or already verified. Solo only on conversational turns or trivial mechanical edits. When a reminder says ultracode is off, revert to the opt-in rule above.
When you do call it, the right move is often **hybrid**: scout inline first (list the files, find the channels, scope the diff) to discover the work-list, then call Workflow to pipeline over it. You don't need to know the shape before the *task* — only before the *orchestration step*.
Common single-phase workflows you can chain across turns:
- **Understand** — parallel readers over relevant subsystems → structured map
- **Design** — judge panel of N independent approaches → scored synthesis
- **Review** — dimensions → find → adversarially verify (example below)
- **Research** — multi-modal sweep → deep-read → synthesize
- **Migrate** — discover sites → transform each (worktree isolation) → verify
For larger work, run several in sequence — read each result before deciding the next phase. You stay in the loop; each workflow is one well-scoped fan-out.
Pass the script inline via \`script\` — do not Write it to a file first. Every invocation automatically persists its script to a file under the session directory and returns the path in the tool result. To iterate on a workflow, edit that file with Write/Edit and re-invoke Workflow with \`{scriptPath: "<path>"}\` instead of resending the full script.
Every script must begin with \`export const meta = {...}\`:
export const meta = {
name: 'find-flaky-tests',
description: 'Find flaky tests and propose fixes', // one-line, shown in permission dialog
phases: [ // one entry per phase() call
{ title: 'Scan', detail: 'grep test logs for retries' },
{ title: 'Fix', detail: 'one agent per flaky test' },
],
}
// script body starts here — use agent()/parallel()/pipeline()/phase()/log()
phase('Scan')
const flaky = await agent('grep CI logs for retry markers', {schema: FLAKY_SCHEMA})
...
The \`meta\` object must be a PURE LITERAL — no variables, function calls, spreads, or template interpolation. Required fields: \`name\`, \`description\`. Optional: \`whenToUse\` (shown in the workflow list), \`phases\`. Use the SAME phase titles in meta.phases as in phase() calls — titles are matched exactly; a phase() call with no matching meta entry just gets its own progress group. Add \`model\` to a phase entry when that phase uses a specific model override.
Script body hooks:
- agent(prompt: string, opts?: {label?: string, phase?: string, schema?: object, model?: string, effort?: string, isolation?: 'worktree', agentType?: string}): Promise<any> — spawn a subagent. Without schema, returns its final text as a string. With schema (a JSON Schema), the subagent is forced to call a StructuredOutput tool and agent() returns the validated object — no parsing needed. Returns null if the user skips the agent mid-run or the subagent dies on a terminal API error after retries (filter with .filter(Boolean)). opts.label overrides the display label. opts.phase explicitly assigns this agent to a progress group (use this inside pipeline()/parallel() stages to avoid races on the global phase() state — same phase string → same group box). opts.model overrides the model for this agent call. Default to omitting it — the agent inherits the main-loop model, which is almost always correct. opts.effort overrides the reasoning effort for this agent call. opts.isolation: 'worktree' runs the agent in a fresh git worktree — EXPENSIVE, use ONLY when agents mutate files in parallel and would otherwise conflict; the worktree is auto-removed if unchanged. opts.agentType uses a custom subagent type (e.g. 'general-purpose', 'code-reviewer') instead of the default workflow subagent; composes with schema.
- pipeline(items, stage1, stage2, ...): Promise<any[]> — run each item through all stages independently, NO barrier between stages. Item A can be in stage 3 while item B is still in stage 1. This is the DEFAULT for multi-stage work. Wall-clock = slowest single-item chain, not sum-of-slowest-per-stage. Every stage callback receives (prevResult, originalItem, index). A stage that throws drops that item to \`null\` and skips its remaining stages.
- parallel(thunks: Array<() => Promise<any>>): Promise<any[]> — run tasks concurrently. This is a BARRIER: awaits all thunks before returning. A thunk that throws resolves to \`null\` in the result array — the call itself never rejects, so \`.filter(Boolean)\` before using the results. Use ONLY when you genuinely need all results together.
- log(message: string): void — emit a progress message to the user
- phase(title: string): void — start a new phase; subsequent agent() calls are grouped under this title in the progress display
- args: any — the value passed as Workflow's \`args\` input, verbatim (undefined if not provided). Pass arrays/objects as actual JSON values in the tool call, NOT as a JSON-encoded string.
- budget: {total: number|null, spent(): number, remaining(): number} — the turn's token target. \`budget.total\` is null if no target was set. The target is a HARD ceiling: once \`spent()\` reaches \`total\`, further \`agent()\` calls throw.
- workflow(nameOrRef: string | {scriptPath: string}, args?: any): Promise<any> — run a saved workflow inline as a sub-step and return whatever it returns. The child shares this run's concurrency cap, agent counter, abort signal, and token budget. Nesting is one level only.
Subagents are told their final text IS the return value (not a human-facing message), so they return raw data. For structured output, use the schema option — validation happens at the tool-call layer so the model retries on mismatch.
Scripts are plain JavaScript, NOT TypeScript — type annotations (\`: string[]\`), interfaces, and generics fail to parse. The script body runs in an async context — use await directly. Standard JS built-ins (JSON, Math, Array, etc.) are available — EXCEPT \`Date.now()\`/\`Math.random()\`/argless \`new Date()\`, which throw (they would break resume); pass timestamps in via \`args\`, stamp results after the workflow returns, and for randomness vary the agent prompt/label by index. No filesystem or Node.js API access.
DEFAULT TO pipeline(). Only reach for a barrier (parallel between stages) when you genuinely need ALL prior-stage results together.
A barrier is correct ONLY when stage N needs cross-item context from all of stage N-1:
- Dedup/merge across the full result set before expensive downstream work
- Early-exit if the total count is zero ("0 bugs found → skip verification entirely")
- Stage N's prompt references "the other findings" for comparison
A barrier is NOT justified by:
- "I need to flatten/map/filter first" — do it inside a pipeline stage: pipeline(items, stageA, r => transform([r]).flat(), stageB)
- "The stages are conceptually separate" — that's what pipeline() models. Separate stages ≠ synchronized stages.
- "It's cleaner code" — barrier latency is real.
The canonical multi-stage pattern — pipeline by default, each dimension verifies as soon as its review completes:
export const meta = {
name: 'review-changes',
description: 'Review changed files across dimensions, verify each finding',
phases: [{ title: 'Review' }, { title: 'Verify' }],
}
const DIMENSIONS = [{key: 'bugs', prompt: '...'}, {key: 'perf', prompt: '...'}]
const results = await pipeline(
DIMENSIONS,
d => agent(d.prompt, {label: \`review:\${d.key}\`, phase: 'Review', schema: FINDINGS_SCHEMA}),
review => parallel(review.findings.map(f => () =>
agent(\`Adversarially verify: \${f.title}\`, {label: \`verify:\${f.file}\`, phase: 'Verify', schema: VERDICT_SCHEMA})
.then(v => ({...f, verdict: v}))
))
)
const confirmed = results.flat().filter(Boolean).filter(f => f.verdict?.isReal)
return { confirmed }
When a barrier IS correct — dedup across all findings before expensive verification:
const all = await parallel(DIMENSIONS.map(d => () => agent(d.prompt, {schema: FINDINGS_SCHEMA})))
const deduped = dedupeByFileAndLine(all.filter(Boolean).flatMap(r => r.findings))
const verified = await parallel(deduped.map(f => () => agent(verifyPrompt(f), {schema: VERDICT_SCHEMA})))
Loop-until-count pattern — accumulate to a target:
const bugs = []
while (bugs.length < 10) {
const result = await agent("Find bugs in this codebase.", {schema: BUGS_SCHEMA})
bugs.push(...result.bugs)
log(\`\${bugs.length}/10 found\`)
}
Loop-until-budget pattern — guard on budget.total: with no target set, remaining() is Infinity and the loop would run straight to the agent cap.
while (budget.total && budget.remaining() > 50_000) { ... }
Composing patterns — exhaustive review (find → dedup vs seen → diverse-lens panel → loop-until-dry):
const seen = new Set(), confirmed = []
let dry = 0
while (dry < 2) {
const found = (await parallel(FINDERS.map(f => () =>
agent(f.prompt, {phase: 'Find', schema: BUGS})))).filter(Boolean).flatMap(r => r.bugs)
const fresh = found.filter(b => !seen.has(key(b)))
if (!fresh.length) { dry++; continue }
dry = 0; fresh.forEach(b => seen.add(key(b)))
const judged = await parallel(fresh.map(b => () =>
parallel(['correctness','security','repro'].map(lens => () =>
agent(\`Judge "\${b.desc}" via the \${lens} lens — real?\`, {phase: 'Verify', schema: VERDICT})))
.then(vs => ({ b, real: vs.filter(Boolean).filter(v => v.real).length >= 2 }))))
confirmed.push(...judged.filter(v => v.real).map(v => v.b))
}
return confirmed
// dedup vs \`seen\`, NOT \`confirmed\` — else judge-rejected findings reappear every round and it never converges.
Quality patterns — common shapes; pick by task and compose freely:
- Adversarial verify: spawn N independent skeptics per finding, each prompted to REFUTE. Kill if ≥majority refute.
- Perspective-diverse verify: give each verifier a distinct lens (correctness, security, perf, does-it-reproduce) instead of N identical refuters.
- Judge panel: generate N independent attempts from different angles, score with parallel judges, synthesize from the winner while grafting the best ideas from runners-up.
- Loop-until-dry: for unknown-size discovery, keep spawning finders until K consecutive rounds return nothing new.
- Multi-modal sweep: parallel agents each searching a different way (by-container, by-content, by-entity, by-time).
- Completeness critic: a final agent that asks "what's missing — modality not run, claim unverified, source unread?"
- No silent caps: if a workflow bounds coverage (top-N, no-retry, sampling), \`log()\` what was dropped.
Scale to what the user asked for. "find any bugs" → a few finders, single-vote verify. "thoroughly audit this" or "be comprehensive" → larger finder pool, 3–5 vote adversarial pass, synthesis stage.
Use this tool for multi-step orchestration where control flow should be deterministic (loops, conditionals, fan-out) rather than model-driven.
## Resume
The tool result includes a runId. To resume after a pause, kill, or script edit, relaunch with Workflow({scriptPath, resumeFromRunId}) — the longest unchanged prefix of agent() calls returns cached results instantly; the first edited/new call and everything after it runs live. Same script + same args → 100% cache hit. Before diagnosing why a completed workflow returned an empty or unexpected result, Read <transcriptDir>/journal.jsonl — it records each agent's actual return value.
Concurrent agent() calls are capped at ${getWorkflowConcurrency()} per workflow — excess calls queue and run as slots free up. Total agent count across a workflow's lifetime is capped at ${WORKFLOW_MAX_AGENTS}. A single parallel()/pipeline() call accepts at most ${WORKFLOW_MAX_FANOUT} items.
${describeWorkflowSizeGuideline()}`
}
+100
View File
@@ -0,0 +1,100 @@
import { readFile } from 'fs/promises'
import type { CanUseToolFn } from '../../hooks/useCanUseTool.js'
import type { ToolUseContext } from '../../Tool.js'
import { findWorkflowByName } from '../../utils/workflows/discovery.js'
import type { WorkflowSharedCounters } from '../../utils/workflows/harness.js'
import type { WorkflowJournal } from '../../utils/workflows/journal.js'
import {
executeWorkflowScript,
prepareWorkflowScript,
} from '../../utils/workflows/runtime.js'
import type {
WorkflowProgressEvent,
WorkflowTokenBudget,
} from '../../utils/workflows/types.js'
export type NestedWorkflowDeps = {
toolUseContext: ToolUseContext
canUseTool: CanUseToolFn
runId: string
tokenBudget?: WorkflowTokenBudget
journal?: WorkflowJournal
shared: WorkflowSharedCounters
onProgress: (event: WorkflowProgressEvent) => void
onAgentController: (
agentKey: string,
controller: AbortController | undefined,
) => void
}
/**
* Build the `workflow(nameOrRef, args)` global.
*
* A child runs inside the parent's run: same run id, same journal, same
* concurrency pool and agent-index sequence (so its progress rows sit
* alongside the parent's instead of overwriting them), and the same token
* budget. Nesting is one level — a child's own `workflow()` throws, which is
* what keeps the agent cap meaningful.
*/
export function createNestedWorkflowRunner(
deps: NestedWorkflowDeps,
): (nameOrRef: unknown, args: unknown) => Promise<unknown> {
return async (nameOrRef, args) => {
const script = await resolveNestedScript(nameOrRef)
const prepared = prepareWorkflowScript(script)
if (!prepared.ok) {
throw new Error(`workflow(): ${prepared.error}`)
}
deps.onProgress({
type: 'workflow_log',
message: `workflow(${prepared.meta.name}) started`,
})
const outcome = await executeWorkflowScript({
vmScript: prepared.vmScript,
toolUseContext: deps.toolUseContext,
canUseTool: deps.canUseTool,
runId: deps.runId,
args,
seedPhaseTitles: prepared.meta.phases?.map(phase => phase.title),
tokenBudget: deps.tokenBudget,
journal: deps.journal,
onProgress: deps.onProgress,
onAgentController: deps.onAgentController,
shared: deps.shared,
// One level only: the child gets no `workflow()` of its own.
runNestedWorkflow: undefined,
})
if (outcome.error) {
throw new Error(`workflow(${prepared.meta.name}): ${outcome.error}`)
}
return outcome.result
}
}
async function resolveNestedScript(nameOrRef: unknown): Promise<string> {
if (typeof nameOrRef === 'string') {
const workflow = await findWorkflowByName(nameOrRef)
if (!workflow) throw new Error(`workflow(): unknown workflow '${nameOrRef}'`)
return workflow.script
}
if (nameOrRef !== null && typeof nameOrRef === 'object') {
const scriptPath = (nameOrRef as { scriptPath?: unknown }).scriptPath
if (typeof scriptPath === 'string') {
try {
return await readFile(scriptPath, 'utf8')
} catch (error) {
throw new Error(
`workflow(): failed to read ${scriptPath}: ${
error instanceof Error ? error.message : String(error)
}`,
)
}
}
}
throw new Error(
"workflow() expects a saved workflow name or { scriptPath: '...' }",
)
}
+2
View File
@@ -63,6 +63,8 @@ export type LoadedPlugin = {
skillsPaths?: string[] // Additional skill paths from manifest
outputStylesPath?: string
outputStylesPaths?: string[] // Additional output style paths from manifest
workflowsPath?: string
workflowsPaths?: string[] // Additional dynamic-workflow paths from manifest
hooksConfig?: HooksSettings
mcpServers?: Record<string, McpServerConfig>
lspServers?: Record<string, LspServerConfig>
+132
View File
@@ -200,6 +200,11 @@ const sessionTranscriptModule = feature('KAIROS')
: null
/* eslint-enable @typescript-eslint/no-require-imports */
import { hasUltrathinkKeyword, isUltrathinkEnabled } from './thinking.js'
import {
getWorkflowSizeGuideline,
isWorkflowKeywordTriggerEnabled,
} from './workflows/enabled.js'
import { hasWorkflowKeyword } from './workflows/keyword.js'
import {
tokenCountFromLastAPIResponse,
tokenCountWithEstimation,
@@ -674,6 +679,21 @@ export type Attachment =
type: 'ultrathink_effort'
level: 'high'
}
| {
type: 'workflow_keyword_request'
}
| {
type: 'ultra_effort_enter'
/** `full` on the first turn of a session; `short` on every turn after. */
reminderType: 'full' | 'short'
}
| {
type: 'ultra_effort_exit'
}
| {
type: 'workflow_size_guideline_change'
size: string
}
| {
type: 'deferred_tools_delta'
addedNames: string[]
@@ -825,6 +845,25 @@ export async function getAttachments(
maybe('ultrathink_effort', () =>
Promise.resolve(getUltrathinkEffortAttachment(input)),
),
maybe('workflow_keyword_request', () =>
Promise.resolve(
getWorkflowKeywordAttachment(
input,
toolUseContext.getAppState().suppressWorkflowKeyword === true,
),
),
),
maybe('ultra_effort_enter', () =>
Promise.resolve(
getUltracodeEffortAttachments(
messages,
toolUseContext.getAppState().ultracode === true,
),
),
),
maybe('workflow_size_guideline_change', () =>
Promise.resolve(getWorkflowSizeGuidelineAttachment(messages)),
),
maybe('deferred_tools_delta', () =>
Promise.resolve(
getDeferredToolsDeltaAttachment(
@@ -1431,6 +1470,99 @@ export function getDateChangeAttachments(
return [{ type: 'date_change', newDate: currentDate }]
}
/**
* Opt this turn into orchestration when the user typed `ultracode`.
*
* Only a prompt the human actually typed counts — a webhook payload or a
* relayed PR comment containing the word must not spend a hundred agents.
*/
function getWorkflowKeywordAttachment(
input: string | null,
suppressed: boolean,
): Attachment[] {
if (suppressed) return []
if (!input || !isWorkflowKeywordTriggerEnabled()) return []
if (!hasWorkflowKeyword(input)) return []
logEvent('tengu_workflow_keyword', {})
return [{ type: 'workflow_keyword_request' }]
}
/**
* Announce ultracode turning on or off, once per transition.
*
* The model needs a standing instruction while it is on, but repeating the
* full paragraph every turn is pure token cost — so the transition gets the
* full text and later turns get a one-line reminder.
*/
function getUltracodeEffortAttachments(
messages: Message[] | undefined,
ultracodeActive: boolean,
): Attachment[] {
let lastState: 'enter' | 'exit' | 'none' = 'none'
for (let i = (messages?.length ?? 0) - 1; i >= 0; i--) {
const message = messages?.[i]
if (message?.type !== 'attachment') continue
if (message.attachment.type === 'ultra_effort_enter') {
lastState = 'enter'
break
}
if (message.attachment.type === 'ultra_effort_exit') {
lastState = 'exit'
break
}
}
if (ultracodeActive) {
return [
{
type: 'ultra_effort_enter',
reminderType: lastState === 'enter' ? 'short' : 'full',
},
]
}
return lastState === 'enter' ? [{ type: 'ultra_effort_exit' }] : []
}
/**
* Tell the model when the size guideline changed mid-session.
*
* The guideline is baked into the Workflow tool prompt, which the model may
* have cached from an earlier turn — without this, changing it in `/config`
* has no visible effect until the next session.
*/
function getWorkflowSizeGuidelineAttachment(
messages: Message[] | undefined,
): Attachment[] {
const current = getWorkflowSizeGuideline()
for (let i = (messages?.length ?? 0) - 1; i >= 0; i--) {
const message = messages?.[i]
if (message?.type !== 'attachment') continue
if (message.attachment.type !== 'workflow_size_guideline_change') continue
return message.attachment.size === current
? []
: [{ type: 'workflow_size_guideline_change', size: current }]
}
// Nothing announced yet: the tool prompt already carries the guideline on
// the first turn, so only a later change is worth a reminder.
return []
}
/**
* Test seam for the two workflow attachment producers.
*
* Both are pure given their inputs, but they sit behind `getAttachments()`,
* which needs a full ToolUseContext and a live session. Exposing them keeps
* the tests on the real functions instead of a re-implementation.
*/
export const getAttachmentsForTesting = {
workflowKeyword: (input: string | null, opts: { suppressed: boolean }) =>
getWorkflowKeywordAttachment(input, opts.suppressed),
ultracodeEffort: (messages: Message[] | undefined, ultracodeActive: boolean) =>
getUltracodeEffortAttachments(messages, ultracodeActive),
workflowSizeGuideline: (messages: Message[] | undefined) =>
getWorkflowSizeGuidelineAttachment(messages),
}
function getUltrathinkEffortAttachment(input: string | null): Attachment[] {
if (!isUltrathinkEnabled() || !input || !hasUltrathinkKeyword(input)) {
return []
+11
View File
@@ -383,6 +383,15 @@ export type GlobalConfig = {
// Terminal progress bar configuration (OSC 9;4)
terminalProgressBarEnabled: boolean
// How many subagents Claude should aim for when writing a dynamic workflow.
// Set from /config; a settings-file value overrides it and hides the row.
workflowSizeGuideline?: 'unrestricted' | 'small' | 'medium' | 'large'
// Recorded the first time the user approves a workflow in auto permission
// mode. Auto mode is already "stop asking me about routine things", so the
// launch prompt asks once and then stays out of the way.
hasAcceptedWorkflowsInAutoMode?: boolean
// Terminal tab status indicator (OSC 21337). When on, emits a colored
// dot + status text to the tab sidebar and drops the spinner prefix
// from the title (the dot makes it redundant).
@@ -647,6 +656,8 @@ export const GLOBAL_CONFIG_KEYS = [
'autoInstallIdeExtension',
'fileCheckpointingEnabled',
'terminalProgressBarEnabled',
'workflowSizeGuideline',
'hasAcceptedWorkflowsInAutoMode',
'showStatusInTerminalTab',
'taskCompleteNotifEnabled',
'inputNeededNotifEnabled',
+44
View File
@@ -1,3 +1,10 @@
import { describeWorkflowSizeGuideline } from './workflows/enabled.js'
import {
ULTRACODE_ENTER_REMINDER,
ULTRACODE_EXIT_REMINDER,
ULTRACODE_STILL_ON_REMINDER,
WORKFLOW_KEYWORD_REMINDER,
} from './workflows/ultracode.js'
import { feature } from 'bun:bundle'
import type { BetaUsage as Usage } from '@anthropic-ai/sdk/resources/beta/messages/messages.mjs'
import type {
@@ -4319,6 +4326,43 @@ You have exited auto mode. The user may now want to interact more directly. You
}),
])
}
case 'workflow_keyword_request': {
return wrapMessagesInSystemReminder([
createUserMessage({
content: WORKFLOW_KEYWORD_REMINDER,
isMeta: true,
}),
])
}
case 'ultra_effort_enter': {
return wrapMessagesInSystemReminder([
createUserMessage({
content:
attachment.reminderType === 'full'
? ULTRACODE_ENTER_REMINDER
: ULTRACODE_STILL_ON_REMINDER,
isMeta: true,
}),
])
}
case 'ultra_effort_exit': {
return wrapMessagesInSystemReminder([
createUserMessage({
content: ULTRACODE_EXIT_REMINDER,
isMeta: true,
}),
])
}
case 'workflow_size_guideline_change': {
return wrapMessagesInSystemReminder([
createUserMessage({
content: describeWorkflowSizeGuideline(
attachment.size as Parameters<typeof describeWorkflowSizeGuideline>[0],
),
isMeta: true,
}),
])
}
case 'deferred_tools_delta': {
const parts: string[] = []
if (attachment.addedLines.length > 0) {
+3 -5
View File
@@ -40,11 +40,9 @@ const VERIFY_PLAN_EXECUTION_TOOL_NAME =
require('../../tools/VerifyPlanExecutionTool/constants.js') as typeof import('../../tools/VerifyPlanExecutionTool/constants.js')
).VERIFY_PLAN_EXECUTION_TOOL_NAME
: null
const WORKFLOW_TOOL_NAME = feature('WORKFLOW_SCRIPTS')
? (
require('../../tools/WorkflowTool/constants.js') as typeof import('../../tools/WorkflowTool/constants.js')
).WORKFLOW_TOOL_NAME
: null
const WORKFLOW_TOOL_NAME = (
require('../../tools/WorkflowTool/constants.js') as typeof import('../../tools/WorkflowTool/constants.js')
).WORKFLOW_TOOL_NAME
/* eslint-enable @typescript-eslint/no-require-imports */
/**
+25
View File
@@ -1591,6 +1591,31 @@ export async function createPluginFromPath(
plugin.outputStylesPath = outputStylesPath
}
// Step 4c-2: Register dynamic workflows shipped by the plugin. Namespaced by
// plugin name at command-build time, so two plugins can both ship `review`.
const workflowsPath = join(pluginPath, 'workflows')
if (await pathExists(workflowsPath)) {
plugin.workflowsPath = workflowsPath
}
if (manifest.workflows) {
const declared = Array.isArray(manifest.workflows)
? manifest.workflows
: [manifest.workflows]
const validPaths = await validatePluginPaths(
declared,
pluginPath,
manifest.name,
source,
'workflows',
'Workflow',
'specified in manifest but',
errors,
)
if (validPaths.length > 0) {
plugin.workflowsPaths = validPaths
}
}
// Step 4e: Process additional output style paths from manifest
if (manifest.outputStyles) {
const outputStylePaths = Array.isArray(manifest.outputStyles)
+24
View File
@@ -523,6 +523,29 @@ const PluginManifestOutputStylesSchema = lazySchema(() =>
}),
)
/**
* Schema for additional dynamic-workflow paths in a plugin manifest.
*
* Plugins ship workflows in `workflows/` at the plugin root; this field points
* at extra directories or files, matching how commands/agents/skills work.
*/
const PluginManifestWorkflowsSchema = lazySchema(() =>
z.object({
workflows: z.union([
RelativePath().describe(
'Path to additional workflows directory or file (in addition to those in the workflows/ directory, if it exists), relative to the plugin root',
),
z
.array(
RelativePath().describe(
'Path to additional workflows directory or file, relative to the plugin root',
),
)
.describe('List of paths to additional workflows directories or files'),
]),
}),
)
// Helper validators for LSP config
const nonEmptyString = lazySchema(() => z.string().min(1))
const fileExtension = lazySchema(() =>
@@ -889,6 +912,7 @@ export const PluginManifestSchema = lazySchema(() =>
...PluginManifestAgentsSchema().partial().shape,
...PluginManifestSkillsSchema().partial().shape,
...PluginManifestOutputStylesSchema().partial().shape,
...PluginManifestWorkflowsSchema().partial().shape,
...PluginManifestChannelsSchema().partial().shape,
...PluginManifestMcpServerSchema().partial().shape,
...PluginManifestLspServerSchema().partial().shape,
+17
View File
@@ -278,6 +278,23 @@ export type AgentMetadata = {
* agent's activity and completion under the wrong card. Optional —
* older metadata files lack this field. */
toolUseId?: string
/**
* Which workflow run and phase this agent belongs to.
*
* A workflow's shape — phases, and which agents ran in each — exists only in
* the live progress stream, so reopening a finished session had nothing to
* rebuild it from. This sidecar is written before the agent starts and
* outlives the process, which makes it the record. Absent for ordinary
* subagents and for workflow runs from before this was added.
*/
workflow?: {
runId: string
name: string
phaseIndex: number
phaseTitle?: string
/** The runtime's sequential agent number within the run. */
agentIndex: number
}
}
/**
+31
View File
@@ -460,6 +460,37 @@ export const SettingsSchema = lazySchema(() =>
.boolean()
.optional()
.describe('Disable all hooks and statusLine execution'),
// Dynamic workflows: the Workflow tool, /workflows, and saved workflow
// commands. Honored in managed settings as an org-wide kill switch.
disableWorkflows: z
.boolean()
.optional()
.describe(
'Disable dynamic workflows: the Workflow tool, bundled workflow commands, ' +
'and saved workflows in .claude/workflows.',
),
enableWorkflows: z
.boolean()
.optional()
.describe(
'Explicitly enable dynamic workflows. Only consulted when they are not ' +
'disabled; `disableWorkflows` and CLAUDE_CODE_DISABLE_WORKFLOWS win.',
),
workflowKeywordTriggerEnabled: z
.boolean()
.optional()
.describe(
'Enable the "ultracode" keyword trigger: including the keyword in a prompt ' +
'opts that turn into the Workflow tool. Set to false to disable the trigger. Default: true.',
),
workflowSizeGuideline: z
.enum(['unrestricted', 'small', 'medium', 'large'])
.optional()
.describe(
'How many subagents Claude should aim for when writing a dynamic workflow. ' +
'small < 5, medium < 15, large < 50, unrestricted lets Claude size it to the task. ' +
'Advice to the model, not a runtime cap.',
),
// Opt out of the cross-client `.agents/skills` convention (agentskills.io),
// which is scanned alongside `.claude/skills` by default.
disableAgentSkillsDirectory: z
+2
View File
@@ -311,6 +311,8 @@ function getStatusText(status: TaskStatus): string {
return 'was stopped'
case 'running':
return 'is running'
case 'paused':
return 'is paused'
case 'pending':
return 'is pending'
}
+30
View File
@@ -0,0 +1,30 @@
import { getGlobalConfig, saveGlobalConfig } from '../config.js'
/**
* Whether the user has already approved running workflows in auto mode.
*
* Auto mode means "stop asking me about routine things". Prompting on every
* workflow launch there would defeat the mode, so the launch prompt appears
* once and the answer is remembered for the machine.
*/
export function hasAcceptedWorkflowsInAutoMode(): boolean {
try {
return getGlobalConfig().hasAcceptedWorkflowsInAutoMode === true
} catch {
// Config access can be closed early in the process lifetime; a missing
// consent record just means we ask.
return false
}
}
export function recordWorkflowAutoModeConsent(): void {
try {
if (getGlobalConfig().hasAcceptedWorkflowsInAutoMode === true) return
saveGlobalConfig(config => ({
...config,
hasAcceptedWorkflowsInAutoMode: true,
}))
} catch {
// Failing to persist consent only costs one extra prompt next time.
}
}
+172
View File
@@ -0,0 +1,172 @@
/**
* `/deep-research` — the bundled workflow.
*
* Investigates one question across many sources: fan out searches on distinct
* angles, deep-read what they surface, then have independent verifiers vote on
* every claim before it reaches the report. The voting stage is the point —
* a single research pass reports whatever the first plausible page said.
*/
export const DEEP_RESEARCH_SCRIPT = `export const meta = {
name: 'deep-research',
description: 'Research a question across many sources and return a cited, cross-checked report',
whenToUse: 'Use for questions that need several independent sources weighed against each other.',
phases: [
{ title: 'Survey', detail: 'search the question from several angles' },
{ title: 'Read', detail: 'deep-read the most promising sources' },
{ title: 'Verify', detail: 'independent verifiers vote on each claim' },
{ title: 'Report', detail: 'synthesize a cited report' },
],
}
const question = typeof args === 'string' ? args : (args && args.question) || ''
if (!question) return { error: 'deep-research needs a question. Run /deep-research <question>.' }
const ANGLE_SCHEMA = {
type: 'object',
required: ['angles'],
properties: {
angles: {
type: 'array',
items: {
type: 'object',
required: ['angle', 'query'],
properties: { angle: { type: 'string' }, query: { type: 'string' } },
},
},
},
}
const SOURCE_SCHEMA = {
type: 'object',
required: ['sources'],
properties: {
sources: {
type: 'array',
items: {
type: 'object',
required: ['url', 'title', 'why'],
properties: { url: { type: 'string' }, title: { type: 'string' }, why: { type: 'string' } },
},
},
},
}
const CLAIM_SCHEMA = {
type: 'object',
required: ['claims'],
properties: {
claims: {
type: 'array',
items: {
type: 'object',
required: ['claim', 'url'],
properties: {
claim: { type: 'string' },
url: { type: 'string' },
quote: { type: 'string' },
},
},
},
},
}
const VERDICT_SCHEMA = {
type: 'object',
required: ['verdict', 'reason'],
properties: {
verdict: { type: 'string', enum: ['supported', 'refuted', 'unverifiable'] },
reason: { type: 'string' },
},
}
phase('Survey')
const plan = await agent(
'Break this research question into 4 distinct search angles that would surface DIFFERENT sources, ' +
'not rephrasings of each other. Question: ' + question,
{ label: 'plan angles', phase: 'Survey', schema: ANGLE_SCHEMA },
)
const angles = (plan && plan.angles ? plan.angles : []).slice(0, 4)
if (angles.length === 0) return { error: 'could not decompose the question into search angles' }
log(angles.length + ' angles: ' + angles.map(a => a.angle).join(', '))
const perAngle = await pipeline(
angles,
angle =>
agent(
'Use WebSearch to find the best sources for this angle on "' + question + '".\\n' +
'Angle: ' + angle.angle + '\\nSuggested query: ' + angle.query + '\\n' +
'Return the 3 most credible, most specific sources. Prefer primary sources over summaries.',
{ label: angle.angle, phase: 'Survey', schema: SOURCE_SCHEMA },
),
found =>
parallel(
(found && found.sources ? found.sources : []).slice(0, 3).map(source => () =>
agent(
'WebFetch ' + source.url + ' and extract every claim in it that bears on: ' + question + '\\n' +
'Quote the sentence each claim comes from. Do not infer beyond the text. ' +
'If the page does not load or does not address the question, return an empty claims array.',
{ label: source.title, phase: 'Read', schema: CLAIM_SCHEMA },
),
),
),
)
const claims = []
const seen = new Set()
for (const batch of perAngle.flat().filter(Boolean)) {
for (const claim of batch.claims || []) {
const key = (claim.claim || '').toLowerCase().replace(/\\s+/g, ' ').trim()
if (!key || seen.has(key)) continue
seen.add(key)
claims.push(claim)
}
}
log(claims.length + ' distinct claims to verify')
if (claims.length === 0) return { question, claims: [], unverified: [], note: 'no claims found' }
phase('Verify')
const judged = await pipeline(claims, (claim, _original, index) =>
parallel(
['does an independent source confirm this', 'does any source contradict this'].map(
(lens, lensIndex) => () =>
agent(
'Verify this claim independently of where it came from.\\n' +
'Claim: ' + claim.claim + '\\nOriginally from: ' + claim.url + '\\n' +
'Lens: ' + lens + '\\n' +
'Search for and read at least one source OTHER than the original. ' +
'Answer "unverifiable" if you cannot reach a source — do not guess, and do not ' +
'treat a failed fetch as a refutation.',
{ label: 'claim ' + (index + 1) + '.' + (lensIndex + 1), phase: 'Verify', schema: VERDICT_SCHEMA },
),
),
).then(votes => ({ claim, votes: votes.filter(Boolean) })),
)
const supported = []
const unverified = []
for (const entry of judged.filter(Boolean)) {
const votes = entry.votes
const refuted = votes.filter(v => v.verdict === 'refuted').length
const confirms = votes.filter(v => v.verdict === 'supported').length
if (refuted > confirms) continue
if (confirms === 0) {
unverified.push({ claim: entry.claim.claim, url: entry.claim.url, reason: (votes[0] && votes[0].reason) || 'no verifier reached a source' })
continue
}
supported.push({ claim: entry.claim.claim, url: entry.claim.url, quote: entry.claim.quote })
}
phase('Report')
const report = await agent(
'Write a cited report answering: ' + question + '\\n\\n' +
'Use ONLY these cross-checked claims, citing the URL after each statement:\\n' +
JSON.stringify(supported, null, 2) +
'\\n\\nThese claims could not be verified — list them at the end under "Unverified", do not treat them as fact:\\n' +
JSON.stringify(unverified, null, 2) +
'\\n\\nBe direct. Lead with the answer. Markdown, no preamble.',
{ label: 'synthesize', phase: 'Report' },
)
return { question, report, supported, unverified }
`
+39
View File
@@ -0,0 +1,39 @@
import { parseWorkflowScript } from '../meta.js'
import type { WorkflowDefinition } from '../types.js'
import { DEEP_RESEARCH_SCRIPT } from './deepResearch.js'
const BUNDLED_SCRIPTS = [DEEP_RESEARCH_SCRIPT]
let cached: WorkflowDefinition[] | null = null
/**
* Workflows that ship with the CLI.
*
* Their metadata is parsed from the same source the runtime executes, so a
* bundled script whose `meta` drifts out of shape fails the parser here rather
* than at run time.
*/
export function getBundledWorkflows(): WorkflowDefinition[] {
if (cached) return cached
const definitions: WorkflowDefinition[] = []
for (const script of BUNDLED_SCRIPTS) {
const parsed = parseWorkflowScript(script)
if ('error' in parsed) {
throw new Error(`Bundled workflow failed to parse: ${parsed.error}`)
}
definitions.push({
source: 'built-in',
name: parsed.meta.name,
description: parsed.meta.description,
whenToUse: parsed.meta.whenToUse,
phases: parsed.meta.phases,
script,
})
}
cached = definitions
return definitions
}
export function resetBundledWorkflowsForTesting(): void {
cached = null
}
+74
View File
@@ -0,0 +1,74 @@
import vm from 'vm'
import {
WORKFLOW_DATE_BANNED_MESSAGE,
WORKFLOW_IMPORT_BANNED_MESSAGE,
WORKFLOW_RANDOM_BANNED_MESSAGE,
} from './constants.js'
export type WorkflowCompileResult =
| { ok: true; vmScript: vm.Script }
| { ok: false; error: string }
/**
* Remove the two sources of nondeterminism that would break resume.
*
* A resumed run replays cached agent results in start order. A script whose
* control flow depends on wall-clock time or randomness would take a different
* branch on replay and silently consume the wrong cache entries, so both are
* made to throw at the first call rather than shimmed into something plausible.
*
* Installed on the context (not prepended to the script) so `globalThis.Date`
* and an aliased `const r = Math.random` are covered too.
*/
export function installDeterminismGuards(context: vm.Context): void {
vm.runInContext(
`(() => {
const dateMessage = ${JSON.stringify(WORKFLOW_DATE_BANNED_MESSAGE)}
const randomMessage = ${JSON.stringify(WORKFLOW_RANDOM_BANNED_MESSAGE)}
const RealDate = Date
globalThis.Date = new Proxy(RealDate, {
construct(target, argv, newTarget) {
if (argv.length === 0) throw new Error(dateMessage)
return Reflect.construct(target, argv, newTarget === undefined ? target : newTarget)
},
get(target, prop, receiver) {
if (prop === 'now') throw new Error(dateMessage)
return Reflect.get(target, prop, receiver)
},
})
Math.random = () => {
throw new Error(randomMessage)
}
})()`,
context,
{ filename: 'workflow-guards.js' },
)
}
/**
* Compile a script body into a `vm.Script` that evaluates to a promise.
*
* The body runs inside an async IIFE so a bare top-level `return` resolves the
* run and `await` works without the script declaring anything. Dynamic
* `import()` is refused at the module-resolution hook rather than by a lint
* pass, so there is no spelling of it that gets through.
*/
export function compileWorkflowScript(
scriptBody: string,
): WorkflowCompileResult {
const source = `(async () => {\n${scriptBody}\n})()`
try {
const vmScript = new vm.Script(source, {
filename: 'workflow.js',
importModuleDynamically: () => {
throw new Error(WORKFLOW_IMPORT_BANNED_MESSAGE)
},
})
return { ok: true, vmScript }
} catch (error) {
return {
ok: false,
error: `SyntaxError: ${error instanceof Error ? error.message : String(error)}`,
}
}
}
+94
View File
@@ -0,0 +1,94 @@
import { cpus } from 'os'
/** Hard cap on `agent()` calls per run — a backstop against runaway loops. */
export const WORKFLOW_MAX_AGENTS = 1000
/** A single parallel()/pipeline() call may not fan out wider than this. */
export const WORKFLOW_MAX_FANOUT = 4096
/** Scripts larger than this are rejected before parsing (512 KiB). */
export const WORKFLOW_SCRIPT_MAX_BYTES = 524_288
/**
* Wall-clock budget for the *synchronous* portion of the script. Awaiting an
* agent does not consume it; a `while (true) {}` with no await does.
*/
export const WORKFLOW_SYNC_TIMEOUT_MS = 30_000
/** An agent that emits no progress for this long is reported as stalled. */
export const WORKFLOW_AGENT_STALL_MS = 180_000
/** Prompt/result previews shown in the progress view are clipped to this. */
export const WORKFLOW_PREVIEW_MAX_CHARS = 400
/** Labels are clipped to this when derived from the prompt. */
export const WORKFLOW_LABEL_MAX_CHARS = 60
/** Upper bound on log lines carried in the final result. */
export const WORKFLOW_MAX_COLLECTED_LOGS = 1000
/** Progress rows retained in task state; logs beyond this are dropped first. */
export const WORKFLOW_MAX_PROGRESS_ROWS = 500
/** Progress events are coalesced on this interval before touching AppState. */
export const WORKFLOW_PROGRESS_BATCH_MS = 16
/** Minimum spacing between task-panel progress emissions. */
export const WORKFLOW_PANEL_EMIT_INTERVAL_MS = 10_000
/** Agent type used for every subagent a workflow script spawns. */
export const WORKFLOW_SUBAGENT_TYPE = 'workflow-subagent'
/** Run ids look like `wf_<8 hex>-<3 hex>`; agent ids append `-<index>`. */
export const WORKFLOW_RUN_ID_PATTERN = /^wf_[a-z0-9-]{6,}$/
/**
* Concurrency ceiling for in-flight agents. Bounded by CPU count so a fan-out
* of 500 items does not spawn 500 model streams on a laptop.
*/
export function getWorkflowConcurrency(
cpuCount: number = cpus().length,
): number {
return Math.min(16, Math.max(2, cpuCount - 2))
}
export const WORKFLOW_DATE_BANNED_MESSAGE =
'Date.now() / new Date() are unavailable in workflow scripts (breaks resume). ' +
'Stamp results after the workflow returns, or pass timestamps via args.'
export const WORKFLOW_RANDOM_BANNED_MESSAGE =
'Math.random() is unavailable in workflow scripts (breaks resume). ' +
'For N independent samples, include the index in the agent label or prompt.'
export const WORKFLOW_IMPORT_BANNED_MESSAGE =
'import() is not available in workflow scripts.'
export const WORKFLOW_AGENT_CAP_MESSAGE =
`Workflow agent() call cap reached (${WORKFLOW_MAX_AGENTS}). This usually means a loop using ` +
'budget.remaining() never terminates because no token budget was set — remaining() returns ' +
'Infinity when budget.total is null. Add a hard iteration cap to the loop, or pass a token budget.'
/** System prompt for a workflow subagent that returns free text. */
export const WORKFLOW_SUBAGENT_PROMPT =
'You are a subagent spawned by a workflow orchestration script. Use the tools available to complete the task.\n' +
'NOTE: You are running inside a workflow script. Your final text response is returned verbatim as a string ' +
'to the calling script — it is your return value, not a message to a human. Output the literal result; do not ' +
'output confirmations like "Done." Be concise — the script will parse your output.'
/** System prompt for a workflow subagent forced through StructuredOutput. */
export function workflowStructuredSubagentPrompt(toolName: string): string {
return (
'You are a subagent spawned by a workflow orchestration script. Use the tools available to complete the task.\n' +
`- After calling ${toolName} successfully, end your turn. No acknowledgment needed.`
)
}
/** Appended to a schema-bearing agent prompt so the model uses the tool. */
export function workflowStructuredOutputNote(toolName: string): string {
return (
`\nNOTE: You are running inside a workflow script. You MUST return your final answer by calling the ${toolName} ` +
"tool exactly once — the tool's input schema defines the required shape. Do your work, then call " +
`${toolName}; do NOT put your answer in a text response (the script reads ONLY the tool call). ` +
`If validation fails, read the error and call ${toolName} again with a corrected shape.`
)
}
+100
View File
@@ -0,0 +1,100 @@
import { afterEach, beforeEach, describe, expect, test } from 'bun:test'
import { mkdir, mkdtemp, rm, writeFile } from 'fs/promises'
import { tmpdir } from 'os'
import { join } from 'path'
import { loadWorkflows } from './discovery.js'
import { getUserWorkflowsDir } from './paths.js'
function script(name: string, description: string): string {
return `export const meta = { name: '${name}', description: '${description}' }\nreturn null\n`
}
describe('loadWorkflows', () => {
let root: string
let configDir: string
let projectDir: string
const originalConfigDir = process.env.CLAUDE_CONFIG_DIR
beforeEach(async () => {
root = await mkdtemp(join(tmpdir(), 'wf-discovery-'))
configDir = join(root, 'config')
projectDir = join(root, 'project')
// Redirect the personal workflows dir away from the developer's ~/.claude.
process.env.CLAUDE_CONFIG_DIR = configDir
await mkdir(join(configDir, 'workflows'), { recursive: true })
await mkdir(join(projectDir, '.claude', 'workflows'), { recursive: true })
// getProjectDirsUpToHome stops at the git root; without one it walks to
// home, so mark this temp dir as a repository.
await mkdir(join(projectDir, '.git'), { recursive: true })
})
afterEach(async () => {
if (originalConfigDir === undefined) delete process.env.CLAUDE_CONFIG_DIR
else process.env.CLAUDE_CONFIG_DIR = originalConfigDir
await rm(root, { recursive: true, force: true })
})
test('resolves the personal workflows dir under CLAUDE_CONFIG_DIR', () => {
expect(getUserWorkflowsDir()).toBe(join(configDir, 'workflows'))
})
test('includes bundled, personal, and project workflows', async () => {
await writeFile(
join(configDir, 'workflows', 'mine.js'),
script('mine', 'Personal one'),
)
await writeFile(
join(projectDir, '.claude', 'workflows', 'ours.js'),
script('ours', 'Project one'),
)
const found = await loadWorkflows(projectDir)
const byName = new Map(found.map(entry => [entry.name, entry]))
expect(byName.get('deep-research')?.source).toBe('built-in')
expect(byName.get('mine')?.source).toBe('userSettings')
expect(byName.get('ours')?.source).toBe('projectSettings')
})
test('a project workflow shadows a personal one with the same name', async () => {
await writeFile(
join(configDir, 'workflows', 'review.js'),
script('review', 'Personal review'),
)
await writeFile(
join(projectDir, '.claude', 'workflows', 'review.js'),
script('review', 'Project review'),
)
const found = await loadWorkflows(projectDir)
const review = found.filter(entry => entry.name === 'review')
expect(review).toHaveLength(1)
expect(review[0]?.description).toBe('Project review')
expect(review[0]?.source).toBe('projectSettings')
})
test('skips files that are not .js or whose meta is invalid', async () => {
await writeFile(
join(configDir, 'workflows', 'good.js'),
script('good', 'Fine'),
)
await writeFile(
join(configDir, 'workflows', 'broken.js'),
'const x = 1\nexport const meta = { name: "late", description: "b" }\n',
)
await writeFile(
join(configDir, 'workflows', 'ignored.ts'),
script('ignored', 'Wrong extension'),
)
const names = (await loadWorkflows(projectDir)).map(entry => entry.name)
expect(names).toContain('good')
expect(names).not.toContain('late')
expect(names).not.toContain('ignored')
})
test('a missing workflows directory is not an error', async () => {
await rm(join(configDir, 'workflows'), { recursive: true, force: true })
const names = (await loadWorkflows(projectDir)).map(entry => entry.name)
expect(names).toEqual(['deep-research'])
})
})
+127
View File
@@ -0,0 +1,127 @@
import { readdir, readFile, stat } from 'fs/promises'
import { join } from 'path'
import { getOriginalCwd } from '../../bootstrap/state.js'
import { logForDebugging } from '../debug.js'
import { getProjectDirsUpToHome } from '../markdownConfigLoader.js'
import { getBundledWorkflows } from './bundled/index.js'
import { loadPluginWorkflows } from './pluginWorkflows.js'
import { WORKFLOW_SCRIPT_MAX_BYTES } from './constants.js'
import { parseWorkflowScript } from './meta.js'
import { getUserWorkflowsDir } from './paths.js'
import type { WorkflowDefinition, WorkflowSource } from './types.js'
/**
* Every workflow runnable as a `/` command, most specific definition winning.
*
* Precedence, lowest to highest: bundled → plugin → personal
* (`~/.claude/workflows`) → project, and within project the directory closest
* to the working directory.
* A project workflow shadowing a personal one is deliberate — it is how a repo
* pins the version of a review its contributors run.
*/
export async function loadWorkflows(
cwd: string = getOriginalCwd(),
): Promise<WorkflowDefinition[]> {
const projectDirs = safeProjectDirs(cwd)
const [plugins, personal, ...projectResults] = await Promise.all([
loadPluginWorkflows(),
loadWorkflowsFromDir(getUserWorkflowsDir(), 'userSettings'),
...projectDirs.map(dir => loadWorkflowsFromDir(dir, 'projectSettings')),
])
const byName = new Map<string, WorkflowDefinition>()
for (const workflow of getBundledWorkflows()) byName.set(workflow.name, workflow)
// Plugin workflows are namespaced `plugin:name`, so they never collide with
// a personal or project workflow and never need to lose a precedence fight.
for (const workflow of plugins) byName.set(workflow.name, workflow)
for (const workflow of personal ?? []) byName.set(workflow.name, workflow)
// projectDirs is ordered most-specific first, so apply in reverse and let
// the closest directory overwrite the ones above it.
for (let i = projectResults.length - 1; i >= 0; i--) {
for (const workflow of projectResults[i] ?? []) {
byName.set(workflow.name, workflow)
}
}
return [...byName.values()].sort((a, b) => a.name.localeCompare(b.name))
}
export async function findWorkflowByName(
name: string,
cwd?: string,
): Promise<WorkflowDefinition | undefined> {
return (await loadWorkflows(cwd)).find(workflow => workflow.name === name)
}
/**
* Read every `.js` workflow in one directory.
*
* Files are validated by parsing their `meta` literal, so a malformed script
* is skipped with a log line rather than breaking `/` autocomplete for the
* rest of the directory. Nothing here executes the script.
*/
export async function loadWorkflowsFromDir(
dir: string,
source: WorkflowSource,
namePrefix?: string,
): Promise<WorkflowDefinition[]> {
let entries: Awaited<ReturnType<typeof readdir>>
try {
entries = await readdir(dir, { withFileTypes: true })
} catch {
return []
}
const loaded = await Promise.all(
entries.map(async entry => {
if (!entry.isFile() && !entry.isSymbolicLink()) return null
if (!entry.name.endsWith('.js')) return null
const filePath = join(dir, entry.name)
try {
const stats = await stat(filePath)
if (stats.size > WORKFLOW_SCRIPT_MAX_BYTES) {
logForDebugging(
`Workflow ${filePath} exceeds ${WORKFLOW_SCRIPT_MAX_BYTES} bytes — skipping`,
)
return null
}
const script = await readFile(filePath, 'utf8')
const parsed = parseWorkflowScript(script)
if ('error' in parsed) {
logForDebugging(
`Workflow ${filePath} has invalid meta: ${parsed.error} — skipping`,
)
return null
}
return {
source,
name: namePrefix ? `${namePrefix}:${parsed.meta.name}` : parsed.meta.name,
description: parsed.meta.description,
whenToUse: parsed.meta.whenToUse,
phases: parsed.meta.phases,
script,
filePath,
} satisfies WorkflowDefinition
} catch (error) {
logForDebugging(
`Workflow ${filePath} could not be read: ${error instanceof Error ? error.message : String(error)}`,
)
return null
}
}),
)
return loaded.filter((entry): entry is WorkflowDefinition => entry !== null)
}
function safeProjectDirs(cwd: string): string[] {
try {
return getProjectDirsUpToHome('workflows', cwd)
} catch (error) {
logForDebugging(
`loadWorkflows: project-dir walk failed: ${error instanceof Error ? error.message : String(error)}`,
)
return []
}
}
+209
View File
@@ -0,0 +1,209 @@
import { afterEach, beforeEach, describe, expect, test } from 'bun:test'
import { mkdtemp, rm, writeFile } from 'fs/promises'
import { mkdirSync } from 'fs'
import { tmpdir } from 'os'
import { join } from 'path'
import { getIsInteractive, setIsInteractive } from '../../bootstrap/state.js'
import { resetSettingsCache } from '../settings/settingsCache.js'
import {
areWorkflowsEnabled,
describeWorkflowSizeGuideline,
getLargeWorkflowWarning,
getWorkflowsDisabledReason,
getWorkflowSizeGuideline,
isWorkflowKeywordTriggerEnabled,
} from './enabled.js'
let home: string
let configDir: string
const original = {
configDir: process.env.CLAUDE_CONFIG_DIR,
disable: process.env.CLAUDE_CODE_DISABLE_WORKFLOWS,
warnAgents: process.env.CLAUDE_CODE_WORKFLOW_SIZE_WARNING_AGENTS,
warnTokens: process.env.CLAUDE_CODE_WORKFLOW_SIZE_WARNING_TOKENS,
}
async function writeSettings(value: Record<string, unknown>): Promise<void> {
await writeFile(join(configDir, 'settings.json'), JSON.stringify(value), 'utf8')
resetSettingsCache()
}
describe('workflow gates', () => {
beforeEach(async () => {
home = await mkdtemp(join(tmpdir(), 'wf-enabled-'))
configDir = join(home, 'claude')
mkdirSync(configDir, { recursive: true })
process.env.CLAUDE_CONFIG_DIR = configDir
delete process.env.CLAUDE_CODE_DISABLE_WORKFLOWS
delete process.env.CLAUDE_CODE_WORKFLOW_SIZE_WARNING_AGENTS
delete process.env.CLAUDE_CODE_WORKFLOW_SIZE_WARNING_TOKENS
resetSettingsCache()
})
afterEach(async () => {
for (const [key, value] of [
['CLAUDE_CONFIG_DIR', original.configDir],
['CLAUDE_CODE_DISABLE_WORKFLOWS', original.disable],
['CLAUDE_CODE_WORKFLOW_SIZE_WARNING_AGENTS', original.warnAgents],
['CLAUDE_CODE_WORKFLOW_SIZE_WARNING_TOKENS', original.warnTokens],
] as const) {
if (value === undefined) delete process.env[key]
else process.env[key] = value
}
resetSettingsCache()
await rm(home, { recursive: true, force: true })
})
test('enabled by default', async () => {
await writeSettings({})
expect(areWorkflowsEnabled()).toBe(true)
expect(getWorkflowsDisabledReason()).toBeNull()
})
test('disableWorkflows and enableWorkflows:false both turn it off', async () => {
await writeSettings({ disableWorkflows: true })
expect(getWorkflowsDisabledReason()).toBe('settings')
await writeSettings({ enableWorkflows: false })
expect(getWorkflowsDisabledReason()).toBe('settings')
await writeSettings({ enableWorkflows: true })
expect(getWorkflowsDisabledReason()).toBeNull()
})
test('the env var wins over settings and is reported separately', async () => {
await writeSettings({ enableWorkflows: true })
process.env.CLAUDE_CODE_DISABLE_WORKFLOWS = '1'
expect(getWorkflowsDisabledReason()).toBe('env')
})
test('the keyword trigger is on by default and off when set false', async () => {
const wasInteractive = getIsInteractive()
setIsInteractive(true)
try {
await writeSettings({})
expect(isWorkflowKeywordTriggerEnabled()).toBe(true)
await writeSettings({ workflowKeywordTriggerEnabled: false })
expect(isWorkflowKeywordTriggerEnabled()).toBe(false)
} finally {
setIsInteractive(wasInteractive)
}
})
test('a non-interactive session never honours the keyword', async () => {
await writeSettings({ workflowKeywordTriggerEnabled: true })
const wasInteractive = getIsInteractive()
setIsInteractive(false)
try {
expect(isWorkflowKeywordTriggerEnabled()).toBe(false)
} finally {
setIsInteractive(wasInteractive)
}
})
test('an interactive session honours it', async () => {
await writeSettings({ workflowKeywordTriggerEnabled: true })
const wasInteractive = getIsInteractive()
setIsInteractive(true)
try {
expect(isWorkflowKeywordTriggerEnabled()).toBe(true)
} finally {
setIsInteractive(wasInteractive)
}
})
test('disabling workflows also disables the keyword', async () => {
const wasInteractive = getIsInteractive()
setIsInteractive(true)
try {
await writeSettings({
disableWorkflows: true,
workflowKeywordTriggerEnabled: true,
})
expect(isWorkflowKeywordTriggerEnabled()).toBe(false)
} finally {
setIsInteractive(wasInteractive)
}
})
test('a settings-file guideline wins and shows in the prompt', async () => {
await writeSettings({ workflowSizeGuideline: 'small' })
expect(getWorkflowSizeGuideline()).toBe('small')
expect(describeWorkflowSizeGuideline()).toContain('under 5 agents')
})
test('the default guideline is medium', async () => {
await writeSettings({})
expect(getWorkflowSizeGuideline()).toBe('medium')
})
})
describe('getLargeWorkflowWarning', () => {
beforeEach(async () => {
home = await mkdtemp(join(tmpdir(), 'wf-warn-'))
configDir = join(home, 'claude')
mkdirSync(configDir, { recursive: true })
process.env.CLAUDE_CONFIG_DIR = configDir
delete process.env.CLAUDE_CODE_WORKFLOW_SIZE_WARNING_AGENTS
delete process.env.CLAUDE_CODE_WORKFLOW_SIZE_WARNING_TOKENS
await writeSettings({})
})
afterEach(async () => {
if (original.configDir === undefined) delete process.env.CLAUDE_CONFIG_DIR
else process.env.CLAUDE_CONFIG_DIR = original.configDir
delete process.env.CLAUDE_CODE_WORKFLOW_SIZE_WARNING_AGENTS
resetSettingsCache()
await rm(home, { recursive: true, force: true })
})
const base = { startedAgents: 4, totalTokens: 40_000, ultracodeActive: false }
test('stays quiet for an ordinary run', () => {
expect(
getLargeWorkflowWarning({ ...base, scheduledAgents: 6 }),
).toBeUndefined()
})
test('fires past the default 25-agent threshold', () => {
const warning = getLargeWorkflowWarning({ ...base, scheduledAgents: 26 })
expect(warning?.axis).toBe('agents')
expect(warning?.agentCap).toBe(25)
})
test('projects tokens from the agents that have run so far', () => {
// 4 agents at 500k each → 40 scheduled projects to 20M, far over the cap.
const warning = getLargeWorkflowWarning({
scheduledAgents: 40,
startedAgents: 4,
totalTokens: 2_000_000,
ultracodeActive: false,
})
expect(warning?.axis).toBe('both')
expect(warning?.projectedTokens).toBe(20_000_000)
})
test('an explicit size guideline replaces the agent threshold', async () => {
await writeSettings({ workflowSizeGuideline: 'small' })
const warning = getLargeWorkflowWarning({ ...base, scheduledAgents: 6 })
expect(warning?.agentCap).toBe(5)
expect(warning?.capFromGuideline).toBe(true)
})
test('the env override beats the guideline', async () => {
await writeSettings({ workflowSizeGuideline: 'small' })
process.env.CLAUDE_CODE_WORKFLOW_SIZE_WARNING_AGENTS = '100'
expect(
getLargeWorkflowWarning({ ...base, scheduledAgents: 40 }),
).toBeUndefined()
})
test('ultracode suppresses it — that mode already opted in to scale', () => {
expect(
getLargeWorkflowWarning({
scheduledAgents: 500,
startedAgents: 10,
totalTokens: 9_000_000,
ultracodeActive: true,
}),
).toBeUndefined()
})
})
+237
View File
@@ -0,0 +1,237 @@
import { getIsNonInteractiveSession } from '../../bootstrap/state.js'
import { getGlobalConfig } from '../config.js'
import { isEnvTruthy } from '../envUtils.js'
import {
getSettings_DEPRECATED,
getSettingsForSource,
} from '../settings/settings.js'
export const WORKFLOW_SIZE_GUIDELINES = [
'unrestricted',
'small',
'medium',
'large',
] as const
export type WorkflowSizeGuideline = (typeof WORKFLOW_SIZE_GUIDELINES)[number]
export const DEFAULT_WORKFLOW_SIZE_GUIDELINE: WorkflowSizeGuideline = 'medium'
/** Agent count each guideline asks Claude to aim for. Advice, not a cap. */
export const WORKFLOW_SIZE_AGENT_TARGETS: Record<
WorkflowSizeGuideline,
number | null
> = {
unrestricted: null,
small: 5,
medium: 15,
large: 50,
}
export type WorkflowsDisabledReason = 'env' | 'managed' | 'settings'
/** Agent count that triggers the "Large workflow" advisory, absent a guideline. */
export const WORKFLOW_LARGE_AGENT_THRESHOLD = 25
/** Projected-token count that triggers the same advisory. */
export const WORKFLOW_LARGE_TOKEN_THRESHOLD = 1_500_000
/** Token estimate per not-yet-started agent, before any real usage is known. */
export const WORKFLOW_ASSUMED_TOKENS_PER_AGENT = 70_000
/**
* Why dynamic workflows are unavailable, or `null` when they are available.
*
* Managed settings are checked separately from user settings so the CLI can
* say *who* turned the feature off — a user who cannot re-enable it needs to
* know it was their organisation, not a toggle they flipped.
*/
export function getWorkflowsDisabledReason(): WorkflowsDisabledReason | null {
if (isEnvTruthy(process.env.CLAUDE_CODE_DISABLE_WORKFLOWS)) return 'env'
if (getSettingsForSource('policySettings')?.disableWorkflows === true) {
return 'managed'
}
const merged = getSettings_DEPRECATED()
if (merged.disableWorkflows === true) return 'settings'
// `enableWorkflows: false` is the /config toggle's off position. It is a
// separate key from `disableWorkflows` so an org can hard-disable while a
// user's toggle stays untouched underneath.
if (merged.enableWorkflows === false) return 'settings'
return null
}
export function areWorkflowsEnabled(): boolean {
return getWorkflowsDisabledReason() === null
}
/**
* Whether typing `ultracode` opts a turn into orchestration.
*
* On by default; the `/config` row writes `false` to turn it off. Independent
* of whether workflows themselves are enabled — a disabled feature has no
* keyword to suppress.
*/
export function isWorkflowKeywordTriggerEnabled(): boolean {
if (!areWorkflowsEnabled()) return false
// The keyword is an opt-in only in a prompt the user typed. A `-p` run, an
// SDK caller, a scheduled task, or a relayed PR comment can all contain the
// word "ultracode" without anyone having asked for a hundred agents.
if (getIsNonInteractiveSession()) return false
return getSettings_DEPRECATED().workflowKeywordTriggerEnabled ?? true
}
export type LargeWorkflowWarning = {
axis: 'agents' | 'tokens' | 'both'
scheduledAgents: number
totalTokens: number
projectedTokens: number
agentCap: number
tokenCap: number
/** True when the agent cap came from the size guideline, not the default. */
capFromGuideline: boolean
}
/**
* Advisory shown when a run grows past what the user asked for.
*
* It never pauses or limits anything — the point is that a runaway script is
* visible before it has spent the tokens, so the user can stop it from
* `/workflows`. Suppressed under ultracode: turning that on already opts in
* to large runs, so the warning would fire on every single run.
*/
export function getLargeWorkflowWarning(params: {
scheduledAgents: number
startedAgents: number
totalTokens: number
ultracodeActive: boolean
}): LargeWorkflowWarning | undefined {
if (params.ultracodeActive) return undefined
const guideline = getWorkflowSizeGuideline()
const guidelineCap = isWorkflowSizeGuidelineExplicit()
? WORKFLOW_SIZE_AGENT_TARGETS[guideline] ?? undefined
: undefined
const agentCap =
positiveInt(process.env.CLAUDE_CODE_WORKFLOW_SIZE_WARNING_AGENTS) ??
guidelineCap ??
WORKFLOW_LARGE_AGENT_THRESHOLD
const tokenCap =
positiveInt(process.env.CLAUDE_CODE_WORKFLOW_SIZE_WARNING_TOKENS) ??
WORKFLOW_LARGE_TOKEN_THRESHOLD
const perAgent =
params.startedAgents > 0
? params.totalTokens / params.startedAgents
: WORKFLOW_ASSUMED_TOKENS_PER_AGENT
const projectedTokens = Math.max(
params.totalTokens,
Math.round(perAgent * params.scheduledAgents),
)
const overAgents = params.scheduledAgents > agentCap
const overTokens = params.totalTokens > tokenCap || projectedTokens > tokenCap
if (!overAgents && !overTokens) return undefined
return {
axis: overAgents && overTokens ? 'both' : overAgents ? 'agents' : 'tokens',
scheduledAgents: params.scheduledAgents,
totalTokens: params.totalTokens,
projectedTokens,
agentCap,
tokenCap,
capFromGuideline: overAgents && guidelineCap !== undefined,
}
}
function positiveInt(raw: string | undefined): number | undefined {
if (!raw) return undefined
const value = Number.parseInt(raw, 10)
return Number.isFinite(value) && value > 0 ? value : undefined
}
export function describeWorkflowsDisabled(
reason: WorkflowsDisabledReason,
): string {
switch (reason) {
case 'env':
return 'Dynamic workflows are disabled by CLAUDE_CODE_DISABLE_WORKFLOWS.'
case 'managed':
return 'Dynamic workflows are disabled by managed settings (`disableWorkflows`).'
case 'settings':
return 'Dynamic workflows are disabled in settings (`disableWorkflows`). Turn them back on in /config.'
}
}
/**
* The configured size guideline.
*
* A settings file wins over the interactive `/config` value so an organisation
* can pin the scale; `/config` writes to user settings, which is where the
* fallback lands.
*/
export function getWorkflowSizeGuideline(): WorkflowSizeGuideline {
const fromSettings = getWorkflowSizeGuidelineFromSettings()
if (fromSettings) return fromSettings
const fromConfig = readGlobalGuideline()
if (fromConfig) return fromConfig
return DEFAULT_WORKFLOW_SIZE_GUIDELINE
}
/**
* The `/config` choice, or undefined.
*
* `getGlobalConfig()` throws before bootstrap opens config access, and this is
* read from the tool prompt — which the SDK can build early. A missing
* guideline is not worth failing a turn over.
*/
function readGlobalGuideline(): WorkflowSizeGuideline | undefined {
try {
const value = getGlobalConfig().workflowSizeGuideline
return isSizeGuideline(value) ? value : undefined
} catch {
return undefined
}
}
/**
* The guideline a settings file pins, if any.
*
* A settings file outranks the `/config` choice so an organisation can fix the
* scale; `/config` hides its row entirely in that case rather than offering a
* control that would silently do nothing.
*/
export function getWorkflowSizeGuidelineFromSettings():
| WorkflowSizeGuideline
| undefined {
const managed = getSettingsForSource('policySettings')?.workflowSizeGuideline
if (isSizeGuideline(managed)) return managed
const merged = getSettings_DEPRECATED().workflowSizeGuideline
if (isSizeGuideline(merged)) return merged
return undefined
}
/** True when the guideline was chosen, rather than falling back to the default. */
export function isWorkflowSizeGuidelineExplicit(): boolean {
if (getWorkflowSizeGuidelineFromSettings() !== undefined) return true
return readGlobalGuideline() !== undefined
}
/** Sentence appended to the Workflow tool prompt so Claude sizes runs to taste. */
export function describeWorkflowSizeGuideline(
guideline: WorkflowSizeGuideline = getWorkflowSizeGuideline(),
): string {
const target = WORKFLOW_SIZE_AGENT_TARGETS[guideline]
if (target === null) {
return 'This session has no workflow size guideline: size the workflow to the task.'
}
return (
`This session has the workflow size guideline: ${guideline} — keep workflows under ${target} agents. ` +
"This is a guideline, not a hard limit — follow it unless the user's prompt calls for a different scale. " +
'The user can raise or remove it with "Dynamic workflow size" in /config.'
)
}
function isSizeGuideline(value: unknown): value is WorkflowSizeGuideline {
return (
typeof value === 'string' &&
(WORKFLOW_SIZE_GUIDELINES as readonly string[]).includes(value)
)
}
+39
View File
@@ -0,0 +1,39 @@
/**
* Describe a thrown value that may have crossed the VM boundary.
*
* An `Error` constructed inside the workflow sandbox belongs to that realm, so
* `instanceof Error` is false in the host and the usual `err.message` narrowing
* never fires — the value would render as `{}`. Reading the fields
* defensively is the only way to keep the script author's own error text.
*/
export function describeThrown(value: unknown): string {
if (typeof value === 'string') return value
if (value === null || value === undefined) return String(value)
if (typeof value !== 'object') return String(value)
const message = readString(value, 'message')
const name = readString(value, 'name')
if (message) return name && name !== 'Error' ? `${name}: ${message}` : message
if (name) return name
try {
return JSON.stringify(value) ?? '[object]'
} catch {
return '[unprintable thrown value]'
}
}
/** The `name` of a thrown value, used to recognise the runtime's own errors. */
export function thrownName(value: unknown): string | undefined {
if (value === null || typeof value !== 'object') return undefined
return readString(value, 'name')
}
function readString(target: object, key: string): string | undefined {
try {
const value = (target as Record<string, unknown>)[key]
return typeof value === 'string' && value !== '' ? value : undefined
} catch {
return undefined
}
}
+588
View File
@@ -0,0 +1,588 @@
import type { CanUseToolFn } from '../../hooks/useCanUseTool.js'
import type { ToolUseContext } from '../../Tool.js'
import { logForDebugging } from '../debug.js'
import { describeThrown, thrownName } from './errors.js'
import {
WORKFLOW_AGENT_CAP_MESSAGE,
WORKFLOW_AGENT_STALL_MS,
WORKFLOW_LABEL_MAX_CHARS,
WORKFLOW_MAX_AGENTS,
WORKFLOW_MAX_FANOUT,
WORKFLOW_PREVIEW_MAX_CHARS,
getWorkflowConcurrency,
} from './constants.js'
import {
WorkflowJournal,
workflowCacheKey,
type WorkflowJournalSnapshot,
} from './journal.js'
import { createLimiter, type Limiter } from './limiter.js'
import {
createWorkflowAgentController,
runWorkflowAgent,
type WorkflowAgentRunParams,
type WorkflowAgentRunResult,
} from './runWorkflowAgent.js'
import type {
WorkflowAgentEvent,
WorkflowAgentOptions,
WorkflowProgressEvent,
} from './types.js'
/** Thrown when a script's loop keeps calling `agent()` past the hard cap. */
export class WorkflowAgentCapError extends Error {
constructor() {
super(WORKFLOW_AGENT_CAP_MESSAGE)
this.name = 'WorkflowAgentCapError'
}
}
/** Thrown when the turn's token target is exhausted mid-run. */
export class WorkflowBudgetExceededError extends Error {
constructor(spent: number, total: number) {
super(
`Workflow token budget exceeded (${spent.toLocaleString()} / ${total.toLocaleString()} output tokens). ` +
'Stopping further agent() calls. In-flight agents will complete; their results are preserved.',
)
this.name = 'WorkflowBudgetExceededError'
}
}
export type WorkflowHarnessParams = {
toolUseContext: ToolUseContext
canUseTool: CanUseToolFn
runId: string
/** `meta.name`, recorded on each agent's sidecar metadata. */
workflowName: string
emit: (event: WorkflowProgressEvent) => void
/** Titles from `meta.phases`, seeded so the phase list exists before agents run. */
seedPhaseTitles?: string[]
tokenBudget?: { total: number | null; getTurnSpent: () => number }
journal?: WorkflowJournal
journalSnapshot?: WorkflowJournalSnapshot
/** Registers/unregisters a running agent so the UI can skip or retry it. */
onAgentController: (
agentKey: string,
controller: AbortController | undefined,
) => void
abortSignal?: AbortSignal
/** Overridable so tests can drive the harness without a live model. */
runAgentImpl?: (
params: WorkflowAgentRunParams,
) => Promise<WorkflowAgentRunResult>
/**
* Counters a nested `workflow()` child shares with its parent.
*
* Without this a child would get its own concurrency slot pool (doubling the
* agents actually in flight) and its own index sequence (whose progress rows
* would overwrite the parent's, since rows are keyed by index).
*/
shared?: WorkflowSharedCounters
}
export type WorkflowSharedCounters = {
limiter: Limiter
nextAgentIndex: () => number
getAgentCount: () => number
}
/** Counters for a top-level run; nested children are handed the same object. */
export function createWorkflowSharedCounters(): WorkflowSharedCounters {
let agentCount = 0
return {
limiter: createLimiter(getWorkflowConcurrency()),
nextAgentIndex: () => ++agentCount,
getAgentCount: () => agentCount,
}
}
export type WorkflowHarness = {
agent: (prompt: unknown, opts?: unknown) => Promise<unknown>
parallel: (thunks: unknown) => Promise<unknown[]>
pipeline: (items: unknown, ...stages: unknown[]) => Promise<unknown[]>
log: (message: unknown) => void
phase: (title: unknown) => void
getAgentCount: () => number
getFailures: () => string[]
recordFailure: (message: string) => void
}
/**
* Build the globals a workflow script sees.
*
* Everything here is the host side of the VM boundary: values arriving from
* the script are untrusted (they can be cross-realm objects, proxies, or
* getters that throw), so each one is read defensively before use.
*/
export function createWorkflowHarness(
params: WorkflowHarnessParams,
): WorkflowHarness {
const {
toolUseContext,
canUseTool,
runId,
workflowName,
emit,
seedPhaseTitles,
tokenBudget,
journal,
journalSnapshot,
onAgentController,
abortSignal,
runAgentImpl = runWorkflowAgent,
shared = createWorkflowSharedCounters(),
} = params
const runAgentLimiter = shared.limiter
const failures: string[] = []
const phaseIndexByTitle = new Map<string, number>()
let phaseCounter = 0
let currentPhase: string | undefined
let capReported = false
let budgetReported = false
/** Journal replay stops permanently at the first cache miss (see resume docs). */
let cacheExhausted = false
let previousCacheKey = ''
function resolvePhase(title: string, kind: 'meta' | 'script'): number {
const existing = phaseIndexByTitle.get(title)
if (existing !== undefined) return existing
const index = ++phaseCounter
phaseIndexByTitle.set(title, index)
emit({ type: 'workflow_phase', index, title, kind })
return index
}
for (const title of seedPhaseTitles ?? []) {
if (typeof title === 'string' && title !== '') resolvePhase(title, 'meta')
}
function assertAgentCap(): void {
if (shared.getAgentCount() < WORKFLOW_MAX_AGENTS) return
if (!capReported) {
capReported = true
logForDebugging(`Workflow ${runId} hit the ${WORKFLOW_MAX_AGENTS}-agent cap`)
}
throw new WorkflowAgentCapError()
}
function assertBudget(): void {
const total = tokenBudget?.total
if (total == null || total <= 0) return
const spent = tokenBudget.getTurnSpent()
if (spent < total) return
if (!budgetReported) {
budgetReported = true
logForDebugging(`Workflow ${runId} exhausted its ${total} token budget`)
}
throw new WorkflowBudgetExceededError(spent, total)
}
function neverResolves<T>(): Promise<T> {
// An aborted run must not let the script continue past the await: the task
// is already terminal, so the cleanest stop is a promise that never settles.
return new Promise<T>(() => {})
}
const agent = async (
rawPrompt: unknown,
rawOpts?: unknown,
): Promise<unknown> => {
if (abortSignal?.aborted) return neverResolves()
const prompt = coerceToString(rawPrompt)
const opts = readAgentOptions(rawOpts)
assertAgentCap()
assertBudget()
const index = shared.nextAgentIndex()
const label = deriveLabel(prompt, opts.label)
const phaseTitle = opts.phase ?? currentPhase
const phaseIndex =
phaseTitle !== undefined ? resolvePhase(phaseTitle, 'script') : undefined
const model = opts.model ?? toolUseContext.options.mainLoopModel
const promptPreview = clip(prompt, WORKFLOW_PREVIEW_MAX_CHARS)
if (opts.isolation === 'remote') {
throw new Error(
"agent({isolation:'remote'}) is not available in this build",
)
}
const cacheKey = journal
? workflowCacheKey(prompt, opts, previousCacheKey)
: undefined
if (cacheKey !== undefined) previousCacheKey = cacheKey
if (cacheKey !== undefined && !cacheExhausted) {
const cached = journalSnapshot?.results.get(cacheKey)
if (cached !== undefined) {
const now = Date.now()
emit({
type: 'workflow_agent',
index,
label,
state: 'done',
phaseIndex,
phaseTitle,
model,
agentId: cached.agentId,
startedAt: now,
lastProgressAt: now,
cached: true,
promptPreview,
resultPreview: clip(
coerceToString(cached.result),
WORKFLOW_PREVIEW_MAX_CHARS,
),
})
return cached.result
}
// First miss: every later agent started after this one, so none of their
// cached results are valid any more.
cacheExhausted = true
}
const queuedAt = Date.now()
const baseEvent: WorkflowAgentEvent = {
type: 'workflow_agent',
index,
label,
state: 'start',
phaseIndex,
phaseTitle,
model,
queuedAt,
lastProgressAt: queuedAt,
promptPreview,
...(opts.agentType ? { agentType: opts.agentType } : {}),
...(opts.isolation === 'worktree' ? { isolation: 'worktree' as const } : {}),
}
emit(baseEvent)
const agentKey = `${runId}-${index}`
return runAgentLimiter(async () => {
if (abortSignal?.aborted) return neverResolves()
assertBudget()
const controller = createWorkflowAgentController(abortSignal)
onAgentController(agentKey, controller)
const startedAt = Date.now()
let tokens = 0
let toolCalls = 0
let agentId: string | undefined
const stallMs = opts.stallMs ?? WORKFLOW_AGENT_STALL_MS
let lastProgressAt = startedAt
const stallTimer = setInterval(() => {
if (Date.now() - lastProgressAt < stallMs) return
emit({
...baseEvent,
state: 'progress',
startedAt,
lastProgressAt,
tokens,
toolCalls,
agentId,
})
}, Math.max(5_000, Math.floor(stallMs / 2)))
stallTimer.unref?.()
emit({ ...baseEvent, state: 'progress', startedAt, lastProgressAt })
try {
const result = await runAgentImpl({
prompt,
opts,
toolUseContext,
canUseTool,
runId,
// Written to the agent's sidecar before it starts, so a finished
// run can be rebuilt from disk long after the progress stream is
// gone. Nothing else persists which phase an agent belonged to.
workflow: {
runId,
name: workflowName,
phaseIndex: phaseIndex ?? 0,
...(phaseTitle ? { phaseTitle } : {}),
agentIndex: index,
},
abortController: controller,
onAgentId: id => {
agentId = id
void journal
?.append({ type: 'started', key: cacheKey ?? '', agentId: id })
.catch(error =>
logForDebugging(`workflow journal started-append failed: ${error}`),
)
},
onProgress: progress => {
tokens = progress.tokens
toolCalls = progress.toolCalls
lastProgressAt = Date.now()
emit({
...baseEvent,
state: 'progress',
startedAt,
lastProgressAt,
tokens,
toolCalls,
agentId,
...(progress.lastToolName
? { lastToolName: progress.lastToolName }
: {}),
})
},
})
emit({
...baseEvent,
state: 'done',
agentId: result.agentId,
startedAt,
lastProgressAt: Date.now(),
durationMs: Date.now() - startedAt,
tokens: result.tokens,
toolCalls: result.toolCalls,
resultPreview: clip(
coerceToString(result.value),
WORKFLOW_PREVIEW_MAX_CHARS,
),
})
if (cacheKey !== undefined) {
void journal
?.append({
type: 'result',
key: cacheKey,
agentId: result.agentId,
result: result.value,
})
.catch(error =>
logForDebugging(`workflow journal result-append failed: ${error}`),
)
}
return result.value
} catch (error) {
const skipped =
describeThrown(controller.signal.reason).includes('user-skip')
const message = skipped ? 'skipped by user' : describeThrown(error)
emit({
...baseEvent,
state: 'error',
agentId,
startedAt,
lastProgressAt: Date.now(),
durationMs: Date.now() - startedAt,
tokens,
toolCalls,
error: message,
...(skipped ? { skipped: true } : {}),
})
if (abortSignal?.aborted) return neverResolves()
// A skipped agent resolves to null so the surrounding pipeline keeps
// going; a real failure propagates so the script can react.
if (skipped) return null
throw error
} finally {
clearInterval(stallTimer)
onAgentController(agentKey, undefined)
}
})
}
const parallel = async (rawThunks: unknown): Promise<unknown[]> => {
if (abortSignal?.aborted) return neverResolves()
const thunks = readArray(rawThunks, 'parallel() expects an array of functions')
if (thunks.length === 0) return []
assertAgentCap()
assertBudget()
for (const thunk of thunks) {
if (typeof thunk !== 'function') {
throw new TypeError(
'parallel() expects an array of functions, not promises. Wrap each call: () => agent(...)',
)
}
}
const settled = await Promise.allSettled(
thunks.map(thunk => {
try {
return Promise.resolve((thunk as () => unknown)())
} catch (error) {
return Promise.reject(error)
}
}),
)
return collectSettled(settled, 'parallel')
}
const pipeline = async (
rawItems: unknown,
...rawStages: unknown[]
): Promise<unknown[]> => {
if (abortSignal?.aborted) return neverResolves()
const items = readArray(
rawItems,
'pipeline() expects an array as the first argument',
)
if (items.length === 0) return []
assertAgentCap()
assertBudget()
const stages = rawStages.flat()
for (const stage of stages) {
if (typeof stage !== 'function') {
throw new TypeError(
'pipeline() stages must be functions: pipeline(items, item => ..., result => ...)',
)
}
}
const settled = await Promise.allSettled(
items.map(async (item, index) => {
let value: unknown = item
for (const stage of stages) {
if (value === null) break
value = await (stage as (
prev: unknown,
original: unknown,
index: number,
) => unknown)(value, item, index)
}
return value
}),
)
return collectSettled(settled, 'pipeline')
}
/**
* Turn a settled batch into the script-visible array.
*
* A rejected slot becomes `null` rather than rejecting the whole call — one
* bad file in a 500-file migration should not discard the other 499. The
* failure is still recorded so it reaches the final report.
*/
function collectSettled(
settled: PromiseSettledResult<unknown>[],
kind: 'parallel' | 'pipeline',
): unknown[] {
let budgetDrops = 0
const values = settled.map((entry, index) => {
if (entry.status === 'fulfilled') return entry.value
const reason = entry.reason
if (thrownName(reason) === 'WorkflowBudgetExceededError') {
budgetDrops++
return null
}
const message = `${kind}[${index}] failed: ${describeThrown(reason)}`
failures.push(message)
emit({ type: 'workflow_log', message })
return null
})
if (budgetDrops > 0) {
failures.push(
`${kind}: ${budgetDrops} slot${budgetDrops === 1 ? '' : 's'} dropped — token budget exceeded`,
)
}
return values
}
const log = (rawMessage: unknown): void => {
emit({ type: 'workflow_log', message: coerceToString(rawMessage) })
}
const phase = (rawTitle: unknown): void => {
const title = coerceToString(rawTitle)
if (title === '') return
currentPhase = title
resolvePhase(title, 'script')
}
return {
agent,
parallel,
pipeline,
log,
phase,
getAgentCount: shared.getAgentCount,
getFailures: () => failures,
recordFailure: message => failures.push(message),
}
}
/**
* Copy the options object out of the VM before use.
*
* The script owns that object and could keep mutating it (or define getters
* that throw) while the agent runs; a flat snapshot of the known keys is the
* only version the runtime trusts. `schema` is passed through by reference
* because `createSyntheticOutputTool` caches compiled validators on identity.
*/
function readAgentOptions(raw: unknown): WorkflowAgentOptions {
if (raw === null || typeof raw !== 'object') return {}
const opts: WorkflowAgentOptions = {}
const label = readProp(raw, 'label')
if (label != null) opts.label = String(label)
const phase = readProp(raw, 'phase')
if (phase != null) opts.phase = String(phase)
const model = readProp(raw, 'model')
if (model != null) opts.model = String(model)
const effort = readProp(raw, 'effort')
if (effort != null) opts.effort = String(effort)
const agentType = readProp(raw, 'agentType')
if (agentType != null) opts.agentType = String(agentType)
const isolation = readProp(raw, 'isolation')
if (isolation === 'worktree' || isolation === 'remote') {
opts.isolation = isolation
}
const stallMs = readProp(raw, 'stallMs')
if (typeof stallMs === 'number' && Number.isFinite(stallMs)) {
opts.stallMs = stallMs
}
const schema = readProp(raw, 'schema')
if (schema != null && typeof schema === 'object') opts.schema = schema
return opts
}
function readProp(target: unknown, key: string): unknown {
try {
return (target as Record<string, unknown>)[key]
} catch {
return undefined
}
}
function readArray(value: unknown, message: string): unknown[] {
if (!Array.isArray(value)) throw new TypeError(message)
if (value.length > WORKFLOW_MAX_FANOUT) {
throw new RangeError(
`A single parallel()/pipeline() call accepts at most ${WORKFLOW_MAX_FANOUT} items (got ${value.length})`,
)
}
return [...value]
}
function deriveLabel(prompt: string, explicit: string | undefined): string {
const source = explicit ?? prompt.slice(0, WORKFLOW_LABEL_MAX_CHARS)
return source.replace(/\s+/g, ' ').trim() || 'agent'
}
function clip(value: string, max: number): string | undefined {
const trimmed = value.trim()
if (trimmed === '') return undefined
return trimmed.length > max ? `${trimmed.slice(0, max)}…` : trimmed
}
/** Render any script value as a string without invoking its toString trap. */
function coerceToString(value: unknown): string {
if (typeof value === 'string') return value
if (value === null) return 'null'
if (value === undefined) return 'undefined'
if (typeof value === 'function') return '[function]'
if (typeof value === 'object') {
try {
return JSON.stringify(value) ?? '[object]'
} catch {
return '[object]'
}
}
return String(value)
}
+144
View File
@@ -0,0 +1,144 @@
import { appendFile, mkdir, readFile } from 'fs/promises'
import { dirname } from 'path'
import { logForDebugging } from '../debug.js'
import { getWorkflowJournalPath } from './paths.js'
/**
* One line per agent lifecycle transition, appended as the run progresses.
*
* `started` is written before the subagent is spawned and `result` after it
* returns. Resume needs both: an agent that has a `started` line but no
* `result` was in flight when the run stopped, which is what makes every agent
* after it re-run instead of replaying.
*/
export type WorkflowJournalEntry =
| { type: 'started'; key: string; agentId: string }
| { type: 'result'; key: string; agentId: string; result: unknown }
export type WorkflowJournalSnapshot = {
/** key → the cached return value of a completed agent. */
results: Map<string, { agentId: string; result: unknown }>
/** key → agent ids that started but never produced a result. */
started: Map<string, string[]>
}
/**
* Append-only record of a run's agent results, used to replay a resumed run.
*
* Writes are serialized through a promise chain: `agent()` calls resolve
* concurrently, and two overlapping appends to the same file can interleave
* partial lines.
*/
export class WorkflowJournal {
private readonly path: string
private writeChain: Promise<void> = Promise.resolve()
private dirReady = false
/** Takes an absolute path so tests never resolve against the real session dir. */
constructor(path: string) {
this.path = path
}
get filePath(): string {
return this.path
}
/**
* Wait for every queued append to hit disk.
*
* Appends are fired without awaiting so an agent's result reaches the script
* immediately; the run must flush before it settles or the last result can
* be missing from a resume.
*/
async flush(): Promise<void> {
await this.writeChain
}
async append(entry: WorkflowJournalEntry): Promise<void> {
const line = `${safeStringify(entry)}\n`
this.writeChain = this.writeChain.then(async () => {
if (!this.dirReady) {
await mkdir(dirname(this.path), { recursive: true })
this.dirReady = true
}
await appendFile(this.path, line, 'utf8')
})
return this.writeChain
}
/**
* Read back a previous run's journal. A truncated or corrupt trailing line
* is skipped rather than failing the resume — the worst case is that one
* agent re-runs.
*/
async load(): Promise<WorkflowJournalSnapshot> {
const snapshot: WorkflowJournalSnapshot = {
results: new Map(),
started: new Map(),
}
let raw: string
try {
raw = await readFile(this.path, 'utf8')
} catch {
return snapshot
}
for (const line of raw.split('\n')) {
if (line.trim() === '') continue
let entry: WorkflowJournalEntry
try {
entry = JSON.parse(line) as WorkflowJournalEntry
} catch {
logForDebugging(`workflow journal: skipping malformed line in ${this.path}`)
continue
}
if (entry.type === 'result') {
snapshot.results.set(entry.key, {
agentId: entry.agentId,
result: entry.result,
})
snapshot.started.delete(entry.key)
} else if (entry.type === 'started') {
if (snapshot.results.has(entry.key)) continue
const existing = snapshot.started.get(entry.key) ?? []
existing.push(entry.agentId)
snapshot.started.set(entry.key, existing)
}
}
return snapshot
}
}
/** The journal for a run, resolved against the current session's directory. */
export function createRunJournal(runId: string): WorkflowJournal {
return new WorkflowJournal(getWorkflowJournalPath(runId))
}
/**
* Cache key for one `agent()` call.
*
* Chained through the previous key so position in the call order is part of
* the identity: two identical `agent('review')` calls in a loop must not share
* a cache entry, and inserting a call ahead of them must invalidate both.
*/
export function workflowCacheKey(
prompt: string,
opts: unknown,
previousKey: string,
): string {
return `${previousKey}|${prompt}|${safeStringify(opts ?? null)}`
}
function safeStringify(value: unknown): string {
const seen = new WeakSet<object>()
return (
JSON.stringify(value, (_key, val) => {
if (typeof val === 'bigint') return val.toString()
if (typeof val === 'function') return undefined
if (val !== null && typeof val === 'object') {
if (seen.has(val as object)) return '[Circular]'
seen.add(val as object)
}
return val
}) ?? 'null'
)
}
+51
View File
@@ -0,0 +1,51 @@
import { describe, expect, test } from 'bun:test'
import { findKeywordRanges, hasWorkflowKeyword } from './keyword.js'
describe('ultracode keyword matcher', () => {
test.each([
['bare', 'ultracode: audit every endpoint'],
['mid-sentence', 'please ultracode this repo'],
['capitalised', 'Ultracode the migration'],
['after a comma', 'ok, ultracode, go'],
['at the end', 'audit the routes ultracode'],
])('fires on %s', (_label, text) => {
expect(hasWorkflowKeyword(text)).toBe(true)
})
test.each([
['inside backticks', 'run `ultracode` to see the keyword'],
['inside double quotes', 'the word "ultracode" opts a turn in'],
['inside single quotes', "the word 'ultracode' opts a turn in"],
['inside a tag', '<ultracode> is not a real tag'],
['inside braces', 'config is { mode: ultracode }'],
['inside brackets', 'see [ultracode] in the docs'],
['inside parens', 'the flag (ultracode) exists'],
['as a path segment', 'open docs/ultracode/readme.md'],
['as a filename', 'edit ultracode.ts'],
['as a flag', 'pass --ultracode to the CLI'],
['hyphenated', 'the ultracode-runner package'],
['as a question', 'what is ultracode?'],
['as a substring', 'superultracoded is a different word'],
])('does not fire %s', (_label, text) => {
expect(hasWorkflowKeyword(text)).toBe(false)
})
test('an apostrophe inside a word does not open a quoted span', () => {
expect(hasWorkflowKeyword("don't stop — ultracode this")).toBe(true)
})
test('reports the range so the composer can highlight it', () => {
const ranges = findKeywordRanges('please Ultracode this repo')
expect(ranges).toEqual([{ word: 'Ultracode', start: 7, end: 16 }])
})
test('finds every standalone occurrence', () => {
expect(findKeywordRanges('ultracode now, ultracode later')).toHaveLength(2)
})
test('empty and missing input never fires', () => {
expect(hasWorkflowKeyword('')).toBe(false)
expect(hasWorkflowKeyword(null)).toBe(false)
expect(hasWorkflowKeyword(undefined)).toBe(false)
})
})
+99
View File
@@ -0,0 +1,99 @@
/**
* The `ultracode` keyword trigger.
*
* Typing `ultracode` in a prompt opts that turn into multi-agent orchestration.
* The matcher has to be conservative: the word appears in file paths, code
* fences, and quoted prose far more often than it appears as an instruction,
* and a false positive silently turns a one-line question into a run that
* spawns dozens of agents.
*/
export const WORKFLOW_KEYWORD = 'ultracode'
export type KeywordRange = {
word: string
start: number
end: number
}
/** Spans a match must not fall inside — quoted text, code, and bracketed args. */
const SPAN_DELIMITERS: Record<string, string> = {
'`': '`',
'"': '"',
'<': '>',
'{': '}',
'[': ']',
'(': ')',
"'": "'",
}
function isWordChar(char: string | undefined): boolean {
return char !== undefined && /[A-Za-z0-9_]/.test(char)
}
/**
* Spans of `text` that should be treated as opaque.
*
* `<` only opens a span when it looks like a tag, and `'` only when it is not
* an apostrophe inside a word — otherwise "don't" would swallow the rest of
* the sentence and hide a real keyword behind it.
*/
function findOpaqueSpans(text: string): Array<{ start: number; end: number }> {
const spans: Array<{ start: number; end: number }> = []
let open: string | null = null
let openIndex = 0
for (let i = 0; i < text.length; i++) {
const char = text[i]!
if (open !== null) {
if (char !== SPAN_DELIMITERS[open]) continue
if (open === "'" && isWordChar(text[i + 1])) continue
spans.push({ start: openIndex, end: i + 1 })
open = null
continue
}
const opensTag = char === '<' && i + 1 < text.length && /[a-zA-Z/]/.test(text[i + 1]!)
const opensQuote = char === "'" && !isWordChar(text[i - 1])
const opensOther = char !== '<' && char !== "'" && char in SPAN_DELIMITERS
if (opensTag || opensQuote || opensOther) {
open = char
openIndex = i
}
}
return spans
}
/**
* Every place `keyword` appears as a standalone word the user meant.
*
* Returned as ranges, not a boolean, because the composer highlights the word
* so the user can see the turn was opted in before they send it.
*/
export function findKeywordRanges(
text: string,
keyword: string = WORKFLOW_KEYWORD,
): KeywordRange[] {
const opaque = findOpaqueSpans(text)
const matches: KeywordRange[] = []
for (const match of text.matchAll(new RegExp(`\\b${keyword}\\b`, 'gi'))) {
if (match.index === undefined) continue
const start = match.index
const end = start + match[0].length
if (opaque.some(span => start >= span.start && start < span.end)) continue
const before = text[start - 1]
const after = text[end]
// Path- and flag-like neighbours: `--ultracode`, `foo/ultracode`,
// `ultracode-runner`, `ultracode?`.
if (before === '/' || before === '\\' || before === '-') continue
if (after === '/' || after === '\\' || after === '-' || after === '?') continue
// `ultracode.js` — a filename, not an instruction.
if (after === '.' && isWordChar(text[end + 1])) continue
matches.push({ word: match[0], start, end })
}
return matches
}
export function hasWorkflowKeyword(text: string | null | undefined): boolean {
if (!text) return false
return findKeywordRanges(text).length > 0
}
@@ -0,0 +1,157 @@
import { afterEach, beforeEach, describe, expect, test } from 'bun:test'
import { mkdirSync } from 'fs'
import { mkdtemp, rm, writeFile } from 'fs/promises'
import { tmpdir } from 'os'
import { join } from 'path'
import { getIsInteractive, setIsInteractive } from '../../bootstrap/state.js'
import type { AppState } from '../../state/AppState.js'
import type { ToolUseContext } from '../../Tool.js'
import { getAttachmentsForTesting } from '../attachments.js'
import { resetSettingsCache } from '../settings/settingsCache.js'
let home: string
let configDir: string
let wasInteractive: boolean
const originalConfigDir = process.env.CLAUDE_CONFIG_DIR
async function writeSettings(value: Record<string, unknown>): Promise<void> {
await writeFile(join(configDir, 'settings.json'), JSON.stringify(value), 'utf8')
resetSettingsCache()
}
function contextWith(state: Partial<AppState>): ToolUseContext {
return {
getAppState: () => state as AppState,
} as unknown as ToolUseContext
}
/**
* The keyword path has three independent off-switches (the setting, the
* per-prompt `opt+w` dismissal, and the matcher itself). Each one has silently
* regressed in the official CLI at least once, and the failure mode is the
* same either way: a one-line prompt quietly becomes a run that spawns dozens
* of agents, or the keyword stops working with no error anywhere.
*/
describe('workflow keyword attachment', () => {
beforeEach(async () => {
home = await mkdtemp(join(tmpdir(), 'wf-keyword-attach-'))
configDir = join(home, 'claude')
mkdirSync(configDir, { recursive: true })
process.env.CLAUDE_CONFIG_DIR = configDir
// The keyword is interactive-only, so these cases have to run as if the
// user were at the prompt.
wasInteractive = getIsInteractive()
setIsInteractive(true)
await writeSettings({})
})
afterEach(async () => {
setIsInteractive(wasInteractive)
if (originalConfigDir === undefined) delete process.env.CLAUDE_CONFIG_DIR
else process.env.CLAUDE_CONFIG_DIR = originalConfigDir
resetSettingsCache()
await rm(home, { recursive: true, force: true })
})
test('fires on a typed keyword', () => {
expect(
getAttachmentsForTesting.workflowKeyword('ultracode: audit the routes', {
suppressed: false,
}),
).toEqual([{ type: 'workflow_keyword_request' }])
})
test('opt+w suppression wins over a present keyword', () => {
expect(
getAttachmentsForTesting.workflowKeyword('ultracode: audit the routes', {
suppressed: true,
}),
).toEqual([])
})
test('the setting turns the trigger off entirely', async () => {
await writeSettings({ workflowKeywordTriggerEnabled: false })
expect(
getAttachmentsForTesting.workflowKeyword('ultracode: audit', {
suppressed: false,
}),
).toEqual([])
})
test('a quoted mention of the word is not an opt-in', () => {
expect(
getAttachmentsForTesting.workflowKeyword(
'what does the "ultracode" keyword do?',
{ suppressed: false },
),
).toEqual([])
})
})
describe('ultracode effort attachments', () => {
test('announces the transition once, then keeps it short', () => {
const enter = getAttachmentsForTesting.ultracodeEffort([], true)
expect(enter).toEqual([{ type: 'ultra_effort_enter', reminderType: 'full' }])
const messages = [
{ type: 'attachment', attachment: { type: 'ultra_effort_enter' } },
] as never
expect(getAttachmentsForTesting.ultracodeEffort(messages, true)).toEqual([
{ type: 'ultra_effort_enter', reminderType: 'short' },
])
})
test('announces the exit only after it was announced on', () => {
expect(getAttachmentsForTesting.ultracodeEffort([], false)).toEqual([])
const messages = [
{ type: 'attachment', attachment: { type: 'ultra_effort_enter' } },
] as never
expect(getAttachmentsForTesting.ultracodeEffort(messages, false)).toEqual([
{ type: 'ultra_effort_exit' },
])
})
test('does not repeat the exit once it has been announced', () => {
const messages = [
{ type: 'attachment', attachment: { type: 'ultra_effort_enter' } },
{ type: 'attachment', attachment: { type: 'ultra_effort_exit' } },
] as never
expect(getAttachmentsForTesting.ultracodeEffort(messages, false)).toEqual([])
})
test('reads the flag from AppState, not from a parameter default', () => {
const context = contextWith({ ultracode: true })
expect(context.getAppState().ultracode).toBe(true)
})
})
describe('workflow size guideline reminder', () => {
test('stays silent on the first turn — the tool prompt already carries it', () => {
expect(getAttachmentsForTesting.workflowSizeGuideline([])).toEqual([])
})
test('announces a change away from what was last announced', () => {
const messages = [
{
type: 'attachment',
attachment: { type: 'workflow_size_guideline_change', size: 'small' },
},
] as never
// The live guideline in this environment is the default, so a previously
// announced 'small' must produce a fresh announcement.
expect(getAttachmentsForTesting.workflowSizeGuideline(messages)).toEqual([
{ type: 'workflow_size_guideline_change', size: 'medium' },
])
})
test('does not repeat an announcement that still holds', () => {
const messages = [
{
type: 'attachment',
attachment: { type: 'workflow_size_guideline_change', size: 'medium' },
},
] as never
expect(getAttachmentsForTesting.workflowSizeGuideline(messages)).toEqual([])
})
})
+34
View File
@@ -0,0 +1,34 @@
/**
* A minimal FIFO concurrency gate.
*
* `parallel()` and `pipeline()` hand the runtime as many items as the script
* asks for; this is what keeps only N model streams alive at a time. Queued
* work runs in submission order so a fan-out's progress view fills top-down
* instead of at random.
*/
export type Limiter = <T>(task: () => Promise<T>) => Promise<T>
export function createLimiter(concurrency: number): Limiter {
const max = Math.max(1, Math.floor(concurrency))
const queue: Array<() => void> = []
let active = 0
// Release hands the slot directly to the next waiter instead of decrementing
// and letting it re-check: a caller that arrives during the microtask gap
// would otherwise see a free slot that is already spoken for.
const release = (): void => {
const next = queue.shift()
if (next) next()
else active--
}
return async <T>(task: () => Promise<T>): Promise<T> => {
if (active < max) active++
else await new Promise<void>(resolve => queue.push(resolve))
try {
return await task()
} finally {
release()
}
}
}
+112
View File
@@ -0,0 +1,112 @@
import { describe, expect, test } from 'bun:test'
import { parseWorkflowScript, usesBannedNondeterminism } from './meta.js'
describe('parseWorkflowScript', () => {
test('extracts meta and leaves the body executable', () => {
const parsed = parseWorkflowScript(
[
"export const meta = { name: 'audit', description: 'Audit routes' }",
"const found = await agent('list files')",
'return found',
].join('\n'),
)
expect(parsed).not.toHaveProperty('error')
if ('error' in parsed) throw new Error(parsed.error)
expect(parsed.meta.name).toBe('audit')
expect(parsed.meta.description).toBe('Audit routes')
expect(parsed.scriptBody).toBe(
"const found = await agent('list files')\nreturn found",
)
})
test('keeps phases with their detail and model', () => {
const parsed = parseWorkflowScript(
[
'export const meta = {',
" name: 'review',",
" description: 'Review the diff',",
" whenToUse: 'before merging',",
' phases: [',
" { title: 'Find', detail: 'one agent per file' },",
" { title: 'Verify', model: 'sonnet' },",
' ],',
'}',
'return []',
].join('\n'),
)
if ('error' in parsed) throw new Error(parsed.error)
expect(parsed.meta.whenToUse).toBe('before merging')
expect(parsed.meta.phases).toEqual([
{ title: 'Find', detail: 'one agent per file' },
{ title: 'Verify', model: 'sonnet' },
])
})
test('rejects a script whose first statement is not the meta export', () => {
const parsed = parseWorkflowScript(
["const x = 1", "export const meta = { name: 'a', description: 'b' }"].join(
'\n',
),
)
expect(parsed).toEqual({
error:
'`export const meta = { name, description, phases }` must be the FIRST statement in the script',
})
})
test.each([
['a call', "export const meta = { name: 'a', description: makeIt() }\n"],
['an identifier', "export const meta = { name: 'a', description: other }\n"],
[
'interpolation',
'export const meta = { name: `a${1}`, description: "b" }\n',
],
[
'a spread',
"export const meta = { ...base, name: 'a', description: 'b' }\n",
],
])('rejects meta containing %s', (_label, script) => {
const parsed = parseWorkflowScript(script)
expect('error' in parsed && parsed.error).toContain('pure literal')
})
test('rejects a name that would not be a safe command', () => {
const parsed = parseWorkflowScript(
"export const meta = { name: '../escape', description: 'b' }\n",
)
expect('error' in parsed && parsed.error).toContain('meta.name')
})
test('reports TypeScript syntax with the plain-JavaScript hint', () => {
const parsed = parseWorkflowScript(
[
"export const meta = { name: 'a', description: 'b' }",
'const files: string[] = []',
].join('\n'),
)
expect('error' in parsed && parsed.error).toContain('plain JavaScript')
})
test('rejects a script over the byte limit', () => {
const script = `export const meta = { name: 'a', description: 'b' }\n// ${'x'.repeat(
600_000,
)}`
expect('error' in parseWorkflowScript(script)).toBe(true)
})
})
describe('usesBannedNondeterminism', () => {
test.each([
['Date.now()', 'const t = Date.now()'],
['new Date()', 'const t = new Date()'],
['Math.random()', 'const r = Math.random()'],
])('flags %s', (_label, body) => {
expect(usesBannedNondeterminism(body)).toBe(true)
})
test('allows a Date built from an explicit timestamp', () => {
expect(usesBannedNondeterminism('const t = new Date(args.stamp)')).toBe(
false,
)
})
})
+276
View File
@@ -0,0 +1,276 @@
import { parse } from 'acorn'
import { simple as walkSimple } from 'acorn-walk'
import { WORKFLOW_SCRIPT_MAX_BYTES } from './constants.js'
import type { WorkflowMeta, WorkflowPhaseMeta } from './types.js'
/* eslint-disable @typescript-eslint/no-explicit-any */
type AnyNode = any
/* eslint-enable @typescript-eslint/no-explicit-any */
export type WorkflowParseResult =
| { meta: WorkflowMeta; scriptBody: string }
| { error: string }
const PLAIN_JS_HINT =
'Workflow scripts must be plain JavaScript — common causes are TypeScript syntax ' +
'(type annotations, interfaces, generics) and broken string quoting or escaping.'
const CARET_WINDOW = 80
const ACORN_OPTIONS = {
ecmaVersion: 'latest',
sourceType: 'module',
allowAwaitOutsideFunction: true,
allowReturnOutsideFunction: true,
} as const
/**
* Split a workflow script into its `meta` literal and the executable body.
*
* `meta` must be the very first statement and a pure literal: the discovery
* path reads it for every saved workflow at `/` autocomplete time, so it can
* never be allowed to run code or reference anything outside itself.
*/
export function parseWorkflowScript(script: string): WorkflowParseResult {
if (Buffer.byteLength(script, 'utf8') > WORKFLOW_SCRIPT_MAX_BYTES) {
return { error: `Script exceeds ${WORKFLOW_SCRIPT_MAX_BYTES} bytes` }
}
let program: AnyNode
try {
program = parse(script, ACORN_OPTIONS as never)
} catch (error) {
return { error: formatParseError(error, script) }
}
const first = program.body[0]
if (!first || first.type !== 'ExportNamedDeclaration' || !isMetaExport(first)) {
return {
error:
'`export const meta = { name, description, phases }` must be the FIRST statement in the script',
}
}
const init = first.declaration.declarations[0].init
let literal: unknown
try {
literal = evaluateObjectLiteral(init)
} catch (error) {
return {
error: `meta must be a pure literal: ${
error instanceof Error ? error.message : String(error)
}`,
}
}
const validated = validateMeta(literal)
if ('error' in validated) return validated
// Drop the trailing `;` and the newline that ended the export so line
// numbers in runtime stack traces line up with the body the user wrote.
const scriptBody = script.slice(first.end).replace(/^[;\s]*\n/, '').trimStart()
return { meta: validated.meta, scriptBody }
}
/**
* True when the script reads `Date.now()`, `new Date()` or `Math.random()`.
* Those are unavailable at runtime (they would make a resume replay diverge),
* so the caller surfaces a targeted hint instead of a bare TypeError.
*/
export function usesBannedNondeterminism(script: string): boolean {
let found = false
try {
const program = parse(script, ACORN_OPTIONS as never)
walkSimple(program as never, {
MemberExpression(node: AnyNode) {
if (
node.computed ||
node.object.type !== 'Identifier' ||
node.property.type !== 'Identifier'
) {
return
}
const object = node.object.name
const property = node.property.name
if (
(object === 'Date' && property === 'now') ||
(object === 'Math' && property === 'random')
) {
found = true
}
},
NewExpression(node: AnyNode) {
if (
node.callee.type === 'Identifier' &&
node.callee.name === 'Date' &&
node.arguments.length === 0
) {
found = true
}
},
})
} catch {
return false
}
return found
}
function isMetaExport(node: AnyNode): boolean {
const declaration = node.declaration
if (!declaration || declaration.type !== 'VariableDeclaration') return false
if (declaration.kind !== 'const' || declaration.declarations.length !== 1) {
return false
}
const declarator = declaration.declarations[0]
return (
declarator.id.type === 'Identifier' &&
declarator.id.name === 'meta' &&
declarator.init?.type === 'ObjectExpression'
)
}
/**
* Evaluate an ObjectExpression made only of literals. Anything that could
* observe or mutate the outside world (identifiers, calls, spread, getters)
* throws, which the caller reports as "meta must be a pure literal".
*/
function evaluateObjectLiteral(node: AnyNode): Record<string, unknown> {
const result: Record<string, unknown> = {}
for (const property of node.properties) {
if (property.type === 'SpreadElement') {
throw new Error('spread not allowed in meta')
}
if (property.kind !== 'init' || property.method) {
throw new Error('only plain key: value pairs are allowed in meta')
}
let key: string
if (property.computed) {
throw new Error('computed keys not allowed in meta')
} else if (property.key.type === 'Identifier') {
key = property.key.name
} else if (property.key.type === 'Literal') {
key = String(property.key.value)
} else {
throw new Error('unsupported key in meta')
}
result[key] = evaluateLiteral(property.value)
}
return result
}
function evaluateLiteral(node: AnyNode): unknown {
switch (node.type) {
case 'Literal':
return node.value
case 'ArrayExpression':
return node.elements.map((element: AnyNode) => {
if (element === null) throw new Error('sparse arrays not allowed')
if (element.type === 'SpreadElement') {
throw new Error('spread not allowed in meta')
}
return evaluateLiteral(element)
})
case 'ObjectExpression':
return evaluateObjectLiteral(node)
case 'TemplateLiteral': {
if (node.expressions.length > 0) {
throw new Error('template interpolation not allowed in meta')
}
return node.quasis
.map((quasi: AnyNode) => quasi.value.cooked ?? '')
.join('')
}
case 'UnaryExpression': {
if (node.operator === '-' && node.argument.type === 'Literal') {
return -(node.argument.value as number)
}
throw new Error(`unsupported expression: ${node.operator}`)
}
default:
throw new Error(`unsupported expression: ${node.type}`)
}
}
function validateMeta(
value: unknown,
): { meta: WorkflowMeta } | { error: string } {
if (typeof value !== 'object' || value === null) {
return { error: 'meta must be an object literal' }
}
const raw = value as Record<string, unknown>
const name = raw.name
if (typeof name !== 'string' || name.trim() === '') {
return { error: 'meta.name must be a non-empty string' }
}
if (!/^[a-zA-Z0-9][a-zA-Z0-9_-]*$/.test(name)) {
return {
error: `meta.name '${name}' must start with a letter or digit and contain only letters, digits, '-' and '_'`,
}
}
const description = raw.description
if (typeof description !== 'string' || description.trim() === '') {
return { error: 'meta.description must be a non-empty string' }
}
const meta: WorkflowMeta = { name, description }
if (typeof raw.whenToUse === 'string') meta.whenToUse = raw.whenToUse
if (typeof raw.title === 'string') meta.title = raw.title
if (typeof raw.model === 'string') meta.model = raw.model
if (raw.phases !== undefined) {
if (!Array.isArray(raw.phases)) {
return { error: 'meta.phases must be an array' }
}
const phases: WorkflowPhaseMeta[] = []
for (const entry of raw.phases) {
if (typeof entry !== 'object' || entry === null) {
return { error: 'each meta.phases entry must be an object' }
}
const phase = entry as Record<string, unknown>
if (typeof phase.title !== 'string' || phase.title.trim() === '') {
return { error: 'each meta.phases entry needs a non-empty title' }
}
const parsed: WorkflowPhaseMeta = { title: phase.title }
if (typeof phase.detail === 'string') parsed.detail = phase.detail
if (typeof phase.model === 'string') parsed.model = phase.model
phases.push(parsed)
}
meta.phases = phases
}
return { meta }
}
function formatParseError(error: unknown, script: string): string {
const message = error instanceof Error ? error.message : String(error)
const loc = hasLoc(error) ? error.loc : undefined
const line = loc ? script.split('\n')[loc.line - 1] : undefined
if (!loc || line === undefined) {
return `Script parse error: ${message}. ${PLAIN_JS_HINT}`
}
const column = Math.max(0, Math.min(loc.column, line.length))
const start = Math.max(
0,
Math.min(column - Math.floor(CARET_WINDOW / 2), line.length - CARET_WINDOW),
)
const excerpt = line.slice(start, start + CARET_WINDOW)
const caret = `${' '.repeat(column - start)}^`
return `Script parse error: ${message}\n${excerpt}\n${caret}\n${PLAIN_JS_HINT}`
}
function hasLoc(
error: unknown,
): error is { loc: { line: number; column: number } } {
if (typeof error !== 'object' || error === null || !('loc' in error)) {
return false
}
const loc = (error as { loc: unknown }).loc
return (
typeof loc === 'object' &&
loc !== null &&
'line' in loc &&
typeof (loc as { line: unknown }).line === 'number' &&
'column' in loc &&
typeof (loc as { column: unknown }).column === 'number'
)
}
+134
View File
@@ -0,0 +1,134 @@
import { describe, expect, test } from 'bun:test'
import type { CanUseToolFn } from '../../hooks/useCanUseTool.js'
import type { ToolUseContext } from '../../Tool.js'
import { createWorkflowSharedCounters } from './harness.js'
import { executeWorkflowScript, prepareWorkflowScript } from './runtime.js'
import type {
WorkflowAgentRunParams,
WorkflowAgentRunResult,
} from './runWorkflowAgent.js'
import type { WorkflowProgressEvent } from './types.js'
const CHILD = [
"export const meta = { name: 'child', description: 'Child run' }",
"const out = await agent('child-' + (args ?? 'none'))",
'return { child: out }',
].join('\n')
async function runParent(
parentBody: string,
options: { childScript?: string } = {},
) {
const parent = prepareWorkflowScript(
`export const meta = { name: 'parent', description: 'Parent run' }\n${parentBody}`,
)
if (!parent.ok) throw new Error(parent.error)
const events: WorkflowProgressEvent[] = []
const shared = createWorkflowSharedCounters()
let seq = 0
const runAgentImpl = async (
params: WorkflowAgentRunParams,
): Promise<WorkflowAgentRunResult> => ({
agentId: `a${++seq}`,
value: `ran:${params.prompt}`,
tokens: 1,
toolCalls: 0,
})
const toolUseContext = {
abortController: new AbortController(),
options: { mainLoopModel: 'test-model' },
} as unknown as ToolUseContext
const canUseTool = (() => {}) as unknown as CanUseToolFn
const outcome = await executeWorkflowScript({
vmScript: parent.vmScript,
toolUseContext,
canUseTool,
runId: 'wf_nested00-abc',
shared,
onProgress: event => events.push(event),
onAgentController: () => {},
runAgentImpl,
runNestedWorkflow: async (_nameOrRef, args) => {
const child = prepareWorkflowScript(options.childScript ?? CHILD)
if (!child.ok) throw new Error(child.error)
const result = await executeWorkflowScript({
vmScript: child.vmScript,
toolUseContext,
canUseTool,
runId: 'wf_nested00-abc',
args,
shared,
onProgress: event => events.push(event),
onAgentController: () => {},
runAgentImpl,
runNestedWorkflow: undefined,
})
if (result.error) throw new Error(result.error)
return result.result
},
})
return { outcome, events, shared }
}
describe('nested workflow()', () => {
test('returns the child result and passes args through', async () => {
const { outcome } = await runParent(
"const child = await workflow('child', 'from-parent')\nreturn child",
)
expect(outcome.error).toBeUndefined()
expect(outcome.result).toEqual({ child: 'ran:child-from-parent' })
})
test('parent and child agents share one index sequence', async () => {
const { outcome, events, shared } = await runParent(
[
"const a = await agent('parent-1')",
"const child = await workflow('child', 'x')",
"const b = await agent('parent-2')",
'return [a, child, b]',
].join('\n'),
)
expect(outcome.error).toBeUndefined()
const indexes = [
...new Set(
events
.filter(event => event.type === 'workflow_agent')
.map(event => event.index),
),
].sort((a, b) => a - b)
// Three agents in total across both scripts, no index reused — a reused
// index would overwrite the parent's row in the progress view.
expect(indexes).toEqual([1, 2, 3])
expect(shared.getAgentCount()).toBe(3)
})
test("the child's failure surfaces as the parent's failure", async () => {
const { outcome } = await runParent(
"return await workflow('child')",
{ childScript: "export const meta = { name: 'child', description: 'Boom' }\nthrow new Error('child exploded')\n" },
)
expect(outcome.error).toContain('child exploded')
})
test('workflow() is unavailable when the runner is not supplied', async () => {
const prepared = prepareWorkflowScript(
"export const meta = { name: 'p', description: 'p' }\nreturn await workflow('child')\n",
)
if (!prepared.ok) throw new Error(prepared.error)
const outcome = await executeWorkflowScript({
vmScript: prepared.vmScript,
toolUseContext: {
abortController: new AbortController(),
options: { mainLoopModel: 'test-model' },
} as unknown as ToolUseContext,
canUseTool: (() => {}) as unknown as CanUseToolFn,
runId: 'wf_nested00-abc',
onAgentController: () => {},
})
expect(outcome.error).toBe('workflow() is not available in this run')
})
})
+54
View File
@@ -0,0 +1,54 @@
import { randomBytes } from 'crypto'
import { join } from 'path'
import {
getOriginalCwd,
getSessionId,
getSessionProjectDir,
} from '../../bootstrap/state.js'
import { getClaudeConfigHomeDir } from '../envUtils.js'
import { getProjectDir } from '../sessionStorage.js'
/** `~/.claude/workflows` — personal workflows, available in every project. */
export function getUserWorkflowsDir(): string {
return join(getClaudeConfigHomeDir(), 'workflows')
}
/**
* A fresh run id. Shaped `wf_<8 hex>-<3 hex>` so it is short enough to read in
* the progress view and still collision-free within a session.
*/
export function createWorkflowRunId(): string {
const bytes = randomBytes(6).toString('hex')
return `wf_${bytes.slice(0, 8)}-${bytes.slice(8, 11)}`
}
/** Subdirectory (relative to `subagents/`) holding this run's agent transcripts. */
export function getWorkflowTranscriptSubdir(runId: string): string {
return join('workflows', runId)
}
/**
* Absolute directory for a run's artifacts: the resume journal plus one
* `agent-<id>.jsonl` transcript per subagent. Lives beside the session's other
* subagent transcripts so existing transcript tooling can read it unchanged.
*/
export function getWorkflowTranscriptDir(runId: string): string {
const projectDir = getSessionProjectDir() ?? getProjectDir(getOriginalCwd())
return join(
projectDir,
getSessionId(),
'subagents',
getWorkflowTranscriptSubdir(runId),
)
}
/** Where the executed script is persisted so a run can be read back or edited. */
export function getWorkflowScriptPath(runId: string, name: string): string {
const projectDir = getSessionProjectDir() ?? getProjectDir(getOriginalCwd())
const safeName = name.replace(/[^a-zA-Z0-9._-]/g, '-').slice(0, 64) || 'workflow'
return join(projectDir, getSessionId(), 'workflows', `${safeName}.${runId}.js`)
}
export function getWorkflowJournalPath(runId: string): string {
return join(getWorkflowTranscriptDir(runId), 'journal.jsonl')
}
+51
View File
@@ -0,0 +1,51 @@
import { stat } from 'fs/promises'
import { dirname } from 'path'
import { logForDebugging } from '../debug.js'
import { loadAllPluginsCacheOnly } from '../plugins/pluginLoader.js'
import { loadWorkflowsFromDir } from './discovery.js'
import type { WorkflowDefinition } from './types.js'
/**
* Dynamic workflows shipped by enabled plugins.
*
* Names are namespaced `plugin:workflow` — two plugins can both ship a
* `release-audit` and neither has to lose. Reads the plugin cache rather than
* re-scanning: this runs on every `/` autocomplete.
*/
export async function loadPluginWorkflows(): Promise<WorkflowDefinition[]> {
let plugins: Awaited<ReturnType<typeof loadAllPluginsCacheOnly>>['plugins']
try {
plugins = (await loadAllPluginsCacheOnly()).plugins ?? []
} catch (error) {
logForDebugging(
`loadPluginWorkflows: plugin cache unavailable: ${error instanceof Error ? error.message : String(error)}`,
)
return []
}
const found: WorkflowDefinition[] = []
for (const plugin of plugins) {
if (plugin.enabled === false) continue
const dirs = [
...(plugin.workflowsPath ? [plugin.workflowsPath] : []),
...(plugin.workflowsPaths ?? []),
]
for (const path of dirs) {
const dir = await resolveDirectory(path)
if (!dir) continue
found.push(
...(await loadWorkflowsFromDir(dir, 'plugin', plugin.manifest.name)),
)
}
}
return found
}
/** A manifest entry may point at a single `.js` file rather than a directory. */
async function resolveDirectory(path: string): Promise<string | undefined> {
try {
return (await stat(path)).isDirectory() ? path : dirname(path)
} catch {
return undefined
}
}
+60
View File
@@ -0,0 +1,60 @@
import { describe, expect, test } from 'bun:test'
import { normalizeAttachmentForAPI } from '../messages.js'
import {
ULTRACODE_ENTER_REMINDER,
ULTRACODE_EXIT_REMINDER,
ULTRACODE_STILL_ON_REMINDER,
WORKFLOW_KEYWORD_REMINDER,
} from './ultracode.js'
/**
* The reminders are the only channel that tells the model a turn was opted
* into orchestration. If one stops rendering, the Workflow tool's opt-in rule
* silently stops firing and nothing else in the system notices.
*/
function renderedText(
attachment: Parameters<typeof normalizeAttachmentForAPI>[0],
): string {
return normalizeAttachmentForAPI(attachment)
.map(message =>
typeof message.message.content === 'string'
? message.message.content
: message.message.content
.map(block => ('text' in block ? block.text : ''))
.join('\n'),
)
.join('\n')
}
describe('workflow reminders', () => {
test('the keyword reminder names the tool the model must call', () => {
const text = renderedText({ type: 'workflow_keyword_request' })
expect(text).toContain(WORKFLOW_KEYWORD_REMINDER)
expect(text).toContain('Workflow tool')
expect(text).toContain('<system-reminder>')
})
test('entering ultracode gets the full standing instruction', () => {
const text = renderedText({
type: 'ultra_effort_enter',
reminderType: 'full',
})
expect(text).toContain(ULTRACODE_ENTER_REMINDER)
expect(text).toContain('every substantive task')
})
test('staying in ultracode gets the short form, not the paragraph', () => {
const text = renderedText({
type: 'ultra_effort_enter',
reminderType: 'short',
})
expect(text).toContain(ULTRACODE_STILL_ON_REMINDER)
expect(text).not.toContain('token cost is not a constraint')
})
test('leaving ultracode restores the opt-in rule explicitly', () => {
const text = renderedText({ type: 'ultra_effort_exit' })
expect(text).toContain(ULTRACODE_EXIT_REMINDER)
expect(text).toContain('opt-in rule applies again')
})
})
+142
View File
@@ -0,0 +1,142 @@
import { describe, expect, test } from 'bun:test'
import { mkdtemp, rm } from 'fs/promises'
import { tmpdir } from 'os'
import { join } from 'path'
import type { CanUseToolFn } from '../../hooks/useCanUseTool.js'
import type { ToolUseContext } from '../../Tool.js'
import { createWorkflowHarness } from './harness.js'
import { WorkflowJournal } from './journal.js'
import type {
WorkflowAgentRunParams,
WorkflowAgentRunResult,
} from './runWorkflowAgent.js'
import type { WorkflowProgressEvent } from './types.js'
function makeHarness(options: {
journal?: WorkflowJournal
journalSnapshot?: Awaited<ReturnType<WorkflowJournal['load']>>
onRun?: (params: WorkflowAgentRunParams) => void
}) {
const events: WorkflowProgressEvent[] = []
let seq = 0
const harness = createWorkflowHarness({
toolUseContext: {
options: { mainLoopModel: 'test-model' },
} as unknown as ToolUseContext,
canUseTool: (() => {}) as unknown as CanUseToolFn,
runId: 'wf_temp0000-aaa',
emit: event => events.push(event),
onAgentController: () => {},
journal: options.journal,
journalSnapshot: options.journalSnapshot,
runAgentImpl: async (
params: WorkflowAgentRunParams,
): Promise<WorkflowAgentRunResult> => {
options.onRun?.(params)
return {
agentId: `live-${++seq}`,
value: `live:${params.prompt}`,
tokens: 1,
toolCalls: 0,
}
},
})
return { harness, events }
}
describe('workflow resume', () => {
test('replays cached results and stops replaying at the first miss', async () => {
const dir = await mkdtemp(join(tmpdir(), 'wf-journal-'))
try {
const journal = new WorkflowJournal(join(dir, 'journal.jsonl'))
// First run: three agents, all recorded.
const first = makeHarness({ journal })
const a1 = await first.harness.agent('one')
const a2 = await first.harness.agent('two')
const a3 = await first.harness.agent('three')
expect([a1, a2, a3]).toEqual(['live:one', 'live:two', 'live:three'])
await journal.flush()
const snapshot = await journal.load()
expect(snapshot.results.size).toBe(3)
// Second run: the middle prompt changed, so it and everything after it
// must run live again even though the third call is byte-identical.
const liveRuns: string[] = []
const second = makeHarness({
journal,
journalSnapshot: snapshot,
onRun: params => liveRuns.push(params.prompt),
})
const b1 = await second.harness.agent('one')
const b2 = await second.harness.agent('CHANGED')
const b3 = await second.harness.agent('three')
expect(b1).toBe('live:one')
expect(liveRuns).toEqual(['CHANGED', 'three'])
expect(b2).toBe('live:CHANGED')
expect(b3).toBe('live:three')
const cachedEvents = second.events.filter(
event => event.type === 'workflow_agent' && event.cached === true,
)
expect(cachedEvents).toHaveLength(1)
} finally {
await rm(dir, { recursive: true, force: true })
}
})
test('an unchanged script replays every agent from cache', async () => {
const dir = await mkdtemp(join(tmpdir(), 'wf-journal-'))
try {
const journal = new WorkflowJournal(join(dir, 'journal.jsonl'))
const first = makeHarness({ journal })
await first.harness.agent('one')
await first.harness.agent('two')
await journal.flush()
const snapshot = await journal.load()
const liveRuns: string[] = []
const second = makeHarness({
journal,
journalSnapshot: snapshot,
onRun: params => liveRuns.push(params.prompt),
})
expect(await second.harness.agent('one')).toBe('live:one')
expect(await second.harness.agent('two')).toBe('live:two')
expect(liveRuns).toEqual([])
expect(second.harness.getAgentCount()).toBe(2)
} finally {
await rm(dir, { recursive: true, force: true })
}
})
test('an agent that started but never finished is not replayed', async () => {
const dir = await mkdtemp(join(tmpdir(), 'wf-journal-'))
try {
const path = join(dir, 'journal.jsonl')
const journal = new WorkflowJournal(path)
await journal.append({
type: 'started',
key: '|one|null',
agentId: 'interrupted',
})
const snapshot = await journal.load()
expect(snapshot.results.size).toBe(0)
expect(snapshot.started.get('|one|null')).toEqual(['interrupted'])
const liveRuns: string[] = []
const { harness } = makeHarness({
journal,
journalSnapshot: snapshot,
onRun: params => liveRuns.push(params.prompt),
})
await harness.agent('one')
expect(liveRuns).toEqual(['one'])
} finally {
await rm(dir, { recursive: true, force: true })
}
})
})
+336
View File
@@ -0,0 +1,336 @@
import type { CanUseToolFn } from '../../hooks/useCanUseTool.js'
import type { Tool, ToolUseContext } from '../../Tool.js'
import { toolMatchesName } from '../../Tool.js'
import {
createActivityDescriptionResolver,
createProgressTracker,
getTokenCountFromTracker,
updateProgressFromMessage,
} from '../../tasks/LocalAgentTask/LocalAgentTask.js'
import type {
AgentDefinition,
BuiltInAgentDefinition,
} from '../../tools/AgentTool/loadAgentsDir.js'
import { getLastToolUseName } from '../../tools/AgentTool/agentToolUtils.js'
import { runAgent } from '../../tools/AgentTool/runAgent.js'
import {
createSyntheticOutputTool,
SYNTHETIC_OUTPUT_TOOL_NAME,
} from '../../tools/SyntheticOutputTool/SyntheticOutputTool.js'
import { assembleToolPool } from '../../tools.js'
import type { Message } from '../../types/message.js'
import { createAbortController } from '../abortController.js'
import { runWithCwdOverride } from '../cwd.js'
import { logForDebugging } from '../debug.js'
import { createUserMessage, extractTextContent } from '../messages.js'
import { createAgentId } from '../uuid.js'
import type { AgentMetadata } from '../sessionStorage.js'
import {
createAgentWorktree,
hasWorktreeChanges,
removeAgentWorktree,
} from '../worktree.js'
import {
WORKFLOW_SUBAGENT_PROMPT,
WORKFLOW_SUBAGENT_TYPE,
workflowStructuredOutputNote,
workflowStructuredSubagentPrompt,
} from './constants.js'
import type { WorkflowAgentOptions } from './types.js'
/**
* The agent definition every workflow subagent runs under.
*
* `permissionMode: 'acceptEdits'` is deliberate and independent of the
* session's mode: the script is the orchestrator and there is nobody to
* approve each of a hundred file edits. Tool calls still go through the
* session allowlist, so shell/MCP calls can prompt.
*/
function buildWorkflowAgentDefinition(
schemaToolName: string | undefined,
): BuiltInAgentDefinition {
return {
agentType: WORKFLOW_SUBAGENT_TYPE,
whenToUse: 'Internal subagent for workflow script orchestration.',
tools: ['*'],
source: 'built-in',
baseDir: 'built-in',
permissionMode: 'acceptEdits',
getSystemPrompt: () =>
schemaToolName
? workflowStructuredSubagentPrompt(schemaToolName)
: WORKFLOW_SUBAGENT_PROMPT,
}
}
export type WorkflowAgentRunParams = {
prompt: string
opts?: WorkflowAgentOptions
toolUseContext: ToolUseContext
canUseTool: CanUseToolFn
runId: string
/** Workflow/phase provenance stamped onto the agent's sidecar metadata. */
workflow?: AgentMetadata['workflow']
/** Per-agent controller so the user can skip or restart a single agent. */
abortController: AbortController
/** Called once the agent id is known, before the first model request. */
onAgentId: (agentId: string) => void
/** Called on every assistant message with the running token/tool totals. */
onProgress: (progress: {
tokens: number
toolCalls: number
lastToolName?: string
}) => void
}
export type WorkflowAgentRunResult = {
agentId: string
/** Structured output object when a schema was given, otherwise the final text. */
value: unknown
tokens: number
toolCalls: number
worktreePath?: string
}
/**
* Run one `agent()` call to completion and return its value to the script.
*
* With a `schema`, the agent is forced through `StructuredOutput` and the
* validated object is returned — validation happens at the tool-call layer so
* the model retries a bad shape itself instead of the script parsing prose.
*/
export async function runWorkflowAgent(
params: WorkflowAgentRunParams,
): Promise<WorkflowAgentRunResult> {
const {
prompt,
opts,
toolUseContext,
canUseTool,
runId,
workflow,
abortController,
onAgentId,
onProgress,
} = params
const agentDefinition = resolveAgentDefinition(toolUseContext, opts)
const schemaTool = buildSchemaTool(opts?.schema)
if (schemaTool && 'error' in schemaTool) {
throw new Error(`agent({schema}): ${schemaTool.error}`)
}
const appState = toolUseContext.getAppState()
const workerPermissionContext = {
...appState.toolPermissionContext,
mode: agentDefinition.permissionMode ?? ('acceptEdits' as const),
}
const basePool = assembleToolPool(workerPermissionContext, appState.mcp.tools)
// A parent StructuredOutput (from --json-schema) carries a different schema;
// leaving it in would let the agent satisfy the wrong contract.
const availableTools: Tool[] = schemaTool
? [
...basePool.filter(
tool => !toolMatchesName(tool, SYNTHETIC_OUTPUT_TOOL_NAME),
),
schemaTool.tool,
]
: [...basePool]
const agentId = createAgentId()
onAgentId(agentId)
const promptText = schemaTool
? `${prompt}${workflowStructuredOutputNote(SYNTHETIC_OUTPUT_TOOL_NAME)}`
: prompt
let worktree: Awaited<ReturnType<typeof createAgentWorktree>> | null = null
if (opts?.isolation === 'worktree') {
worktree = await createAgentWorktree(`agent-${agentId.slice(0, 8)}`)
}
const tracker = createProgressTracker()
const resolveActivity = createActivityDescriptionResolver(
toolUseContext.options.tools,
)
const messages: Message[] = []
let structuredOutput: unknown
const runInCwd = <T,>(fn: () => T): T =>
worktree ? runWithCwdOverride(worktree.worktreePath, fn) : fn()
try {
const iterator = runInCwd(() =>
runAgent({
agentDefinition,
promptMessages: [createUserMessage({ content: promptText })],
toolUseContext: {
...toolUseContext,
abortController,
agentId,
agentType: agentDefinition.agentType,
options: {
...toolUseContext.options,
tools: availableTools,
...(opts?.model ? { mainLoopModel: opts.model } : {}),
},
},
canUseTool,
isAsync: false,
canShowPermissionPrompts: true,
querySource: 'workflow_agent',
model: opts?.model,
availableTools,
description: opts?.label ?? prompt.slice(0, 60),
...(workflow ? { workflow } : {}),
// Deliberately no `transcriptSubdir`. A workflow agent is an ordinary
// subagent run by the same runner, so its transcript belongs in the
// session's flat `subagents/` directory with every other one — that is
// the only place the reader looks, and routing these into
// `subagents/workflows/<runId>/` is what made them unopenable in the
// UI. The run's journal still lives in the per-run directory.
...(worktree ? { worktreePath: worktree.worktreePath } : {}),
override: { agentId, abortController },
}),
)
for await (const message of iterator) {
messages.push(message)
if (
message.type === 'attachment' &&
message.attachment.type === 'structured_output'
) {
structuredOutput = message.attachment.data
continue
}
if (message.type !== 'assistant') continue
updateProgressFromMessage(
tracker,
message,
resolveActivity,
toolUseContext.options.tools,
)
onProgress({
tokens: getTokenCountFromTracker(tracker),
toolCalls: tracker.toolUseCount,
lastToolName: getLastToolUseName(message),
})
}
} finally {
if (worktree) {
await cleanupWorkflowWorktree(worktree).catch(error =>
logForDebugging(
`Workflow worktree cleanup failed: ${error instanceof Error ? error.message : String(error)}`,
),
)
}
}
const value = schemaTool
? structuredOutput
: extractFinalText(messages)
if (schemaTool && structuredOutput === undefined) {
throw new Error(
`agent({schema}): the subagent ended without calling ${SYNTHETIC_OUTPUT_TOOL_NAME}`,
)
}
return {
agentId,
value: value ?? null,
tokens: getTokenCountFromTracker(tracker),
toolCalls: tracker.toolUseCount,
...(worktree ? { worktreePath: worktree.worktreePath } : {}),
}
}
/**
* Resolve `opts.agentType` against the session's active agents.
*
* A named agent keeps its own system prompt and tool list — the workflow only
* supplies the prompt — so a script can reuse `code-reviewer` or a custom
* agent instead of the generic workflow subagent.
*/
function resolveAgentDefinition(
toolUseContext: ToolUseContext,
opts: WorkflowAgentOptions | undefined,
): AgentDefinition {
const requested = opts?.agentType
const schemaToolName = opts?.schema ? SYNTHETIC_OUTPUT_TOOL_NAME : undefined
if (requested == null) return buildWorkflowAgentDefinition(schemaToolName)
const activeAgents = toolUseContext.options.agentDefinitions.activeAgents
const match = activeAgents.find(agent => agent.agentType === requested)
if (!match) {
const available = activeAgents.map(agent => agent.agentType).join(', ')
throw new Error(
`agent({agentType}): agent type '${requested}' not found. Available agents: ${available}`,
)
}
if (!schemaToolName) return match
// Custom agent + schema: keep its prompt but append the StructuredOutput
// instruction so the two contracts don't fight.
const basePrompt = match.getSystemPrompt({
toolUseContext: { options: toolUseContext.options },
} as never)
return {
...match,
getSystemPrompt: () =>
`${basePrompt}\n${workflowStructuredOutputNote(schemaToolName)}`,
} as AgentDefinition
}
function buildSchemaTool(
schema: unknown,
): { tool: Tool } | { error: string } | undefined {
if (schema == null) return undefined
if (typeof schema !== 'object' || Array.isArray(schema)) {
return { error: 'schema must be a JSON Schema object' }
}
return createSyntheticOutputTool(schema as Record<string, unknown>) as
| { tool: Tool }
| { error: string }
}
/**
* The agent's final text. Falls back to the last assistant message that had
* any text, so a run that ended on a bare tool_use block still returns
* something instead of an empty string.
*/
function extractFinalText(messages: Message[]): string {
for (let i = messages.length - 1; i >= 0; i--) {
const message = messages[i]
if (message?.type !== 'assistant') continue
const text = extractTextContent(message.message.content, '\n').trim()
if (text !== '') return text
}
return ''
}
async function cleanupWorkflowWorktree(
worktree: Awaited<ReturnType<typeof createAgentWorktree>>,
): Promise<void> {
const { worktreePath, worktreeBranch, headCommit, gitRoot, hookBased } =
worktree
if (hookBased) return
if (!headCommit) return
if (await hasWorktreeChanges(worktreePath, headCommit)) return
await removeAgentWorktree(worktreePath, worktreeBranch, gitRoot)
}
/** Exposed so the harness can create per-agent controllers consistently. */
export function createWorkflowAgentController(
parent: AbortSignal | undefined,
): AbortController {
const controller = createAbortController()
if (parent) {
if (parent.aborted) controller.abort(parent.reason)
else
parent.addEventListener('abort', () => controller.abort(parent.reason), {
once: true,
})
}
return controller
}
+287
View File
@@ -0,0 +1,287 @@
import { describe, expect, test } from 'bun:test'
import type { ToolUseContext } from '../../Tool.js'
import type { CanUseToolFn } from '../../hooks/useCanUseTool.js'
import { executeWorkflowScript, prepareWorkflowScript } from './runtime.js'
import type {
WorkflowAgentRunParams,
WorkflowAgentRunResult,
} from './runWorkflowAgent.js'
import type { WorkflowProgressEvent } from './types.js'
/**
* Drive a whole script through the real VM sandbox with a scripted stand-in
* for the subagent. The harness/runtime seam is what these tests own; the
* model call itself is covered by the CLI smoke lane.
*/
async function run(
script: string,
options: {
agent?: (params: WorkflowAgentRunParams) => Promise<unknown>
args?: unknown
abortController?: AbortController
} = {},
) {
const prepared = prepareWorkflowScript(script)
if (!prepared.ok) throw new Error(prepared.error)
const events: WorkflowProgressEvent[] = []
const abortController = options.abortController ?? new AbortController()
let seq = 0
const outcome = await executeWorkflowScript({
vmScript: prepared.vmScript,
toolUseContext: {
abortController,
options: { mainLoopModel: 'test-model' },
} as unknown as ToolUseContext,
canUseTool: (() => {}) as unknown as CanUseToolFn,
runId: 'wf_test0000-abc',
workflowName: prepared.meta.name,
args: options.args,
seedPhaseTitles: prepared.meta.phases?.map(phase => phase.title),
onProgress: event => events.push(event),
onAgentController: () => {},
runAgentImpl: async (
params: WorkflowAgentRunParams,
): Promise<WorkflowAgentRunResult> => {
const value = options.agent
? await options.agent(params)
: `ran: ${params.prompt}`
return {
agentId: `agent-${++seq}`,
value,
tokens: 10,
toolCalls: 1,
}
},
})
return { outcome, events, meta: prepared.meta }
}
const META = "export const meta = { name: 'demo', description: 'A demo' }\n"
describe('executeWorkflowScript', () => {
test('returns the script result and counts the agents it spawned', async () => {
const { outcome } = await run(
`${META}
const a = await agent('first')
const b = await agent('second')
return { a, b }`,
)
expect(outcome.error).toBeUndefined()
expect(outcome.result).toEqual({ a: 'ran: first', b: 'ran: second' })
expect(outcome.agentCount).toBe(2)
})
test('pipeline threads each item through every stage', async () => {
const { outcome } = await run(
`${META}
const out = await pipeline(
['a.ts', 'b.ts'],
file => agent('review ' + file, { label: file, phase: 'Review' }),
(review, original, index) => agent('verify ' + original + '#' + index, { phase: 'Verify' }),
)
return out`,
{ agent: async params => params.prompt },
)
expect(outcome.result).toEqual([
'verify a.ts#0',
'verify b.ts#1',
])
expect(outcome.agentCount).toBe(4)
})
test('a failing slot becomes null instead of failing the whole batch', async () => {
const { outcome } = await run(
`${META}
const out = await parallel([
() => agent('ok'),
() => agent('boom'),
])
return out`,
{
agent: async params => {
if (params.prompt === 'boom') throw new Error('agent exploded')
return 'fine'
},
},
)
expect(outcome.result).toEqual(['fine', null])
expect(outcome.failures).toEqual(['parallel[1] failed: agent exploded'])
})
test('parallel rejects promises passed instead of thunks', async () => {
const { outcome } = await run(
`${META}
return await parallel([agent('a')])`,
)
expect(outcome.error).toContain('Wrap each call: () => agent(...)')
})
test('emits phase rows for meta.phases and for phase() calls', async () => {
const { events } = await run(
`export const meta = { name: 'demo', description: 'A demo', phases: [{ title: 'Scan' }] }
phase('Scan')
await agent('one')
phase('Fix')
await agent('two')
return null`,
)
const phases = events.filter(event => event.type === 'workflow_phase')
expect(phases).toEqual([
{ type: 'workflow_phase', index: 1, title: 'Scan', kind: 'meta' },
{ type: 'workflow_phase', index: 2, title: 'Fix', kind: 'script' },
])
const agentPhases = events
.filter(event => event.type === 'workflow_agent' && event.state === 'done')
.map(event => (event.type === 'workflow_agent' ? event.phaseTitle : null))
expect(agentPhases).toEqual(['Scan', 'Fix'])
})
test('log() and console.log both reach the progress stream', async () => {
const { outcome, events } = await run(
`${META}
log('from log')
console.log('from console', { n: 1 })
return 'done'`,
)
expect(outcome.logs).toEqual(['from log', 'from console {"n":1}'])
expect(
events.filter(event => event.type === 'workflow_log'),
).toHaveLength(2)
})
test('args arrive as structured data, not a JSON string', async () => {
const { outcome } = await run(
`${META}
return args.filter(n => n > 1)`,
{ args: [1, 2, 3] },
)
expect(outcome.result).toEqual([2, 3])
})
test('args is undefined when the caller passed none', async () => {
const { outcome } = await run(`${META}
return typeof args`)
expect(outcome.result).toBe('undefined')
})
test('budget.remaining is Infinity with no target', async () => {
const { outcome } = await run(`${META}
return { total: budget.total, infinite: budget.remaining() === Infinity }`)
expect(outcome.result).toEqual({ total: null, infinite: true })
})
test('Date.now, new Date and Math.random are unavailable', async () => {
for (const expression of ['Date.now()', 'new Date()', 'Math.random()']) {
const { outcome } = await run(`${META}
return ${expression}`)
expect(outcome.error).toContain('unavailable in workflow scripts')
}
})
test('a Date built from an explicit timestamp still works', async () => {
const { outcome } = await run(`${META}
return new Date(0).toISOString()`)
expect(outcome.result).toBe('1970-01-01T00:00:00.000Z')
})
test('eval and new Function are blocked', async () => {
const { outcome } = await run(`${META}
return eval('1 + 1')`)
expect(outcome.error).toBeDefined()
})
test('script values cannot reach the host realm through a constructor', async () => {
const { outcome } = await run(
`${META}
const out = await parallel([() => agent('a')])
const Ctor = out.constructor.constructor
return typeof Ctor('return process')().version`,
)
// codeGeneration is disabled, so the Function constructor cannot compile —
// and out.constructor is the VM's Array, not the host's.
expect(outcome.error).toBeDefined()
})
test('import() is refused before the run starts', () => {
const prepared = prepareWorkflowScript(`${META}
const mod = await import('fs')
return mod`)
expect(prepared.ok).toBe(true)
})
test('a rejected agent propagates when awaited directly', async () => {
const { outcome } = await run(
`${META}
return await agent('boom')`,
{
agent: async () => {
throw new Error('agent exploded')
},
},
)
expect(outcome.error).toBe('agent exploded')
})
test('a synchronous infinite loop is cut off by the sync timeout', async () => {
const prepared = prepareWorkflowScript(`${META}
while (true) {}`)
if (!prepared.ok) throw new Error(prepared.error)
const outcome = await executeWorkflowScript({
vmScript: prepared.vmScript,
toolUseContext: {
abortController: new AbortController(),
options: { mainLoopModel: 'test-model' },
} as unknown as ToolUseContext,
canUseTool: (() => {}) as unknown as CanUseToolFn,
runId: 'wf_test0000-abc',
onAgentController: () => {},
syncTimeoutMs: 200,
})
expect(outcome.error).toContain('timed out')
})
test('a workflow result that is a function is rejected', async () => {
const { outcome } = await run(`${META}
return () => 1`)
expect(outcome.error).toBe('workflow result cannot be a function')
})
})
describe('workflow agent provenance', () => {
test('stamps every agent with the run, name and phase it belongs to', async () => {
// This is the only record of a run's shape once the process exits — the
// desktop rebuilds a finished run's phases from it. It shipped with
// `name` silently undefined because nothing type-checks src/.
const seen: Array<WorkflowAgentRunParams['workflow']> = []
await run(
`export const meta = { name: 'demo', description: 'A demo' }
phase('Scan')
await agent('one', { label: 'scan:one' })
phase('Verify')
await agent('two', { label: 'verify:two' })`,
{
agent: async (params: WorkflowAgentRunParams) => {
seen.push(params.workflow)
return 'ok'
},
},
)
expect(seen).toHaveLength(2)
expect(seen[0]).toMatchObject({
runId: 'wf_test0000-abc',
name: 'demo',
phaseTitle: 'Scan',
agentIndex: 1,
})
expect(seen[1]).toMatchObject({
name: 'demo',
phaseTitle: 'Verify',
agentIndex: 2,
})
// Distinct phases must get distinct indices or the rebuild collapses them.
expect(seen[0]!.phaseIndex).not.toBe(seen[1]!.phaseIndex)
})
})
+404
View File
@@ -0,0 +1,404 @@
import vm from 'vm'
import type { CanUseToolFn } from '../../hooks/useCanUseTool.js'
import type { ToolUseContext } from '../../Tool.js'
import { logForDebugging } from '../debug.js'
import { compileWorkflowScript, installDeterminismGuards } from './compile.js'
import {
WORKFLOW_MAX_COLLECTED_LOGS,
WORKFLOW_SYNC_TIMEOUT_MS,
} from './constants.js'
import { describeThrown } from './errors.js'
import {
createWorkflowHarness,
createWorkflowSharedCounters,
type WorkflowHarness,
type WorkflowHarnessParams,
type WorkflowSharedCounters,
} from './harness.js'
import type { WorkflowJournal, WorkflowJournalSnapshot } from './journal.js'
import { parseWorkflowScript } from './meta.js'
import type {
WorkflowMeta,
WorkflowProgressEvent,
WorkflowRunOutcome,
WorkflowTokenBudget,
} from './types.js'
export type WorkflowExecutionParams = {
vmScript: vm.Script
toolUseContext: ToolUseContext
canUseTool: CanUseToolFn
runId: string
/** `meta.name`, persisted with each agent so a finished run is identifiable. */
workflowName: string
args?: unknown
seedPhaseTitles?: string[]
tokenBudget?: WorkflowTokenBudget
journal?: WorkflowJournal
journalSnapshot?: WorkflowJournalSnapshot
onProgress?: (event: WorkflowProgressEvent) => void
onAgentController: (
agentKey: string,
controller: AbortController | undefined,
) => void
/** Runs a saved workflow inline; `undefined` disables the `workflow()` global. */
runNestedWorkflow?: (nameOrRef: unknown, args: unknown) => Promise<unknown>
syncTimeoutMs?: number
/** Overridable so tests can drive a whole run without a live model. */
runAgentImpl?: WorkflowHarnessParams['runAgentImpl']
/** Concurrency and agent-index counters, shared with any nested run. */
shared?: WorkflowSharedCounters
}
/**
* Execute a compiled workflow script and return its result.
*
* The script runs in a `node:vm` context that holds nothing but the harness
* globals. Every value that crosses the boundary is rebuilt on the far side:
* handing VM code a live host object would expose `obj.constructor.constructor`
* — the host `Function` constructor — and with it the whole process.
*/
export async function executeWorkflowScript(
params: WorkflowExecutionParams,
): Promise<WorkflowRunOutcome> {
const startedAt = Date.now()
const logs: string[] = []
const abortSignal = params.toolUseContext.abortController?.signal
const emit = (event: WorkflowProgressEvent): void => {
if (
event.type === 'workflow_log' &&
logs.length < WORKFLOW_MAX_COLLECTED_LOGS
) {
logs.push(event.message)
}
params.onProgress?.(event)
}
const harness = createWorkflowHarness({
toolUseContext: params.toolUseContext,
canUseTool: params.canUseTool,
runId: params.runId,
workflowName: params.workflowName,
emit,
seedPhaseTitles: params.seedPhaseTitles,
tokenBudget: params.tokenBudget,
journal: params.journal,
journalSnapshot: params.journalSnapshot,
onAgentController: params.onAgentController,
abortSignal,
runAgentImpl: params.runAgentImpl,
shared: params.shared ?? createWorkflowSharedCounters(),
})
const sandbox = createWorkflowSandbox({
harness,
tokenBudget: params.tokenBudget,
args: params.args,
emit,
abortSignal,
runNestedWorkflow: params.runNestedWorkflow,
})
let detachAbort: (() => void) | undefined
try {
const pending = params.vmScript.runInContext(sandbox.context, {
timeout: params.syncTimeoutMs ?? WORKFLOW_SYNC_TIMEOUT_MS,
})
const settled = sandbox.awaitInVm(pending) as Promise<{ v: unknown }>
settled.catch(() => {})
let raced: { v: unknown }
if (abortSignal) {
raced = await Promise.race([
settled,
new Promise<never>((_resolve, reject) => {
const onAbort = () => reject(new Error('Workflow aborted'))
if (abortSignal.aborted) onAbort()
else {
abortSignal.addEventListener('abort', onAbort, { once: true })
detachAbort = () =>
abortSignal.removeEventListener('abort', onAbort)
}
}),
])
} else {
raced = await settled
}
const result = sandbox.exportValue(raced.v)
await params.journal?.flush()
return {
result,
agentCount: harness.getAgentCount(),
logs,
failures: harness.getFailures(),
durationMs: Date.now() - startedAt,
}
} catch (error) {
const message = describeError(error)
logForDebugging(`Workflow ${params.runId} script error: ${message}`)
await params.journal?.flush().catch(() => {})
return {
result: null,
agentCount: harness.getAgentCount(),
logs,
failures: harness.getFailures(),
durationMs: Date.now() - startedAt,
error: message,
}
} finally {
detachAbort?.()
sandbox.dispose()
}
}
type SandboxParams = {
harness: WorkflowHarness
tokenBudget?: WorkflowTokenBudget
args?: unknown
emit: (event: WorkflowProgressEvent) => void
abortSignal?: AbortSignal
runNestedWorkflow?: (nameOrRef: unknown, args: unknown) => Promise<unknown>
}
type WorkflowSandbox = {
context: vm.Context
awaitInVm: (value: unknown) => Promise<{ v: unknown }>
exportValue: (value: unknown) => unknown
dispose: () => void
}
function createWorkflowSandbox(params: SandboxParams): WorkflowSandbox {
const { harness, tokenBudget, emit, abortSignal, runNestedWorkflow } = params
// codeGeneration off: `eval` and `new Function` inside the script would
// otherwise reconstruct anything the sandbox withholds.
const context = vm.createContext(Object.create(null) as object, {
codeGeneration: { strings: false, wasm: false },
})
installDeterminismGuards(context)
const evalInVm = <T>(source: string): T =>
vm.runInContext(source, context, { filename: 'workflow-bridge.js' }) as T
const awaitInVm = evalInVm<(value: unknown) => Promise<{ v: unknown }>>(
'(async v => ({ __proto__: null, v: await v }))',
)
const parseInVm = evalInVm<(json: string) => unknown>('(json => JSON.parse(json))')
const stringifyInVm = evalInVm<(value: unknown) => string | undefined>(
'(value => JSON.stringify(value, (_k, v) => typeof v === "function" ? undefined : v))',
)
const newArrayInVm = evalInVm<(length: number) => unknown[]>(
'(length => new Array(length))',
)
const setIndexInVm = evalInVm<
(array: unknown, index: number, value: unknown) => void
>('((array, index, value) => { array[index] = value })')
// Host functions are never exposed directly: the script would reach the host
// realm through `fn.constructor`. Each one is re-wrapped by a VM arrow whose
// closure holds the host reference out of reach.
const wrapAsyncInVm = evalInVm<
(hostFn: (...args: unknown[]) => Promise<unknown>) => unknown
>('(hostFn => async (...args) => hostFn(...args))')
const wrapSyncInVm = evalInVm<(hostFn: (...args: unknown[]) => void) => unknown>(
'(hostFn => (...args) => { hostFn(...args) })',
)
/** Rebuild a host value inside the VM realm. */
const intake = (value: unknown): unknown => {
if (value === undefined) return undefined
if (value === null) return null
const primitive = typeof value
if (primitive === 'string' || primitive === 'number' || primitive === 'boolean') {
return value
}
let json: string | undefined
try {
json = JSON.stringify(value)
} catch {
json = undefined
}
if (json === undefined) return null
return parseInVm(json)
}
/** Rebuild a VM array of already-VM elements without copying the elements. */
const intakeArray = (values: unknown[]): unknown[] => {
const array = newArrayInVm(values.length)
for (let i = 0; i < values.length; i++) setIndexInVm(array, i, values[i])
return array
}
const timers = new Set<NodeJS.Timeout>()
const clearAllTimers = (): void => {
for (const timer of timers) clearTimeout(timer)
timers.clear()
}
abortSignal?.addEventListener('abort', clearAllTimers, { once: true })
const defineGlobal = (name: string, value: unknown): void => {
Object.defineProperty(context, name, {
value,
writable: true,
enumerable: true,
configurable: true,
})
}
defineGlobal(
'agent',
wrapAsyncInVm(async (prompt, opts) =>
intake(await harness.agent(prompt, opts)),
),
)
defineGlobal(
'parallel',
wrapAsyncInVm(async thunks => intakeArray(await harness.parallel(thunks))),
)
defineGlobal(
'pipeline',
wrapAsyncInVm(async (items, ...stages) =>
intakeArray(await harness.pipeline(items, ...stages)),
),
)
defineGlobal('log', wrapSyncInVm(message => harness.log(message)))
defineGlobal('phase', wrapSyncInVm(title => harness.phase(title)))
defineGlobal(
'workflow',
wrapAsyncInVm(async (nameOrRef, args) => {
if (!runNestedWorkflow) {
throw new Error('workflow() is not available in this run')
}
return intake(await runNestedWorkflow(nameOrRef, args))
}),
)
// budget is built inside the VM so `budget.spent.constructor` resolves to the
// VM's Function, not the host's.
const makeBudget = evalInVm<
(
total: number | null,
spent: () => number,
remaining: () => number,
) => unknown
>(
'((total, spent, remaining) => Object.freeze({ __proto__: null, total, spent: () => spent(), remaining: () => remaining() }))',
)
defineGlobal(
'budget',
makeBudget(
tokenBudget?.total ?? null,
() => tokenBudget?.getTurnSpent() ?? 0,
() =>
tokenBudget?.total == null
? Number.POSITIVE_INFINITY
: Math.max(0, tokenBudget.total - tokenBudget.getTurnSpent()),
),
)
const makeConsole = evalInVm<(write: (line: string) => void) => unknown>(
`(write => {
const render = args => args.map(a => {
if (typeof a === 'string') return a
try { return JSON.stringify(a) ?? String(a) } catch { return '[object]' }
}).join(' ')
const emit = (...args) => { write(render(args)) }
return Object.freeze({ __proto__: null, log: emit, info: emit, warn: emit, error: emit, debug: emit })
})`,
)
defineGlobal(
'console',
makeConsole(line => emit({ type: 'workflow_log', message: line })),
)
const makeTimers = evalInVm<
(
set: (callback: unknown, ms: unknown) => unknown,
clear: (handle: unknown) => void,
) => { setTimeout: unknown; clearTimeout: unknown }
>(
`((set, clear) => ({
__proto__: null,
setTimeout: (callback, ms) => set(callback, ms),
clearTimeout: handle => { clear(handle) },
}))`,
)
const vmTimers = makeTimers(
(callback, ms) => {
if (typeof callback !== 'function') return 0
const delay = typeof ms === 'number' && Number.isFinite(ms) ? ms : 0
const timer = setTimeout(() => {
timers.delete(timer)
try {
;(callback as () => void)()
} catch (error) {
logForDebugging(`workflow setTimeout callback threw: ${describeError(error)}`)
}
}, Math.max(0, delay))
timer.unref?.()
timers.add(timer)
return timer
},
handle => {
if (handle && typeof handle === 'object') {
clearTimeout(handle as NodeJS.Timeout)
timers.delete(handle as NodeJS.Timeout)
}
},
)
defineGlobal('setTimeout', vmTimers.setTimeout)
defineGlobal('clearTimeout', vmTimers.clearTimeout)
if (params.args !== undefined) {
const json = JSON.stringify(params.args)
defineGlobal('args', json === undefined ? undefined : parseInVm(json))
} else {
defineGlobal('args', undefined)
}
return {
context,
awaitInVm,
exportValue(value: unknown): unknown {
if (typeof value === 'function') {
throw new Error('workflow result cannot be a function')
}
if (value === undefined) return null
const json = stringifyInVm(value)
if (json === undefined) return null
return JSON.parse(json) as unknown
},
dispose: clearAllTimers,
}
}
const describeError = describeThrown
/**
* Parse + compile a script in one step. Used by the tool, the resume path, and
* the tests so all three reject the same scripts for the same reasons.
*/
export function prepareWorkflowScript(
script: string,
): PreparedWorkflowScript {
const parsed = parseWorkflowScript(script)
if ('error' in parsed) return { ok: false, error: parsed.error }
const compiled = compileWorkflowScript(parsed.scriptBody)
if (!compiled.ok) return { ok: false, error: compiled.error }
return {
ok: true,
meta: parsed.meta,
scriptBody: parsed.scriptBody,
vmScript: compiled.vmScript,
}
}
export type PreparedWorkflowScript =
| {
ok: true
meta: WorkflowMeta
scriptBody: string
vmScript: vm.Script
}
| { ok: false; error: string }
+105
View File
@@ -0,0 +1,105 @@
import { afterEach, beforeEach, describe, expect, test } from 'bun:test'
import { mkdirSync, symlinkSync } from 'fs'
import { mkdtemp, readFile, rm, writeFile } from 'fs/promises'
import { tmpdir } from 'os'
import { join } from 'path'
import { resolveProjectWorkflowsDir, saveWorkflowScript } from './save.js'
let root: string
let configDir: string
let repo: string
const originalConfigDir = process.env.CLAUDE_CONFIG_DIR
const SCRIPT = [
"export const meta = { name: 'my-review', description: 'Review the diff' }",
"return await agent('review')",
].join('\n')
describe('saveWorkflowScript', () => {
beforeEach(async () => {
root = await mkdtemp(join(tmpdir(), 'wf-save-'))
configDir = join(root, 'claude')
repo = join(root, 'repo')
mkdirSync(configDir, { recursive: true })
mkdirSync(join(repo, '.git'), { recursive: true })
process.env.CLAUDE_CONFIG_DIR = configDir
})
afterEach(async () => {
if (originalConfigDir === undefined) delete process.env.CLAUDE_CONFIG_DIR
else process.env.CLAUDE_CONFIG_DIR = originalConfigDir
await rm(root, { recursive: true, force: true })
})
test('saves to the personal directory under CLAUDE_CONFIG_DIR', async () => {
const result = await saveWorkflowScript({ script: SCRIPT, scope: 'user' })
expect(result).toEqual({
name: 'my-review',
filePath: join(configDir, 'workflows', 'my-review.js'),
})
expect(await readFile(join(configDir, 'workflows', 'my-review.js'), 'utf8')).toBe(
SCRIPT,
)
})
test('saves to the repository root when no .claude/workflows exists yet', async () => {
const result = await saveWorkflowScript({
script: SCRIPT,
scope: 'project',
cwd: join(repo, 'packages', 'api'),
})
expect('filePath' in result && result.filePath).toBe(
join(repo, '.claude', 'workflows', 'my-review.js'),
)
})
test('prefers the closest existing .claude/workflows in a monorepo', async () => {
const pkg = join(repo, 'packages', 'api')
mkdirSync(join(pkg, '.claude', 'workflows'), { recursive: true })
mkdirSync(join(repo, '.claude', 'workflows'), { recursive: true })
expect(resolveProjectWorkflowsDir(pkg)).toBe(
join(pkg, '.claude', 'workflows'),
)
const result = await saveWorkflowScript({
script: SCRIPT,
scope: 'project',
cwd: pkg,
})
expect('filePath' in result && result.filePath).toBe(
join(pkg, '.claude', 'workflows', 'my-review.js'),
)
})
test('rejects a script whose meta will not parse', async () => {
const result = await saveWorkflowScript({
script: 'const x = 1\n',
scope: 'user',
})
expect('error' in result && result.error).toContain('FIRST statement')
})
test('refuses to write through a symlinked target file', async () => {
const outside = join(root, 'outside.js')
await writeFile(outside, '// untouched\n', 'utf8')
mkdirSync(join(configDir, 'workflows'), { recursive: true })
symlinkSync(outside, join(configDir, 'workflows', 'my-review.js'))
const result = await saveWorkflowScript({ script: SCRIPT, scope: 'user' })
expect('error' in result && result.error).toContain('symlink')
expect(await readFile(outside, 'utf8')).toBe('// untouched\n')
})
test('refuses when the project .claude directory is itself a symlink', async () => {
const elsewhere = join(root, 'elsewhere')
mkdirSync(elsewhere, { recursive: true })
symlinkSync(elsewhere, join(repo, '.claude'))
const result = await saveWorkflowScript({
script: SCRIPT,
scope: 'project',
cwd: repo,
})
expect('error' in result && result.error).toContain('symlink')
})
})
+98
View File
@@ -0,0 +1,98 @@
import { lstat, mkdir, writeFile } from 'fs/promises'
import { join } from 'path'
import { getOriginalCwd } from '../../bootstrap/state.js'
import { getProjectDirsUpToHome } from '../markdownConfigLoader.js'
import { findGitRoot } from '../git.js'
import { parseWorkflowScript } from './meta.js'
import { getUserWorkflowsDir } from './paths.js'
export type WorkflowSaveScope = 'user' | 'project'
export type WorkflowSaveResult =
| { name: string; filePath: string }
| { error: string }
/**
* Save a run's script as a `/name` command.
*
* Refuses to write through a symlink. For the project scope the check is
* wider than for the personal one: `.claude` there is repo-controlled, so a
* symlinked `.claude` or `.claude/workflows` could redirect the write out of
* the repository entirely. A `~/.claude` managed by a dotfiles tool is a
* normal setup, so only the target file itself is checked there.
*/
export async function saveWorkflowScript(params: {
script: string
scope: WorkflowSaveScope
cwd?: string
}): Promise<WorkflowSaveResult> {
const parsed = parseWorkflowScript(params.script)
if ('error' in parsed) return { error: parsed.error }
const cwd = params.cwd ?? getOriginalCwd()
const dir =
params.scope === 'project'
? resolveProjectWorkflowsDir(cwd)
: getUserWorkflowsDir()
if (params.scope === 'project') {
for (const candidate of [
join(dirOf(dir), '.claude'),
dir,
join(dir, `${parsed.meta.name}.js`),
]) {
const link = await isSymlink(candidate)
if (link) return { error: `Refusing to write through a symlink: ${candidate}` }
}
} else if (await isSymlink(join(dir, `${parsed.meta.name}.js`))) {
return {
error: `Refusing to write through a symlink: ${join(dir, `${parsed.meta.name}.js`)}`,
}
}
const filePath = join(dir, `${parsed.meta.name}.js`)
try {
await mkdir(dir, { recursive: true })
await writeFile(filePath, params.script, 'utf8')
} catch (error) {
return {
error: error instanceof Error ? error.message : String(error),
}
}
return { name: parsed.meta.name, filePath }
}
/**
* Where a project save lands in a monorepo.
*
* The closest existing `.claude/workflows` between cwd and the repo root wins,
* so a workflow saved while working in `packages/api` stays with that package
* instead of being hoisted to the root and shown to every other package.
*/
export function resolveProjectWorkflowsDir(cwd: string): string {
const existing = safeProjectDirs(cwd)
if (existing.length > 0) return existing[0]!
const root = findGitRoot(cwd) ?? cwd
return join(root, '.claude', 'workflows')
}
function safeProjectDirs(cwd: string): string[] {
try {
return getProjectDirsUpToHome('workflows', cwd)
} catch {
return []
}
}
function dirOf(workflowsDir: string): string {
// `<x>/.claude/workflows` → `<x>`; the caller re-appends `.claude`.
return join(workflowsDir, '..', '..')
}
async function isSymlink(path: string): Promise<boolean> {
try {
return (await lstat(path)).isSymbolicLink()
} catch {
return false
}
}
+147
View File
@@ -0,0 +1,147 @@
/**
* Shared shapes for dynamic workflows.
*
* A dynamic workflow is a plain-JavaScript script that orchestrates subagents.
* The script never touches the filesystem or the network itself: it calls
* `agent()` / `parallel()` / `pipeline()` and the runtime spawns real subagents
* on its behalf. Everything the UI renders while a run is in flight travels as
* `WorkflowProgressEvent`s, so these shapes are the contract between the
* runtime, the Ink TUI, the local server, and the desktop app.
*/
/** One entry of `meta.phases` — the phase list shown before a run is approved. */
export type WorkflowPhaseMeta = {
title: string
detail?: string
model?: string
}
/**
* The `export const meta = {...}` block every script must start with.
* Parsed as a pure literal (no identifiers, calls, or interpolation) so that
* listing saved workflows never executes untrusted code.
*/
export type WorkflowMeta = {
name: string
description: string
whenToUse?: string
title?: string
phases?: WorkflowPhaseMeta[]
model?: string
}
/** Where a workflow definition was loaded from. */
export type WorkflowSource =
| 'built-in'
| 'userSettings'
| 'projectSettings'
| 'plugin'
| 'inline'
/** A discovered, runnable workflow definition. */
export type WorkflowDefinition = {
source: WorkflowSource
name: string
description: string
whenToUse?: string
phases?: WorkflowPhaseMeta[]
script: string
filePath?: string
}
export type WorkflowAgentRunState = 'start' | 'progress' | 'done' | 'error'
/** Lifecycle of one `agent()` call. Keyed by `index` (1-based, run-scoped). */
export type WorkflowAgentEvent = {
type: 'workflow_agent'
index: number
label: string
state: WorkflowAgentRunState
phaseIndex?: number
phaseTitle?: string
agentType?: string
isolation?: 'worktree' | 'remote'
model?: string
agentId?: string
queuedAt?: number
startedAt?: number
lastProgressAt?: number
durationMs?: number
tokens?: number
toolCalls?: number
/** Replayed from the resume journal instead of re-run. */
cached?: boolean
/** Stopped by the user rather than failed on its own. */
skipped?: boolean
/** Refused before spawning (permission rule or safety check). */
blocked?: boolean
error?: string
promptPreview?: string
resultPreview?: string
lastToolName?: string
}
/** A `phase()` call, or a seeded entry from `meta.phases`. */
export type WorkflowPhaseEvent = {
type: 'workflow_phase'
index: number
title: string
/** `'meta'` for entries seeded from meta.phases, `'script'` for phase() calls. */
kind?: 'meta' | 'script'
}
/** A `log()` call, or a runtime note (a failed pipeline slot, a stall warning). */
export type WorkflowLogEvent = {
type: 'workflow_log'
message: string
}
export type WorkflowProgressEvent =
| WorkflowAgentEvent
| WorkflowPhaseEvent
| WorkflowLogEvent
export function isWorkflowAgentEvent(
event: WorkflowProgressEvent,
): event is WorkflowAgentEvent {
return event.type === 'workflow_agent'
}
export function isWorkflowPhaseEvent(
event: WorkflowProgressEvent,
): event is WorkflowPhaseEvent {
return event.type === 'workflow_phase'
}
/** Progress rows worth persisting in the run summary — logs are transient. */
export function isDurableWorkflowEvent(event: WorkflowProgressEvent): boolean {
return event.type !== 'workflow_log'
}
/** Options a script may pass as the second argument to `agent()`. */
export type WorkflowAgentOptions = {
label?: string
phase?: string
schema?: unknown
model?: string
effort?: string
isolation?: 'worktree' | 'remote'
agentType?: string
stallMs?: number
}
/** Outcome of one script execution, before it is turned into a tool result. */
export type WorkflowRunOutcome = {
result: unknown
agentCount: number
logs: string[]
failures: string[]
durationMs: number
error?: string
}
/** The token allowance a run may spend, threaded in from the parent turn. */
export type WorkflowTokenBudget = {
total: number | null
getTurnSpent: () => number
}
+98
View File
@@ -0,0 +1,98 @@
import { afterEach, beforeEach, describe, expect, test } from 'bun:test'
import { mkdirSync } from 'fs'
import { mkdtemp, rm, writeFile } from 'fs/promises'
import { tmpdir } from 'os'
import { join } from 'path'
import { executeEffort } from '../../commands/effort/effort.js'
import { resetSettingsCache } from '../settings/settingsCache.js'
import { modelSupportsXHighEffort } from '../effort.js'
import { canEnableUltracode } from './ultracode.js'
/**
* A model the capability table accepts in this environment.
*
* `modelSupportsXHighEffort` is gated on provider trust, so hardcoding a name
* makes the test depend on the developer's provider config rather than on
* ultracode's own logic.
*/
const XHIGH_MODEL = ['claude-opus-4-7', 'claude-sonnet-5', 'claude-fable-5'].find(
modelSupportsXHighEffort,
)
let home: string
let configDir: string
const originalConfigDir = process.env.CLAUDE_CONFIG_DIR
const originalEffortEnv = process.env.CLAUDE_CODE_EFFORT_LEVEL
async function writeSettings(value: Record<string, unknown>): Promise<void> {
await writeFile(join(configDir, 'settings.json'), JSON.stringify(value), 'utf8')
resetSettingsCache()
}
describe('ultracode', () => {
beforeEach(async () => {
home = await mkdtemp(join(tmpdir(), 'wf-ultracode-'))
configDir = join(home, 'claude')
mkdirSync(configDir, { recursive: true })
process.env.CLAUDE_CONFIG_DIR = configDir
delete process.env.CLAUDE_CODE_EFFORT_LEVEL
await writeSettings({})
})
afterEach(async () => {
if (originalConfigDir === undefined) delete process.env.CLAUDE_CONFIG_DIR
else process.env.CLAUDE_CONFIG_DIR = originalConfigDir
if (originalEffortEnv === undefined) delete process.env.CLAUDE_CODE_EFFORT_LEVEL
else process.env.CLAUDE_CODE_EFFORT_LEVEL = originalEffortEnv
resetSettingsCache()
await rm(home, { recursive: true, force: true })
})
test('requires an xhigh-capable model', () => {
const refused = canEnableUltracode('deepseek-v4-flash')
expect(refused.ok).toBe(false)
if (refused.ok) return
expect(refused.reason).toBe('model')
})
test('requires workflows to be enabled', async () => {
await writeSettings({ disableWorkflows: true })
const refused = canEnableUltracode(XHIGH_MODEL ?? 'claude-opus-4-7')
expect(refused.ok).toBe(false)
if (refused.ok) return
// The workflow gate is checked before the model gate, so this reason holds
// whether or not the capability table trusts the model here.
expect(refused.reason).toBe('workflows-disabled')
})
test('/effort ultracode resolves to xhigh plus the session flag', () => {
if (!XHIGH_MODEL) {
// No xhigh-capable model is trusted in this environment; the refusal
// path is covered above and asserting here would test the provider
// config, not ultracode.
expect(canEnableUltracode('claude-opus-4-7').ok).toBe(false)
return
}
const result = executeEffort('ultracode', XHIGH_MODEL)
expect(result.effortUpdate).toEqual({ value: 'xhigh', ultracode: true })
expect(result.message).toContain('this session only')
})
test('/effort ultracode explains itself when the model cannot do xhigh', () => {
const result = executeEffort('ultracode', 'deepseek-v4-flash')
expect(result.effortUpdate).toBeUndefined()
expect(result.message).toContain("doesn't support")
})
test('/effort ultracode is refused when workflows are off', async () => {
await writeSettings({ enableWorkflows: false })
const result = executeEffort('ultracode', XHIGH_MODEL ?? 'claude-opus-4-7')
expect(result.effortUpdate).toBeUndefined()
expect(result.message).toContain('dynamic workflows enabled')
})
test('an ordinary level never sets the flag', () => {
const result = executeEffort('high', XHIGH_MODEL ?? 'claude-opus-4-7')
expect(result.effortUpdate?.ultracode).toBeUndefined()
})
})
+62
View File
@@ -0,0 +1,62 @@
/**
* Ultracode: xhigh effort plus standing dynamic-workflow orchestration.
*
* Not a sixth effort level — it is `xhigh` with a flag beside it. Keeping it
* out of `EffortLevel` matters: every model-capability check, settings write,
* and picker row already reasons about the five real levels, and a sixth would
* have to be special-cased in all of them to mean "xhigh, but also…".
*/
import { modelSupportsXHighEffort } from '../effort.js'
import { areWorkflowsEnabled } from './enabled.js'
export const ULTRACODE_EFFORT_ARG = 'ultracode'
/** Effort ultracode resolves to once the flag is recorded separately. */
export const ULTRACODE_EFFORT_LEVEL = 'xhigh' as const
export const ULTRACODE_MENU_DESCRIPTION =
'xhigh effort + dynamic workflows for maximum thoroughness'
export const ULTRACODE_ENTER_REMINDER =
'Ultracode is on: optimize for the most exhaustive, correct answer — not the fastest or cheapest. ' +
"Use the Workflow tool on every substantive task; token cost is not a constraint. See the Workflow tool's " +
'**Ultracode** section and quality patterns. Solo only on conversational/trivial turns.'
export const ULTRACODE_STILL_ON_REMINDER =
'Ultracode is still on — use the Workflow tool; see its Ultracode section.'
export const ULTRACODE_EXIT_REMINDER =
"Ultracode is off — the Workflow tool's standard opt-in rule applies again."
export const WORKFLOW_KEYWORD_REMINDER =
'The user included the keyword "ultracode", opting this turn into multi-agent orchestration — ' +
'use the Workflow tool to fulfill the request.'
export type UltracodeRequest =
| { ok: true }
| { ok: false; reason: 'workflows-disabled' | 'model' }
/**
* Whether ultracode can be turned on right now.
*
* Both preconditions are real: without workflows there is nothing to
* orchestrate, and without xhigh the "ultra" half of the level is a lie.
*/
export function canEnableUltracode(model: string): UltracodeRequest {
if (!areWorkflowsEnabled()) return { ok: false, reason: 'workflows-disabled' }
if (!modelSupportsXHighEffort(model)) return { ok: false, reason: 'model' }
return { ok: true }
}
export function describeUltracodeRefusal(
reason: Exclude<UltracodeRequest, { ok: true }>['reason'],
model: string,
): string {
switch (reason) {
case 'workflows-disabled':
return 'Ultracode needs dynamic workflows enabled (see /config).'
case 'model':
return `Ultracode runs at xhigh effort, which ${model} doesn't support — switch to an xhigh-capable model.`
}
}