diff --git a/bun.lock b/bun.lock index 8f4c5574..7e4ee14a 100644 --- a/bun.lock +++ b/bun.lock @@ -18,6 +18,8 @@ "@opentelemetry/sdk-metrics": "^2.6.1", "@opentelemetry/sdk-trace-base": "^2.6.1", "@opentelemetry/semantic-conventions": "^1.40.0", + "acorn": "^8.16.0", + "acorn-walk": "^8.3.4", "ajv": "^8.18.0", "asciichart": "^1.5.25", "auto-bind": "^5.0.1", @@ -370,6 +372,8 @@ "acorn": ["acorn@8.16.0", "https://registry.npmmirror.com/acorn/-/acorn-8.16.0.tgz", { "bin": { "acorn": "bin/acorn" } }, "sha512-UVJyE9MttOsBQIDKw1skb9nAwQuR5wuGD3+82K6JgJlm/Y+KI92oNsMNGZCYdDsVtRHSak0pcV5Dno5+4jh9sw=="], + "acorn-walk": ["acorn-walk@8.3.5", "https://registry.npmmirror.com/acorn-walk/-/acorn-walk-8.3.5.tgz", { "dependencies": { "acorn": "^8.11.0" } }, "sha512-HEHNfbars9v4pgpW6SO1KSPkfoS0xVOM/9UzkJltjlsHZmJasxg8aXkuZa7SMf8vKGIBhpUsPluQSqhJFCqebw=="], + "agent-base": ["agent-base@8.0.0", "https://registry.npmmirror.com/agent-base/-/agent-base-8.0.0.tgz", {}, "sha512-QT8i0hCz6C/KQ+KTAbSNwCHDGdmUJl2tp2ZpNlGSWCfhUNVbYG2WLE3MdZGBAgXPV4GAvjGMxo+C1hroyxmZEg=="], "ajv": ["ajv@8.18.0", "https://registry.npmmirror.com/ajv/-/ajv-8.18.0.tgz", { "dependencies": { "fast-deep-equal": "^3.1.3", "fast-uri": "^3.0.1", "json-schema-traverse": "^1.0.0", "require-from-string": "^2.0.2" } }, "sha512-PlXPeEWMXMZ7sPYOHqmDyCJzcfNrUr3fGNKtezX14ykXOEIvyK81d+qydx89KY5O71FKMPaQ2vBfBFI5NHR63A=="], diff --git a/desktop/src/api/subagents.ts b/desktop/src/api/subagents.ts index 37a6cecc..e169c3c8 100644 --- a/desktop/src/api/subagents.ts +++ b/desktop/src/api/subagents.ts @@ -35,7 +35,36 @@ export type SubagentRunResponse = { canSendMessage?: boolean } +/** + * Marks a subagent addressed by agent id rather than by the `Agent` tool call + * that spawned it. + * + * Workflow agents are spawned by the workflow runtime, so no such tool call + * exists. Carrying the distinction in the identifier means the tab id, the + * page, and the return path all stay exactly as they are for every other + * subagent — only the fetch differs. + */ +export const AGENT_ID_REF_PREFIX = 'agent:' + +export function isAgentIdRef(ref: string): boolean { + return ref.startsWith(AGENT_ID_REF_PREFIX) +} + +export function toAgentIdRef(agentId: string): string { + return `${AGENT_ID_REF_PREFIX}${agentId}` +} + +export function readAgentIdRef(ref: string): string { + return ref.slice(AGENT_ID_REF_PREFIX.length) +} + export const subagentsApi = { + getRunByAgent(sessionId: string, agentId: string) { + return api.get( + `/api/sessions/${encodeURIComponent(sessionId)}/subagents/by-agent/${encodeURIComponent(agentId)}`, + ) + }, + getRunByTool(sessionId: string, toolUseId: string, taskId?: string) { const query = taskId ? `?taskId=${encodeURIComponent(taskId)}` : '' return api.get( diff --git a/desktop/src/api/workflows.ts b/desktop/src/api/workflows.ts new file mode 100644 index 00000000..16f52a5a --- /dev/null +++ b/desktop/src/api/workflows.ts @@ -0,0 +1,71 @@ +import { api } from './client' +import type { + ReconstructedWorkflowRun, + WorkflowDefinition, + WorkflowRunDetail, + WorkflowRunSummary, +} from '../types/workflow' + +export type WorkflowValidateResult = { + ok: boolean + error?: string + name?: string + description?: string + phases?: { title: string; detail?: string; model?: string }[] +} + +export const workflowsApi = { + list(cwd?: string) { + const query = cwd ? `?cwd=${encodeURIComponent(cwd)}` : '' + return api.get<{ workflows: WorkflowDefinition[] }>(`/api/workflows${query}`) + }, + + get(name: string, cwd?: string) { + const query = cwd ? `?cwd=${encodeURIComponent(cwd)}` : '' + return api.get( + `/api/workflows/${encodeURIComponent(name)}${query}`, + ) + }, + + /** Finished runs for one session, rebuilt from what the CLI left on disk. */ + sessionRuns(sessionId: string) { + return api.get<{ runs: ReconstructedWorkflowRun[] }>( + `/api/workflows/session-runs/${encodeURIComponent(sessionId)}`, + ) + }, + + listRuns(options?: { sessionId?: string; limit?: number }) { + const params = new URLSearchParams() + if (options?.sessionId) params.set('sessionId', options.sessionId) + if (options?.limit) params.set('limit', String(options.limit)) + const query = params.toString() + return api.get<{ runs: WorkflowRunSummary[] }>( + `/api/workflows/runs${query ? `?${query}` : ''}`, + ) + }, + + getRun(sessionId: string, runId: string) { + return api.get( + `/api/workflows/runs/${encodeURIComponent(sessionId)}/${encodeURIComponent(runId)}`, + ) + }, + + validate(script: string) { + return api.post('/api/workflows/validate', { script }) + }, + + save(script: string, scope: 'user' | 'project', cwd?: string) { + return api.post<{ ok: true; name: string; filePath: string }>( + '/api/workflows/save', + { script, scope, cwd }, + ) + }, + + remove(name: string, scope: 'user' | 'project', cwd?: string) { + const params = new URLSearchParams({ scope }) + if (cwd) params.set('cwd', cwd) + return api.delete<{ ok: true }>( + `/api/workflows/${encodeURIComponent(name)}?${params.toString()}`, + ) + }, +} diff --git a/desktop/src/components/activity/SessionActivityPanel.test.tsx b/desktop/src/components/activity/SessionActivityPanel.test.tsx index 697100b7..2af06d25 100644 --- a/desktop/src/components/activity/SessionActivityPanel.test.tsx +++ b/desktop/src/components/activity/SessionActivityPanel.test.tsx @@ -34,6 +34,7 @@ vi.mock('../../i18n', () => ({ 'session.activity.details.usage': 'Usage', 'session.activity.section.tasks': 'Tasks', 'session.activity.section.team': 'Team', + 'session.activity.section.workflow': 'Workflow', 'session.activity.section.backgroundTasks': 'Background Tasks', 'session.activity.section.subagents': 'SubAgents', 'session.activity.section.sources': 'Sources', @@ -75,6 +76,7 @@ function model(overrides: Partial = {}): SessionActivityMo badgeCount: 1, sections: { output: { id: 'output', title: 'Output', emptyLabel: 'No output', rows: [] }, + workflow: { id: 'workflow', title: 'Workflow', emptyLabel: 'No workflow running', rows: [] }, tasks: { id: 'tasks', title: 'Tasks', @@ -844,6 +846,73 @@ describe('SessionActivityPanel', () => { expect(onOpenSubagent).not.toHaveBeenCalled() }) + it('renders a workflow as phase headers with their agents, each opening the subagent page', () => { + const onOpenSubagent = vi.fn() + render( + , + ) + + // The phase is a heading, not something you can open. + const header = screen.getByTestId('workflow-phase-header') + expect(header).toHaveTextContent('Survey') + expect(header).toHaveTextContent('2/2') + + // Its agent opens the ordinary subagent page, addressed by agent id + // because a workflow agent has no parent Agent tool call. + fireEvent.click(screen.getByRole('button', { name: /open run survey response\.js/i })) + expect(onOpenSubagent).toHaveBeenCalledWith( + expect.objectContaining({ toolUseId: 'agent:a11', title: 'survey response.js' }), + ) + + // A queued agent has no transcript yet, so it must not offer to open one. + expect( + screen.queryByRole('button', { name: /open run check response #2/i }), + ).not.toBeInTheDocument() + expect(screen.getByText('check response #2')).toBeInTheDocument() + }) + it('does not render when closed', () => { render() diff --git a/desktop/src/components/activity/SessionActivityPanel.tsx b/desktop/src/components/activity/SessionActivityPanel.tsx index 632a4b2c..64d996d8 100644 --- a/desktop/src/components/activity/SessionActivityPanel.tsx +++ b/desktop/src/components/activity/SessionActivityPanel.tsx @@ -1,5 +1,5 @@ import { useCallback, useEffect, useMemo, useRef, useState } from 'react' -import { Check, ChevronRight, Circle, FileText, LoaderCircle, Square, Terminal, Users, X } from 'lucide-react' +import { Check, ChevronRight, Circle, FileText, LoaderCircle, Square, Terminal, Users, X, Zap } from 'lucide-react' import { Badge, StatusDot, type Tone } from '@/components/ui/Badge' import { Button } from '@/components/ui/Button' import { IconButton } from '@/components/ui/IconButton' @@ -72,6 +72,8 @@ function getSectionTitle(sectionId: ActivitySectionId, t: TranslationFn): string return t('session.activity.section.tasks') case 'team': return t('session.activity.section.team') + case 'workflow': + return t('session.activity.section.workflow') case 'backgroundTasks': return t('session.activity.section.backgroundTasks') case 'subagents': @@ -92,6 +94,8 @@ function getSectionRowsClassName(sectionId: ActivitySectionId, rowCount: number) return base case 'team': return base + case 'workflow': + return base case 'backgroundTasks': return base case 'subagents': @@ -201,6 +205,8 @@ function getRowIcon(row: ActivityRow) { switch (row.section) { case 'team': return Users + case 'workflow': + return Zap case 'backgroundTasks': return Terminal case 'subagents': @@ -255,7 +261,9 @@ function ActivityRowIcon({ sessionId: string status?: ActivityRow['status'] }) { - if (row.section === 'subagents') { + // Workflow agents get the same mascot as any other subagent — they are the + // same thing, and giving them a different glyph would imply otherwise. + if (row.section === 'subagents' || (row.section === 'workflow' && row.group)) { return } @@ -328,6 +336,41 @@ function BackgroundTaskStopButton({ ) } +/** + * A phase heading inside the workflow section. + * + * Deliberately not a card: the section is already a bordered list, and boxing + * each phase inside it turned three stages into three nested frames. A rule + * plus the settled count carries the grouping on its own. + */ +function WorkflowPhaseHeader({ + label, + status, + done, + total, +}: { + label: string + status: ActivityRow['status'] + done: number + total: number +}) { + return ( +
+ + {label} + +
+ ) +} + function ActivityRowView({ row, sessionId, @@ -348,6 +391,19 @@ function ActivityRowView({ selected?: boolean }) { const t = useTranslation() + // A workflow phase is a heading over its agents, not a row you can open. + // Rendering it as one made a fan-out read as a flat list where the stage + // boundaries were invisible. + if (row.groupProgress) { + return ( + + ) + } const isTask = row.section === 'tasks' const isStoppingSubagent = row.section === 'subagents' && row.status === 'running' && stoppingBackgroundTask const displayStatus: ActivityRow['status'] = isStoppingSubagent ? 'pending' : row.status @@ -430,7 +486,13 @@ function ActivityRowView({ ) } - if (row.section === 'subagents' && row.openable && row.toolUseId) { + // A workflow agent is an ordinary subagent, so it opens the same page by the + // same handler — there is nothing workflow-specific to render for one. + if ( + (row.section === 'subagents' || row.section === 'workflow') && + row.openable && + row.toolUseId + ) { const openButton = ( diff --git a/desktop/src/components/activity/sessionActivityModel.test.ts b/desktop/src/components/activity/sessionActivityModel.test.ts index d373f59f..285a546f 100644 --- a/desktop/src/components/activity/sessionActivityModel.test.ts +++ b/desktop/src/components/activity/sessionActivityModel.test.ts @@ -1720,3 +1720,110 @@ describe('buildSessionActivityModel', () => { expect(model.badgeCount).toBe(1) }) }) + +describe('workflow section', () => { + const AGENTS = [ + { type: 'workflow_agent', index: 1, label: 'survey response.js', state: 'done', phaseIndex: 1, phaseTitle: 'Survey', agentId: 'a11', tokens: 24_100 }, + { type: 'workflow_agent', index: 2, label: 'survey request.js', state: 'done', phaseIndex: 1, phaseTitle: 'Survey', agentId: 'a12' }, + { type: 'workflow_agent', index: 3, label: 'check response #1', state: 'progress', phaseIndex: 2, phaseTitle: 'Cross-check', agentId: 'a13' }, + // Queued: accepted by the runtime but never given a slot, so no transcript. + { type: 'workflow_agent', index: 4, label: 'check response #2', state: 'start', phaseIndex: 2, phaseTitle: 'Cross-check' }, + ] + + function run(overrides: Record = {}) { + return { + taskId: 'w1', + sessionId: 'session-1', + workflowName: 'route-survey', + status: 'running', + startedAt: 0, + updatedAt: 0, + agentCount: 4, + totalTokens: 0, + toolCalls: 0, + progress: [ + { type: 'workflow_phase', index: 1, title: 'Survey' }, + { type: 'workflow_phase', index: 2, title: 'Cross-check' }, + ...AGENTS, + ], + ...overrides, + } as never + } + + function build() { + return buildSessionActivityModel({ + sessionId: 'session-1', + tasks: [], + completedAndDismissed: false, + backgroundTasks: [], + agentNotifications: [], + workflowRuns: [run()], + }) + } + + it('lays each phase out as a header followed by its agents', () => { + const rows = build().sections.workflow.rows + expect(rows.map((row) => [row.label, row.groupProgress ? 'phase' : row.group])).toEqual([ + ['Survey', 'phase'], + ['survey response.js', 'Survey'], + ['survey request.js', 'Survey'], + ['Cross-check', 'phase'], + ['check response #1', 'Cross-check'], + ['check response #2', 'Cross-check'], + ]) + }) + + it('counts settled agents on the phase header', () => { + const headers = build().sections.workflow.rows.filter((row) => row.groupProgress) + expect(headers[0]!.groupProgress).toEqual({ done: 2, total: 2 }) + expect(headers[0]!.status).toBe('completed') + expect(headers[1]!.groupProgress).toEqual({ done: 0, total: 2 }) + expect(headers[1]!.status).toBe('running') + }) + + it('opens each agent through the ordinary subagent route', () => { + // A workflow agent is a subagent run by the same runner, so the row carries + // the reference the existing page opens with rather than anything bespoke. + const rows = build().sections.workflow.rows + const running = rows.find((row) => row.label === 'check response #1')! + expect(running.openable).toBe(true) + expect(running.toolUseId).toBe('agent:a13') + + // Queued agents have no transcript yet — offering to open one would 404. + const queued = rows.find((row) => row.label === 'check response #2')! + expect(queued.openable).toBe(false) + expect(queued.toolUseId).toBeUndefined() + }) + + it('labels an unphased group with the run name instead of "Phase 0"', () => { + // Runs recorded before phases were persisted come back ungrouped. The + // workflow name identifies them; a bare index does not. + const model = buildSessionActivityModel({ + sessionId: 'session-1', + tasks: [], + completedAndDismissed: false, + backgroundTasks: [], + agentNotifications: [], + workflowRuns: [run({ + workflowName: 'review-last-month', + progress: [ + { type: 'workflow_agent', index: 1, label: 'review:security', state: 'done', phaseIndex: 0, agentId: 'a1' }, + ], + })], + }) + const header = model.sections.workflow.rows.find((row) => row.groupProgress)! + expect(header.label).toBe('review-last-month') + }) + + it('badges only the agents, never the phase headers', () => { + // One running plus one queued agent. The Cross-check header is also + // "running", but counting it would double-count the very agents beneath + // it — the badge is a count of work, not of headings. + expect(build().badgeCount).toBe(2) + }) + + it('shows the workflow above the individual subagents it spawned', () => { + const order = getVisibleActivitySections(build()).map((section) => section.id) + expect(order).toEqual(['workflow']) + }) +}) diff --git a/desktop/src/components/activity/sessionActivityModel.ts b/desktop/src/components/activity/sessionActivityModel.ts index a0b3f27b..e44740b8 100644 --- a/desktop/src/components/activity/sessionActivityModel.ts +++ b/desktop/src/components/activity/sessionActivityModel.ts @@ -3,10 +3,12 @@ import type { TaskSummaryItem, UIMessage } from '../../types/chat' import type { CLITask, TaskStatus } from '../../types/cliTask' import type { TeamMember } from '../../types/team' import { createBackgroundTaskDismissKey } from '../../lib/backgroundTasks' +import { toAgentIdRef } from '../../api/subagents' +import type { WorkflowAgentEvent, WorkflowRun } from '../../types/workflow' export type ActivityStatus = TaskStatus | BackgroundAgentTask['status'] | TeamMember['status'] -export type ActivitySectionId = 'output' | 'tasks' | 'team' | 'backgroundTasks' | 'subagents' | 'sources' +export type ActivitySectionId = 'output' | 'tasks' | 'team' | 'workflow' | 'backgroundTasks' | 'subagents' | 'sources' export type ActivityRow = { id: string @@ -16,6 +18,14 @@ export type ActivityRow = { description?: string summary?: string toolUseId?: string + /** + * Phase this row belongs to, for sections that group. A workflow is phases + * of N agents, and the grouping is the only thing that makes a fan-out + * readable — twelve flat rows say nothing about which stage they belong to. + */ + group?: string + /** Set on a group's header row; agents under it carry `group` instead. */ + groupProgress?: { done: number; total: number } taskId?: string taskType?: BackgroundAgentTask['taskType'] workflowName?: string @@ -55,6 +65,8 @@ export type BuildSessionActivityModelInput = { dismissedBackgroundTaskKeys?: Set agentNotifications: AgentTaskNotification[] teamMembers?: TeamMember[] + /** Live workflow runs for this session, newest first. */ + workflowRuns?: WorkflowRun[] } /** @@ -64,6 +76,9 @@ export type BuildSessionActivityModelInput = { */ export const VISIBLE_ACTIVITY_SECTION_ORDER = [ 'tasks', + // A running workflow is the turn's whole shape, so it sits above the + // individual agents it spawned rather than among them. + 'workflow', 'subagents', 'team', 'backgroundTasks', @@ -76,6 +91,7 @@ const SECTION_META: Record { output: createSection('output'), tasks: createSection('tasks'), team: createSection('team'), + workflow: createSection('workflow'), backgroundTasks: createSection('backgroundTasks'), subagents: createSection('subagents'), sources: createSection('sources'), @@ -764,6 +781,96 @@ function mergeNotificationRow(existing: ActivityRow | undefined, notification: A } } +/** + * Flatten a workflow run into phase headers each followed by its agents. + * + * The agents are ordinary subagents, so every row carries the reference the + * existing subagent page opens with — there is nothing workflow-specific to + * render for one of them. An agent that has not been given a concurrency slot + * yet has no transcript to open, so it is listed but not openable. + */ +function buildWorkflowRows(run: WorkflowRun): ActivityRow[] { + const phaseTitles = new Map() + const agentsByPhase = new Map() + + for (const event of run.progress) { + if (event.type === 'workflow_phase') { + if (!phaseTitles.has(event.index)) phaseTitles.set(event.index, event.title) + if (!agentsByPhase.has(event.index)) agentsByPhase.set(event.index, []) + continue + } + const phaseIndex = event.phaseIndex ?? 0 + if (!phaseTitles.has(phaseIndex)) { + phaseTitles.set(phaseIndex, event.phaseTitle ?? '') + } + const bucket = agentsByPhase.get(phaseIndex) ?? [] + bucket.push(event) + agentsByPhase.set(phaseIndex, bucket) + } + + const rows: ActivityRow[] = [] + for (const [phaseIndex, title] of [...phaseTitles.entries()].sort(([a], [b]) => a - b)) { + const agents = (agentsByPhase.get(phaseIndex) ?? []) + .slice() + .sort((a, b) => a.index - b.index) + if (agents.length === 0 && !title) continue + // Agents emitted before any `phase()` call — and every agent of a run + // recorded before phases were persisted — have no title. The run's name + // says more about them than "Phase 0" does, and it also tells two runs in + // the same session apart. + const groupLabel = title || run.workflowName + const done = agents.filter( + agent => agent.state === 'done' || agent.state === 'error', + ).length + + rows.push({ + id: `${run.taskId}-phase-${phaseIndex}`, + section: 'workflow', + label: groupLabel, + status: workflowPhaseStatus(agents), + groupProgress: { done, total: agents.length }, + workflowName: run.workflowName, + openable: false, + }) + + for (const agent of agents) { + rows.push({ + id: `${run.taskId}-agent-${agent.index}`, + section: 'workflow', + label: agent.label, + status: workflowAgentStatus(agent), + group: groupLabel, + summary: agent.resultPreview, + toolUseId: agent.agentId ? toAgentIdRef(agent.agentId) : undefined, + taskType: 'local_agent', + workflowName: run.workflowName, + usage: agent.tokens ? { totalTokens: agent.tokens } : undefined, + openable: Boolean(agent.agentId), + }) + } + } + + return rows +} + +function workflowAgentStatus(agent: WorkflowAgentEvent): ActivityStatus { + if (agent.state === 'done') return 'completed' + if (agent.state === 'error') return 'failed' + if (agent.state === 'progress') return 'running' + return 'pending' +} + +function workflowPhaseStatus(agents: WorkflowAgentEvent[]): ActivityStatus { + if (agents.length === 0) return 'pending' + if (agents.some(agent => agent.state === 'progress')) return 'running' + if (agents.every(agent => agent.state === 'done' || agent.state === 'error')) { + return agents.some(agent => agent.state === 'error') ? 'failed' : 'completed' + } + return agents.some(agent => agent.state === 'done' || agent.state === 'error') + ? 'running' + : 'pending' +} + function buildOutputRow(key: string, outputFile: string): ActivityRow { return { id: `output-${key}`, @@ -788,6 +895,15 @@ export function buildSessionActivityModel(input: BuildSessionActivityModelInput) } } + for (const run of input.workflowRuns ?? []) { + sections.workflow.rows.push(...buildWorkflowRows(run)) + } + for (const row of sections.workflow.rows) { + if (isBadgeStatus(row.status) && !row.groupProgress) { + badgeCount += 1 + } + } + for (const member of input.teamMembers ?? []) { sections.team.rows.push(buildTeamRow(member)) } diff --git a/desktop/src/i18n/locales/en.ts b/desktop/src/i18n/locales/en.ts index 7325e154..1f3f71eb 100644 --- a/desktop/src/i18n/locales/en.ts +++ b/desktop/src/i18n/locales/en.ts @@ -2327,6 +2327,7 @@ Row 9, all 8 cells: continuing from straight down, turning left through lower-le 'session.activity.section.tasks': 'Tasks', 'session.activity.tasksProgress': 'Task progress {completed}/{total}', 'session.activity.section.team': 'Team', + 'session.activity.section.workflow': 'Workflow', 'session.activity.section.backgroundTasks': 'Background Tasks', 'session.activity.section.subagents': 'SubAgents', 'session.activity.section.sources': 'Sources', @@ -2821,6 +2822,44 @@ Row 9, all 8 cells: continuing from straight down, turning left through lower-le 'browser.selection.navigationBody': 'Navigating will discard the {count} selected elements and revert their previews.', 'browser.selection.navigationContinue': 'Discard and continue', 'browser.selection.navigationDiscarded': 'Cleared {count} selections after the page changed.', + 'workflows.status.running': 'Running', + 'workflows.status.completed': 'Completed', + 'workflows.status.failed': 'Failed', + 'workflows.status.stopped': 'Stopped', + 'workflows.report.title': 'Dynamic workflow run', + 'workflows.report.progressLabel': 'Progress of workflow {name}', + 'workflows.report.waiting': 'Waiting for the first agent…', + 'workflows.report.result': 'Result', + 'workflows.ungroupedPhase': 'Ungrouped', + 'workflows.meta.agents': '{count} agents', + 'workflows.meta.tokens': '{tokens} tokens', + 'workflows.meta.tools': '{count} tool calls', + 'workflows.phase.counts': '{done}/{total} done', + 'workflows.phase.failed': '{count} failed', + 'workflows.agent.cached': 'cached', + 'workflows.dock.running': '{count} running', + 'workflows.dock.expand': 'Show workflow details', + 'workflows.dock.collapse': 'Hide workflow details', + 'workflows.launcher.title': 'Run a workflow', + 'workflows.launcher.hint': 'A workflow orchestrates many subagents from a saved script. It runs in the background and reports once.', + 'workflows.launcher.empty': 'No saved workflows. Save one from a run, or drop a script in .claude/workflows/.', + 'workflows.launcher.run': 'Run', + 'workflows.launcher.argsLabel': 'Input passed to the script as args', + 'workflows.workbench.title': 'Dynamic workflow workbench', + 'workflows.workbench.close': 'Close workflow workbench', + 'workflows.workbench.open': 'Open workflow workbench', + 'workflows.workbench.closeDetail': 'Close agent details', + 'workflows.workbench.counts': '{done}/{total} agents settled', + 'workflows.workbench.runningCount': '{count} running', + 'workflows.workbench.queued': '{count} queued', + 'workflows.workbench.prompt': 'Prompt', + 'workflows.workbench.detailTitle': 'Agent', + 'workflows.history.title': 'Workflow history', + 'workflows.history.empty': 'No past runs on this machine yet.', + 'workflows.history.pick': 'Pick a run to see its script and per-agent results.', + 'workflows.history.agents': '{count} agents recorded', + 'workflows.history.script': 'Script', + 'workflows.launcher.argsPlaceholder': 'e.g. a question, or a list of paths', } as const export type TranslationKey = keyof typeof en diff --git a/desktop/src/i18n/locales/jp.ts b/desktop/src/i18n/locales/jp.ts index 06cc7a0a..8bdd65c1 100644 --- a/desktop/src/i18n/locales/jp.ts +++ b/desktop/src/i18n/locales/jp.ts @@ -2329,6 +2329,7 @@ export const jp: Record = { 'session.activity.section.tasks': 'タスク', 'session.activity.tasksProgress': 'タスク進捗 {completed}/{total}', 'session.activity.section.team': 'チーム', + 'session.activity.section.workflow': 'ワークフロー', 'session.activity.section.backgroundTasks': 'バックグラウンドタスク', 'session.activity.section.subagents': 'SubAgent', 'session.activity.section.sources': 'ソース', @@ -2823,4 +2824,42 @@ export const jp: Record = { 'browser.selection.navigationBody': '移動すると、選択した {count} 件が破棄され、ページ上のプレビュー変更も元に戻ります。', 'browser.selection.navigationContinue': '破棄して続行', 'browser.selection.navigationDiscarded': 'ページが変わったため、{count} 件の選択をクリアしました。', + 'workflows.status.running': '実行中', + 'workflows.status.completed': '完了', + 'workflows.status.failed': '失敗', + 'workflows.status.stopped': '停止', + 'workflows.report.title': '動的ワークフローの実行', + 'workflows.report.progressLabel': 'ワークフロー {name} の進捗', + 'workflows.report.waiting': '最初のエージェントを待機中…', + 'workflows.report.result': '結果', + 'workflows.ungroupedPhase': '未分類', + 'workflows.meta.agents': 'エージェント {count} 件', + 'workflows.meta.tokens': '{tokens} トークン', + 'workflows.meta.tools': 'ツール呼び出し {count} 回', + 'workflows.phase.counts': '{done}/{total} 完了', + 'workflows.phase.failed': '{count} 件失敗', + 'workflows.agent.cached': 'キャッシュ', + 'workflows.dock.running': '{count} 件実行中', + 'workflows.dock.expand': 'ワークフローの詳細を表示', + 'workflows.dock.collapse': 'ワークフローの詳細を隠す', + 'workflows.launcher.title': 'ワークフローを実行', + 'workflows.launcher.hint': 'ワークフローは保存済みスクリプトから多数のサブエージェントを編成し、バックグラウンドで実行して最後にまとめて報告します。', + 'workflows.launcher.empty': '保存済みのワークフローがありません。実行から保存するか、.claude/workflows/ にスクリプトを置いてください。', + 'workflows.launcher.run': '実行', + 'workflows.launcher.argsLabel': 'args としてスクリプトに渡す入力', + 'workflows.workbench.title': '動的ワークフロー ワークベンチ', + 'workflows.workbench.close': 'ワークフロー ワークベンチを閉じる', + 'workflows.workbench.open': 'ワークフロー ワークベンチを開く', + 'workflows.workbench.closeDetail': 'エージェント詳細を閉じる', + 'workflows.workbench.counts': 'エージェント {done}/{total} 件が終了', + 'workflows.workbench.runningCount': '{count} 件実行中', + 'workflows.workbench.queued': '{count} 件待機中', + 'workflows.workbench.prompt': 'プロンプト', + 'workflows.workbench.detailTitle': 'エージェント', + 'workflows.history.title': 'ワークフロー履歴', + 'workflows.history.empty': 'このマシンにはまだ実行履歴がありません。', + 'workflows.history.pick': '実行を選ぶと、スクリプトと各エージェントの結果を表示します。', + 'workflows.history.agents': '{count} 個のエージェントを記録', + 'workflows.history.script': 'スクリプト', + 'workflows.launcher.argsPlaceholder': '例: 質問、またはパスのリスト', } diff --git a/desktop/src/i18n/locales/kr.ts b/desktop/src/i18n/locales/kr.ts index fc916dcb..3de2ba1b 100644 --- a/desktop/src/i18n/locales/kr.ts +++ b/desktop/src/i18n/locales/kr.ts @@ -2329,6 +2329,7 @@ export const kr: Record = { 'session.activity.section.tasks': '작업', 'session.activity.tasksProgress': '작업 진행률 {completed}/{total}', 'session.activity.section.team': '팀', + 'session.activity.section.workflow': '워크플로', 'session.activity.section.backgroundTasks': '백그라운드 작업', 'session.activity.section.subagents': 'SubAgent', 'session.activity.section.sources': '소스', @@ -2823,4 +2824,42 @@ export const kr: Record = { 'browser.selection.navigationBody': '이동하면 선택한 요소 {count}개가 삭제되고 페이지의 미리보기 수정도 되돌아갑니다.', 'browser.selection.navigationContinue': '삭제하고 계속', 'browser.selection.navigationDiscarded': '페이지가 변경되어 선택 {count}개를 지웠습니다.', + 'workflows.status.running': '실행 중', + 'workflows.status.completed': '완료', + 'workflows.status.failed': '실패', + 'workflows.status.stopped': '중지됨', + 'workflows.report.title': '동적 워크플로 실행', + 'workflows.report.progressLabel': '워크플로 {name} 진행률', + 'workflows.report.waiting': '첫 에이전트를 기다리는 중…', + 'workflows.report.result': '결과', + 'workflows.ungroupedPhase': '미분류', + 'workflows.meta.agents': '에이전트 {count}개', + 'workflows.meta.tokens': '{tokens} 토큰', + 'workflows.meta.tools': '도구 호출 {count}회', + 'workflows.phase.counts': '{done}/{total} 완료', + 'workflows.phase.failed': '{count}개 실패', + 'workflows.agent.cached': '캐시', + 'workflows.dock.running': '{count}개 실행 중', + 'workflows.dock.expand': '워크플로 세부 정보 펼치기', + 'workflows.dock.collapse': '워크플로 세부 정보 접기', + 'workflows.launcher.title': '워크플로 실행', + 'workflows.launcher.hint': '워크플로는 저장된 스크립트로 여러 서브에이전트를 편성해 백그라운드에서 실행하고 끝나면 한 번에 보고합니다.', + 'workflows.launcher.empty': '저장된 워크플로가 없습니다. 실행에서 저장하거나 .claude/workflows/ 에 스크립트를 넣으세요.', + 'workflows.launcher.run': '실행', + 'workflows.launcher.argsLabel': 'args 로 스크립트에 전달할 입력', + 'workflows.workbench.title': '동적 워크플로 워크벤치', + 'workflows.workbench.close': '워크플로 워크벤치 닫기', + 'workflows.workbench.open': '워크플로 워크벤치 열기', + 'workflows.workbench.closeDetail': '에이전트 세부 정보 닫기', + 'workflows.workbench.counts': '에이전트 {done}/{total}개 종료', + 'workflows.workbench.runningCount': '{count}개 실행 중', + 'workflows.workbench.queued': '{count}개 대기 중', + 'workflows.workbench.prompt': '프롬프트', + 'workflows.workbench.detailTitle': '에이전트', + 'workflows.history.title': '워크플로 기록', + 'workflows.history.empty': '이 컴퓨터에는 아직 실행 기록이 없습니다.', + 'workflows.history.pick': '실행을 선택하면 스크립트와 에이전트별 결과를 볼 수 있습니다.', + 'workflows.history.agents': '{count}개 에이전트 기록됨', + 'workflows.history.script': '스크립트', + 'workflows.launcher.argsPlaceholder': '예: 질문 또는 경로 목록', } diff --git a/desktop/src/i18n/locales/zh-TW.ts b/desktop/src/i18n/locales/zh-TW.ts index 70ebe7f4..45cfff27 100644 --- a/desktop/src/i18n/locales/zh-TW.ts +++ b/desktop/src/i18n/locales/zh-TW.ts @@ -2328,6 +2328,7 @@ export const zh: Record = { 'session.activity.section.tasks': '任務', 'session.activity.tasksProgress': '任務進度 {completed}/{total}', 'session.activity.section.team': '團隊', + 'session.activity.section.workflow': '工作流程', 'session.activity.section.backgroundTasks': '後台任務', 'session.activity.section.subagents': 'SubAgent', 'session.activity.section.sources': '來源', @@ -2822,4 +2823,42 @@ export const zh: Record = { 'browser.selection.navigationBody': '繼續導覽會捨棄已選的 {count} 個元素,並還原頁面中的預覽修改。', 'browser.selection.navigationContinue': '捨棄並繼續', 'browser.selection.navigationDiscarded': '頁面已變更,已清除 {count} 個選取。', + 'workflows.status.running': '執行中', + 'workflows.status.completed': '已完成', + 'workflows.status.failed': '失敗', + 'workflows.status.stopped': '已停止', + 'workflows.report.title': '動態工作流執行', + 'workflows.report.progressLabel': '工作流 {name} 的進度', + 'workflows.report.waiting': '等待第一個 agent…', + 'workflows.report.result': '結果', + 'workflows.ungroupedPhase': '未分組', + 'workflows.meta.agents': '{count} 個 agent', + 'workflows.meta.tokens': '{tokens} tokens', + 'workflows.meta.tools': '{count} 次工具呼叫', + 'workflows.phase.counts': '{done}/{total} 已完成', + 'workflows.phase.failed': '{count} 個失敗', + 'workflows.agent.cached': '快取', + 'workflows.dock.running': '{count} 個執行中', + 'workflows.dock.expand': '展開工作流詳情', + 'workflows.dock.collapse': '收合工作流詳情', + 'workflows.launcher.title': '執行工作流', + 'workflows.launcher.hint': '工作流透過已儲存的腳本編排多個子 agent,在背景執行並在結束時統一回報。', + 'workflows.launcher.empty': '還沒有已儲存的工作流。可以從某次執行儲存,或把腳本放進 .claude/workflows/。', + 'workflows.launcher.run': '執行', + 'workflows.launcher.argsLabel': '作為 args 傳給腳本的輸入', + 'workflows.workbench.title': '動態工作流工作台', + 'workflows.workbench.close': '關閉工作流工作台', + 'workflows.workbench.open': '開啟工作流工作台', + 'workflows.workbench.closeDetail': '關閉 agent 詳情', + 'workflows.workbench.counts': '{done}/{total} 個 agent 已結束', + 'workflows.workbench.runningCount': '{count} 個執行中', + 'workflows.workbench.queued': '{count} 個排隊中', + 'workflows.workbench.prompt': '提示詞', + 'workflows.workbench.detailTitle': 'Agent', + 'workflows.history.title': '工作流程歷史', + 'workflows.history.empty': '本機還沒有歷史執行記錄。', + 'workflows.history.pick': '選擇一次執行,查看它的腳本和各 agent 的結果。', + 'workflows.history.agents': '已記錄 {count} 個 agent', + 'workflows.history.script': '腳本', + 'workflows.launcher.argsPlaceholder': '例如一個問題,或一組路徑', } diff --git a/desktop/src/i18n/locales/zh.ts b/desktop/src/i18n/locales/zh.ts index 78a3e95b..cf74e26c 100644 --- a/desktop/src/i18n/locales/zh.ts +++ b/desktop/src/i18n/locales/zh.ts @@ -2328,6 +2328,7 @@ export const zh: Record = { 'session.activity.section.tasks': '任务', 'session.activity.tasksProgress': '任务进度 {completed}/{total}', 'session.activity.section.team': '团队', + 'session.activity.section.workflow': '工作流', 'session.activity.section.backgroundTasks': '后台任务', 'session.activity.section.subagents': 'SubAgent', 'session.activity.section.sources': '来源', @@ -2822,4 +2823,42 @@ export const zh: Record = { 'browser.selection.navigationBody': '继续导航会丢弃已选的 {count} 个元素,并还原页面中的预览修改。', 'browser.selection.navigationContinue': '丢弃并继续', 'browser.selection.navigationDiscarded': '页面已变化,已清空 {count} 个选择。', + 'workflows.status.running': '运行中', + 'workflows.status.completed': '已完成', + 'workflows.status.failed': '失败', + 'workflows.status.stopped': '已停止', + 'workflows.report.title': '动态工作流运行', + 'workflows.report.progressLabel': '工作流 {name} 的进度', + 'workflows.report.waiting': '等待第一个 agent…', + 'workflows.report.result': '结果', + 'workflows.ungroupedPhase': '未分组', + 'workflows.meta.agents': '{count} 个 agent', + 'workflows.meta.tokens': '{tokens} tokens', + 'workflows.meta.tools': '{count} 次工具调用', + 'workflows.phase.counts': '{done}/{total} 已完成', + 'workflows.phase.failed': '{count} 个失败', + 'workflows.agent.cached': '缓存', + 'workflows.dock.running': '{count} 个运行中', + 'workflows.dock.expand': '展开工作流详情', + 'workflows.dock.collapse': '收起工作流详情', + 'workflows.launcher.title': '运行工作流', + 'workflows.launcher.hint': '工作流通过已保存的脚本编排多个子 agent,在后台运行并在结束时统一汇报。', + 'workflows.launcher.empty': '还没有已保存的工作流。可以从某次运行保存,或把脚本放进 .claude/workflows/。', + 'workflows.launcher.run': '运行', + 'workflows.launcher.argsLabel': '作为 args 传给脚本的输入', + 'workflows.workbench.title': '动态工作流工作台', + 'workflows.workbench.close': '关闭工作流工作台', + 'workflows.workbench.open': '打开工作流工作台', + 'workflows.workbench.closeDetail': '关闭 agent 详情', + 'workflows.workbench.counts': '{done}/{total} 个 agent 已结束', + 'workflows.workbench.runningCount': '{count} 个运行中', + 'workflows.workbench.queued': '{count} 个排队中', + 'workflows.workbench.prompt': '提示词', + 'workflows.workbench.detailTitle': 'Agent', + 'workflows.history.title': '工作流历史', + 'workflows.history.empty': '本机还没有历史运行记录。', + 'workflows.history.pick': '选择一次运行,查看它的脚本和各 agent 的结果。', + 'workflows.history.agents': '已记录 {count} 个 agent', + 'workflows.history.script': '脚本', + 'workflows.launcher.argsPlaceholder': '例如一个问题,或一组路径', } diff --git a/desktop/src/pages/ActiveSession.tsx b/desktop/src/pages/ActiveSession.tsx index 2ad57706..62d6e0d8 100644 --- a/desktop/src/pages/ActiveSession.tsx +++ b/desktop/src/pages/ActiveSession.tsx @@ -41,6 +41,7 @@ import { AgentTeamsReport } from '../components/agentTeams/AgentTeamsReport' import { AgentTeamsStrip } from '../components/agentTeams/AgentTeamsSummary' import { SessionActivityPanel } from '../components/activity/SessionActivityPanel' import { buildSessionActivityModel, hasVisibleSessionActivity } from '../components/activity/sessionActivityModel' +import { runsForSession, useWorkflowStore } from '../stores/workflowStore' import { TerminalSettings } from './TerminalSettings' import type { SessionListItem } from '../types/session' import type { ActiveGoalState, TokenUsage } from '../types/chat' @@ -519,6 +520,14 @@ export function ActiveSession() { [dismissedBackgroundTaskKeyList], ) const agentTaskNotifications = sessionState?.agentTaskNotifications ?? EMPTY_AGENT_TASK_NOTIFICATIONS + // Subscribe to the stable `runs` record and derive the per-session list here: + // a selector that filtered would allocate a new array on every store read, + // which zustand compares by identity and would re-render forever. + const allWorkflowRuns = useWorkflowStore(state => state.runs) + const workflowRuns = useMemo( + () => (activeTabId ? runsForSession({ runs: allWorkflowRuns }, activeTabId) : []), + [allWorkflowRuns, activeTabId], + ) const activeGoal = sessionState?.activeGoal ?? null const isEmpty = messages.length === 0 && @@ -570,6 +579,7 @@ export function ActiveSession() { backgroundTasks, dismissedBackgroundTaskKeys, agentNotifications: Object.values(agentTaskNotifications), + workflowRuns, }) }, [ activeTabId, @@ -581,6 +591,7 @@ export function ActiveSession() { dismissedBackgroundTaskKeys, messages, trackedTaskSessionId, + workflowRuns, ]) const hasVisibleActivity = activityModel ? hasVisibleSessionActivity(activityModel) : false const hasAutoOpenActivity = activityModel ? activityModel.badgeCount > 0 : false diff --git a/desktop/src/pages/SubagentRunPage.test.tsx b/desktop/src/pages/SubagentRunPage.test.tsx index a0cea7a4..607d3ac1 100644 --- a/desktop/src/pages/SubagentRunPage.test.tsx +++ b/desktop/src/pages/SubagentRunPage.test.tsx @@ -5,9 +5,13 @@ import type { SubagentRunResponse } from '../api/subagents' import { useSettingsStore } from '../stores/settingsStore' import { setComposerText } from '../components/chat/composerTestUtils' -vi.mock('../api/subagents', () => ({ +vi.mock('../api/subagents', async (importOriginal) => ({ + // Keep the real ref helpers: they decide whether the page fetches by tool + // call or by agent id, so stubbing them would test a fiction. + ...(await importOriginal()), subagentsApi: { getRunByTool: vi.fn(), + getRunByAgent: vi.fn(), sendMessage: vi.fn(), }, })) @@ -70,6 +74,7 @@ describe('SubagentRunPage', () => { cleanup() vi.useRealTimers() vi.mocked(subagentsApi.getRunByTool).mockReset() + vi.mocked(subagentsApi.getRunByAgent).mockReset() vi.mocked(subagentsApi.sendMessage).mockReset() }) @@ -86,6 +91,25 @@ describe('SubagentRunPage', () => { expect(useTabStore.getState().tabs.map((tab) => tab.sessionId)).toEqual(['session-1']) }) + it('fetches a workflow agent by agent id, not by tool call', async () => { + // Workflow agents are spawned by the workflow runtime, so no parent Agent + // tool call exists to look them up by. Same page, same rendering — only + // the lookup differs. + vi.mocked(subagentsApi.getRunByAgent).mockResolvedValue(subagentRun()) + + render( + , + ) + + expect(await screen.findByText('survey response.js')).toBeInTheDocument() + expect(subagentsApi.getRunByAgent).toHaveBeenCalledWith('session-1', 'wfagent1') + expect(subagentsApi.getRunByTool).not.toHaveBeenCalled() + }) + it('renders SubAgent run details', async () => { vi.mocked(subagentsApi.getRunByTool).mockResolvedValue(subagentRun({ outputFile: '/tmp/result.md', diff --git a/desktop/src/pages/SubagentRunPage.tsx b/desktop/src/pages/SubagentRunPage.tsx index e6e35f56..a493b16c 100644 --- a/desktop/src/pages/SubagentRunPage.tsx +++ b/desktop/src/pages/SubagentRunPage.tsx @@ -1,6 +1,8 @@ import { useCallback, useEffect, useRef, useState } from 'react' import { ArrowLeft, RefreshCw } from 'lucide-react' import { + isAgentIdRef, + readAgentIdRef, subagentsApi, type SubagentRunResponse, type SubagentRunStatus, @@ -68,7 +70,11 @@ export function SubagentRunPage({ setError(null) if (options?.resetData) setData(null) try { - const nextData = await subagentsApi.getRunByTool(sourceSessionId, toolUseId, resolvedTaskId) + // A workflow agent has no parent Agent tool call, so it is addressed by + // agent id. Everything downstream of this line is identical. + const nextData = isAgentIdRef(toolUseId) + ? await subagentsApi.getRunByAgent(sourceSessionId, readAgentIdRef(toolUseId)) + : await subagentsApi.getRunByTool(sourceSessionId, toolUseId, resolvedTaskId) if (requestIdRef.current !== requestId) return setData(nextData) } catch (err) { diff --git a/desktop/src/stores/chatStore.ts b/desktop/src/stores/chatStore.ts index 97701d01..f09fb73f 100644 --- a/desktop/src/stores/chatStore.ts +++ b/desktop/src/stores/chatStore.ts @@ -5,6 +5,7 @@ import { subagentsApi } from '../api/subagents' import { useTeamStore } from './teamStore' import { useSessionStore } from './sessionStore' import { useCLITaskStore } from './cliTaskStore' +import { useWorkflowStore } from './workflowStore' import { useSessionRuntimeStore } from './sessionRuntimeStore' import { useTabStore } from './tabStore' import { randomSpinnerVerb } from '../config/spinnerVerbs' @@ -1796,6 +1797,11 @@ export const useChatStore = create((set, get) => ({ const existingLoad = historyLoadsInFlight.get(sessionId) if (existingLoad) return existingLoad + // Workflow runs are rebuilt from disk alongside the transcript. Without + // this, reopening a session that ran a workflow showed no trace of it — + // the progress stream that populates the panel is live-only. + void useWorkflowStore.getState().hydrateSession(sessionId) + const requestedMutationEpoch = get().sessions[sessionId]?.historyMutationEpoch ?? 0 let load!: Promise load = (async () => { @@ -3276,6 +3282,7 @@ export const useChatStore = create((set, get) => ({ clearPendingTaskToolUseIds(sessionId) clearPendingToolParentUseIds(sessionId) useCLITaskStore.getState().clearTasks(sessionId) + useWorkflowStore.getState().clearSession(sessionId) useSessionStore.getState().updateSessionTitle(sessionId, 'New Session') useSessionStore.getState().updateSessionMessageCount(sessionId, 0) useTabStore.getState().updateTabTitle(sessionId, 'New Session') @@ -3355,6 +3362,10 @@ export const useChatStore = create((set, get) => ({ } } if ((msg.subtype === 'task_started' || msg.subtype === 'task_progress') && msg.data && typeof msg.data === 'object') { + // Workflow runs also flow through the generic background-task path + // below; the workflow store keeps the phase/agent detail that the + // generic task shape has nowhere to put. + useWorkflowStore.getState().handleTaskEvent(sessionId, msg.subtype, msg.data) const taskEvent = normalizeBackgroundAgentTaskEvent(msg.data, msg.subtype) if (taskEvent) { const now = Date.now() @@ -3393,6 +3404,7 @@ export const useChatStore = create((set, get) => ({ } } if (msg.subtype === 'task_notification' && msg.data && typeof msg.data === 'object') { + useWorkflowStore.getState().handleTaskEvent(sessionId, 'task_notification', msg.data) const data = msg.data as Record const taskEvent = normalizeBackgroundAgentTaskEvent(data, 'task_notification') const toolUseId = diff --git a/desktop/src/stores/workflowStore.test.ts b/desktop/src/stores/workflowStore.test.ts new file mode 100644 index 00000000..7f0fb0f9 --- /dev/null +++ b/desktop/src/stores/workflowStore.test.ts @@ -0,0 +1,356 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' + +const sessionRunsMock = vi.hoisted(() => vi.fn()) +vi.mock('../api/workflows', async importOriginal => { + const actual = await importOriginal() + return { + ...actual, + workflowsApi: { ...actual.workflowsApi, sessionRuns: sessionRunsMock }, + } +}) +import { + activeRunForSession, + groupRunPhases, + runCompletion, + runsForSession, + useWorkflowStore, +} from './workflowStore' + +const SESSION = 'session-1' +const TASK = 'w1234abcd' + +function taskStarted(overrides: Record = {}) { + return { + task_id: TASK, + task_type: 'local_workflow', + workflow_name: 'audit-routes', + description: 'Audit every route handler', + ...overrides, + } +} + +function progress( + rows: Array>, + usage?: { total_tokens?: number; tool_uses?: number }, +) { + return { + task_id: TASK, + workflow_progress: rows, + ...(usage ? { usage } : {}), + } +} + +function agentRow( + index: number, + state: string, + extra: Record = {}, +) { + return { + type: 'workflow_agent', + index, + label: `agent ${index}`, + state, + ...extra, + } +} + +describe('workflowStore', () => { + beforeEach(() => { + useWorkflowStore.setState({ + runs: {}, + definitions: [], + definitionsLoading: false, + definitionsError: null, + history: [], + historyLoading: false, + openRunId: null, + }) + }) + + const store = () => useWorkflowStore.getState() + + it('creates a run from task_started and tracks it for the session', () => { + store().handleTaskEvent(SESSION, 'task_started', taskStarted()) + const runs = runsForSession(useWorkflowStore.getState(), SESSION) + expect(runs).toHaveLength(1) + expect(runs[0]).toMatchObject({ + taskId: TASK, + workflowName: 'audit-routes', + description: 'Audit every route handler', + status: 'running', + }) + expect(activeRunForSession(useWorkflowStore.getState(), SESSION)?.taskId).toBe( + TASK, + ) + }) + + it('still ignores a bare notification for a task it never tracked', () => { + store().handleTaskEvent(SESSION, 'task_notification', { + task_id: 'b0000009', + status: 'completed', + }) + expect(runsForSession(useWorkflowStore.getState(), SESSION)).toHaveLength(0) + }) + + it('ignores task events that are not workflows', () => { + store().handleTaskEvent(SESSION, 'task_started', { + task_id: 'b0000001', + task_type: 'local_bash', + description: 'npm test', + }) + expect(runsForSession(useWorkflowStore.getState(), SESSION)).toHaveLength(0) + }) + + it('replaces an agent row in place instead of appending a new one', () => { + store().handleTaskEvent(SESSION, 'task_started', taskStarted()) + store().handleTaskEvent( + SESSION, + 'task_progress', + progress([agentRow(1, 'progress', { tokens: 100 })]), + ) + store().handleTaskEvent( + SESSION, + 'task_progress', + progress([agentRow(1, 'done', { tokens: 400 })]), + ) + + const run = useWorkflowStore.getState().runs[TASK]! + const agents = run.progress.filter(row => row.type === 'workflow_agent') + expect(agents).toHaveLength(1) + expect(agents[0]).toMatchObject({ index: 1, state: 'done', tokens: 400 }) + expect(run.agentCount).toBe(1) + }) + + it('keeps the object identity stable when an event changes nothing', () => { + store().handleTaskEvent(SESSION, 'task_started', taskStarted()) + const before = useWorkflowStore.getState().runs[TASK] + store().handleTaskEvent(SESSION, 'task_progress', progress([])) + expect(useWorkflowStore.getState().runs[TASK]).toBe(before) + }) + + it('carries usage totals from the event', () => { + store().handleTaskEvent(SESSION, 'task_started', taskStarted()) + store().handleTaskEvent( + SESSION, + 'task_progress', + progress([agentRow(1, 'done')], { total_tokens: 1234, tool_uses: 7 }), + ) + expect(useWorkflowStore.getState().runs[TASK]).toMatchObject({ + totalTokens: 1234, + toolCalls: 7, + }) + }) + + it('settles the run on task_notification and records the result', () => { + store().handleTaskEvent(SESSION, 'task_started', taskStarted()) + // The CLI's terminal notification carries no task_type / workflow_name / + // workflow_progress — only the task id. Dropping it here is what left the + // dock reading "1 running" after a finished run. + store().handleTaskEvent(SESSION, 'task_notification', { + task_id: TASK, + status: 'completed', + result: '["ALPHA","BETA"]', + summary: 'Dynamic workflow "route-survey" completed', + }) + const run = useWorkflowStore.getState().runs[TASK]! + expect(run.status).toBe('completed') + expect(run.result).toBe('["ALPHA","BETA"]') + expect(run.endedAt).toBeGreaterThan(0) + expect(activeRunForSession(useWorkflowStore.getState(), SESSION)).toBeNull() + }) + + it('a late progress event does not revive a settled run', () => { + store().handleTaskEvent(SESSION, 'task_started', taskStarted()) + store().handleTaskEvent(SESSION, 'task_notification', { + task_id: TASK, + status: 'failed', + error: 'boom', + }) + store().handleTaskEvent( + SESSION, + 'task_progress', + progress([agentRow(2, 'progress')]), + ) + const run = useWorkflowStore.getState().runs[TASK]! + expect(run.status).toBe('failed') + expect(run.error).toBe('boom') + }) + + it('surfaces a failure reason that only arrived in the summary', () => { + // Seen on a real failed run: the CLI's terminal event has no `error` + // field, so the reason only ever comes through `summary`. Reading it as + // the description put it in a truncated one-liner and left the error + // banner empty — a red "failed" badge with no explanation. + store().handleTaskEvent(SESSION, 'task_started', taskStarted()) + store().handleTaskEvent(SESSION, 'task_notification', { + task_id: TASK, + status: 'failed', + summary: + 'Dynamic workflow "audit-routes" failed: deliberate failure after the probe settled', + }) + const run = useWorkflowStore.getState().runs[TASK]! + expect(run.error).toContain('deliberate failure after the probe settled') + // The run's own description must survive the notification. + expect(run.description).toBe('Audit every route handler') + }) + + it('keeps the description when a run completes', () => { + store().handleTaskEvent(SESSION, 'task_started', taskStarted()) + store().handleTaskEvent(SESSION, 'task_notification', { + task_id: TASK, + status: 'completed', + summary: 'Dynamic workflow "audit-routes" completed', + result: '{}', + }) + const run = useWorkflowStore.getState().runs[TASK]! + expect(run.description).toBe('Audit every route handler') + // Boilerplate completion text is not an error. + expect(run.error).toBeUndefined() + }) + + it('maps a killed CLI status onto stopped', () => { + store().handleTaskEvent(SESSION, 'task_started', taskStarted()) + store().handleTaskEvent(SESSION, 'task_notification', { + task_id: TASK, + status: 'killed', + }) + expect(useWorkflowStore.getState().runs[TASK]?.status).toBe('stopped') + }) + + it('clearSession drops only that session and closes an open run', () => { + store().handleTaskEvent(SESSION, 'task_started', taskStarted()) + store().handleTaskEvent('other', 'task_started', taskStarted({ task_id: 'w9' })) + store().openRun(TASK) + + store().clearSession(SESSION) + expect(runsForSession(useWorkflowStore.getState(), SESSION)).toHaveLength(0) + expect(runsForSession(useWorkflowStore.getState(), 'other')).toHaveLength(1) + expect(useWorkflowStore.getState().openRunId).toBeNull() + }) + + it('groups agents under the phase they reported', () => { + store().handleTaskEvent(SESSION, 'task_started', taskStarted()) + store().handleTaskEvent( + SESSION, + 'task_progress', + progress([ + { type: 'workflow_phase', index: 1, title: 'Review', kind: 'meta' }, + { type: 'workflow_phase', index: 2, title: 'Verify', kind: 'script' }, + agentRow(1, 'done', { phaseIndex: 1, phaseTitle: 'Review' }), + agentRow(2, 'progress', { phaseIndex: 2, phaseTitle: 'Verify' }), + agentRow(3, 'done'), + ]), + ) + const run = useWorkflowStore.getState().runs[TASK]! + const { phases, ungrouped } = groupRunPhases(run) + expect(phases.map(phase => phase.title)).toEqual(['Review', 'Verify']) + expect(phases[0]?.agents.map(agent => agent.index)).toEqual([1]) + expect(phases[1]?.agents.map(agent => agent.index)).toEqual([2]) + expect(ungrouped.map(agent => agent.index)).toEqual([3]) + }) + + it('reports completion as the settled fraction of agents', () => { + store().handleTaskEvent(SESSION, 'task_started', taskStarted()) + store().handleTaskEvent( + SESSION, + 'task_progress', + progress([ + agentRow(1, 'done'), + agentRow(2, 'error'), + agentRow(3, 'progress'), + agentRow(4, 'start'), + ]), + ) + expect(runCompletion(useWorkflowStore.getState().runs[TASK]!)).toBe(0.5) + }) + + it('recognises a workflow from workflow_progress even without task_type', () => { + store().handleTaskEvent( + SESSION, + 'task_progress', + progress([agentRow(1, 'progress')]), + ) + expect(runsForSession(useWorkflowStore.getState(), SESSION)).toHaveLength(1) + }) +}) + +describe('hydrateSession', () => { + const RECONSTRUCTED = { + runId: 'wf_06ee51bf-6b1', + workflowName: 'review-last-month', + startedAt: 1_700_000_000_000, + agents: [ + { agentId: 'a1', label: 'review:imagegen', phaseIndex: 1, phaseTitle: '维度审查', agentIndex: 1 }, + { agentId: 'a2', label: 'review:security', phaseIndex: 1, phaseTitle: '维度审查', agentIndex: 2 }, + { agentId: 'a3', label: 'verify:security', phaseIndex: 2, phaseTitle: '对抗验证', agentIndex: 3 }, + ], + } + + beforeEach(() => { + sessionRunsMock.mockReset() + useWorkflowStore.setState({ runs: {}, openRunId: null }) + }) + + it('rebuilds a finished run so a reopened session still shows it', async () => { + // The whole point: the live progress stream is gone by now. + sessionRunsMock.mockResolvedValue({ runs: [RECONSTRUCTED] }) + await useWorkflowStore.getState().hydrateSession(SESSION) + + const runs = runsForSession(useWorkflowStore.getState(), SESSION) + expect(runs).toHaveLength(1) + const { phases } = groupRunPhases(runs[0]!) + expect(phases.map(phase => phase.title)).toEqual(['维度审查', '对抗验证']) + expect(phases[0]!.agents.map(agent => agent.label)).toEqual([ + 'review:imagegen', + 'review:security', + ]) + // Every reconstructed agent has a transcript, so it is openable. + expect(phases[0]!.agents.every(agent => agent.agentId)).toBe(true) + }) + + it('leaves an unrecorded phase title empty rather than inventing one', () => { + // The renderer falls back to the run name; filling in "Phase 0" here made + // that impossible and showed a meaningless heading. + sessionRunsMock.mockResolvedValue({ + runs: [{ + ...RECONSTRUCTED, + agents: [{ agentId: 'a1', label: 'review:security', phaseIndex: 0, agentIndex: 1 }], + }], + }) + return useWorkflowStore.getState().hydrateSession(SESSION).then(() => { + const run = runsForSession(useWorkflowStore.getState(), SESSION)[0]! + const phase = run.progress.find((row) => row.type === 'workflow_phase')! + expect(phase.title).toBe('') + }) + }) + + it('never overwrites a run that is still streaming live', async () => { + useWorkflowStore.getState().handleTaskEvent(SESSION, 'task_started', { + task_id: TASK, + task_type: 'local_workflow', + workflow_name: 'review-last-month', + workflow_run_id: RECONSTRUCTED.runId, + }) + useWorkflowStore.getState().handleTaskEvent( + SESSION, + 'task_progress', + progress([agentRow(1, 'progress')]), + ) + sessionRunsMock.mockResolvedValue({ runs: [RECONSTRUCTED] }) + + await useWorkflowStore.getState().hydrateSession(SESSION) + + const runs = runsForSession(useWorkflowStore.getState(), SESSION) + // One run, still the live one — a reconstruction would have marked its + // in-flight agent as done. + expect(runs).toHaveLength(1) + expect(runs[0]!.status).toBe('running') + expect(runs[0]!.taskId).toBe(TASK) + }) + + it('stays quiet when the server cannot rebuild anything', async () => { + sessionRunsMock.mockRejectedValue(new Error('offline')) + await useWorkflowStore.getState().hydrateSession(SESSION) + expect(runsForSession(useWorkflowStore.getState(), SESSION)).toHaveLength(0) + }) +}) diff --git a/desktop/src/stores/workflowStore.ts b/desktop/src/stores/workflowStore.ts new file mode 100644 index 00000000..176b69fb --- /dev/null +++ b/desktop/src/stores/workflowStore.ts @@ -0,0 +1,508 @@ +import { create } from 'zustand' +import { workflowsApi } from '../api/workflows' +import type { + ReconstructedWorkflowRun, + WorkflowAgentEvent, + WorkflowDefinition, + WorkflowPhaseGroup, + WorkflowProgressEvent, + WorkflowRun, + WorkflowRunStatus, + WorkflowRunSummary, +} from '../types/workflow' + +/** Cap kept in sync with the CLI's own progress-row budget. */ +const MAX_PROGRESS_ROWS = 500 + +type WorkflowStore = { + /** Live and just-finished runs, keyed by CLI task id. */ + runs: Record + definitions: WorkflowDefinition[] + definitionsLoading: boolean + definitionsError: string | null + history: WorkflowRunSummary[] + historyLoading: boolean + /** Task id of the run the workflow panel is showing, if any. */ + openRunId: string | null + + handleTaskEvent( + sessionId: string, + subtype: 'task_started' | 'task_progress' | 'task_notification', + data: unknown, + ): void + clearSession(sessionId: string): void + hydrateSession(sessionId: string): Promise + openRun(taskId: string | null): void + loadDefinitions(cwd?: string): Promise + loadHistory(sessionId?: string): Promise +} + +export const useWorkflowStore = create(set => ({ + runs: {}, + definitions: [], + definitionsLoading: false, + definitionsError: null, + history: [], + historyLoading: false, + openRunId: null, + + handleTaskEvent(sessionId, subtype, data) { + // A run we already track is identified by its task id alone. The terminal + // `task_notification` the CLI emits carries no task_type, workflow_name or + // workflow_progress (print.ts builds it from the notification XML), so + // re-deriving "is this a workflow" from the payload would drop it and the + // run would sit at "running" forever. + const known = Boolean( + typeof data === 'object' && + data !== null && + typeof (data as { task_id?: unknown }).task_id === 'string' && + useWorkflowStore.getState().runs[(data as { task_id: string }).task_id], + ) + const event = parseWorkflowTaskEvent(sessionId, subtype, data, known) + if (!event) return + set(state => { + const existing = state.runs[event.taskId] + const merged = mergeRun(existing, event) + if (merged === existing) return state + return { runs: { ...state.runs, [event.taskId]: merged } } + }) + }, + + /** + * Load a session's finished runs from disk. + * + * Live progress only exists while a run is happening, so a session reopened + * later showed no workflow at all. This fills that in from the per-agent + * sidecars the CLI persisted. A run already being tracked live is left + * alone — the live stream is strictly better than the reconstruction. + */ + async hydrateSession(sessionId) { + try { + const { runs } = await workflowsApi.sessionRuns(sessionId) + if (runs.length === 0) return + set(state => { + const liveRunIds = new Set( + Object.values(state.runs) + .filter(run => run.sessionId === sessionId) + .map(run => run.runId) + .filter((runId): runId is string => Boolean(runId)), + ) + const next = { ...state.runs } + let added = false + for (const run of runs) { + if (liveRunIds.has(run.runId) || next[run.runId]) continue + added = true + next[run.runId] = reconstructedToRun(sessionId, run) + } + return added ? { runs: next } : state + }) + } catch { + // History is an enhancement; a session that cannot load it still works. + } + }, + + clearSession(sessionId) { + set(state => { + const kept: Record = {} + let removed = false + for (const [taskId, run] of Object.entries(state.runs)) { + if (run.sessionId === sessionId) { + removed = true + continue + } + kept[taskId] = run + } + if (!removed) return state + const openRunId = + state.openRunId && kept[state.openRunId] ? state.openRunId : null + return { runs: kept, openRunId } + }) + }, + + openRun(taskId) { + set({ openRunId: taskId }) + }, + + async loadDefinitions(cwd) { + set({ definitionsLoading: true, definitionsError: null }) + try { + const { workflows } = await workflowsApi.list(cwd) + set({ definitions: workflows, definitionsLoading: false }) + } catch (error) { + set({ + definitionsLoading: false, + definitionsError: + error instanceof Error ? error.message : 'Failed to load workflows', + }) + } + }, + + async loadHistory(sessionId) { + set({ historyLoading: true }) + try { + const { runs } = await workflowsApi.listRuns({ sessionId, limit: 50 }) + set({ history: runs, historyLoading: false }) + } catch { + set({ historyLoading: false }) + } + }, +})) + + +/** + * Turn a disk-reconstructed run into the shape the panel renders. + * + * Every agent listed produced a transcript, so `done` is a fact rather than an + * assumption. Per-agent token counts and the run's own outcome are not + * persisted, so they are left empty instead of being invented — the panel + * shows neither for a historical run. + */ +function reconstructedToRun( + sessionId: string, + run: ReconstructedWorkflowRun, +): WorkflowRun { + const phases = new Map() + const progress: WorkflowProgressEvent[] = [] + + for (const agent of run.agents) { + if (!phases.has(agent.phaseIndex)) { + phases.set(agent.phaseIndex, agent.phaseTitle) + } + } + for (const [index, title] of [...phases.entries()].sort(([a], [b]) => a - b)) { + progress.push({ + type: 'workflow_phase', + index, + // Left empty rather than invented when the run never recorded a phase + // title, so the renderer can fall back to something meaningful. Filling + // in "Phase 0" here hid that the title was simply unknown. + title: title ?? '', + }) + } + for (const agent of run.agents) { + progress.push({ + type: 'workflow_agent', + index: agent.agentIndex, + label: agent.label, + state: 'done', + phaseIndex: agent.phaseIndex, + ...(agent.phaseTitle ? { phaseTitle: agent.phaseTitle } : {}), + agentId: agent.agentId, + }) + } + + return { + // Keyed by run id: a reconstructed run has no CLI task id, and the run id + // is what keeps it from colliding with a live run for the same workflow. + taskId: run.runId, + sessionId, + runId: run.runId, + workflowName: run.workflowName, + status: 'completed', + startedAt: run.startedAt, + updatedAt: run.startedAt, + endedAt: run.startedAt, + agentCount: run.agents.length, + totalTokens: 0, + toolCalls: 0, + progress, + } +} + +// ── Selectors ──────────────────────────────────────────────────────────────── + +export function runsForSession( + state: Pick, + sessionId: string, +): WorkflowRun[] { + return Object.values(state.runs) + .filter(run => run.sessionId === sessionId) + .sort((a, b) => b.startedAt - a.startedAt) +} + +export function activeRunForSession( + state: Pick, + sessionId: string, +): WorkflowRun | null { + return ( + runsForSession(state, sessionId).find(run => run.status === 'running') ?? null + ) +} + +/** + * Fold a run's flat progress rows into phases. + * + * Agents emitted before any `phase()` call carry no `phaseIndex`; they are + * grouped under a synthetic phase rather than dropped, so a script that never + * calls `phase()` still shows all of its agents. + */ +export function groupRunPhases(run: WorkflowRun): { + phases: WorkflowPhaseGroup[] + ungrouped: WorkflowAgentEvent[] +} { + const byIndex = new Map() + const ungrouped: WorkflowAgentEvent[] = [] + + for (const event of run.progress) { + if (event.type === 'workflow_phase') { + if (!byIndex.has(event.index)) { + byIndex.set(event.index, { + index: event.index, + title: event.title, + agents: [], + }) + } + continue + } + if (event.phaseIndex === undefined) { + ungrouped.push(event) + continue + } + const group = byIndex.get(event.phaseIndex) ?? { + index: event.phaseIndex, + title: event.phaseTitle ?? `Phase ${event.phaseIndex}`, + agents: [], + } + group.agents.push(event) + byIndex.set(event.phaseIndex, group) + } + + return { + phases: [...byIndex.values()].sort((a, b) => a.index - b.index), + ungrouped, + } +} + +/** Fraction of a run's agents that have settled, for the progress bar. */ +export function runCompletion(run: WorkflowRun): number { + const agents = run.progress.filter( + (event): event is WorkflowAgentEvent => event.type === 'workflow_agent', + ) + if (agents.length === 0) return run.status === 'running' ? 0 : 1 + const settled = agents.filter( + agent => agent.state === 'done' || agent.state === 'error', + ).length + return settled / agents.length +} + +// ── Event parsing ──────────────────────────────────────────────────────────── + +type ParsedEvent = { + taskId: string + sessionId: string + runId?: string + workflowName?: string + description?: string + status?: WorkflowRunStatus + progress: WorkflowProgressEvent[] + totalTokens?: number + toolCalls?: number + result?: string + error?: string +} + +/** + * Read a CLI task event, keeping only the ones that belong to a workflow. + * + * `task_progress` does not carry `task_type`, and the terminal + * `task_notification` carries neither that nor `workflow_progress`. A run is + * therefore recognised either by its workflow markers or — once `task_started` + * has registered it — by its task id alone. + */ +export function parseWorkflowTaskEvent( + sessionId: string, + subtype: 'task_started' | 'task_progress' | 'task_notification', + data: unknown, + /** True when a run with this task id is already tracked. */ + known = false, +): ParsedEvent | null { + if (typeof data !== 'object' || data === null) return null + const payload = data as Record + const taskId = readString(payload.task_id) + if (!taskId) return null + + const taskType = readString(payload.task_type) + const workflowName = readString(payload.workflow_name) + const progress = readProgress(payload.workflow_progress) + const isWorkflow = + known || + taskType === 'local_workflow' || + Boolean(workflowName) || + progress.length > 0 + if (!isWorkflow) return null + + const usage = + typeof payload.usage === 'object' && payload.usage !== null + ? (payload.usage as Record) + : undefined + + const terminal = subtype === 'task_notification' + const status = terminal + ? normalizeStatus(readString(payload.status)) + : 'running' + const summary = readString(payload.summary) + + return { + taskId, + sessionId, + runId: readString(payload.workflow_run_id), + workflowName, + // A terminal notification's `summary` describes the outcome, not the run. + // Letting it through as the description replaced "Survey Express + // request/response helpers…" with `Dynamic workflow "…" failed: …` in a + // single truncated line. + description: terminal ? undefined : (summary ?? readString(payload.description)), + status, + progress, + totalTokens: readNumber(usage?.total_tokens), + toolCalls: readNumber(usage?.tool_uses), + result: readString(payload.result), + // The CLI's terminal event carries no `error` field — the reason a run + // failed only ever arrives inside `summary`. Without this a failed run + // showed a red badge and no explanation anywhere. + error: + readString(payload.error) ?? + (status === 'failed' || status === 'stopped' ? summary : undefined), + } +} + +/** + * Apply one event to a run. + * + * Agent and phase rows are keyed by `type:index` and replaced in place, so a + * run that emits thousands of token-count updates for twenty agents keeps + * twenty rows. Returning the previous object unchanged when nothing moved is + * what stops zustand subscribers from re-rendering on every heartbeat. + */ +function mergeRun( + existing: WorkflowRun | undefined, + event: ParsedEvent, +): WorkflowRun { + const now = Date.now() + const base: WorkflowRun = existing ?? { + taskId: event.taskId, + sessionId: event.sessionId, + runId: event.runId, + workflowName: event.workflowName ?? 'workflow', + description: event.description, + status: 'running', + startedAt: now, + updatedAt: now, + agentCount: 0, + totalTokens: 0, + toolCalls: 0, + progress: [], + } + + // A terminal run must not be revived by a late progress event. + if (base.status !== 'running' && event.status === 'running') return base + + let progress = base.progress + if (event.progress.length > 0) { + const next = [...base.progress] + const indexByKey = new Map() + for (let i = 0; i < next.length; i++) { + const row = next[i]! + indexByKey.set(`${row.type}:${row.index}`, i) + } + for (const row of event.progress) { + const key = `${row.type}:${row.index}` + const at = indexByKey.get(key) + if (at === undefined) { + indexByKey.set(key, next.length) + next.push(row) + } else { + next[at] = row + } + } + progress = next.length > MAX_PROGRESS_ROWS + ? next.slice(next.length - MAX_PROGRESS_ROWS) + : next + } + + const agentCount = progress.reduce( + (max, row) => (row.type === 'workflow_agent' ? Math.max(max, row.index) : max), + 0, + ) + const status = event.status ?? base.status + const isTerminal = status !== 'running' + + const merged: WorkflowRun = { + ...base, + runId: event.runId ?? base.runId, + workflowName: event.workflowName ?? base.workflowName, + description: event.description ?? base.description, + status, + updatedAt: now, + endedAt: isTerminal ? (base.endedAt ?? now) : undefined, + agentCount: Math.max(base.agentCount, agentCount), + totalTokens: event.totalTokens ?? base.totalTokens, + toolCalls: event.toolCalls ?? base.toolCalls, + progress, + result: event.result ?? base.result, + error: event.error ?? base.error, + } + + return hasChanged(base, merged) ? merged : base +} + +function hasChanged(before: WorkflowRun, after: WorkflowRun): boolean { + return ( + before.progress !== after.progress || + before.status !== after.status || + before.totalTokens !== after.totalTokens || + before.toolCalls !== after.toolCalls || + before.agentCount !== after.agentCount || + before.result !== after.result || + before.error !== after.error || + before.workflowName !== after.workflowName || + before.description !== after.description || + before.runId !== after.runId + ) +} + +function readProgress(value: unknown): WorkflowProgressEvent[] { + if (!Array.isArray(value)) return [] + const rows: WorkflowProgressEvent[] = [] + for (const entry of value) { + if (typeof entry !== 'object' || entry === null) continue + const row = entry as Record + const index = readNumber(row.index) + if (index === undefined) continue + if (row.type === 'workflow_phase' && typeof row.title === 'string') { + rows.push({ + type: 'workflow_phase', + index, + title: row.title, + kind: row.kind === 'meta' || row.kind === 'script' ? row.kind : undefined, + }) + continue + } + if (row.type === 'workflow_agent' && typeof row.label === 'string') { + rows.push({ ...(row as unknown as WorkflowAgentEvent), index }) + } + } + return rows +} + +function normalizeStatus(value: string | undefined): WorkflowRunStatus { + switch (value) { + case 'completed': + return 'completed' + case 'failed': + return 'failed' + case 'killed': + case 'stopped': + return 'stopped' + default: + return 'running' + } +} + +function readString(value: unknown): string | undefined { + return typeof value === 'string' && value.trim() !== '' ? value : undefined +} + +function readNumber(value: unknown): number | undefined { + return typeof value === 'number' && Number.isFinite(value) ? value : undefined +} diff --git a/desktop/src/types/workflow.ts b/desktop/src/types/workflow.ts new file mode 100644 index 00000000..5d81d38e --- /dev/null +++ b/desktop/src/types/workflow.ts @@ -0,0 +1,132 @@ +/** + * Desktop-side mirror of the CLI's dynamic-workflow progress protocol. + * + * These shapes arrive over the session WebSocket inside + * `system_notification` → `task_started` / `task_progress` / `task_notification` + * payloads. They are declared here rather than imported from `src/` because the + * desktop bundle must not pull in the CLI runtime. + */ + +export type WorkflowAgentRunState = 'start' | 'progress' | 'done' | 'error' + +export type WorkflowAgentEvent = { + type: 'workflow_agent' + index: number + label: string + state: WorkflowAgentRunState + phaseIndex?: number + phaseTitle?: string + agentType?: string + isolation?: 'worktree' | 'remote' + model?: string + agentId?: string + queuedAt?: number + startedAt?: number + lastProgressAt?: number + durationMs?: number + tokens?: number + toolCalls?: number + cached?: boolean + skipped?: boolean + blocked?: boolean + error?: string + promptPreview?: string + resultPreview?: string + lastToolName?: string +} + +export type WorkflowPhaseEvent = { + type: 'workflow_phase' + index: number + title: string + kind?: 'meta' | 'script' +} + +export type WorkflowProgressEvent = WorkflowAgentEvent | WorkflowPhaseEvent + +export type WorkflowRunStatus = 'running' | 'completed' | 'failed' | 'stopped' + +/** One workflow run as the desktop knows it, live or just finished. */ +export type WorkflowRun = { + taskId: string + sessionId: string + runId?: string + workflowName: string + description?: string + status: WorkflowRunStatus + startedAt: number + updatedAt: number + endedAt?: number + agentCount: number + totalTokens: number + toolCalls: number + /** Keyed `workflow_agent:` / `workflow_phase:`, newest value wins. */ + progress: WorkflowProgressEvent[] + result?: string + error?: string +} + +export type WorkflowPhaseMeta = { + title: string + detail?: string + model?: string +} + +export type WorkflowDefinition = { + name: string + description: string + whenToUse?: string + source: 'built-in' | 'userSettings' | 'projectSettings' | 'plugin' | 'inline' + phases?: WorkflowPhaseMeta[] + filePath?: string + script?: string +} + +export type WorkflowRunSummary = { + runId: string + sessionId: string + workflowName: string + scriptPath: string + startedAt: number + completedAgents: number +} + +export type WorkflowRunDetail = WorkflowRunSummary & { + script: string + description?: string + phases?: WorkflowPhaseMeta[] + agents: Array<{ key: string; agentId: string; result: unknown }> +} + +/** A phase with the agents that reported under it. */ +export type WorkflowPhaseGroup = { + index: number + title: string + agents: WorkflowAgentEvent[] +} + +export function isWorkflowAgentEvent( + event: WorkflowProgressEvent, +): event is WorkflowAgentEvent { + return event.type === 'workflow_agent' +} + +/** + * A finished run rebuilt from disk, as the server reconstructs it. + * + * Carries only what outlives the process: which agents ran, under which + * phase. There is no live state here — every agent listed has a transcript, + * which is the fact that it ran. + */ +export type ReconstructedWorkflowRun = { + runId: string + workflowName: string + startedAt: number + agents: Array<{ + agentId: string + label: string + phaseIndex: number + phaseTitle?: string + agentIndex: number + }> +} diff --git a/package.json b/package.json index 342b70d6..63cd479b 100644 --- a/package.json +++ b/package.json @@ -59,6 +59,8 @@ "@opentelemetry/sdk-metrics": "^2.6.1", "@opentelemetry/sdk-trace-base": "^2.6.1", "@opentelemetry/semantic-conventions": "^1.40.0", + "acorn": "^8.16.0", + "acorn-walk": "^8.3.4", "ajv": "^8.18.0", "asciichart": "^1.5.25", "auto-bind": "^5.0.1", diff --git a/src/Task.ts b/src/Task.ts index 196caf36..c9ac32e0 100644 --- a/src/Task.ts +++ b/src/Task.ts @@ -15,6 +15,9 @@ export type TaskType = export type TaskStatus = | 'pending' | 'running' + // Only dynamic workflows pause: the run is stopped but its journal is intact, + // so resuming replays completed agents instead of re-running them. + | 'paused' | 'completed' | 'failed' | 'killed' diff --git a/src/commands.ts b/src/commands.ts index 85cf49be..af55f893 100644 --- a/src/commands.ts +++ b/src/commands.ts @@ -84,11 +84,9 @@ const voiceCommand = feature('VOICE_MODE') const forceSnip = feature('HISTORY_SNIP') ? require('./commands/force-snip.js').default : null -const workflowsCmd = feature('WORKFLOW_SCRIPTS') - ? ( - require('./commands/workflows/index.js') as typeof import('./commands/workflows/index.js') - ).default - : null +const workflowsCmd = ( + require('./commands/workflows/index.js') as typeof import('./commands/workflows/index.js') +).default const webCmd = feature('CCR_REMOTE_SETUP') ? ( require('./commands/remote-setup/index.js') as typeof import('./commands/remote-setup/index.js') @@ -400,11 +398,9 @@ async function getSkills(cwd: string): Promise<{ } /* eslint-disable @typescript-eslint/no-require-imports */ -const getWorkflowCommands = feature('WORKFLOW_SCRIPTS') - ? ( - require('./tools/WorkflowTool/createWorkflowCommand.js') as typeof import('./tools/WorkflowTool/createWorkflowCommand.js') - ).getWorkflowCommands - : null +const getWorkflowCommands = ( + require('./tools/WorkflowTool/createWorkflowCommand.js') as typeof import('./tools/WorkflowTool/createWorkflowCommand.js') +).getWorkflowCommands /* eslint-enable @typescript-eslint/no-require-imports */ /** diff --git a/src/commands/effort/effort.tsx b/src/commands/effort/effort.tsx index 35acf91f..f34b2f6f 100644 --- a/src/commands/effort/effort.tsx +++ b/src/commands/effort/effort.tsx @@ -6,13 +6,45 @@ import { useAppState, useSetAppState } from '../../state/AppState.js'; import type { LocalJSXCommandOnDone } from '../../types/command.js'; import { type EffortValue, getDisplayedEffortLevel, getEffortEnvOverride, getEffortValueDescription, isEffortLevel, toPersistableEffort } from '../../utils/effort.js'; import { updateSettingsForSource } from '../../utils/settings/settings.js'; +import { canEnableUltracode, describeUltracodeRefusal, ULTRACODE_EFFORT_ARG, ULTRACODE_EFFORT_LEVEL, ULTRACODE_MENU_DESCRIPTION } from '../../utils/workflows/ultracode.js'; const COMMON_HELP_ARGS = ['help', '-h', '--help']; type EffortCommandResult = { message: string; effortUpdate?: { value: EffortValue | undefined; + /** Session-only flag that rides alongside xhigh. Never persisted. */ + ultracode?: boolean; }; }; + +/** + * `/effort ultracode` — xhigh plus standing workflow orchestration. + * + * Session-only by design: it multiplies token spend on every turn, so it must + * not survive into a session the user did not opt in from. + */ +function setUltracode(model: string): EffortCommandResult { + const allowed = canEnableUltracode(model); + if (!allowed.ok) { + return { + message: `${describeUltracodeRefusal(allowed.reason, model)} Valid options are: low, medium, high, xhigh, max, ultracode, auto` + }; + } + logEvent('tengu_effort_command', { + effort: ULTRACODE_EFFORT_ARG as AnalyticsMetadata_I_VERIFIED_THIS_IS_NOT_CODE_OR_FILEPATHS + }); + const envOverride = getEffortEnvOverride(); + if (envOverride !== undefined && envOverride !== null && envOverride !== ULTRACODE_EFFORT_LEVEL) { + return { + message: `CLAUDE_CODE_EFFORT_LEVEL=${process.env.CLAUDE_CODE_EFFORT_LEVEL} overrides effort this session — clear it and ultracode takes over`, + effortUpdate: { value: ULTRACODE_EFFORT_LEVEL, ultracode: true } + }; + } + return { + message: `Set effort level to ultracode (this session only): ${ULTRACODE_MENU_DESCRIPTION}`, + effortUpdate: { value: ULTRACODE_EFFORT_LEVEL, ultracode: true } + }; +} function setEffortValue(effortValue: EffortValue): EffortCommandResult { const persistable = toPersistableEffort(effortValue); if (persistable !== undefined) { @@ -104,8 +136,11 @@ function unsetEffortLevel(): EffortCommandResult { } }; } -export function executeEffort(args: string): EffortCommandResult { +export function executeEffort(args: string, model = ''): EffortCommandResult { const normalized = args.toLowerCase(); + if (normalized === ULTRACODE_EFFORT_ARG) { + return setUltracode(model); + } if (normalized === 'auto' || normalized === 'unset') { return unsetEffortLevel(); } @@ -149,7 +184,8 @@ function ApplyEffortAndClose(t0) { if (effortUpdate) { setAppState(prev => ({ ...prev, - effortValue: effortUpdate.value + effortValue: effortUpdate.value, + ultracode: effortUpdate.ultracode ?? false })); } onDone(message); @@ -168,16 +204,28 @@ function ApplyEffortAndClose(t0) { React.useEffect(t1, t2); return null; } +function RunEffort({ + args, + onDone +}: { + args: string; + onDone: (result: string) => void; +}): React.ReactNode { + // ultracode is only valid on an xhigh-capable model, so the command needs + // the session's model — which is only readable from a hook. + const model = useMainLoopModel(); + const result = React.useMemo(() => executeEffort(args, model), [args, model]); + return ; +} + export async function call(onDone: LocalJSXCommandOnDone, _context: unknown, args?: string): Promise { args = args?.trim() || ''; if (COMMON_HELP_ARGS.includes(args)) { - onDone('Usage: /effort [low|medium|high|max|auto]\n\nEffort levels:\n- low: Quick, straightforward implementation\n- medium: Balanced approach with standard testing\n- high: Comprehensive implementation with extensive testing\n- max: Maximum capability with deepest reasoning\n- auto: Use the default effort level for your model'); + onDone('Usage: /effort [low|medium|high|xhigh|max|ultracode|auto]\n\nEffort levels:\n- low: Quick, straightforward implementation\n- medium: Balanced approach with standard testing\n- high: Comprehensive implementation with extensive testing\n- xhigh: Extra-high reasoning for especially complex tasks\n- max: Maximum capability with deepest reasoning\n- ultracode: xhigh effort + dynamic workflows for maximum thoroughness (this session only)\n- auto: Use the default effort level for your model'); return; } if (!args || args === 'current' || args === 'status') { return ; } - const result = executeEffort(args); - return ; + return ; } -//# sourceMappingURL=data:application/json;charset=utf-8;base64,eyJ2ZXJzaW9uIjozLCJuYW1lcyI6WyJSZWFjdCIsInVzZU1haW5Mb29wTW9kZWwiLCJBbmFseXRpY3NNZXRhZGF0YV9JX1ZFUklGSUVEX1RISVNfSVNfTk9UX0NPREVfT1JfRklMRVBBVEhTIiwibG9nRXZlbnQiLCJ1c2VBcHBTdGF0ZSIsInVzZVNldEFwcFN0YXRlIiwiTG9jYWxKU1hDb21tYW5kT25Eb25lIiwiRWZmb3J0VmFsdWUiLCJnZXREaXNwbGF5ZWRFZmZvcnRMZXZlbCIsImdldEVmZm9ydEVudk92ZXJyaWRlIiwiZ2V0RWZmb3J0VmFsdWVEZXNjcmlwdGlvbiIsImlzRWZmb3J0TGV2ZWwiLCJ0b1BlcnNpc3RhYmxlRWZmb3J0IiwidXBkYXRlU2V0dGluZ3NGb3JTb3VyY2UiLCJDT01NT05fSEVMUF9BUkdTIiwiRWZmb3J0Q29tbWFuZFJlc3VsdCIsIm1lc3NhZ2UiLCJlZmZvcnRVcGRhdGUiLCJ2YWx1ZSIsInNldEVmZm9ydFZhbHVlIiwiZWZmb3J0VmFsdWUiLCJwZXJzaXN0YWJsZSIsInVuZGVmaW5lZCIsInJlc3VsdCIsImVmZm9ydExldmVsIiwiZXJyb3IiLCJlZmZvcnQiLCJlbnZPdmVycmlkZSIsImVudlJhdyIsInByb2Nlc3MiLCJlbnYiLCJDTEFVREVfQ09ERV9FRkZPUlRfTEVWRUwiLCJkZXNjcmlwdGlvbiIsInN1ZmZpeCIsInNob3dDdXJyZW50RWZmb3J0IiwiYXBwU3RhdGVFZmZvcnQiLCJtb2RlbCIsImVmZmVjdGl2ZVZhbHVlIiwibGV2ZWwiLCJ1bnNldEVmZm9ydExldmVsIiwiZXhlY3V0ZUVmZm9ydCIsImFyZ3MiLCJub3JtYWxpemVkIiwidG9Mb3dlckNhc2UiLCJTaG93Q3VycmVudEVmZm9ydCIsInQwIiwib25Eb25lIiwiX3RlbXAiLCJzIiwiQXBwbHlFZmZvcnRBbmRDbG9zZSIsIiQiLCJfYyIsInNldEFwcFN0YXRlIiwidDEiLCJ0MiIsInByZXYiLCJ1c2VFZmZlY3QiLCJjYWxsIiwiX2NvbnRleHQiLCJQcm9taXNlIiwiUmVhY3ROb2RlIiwidHJpbSIsImluY2x1ZGVzIl0sInNvdXJjZXMiOlsiZWZmb3J0LnRzeCJdLCJzb3VyY2VzQ29udGVudCI6WyJpbXBvcnQgKiBhcyBSZWFjdCBmcm9tICdyZWFjdCdcbmltcG9ydCB7IHVzZU1haW5Mb29wTW9kZWwgfSBmcm9tICcuLi8uLi9ob29rcy91c2VNYWluTG9vcE1vZGVsLmpzJ1xuaW1wb3J0IHtcbiAgdHlwZSBBbmFseXRpY3NNZXRhZGF0YV9JX1ZFUklGSUVEX1RISVNfSVNfTk9UX0NPREVfT1JfRklMRVBBVEhTLFxuICBsb2dFdmVudCxcbn0gZnJvbSAnLi4vLi4vc2VydmljZXMvYW5hbHl0aWNzL2luZGV4LmpzJ1xuaW1wb3J0IHsgdXNlQXBwU3RhdGUsIHVzZVNldEFwcFN0YXRlIH0gZnJvbSAnLi4vLi4vc3RhdGUvQXBwU3RhdGUuanMnXG5pbXBvcnQgdHlwZSB7IExvY2FsSlNYQ29tbWFuZE9uRG9uZSB9IGZyb20gJy4uLy4uL3R5cGVzL2NvbW1hbmQuanMnXG5pbXBvcnQge1xuICB0eXBlIEVmZm9ydFZhbHVlLFxuICBnZXREaXNwbGF5ZWRFZmZvcnRMZXZlbCxcbiAgZ2V0RWZmb3J0RW52T3ZlcnJpZGUsXG4gIGdldEVmZm9ydFZhbHVlRGVzY3JpcHRpb24sXG4gIGlzRWZmb3J0TGV2ZWwsXG4gIHRvUGVyc2lzdGFibGVFZmZvcnQsXG59IGZyb20gJy4uLy4uL3V0aWxzL2VmZm9ydC5qcydcbmltcG9ydCB7IHVwZGF0ZVNldHRpbmdzRm9yU291cmNlIH0gZnJvbSAnLi4vLi4vdXRpbHMvc2V0dGluZ3Mvc2V0dGluZ3MuanMnXG5cbmNvbnN0IENPTU1PTl9IRUxQX0FSR1MgPSBbJ2hlbHAnLCAnLWgnLCAnLS1oZWxwJ11cblxudHlwZSBFZmZvcnRDb21tYW5kUmVzdWx0ID0ge1xuICBtZXNzYWdlOiBzdHJpbmdcbiAgZWZmb3J0VXBkYXRlPzogeyB2YWx1ZTogRWZmb3J0VmFsdWUgfCB1bmRlZmluZWQgfVxufVxuXG5mdW5jdGlvbiBzZXRFZmZvcnRWYWx1ZShlZmZvcnRWYWx1ZTogRWZmb3J0VmFsdWUpOiBFZmZvcnRDb21tYW5kUmVzdWx0IHtcbiAgY29uc3QgcGVyc2lzdGFibGUgPSB0b1BlcnNpc3RhYmxlRWZmb3J0KGVmZm9ydFZhbHVlKVxuICBpZiAocGVyc2lzdGFibGUgIT09IHVuZGVmaW5lZCkge1xuICAgIGNvbnN0IHJlc3VsdCA9IHVwZGF0ZVNldHRpbmdzRm9yU291cmNlKCd1c2VyU2V0dGluZ3MnLCB7XG4gICAgICBlZmZvcnRMZXZlbDogcGVyc2lzdGFibGUsXG4gICAgfSlcbiAgICBpZiAocmVzdWx0LmVycm9yKSB7XG4gICAgICByZXR1cm4ge1xuICAgICAgICBtZXNzYWdlOiBgRmFpbGVkIHRvIHNldCBlZmZvcnQgbGV2ZWw6ICR7cmVzdWx0LmVycm9yLm1lc3NhZ2V9YCxcbiAgICAgIH1cbiAgICB9XG4gIH1cbiAgbG9nRXZlbnQoJ3Rlbmd1X2VmZm9ydF9jb21tYW5kJywge1xuICAgIGVmZm9ydDpcbiAgICAgIGVmZm9ydFZhbHVlIGFzIEFuYWx5dGljc01ldGFkYXRhX0lfVkVSSUZJRURfVEhJU19JU19OT1RfQ09ERV9PUl9GSUxFUEFUSFMsXG4gIH0pXG5cbiAgLy8gRW52IHZhciB3aW5zIGF0IHJlc29sdmVBcHBsaWVkRWZmb3J0IHRpbWUuIE9ubHkgZmxhZyBpdCB3aGVuIGl0IGFjdHVhbGx5XG4gIC8vIGNvbmZsaWN0cyDigJQgaWYgZW52IG1hdGNoZXMgd2hhdCB0aGUgdXNlciBqdXN0IGFza2VkIGZvciwgdGhlIG91dGNvbWUgaXNcbiAgLy8gdGhlIHNhbWUsIHNvIFwiU2V0IGVmZm9ydCB0byBYXCIgaXMgdHJ1ZSBhbmQgdGhlIG5vdGUgaXMgbm9pc2UuXG4gIGNvbnN0IGVudk92ZXJyaWRlID0gZ2V0RWZmb3J0RW52T3ZlcnJpZGUoKVxuICBpZiAoZW52T3ZlcnJpZGUgIT09IHVuZGVmaW5lZCAmJiBlbnZPdmVycmlkZSAhPT0gZWZmb3J0VmFsdWUpIHtcbiAgICBjb25zdCBlbnZSYXcgPSBwcm9jZXNzLmVudi5DTEFVREVfQ09ERV9FRkZPUlRfTEVWRUxcbiAgICBpZiAocGVyc2lzdGFibGUgPT09IHVuZGVmaW5lZCkge1xuICAgICAgcmV0dXJuIHtcbiAgICAgICAgbWVzc2FnZTogYE5vdCBhcHBsaWVkOiBDTEFVREVfQ09ERV9FRkZPUlRfTEVWRUw9JHtlbnZSYXd9IG92ZXJyaWRlcyBlZmZvcnQgdGhpcyBzZXNzaW9uLCBhbmQgJHtlZmZvcnRWYWx1ZX0gaXMgc2Vzc2lvbi1vbmx5IChub3RoaW5nIHNhdmVkKWAsXG4gICAgICAgIGVmZm9ydFVwZGF0ZTogeyB2YWx1ZTogZWZmb3J0VmFsdWUgfSxcbiAgICAgIH1cbiAgICB9XG4gICAgcmV0dXJuIHtcbiAgICAgIG1lc3NhZ2U6IGBDTEFVREVfQ09ERV9FRkZPUlRfTEVWRUw9JHtlbnZSYXd9IG92ZXJyaWRlcyB0aGlzIHNlc3Npb24g4oCUIGNsZWFyIGl0IGFuZCAke2VmZm9ydFZhbHVlfSB0YWtlcyBvdmVyYCxcbiAgICAgIGVmZm9ydFVwZGF0ZTogeyB2YWx1ZTogZWZmb3J0VmFsdWUgfSxcbiAgICB9XG4gIH1cblxuICBjb25zdCBkZXNjcmlwdGlvbiA9IGdldEVmZm9ydFZhbHVlRGVzY3JpcHRpb24oZWZmb3J0VmFsdWUpXG4gIGNvbnN0IHN1ZmZpeCA9IHBlcnNpc3RhYmxlICE9PSB1bmRlZmluZWQgPyAnJyA6ICcgKHRoaXMgc2Vzc2lvbiBvbmx5KSdcbiAgcmV0dXJuIHtcbiAgICBtZXNzYWdlOiBgU2V0IGVmZm9ydCBsZXZlbCB0byAke2VmZm9ydFZhbHVlfSR7c3VmZml4fTogJHtkZXNjcmlwdGlvbn1gLFxuICAgIGVmZm9ydFVwZGF0ZTogeyB2YWx1ZTogZWZmb3J0VmFsdWUgfSxcbiAgfVxufVxuXG5leHBvcnQgZnVuY3Rpb24gc2hvd0N1cnJlbnRFZmZvcnQoXG4gIGFwcFN0YXRlRWZmb3J0OiBFZmZvcnRWYWx1ZSB8IHVuZGVmaW5lZCxcbiAgbW9kZWw6IHN0cmluZyxcbik6IEVmZm9ydENvbW1hbmRSZXN1bHQge1xuICBjb25zdCBlbnZPdmVycmlkZSA9IGdldEVmZm9ydEVudk92ZXJyaWRlKClcbiAgY29uc3QgZWZmZWN0aXZlVmFsdWUgPVxuICAgIGVudk92ZXJyaWRlID09PSBudWxsID8gdW5kZWZpbmVkIDogKGVudk92ZXJyaWRlID8/IGFwcFN0YXRlRWZmb3J0KVxuICBpZiAoZWZmZWN0aXZlVmFsdWUgPT09IHVuZGVmaW5lZCkge1xuICAgIGNvbnN0IGxldmVsID0gZ2V0RGlzcGxheWVkRWZmb3J0TGV2ZWwobW9kZWwsIGFwcFN0YXRlRWZmb3J0KVxuICAgIHJldHVybiB7IG1lc3NhZ2U6IGBFZmZvcnQgbGV2ZWw6IGF1dG8gKGN1cnJlbnRseSAke2xldmVsfSlgIH1cbiAgfVxuICBjb25zdCBkZXNjcmlwdGlvbiA9IGdldEVmZm9ydFZhbHVlRGVzY3JpcHRpb24oZWZmZWN0aXZlVmFsdWUpXG4gIHJldHVybiB7XG4gICAgbWVzc2FnZTogYEN1cnJlbnQgZWZmb3J0IGxldmVsOiAke2VmZmVjdGl2ZVZhbHVlfSAoJHtkZXNjcmlwdGlvbn0pYCxcbiAgfVxufVxuXG5mdW5jdGlvbiB1bnNldEVmZm9ydExldmVsKCk6IEVmZm9ydENvbW1hbmRSZXN1bHQge1xuICBjb25zdCByZXN1bHQgPSB1cGRhdGVTZXR0aW5nc0ZvclNvdXJjZSgndXNlclNldHRpbmdzJywge1xuICAgIGVmZm9ydExldmVsOiB1bmRlZmluZWQsXG4gIH0pXG4gIGlmIChyZXN1bHQuZXJyb3IpIHtcbiAgICByZXR1cm4ge1xuICAgICAgbWVzc2FnZTogYEZhaWxlZCB0byBzZXQgZWZmb3J0IGxldmVsOiAke3Jlc3VsdC5lcnJvci5tZXNzYWdlfWAsXG4gICAgfVxuICB9XG4gIGxvZ0V2ZW50KCd0ZW5ndV9lZmZvcnRfY29tbWFuZCcsIHtcbiAgICBlZmZvcnQ6XG4gICAgICAnYXV0bycgYXMgQW5hbHl0aWNzTWV0YWRhdGFfSV9WRVJJRklFRF9USElTX0lTX05PVF9DT0RFX09SX0ZJTEVQQVRIUyxcbiAgfSlcbiAgLy8gZW52PWF1dG8vdW5zZXQgKG51bGwpIG1hdGNoZXMgd2hhdCAvZWZmb3J0IGF1dG8gYXNrcyBmb3IsIHNvIG9ubHkgd2FyblxuICAvLyB3aGVuIGVudiBpcyBwaW5uaW5nIGEgc3BlY2lmaWMgbGV2ZWwgdGhhdCB3aWxsIGtlZXAgb3ZlcnJpZGluZy5cbiAgY29uc3QgZW52T3ZlcnJpZGUgPSBnZXRFZmZvcnRFbnZPdmVycmlkZSgpXG4gIGlmIChlbnZPdmVycmlkZSAhPT0gdW5kZWZpbmVkICYmIGVudk92ZXJyaWRlICE9PSBudWxsKSB7XG4gICAgY29uc3QgZW52UmF3ID0gcHJvY2Vzcy5lbnYuQ0xBVURFX0NPREVfRUZGT1JUX0xFVkVMXG4gICAgcmV0dXJuIHtcbiAgICAgIG1lc3NhZ2U6IGBDbGVhcmVkIGVmZm9ydCBmcm9tIHNldHRpbmdzLCBidXQgQ0xBVURFX0NPREVfRUZGT1JUX0xFVkVMPSR7ZW52UmF3fSBzdGlsbCBjb250cm9scyB0aGlzIHNlc3Npb25gLFxuICAgICAgZWZmb3J0VXBkYXRlOiB7IHZhbHVlOiB1bmRlZmluZWQgfSxcbiAgICB9XG4gIH1cbiAgcmV0dXJuIHtcbiAgICBtZXNzYWdlOiAnRWZmb3J0IGxldmVsIHNldCB0byBhdXRvJyxcbiAgICBlZmZvcnRVcGRhdGU6IHsgdmFsdWU6IHVuZGVmaW5lZCB9LFxuICB9XG59XG5cbmV4cG9ydCBmdW5jdGlvbiBleGVjdXRlRWZmb3J0KGFyZ3M6IHN0cmluZyk6IEVmZm9ydENvbW1hbmRSZXN1bHQge1xuICBjb25zdCBub3JtYWxpemVkID0gYXJncy50b0xvd2VyQ2FzZSgpXG4gIGlmIChub3JtYWxpemVkID09PSAnYXV0bycgfHwgbm9ybWFsaXplZCA9PT0gJ3Vuc2V0Jykge1xuICAgIHJldHVybiB1bnNldEVmZm9ydExldmVsKClcbiAgfVxuXG4gIGlmICghaXNFZmZvcnRMZXZlbChub3JtYWxpemVkKSkge1xuICAgIHJldHVybiB7XG4gICAgICBtZXNzYWdlOiBgSW52YWxpZCBhcmd1bWVudDogJHthcmdzfS4gVmFsaWQgb3B0aW9ucyBhcmU6IGxvdywgbWVkaXVtLCBoaWdoLCBtYXgsIGF1dG9gLFxuICAgIH1cbiAgfVxuXG4gIHJldHVybiBzZXRFZmZvcnRWYWx1ZShub3JtYWxpemVkKVxufVxuXG5mdW5jdGlvbiBTaG93Q3VycmVudEVmZm9ydCh7XG4gIG9uRG9uZSxcbn06IHtcbiAgb25Eb25lOiAocmVzdWx0OiBzdHJpbmcpID0+IHZvaWRcbn0pOiBSZWFjdC5SZWFjdE5vZGUge1xuICBjb25zdCBlZmZvcnRWYWx1ZSA9IHVzZUFwcFN0YXRlKHMgPT4gcy5lZmZvcnRWYWx1ZSlcbiAgY29uc3QgbW9kZWwgPSB1c2VNYWluTG9vcE1vZGVsKClcbiAgY29uc3QgeyBtZXNzYWdlIH0gPSBzaG93Q3VycmVudEVmZm9ydChlZmZvcnRWYWx1ZSwgbW9kZWwpXG4gIG9uRG9uZShtZXNzYWdlKVxuICByZXR1cm4gbnVsbFxufVxuXG5mdW5jdGlvbiBBcHBseUVmZm9ydEFuZENsb3NlKHtcbiAgcmVzdWx0LFxuICBvbkRvbmUsXG59OiB7XG4gIHJlc3VsdDogRWZmb3J0Q29tbWFuZFJlc3VsdFxuICBvbkRvbmU6IChyZXN1bHQ6IHN0cmluZykgPT4gdm9pZFxufSk6IFJlYWN0LlJlYWN0Tm9kZSB7XG4gIGNvbnN0IHNldEFwcFN0YXRlID0gdXNlU2V0QXBwU3RhdGUoKVxuICBjb25zdCB7IGVmZm9ydFVwZGF0ZSwgbWVzc2FnZSB9ID0gcmVzdWx0XG4gIFJlYWN0LnVzZUVmZmVjdCgoKSA9PiB7XG4gICAgaWYgKGVmZm9ydFVwZGF0ZSkge1xuICAgICAgc2V0QXBwU3RhdGUocHJldiA9PiAoe1xuICAgICAgICAuLi5wcmV2LFxuICAgICAgICBlZmZvcnRWYWx1ZTogZWZmb3J0VXBkYXRlLnZhbHVlLFxuICAgICAgfSkpXG4gICAgfVxuICAgIG9uRG9uZShtZXNzYWdlKVxuICB9LCBbc2V0QXBwU3RhdGUsIGVmZm9ydFVwZGF0ZSwgbWVzc2FnZSwgb25Eb25lXSlcbiAgcmV0dXJuIG51bGxcbn1cblxuZXhwb3J0IGFzeW5jIGZ1bmN0aW9uIGNhbGwoXG4gIG9uRG9uZTogTG9jYWxKU1hDb21tYW5kT25Eb25lLFxuICBfY29udGV4dDogdW5rbm93bixcbiAgYXJncz86IHN0cmluZyxcbik6IFByb21pc2U8UmVhY3QuUmVhY3ROb2RlPiB7XG4gIGFyZ3MgPSBhcmdzPy50cmltKCkgfHwgJydcblxuICBpZiAoQ09NTU9OX0hFTFBfQVJHUy5pbmNsdWRlcyhhcmdzKSkge1xuICAgIG9uRG9uZShcbiAgICAgICdVc2FnZTogL2VmZm9ydCBbbG93fG1lZGl1bXxoaWdofG1heHxhdXRvXVxcblxcbkVmZm9ydCBsZXZlbHM6XFxuLSBsb3c6IFF1aWNrLCBzdHJhaWdodGZvcndhcmQgaW1wbGVtZW50YXRpb25cXG4tIG1lZGl1bTogQmFsYW5jZWQgYXBwcm9hY2ggd2l0aCBzdGFuZGFyZCB0ZXN0aW5nXFxuLSBoaWdoOiBDb21wcmVoZW5zaXZlIGltcGxlbWVudGF0aW9uIHdpdGggZXh0ZW5zaXZlIHRlc3RpbmdcXG4tIG1heDogTWF4aW11bSBjYXBhYmlsaXR5IHdpdGggZGVlcGVzdCByZWFzb25pbmcgKE9wdXMgNC42IG9ubHkpXFxuLSBhdXRvOiBVc2UgdGhlIGRlZmF1bHQgZWZmb3J0IGxldmVsIGZvciB5b3VyIG1vZGVsJyxcbiAgICApXG4gICAgcmV0dXJuXG4gIH1cblxuICBpZiAoIWFyZ3MgfHwgYXJncyA9PT0gJ2N1cnJlbnQnIHx8IGFyZ3MgPT09ICdzdGF0dXMnKSB7XG4gICAgcmV0dXJuIDxTaG93Q3VycmVudEVmZm9ydCBvbkRvbmU9e29uRG9uZX0gLz5cbiAgfVxuXG4gIGNvbnN0IHJlc3VsdCA9IGV4ZWN1dGVFZmZvcnQoYXJncylcbiAgcmV0dXJuIDxBcHBseUVmZm9ydEFuZENsb3NlIHJlc3VsdD17cmVzdWx0fSBvbkRvbmU9e29uRG9uZX0gLz5cbn1cbiJdLCJtYXBwaW5ncyI6IjtBQUFBLE9BQU8sS0FBS0EsS0FBSyxNQUFNLE9BQU87QUFDOUIsU0FBU0MsZ0JBQWdCLFFBQVEsaUNBQWlDO0FBQ2xFLFNBQ0UsS0FBS0MsMERBQTBELEVBQy9EQyxRQUFRLFFBQ0gsbUNBQW1DO0FBQzFDLFNBQVNDLFdBQVcsRUFBRUMsY0FBYyxRQUFRLHlCQUF5QjtBQUNyRSxjQUFjQyxxQkFBcUIsUUFBUSx3QkFBd0I7QUFDbkUsU0FDRSxLQUFLQyxXQUFXLEVBQ2hCQyx1QkFBdUIsRUFDdkJDLG9CQUFvQixFQUNwQkMseUJBQXlCLEVBQ3pCQyxhQUFhLEVBQ2JDLG1CQUFtQixRQUNkLHVCQUF1QjtBQUM5QixTQUFTQyx1QkFBdUIsUUFBUSxrQ0FBa0M7QUFFMUUsTUFBTUMsZ0JBQWdCLEdBQUcsQ0FBQyxNQUFNLEVBQUUsSUFBSSxFQUFFLFFBQVEsQ0FBQztBQUVqRCxLQUFLQyxtQkFBbUIsR0FBRztFQUN6QkMsT0FBTyxFQUFFLE1BQU07RUFDZkMsWUFBWSxDQUFDLEVBQUU7SUFBRUMsS0FBSyxFQUFFWCxXQUFXLEdBQUcsU0FBUztFQUFDLENBQUM7QUFDbkQsQ0FBQztBQUVELFNBQVNZLGNBQWNBLENBQUNDLFdBQVcsRUFBRWIsV0FBVyxDQUFDLEVBQUVRLG1CQUFtQixDQUFDO0VBQ3JFLE1BQU1NLFdBQVcsR0FBR1QsbUJBQW1CLENBQUNRLFdBQVcsQ0FBQztFQUNwRCxJQUFJQyxXQUFXLEtBQUtDLFNBQVMsRUFBRTtJQUM3QixNQUFNQyxNQUFNLEdBQUdWLHVCQUF1QixDQUFDLGNBQWMsRUFBRTtNQUNyRFcsV0FBVyxFQUFFSDtJQUNmLENBQUMsQ0FBQztJQUNGLElBQUlFLE1BQU0sQ0FBQ0UsS0FBSyxFQUFFO01BQ2hCLE9BQU87UUFDTFQsT0FBTyxFQUFFLCtCQUErQk8sTUFBTSxDQUFDRSxLQUFLLENBQUNULE9BQU87TUFDOUQsQ0FBQztJQUNIO0VBQ0Y7RUFDQWIsUUFBUSxDQUFDLHNCQUFzQixFQUFFO0lBQy9CdUIsTUFBTSxFQUNKTixXQUFXLElBQUlsQjtFQUNuQixDQUFDLENBQUM7O0VBRUY7RUFDQTtFQUNBO0VBQ0EsTUFBTXlCLFdBQVcsR0FBR2xCLG9CQUFvQixDQUFDLENBQUM7RUFDMUMsSUFBSWtCLFdBQVcsS0FBS0wsU0FBUyxJQUFJSyxXQUFXLEtBQUtQLFdBQVcsRUFBRTtJQUM1RCxNQUFNUSxNQUFNLEdBQUdDLE9BQU8sQ0FBQ0MsR0FBRyxDQUFDQyx3QkFBd0I7SUFDbkQsSUFBSVYsV0FBVyxLQUFLQyxTQUFTLEVBQUU7TUFDN0IsT0FBTztRQUNMTixPQUFPLEVBQUUseUNBQXlDWSxNQUFNLHVDQUF1Q1IsV0FBVyxrQ0FBa0M7UUFDNUlILFlBQVksRUFBRTtVQUFFQyxLQUFLLEVBQUVFO1FBQVk7TUFDckMsQ0FBQztJQUNIO0lBQ0EsT0FBTztNQUNMSixPQUFPLEVBQUUsNEJBQTRCWSxNQUFNLDBDQUEwQ1IsV0FBVyxhQUFhO01BQzdHSCxZQUFZLEVBQUU7UUFBRUMsS0FBSyxFQUFFRTtNQUFZO0lBQ3JDLENBQUM7RUFDSDtFQUVBLE1BQU1ZLFdBQVcsR0FBR3RCLHlCQUF5QixDQUFDVSxXQUFXLENBQUM7RUFDMUQsTUFBTWEsTUFBTSxHQUFHWixXQUFXLEtBQUtDLFNBQVMsR0FBRyxFQUFFLEdBQUcsc0JBQXNCO0VBQ3RFLE9BQU87SUFDTE4sT0FBTyxFQUFFLHVCQUF1QkksV0FBVyxHQUFHYSxNQUFNLEtBQUtELFdBQVcsRUFBRTtJQUN0RWYsWUFBWSxFQUFFO01BQUVDLEtBQUssRUFBRUU7SUFBWTtFQUNyQyxDQUFDO0FBQ0g7QUFFQSxPQUFPLFNBQVNjLGlCQUFpQkEsQ0FDL0JDLGNBQWMsRUFBRTVCLFdBQVcsR0FBRyxTQUFTLEVBQ3ZDNkIsS0FBSyxFQUFFLE1BQU0sQ0FDZCxFQUFFckIsbUJBQW1CLENBQUM7RUFDckIsTUFBTVksV0FBVyxHQUFHbEIsb0JBQW9CLENBQUMsQ0FBQztFQUMxQyxNQUFNNEIsY0FBYyxHQUNsQlYsV0FBVyxLQUFLLElBQUksR0FBR0wsU0FBUyxHQUFJSyxXQUFXLElBQUlRLGNBQWU7RUFDcEUsSUFBSUUsY0FBYyxLQUFLZixTQUFTLEVBQUU7SUFDaEMsTUFBTWdCLEtBQUssR0FBRzlCLHVCQUF1QixDQUFDNEIsS0FBSyxFQUFFRCxjQUFjLENBQUM7SUFDNUQsT0FBTztNQUFFbkIsT0FBTyxFQUFFLGlDQUFpQ3NCLEtBQUs7SUFBSSxDQUFDO0VBQy9EO0VBQ0EsTUFBTU4sV0FBVyxHQUFHdEIseUJBQXlCLENBQUMyQixjQUFjLENBQUM7RUFDN0QsT0FBTztJQUNMckIsT0FBTyxFQUFFLHlCQUF5QnFCLGNBQWMsS0FBS0wsV0FBVztFQUNsRSxDQUFDO0FBQ0g7QUFFQSxTQUFTTyxnQkFBZ0JBLENBQUEsQ0FBRSxFQUFFeEIsbUJBQW1CLENBQUM7RUFDL0MsTUFBTVEsTUFBTSxHQUFHVix1QkFBdUIsQ0FBQyxjQUFjLEVBQUU7SUFDckRXLFdBQVcsRUFBRUY7RUFDZixDQUFDLENBQUM7RUFDRixJQUFJQyxNQUFNLENBQUNFLEtBQUssRUFBRTtJQUNoQixPQUFPO01BQ0xULE9BQU8sRUFBRSwrQkFBK0JPLE1BQU0sQ0FBQ0UsS0FBSyxDQUFDVCxPQUFPO0lBQzlELENBQUM7RUFDSDtFQUNBYixRQUFRLENBQUMsc0JBQXNCLEVBQUU7SUFDL0J1QixNQUFNLEVBQ0osTUFBTSxJQUFJeEI7RUFDZCxDQUFDLENBQUM7RUFDRjtFQUNBO0VBQ0EsTUFBTXlCLFdBQVcsR0FBR2xCLG9CQUFvQixDQUFDLENBQUM7RUFDMUMsSUFBSWtCLFdBQVcsS0FBS0wsU0FBUyxJQUFJSyxXQUFXLEtBQUssSUFBSSxFQUFFO0lBQ3JELE1BQU1DLE1BQU0sR0FBR0MsT0FBTyxDQUFDQyxHQUFHLENBQUNDLHdCQUF3QjtJQUNuRCxPQUFPO01BQ0xmLE9BQU8sRUFBRSw4REFBOERZLE1BQU0sOEJBQThCO01BQzNHWCxZQUFZLEVBQUU7UUFBRUMsS0FBSyxFQUFFSTtNQUFVO0lBQ25DLENBQUM7RUFDSDtFQUNBLE9BQU87SUFDTE4sT0FBTyxFQUFFLDBCQUEwQjtJQUNuQ0MsWUFBWSxFQUFFO01BQUVDLEtBQUssRUFBRUk7SUFBVTtFQUNuQyxDQUFDO0FBQ0g7QUFFQSxPQUFPLFNBQVNrQixhQUFhQSxDQUFDQyxJQUFJLEVBQUUsTUFBTSxDQUFDLEVBQUUxQixtQkFBbUIsQ0FBQztFQUMvRCxNQUFNMkIsVUFBVSxHQUFHRCxJQUFJLENBQUNFLFdBQVcsQ0FBQyxDQUFDO0VBQ3JDLElBQUlELFVBQVUsS0FBSyxNQUFNLElBQUlBLFVBQVUsS0FBSyxPQUFPLEVBQUU7SUFDbkQsT0FBT0gsZ0JBQWdCLENBQUMsQ0FBQztFQUMzQjtFQUVBLElBQUksQ0FBQzVCLGFBQWEsQ0FBQytCLFVBQVUsQ0FBQyxFQUFFO0lBQzlCLE9BQU87TUFDTDFCLE9BQU8sRUFBRSxxQkFBcUJ5QixJQUFJO0lBQ3BDLENBQUM7RUFDSDtFQUVBLE9BQU90QixjQUFjLENBQUN1QixVQUFVLENBQUM7QUFDbkM7QUFFQSxTQUFBRSxrQkFBQUMsRUFBQTtFQUEyQjtJQUFBQztFQUFBLElBQUFELEVBSTFCO0VBQ0MsTUFBQXpCLFdBQUEsR0FBb0JoQixXQUFXLENBQUMyQyxLQUFrQixDQUFDO0VBQ25ELE1BQUFYLEtBQUEsR0FBY25DLGdCQUFnQixDQUFDLENBQUM7RUFDaEM7SUFBQWU7RUFBQSxJQUFvQmtCLGlCQUFpQixDQUFDZCxXQUFXLEVBQUVnQixLQUFLLENBQUM7RUFDekRVLE1BQU0sQ0FBQzlCLE9BQU8sQ0FBQztFQUFBLE9BQ1IsSUFBSTtBQUFBO0FBVGIsU0FBQStCLE1BQUFDLENBQUE7RUFBQSxPQUt1Q0EsQ0FBQyxDQUFBNUIsV0FBWTtBQUFBO0FBT3BELFNBQUE2QixvQkFBQUosRUFBQTtFQUFBLE1BQUFLLENBQUEsR0FBQUMsRUFBQTtFQUE2QjtJQUFBNUIsTUFBQTtJQUFBdUI7RUFBQSxJQUFBRCxFQU01QjtFQUNDLE1BQUFPLFdBQUEsR0FBb0IvQyxjQUFjLENBQUMsQ0FBQztFQUNwQztJQUFBWSxZQUFBO0lBQUFEO0VBQUEsSUFBa0NPLE1BQU07RUFBQSxJQUFBOEIsRUFBQTtFQUFBLElBQUFDLEVBQUE7RUFBQSxJQUFBSixDQUFBLFFBQUFqQyxZQUFBLElBQUFpQyxDQUFBLFFBQUFsQyxPQUFBLElBQUFrQyxDQUFBLFFBQUFKLE1BQUEsSUFBQUksQ0FBQSxRQUFBRSxXQUFBO0lBQ3hCQyxFQUFBLEdBQUFBLENBQUE7TUFDZCxJQUFJcEMsWUFBWTtRQUNkbUMsV0FBVyxDQUFDRyxJQUFBLEtBQVM7VUFBQSxHQUNoQkEsSUFBSTtVQUFBbkMsV0FBQSxFQUNNSCxZQUFZLENBQUFDO1FBQzNCLENBQUMsQ0FBQyxDQUFDO01BQUE7TUFFTDRCLE1BQU0sQ0FBQzlCLE9BQU8sQ0FBQztJQUFBLENBQ2hCO0lBQUVzQyxFQUFBLElBQUNGLFdBQVcsRUFBRW5DLFlBQVksRUFBRUQsT0FBTyxFQUFFOEIsTUFBTSxDQUFDO0lBQUFJLENBQUEsTUFBQWpDLFlBQUE7SUFBQWlDLENBQUEsTUFBQWxDLE9BQUE7SUFBQWtDLENBQUEsTUFBQUosTUFBQTtJQUFBSSxDQUFBLE1BQUFFLFdBQUE7SUFBQUYsQ0FBQSxNQUFBRyxFQUFBO0lBQUFILENBQUEsTUFBQUksRUFBQTtFQUFBO0lBQUFELEVBQUEsR0FBQUgsQ0FBQTtJQUFBSSxFQUFBLEdBQUFKLENBQUE7RUFBQTtFQVIvQ2xELEtBQUssQ0FBQXdELFNBQVUsQ0FBQ0gsRUFRZixFQUFFQyxFQUE0QyxDQUFDO0VBQUEsT0FDekMsSUFBSTtBQUFBO0FBR2IsT0FBTyxlQUFlRyxJQUFJQSxDQUN4QlgsTUFBTSxFQUFFeEMscUJBQXFCLEVBQzdCb0QsUUFBUSxFQUFFLE9BQU8sRUFDakJqQixJQUFhLENBQVIsRUFBRSxNQUFNLENBQ2QsRUFBRWtCLE9BQU8sQ0FBQzNELEtBQUssQ0FBQzRELFNBQVMsQ0FBQyxDQUFDO0VBQzFCbkIsSUFBSSxHQUFHQSxJQUFJLEVBQUVvQixJQUFJLENBQUMsQ0FBQyxJQUFJLEVBQUU7RUFFekIsSUFBSS9DLGdCQUFnQixDQUFDZ0QsUUFBUSxDQUFDckIsSUFBSSxDQUFDLEVBQUU7SUFDbkNLLE1BQU0sQ0FDSixrVkFDRixDQUFDO0lBQ0Q7RUFDRjtFQUVBLElBQUksQ0FBQ0wsSUFBSSxJQUFJQSxJQUFJLEtBQUssU0FBUyxJQUFJQSxJQUFJLEtBQUssUUFBUSxFQUFFO0lBQ3BELE9BQU8sQ0FBQyxpQkFBaUIsQ0FBQyxNQUFNLENBQUMsQ0FBQ0ssTUFBTSxDQUFDLEdBQUc7RUFDOUM7RUFFQSxNQUFNdkIsTUFBTSxHQUFHaUIsYUFBYSxDQUFDQyxJQUFJLENBQUM7RUFDbEMsT0FBTyxDQUFDLG1CQUFtQixDQUFDLE1BQU0sQ0FBQyxDQUFDbEIsTUFBTSxDQUFDLENBQUMsTUFBTSxDQUFDLENBQUN1QixNQUFNLENBQUMsR0FBRztBQUNoRSIsImlnbm9yZUxpc3QiOltdfQ== diff --git a/src/commands/workflows/index.ts b/src/commands/workflows/index.ts index c170c49a..4f383664 100644 --- a/src/commands/workflows/index.ts +++ b/src/commands/workflows/index.ts @@ -1,34 +1,12 @@ -// @generated stub from scan-missing-imports -// 该文件自动生成,对应 ant-internal 的 feature() gated 模块。 -// 所有外部 build 的代码路径在 DCE 后都不会真的执行这里的代码,这只是 -// bun build resolver 的占位符。 -const __target = function noop() {} -const __handler: ProxyHandler = { - get(_t, prop) { - if (prop === '__esModule') return true - if (prop === 'default') return new Proxy(__target, __handler) - if (prop === Symbol.toPrimitive) return () => undefined - if (prop === Symbol.iterator) return function* () {} - if (prop === Symbol.asyncIterator) return async function* () {} - if (prop === 'then') return undefined - return new Proxy(__target, __handler) - }, - apply() { - return new Proxy(__target, __handler) - }, - construct() { - return new Proxy(__target, __handler) - }, -} -const stub: any = new Proxy(__target, __handler) -export default stub -export const __stubMissing = true -// 兼容常见的命名导出 —— 没列在这里的也会通过 default Proxy 兜底 -export const createCachedMCState = stub -export const isCachedMicrocompactEnabled = stub -export const isModelSupportedForCacheEditing = stub -export const getCachedMCConfig = stub -export const markToolsSentToAPI = stub -export const resetCachedMCState = stub -export const checkProtectedNamespace = stub -export const getCoordinatorUserContext = stub +import type { Command } from '../../commands.js' +import { areWorkflowsEnabled } from '../../utils/workflows/enabled.js' + +const workflows = { + type: 'local-jsx', + name: 'workflows', + description: 'Watch and manage dynamic workflow runs', + isEnabled: () => areWorkflowsEnabled(), + load: () => import('./workflows.js'), +} satisfies Command + +export default workflows diff --git a/src/commands/workflows/workflows.tsx b/src/commands/workflows/workflows.tsx new file mode 100644 index 00000000..81a7a0fc --- /dev/null +++ b/src/commands/workflows/workflows.tsx @@ -0,0 +1,200 @@ +import figures from 'figures' +import React, { useCallback, useMemo, useState } from 'react' +import type { LocalJSXCommandContext } from '../../commands.js' +import { Byline } from '../../components/design-system/Byline.js' +import { KeyboardShortcutHint } from '../../components/design-system/KeyboardShortcutHint.js' +import { + getTaskStatusColor, + getTaskStatusIcon, +} from '../../components/tasks/taskStatusUtils.js' +import { WorkflowDetailDialog } from '../../components/tasks/WorkflowDetailDialog.js' +import type { KeyboardEvent } from '../../ink/events/keyboard-event.js' +import { Box, Text } from '../../ink.js' +import { + buildResumePrompt, + killWorkflowTask, + pauseWorkflowTask, + retryWorkflowAgent, + skipWorkflowAgent, + type LocalWorkflowTaskState, +} from '../../tasks/LocalWorkflowTask/LocalWorkflowTask.js' +import type { LocalJSXCommandOnDone } from '../../types/command.js' +import type { DeepImmutable } from '../../types/utils.js' +import { saveWorkflowScript } from '../../utils/workflows/save.js' + +/** + * `/workflows` — list this session's dynamic workflow runs and open one. + * + * Separate from `/tasks` on purpose: a workflow's interesting state is its + * phase/agent tree, which the generic background-task list cannot show, and + * during a large run the workflow rows would bury every other task. + */ +export async function call( + onDone: LocalJSXCommandOnDone, + context: LocalJSXCommandContext, +): Promise { + return +} + +function WorkflowsDialog({ + toolUseContext, + onDone, +}: { + toolUseContext: LocalJSXCommandContext + onDone: LocalJSXCommandOnDone +}): React.ReactNode { + const appState = toolUseContext.getAppState() + const setAppState = toolUseContext.setAppState + const [selectedId, setSelectedId] = useState(null) + const [cursor, setCursor] = useState(0) + const [saveScope, setSaveScope] = useState<'user' | 'project' | null>(null) + + const runs = useMemo(() => { + const all = Object.values(appState.tasks ?? {}).filter( + (task): task is DeepImmutable => + task.type === 'local_workflow', + ) + // Newest first: during a long session the run you just started is the one + // you came here to watch. + return [...all].sort((a, b) => b.startTime - a.startTime) + }, [appState.tasks]) + + const selected = selectedId + ? (runs.find(run => run.id === selectedId) ?? null) + : null + + const close = useCallback(() => onDone(), [onDone]) + + const handleKeyDown = useCallback( + (event: KeyboardEvent) => { + if (selected) return + const key = event.key + if (key === 'up') { + event.preventDefault() + setCursor(prev => Math.max(0, prev - 1)) + return + } + if (key === 'down') { + event.preventDefault() + setCursor(prev => Math.min(Math.max(0, runs.length - 1), prev + 1)) + return + } + if (key === 'return' || key === 'right') { + event.preventDefault() + const run = runs[cursor] + if (run) setSelectedId(run.id) + return + } + if (key === 'x') { + event.preventDefault() + const run = runs[cursor] + if (run && run.status === 'running') killWorkflowTask(run.id, setAppState) + return + } + if (key === 'escape' || key === 'left' || key === ' ') { + event.preventDefault() + close() + } + }, + [close, cursor, runs, selected, setAppState], + ) + + if (selected) { + return ( + setSelectedId(null)} + onKill={ + selected.status === 'running' + ? () => killWorkflowTask(selected.id, setAppState) + : undefined + } + onPause={ + selected.status === 'running' + ? () => { + if (pauseWorkflowTask(selected.id, setAppState)) { + onDone(buildResumePrompt(selected as LocalWorkflowTaskState), { + display: 'system', + }) + } + } + : undefined + } + onSkipAgent={ + selected.status === 'running' + ? key => skipWorkflowAgent(selected.id, key, setAppState) + : undefined + } + onRetryAgent={ + selected.status === 'running' + ? key => retryWorkflowAgent(selected.id, key, setAppState) + : undefined + } + ultracode={appState.ultracode === true} + saveScope={saveScope} + onToggleSaveScope={() => + setSaveScope(prev => + prev === null ? 'project' : prev === 'project' ? 'user' : 'project', + ) + } + onSave={async () => { + const scope = saveScope ?? 'project' + const result = await saveWorkflowScript({ + script: selected.script, + scope, + }) + onDone( + 'error' in result + ? `Could not save workflow: ${result.error}` + : `Saved /${result.name} to ${result.filePath}`, + { display: 'system' }, + ) + }} + /> + ) + } + + return ( + + + Dynamic workflows + {runs.length === 0 ? ( + + No workflow runs in this session. Ask Claude to "use a + workflow", or run a saved one with /<name>. + + ) : ( + runs.map((run, index) => { + const isSelected = index === cursor + return ( + + + {`${isSelected ? figures.pointer : ' '} `} + + + {getTaskStatusIcon(run.status)} + + {` ${run.workflowName ?? 'workflow'}`} + + {` ${run.agentCount} agents · ${run.totalTokens} tok · ${run.status}`} + + + ) + }) + )} + + + + + + + + + ) +} diff --git a/src/components/PromptInput/PromptInput.tsx b/src/components/PromptInput/PromptInput.tsx index 0f7263db..3a8fd0c2 100644 --- a/src/components/PromptInput/PromptInput.tsx +++ b/src/components/PromptInput/PromptInput.tsx @@ -16,6 +16,8 @@ import { companionReservedColumns } from '../../buddy/CompanionSprite.js'; import { findBuddyTriggerPositions, useBuddyNotification } from '../../buddy/useBuddyNotification.js'; import { FastModePicker } from '../../commands/fast/fast.js'; import { isUltrareviewEnabled } from '../../commands/review/ultrareviewEnabled.js'; +import { isWorkflowKeywordTriggerEnabled } from '../../utils/workflows/enabled.js'; +import { findKeywordRanges } from '../../utils/workflows/keyword.js'; import { getNativeCSIuTerminalDisplayName } from '../../commands/terminalSetup/terminalSetup.js'; import { type Command, hasCommand } from '../../commands.js'; import { useIsModalOverlayActive } from '../../context/overlayContext.js'; @@ -518,6 +520,24 @@ function PromptInput({ const ultraplanLaunching = useAppState(s => s.ultraplanLaunching); const ultraplanTriggers = useMemo(() => feature('ULTRAPLAN') && !ultraplanSessionUrl && !ultraplanLaunching ? findUltraplanTriggerPositions(displayedValue) : [], [displayedValue, ultraplanSessionUrl, ultraplanLaunching]); const ultrareviewTriggers = useMemo(() => isUltrareviewEnabled() ? findUltrareviewTriggerPositions(displayedValue) : [], [displayedValue]); + // `ultracode` opts this turn into multi-agent orchestration. The highlight is + // the only warning the user gets before a one-line prompt becomes a run that + // spawns dozens of agents, so it must be visible before submit. + const [ultracodeKeywordDismissed, setUltracodeKeywordDismissed] = useState(false); + const ultracodeTriggers = useMemo(() => isWorkflowKeywordTriggerEnabled() && !ultracodeKeywordDismissed ? findKeywordRanges(displayedValue) : [], [displayedValue, ultracodeKeywordDismissed]); + const hasUltracodeKeyword = useMemo(() => isWorkflowKeywordTriggerEnabled() ? findKeywordRanges(displayedValue).length > 0 : false, [displayedValue]); + // A fresh prompt is opted in again — the dismissal is per-prompt, not sticky. + // The AppState mirror is what actually suppresses the reminder; the local + // flag only drives the highlight, so both have to move together. + useEffect(() => { + if (!hasUltracodeKeyword && ultracodeKeywordDismissed) { + setUltracodeKeywordDismissed(false); + setAppState(prev => prev.suppressWorkflowKeyword ? { + ...prev, + suppressWorkflowKeyword: false + } : prev); + } + }, [hasUltracodeKeyword, ultracodeKeywordDismissed, setAppState]); const btwTriggers = useMemo(() => findBtwTriggerPositions(displayedValue), [displayedValue]); const buddyTriggers = useMemo(() => findBuddyTriggerPositions(displayedValue), [displayedValue]); const slashCommandTriggers = useMemo(() => { @@ -722,6 +742,19 @@ function PromptInput({ } } + // Same rainbow treatment for the ultracode keyword + for (const trigger of ultracodeTriggers) { + for (let i = trigger.start; i < trigger.end; i++) { + highlights.push({ + start: i, + end: i + 1, + color: getRainbowColor(i - trigger.start), + shimmerColor: getRainbowColor(i - trigger.start, true), + priority: 10 + }); + } + } + // Rainbow for /buddy for (const trigger of buddyTriggers) { for (let i = trigger.start; i < trigger.end; i++) { @@ -735,7 +768,7 @@ function PromptInput({ } } return highlights; - }, [isSearchingHistory, historyQuery, historyMatch, historyFailedMatch, cursorOffset, btwTriggers, imageRefPositions, memberMentionHighlights, slashCommandTriggers, tokenBudgetTriggers, slackChannelTriggers, displayedValue, voiceInterimRange, thinkTriggers, ultraplanTriggers, ultrareviewTriggers, buddyTriggers]); + }, [isSearchingHistory, historyQuery, historyMatch, historyFailedMatch, cursorOffset, btwTriggers, imageRefPositions, memberMentionHighlights, slashCommandTriggers, tokenBudgetTriggers, slackChannelTriggers, displayedValue, voiceInterimRange, thinkTriggers, ultraplanTriggers, ultrareviewTriggers, ultracodeTriggers, buddyTriggers]); const { addNotification, removeNotification @@ -766,6 +799,41 @@ function PromptInput({ removeNotification('ultraplan-active'); } }, [addNotification, removeNotification, ultraplanTriggers.length]); + useEffect(() => { + if (ultracodeTriggers.length) { + addNotification({ + key: 'ultracode-active', + text: 'Ultracode: this prompt runs as a workflow · opt+w to undo', + priority: 'immediate', + timeoutMs: 5000 + }); + } else { + removeNotification('ultracode-active'); + } + }, [addNotification, removeNotification, ultracodeTriggers.length]); + useKeybinding('chat:workflowKeywordToggle', () => { + if (!hasUltracodeKeyword) return; + setUltracodeKeywordDismissed(prev => { + const next = !prev; + setAppState(state => ({ + ...state, + suppressWorkflowKeyword: next + })); + if (next) { + addNotification({ + key: 'workflow-keyword-ignored', + text: 'Ultracode keyword ignored for this prompt · opt+w to undo', + priority: 'immediate', + timeoutMs: 5000 + }); + } else { + removeNotification('workflow-keyword-ignored'); + } + return next; + }); + }, { + context: 'Chat' + }); useEffect(() => { if (isUltrareviewEnabled() && ultrareviewTriggers.length) { addNotification({ diff --git a/src/components/Settings/Config.tsx b/src/components/Settings/Config.tsx index 37ee93c9..1bafffe8 100644 --- a/src/components/Settings/Config.tsx +++ b/src/components/Settings/Config.tsx @@ -36,6 +36,7 @@ import { useIsInsideModal } from '../../context/modalContext.js'; import { SearchBox } from '../SearchBox.js'; import { isSupportedTerminal, hasAccessToIDEExtensionDiffFeature } from '../../utils/ide.js'; import { getInitialSettings, getSettingsForSource, updateSettingsForSource } from '../../utils/settings/settings.js'; +import { getWorkflowSizeGuidelineFromSettings, WORKFLOW_SIZE_GUIDELINES } from '../../utils/workflows/enabled.js'; import { getUserMsgOptIn, setUserMsgOptIn } from '../../bootstrap/state.js'; import { DEFAULT_OUTPUT_STYLE_NAME } from 'src/constants/outputStyles.js'; import { isEnvTruthy, isRunningOnHomespace } from 'src/utils/envUtils.js'; @@ -431,6 +432,58 @@ export function Config({ enabled: enabled_3 }); } + }] : []), { + id: 'workflows', + label: 'Dynamic workflows', + value: settingsData?.disableWorkflows === true ? false : settingsData?.enableWorkflows ?? true, + type: 'boolean' as const, + onChange(workflowsEnabled: boolean) { + // Writes `enableWorkflows` and clears `disableWorkflows`: the two keys + // are separate so managed settings can hard-disable the feature without + // stomping the user's own toggle underneath. + updateSettingsForSource('userSettings', { + enableWorkflows: workflowsEnabled ? undefined : false, + disableWorkflows: undefined + }); + setSettingsData(getInitialSettings()); + logEvent('tengu_workflows_setting_changed', { + enabled: workflowsEnabled + }); + } + }, { + id: 'workflowKeywordTriggerEnabled', + label: 'Ultracode keyword trigger', + value: settingsData?.workflowKeywordTriggerEnabled ?? true, + type: 'boolean' as const, + onChange(keywordEnabled: boolean) { + updateSettingsForSource('userSettings', { + workflowKeywordTriggerEnabled: keywordEnabled ? undefined : false + }); + setSettingsData(getInitialSettings()); + logEvent('tengu_workflow_keyword_trigger_setting_changed', { + enabled: keywordEnabled + }); + } + }, ...(getWorkflowSizeGuidelineFromSettings() === undefined ? [{ + id: 'workflowSizeGuideline', + label: 'Dynamic workflow size', + value: globalConfig.workflowSizeGuideline ?? 'medium (default)', + options: [...WORKFLOW_SIZE_GUIDELINES], + type: 'enum' as const, + onChange(size: string) { + const guideline = ((WORKFLOW_SIZE_GUIDELINES as readonly string[]).includes(size) ? size : 'unrestricted') as (typeof WORKFLOW_SIZE_GUIDELINES)[number]; + saveGlobalConfig(currentWf => currentWf.workflowSizeGuideline === guideline ? currentWf : { + ...currentWf, + workflowSizeGuideline: guideline + }); + setGlobalConfig({ + ...getGlobalConfig(), + workflowSizeGuideline: guideline + }); + logEvent('tengu_workflow_size_guideline_changed', { + size: guideline as AnalyticsMetadata_I_VERIFIED_THIS_IS_NOT_CODE_OR_FILEPATHS + }); + } }] : []), { id: 'verbose', label: 'Verbose output', diff --git a/src/components/messages/nullRenderingAttachments.ts b/src/components/messages/nullRenderingAttachments.ts index 933fea06..f83804f5 100644 --- a/src/components/messages/nullRenderingAttachments.ts +++ b/src/components/messages/nullRenderingAttachments.ts @@ -35,6 +35,10 @@ const NULL_RENDERING_TYPES = [ 'companion_intro', 'token_usage', 'ultrathink_effort', + 'workflow_keyword_request', + 'ultra_effort_enter', + 'ultra_effort_exit', + 'workflow_size_guideline_change', 'max_turns_reached', 'task_reminder', 'auto_mode', diff --git a/src/components/permissions/PermissionRequest.tsx b/src/components/permissions/PermissionRequest.tsx index 2def623e..74f2d6a0 100644 --- a/src/components/permissions/PermissionRequest.tsx +++ b/src/components/permissions/PermissionRequest.tsx @@ -35,8 +35,8 @@ import { WebFetchPermissionRequest } from './WebFetchPermissionRequest/WebFetchP /* eslint-disable @typescript-eslint/no-require-imports */ const ReviewArtifactTool = feature('REVIEW_ARTIFACT') ? (require('../../tools/ReviewArtifactTool/ReviewArtifactTool.js') as typeof import('../../tools/ReviewArtifactTool/ReviewArtifactTool.js')).ReviewArtifactTool : null; const ReviewArtifactPermissionRequest = feature('REVIEW_ARTIFACT') ? (require('./ReviewArtifactPermissionRequest/ReviewArtifactPermissionRequest.js') as typeof import('./ReviewArtifactPermissionRequest/ReviewArtifactPermissionRequest.js')).ReviewArtifactPermissionRequest : null; -const WorkflowTool = feature('WORKFLOW_SCRIPTS') ? (require('../../tools/WorkflowTool/WorkflowTool.js') as typeof import('../../tools/WorkflowTool/WorkflowTool.js')).WorkflowTool : null; -const WorkflowPermissionRequest = feature('WORKFLOW_SCRIPTS') ? (require('../../tools/WorkflowTool/WorkflowPermissionRequest.js') as typeof import('../../tools/WorkflowTool/WorkflowPermissionRequest.js')).WorkflowPermissionRequest : null; +const WorkflowTool = (require('../../tools/WorkflowTool/WorkflowTool.js') as typeof import('../../tools/WorkflowTool/WorkflowTool.js')).WorkflowTool; +const WorkflowPermissionRequest = (require('../../tools/WorkflowTool/WorkflowPermissionRequest.js') as typeof import('../../tools/WorkflowTool/WorkflowPermissionRequest.js')).WorkflowPermissionRequest; const MonitorTool = feature('MONITOR_TOOL') ? (require('../../tools/MonitorTool/MonitorTool.js') as typeof import('../../tools/MonitorTool/MonitorTool.js')).MonitorTool : null; const MonitorPermissionRequest = feature('MONITOR_TOOL') ? (require('./MonitorPermissionRequest/MonitorPermissionRequest.js') as typeof import('./MonitorPermissionRequest/MonitorPermissionRequest.js')).MonitorPermissionRequest : null; import type { ContentBlockParam } from '@anthropic-ai/sdk/resources/messages.mjs'; @@ -69,7 +69,7 @@ function permissionComponentForTool(tool: Tool): React.ComponentType = { - get(_t, prop) { - if (prop === '__esModule') return true - if (prop === 'default') return new Proxy(__target, __handler) - if (prop === Symbol.toPrimitive) return () => undefined - if (prop === Symbol.iterator) return function* () {} - if (prop === Symbol.asyncIterator) return async function* () {} - if (prop === 'then') return undefined - return new Proxy(__target, __handler) - }, - apply() { - return new Proxy(__target, __handler) - }, - construct() { - return new Proxy(__target, __handler) - }, -} -const stub: any = new Proxy(__target, __handler) -export default stub -export const __stubMissing = true -// 兼容常见的命名导出 —— 没列在这里的也会通过 default Proxy 兜底 -export const createCachedMCState = stub -export const isCachedMicrocompactEnabled = stub -export const isModelSupportedForCacheEditing = stub -export const getCachedMCConfig = stub -export const markToolsSentToAPI = stub -export const resetCachedMCState = stub -export const checkProtectedNamespace = stub -export const getCoordinatorUserContext = stub diff --git a/src/components/tasks/WorkflowDetailDialog.tsx b/src/components/tasks/WorkflowDetailDialog.tsx new file mode 100644 index 00000000..2ffabc9d --- /dev/null +++ b/src/components/tasks/WorkflowDetailDialog.tsx @@ -0,0 +1,572 @@ +import figures from 'figures' +import React, { useCallback, useMemo, useState } from 'react' +import { useElapsedTime } from '../../hooks/useElapsedTime.js' +import type { KeyboardEvent } from '../../ink/events/keyboard-event.js' +import { Box, Text } from '../../ink.js' +import type { LocalWorkflowTaskState } from '../../tasks/LocalWorkflowTask/LocalWorkflowTask.js' +import type { DeepImmutable } from '../../types/utils.js' +import { getLargeWorkflowWarning } from '../../utils/workflows/enabled.js' +import type { + WorkflowAgentEvent, + WorkflowProgressEvent, +} from '../../utils/workflows/types.js' +import { Byline } from '../design-system/Byline.js' +import { KeyboardShortcutHint } from '../design-system/KeyboardShortcutHint.js' +import { getTaskStatusColor, getTaskStatusIcon } from './taskStatusUtils.js' + +type Props = { + workflow: DeepImmutable + onDone: () => void + onBack?: () => void + onKill?: () => void + onPause?: () => void + onSkipAgent?: (agentKey: string) => void + onRetryAgent?: (agentKey: string) => void + onSave?: () => void + /** `null` until the user presses Tab; then the destination they picked. */ + saveScope?: 'user' | 'project' | null + onToggleSaveScope?: () => void + /** Suppresses the large-run advisory — ultracode already opts in to scale. */ + ultracode?: boolean +} + +type AgentFilter = 'all' | 'running' | 'done' | 'error' +const AGENT_FILTERS: AgentFilter[] = ['all', 'running', 'done', 'error'] + +/** Rows visible in the agent list before it scrolls. */ +const VISIBLE_AGENTS = 12 +/** Log lines shown at the run level. */ +const VISIBLE_LOGS = 6 + +type PhaseGroup = { + index: number + title: string + agents: WorkflowAgentEvent[] +} + +/** + * Live progress view for one dynamic workflow run. + * + * Three levels: the run (phases with totals), a phase (its agents), and one + * agent (its prompt, model, and result). The whole thing is derived from the + * task's `workflowProgress` rows — the runtime is the single source of truth + * and this component holds no state beyond where the cursor is. + */ +export function WorkflowDetailDialog({ + workflow, + onDone, + onBack, + onKill, + onPause, + onSkipAgent, + onRetryAgent, + onSave, + saveScope, + onToggleSaveScope, + ultracode, +}: Props): React.ReactNode { + const elapsed = useElapsedTime( + workflow.startTime, + workflow.status === 'running', + 1000, + 0, + ) + const [selectedPhase, setSelectedPhase] = useState(null) + const [selectedAgentIndex, setSelectedAgentIndex] = useState( + null, + ) + const [cursor, setCursor] = useState(0) + const [filter, setFilter] = useState('all') + + const { phases, orphanAgents, logs } = useMemo( + () => groupProgress(workflow.workflowProgress as WorkflowProgressEvent[]), + [workflow.workflowProgress], + ) + + const activePhase = + selectedPhase === null + ? null + : (phases.find(phase => phase.index === selectedPhase) ?? + (selectedPhase === 0 + ? { index: 0, title: 'Ungrouped', agents: orphanAgents } + : null)) + + const visibleAgents = useMemo( + () => (activePhase ? applyFilter(activePhase.agents, filter) : []), + [activePhase, filter], + ) + + const selectedAgent = + selectedAgentIndex === null + ? null + : (visibleAgents.find(agent => agent.index === selectedAgentIndex) ?? null) + + const rowCount = + activePhase === null + ? phases.length + (orphanAgents.length > 0 ? 1 : 0) + : visibleAgents.length + + const handleKeyDown = useCallback( + (event: KeyboardEvent) => { + const key = event.key + if (key === 'up' || key === 'k') { + event.preventDefault() + setCursor(prev => Math.max(0, prev - 1)) + return + } + if (key === 'down' || key === 'j') { + event.preventDefault() + setCursor(prev => Math.min(Math.max(0, rowCount - 1), prev + 1)) + return + } + if (key === 'return' || key === 'right') { + event.preventDefault() + if (selectedAgent) return + if (activePhase === null) { + const rows = + orphanAgents.length > 0 + ? [...phases, { index: 0, title: 'Ungrouped', agents: orphanAgents }] + : phases + const target = rows[cursor] + if (target) { + setSelectedPhase(target.index) + setCursor(0) + } + return + } + const target = visibleAgents[cursor] + if (target) setSelectedAgentIndex(target.index) + return + } + if (key === 'escape' || key === 'left') { + event.preventDefault() + if (selectedAgentIndex !== null) { + setSelectedAgentIndex(null) + return + } + if (selectedPhase !== null) { + setSelectedPhase(null) + setCursor(0) + return + } + if (onBack) onBack() + else onDone() + return + } + if (key === 'f' && activePhase !== null) { + event.preventDefault() + setFilter(prev => { + const next = AGENT_FILTERS[(AGENT_FILTERS.indexOf(prev) + 1) % AGENT_FILTERS.length] + return next ?? 'all' + }) + setCursor(0) + return + } + if (key === 'x') { + event.preventDefault() + if (selectedAgent && onSkipAgent) { + onSkipAgent(agentKey(workflow.workflowRunId, selectedAgent.index)) + return + } + if (workflow.status === 'running' && onKill) onKill() + return + } + if (key === 'r' && selectedAgent && onRetryAgent) { + event.preventDefault() + onRetryAgent(agentKey(workflow.workflowRunId, selectedAgent.index)) + return + } + if (key === 'p' && workflow.status === 'running' && onPause) { + event.preventDefault() + onPause() + return + } + if (key === 'tab' && onToggleSaveScope) { + event.preventDefault() + onToggleSaveScope() + return + } + if (key === 's' && onSave) { + event.preventDefault() + onSave() + return + } + if (key === ' ') { + event.preventDefault() + onDone() + } + }, + [ + activePhase, + cursor, + onBack, + onDone, + onKill, + onPause, + onRetryAgent, + onSave, + onSkipAgent, + onToggleSaveScope, + orphanAgents, + phases, + rowCount, + selectedAgent, + selectedAgentIndex, + selectedPhase, + visibleAgents, + workflow.status, + workflow.workflowRunId, + ], + ) + + const statusColor = getTaskStatusColor(workflow.status) + const statusIcon = getTaskStatusIcon(workflow.status) + const started = phases + .flatMap(phase => phase.agents) + .concat(orphanAgents) + .filter(agent => agent.state !== 'start').length + const largeWarning = + workflow.status === 'running' + ? getLargeWorkflowWarning({ + scheduledAgents: workflow.agentCount, + startedAgents: started, + totalTokens: workflow.totalTokens, + ultracodeActive: ultracode === true, + }) + : undefined + + return ( + + + + {workflow.workflowName ?? 'Dynamic workflow'} + {` ${workflow.workflowRunId}`} + + {workflow.summary ? {workflow.summary} : null} + + {`${statusIcon} ${workflow.status}`} + + {` · ${workflow.agentCount} agents · ${formatTokens(workflow.totalTokens)} tok · ${workflow.totalToolCalls} tools · ${formatElapsed(elapsed)}`} + + + {workflow.error ? ( + + {workflow.error} + + ) : null} + {largeWarning ? ( + + + {`⚠ Large workflow — ${describeLargeWarning(largeWarning)}. Press x to stop.`} + + + ) : null} + + + {selectedAgent ? ( + + ) : activePhase ? ( + + ) : ( + + )} + + + {!activePhase && logs.length > 0 ? ( + + Log + {logs.slice(-VISIBLE_LOGS).map((line, index) => ( + + {` ${line}`} + + ))} + + ) : null} + + + + + + + {activePhase ? : null} + {workflow.status === 'running' && onPause ? ( + + ) : null} + {selectedAgent && onSkipAgent ? ( + + ) : workflow.status === 'running' && onKill ? ( + + ) : null} + {selectedAgent && onRetryAgent ? ( + + ) : null} + {onSave ? ( + + ) : null} + {onToggleSaveScope ? ( + + ) : null} + + + ) +} + +function PhaseList({ + phases, + orphanAgents, + cursor, +}: { + phases: PhaseGroup[] + orphanAgents: WorkflowAgentEvent[] + cursor: number +}): React.ReactNode { + const rows = + orphanAgents.length > 0 + ? [...phases, { index: 0, title: 'Ungrouped', agents: orphanAgents }] + : phases + if (rows.length === 0) { + return Waiting for the first agent… + } + return ( + + {rows.map((phase, rowIndex) => { + const totals = summarize(phase.agents) + const selected = rowIndex === cursor + return ( + + + {`${selected ? figures.pointer : ' '} ${phase.title}`} + + + {` ${totals.done}/${phase.agents.length} done`} + {totals.errors > 0 ? ` ${totals.errors} failed` : ''} + {` · ${formatTokens(totals.tokens)} tok`} + + + ) + })} + + ) +} + +function AgentList({ + phase, + agents, + cursor, + filter, +}: { + phase: PhaseGroup + agents: WorkflowAgentEvent[] + cursor: number + filter: AgentFilter +}): React.ReactNode { + if (agents.length === 0) { + return {`No ${filter} agents in ${phase.title}.`} + } + const start = Math.max(0, Math.min(cursor - VISIBLE_AGENTS + 1, agents.length - VISIBLE_AGENTS)) + const window = agents.slice(Math.max(0, start), Math.max(0, start) + VISIBLE_AGENTS) + return ( + + {`${phase.title} — ${agents.length} agent(s), filter: ${filter}`} + {window.map(agent => { + const selected = agent.index === agents[cursor]?.index + return ( + + + {`${selected ? figures.pointer : ' '} ${agentIcon(agent)} ${agent.label}`} + + + {agent.cached ? ' cached' : ''} + {agent.tokens ? ` ${formatTokens(agent.tokens)} tok` : ''} + {agent.toolCalls ? ` ${agent.toolCalls} tools` : ''} + {agent.error ? ` ${agent.error}` : ''} + + + ) + })} + + ) +} + +function AgentDetail({ agent }: { agent: WorkflowAgentEvent }): React.ReactNode { + return ( + + {`${agentIcon(agent)} ${agent.label}`} + + {[ + agent.model ? `model ${agent.model}` : null, + agent.agentType ? `agent ${agent.agentType}` : null, + agent.isolation ? `isolation ${agent.isolation}` : null, + agent.agentId ? `id ${agent.agentId}` : null, + ] + .filter(Boolean) + .join(' · ')} + + + {`${formatTokens(agent.tokens ?? 0)} tok · ${agent.toolCalls ?? 0} tools`} + {agent.durationMs ? ` · ${Math.round(agent.durationMs / 1000)}s` : ''} + + {agent.promptPreview ? ( + + Prompt + {agent.promptPreview} + + ) : null} + {agent.resultPreview ? ( + + Result + {agent.resultPreview} + + ) : null} + {agent.error ? ( + + {agent.error} + + ) : null} + + ) +} + +/** + * Fold the flat progress stream into phases. + * + * Agents that ran before any `phase()` call have no `phaseIndex`; they are + * collected separately rather than dropped, so a script that never calls + * `phase()` still shows all its work. + */ +function groupProgress(events: readonly WorkflowProgressEvent[]): { + phases: PhaseGroup[] + orphanAgents: WorkflowAgentEvent[] + logs: string[] +} { + const phaseByIndex = new Map() + const orphanAgents: WorkflowAgentEvent[] = [] + const logs: string[] = [] + + for (const event of events) { + if (event.type === 'workflow_phase') { + if (!phaseByIndex.has(event.index)) { + phaseByIndex.set(event.index, { + index: event.index, + title: event.title, + agents: [], + }) + } + continue + } + if (event.type === 'workflow_log') { + logs.push(event.message) + continue + } + if (event.phaseIndex === undefined) { + orphanAgents.push(event) + continue + } + const phase = phaseByIndex.get(event.phaseIndex) ?? { + index: event.phaseIndex, + title: event.phaseTitle ?? `Phase ${event.phaseIndex}`, + agents: [], + } + phase.agents.push(event) + phaseByIndex.set(event.phaseIndex, phase) + } + + return { + phases: [...phaseByIndex.values()].sort((a, b) => a.index - b.index), + orphanAgents, + logs, + } +} + +function applyFilter( + agents: WorkflowAgentEvent[], + filter: AgentFilter, +): WorkflowAgentEvent[] { + switch (filter) { + case 'running': + return agents.filter( + agent => agent.state === 'start' || agent.state === 'progress', + ) + case 'done': + return agents.filter(agent => agent.state === 'done') + case 'error': + return agents.filter(agent => agent.state === 'error') + default: + return agents + } +} + +function summarize(agents: WorkflowAgentEvent[]): { + done: number + errors: number + tokens: number +} { + let done = 0 + let errors = 0 + let tokens = 0 + for (const agent of agents) { + if (agent.state === 'done') done++ + if (agent.state === 'error') errors++ + tokens += agent.tokens ?? 0 + } + return { done, errors, tokens } +} + +function agentIcon(agent: WorkflowAgentEvent): string { + switch (agent.state) { + case 'done': + return figures.tick + case 'error': + return figures.cross + case 'progress': + return figures.play + default: + return figures.bullet + } +} + +function describeLargeWarning( + warning: NonNullable>, +): string { + const parts: string[] = [] + if (warning.axis !== 'tokens') { + parts.push( + `${warning.scheduledAgents} agents scheduled (over ${warning.agentCap}${warning.capFromGuideline ? ', from your size guideline' : ''})`, + ) + } + if (warning.axis !== 'agents') { + parts.push( + `projected ${formatTokens(warning.projectedTokens)} tokens (over ${formatTokens(warning.tokenCap)})`, + ) + } + return parts.join(' and ') +} + +/** Agent controller key, mirroring the harness's `${runId}-${index}`. */ +function agentKey(runId: string, index: number): string { + return `${runId}-${index}` +} + +function formatTokens(tokens: number): string { + if (tokens >= 1_000_000) return `${(tokens / 1_000_000).toFixed(1)}M` + if (tokens >= 1_000) return `${(tokens / 1_000).toFixed(1)}k` + return String(tokens) +} + +function formatElapsed(seconds: number): string { + if (seconds < 60) return `${seconds}s` + const minutes = Math.floor(seconds / 60) + return `${minutes}m ${seconds % 60}s` +} diff --git a/src/constants/tools.ts b/src/constants/tools.ts index 67dd23fd..93a4be7a 100644 --- a/src/constants/tools.ts +++ b/src/constants/tools.ts @@ -42,7 +42,7 @@ export const ALL_AGENT_DISALLOWED_TOOLS = new Set([ ASK_USER_QUESTION_TOOL_NAME, TASK_STOP_TOOL_NAME, // Prevent recursive workflow execution inside subagents. - ...(feature('WORKFLOW_SCRIPTS') ? [WORKFLOW_TOOL_NAME] : []), + WORKFLOW_TOOL_NAME, ]) export const CUSTOM_AGENT_DISALLOWED_TOOLS = new Set([ diff --git a/src/keybindings/defaultBindings.ts b/src/keybindings/defaultBindings.ts index 8629809d..75ba16fb 100644 --- a/src/keybindings/defaultBindings.ts +++ b/src/keybindings/defaultBindings.ts @@ -70,6 +70,7 @@ export const DEFAULT_BINDINGS: KeybindingBlock[] = [ 'meta+p': 'chat:modelPicker', 'meta+o': 'chat:fastMode', 'meta+t': 'chat:thinkingToggle', + 'meta+w': 'chat:workflowKeywordToggle', enter: 'chat:submit', up: 'history:previous', down: 'history:next', diff --git a/src/keybindings/schema.ts b/src/keybindings/schema.ts index 3e61d63a..981a0f6a 100644 --- a/src/keybindings/schema.ts +++ b/src/keybindings/schema.ts @@ -84,6 +84,7 @@ export const KEYBINDING_ACTIONS = [ 'chat:modelPicker', 'chat:fastMode', 'chat:thinkingToggle', + 'chat:workflowKeywordToggle', 'chat:submit', 'chat:newline', 'chat:undo', diff --git a/src/main.tsx b/src/main.tsx index b7ab7ed5..33f98847 100644 --- a/src/main.tsx +++ b/src/main.tsx @@ -52,6 +52,7 @@ import { getSubscriptionType, isClaudeAISubscriber, prefetchAwsCredentialsAndBed import { checkHasTrustDialogAccepted, getGlobalConfig, getRemoteControlAtStartup, isAutoUpdaterDisabled, saveGlobalConfig } from './utils/config.js'; import { seedEarlyInput, stopCapturingEarlyInput } from './utils/earlyInput.js'; import { getInitialEffortSetting, parseEffortValue } from './utils/effort.js'; +import { ULTRACODE_EFFORT_ARG, ULTRACODE_EFFORT_LEVEL } from './utils/workflows/ultracode.js'; import { getInitialFastModeSetting, isFastModeEnabled, prefetchFastModeStatus, resolveFastModeStatusFromCache } from './utils/fastMode.js'; import { applyConfigEnvironmentVariables } from './utils/managedEnv.js'; import { createSystemMessage, createUserMessage } from './utils/messages.js'; @@ -991,10 +992,16 @@ async function run(): Promise { return Number.isFinite(n) ? n : undefined; }).hideHelp()).option('--from-pr [value]', 'Resume a session linked to a PR by PR number/URL, or open interactive picker with optional search term', value => value || true).option('--no-session-persistence', 'Disable session persistence - sessions will not be saved to disk and cannot be resumed (only works with --print)').addOption(new Option('--resume-session-at ', 'When resuming, only messages up to and including the assistant message with (use with --resume in print mode)').argParser(String).hideHelp()).addOption(new Option('--rewind-files ', 'Restore files to state at the specified user message and exit (requires --resume)').hideHelp()) // @[MODEL LAUNCH]: Update the example model ID in the --model help text. - .option('--model ', `Model for the current session. Provide an alias for the latest model (e.g. 'sonnet' or 'opus') or a model's full name (e.g. 'claude-sonnet-4-6').`).addOption(new Option('--effort ', `Effort level for the current session (low, medium, high, xhigh, max, or an integer)`).argParser((rawValue: string) => { + .option('--model ', `Model for the current session. Provide an alias for the latest model (e.g. 'sonnet' or 'opus') or a model's full name (e.g. 'claude-sonnet-4-6').`).addOption(new Option('--effort ', `Effort level for the current session (low, medium, high, xhigh, max, ultracode, or an integer)`).argParser((rawValue: string) => { + // `ultracode` is passed through as-is: it is not an effort level but a + // request for xhigh plus standing workflow orchestration, resolved later + // where the session's model and workflow availability are known. + if (rawValue.toLowerCase() === ULTRACODE_EFFORT_ARG) { + return ULTRACODE_EFFORT_ARG; + } const value = parseEffortValue(rawValue); if (value === undefined) { - throw new InvalidArgumentError('It must be one of: low, medium, high, xhigh, max, or an integer'); + throw new InvalidArgumentError('It must be one of: low, medium, high, xhigh, max, ultracode, or an integer'); } return value; })).option('--agent ', `Agent for the current session. Overrides the 'agent' setting.`).option('--betas ', 'Beta headers to include in API requests (API key users only)').option('--fallback-model ', 'Enable automatic fallback to specified model when default model is overloaded (only works with --print)').addOption(new Option('--workload ', 'Workload tag for billing-header attribution (cc_workload). Process-scoped; set by SDK daemon callers that spawn subprocesses for cron work. (only works with --print)').hideHelp()).option('--settings ', 'Path to a settings JSON file or a JSON string to load additional settings from').option('--add-dir ', 'Additional directories to allow tool access to').option('--ide', 'Automatically connect to IDE on startup if exactly one valid IDE is available', () => true).option('--strict-mcp-config', 'Only use MCP servers from --mcp-config, ignoring all other MCP configurations', () => true).option('--session-id ', 'Use a specific session ID for the conversation (must be a valid UUID)').option('-n, --name ', 'Set a display name for this session (shown in /resume and terminal title)').option('--agents ', 'JSON object defining custom agents (e.g. \'{"reviewer": {"description": "Reviews code", "prompt": "You are a code reviewer"}}\')').option('--setting-sources ', 'Comma-separated list of setting sources to load (user, project, local).') @@ -2132,7 +2139,11 @@ async function run(): Promise { // An explicit CLI effort remains authoritative. Otherwise a selected // main-thread agent may provide its own effort before falling back to the // session setting. - const effectiveEffort = parseEffortValue(options.effort) ?? mainThreadAgentDefinition?.effort ?? getInitialEffortSetting(); + // `--effort ultracode` is not a sixth level: it resolves to xhigh and + // raises a separate session flag, so every model-capability check keeps + // reasoning about the five real levels. + const ultracodeRequested = String(options.effort ?? '').toLowerCase() === ULTRACODE_EFFORT_ARG; + const effectiveEffort = ultracodeRequested ? ULTRACODE_EFFORT_LEVEL : parseEffortValue(options.effort) ?? mainThreadAgentDefinition?.effort ?? getInitialEffortSetting(); // Compute resolved model for hooks (use user-specified model at launch) setInitialMainLoopModel(getUserSpecifiedModelSetting() || null); @@ -2655,6 +2666,7 @@ async function run(): Promise { }, toolPermissionContext, effortValue: effectiveEffort, + ultracode: ultracodeRequested, ...(isFastModeEnabled() && { fastMode: getInitialFastModeSetting(effectiveModel ?? null) }), @@ -3083,6 +3095,7 @@ async function run(): Promise { }) } : null, effortValue: effectiveEffort, + ultracode: ultracodeRequested, activeOverlays: new Set(), fastMode: getInitialFastModeSetting(resolvedInitialModel), ...(isAdvisorEnabled() && advisorModel && { diff --git a/src/server/__tests__/workflows-api.test.ts b/src/server/__tests__/workflows-api.test.ts new file mode 100644 index 00000000..c438952e --- /dev/null +++ b/src/server/__tests__/workflows-api.test.ts @@ -0,0 +1,250 @@ +import { afterEach, beforeEach, describe, expect, it } from 'bun:test' +import * as fs from 'node:fs/promises' +import * as os from 'node:os' +import * as path from 'node:path' +import { resetSettingsCache } from '../../utils/settings/settingsCache.js' +import { handleWorkflowsApi } from '../api/workflows.js' + +let tmpHome: string +let projectDir: string +let originalConfigDir: string | undefined + +function request( + urlStr: string, + init?: { method?: string; body?: unknown }, +): { req: Request; url: URL; segments: string[] } { + const url = new URL(urlStr, 'http://localhost:3456') + const req = new Request(url.toString(), { + method: init?.method ?? 'GET', + ...(init?.body !== undefined + ? { + body: JSON.stringify(init.body), + headers: { 'content-type': 'application/json' }, + } + : {}), + }) + return { req, url, segments: url.pathname.split('/').filter(Boolean) } +} + +function call( + urlStr: string, + init?: { method?: string; body?: unknown }, +): Promise { + const { req, url, segments } = request(urlStr, init) + return handleWorkflowsApi(req, url, segments) +} + +const VALID_SCRIPT = [ + "export const meta = { name: 'my-audit', description: 'Audit the routes', phases: [{ title: 'Scan' }] }", + "const out = await agent('scan')", + 'return out', +].join('\n') + +describe('Workflows API', () => { + beforeEach(async () => { + tmpHome = await fs.mkdtemp(path.join(os.tmpdir(), 'wf-api-')) + projectDir = path.join(tmpHome, 'project') + await fs.mkdir(path.join(projectDir, '.git'), { recursive: true }) + originalConfigDir = process.env.CLAUDE_CONFIG_DIR + process.env.CLAUDE_CONFIG_DIR = path.join(tmpHome, 'claude') + resetSettingsCache() + }) + + afterEach(async () => { + if (originalConfigDir === undefined) delete process.env.CLAUDE_CONFIG_DIR + else process.env.CLAUDE_CONFIG_DIR = originalConfigDir + resetSettingsCache() + await fs.rm(tmpHome, { recursive: true, force: true }) + }) + + it('lists the bundled workflow', async () => { + const response = await call(`/api/workflows?cwd=${encodeURIComponent(projectDir)}`) + expect(response.status).toBe(200) + const body = (await response.json()) as { + workflows: Array<{ name: string; source: string; phases?: unknown[] }> + } + const deepResearch = body.workflows.find(w => w.name === 'deep-research') + expect(deepResearch?.source).toBe('built-in') + expect(deepResearch?.phases?.length).toBeGreaterThan(0) + // The list stays small: no script bodies. + expect(body.workflows.every(w => !('script' in w) || w.script === undefined)).toBe(true) + }) + + it('returns the script on the detail endpoint', async () => { + const response = await call( + `/api/workflows/deep-research?cwd=${encodeURIComponent(projectDir)}`, + ) + const body = (await response.json()) as { script?: string } + expect(body.script).toContain('export const meta') + }) + + it('validates a script without running it', async () => { + const ok = await ( + await call('/api/workflows/validate', { + method: 'POST', + body: { script: VALID_SCRIPT }, + }) + ).json() + expect(ok).toMatchObject({ ok: true, name: 'my-audit' }) + + const nondeterministic = await ( + await call('/api/workflows/validate', { + method: 'POST', + body: { + script: + "export const meta = { name: 'x', description: 'y' }\nconst t = Date.now()\nreturn t\n", + }, + }) + ).json() + expect(nondeterministic).toMatchObject({ ok: true }) + expect((nondeterministic as { warnings: string[] }).warnings[0]).toContain( + 'Date.now()', + ) + + const bad = await ( + await call('/api/workflows/validate', { + method: 'POST', + body: { script: 'const x = 1\n' }, + }) + ).json() + expect(bad).toMatchObject({ ok: false }) + expect((bad as { error: string }).error).toContain('FIRST statement') + }) + + it('saves a workflow and then lists and deletes it', async () => { + const saved = await ( + await call('/api/workflows/save', { + method: 'POST', + body: { script: VALID_SCRIPT, scope: 'project', cwd: projectDir }, + }) + ).json() + expect(saved).toMatchObject({ ok: true, name: 'my-audit' }) + expect((saved as { filePath: string }).filePath).toBe( + path.join(projectDir, '.claude', 'workflows', 'my-audit.js'), + ) + + const listed = (await ( + await call(`/api/workflows?cwd=${encodeURIComponent(projectDir)}`) + ).json()) as { workflows: Array<{ name: string; source: string }> } + expect( + listed.workflows.find(w => w.name === 'my-audit')?.source, + ).toBe('projectSettings') + + const deleted = await call( + `/api/workflows/my-audit?scope=project&cwd=${encodeURIComponent(projectDir)}`, + { method: 'DELETE' }, + ) + expect(deleted.status).toBe(200) + const after = (await ( + await call(`/api/workflows?cwd=${encodeURIComponent(projectDir)}`) + ).json()) as { workflows: Array<{ name: string }> } + expect(after.workflows.find(w => w.name === 'my-audit')).toBeUndefined() + }) + + it('rejects saving a script that does not compile', async () => { + const response = await call('/api/workflows/save', { + method: 'POST', + body: { + script: + "export const meta = { name: 'broken', description: 'x' }\nconst a: string = 1\n", + scope: 'user', + }, + }) + expect(response.status).toBe(400) + }) + + it('reconstructs a past run from its script and journal', async () => { + const sessionId = '11111111-2222-3333-4444-555555555555' + const sessionDir = path.join( + tmpHome, + 'claude', + 'projects', + '-tmp-project', + sessionId, + ) + await fs.mkdir(path.join(sessionDir, 'workflows'), { recursive: true }) + await fs.writeFile( + path.join(sessionDir, 'workflows', 'my-audit.wf_abc12345-def.js'), + VALID_SCRIPT, + 'utf8', + ) + const journalDir = path.join( + sessionDir, + 'subagents', + 'workflows', + 'wf_abc12345-def', + ) + await fs.mkdir(journalDir, { recursive: true }) + await fs.writeFile( + path.join(journalDir, 'journal.jsonl'), + [ + JSON.stringify({ type: 'started', key: '|scan|null', agentId: 'a1' }), + JSON.stringify({ + type: 'result', + key: '|scan|null', + agentId: 'a1', + result: 'found nothing', + }), + 'not json at all', + ].join('\n'), + 'utf8', + ) + + const runs = (await (await call('/api/workflows/runs')).json()) as { + runs: Array<{ runId: string; workflowName: string; completedAgents: number }> + } + expect(runs.runs).toHaveLength(1) + expect(runs.runs[0]).toMatchObject({ + runId: 'wf_abc12345-def', + workflowName: 'my-audit', + completedAgents: 1, + }) + + const detail = (await ( + await call(`/api/workflows/runs/${sessionId}/wf_abc12345-def`) + ).json()) as { + script: string + description?: string + agents: Array<{ agentId: string; result: unknown; key: string }> + } + expect(detail.script).toBe(VALID_SCRIPT) + expect(detail.description).toBe('Audit the routes') + expect(detail.agents).toHaveLength(1) + expect(detail.agents[0]?.result).toBe('found nothing') + // The raw journal key chains every prior prompt — the API must not leak it. + expect(detail.agents[0]?.key).not.toContain('scan') + }) + + it('404s an unknown run', async () => { + const response = await call('/api/workflows/runs/nope/wf_000000-aaa') + expect(response.status).toBe(404) + }) + + it('refuses to write through a symlinked target', async () => { + const workflowsDir = path.join(tmpHome, 'claude', 'workflows') + await fs.mkdir(workflowsDir, { recursive: true }) + const outside = path.join(tmpHome, 'outside.js') + await fs.writeFile(outside, '// pre-existing\n', 'utf8') + await fs.symlink(outside, path.join(workflowsDir, 'my-audit.js')) + + const response = await call('/api/workflows/save', { + method: 'POST', + body: { script: VALID_SCRIPT, scope: 'user' }, + }) + expect(response.status).toBe(400) + expect(await fs.readFile(outside, 'utf8')).toBe('// pre-existing\n') + }) + + it('403s every route when workflows are disabled', async () => { + await fs.mkdir(path.join(tmpHome, 'claude'), { recursive: true }) + await fs.writeFile( + path.join(tmpHome, 'claude', 'settings.json'), + JSON.stringify({ disableWorkflows: true }), + 'utf8', + ) + resetSettingsCache() + + expect((await call('/api/workflows')).status).toBe(403) + expect((await call('/api/workflows/runs')).status).toBe(403) + }) +}) diff --git a/src/server/api/sessions.ts b/src/server/api/sessions.ts index 33e9d149..77643725 100644 --- a/src/server/api/sessions.ts +++ b/src/server/api/sessions.ts @@ -51,7 +51,7 @@ import { import { registerChangedFileAccessRoot, registerFilesystemAccessRoot } from '../services/filesystemAccessRoots.js' import { findGitRoot } from '../../utils/git.js' import { traceCaptureService, trimTraceCallPreviews } from '../services/traceCaptureService.js' -import { getSubagentRunByTool } from '../services/subagentRunService.js' +import { getSubagentRunByAgentId, getSubagentRunByTool } from '../services/subagentRunService.js' import { isValidPermissionMode } from '../services/settingsService.js' import { handleWorkspaceSearchRoute } from './workspaceSearch.js' import { localIndexCoordinator } from '../services/localIndex/coordinator.js' @@ -227,6 +227,31 @@ export async function handleSessionsApi( } if (subResource === 'subagents') { + // Workflow agents have no parent `Agent` tool call to key off, so they + // are addressed by agent id instead. Same response shape, same page. + if (segments[4] === 'by-agent' && segments[5] && segments.length === 6) { + if (req.method !== 'GET') { + return Response.json( + { error: 'METHOD_NOT_ALLOWED', message: `Method ${req.method} not allowed` }, + { status: 405 }, + ) + } + let agentId: string + try { + agentId = decodeURIComponent(segments[5]) + } catch { + return Response.json( + { error: 'NOT_FOUND', message: 'SubAgent route not found' }, + { status: 404 }, + ) + } + const byAgent = await getSubagentRunByAgentId(sessionId, agentId) + if (!byAgent) { + throw ApiError.notFound(`SubAgent run not found: ${agentId}`) + } + return Response.json(byAgent) + } + const isRunRoute = segments[4] === 'by-tool' && Boolean(segments[5]) const isRunRead = isRunRoute && segments.length === 6 && req.method === 'GET' const isRunMessage = isRunRoute && segments.length === 7 && diff --git a/src/server/api/workflows.ts b/src/server/api/workflows.ts new file mode 100644 index 00000000..5e0efcb6 --- /dev/null +++ b/src/server/api/workflows.ts @@ -0,0 +1,127 @@ +/** + * Dynamic workflows REST API + * + * GET /api/workflows — list runnable workflow definitions + * GET /api/workflows/runs — list past runs (?sessionId=&limit=) + * GET /api/workflows/runs/:sessionId/:runId — one run: script, phases, agent results + * GET /api/workflows/session-runs/:sessionId — finished runs rebuilt from disk + * POST /api/workflows/validate — parse + compile a script without running it + * POST /api/workflows/save — save a script as a /name command + * GET /api/workflows/:name — one definition, including its script + * DELETE /api/workflows/:name — delete a saved workflow (?scope=user|project) + * + * Starting a run is deliberately not here: a run belongs to a conversation + * turn, so the desktop sends the prompt (or `/name`) over the session + * WebSocket and watches `task_progress` events for live phase/agent state. + */ + +import { ApiError, errorResponse } from '../middleware/errorHandler.js' +import { + workflowService, + type WorkflowSaveScope, +} from '../services/workflowService.js' + +export async function handleWorkflowsApi( + req: Request, + url: URL, + segments: string[], +): Promise { + try { + const method = req.method + const first = segments[2] ? decodeURIComponent(segments[2]) : undefined + + if (method === 'GET' && !first) { + const cwd = url.searchParams.get('cwd') ?? undefined + const workflows = await workflowService.listDefinitions(cwd) + return Response.json({ workflows }) + } + + if (method === 'GET' && first === 'runs' && !segments[3]) { + const rawLimit = url.searchParams.get('limit') + const limit = rawLimit ? Number.parseInt(rawLimit, 10) : undefined + const runs = await workflowService.listRuns({ + sessionId: url.searchParams.get('sessionId') ?? undefined, + limit: Number.isSafeInteger(limit) && limit! > 0 ? limit : undefined, + }) + return Response.json({ runs }) + } + + // Everything this session ever ran, rebuilt from disk. The desktop calls + // this when a session is opened so a finished run is still visible. + if (method === 'GET' && first === 'session-runs' && segments[3]) { + const runs = await workflowService.reconstructSessionRuns( + decodeURIComponent(segments[3]), + ) + return Response.json({ runs }) + } + + if (method === 'GET' && first === 'runs' && segments[3] && segments[4]) { + const run = await workflowService.getRun( + decodeURIComponent(segments[3]), + decodeURIComponent(segments[4]), + ) + return Response.json(run) + } + + if (method === 'POST' && first === 'validate') { + const body = await readJson<{ script?: string }>(req) + if (typeof body.script !== 'string') { + throw ApiError.badRequest('`script` is required') + } + return Response.json(workflowService.validate(body.script)) + } + + if (method === 'POST' && first === 'save') { + const body = await readJson<{ + script?: string + scope?: string + cwd?: string + }>(req) + if (typeof body.script !== 'string') { + throw ApiError.badRequest('`script` is required') + } + const saved = await workflowService.saveDefinition({ + script: body.script, + scope: parseScope(body.scope), + cwd: body.cwd, + }) + return Response.json({ ok: true, ...saved }) + } + + if (method === 'GET' && first) { + const cwd = url.searchParams.get('cwd') ?? undefined + return Response.json(await workflowService.getDefinition(first, cwd)) + } + + if (method === 'DELETE' && first) { + await workflowService.deleteDefinition( + first, + parseScope(url.searchParams.get('scope')), + url.searchParams.get('cwd') ?? undefined, + ) + return Response.json({ ok: true }) + } + + throw new ApiError( + 405, + `Method ${method} not allowed on /api/workflows${first ? `/${first}` : ''}`, + 'METHOD_NOT_ALLOWED', + ) + } catch (error) { + return errorResponse(error) + } +} + +async function readJson(req: Request): Promise { + try { + return (await req.json()) as T + } catch { + throw ApiError.badRequest('Invalid JSON body') + } +} + +function parseScope(value: string | null | undefined): WorkflowSaveScope { + if (value === 'project') return 'project' + if (value === 'user' || value == null || value === '') return 'user' + throw ApiError.badRequest(`Invalid scope: ${value}`) +} diff --git a/src/server/router.ts b/src/server/router.ts index af80655d..a8e28674 100644 --- a/src/server/router.ts +++ b/src/server/router.ts @@ -29,6 +29,7 @@ import { handleOpenTargetsApi } from './api/open-targets.js' import { handleMemoryApi } from './api/memory.js' import { handleDesktopUiApi } from './api/desktop-ui.js' import { handleTracesApi } from './api/traces.js' +import { handleWorkflowsApi } from './api/workflows.js' export async function handleApiRequest(req: Request, url: URL): Promise { const path = url.pathname @@ -76,6 +77,9 @@ export async function handleApiRequest(req: Request, url: URL): Promise { await expect(getSubagentRunByTool(sessionId, 'tool-1')).resolves.toBeNull() }) }) + +describe('getSubagentRunByAgentId', () => { + afterEach(async () => { + if (tmpDir) { + await fs.rm(tmpDir, { recursive: true, force: true }) + tmpDir = null + } + delete process.env.CLAUDE_CONFIG_DIR + }) + + it('reads a run that has no parent Agent tool call', async () => { + await setupTmpConfigDir() + const sessionId = 'aaaaaaaa-bbbb-cccc-dddd-ffffffffffff' + const projectDir = '-tmp-workflow-agent' + const agentId = 'wfagent1' + + // A workflow agent is spawned by the workflow runtime, so the parent + // session has no Agent tool_use to key off — only the transcript exists. + await writeSessionFile(projectDir, sessionId, []) + await writeSubagentTranscriptFile(projectDir, sessionId, agentId, [ + { + type: 'user', + message: { role: 'user', content: 'Survey lib/response.js' }, + uuid: 'wf-user', + timestamp: '2026-01-01T00:00:05.000Z', + }, + { + type: 'assistant', + message: { + role: 'assistant', + content: [{ type: 'text', text: 'res.send(403) sends a JSON body' }], + usage: { input_tokens: 11, output_tokens: 7 }, + }, + uuid: 'wf-assistant', + timestamp: '2026-01-01T00:00:06.000Z', + }, + ]) + + const result = await getSubagentRunByAgentId(sessionId, agentId) + + expect(result).toMatchObject({ sessionId, agentId, source: 'subagent-jsonl' }) + expect(result?.messages.length).toBeGreaterThan(0) + // A workflow agent answers once into its script; there is no inbox. + expect(result?.canSendMessage).toBe(false) + }) + + it('returns null when no transcript exists for that agent', async () => { + await setupTmpConfigDir() + const sessionId = 'aaaaaaaa-bbbb-cccc-dddd-000000000000' + await writeSessionFile('-tmp-workflow-missing', sessionId, []) + + expect(await getSubagentRunByAgentId(sessionId, 'nope')).toBeNull() + }) +}) diff --git a/src/server/services/subagentRunService.ts b/src/server/services/subagentRunService.ts index 36bd893f..2ab11192 100644 --- a/src/server/services/subagentRunService.ts +++ b/src/server/services/subagentRunService.ts @@ -340,6 +340,55 @@ async function resolveTranscript( return { agentId: null, messages: [] } } +/** + * Read a subagent run straight from its transcript. + * + * {@link getSubagentRunByTool} starts from an `Agent` tool call in the parent + * conversation, which is how an agent the assistant dispatched is found. A + * workflow's agents are spawned by the workflow runtime instead, so no such + * tool call exists and that lookup can never resolve them — but they are + * ordinary subagents written by the same runner to the same place, so their + * `agent-.jsonl` is all that is needed. + */ +export async function getSubagentRunByAgentId( + sessionId: string, + agentId: string, +): Promise { + const normalized = normalizeAgentIdHint(agentId) + if (!normalized) return null + + const transcript = await resolveTranscript(sessionId, [normalized]) + if (!transcript.agentId) return null + + const messages = transcript.messages + const truncated = truncateSubagentMessages(messages) + const usage = usageFromTranscriptMessages(messages) + const updatedAt = latestTimestamp( + ...messages.map(message => + isRecord(message) && typeof message.timestamp === 'string' + ? message.timestamp + : undefined, + ), + ) + + return { + sessionId, + // The caller addressed this run by agent id; echoing it keeps the response + // self-describing without inventing a tool call that never happened. + toolUseId: normalized, + agentId: transcript.agentId, + status: 'completed', + ...(usage ? { usage } : {}), + messages: truncated.messages, + truncated: truncated.truncated, + ...(updatedAt ? { updatedAt } : {}), + source: 'subagent-jsonl', + // A workflow agent answers once into its script and is gone; there is no + // inbox a follow-up could reach. + canSendMessage: false, + } +} + export async function getSubagentRunByTool( sessionId: string, toolUseId: string, diff --git a/src/server/services/workflowService.ts b/src/server/services/workflowService.ts new file mode 100644 index 00000000..885daa69 --- /dev/null +++ b/src/server/services/workflowService.ts @@ -0,0 +1,500 @@ +/** + * Dynamic workflow service for the local API. + * + * Two things live here: the *definitions* a user can run (bundled, personal, + * project) and the *runs* the CLI has already executed. Runs are reconstructed + * from the artifacts the runtime writes under the session directory — the + * script, the resume journal, and the task output file — so the desktop can + * show a run's history without the CLI process still being alive. Live + * progress arrives separately over the WebSocket as `task_progress` events. + */ + +import { createHash } from 'crypto' +import * as fs from 'fs/promises' +import * as path from 'path' +import { getClaudeConfigHomeDir } from '../../utils/envUtils.js' +import { loadWorkflows } from '../../utils/workflows/discovery.js' +import { areWorkflowsEnabled } from '../../utils/workflows/enabled.js' +import { + parseWorkflowScript, + usesBannedNondeterminism, +} from '../../utils/workflows/meta.js' +import { compileWorkflowScript } from '../../utils/workflows/compile.js' +import type { + WorkflowDefinition, + WorkflowPhaseMeta, + WorkflowProgressEvent, +} from '../../utils/workflows/types.js' +import { ApiError } from '../middleware/errorHandler.js' + +export type WorkflowDefinitionSummary = { + name: string + description: string + whenToUse?: string + source: WorkflowDefinition['source'] + phases?: WorkflowPhaseMeta[] + filePath?: string + /** Present only on the detail endpoint — the list stays small. */ + script?: string +} + +export type WorkflowRunSummary = { + runId: string + sessionId: string + workflowName: string + scriptPath: string + startedAt: number + /** Number of agents with a recorded result in the journal. */ + completedAgents: number + status: 'completed' | 'failed' | 'unknown' +} + +export type WorkflowRunDetail = WorkflowRunSummary & { + script: string + description?: string + phases?: WorkflowPhaseMeta[] + agents: Array<{ key: string; agentId: string; result: unknown }> + progress?: WorkflowProgressEvent[] + logs?: string[] + result?: unknown + error?: string + totalTokens?: number + totalToolCalls?: number +} + +/** A finished run rebuilt from disk, in the shape the desktop's panel needs. */ +export type ReconstructedRun = { + runId: string + workflowName: string + startedAt: number + agents: Array<{ + agentId: string + label: string + phaseIndex: number + phaseTitle?: string + agentIndex: number + }> +} + +export type WorkflowSaveScope = 'user' | 'project' + +const RUN_SCRIPT_PATTERN = /^(.+)\.(wf_[a-z0-9-]{6,})\.js$/ + +class WorkflowService { + private configDir(): string { + return path.resolve(getClaudeConfigHomeDir()) + } + + private projectsDir(): string { + return path.join(this.configDir(), 'projects') + } + + async listDefinitions(cwd?: string): Promise { + this.assertEnabled() + const workflows = await loadWorkflows(cwd) + return workflows.map(workflow => ({ + name: workflow.name, + description: workflow.description, + whenToUse: workflow.whenToUse, + source: workflow.source, + phases: workflow.phases, + filePath: workflow.filePath, + })) + } + + async getDefinition( + name: string, + cwd?: string, + ): Promise { + this.assertEnabled() + const workflows = await loadWorkflows(cwd) + const workflow = workflows.find(entry => entry.name === name) + if (!workflow) throw ApiError.notFound(`Unknown workflow: ${name}`) + return { + name: workflow.name, + description: workflow.description, + whenToUse: workflow.whenToUse, + source: workflow.source, + phases: workflow.phases, + filePath: workflow.filePath, + script: workflow.script, + } + } + + /** + * Parse and compile a script without running it. + * + * The desktop calls this before sending a run so a malformed script is + * reported in the editor instead of coming back as a failed turn. + */ + validate(script: string): { + ok: boolean + error?: string + warnings?: string[] + name?: string + description?: string + phases?: WorkflowPhaseMeta[] + } { + this.assertEnabled() + const parsed = parseWorkflowScript(script) + if ('error' in parsed) return { ok: false, error: parsed.error } + const compiled = compileWorkflowScript(parsed.scriptBody) + if (!compiled.ok) return { ok: false, error: compiled.error } + // A script can compile and still be guaranteed to throw on its first + // Date.now()/Math.random(). Saying so here is the difference between a + // squiggle in the editor and a failed run. + const warnings = usesBannedNondeterminism(parsed.scriptBody) + ? [ + 'Date.now(), new Date() and Math.random() throw at run time — they would make a resume replay diverge.', + ] + : undefined + return { + ok: true, + ...(warnings ? { warnings } : {}), + name: parsed.meta.name, + description: parsed.meta.description, + phases: parsed.meta.phases, + } + } + + /** + * Save a script as a reusable `/name` command. + * + * Refuses to write through a symlink: the target directory is user-owned + * config, and following a link would place the file somewhere the caller + * did not choose. + */ + async saveDefinition(params: { + script: string + scope: WorkflowSaveScope + cwd?: string + }): Promise<{ name: string; filePath: string }> { + this.assertEnabled() + const parsed = parseWorkflowScript(params.script) + if ('error' in parsed) throw ApiError.badRequest(parsed.error) + const compiled = compileWorkflowScript(parsed.scriptBody) + if (!compiled.ok) throw ApiError.badRequest(compiled.error) + + const dir = + params.scope === 'project' + ? path.join(params.cwd ?? process.cwd(), '.claude', 'workflows') + : path.join(this.configDir(), 'workflows') + const filePath = path.join(dir, `${parsed.meta.name}.js`) + + await this.assertNotSymlink(filePath) + await fs.mkdir(dir, { recursive: true }) + await fs.writeFile(filePath, params.script, 'utf8') + return { name: parsed.meta.name, filePath } + } + + async deleteDefinition( + name: string, + scope: WorkflowSaveScope, + cwd?: string, + ): Promise { + this.assertEnabled() + if (!/^[a-zA-Z0-9][a-zA-Z0-9_-]*$/.test(name)) { + throw ApiError.badRequest(`Invalid workflow name: ${name}`) + } + const dir = + scope === 'project' + ? path.join(cwd ?? process.cwd(), '.claude', 'workflows') + : path.join(this.configDir(), 'workflows') + const filePath = path.join(dir, `${name}.js`) + await this.assertNotSymlink(filePath) + try { + await fs.unlink(filePath) + } catch (error) { + if ((error as NodeJS.ErrnoException).code === 'ENOENT') { + throw ApiError.notFound(`No saved workflow at ${filePath}`) + } + throw error + } + } + + /** + * Rebuild a session's workflow runs from what the CLI left on disk. + * + * The live progress stream only exists while a run is happening, so + * reopening a finished session had nothing to show. Each agent's sidecar + * metadata records the run and phase it belonged to and is written before + * the agent starts, which makes it the durable record of a run's shape. + * Runs from before that field existed still list their agents, ungrouped. + */ + async reconstructSessionRuns(sessionId: string): Promise { + this.assertEnabled() + const runs = await this.listRuns({ sessionId }) + if (runs.length === 0) return [] + + const byRunId = new Map() + for (const run of runs) { + byRunId.set(run.runId, { + runId: run.runId, + workflowName: run.workflowName, + startedAt: run.startedAt, + agents: [], + }) + } + + for (const { dir } of await this.findRunDirs(sessionId)) { + // `dir` is the session's `workflows/` script directory; the agent + // sidecars are its sibling `subagents/`. + const subagentsDir = path.join(path.dirname(dir), 'subagents') + await this.collectRunAgents(subagentsDir, byRunId) + + // Runs from before agents recorded their phase wrote their sidecars + // into `subagents/workflows//` instead. The directory name is + // the only provenance they have, so those agents are recovered without + // grouping rather than being dropped entirely. + const legacyRoot = path.join(subagentsDir, 'workflows') + let legacyRunIds: string[] + try { + legacyRunIds = await fs.readdir(legacyRoot) + } catch { + continue + } + for (const legacyRunId of legacyRunIds) { + if (!byRunId.has(legacyRunId)) continue + await this.collectRunAgents( + path.join(legacyRoot, legacyRunId), + byRunId, + legacyRunId, + ) + } + } + + for (const run of byRunId.values()) { + run.agents.sort((a, b) => a.agentIndex - b.agentIndex) + } + // Only runs we could actually rebuild agents for are worth returning; an + // empty shell would render as a phantom section. + return [...byRunId.values()] + .filter(run => run.agents.length > 0) + .sort((a, b) => b.startedAt - a.startedAt) + } + + /** + * Add every workflow agent sidecar in `dir` to the run it belongs to. + * + * `fallbackRunId` covers legacy layouts where provenance came from the + * directory rather than the metadata; those agents land in phase 0 because + * their phase was never recorded anywhere. + */ + private async collectRunAgents( + dir: string, + byRunId: Map, + fallbackRunId?: string, + ): Promise { + let entries: string[] + try { + entries = await fs.readdir(dir) + } catch { + return + } + let fallbackIndex = 0 + for (const entry of entries) { + if (!entry.endsWith('.meta.json')) continue + let meta: { + agentType?: string + description?: string + workflow?: { + runId: string + name: string + phaseIndex: number + phaseTitle?: string + agentIndex: number + } + } + try { + meta = JSON.parse(await fs.readFile(path.join(dir, entry), 'utf8')) + } catch { + continue + } + const runId = meta.workflow?.runId ?? fallbackRunId + if (!runId) continue + const run = byRunId.get(runId) + if (!run) continue + const agentId = entry.replace(/^agent-/, '').replace(/\.meta\.json$/, '') + if (run.agents.some(agent => agent.agentId === agentId)) continue + fallbackIndex += 1 + run.agents.push({ + agentId, + label: + meta.description ?? + `agent ${meta.workflow?.agentIndex ?? fallbackIndex}`, + phaseIndex: meta.workflow?.phaseIndex ?? 0, + ...(meta.workflow?.phaseTitle + ? { phaseTitle: meta.workflow.phaseTitle } + : {}), + agentIndex: meta.workflow?.agentIndex ?? fallbackIndex, + }) + } + } + + /** Every run whose script the CLI persisted, newest first. */ + async listRuns(options?: { + sessionId?: string + limit?: number + }): Promise { + this.assertEnabled() + const runs: WorkflowRunSummary[] = [] + for (const { sessionId, dir } of await this.findRunDirs(options?.sessionId)) { + let entries: string[] + try { + entries = await fs.readdir(dir) + } catch { + continue + } + for (const entry of entries) { + const match = RUN_SCRIPT_PATTERN.exec(entry) + if (!match) continue + const [, workflowName, runId] = match + const scriptPath = path.join(dir, entry) + let startedAt = 0 + try { + startedAt = (await fs.stat(scriptPath)).mtimeMs + } catch { + continue + } + const journal = await this.readJournal(sessionId, runId!) + runs.push({ + runId: runId!, + sessionId, + workflowName: workflowName!, + scriptPath, + startedAt, + completedAgents: journal.length, + status: 'unknown', + }) + } + } + runs.sort((a, b) => b.startedAt - a.startedAt) + return options?.limit ? runs.slice(0, options.limit) : runs + } + + async getRun(sessionId: string, runId: string): Promise { + this.assertEnabled() + const runs = await this.listRuns({ sessionId }) + const summary = runs.find(run => run.runId === runId) + if (!summary) throw ApiError.notFound(`Unknown workflow run: ${runId}`) + + const script = await fs.readFile(summary.scriptPath, 'utf8') + const parsed = parseWorkflowScript(script) + const agents = await this.readJournal(sessionId, runId) + + return { + ...summary, + script, + description: 'error' in parsed ? undefined : parsed.meta.description, + phases: 'error' in parsed ? undefined : parsed.meta.phases, + agents, + } + } + + /** `///workflows` directories to scan. */ + private async findRunDirs( + sessionId?: string, + ): Promise> { + const projectsDir = this.projectsDir() + let projects: string[] + try { + projects = await fs.readdir(projectsDir) + } catch { + return [] + } + + const found: Array<{ sessionId: string; dir: string }> = [] + for (const project of projects) { + const projectPath = path.join(projectsDir, project) + let sessions: string[] + try { + sessions = await fs.readdir(projectPath) + } catch { + continue + } + for (const entry of sessions) { + if (sessionId && entry !== sessionId) continue + const dir = path.join(projectPath, entry, 'workflows') + try { + if (!(await fs.stat(dir)).isDirectory()) continue + } catch { + continue + } + found.push({ sessionId: entry, dir }) + } + } + return found + } + + private async readJournal( + sessionId: string, + runId: string, + ): Promise> { + const dirs = await this.findRunDirs(sessionId) + for (const { dir } of dirs) { + const journalPath = path.join( + path.dirname(dir), + 'subagents', + 'workflows', + runId, + 'journal.jsonl', + ) + let raw: string + try { + raw = await fs.readFile(journalPath, 'utf8') + } catch { + continue + } + const results: Array<{ key: string; agentId: string; result: unknown }> = [] + for (const line of raw.split('\n')) { + if (line.trim() === '') continue + try { + const entry = JSON.parse(line) as { + type: string + key: string + agentId: string + result?: unknown + } + if (entry.type !== 'result') continue + results.push({ + // The raw key chains every prior prompt; hash it so the API stays + // small and does not leak the whole prompt history. + key: createHash('sha256').update(entry.key).digest('hex').slice(0, 12), + agentId: entry.agentId, + result: entry.result, + }) + } catch { + continue + } + } + return results + } + return [] + } + + private async assertNotSymlink(filePath: string): Promise { + try { + const stats = await fs.lstat(filePath) + if (stats.isSymbolicLink()) { + throw ApiError.badRequest( + `Refusing to write through a symlink: ${filePath}`, + ) + } + } catch (error) { + if ((error as NodeJS.ErrnoException).code === 'ENOENT') return + throw error + } + } + + private assertEnabled(): void { + if (!areWorkflowsEnabled()) { + throw new ApiError( + 403, + 'Dynamic workflows are disabled (`disableWorkflows`).', + 'WORKFLOWS_DISABLED', + ) + } + } +} + +export const workflowService = new WorkflowService() diff --git a/src/state/AppStateStore.ts b/src/state/AppStateStore.ts index fd6526e0..a08ff20b 100644 --- a/src/state/AppStateStore.ts +++ b/src/state/AppStateStore.ts @@ -425,6 +425,14 @@ export type AppState = DeepImmutable<{ advisorModel?: string // Effort value effortValue?: EffortValue + // Ultracode: xhigh effort plus standing dynamic-workflow orchestration. + // Session-scoped and never persisted — /effort ultracode sets it, a new + // session starts without it. + ultracode?: boolean + // Set by opt+w in the composer: the user typed `ultracode` but does not want + // this turn orchestrated. Cleared as soon as the keyword leaves the input, + // so it never leaks into the next prompt. + suppressWorkflowKeyword?: boolean // Set synchronously in launchUltraplan before the detached flow starts. // Prevents duplicate launches during the ~5s window before // ultraplanSessionUrl is set by teleportToRemote. Cleared by launchDetached @@ -563,6 +571,8 @@ export function getDefaultAppState(): AppState { authVersion: 0, initialMessage: null, effortValue: undefined, + ultracode: false, + suppressWorkflowKeyword: false, activeOverlays: new Set(), fastMode: false, } diff --git a/src/tasks.ts b/src/tasks.ts index 9ac21998..d5e86ecb 100644 --- a/src/tasks.ts +++ b/src/tasks.ts @@ -3,12 +3,10 @@ import type { Task, TaskType } from './Task.js' import { DreamTask } from './tasks/DreamTask/DreamTask.js' import { LocalAgentTask } from './tasks/LocalAgentTask/LocalAgentTask.js' import { LocalShellTask } from './tasks/LocalShellTask/LocalShellTask.js' +import { LocalWorkflowTask } from './tasks/LocalWorkflowTask/LocalWorkflowTask.js' import { RemoteAgentTask } from './tasks/RemoteAgentTask/RemoteAgentTask.js' /* eslint-disable @typescript-eslint/no-require-imports */ -const LocalWorkflowTask: Task | null = feature('WORKFLOW_SCRIPTS') - ? require('./tasks/LocalWorkflowTask/LocalWorkflowTask.js').LocalWorkflowTask - : null const MonitorMcpTask: Task | null = feature('MONITOR_TOOL') ? require('./tasks/MonitorMcpTask/MonitorMcpTask.js').MonitorMcpTask : null @@ -25,8 +23,8 @@ export function getAllTasks(): Task[] { LocalAgentTask, RemoteAgentTask, DreamTask, + LocalWorkflowTask, ] - if (LocalWorkflowTask) tasks.push(LocalWorkflowTask) if (MonitorMcpTask) tasks.push(MonitorMcpTask) return tasks } diff --git a/src/tasks/LocalWorkflowTask/LocalWorkflowTask.test.ts b/src/tasks/LocalWorkflowTask/LocalWorkflowTask.test.ts new file mode 100644 index 00000000..eca39eff --- /dev/null +++ b/src/tasks/LocalWorkflowTask/LocalWorkflowTask.test.ts @@ -0,0 +1,190 @@ +import { describe, expect, test } from 'bun:test' +import type { AppState } from '../../state/AppState.js' +import type { SetAppState } from '../../Task.js' +import { WORKFLOW_MAX_PROGRESS_ROWS } from '../../utils/workflows/constants.js' +import type { WorkflowProgressEvent } from '../../utils/workflows/types.js' +import { + buildResumePrompt, + buildWorkflowNotification, + registerWorkflowTask, + updateWorkflowProgressBatch, + type LocalWorkflowTaskState, +} from './LocalWorkflowTask.js' + +/** A minimal AppState stand-in: these helpers only read and write `tasks`. */ +function makeStore(task: LocalWorkflowTaskState): { + setAppState: SetAppState + get: () => LocalWorkflowTaskState +} { + let state = { tasks: { [task.id]: task } } as unknown as AppState + return { + setAppState: updater => { + state = updater(state) + }, + get: () => state.tasks[task.id] as LocalWorkflowTaskState, + } +} + +function makeTask(): LocalWorkflowTaskState { + return registerWorkflowTask({ + taskId: 'w0000001', + script: "export const meta = { name: 'demo', description: 'Demo' }\n", + scriptPath: '/tmp/demo.wf_abc12345-def.js', + summary: 'Demo', + workflowName: 'demo', + workflowRunId: 'wf_abc12345-def', + }) +} + +function agent( + index: number, + state: 'start' | 'progress' | 'done' | 'error', + extra: Partial> = {}, +): WorkflowProgressEvent { + return { type: 'workflow_agent', index, label: `a${index}`, state, ...extra } +} + +describe('updateWorkflowProgressBatch', () => { + test('replaces an agent row in place and recomputes the totals', () => { + const store = makeStore(makeTask()) + updateWorkflowProgressBatch( + 'w0000001', + [agent(1, 'progress', { tokens: 10, toolCalls: 1 })], + store.setAppState, + ) + updateWorkflowProgressBatch( + 'w0000001', + [ + agent(1, 'done', { tokens: 40, toolCalls: 3 }), + agent(2, 'progress', { tokens: 5, toolCalls: 0 }), + ], + store.setAppState, + ) + + const task = store.get() + expect(task.workflowProgress.filter(r => r.type === 'workflow_agent')).toHaveLength(2) + expect(task.agentCount).toBe(2) + expect(task.totalTokens).toBe(45) + expect(task.totalToolCalls).toBe(3) + expect(task.progressVersion).toBe(3) + }) + + test('phase rows are keyed separately from agent rows with the same index', () => { + const store = makeStore(makeTask()) + updateWorkflowProgressBatch( + 'w0000001', + [ + { type: 'workflow_phase', index: 1, title: 'Scan', kind: 'meta' }, + agent(1, 'done'), + ], + store.setAppState, + ) + const rows = store.get().workflowProgress + expect(rows).toHaveLength(2) + expect(rows[0]?.type).toBe('workflow_phase') + expect(rows[1]?.type).toBe('workflow_agent') + }) + + test('logs are trimmed before agent rows when the buffer overflows', () => { + const store = makeStore(makeTask()) + const logs: WorkflowProgressEvent[] = Array.from( + { length: WORKFLOW_MAX_PROGRESS_ROWS * 2 + 10 }, + (_unused, i) => ({ type: 'workflow_log', message: `line ${i}` }), + ) + updateWorkflowProgressBatch( + 'w0000001', + [agent(1, 'done', { tokens: 7 }), ...logs], + store.setAppState, + ) + + const rows = store.get().workflowProgress + expect(rows.length).toBeLessThanOrEqual(WORKFLOW_MAX_PROGRESS_ROWS + 1) + expect(rows.some(r => r.type === 'workflow_agent')).toBe(true) + // Oldest logs go first, so the newest line must survive. + const kept = rows.filter(r => r.type === 'workflow_log') + expect(kept.at(-1)).toEqual({ + type: 'workflow_log', + message: `line ${logs.length - 1}`, + }) + expect(store.get().totalTokens).toBe(7) + }) + + test('a settled task ignores late progress', () => { + const task = makeTask() + const store = makeStore({ ...task, status: 'completed' }) + updateWorkflowProgressBatch('w0000001', [agent(1, 'done')], store.setAppState) + expect(store.get().workflowProgress).toHaveLength(0) + }) + + test('an empty batch does not bump the version', () => { + const store = makeStore(makeTask()) + updateWorkflowProgressBatch('w0000001', [], store.setAppState) + expect(store.get().progressVersion).toBe(0) + }) +}) + +describe('buildWorkflowNotification', () => { + const base = { + taskId: 'w0000001', + summary: 'Demo', + agentCount: 3, + totalTokens: 1200, + totalToolCalls: 9, + durationMs: 4500, + transcriptDir: '/tmp/run', + scriptPath: '/tmp/demo.wf_abc12345-def.js', + workflowRunId: 'wf_abc12345-def', + } + + test('a completed run points at the journal and the replay call', () => { + const message = buildWorkflowNotification({ + ...base, + status: 'completed', + result: ['ALPHA'], + }) + expect(message).toContain('completed') + expect(message).toContain('/tmp/run/journal.jsonl') + expect(message).toContain( + "Workflow({scriptPath: '/tmp/demo.wf_abc12345-def.js', resumeFromRunId: 'wf_abc12345-def'})", + ) + expect(message).toContain('["ALPHA"]') + expect(message).toContain( + '3120094500', + ) + }) + + test('a failed run offers recovery instead of the journal note', () => { + const message = buildWorkflowNotification({ + ...base, + status: 'failed', + error: 'agent exploded', + failures: ['parallel[1] failed: boom'], + }) + expect(message).toContain('') + expect(message).not.toContain('journal.jsonl') + expect(message).toContain('\nparallel[1] failed: boom\n') + expect(message).toContain('failed: agent exploded') + }) + + test('args are threaded into the resume call so a replay reruns the same input', () => { + const message = buildWorkflowNotification({ + ...base, + status: 'failed', + error: 'nope', + args: ['a.ts', 'b.ts'], + }) + expect(message).toContain('args: ["a.ts","b.ts"]') + }) +}) + +describe('buildResumePrompt', () => { + test('names the script path and run id', () => { + const prompt = buildResumePrompt({ + ...makeTask(), + args: { q: 1 }, + }) + expect(prompt).toContain("scriptPath: '/tmp/demo.wf_abc12345-def.js'") + expect(prompt).toContain("resumeFromRunId: 'wf_abc12345-def'") + expect(prompt).toContain('args: {"q":1}') + }) +}) diff --git a/src/tasks/LocalWorkflowTask/LocalWorkflowTask.ts b/src/tasks/LocalWorkflowTask/LocalWorkflowTask.ts index c170c49a..6d47be09 100644 --- a/src/tasks/LocalWorkflowTask/LocalWorkflowTask.ts +++ b/src/tasks/LocalWorkflowTask/LocalWorkflowTask.ts @@ -1,34 +1,507 @@ -// @generated stub from scan-missing-imports -// 该文件自动生成,对应 ant-internal 的 feature() gated 模块。 -// 所有外部 build 的代码路径在 DCE 后都不会真的执行这里的代码,这只是 -// bun build resolver 的占位符。 -const __target = function noop() {} -const __handler: ProxyHandler = { - get(_t, prop) { - if (prop === '__esModule') return true - if (prop === 'default') return new Proxy(__target, __handler) - if (prop === Symbol.toPrimitive) return () => undefined - if (prop === Symbol.iterator) return function* () {} - if (prop === Symbol.asyncIterator) return async function* () {} - if (prop === 'then') return undefined - return new Proxy(__target, __handler) - }, - apply() { - return new Proxy(__target, __handler) - }, - construct() { - return new Proxy(__target, __handler) +import { writeFile } from 'fs/promises' +import { + OUTPUT_FILE_TAG, + STATUS_TAG, + SUMMARY_TAG, + TASK_ID_TAG, + TASK_NOTIFICATION_TAG, + TASK_TYPE_TAG, + TOOL_USE_ID_TAG, +} from '../../constants/xml.js' +import type { SetAppState, Task, TaskStateBase } from '../../Task.js' +import { createTaskStateBase } from '../../Task.js' +import { createAbortController } from '../../utils/abortController.js' +import { logForDebugging } from '../../utils/debug.js' +import { enqueuePendingNotification } from '../../utils/messageQueueManager.js' +import { + evictTaskOutput, + getTaskOutputPath, +} from '../../utils/task/diskOutput.js' +import { PANEL_GRACE_MS, updateTaskState } from '../../utils/task/framework.js' +import { WORKFLOW_MAX_PROGRESS_ROWS } from '../../utils/workflows/constants.js' +import { + isDurableWorkflowEvent, + type WorkflowPhaseMeta, + type WorkflowProgressEvent, +} from '../../utils/workflows/types.js' + +export type LocalWorkflowTaskState = TaskStateBase & { + type: 'local_workflow' + /** Full script text as executed — the same bytes written to `scriptPath`. */ + script: string + scriptPath?: string + /** Kept under `prompt` too so generic task consumers can show something. */ + prompt: string + args?: unknown + summary?: string + workflowName?: string + title?: string + phases?: WorkflowPhaseMeta[] + defaultModel?: string + workflowRunId: string + ownerAgentId?: string + workflowProgress: WorkflowProgressEvent[] + /** Bumped on every applied batch so views can diff cheaply. */ + progressVersion: number + agentCount: number + totalTokens: number + totalToolCalls: number + logs: string[] + result?: unknown + error?: string + abortController?: AbortController + /** Per-agent controllers, so a single agent can be skipped or restarted. */ + agentControllers?: Map + evictAfter?: number +} + +export function isLocalWorkflowTask( + task: unknown, +): task is LocalWorkflowTaskState { + return ( + typeof task === 'object' && + task !== null && + 'type' in task && + task.type === 'local_workflow' + ) +} + +export function registerWorkflowTask(params: { + taskId: string + script: string + scriptPath?: string + args?: unknown + summary?: string + workflowName?: string + title?: string + phases?: WorkflowPhaseMeta[] + defaultModel?: string + workflowRunId: string + ownerAgentId?: string + toolUseId?: string + startTime?: number +}): LocalWorkflowTaskState { + const base = createTaskStateBase( + params.taskId, + 'local_workflow', + params.summary ?? 'Dynamic workflow', + params.toolUseId, + ) + return { + ...base, + ...(params.startTime !== undefined ? { startTime: params.startTime } : {}), + type: 'local_workflow', + status: 'running', + script: params.script, + scriptPath: params.scriptPath, + args: params.args, + prompt: params.script, + summary: params.summary, + workflowName: params.workflowName, + title: params.title, + phases: params.phases, + defaultModel: params.defaultModel, + workflowRunId: params.workflowRunId, + ownerAgentId: params.ownerAgentId, + workflowProgress: [], + progressVersion: 0, + agentCount: 0, + totalTokens: 0, + totalToolCalls: 0, + logs: [], + abortController: createAbortController(), + agentControllers: new Map(), + } +} + +/** + * Fold a batch of progress events into the task's state. + * + * Agent and phase rows are keyed by `type:index` and overwritten in place, so + * a run that emits hundreds of updates for the same twenty agents keeps twenty + * rows. Only logs accumulate, and they are the first thing trimmed. + */ +export function updateWorkflowProgressBatch( + taskId: string, + events: WorkflowProgressEvent[], + setAppState: SetAppState, +): void { + if (events.length === 0) return + updateTaskState(taskId, setAppState, task => { + if (task.status !== 'running') return task + + const rows = [...task.workflowProgress] + const rowIndexByKey = new Map() + for (let i = 0; i < rows.length; i++) { + const row = rows[i]! + if (row.type === 'workflow_agent' || row.type === 'workflow_phase') { + rowIndexByKey.set(`${row.type}:${row.index}`, i) + } + } + + let agentCount = task.agentCount + let appendedLogs = false + for (const event of events) { + if (event.type === 'workflow_log') { + rows.push(event) + appendedLogs = true + continue + } + const key = `${event.type}:${event.index}` + const existing = rowIndexByKey.get(key) + if (existing !== undefined) rows[existing] = event + else { + rowIndexByKey.set(key, rows.length) + rows.push(event) + } + if (event.type === 'workflow_agent') { + agentCount = Math.max(agentCount, event.index) + } + } + + const trimmed = + appendedLogs && rows.length > WORKFLOW_MAX_PROGRESS_ROWS * 2 + ? dropOldestLogs(rows, rows.length - WORKFLOW_MAX_PROGRESS_ROWS) + : rows + + let totalTokens = 0 + let totalToolCalls = 0 + for (const row of trimmed) { + if (row.type !== 'workflow_agent') continue + totalTokens += row.tokens ?? 0 + totalToolCalls += row.toolCalls ?? 0 + } + + return { + ...task, + workflowProgress: trimmed, + progressVersion: task.progressVersion + events.length, + agentCount, + totalTokens, + totalToolCalls, + } + }) +} + +function dropOldestLogs( + rows: WorkflowProgressEvent[], + toDrop: number, +): WorkflowProgressEvent[] { + let remaining = toDrop + const kept: WorkflowProgressEvent[] = [] + for (const row of rows) { + if (remaining > 0 && row.type === 'workflow_log') { + remaining-- + continue + } + kept.push(row) + } + return kept +} + +function settleWorkflowTask( + taskId: string, + setAppState: SetAppState, + status: 'completed' | 'failed' | 'killed' | 'paused', + patch: Partial, +): LocalWorkflowTaskState | null { + let settled: LocalWorkflowTaskState | null = null + updateTaskState(taskId, setAppState, task => { + if (task.status !== 'running') return task + settled = task + task.abortController?.abort() + const endTime = Date.now() + return { + ...task, + ...patch, + status, + endTime, + ...(status !== 'paused' ? { evictAfter: endTime + PANEL_GRACE_MS } : {}), + abortController: undefined, + agentControllers: undefined, + } + }) + return settled +} + +export function completeWorkflowTask( + taskId: string, + result: unknown, + agentCount: number, + logs: string[], + setAppState: SetAppState, +): void { + const settled = settleWorkflowTask(taskId, setAppState, 'completed', { + result, + agentCount, + logs, + }) + if (!settled) return + void writeWorkflowOutput(settled, { result, agentCount, logs }) +} + +export function failWorkflowTask( + taskId: string, + error: string, + agentCount: number, + logs: string[], + setAppState: SetAppState, +): void { + const settled = settleWorkflowTask(taskId, setAppState, 'failed', { + error, + agentCount, + logs, + }) + if (!settled) return + void evictTaskOutput(taskId) + void writeWorkflowOutput(settled, { error, agentCount, logs }) +} + +/** Stop the run but keep it resumable: the journal on disk is still valid. */ +export function pauseWorkflowTask( + taskId: string, + setAppState: SetAppState, +): boolean { + return ( + settleWorkflowTask(taskId, setAppState, 'paused', { notified: true }) !== + null + ) +} + +export function killWorkflowTask( + taskId: string, + setAppState: SetAppState, +): boolean { + const settled = settleWorkflowTask(taskId, setAppState, 'killed', { + notified: true, + }) + if (settled) void evictTaskOutput(taskId) + return settled !== null +} + +function abortWorkflowAgent( + taskId: string, + agentKey: string, + reason: 'user-skip' | 'user-retry', + setAppState: SetAppState, +): boolean { + let aborted = false + updateTaskState(taskId, setAppState, task => { + if (task.status !== 'running') return task + const controller = task.agentControllers?.get(agentKey) + if (controller && !controller.signal.aborted) { + controller.abort(new DOMException(reason, 'AbortError')) + aborted = true + } + return task + }) + return aborted +} + +export function skipWorkflowAgent( + taskId: string, + agentKey: string, + setAppState: SetAppState, +): boolean { + return abortWorkflowAgent(taskId, agentKey, 'user-skip', setAppState) +} + +export function retryWorkflowAgent( + taskId: string, + agentKey: string, + setAppState: SetAppState, +): boolean { + return abortWorkflowAgent(taskId, agentKey, 'user-retry', setAppState) +} + +/** The `Workflow(...)` call that picks a stopped run back up. */ +export function buildResumePrompt(task: LocalWorkflowTaskState): string { + const argsPart = + task.args !== undefined ? `, args: ${JSON.stringify(task.args)}` : '' + return ( + `Resume the paused workflow by calling: Workflow({scriptPath: '${task.scriptPath}', ` + + `resumeFromRunId: '${task.workflowRunId}'${argsPart}}) — completed agents return cached results.` + ) +} + +export function enqueueWorkflowNotification(params: { + taskId: string + summary: string + status: 'completed' | 'failed' | 'killed' + result?: unknown + error?: string + failures?: string[] + agentCount: number + totalTokens: number + totalToolCalls: number + durationMs: number + toolUseId?: string + transcriptDir?: string + scriptPath?: string + workflowRunId?: string + args?: unknown + setAppState: SetAppState +}): void { + let shouldEnqueue = false + updateTaskState( + params.taskId, + params.setAppState, + task => { + if (task.notified) return task + shouldEnqueue = true + return { ...task, notified: true } + }, + ) + if (!shouldEnqueue) return + + const message = buildWorkflowNotification(params) + enqueuePendingNotification({ value: message, mode: 'task-notification' }) +} + +/** + * The `` the model reads when a run settles. + * + * The recovery and journal hints matter: without the exact `Workflow({...})` + * call and the journal path, a model looking at an empty result has no way to + * tell "the agents returned nothing" from "post-processing dropped it". + */ +export function buildWorkflowNotification(params: { + taskId: string + summary: string + status: 'completed' | 'failed' | 'killed' + result?: unknown + error?: string + failures?: string[] + agentCount: number + totalTokens: number + totalToolCalls: number + durationMs: number + toolUseId?: string + transcriptDir?: string + scriptPath?: string + workflowRunId?: string + args?: unknown +}): string { + const { + taskId, + summary, + status, + result, + error, + failures, + agentCount, + totalTokens, + totalToolCalls, + durationMs, + toolUseId, + transcriptDir, + scriptPath, + workflowRunId, + args, + } = params + + const headline = + status === 'completed' + ? `Dynamic workflow "${summary}" completed` + : status === 'failed' + ? `Dynamic workflow "${summary}" failed: ${error || 'Unknown error'}` + : `Dynamic workflow "${summary}" was stopped` + + const argsPart = args !== undefined ? `, args: ${JSON.stringify(args)}` : '' + const resumeCall = + scriptPath && workflowRunId + ? `Workflow({scriptPath: '${scriptPath}', resumeFromRunId: '${workflowRunId}'${argsPart}})` + : undefined + + const sections: string[] = [] + if (status !== 'completed') { + const recovery: string[] = [] + if (resumeCall) { + recovery.push(`To resume after editing the script, call: ${resumeCall}`) + } + if (transcriptDir) recovery.push(`Agent transcripts: ${transcriptDir}`) + if (recovery.length > 0) { + sections.push(`\n\n${recovery.join('\n')}\n`) + } + } else if (transcriptDir) { + const notes = [ + `Per-agent results: ${transcriptDir}/journal.jsonl — one {"type":"result",...} line per completed agent with its full return value.`, + 'If the result above is empty or unexpected, Read this file BEFORE diagnosing — do not assume agents returned non-empty results.', + ] + if (resumeCall) { + notes.push( + `To re-run with edited post-processing: ${resumeCall} — agents whose (prompt, opts) are unchanged replay from cache.`, + ) + } + sections.push(`\n\n${notes.join('\n')}\n`) + } + if (failures && failures.length > 0) { + sections.push(`\n\n${failures.join('\n')}\n`) + } + + const resultSection = + result === undefined + ? '' + : `\n${safeJson(result)}` + const toolUseIdLine = toolUseId + ? `\n<${TOOL_USE_ID_TAG}>${toolUseId}` + : '' + + return `<${TASK_NOTIFICATION_TAG}> +<${TASK_ID_TAG}>${taskId}${toolUseIdLine} +<${TASK_TYPE_TAG}>local_workflow +<${OUTPUT_FILE_TAG}>${getTaskOutputPath(taskId)} +<${STATUS_TAG}>${status} +<${SUMMARY_TAG}>${headline}${resultSection}${sections.join('')} +${agentCount}${totalTokens}${totalToolCalls}${durationMs} +` +} + +async function writeWorkflowOutput( + task: LocalWorkflowTaskState, + extra: { result?: unknown; error?: string; agentCount: number; logs: string[] }, +): Promise { + try { + await writeFile( + task.outputFile, + JSON.stringify( + { + summary: task.summary, + workflowName: task.workflowName, + workflowRunId: task.workflowRunId, + agentCount: extra.agentCount, + logs: extra.logs, + result: extra.result, + error: extra.error, + workflowProgress: task.workflowProgress.filter(isDurableWorkflowEvent), + totalTokens: task.totalTokens, + totalToolCalls: task.totalToolCalls, + }, + null, + 2, + ), + 'utf8', + ) + } catch (error) { + logForDebugging( + `Failed to write workflow output for ${task.id}: ${error instanceof Error ? error.message : String(error)}`, + ) + } +} + +function safeJson(value: unknown): string { + if (typeof value === 'string') return value + try { + return JSON.stringify(value) ?? 'null' + } catch { + return '[unserializable result]' + } +} + +export const LocalWorkflowTask: Task = { + name: 'LocalWorkflowTask', + type: 'local_workflow', + async kill(taskId, setAppState) { + killWorkflowTask(taskId, setAppState) }, } -const stub: any = new Proxy(__target, __handler) -export default stub -export const __stubMissing = true -// 兼容常见的命名导出 —— 没列在这里的也会通过 default Proxy 兜底 -export const createCachedMCState = stub -export const isCachedMicrocompactEnabled = stub -export const isModelSupportedForCacheEditing = stub -export const getCachedMCConfig = stub -export const markToolsSentToAPI = stub -export const resetCachedMCState = stub -export const checkProtectedNamespace = stub -export const getCoordinatorUserContext = stub diff --git a/src/tools.ts b/src/tools.ts index 3e4e23d0..cf5b43ee 100644 --- a/src/tools.ts +++ b/src/tools.ts @@ -128,13 +128,14 @@ const SnipTool = feature('HISTORY_SNIP') const ListPeersTool = feature('UDS_INBOX') ? require('./tools/ListPeersTool/ListPeersTool.js').ListPeersTool : null -const WorkflowTool = feature('WORKFLOW_SCRIPTS') - ? (() => { - require('./tools/WorkflowTool/bundled/index.js').initBundledWorkflows() - return require('./tools/WorkflowTool/WorkflowTool.js').WorkflowTool - })() - : null /* eslint-enable custom-rules/no-process-env-top-level, @typescript-eslint/no-require-imports */ +// Lazy require: WorkflowTool -> launchWorkflow -> tools.js (assembleToolPool) +// is a cycle, so it cannot be a top-level import. +/* eslint-disable @typescript-eslint/no-require-imports */ +const getWorkflowTool = () => + require('./tools/WorkflowTool/WorkflowTool.js') + .WorkflowTool as typeof import('./tools/WorkflowTool/WorkflowTool.js').WorkflowTool +/* eslint-enable @typescript-eslint/no-require-imports */ import type { ToolPermissionContext } from './Tool.js' import { getDenyRuleForTool } from './utils/permissions/permissions.js' import { hasEmbeddedSearchTools } from './utils/embeddedTools.js' @@ -236,7 +237,7 @@ export function getAllBaseTools(): Tools { : []), ...(VerifyPlanExecutionTool ? [VerifyPlanExecutionTool] : []), ...(process.env.USER_TYPE === 'ant' && REPLTool ? [REPLTool] : []), - ...(WorkflowTool ? [WorkflowTool] : []), + getWorkflowTool(), ...(SleepTool ? [SleepTool] : []), ...cronTools, ...(RemoteTriggerTool ? [RemoteTriggerTool] : []), diff --git a/src/tools/AgentTool/runAgent.ts b/src/tools/AgentTool/runAgent.ts index d1919f8e..58b57ee9 100644 --- a/src/tools/AgentTool/runAgent.ts +++ b/src/tools/AgentTool/runAgent.ts @@ -59,10 +59,9 @@ import { createUserMessage } from '../../utils/messages.js' import { getAgentModel } from '../../utils/model/agent.js' import type { ModelAlias } from '../../utils/model/aliases.js' import { - clearAgentTranscriptSubdir, recordSidechainTranscript, - setAgentTranscriptSubdir, writeAgentMetadata, + type AgentMetadata, } from '../../utils/sessionStorage.js' import { isRestrictedToPluginOnly, @@ -311,7 +310,7 @@ export async function* runAgent({ spawningToolUseId, persistedAgentType, alreadyPersistedMessageCount, - transcriptSubdir, + workflow, onQueryProgress, }: { agentDefinition: AgentDefinition @@ -379,7 +378,8 @@ export async function* runAgent({ alreadyPersistedMessageCount?: number /** Optional subdirectory under subagents/ to group this agent's transcript * with related ones (e.g. workflows/ for workflow subagents). */ - transcriptSubdir?: string + /** Set when this agent is one step of a dynamic workflow run. */ + workflow?: AgentMetadata['workflow'] /** Optional callback fired on every message yielded by query() — including * stream_event deltas that runAgent otherwise drops. Use to detect liveness * during long single-block streams (e.g. thinking) where no assistant @@ -405,12 +405,6 @@ export async function* runAgent({ const agentId = override?.agentId ? override.agentId : createAgentId() - // Route this agent's transcript into a grouping subdirectory if requested - // (e.g. workflow subagents write to subagents/workflows//). - if (transcriptSubdir) { - setAgentTranscriptSubdir(agentId, transcriptSubdir) - } - // Register agent in Perfetto trace for hierarchy visualization if (isPerfettoTracingEnabled()) { const parentId = toolUseContext.agentId ?? getSessionId() @@ -826,6 +820,7 @@ export async function* runAgent({ ...(worktreePath && { worktreePath }), ...(description && { description }), ...(spawningToolUseId && { toolUseId: spawningToolUseId }), + ...(workflow && { workflow }), }).catch(_err => logForDebugging(`Failed to write agent metadata: ${_err}`)) // Track the last recorded message UUID for parent chain continuity @@ -918,7 +913,6 @@ export async function* runAgent({ // Release perfetto agent registry entry unregisterPerfettoAgent(agentId) // Release transcript subdir mapping - clearAgentTranscriptSubdir(agentId) // Release this agent's todos entry. Without this, every subagent that // called TodoWrite leaves a key in AppState.todos forever (even after all // items complete, the value is [] but the key stays). Whale sessions diff --git a/src/tools/WorkflowTool/WorkflowPermissionRequest.ts b/src/tools/WorkflowTool/WorkflowPermissionRequest.ts deleted file mode 100644 index c170c49a..00000000 --- a/src/tools/WorkflowTool/WorkflowPermissionRequest.ts +++ /dev/null @@ -1,34 +0,0 @@ -// @generated stub from scan-missing-imports -// 该文件自动生成,对应 ant-internal 的 feature() gated 模块。 -// 所有外部 build 的代码路径在 DCE 后都不会真的执行这里的代码,这只是 -// bun build resolver 的占位符。 -const __target = function noop() {} -const __handler: ProxyHandler = { - get(_t, prop) { - if (prop === '__esModule') return true - if (prop === 'default') return new Proxy(__target, __handler) - if (prop === Symbol.toPrimitive) return () => undefined - if (prop === Symbol.iterator) return function* () {} - if (prop === Symbol.asyncIterator) return async function* () {} - if (prop === 'then') return undefined - return new Proxy(__target, __handler) - }, - apply() { - return new Proxy(__target, __handler) - }, - construct() { - return new Proxy(__target, __handler) - }, -} -const stub: any = new Proxy(__target, __handler) -export default stub -export const __stubMissing = true -// 兼容常见的命名导出 —— 没列在这里的也会通过 default Proxy 兜底 -export const createCachedMCState = stub -export const isCachedMicrocompactEnabled = stub -export const isModelSupportedForCacheEditing = stub -export const getCachedMCConfig = stub -export const markToolsSentToAPI = stub -export const resetCachedMCState = stub -export const checkProtectedNamespace = stub -export const getCoordinatorUserContext = stub diff --git a/src/tools/WorkflowTool/WorkflowPermissionRequest.tsx b/src/tools/WorkflowTool/WorkflowPermissionRequest.tsx new file mode 100644 index 00000000..1e061d4f --- /dev/null +++ b/src/tools/WorkflowTool/WorkflowPermissionRequest.tsx @@ -0,0 +1,214 @@ +import React, { useCallback, useMemo, useState } from 'react' +import { getOriginalCwd } from '../../bootstrap/state.js' +import { PermissionDialog } from '../../components/permissions/PermissionDialog.js' +import { + PermissionPrompt, + type PermissionPromptOption, + type ToolAnalyticsContext, +} from '../../components/permissions/PermissionPrompt.js' +import type { PermissionRequestProps } from '../../components/permissions/PermissionRequest.js' +import { PermissionRuleExplanation } from '../../components/permissions/PermissionRuleExplanation.js' +import { + type UnaryEvent, + usePermissionRequestLogging, +} from '../../components/permissions/hooks.js' +import { Box, Text } from '../../ink.js' +import { sanitizeToolNameForAnalytics } from '../../services/analytics/metadata.js' +import { shouldShowAlwaysAllowOptions } from '../../utils/permissions/permissionsLoader.js' +import { recordWorkflowAutoModeConsent } from '../../utils/workflows/autoModeConsent.js' +import { parseWorkflowScript } from '../../utils/workflows/meta.js' +import type { WorkflowMeta } from '../../utils/workflows/types.js' +import { WORKFLOW_TOOL_NAME } from './constants.js' + +type WorkflowOptionValue = 'yes' | 'yes-always' | 'view' | 'no' + +/** + * Approval dialog shown before a dynamic workflow starts. + * + * The point of the dialog is the phase list: it is the only preview the user + * gets of how many agents are about to run and what they will do, and it is + * the last decision point — once the run starts, its subagents' file edits are + * auto-approved. + */ +export function WorkflowPermissionRequest( + props: PermissionRequestProps, +): React.ReactNode { + const { toolUseConfirm, onDone, onReject, workerBadge } = props + + const unaryEvent = useMemo( + () => ({ completion_type: 'tool_use_single', language_name: 'none' }), + [], + ) + usePermissionRequestLogging(toolUseConfirm, unaryEvent) + + const input = toolUseConfirm.input as { + script?: string + name?: string + scriptPath?: string + args?: unknown + } + const script = typeof input.script === 'string' ? input.script : undefined + const meta = useMemo(() => readMeta(input.script), [input.script]) + const workflowName = meta?.name ?? input.name ?? 'workflow' + const description = meta?.description + const phases = meta?.phases ?? [] + const originalCwd = getOriginalCwd() + const showAlwaysAllow = shouldShowAlwaysAllowOptions() && Boolean(input.name) + const [showScript, setShowScript] = useState(false) + + const options = useMemo[]>(() => { + const built: PermissionPromptOption[] = [ + { label: 'Yes, run it', value: 'yes', feedbackConfig: { type: 'accept' } }, + ] + if (showAlwaysAllow) { + built.push({ + label: ( + + Yes, and don't ask again for {workflowName} in{' '} + {originalCwd} + + ), + value: 'yes-always', + }) + } + if (script && !showScript) { + built.push({ label: 'View raw script', value: 'view' }) + } + built.push({ label: 'No', value: 'no', feedbackConfig: { type: 'reject' } }) + return built + }, [showAlwaysAllow, workflowName, originalCwd, script, showScript]) + + const toolAnalyticsContext = useMemo( + () => ({ + toolName: sanitizeToolNameForAnalytics(toolUseConfirm.tool.name), + isMcp: toolUseConfirm.tool.isMcp ?? false, + }), + [toolUseConfirm.tool.name, toolUseConfirm.tool.isMcp], + ) + + const isAutoMode = + toolUseConfirm.toolUseContext.getAppState().toolPermissionContext.mode === + 'auto' + + const handleSelect = useCallback( + (value: WorkflowOptionValue, feedback?: string) => { + // In auto mode a Yes of either kind is the one-time consent — after this + // the launch prompt stops appearing. + if (isAutoMode && (value === 'yes' || value === 'yes-always')) { + recordWorkflowAutoModeConsent() + } + switch (value) { + case 'yes': + toolUseConfirm.onAllow(toolUseConfirm.input, [], feedback) + onDone() + break + case 'yes-always': + toolUseConfirm.onAllow(toolUseConfirm.input, [ + { + type: 'addRules', + rules: [ + { toolName: WORKFLOW_TOOL_NAME, ruleContent: workflowName }, + ], + behavior: 'allow', + destination: 'localSettings', + }, + ]) + onDone() + break + case 'view': + // Stay in the dialog: the whole point is to read the script and then + // decide, so this must not resolve the permission either way. + setShowScript(true) + break + case 'no': + toolUseConfirm.onReject(feedback) + onReject() + onDone() + break + } + }, + [toolUseConfirm, onDone, onReject, workflowName, isAutoMode], + ) + + const handleCancel = useCallback(() => { + toolUseConfirm.onReject() + onReject() + onDone() + }, [toolUseConfirm, onDone, onReject]) + + return ( + + + A workflow spawns many subagents in the background. Their file edits are + auto-approved and the run can use a large number of tokens. + + + {description ? {description} : null} + {phases.length > 0 ? ( + + Phases: + {phases.map((phase, index) => ( + + {` ${index + 1}. ${phase.title}`} + {phase.detail ? ` — ${phase.detail}` : ''} + + ))} + + ) : null} + {input.scriptPath ? ( + + {`Script: ${input.scriptPath}`} + + ) : null} + {showScript && script ? ( + + Raw script: + {clipScript(script)} + + ) : null} + + + + + + + + ) +} + +const SCRIPT_PREVIEW_LINES = 60 + +/** Long scripts are clipped: the dialog must stay smaller than the terminal. */ +function clipScript(script: string): string { + const lines = script.split('\n') + if (lines.length <= SCRIPT_PREVIEW_LINES) return script + const remaining = lines.length - SCRIPT_PREVIEW_LINES + return ( + `${lines.slice(0, SCRIPT_PREVIEW_LINES).join('\n')}\n` + + `… ${remaining} more lines — the full script is persisted under the session directory` + ) +} + +/** + * Read the script's `meta` for the preview. + * + * Parsing can fail here — the tool has not validated the script yet — and a + * bad script should still reach the tool so the model sees the real parse + * error, not a silent refusal in the dialog. + */ +function readMeta(script: string | undefined): WorkflowMeta | undefined { + if (!script) return undefined + const parsed = parseWorkflowScript(script) + return 'error' in parsed ? undefined : parsed.meta +} diff --git a/src/tools/WorkflowTool/WorkflowTool.test.ts b/src/tools/WorkflowTool/WorkflowTool.test.ts new file mode 100644 index 00000000..40ed43ab --- /dev/null +++ b/src/tools/WorkflowTool/WorkflowTool.test.ts @@ -0,0 +1,182 @@ +import { describe, expect, test } from 'bun:test' +import type { ToolUseContext } from '../../Tool.js' +import { getEmptyToolPermissionContext } from '../../Tool.js' +import type { AppState } from '../../state/AppState.js' +import type { PermissionMode } from '../../types/permissions.js' +import { resolveScriptForTesting, WorkflowTool } from './WorkflowTool.js' + +function makeContext(options: { + mode?: PermissionMode + isNonInteractiveSession?: boolean + allowRules?: string[] + ultracode?: boolean +}): ToolUseContext { + const permissionContext = { + ...getEmptyToolPermissionContext(), + mode: options.mode ?? 'default', + ...(options.allowRules + ? { + alwaysAllowRules: { + localSettings: options.allowRules.map( + content => `${WorkflowTool.name}(${content})`, + ), + }, + } + : {}), + } + return { + options: { + isNonInteractiveSession: options.isNonInteractiveSession ?? false, + }, + getAppState: () => + ({ + toolPermissionContext: permissionContext, + ultracode: options.ultracode ?? false, + }) as unknown as AppState, + } as unknown as ToolUseContext +} + +describe('WorkflowTool.checkPermissions', () => { + test('asks before starting a run in default mode', async () => { + const result = await WorkflowTool.checkPermissions( + { name: 'deep-research' }, + makeContext({ mode: 'default' }), + ) + expect(result.behavior).toBe('ask') + if (result.behavior !== 'ask') return + expect(result.message).toContain('spawn many subagents') + // The "don't ask again" option needs a rule to write. + expect(result.suggestions?.[0]).toMatchObject({ + type: 'addRules', + behavior: 'allow', + destination: 'localSettings', + }) + }) + + test('still asks in acceptEdits — a run auto-approves its agents edits', async () => { + const result = await WorkflowTool.checkPermissions( + { name: 'deep-research' }, + makeContext({ mode: 'acceptEdits' }), + ) + expect(result.behavior).toBe('ask') + }) + + test('ultracode skips the launch prompt — it is a standing opt-in', async () => { + const result = await WorkflowTool.checkPermissions( + { name: 'deep-research' }, + makeContext({ mode: 'default', ultracode: true }), + ) + expect(result.behavior).toBe('allow') + }) + + test('allows without prompting under bypassPermissions', async () => { + const result = await WorkflowTool.checkPermissions( + { name: 'deep-research' }, + makeContext({ mode: 'bypassPermissions' }), + ) + expect(result.behavior).toBe('allow') + }) + + test('allows in a non-interactive session, where nobody can answer', async () => { + const result = await WorkflowTool.checkPermissions( + { script: "export const meta = { name: 'x', description: 'y' }\n" }, + makeContext({ isNonInteractiveSession: true }), + ) + expect(result.behavior).toBe('allow') + }) + + test('an allow rule for one workflow does not cover another', async () => { + const context = makeContext({ allowRules: ['deep-research'] }) + const allowed = await WorkflowTool.checkPermissions( + { name: 'deep-research' }, + context, + ) + expect(allowed.behavior).toBe('allow') + + const other = await WorkflowTool.checkPermissions( + { name: 'something-else' }, + context, + ) + expect(other.behavior).toBe('ask') + }) + + test('an inline script has no rule to match, so it always asks', async () => { + const result = await WorkflowTool.checkPermissions( + { script: "export const meta = { name: 'x', description: 'y' }\n" }, + makeContext({ allowRules: ['x'] }), + ) + expect(result.behavior).toBe('ask') + if (result.behavior !== 'ask') return + expect(result.suggestions).toBeUndefined() + }) +}) + +describe('resolveScript', () => { + test('a named workflow carries no scriptPath, so the run gets a session copy', async () => { + const resolved = await resolveScriptForTesting({ name: 'deep-research' }) + expect('script' in resolved && resolved.script).toContain('export const meta') + // Pointing the run at ~/.claude/workflows/.js would make it write + // back over the user's file, and a later resume would replay whatever that + // file says rather than what actually ran. + expect( + (resolved as { scriptPath?: string }).scriptPath, + ).toBeUndefined() + }) + + test('an explicit scriptPath is preserved so a resume reads the edited file', async () => { + const resolved = await resolveScriptForTesting({ + scriptPath: '/definitely/not/here.js', + }) + expect('error' in resolved && resolved.error).toContain('Failed to read') + }) +}) + +describe('WorkflowTool.call', () => { + test('rejects a call with no script, scriptPath, or name', async () => { + await expect( + WorkflowTool.call( + {}, + makeContext({ mode: 'bypassPermissions' }), + (() => {}) as never, + {} as never, + ), + ).rejects.toThrow('requires one of `script`, `scriptPath`, or `name`') + }) + + test('surfaces the parse error for a script without a leading meta export', async () => { + await expect( + WorkflowTool.call( + { script: 'const x = 1\n' }, + makeContext({ mode: 'bypassPermissions' }), + (() => {}) as never, + {} as never, + ), + ).rejects.toThrow('must be the FIRST statement') + }) + + test('points at plain JavaScript when the body has TypeScript syntax', async () => { + await expect( + WorkflowTool.call( + { + script: + "export const meta = { name: 'x', description: 'y' }\nconst files: string[] = []\n", + }, + makeContext({ mode: 'bypassPermissions' }), + (() => {}) as never, + {} as never, + ), + ).rejects.toThrow('plain JavaScript') + }) + + + test('reports a missing named workflow with the available list', async () => { + await expect( + WorkflowTool.call( + { name: 'no-such-workflow' }, + makeContext({ mode: 'bypassPermissions' }), + (() => {}) as never, + {} as never, + ), + ).rejects.toThrow("Unknown workflow 'no-such-workflow'") + }) +}) diff --git a/src/tools/WorkflowTool/WorkflowTool.ts b/src/tools/WorkflowTool/WorkflowTool.ts index c170c49a..ec8f5525 100644 --- a/src/tools/WorkflowTool/WorkflowTool.ts +++ b/src/tools/WorkflowTool/WorkflowTool.ts @@ -1,34 +1,299 @@ -// @generated stub from scan-missing-imports -// 该文件自动生成,对应 ant-internal 的 feature() gated 模块。 -// 所有外部 build 的代码路径在 DCE 后都不会真的执行这里的代码,这只是 -// bun build resolver 的占位符。 -const __target = function noop() {} -const __handler: ProxyHandler = { - get(_t, prop) { - if (prop === '__esModule') return true - if (prop === 'default') return new Proxy(__target, __handler) - if (prop === Symbol.toPrimitive) return () => undefined - if (prop === Symbol.iterator) return function* () {} - if (prop === Symbol.asyncIterator) return async function* () {} - if (prop === 'then') return undefined - return new Proxy(__target, __handler) +import { readFile } from 'fs/promises' +import { z } from 'zod/v4' +import { buildTool, type ToolDef } from '../../Tool.js' +import { generateTaskId } from '../../Task.js' +import { lazySchema } from '../../utils/lazySchema.js' +import { getRuleByContentsForTool } from '../../utils/permissions/permissions.js' +import { hasAcceptedWorkflowsInAutoMode } from '../../utils/workflows/autoModeConsent.js' +import { jsonStringify } from '../../utils/slowOperations.js' +import { findWorkflowByName, loadWorkflows } from '../../utils/workflows/discovery.js' +import { + areWorkflowsEnabled, + describeWorkflowsDisabled, + getWorkflowsDisabledReason, +} from '../../utils/workflows/enabled.js' +import { WORKFLOW_SCRIPT_MAX_BYTES } from '../../utils/workflows/constants.js' +import { createWorkflowRunId } from '../../utils/workflows/paths.js' +import { prepareWorkflowScript } from '../../utils/workflows/runtime.js' +import { launchWorkflow } from './launchWorkflow.js' +import { WORKFLOW_TOOL_NAME } from './constants.js' +import { getWorkflowToolPrompt } from './prompt.js' + +const inputSchema = lazySchema(() => + z.strictObject({ + script: z + .string() + .max(WORKFLOW_SCRIPT_MAX_BYTES) + .optional() + .describe( + 'Self-contained workflow script. Must begin with `export const meta = { name, description, phases }` ' + + '(pure literal, no computed values) followed by the script body using agent()/parallel()/pipeline()/phase().', + ), + scriptPath: z + .string() + .optional() + .describe( + 'Path to a workflow script file on disk. Every Workflow invocation persists its script under the ' + + 'session directory and returns the path in the tool result. Takes precedence over `script` and `name`.', + ), + name: z + .string() + .optional() + .describe( + 'Name of a predefined workflow (built-in or from .claude/workflows/).', + ), + args: z + .unknown() + .optional() + .describe( + 'Optional input value exposed to the script as the global `args`, verbatim. Pass arrays/objects as ' + + 'actual JSON values, NOT as a JSON-encoded string.', + ), + title: z.string().optional().describe('Ignored — set the title in `meta`.'), + description: z + .string() + .optional() + .describe('Ignored — set the description in `meta`.'), + resumeFromRunId: z + .string() + .regex(/^wf_[a-z0-9-]{6,}$/) + .optional() + .describe( + 'Run ID of a prior Workflow invocation to resume from. Completed agent() calls with unchanged ' + + '(prompt, opts) return their cached results instantly; only edited or new calls re-run.', + ), + }), +) +type InputSchema = ReturnType + +// Mirrors the official WorkflowOutput in @anthropic-ai/claude-code's +// sdk-tools.d.ts. Optional fields are optional there too, so a transcript +// written before a field existed still replays without re-validation failing. +const outputSchema = lazySchema(() => + z.object({ + status: z.enum(['async_launched', 'remote_launched']), + taskId: z.string(), + taskType: z.enum(['local_workflow', 'remote_agent']).optional(), + workflowName: z.string().optional(), + runId: z.string().optional(), + summary: z.string().optional(), + transcriptDir: z.string().optional(), + scriptPath: z.string().optional(), + sessionUrl: z.string().optional(), + warning: z.string().optional(), + error: z.string().optional(), + }), +) +type OutputSchema = ReturnType +export type Output = z.infer + +export const WorkflowTool = buildTool({ + name: WORKFLOW_TOOL_NAME, + searchHint: 'orchestrate many subagents from a script', + maxResultSizeChars: 100_000, + userFacingName: () => 'Workflow', + get inputSchema(): InputSchema { + return inputSchema() }, - apply() { - return new Proxy(__target, __handler) + get outputSchema(): OutputSchema { + return outputSchema() }, - construct() { - return new Proxy(__target, __handler) + isEnabled() { + return areWorkflowsEnabled() }, + isConcurrencySafe() { + return false + }, + isReadOnly() { + return false + }, + isOpenWorld() { + return true + }, + toAutoClassifierInput(input) { + return input.script ?? input.scriptPath ?? input.name ?? '' + }, + /** + * A run spawns many agents whose file edits are auto-approved, so the launch + * itself is the only place the user gets to say no. `bypassPermissions` and + * non-interactive runs have nobody to ask; everywhere else prompts unless the + * user has allow-listed this workflow name. + */ + async checkPermissions(input, context) { + const appState = context?.getAppState() + const permissionContext = appState?.toolPermissionContext + if ( + permissionContext?.mode === 'bypassPermissions' || + context?.options.isNonInteractiveSession + ) { + return { behavior: 'allow', updatedInput: input } + } + // Ultracode is a standing instruction to orchestrate every task; prompting + // per run would mean prompting every turn. + if (appState?.ultracode === true) { + return { behavior: 'allow', updatedInput: input } + } + // Auto mode asks once per machine, then remembers. + if (permissionContext?.mode === 'auto' && hasAcceptedWorkflowsInAutoMode()) { + return { behavior: 'allow', updatedInput: input } + } + + const ruleContent = typeof input.name === 'string' ? input.name : undefined + if (ruleContent && permissionContext) { + const denied = getRuleByContentsForTool( + permissionContext, + WorkflowTool, + 'deny', + ).get(ruleContent) + if (denied) { + return { + behavior: 'deny', + message: `${WORKFLOW_TOOL_NAME} denied for workflow "${ruleContent}".`, + decisionReason: { type: 'rule', rule: denied }, + } + } + const allowed = getRuleByContentsForTool( + permissionContext, + WorkflowTool, + 'allow', + ).get(ruleContent) + if (allowed) { + return { + behavior: 'allow', + updatedInput: input, + decisionReason: { type: 'rule', rule: allowed }, + } + } + } + + return { + behavior: 'ask', + message: + 'Claude wants to run a dynamic workflow, which can spawn many subagents and use a large number of tokens.', + ...(ruleContent + ? { + suggestions: [ + { + type: 'addRules' as const, + rules: [{ toolName: WORKFLOW_TOOL_NAME, ruleContent }], + behavior: 'allow' as const, + destination: 'localSettings' as const, + }, + ], + } + : {}), + } + }, + async description(input) { + if (input.name) return `Run the ${input.name} workflow` + return 'Run a dynamic workflow' + }, + async prompt() { + return getWorkflowToolPrompt() + }, + mapToolResultToToolResultBlockParam(output, toolUseID) { + return { + tool_use_id: toolUseID, + type: 'tool_result', + content: jsonStringify(output), + } + }, + async call(input, toolUseContext, canUseTool) { + const disabled = getWorkflowsDisabledReason() + if (disabled) throw new Error(describeWorkflowsDisabled(disabled)) + + const resolved = await resolveScript(input) + if ('error' in resolved) throw new Error(resolved.error) + + const prepared = prepareWorkflowScript(resolved.script) + if (!prepared.ok) throw new Error(prepared.error) + + const workflowRunId = input.resumeFromRunId ?? createWorkflowRunId() + const taskId = generateTaskId('local_workflow') + + const launched = launchWorkflow({ + taskId, + workflowRunId, + script: resolved.script, + scriptPath: resolved.scriptPath, + args: input.args, + meta: prepared.meta, + vmScript: prepared.vmScript, + toolUseContext, + canUseTool, + toolUseId: toolUseContext.toolUseId, + isResume: input.resumeFromRunId !== undefined, + }) + + return { + data: { + status: 'async_launched' as const, + taskId, + taskType: 'local_workflow' as const, + workflowName: prepared.meta.name, + runId: workflowRunId, + summary: prepared.meta.description, + transcriptDir: launched.transcriptDir, + scriptPath: launched.scriptPath, + }, + } + }, +} satisfies ToolDef) + +/** + * Work out which script this call should run. + * + * `scriptPath` wins so an edited run can be relaunched byte-for-byte, then a + * saved `name`, then an inline `script`. Resolving by name here (rather than + * making the model paste the script back) is what lets `/deep-research` and + * saved workflows be one-line calls. + */ +export async function resolveScriptForTesting(input: { + script?: string + scriptPath?: string + name?: string +}): Promise<{ script: string; scriptPath?: string } | { error: string }> { + return resolveScript(input) +} + +async function resolveScript(input: { + script?: string + scriptPath?: string + name?: string +}): Promise<{ script: string; scriptPath?: string } | { error: string }> { + if (input.scriptPath) { + try { + const script = await readFile(input.scriptPath, 'utf8') + return { script, scriptPath: input.scriptPath } + } catch (error) { + return { + error: `Failed to read workflow script file ${input.scriptPath}: ${ + error instanceof Error ? error.message : String(error) + }`, + } + } + } + + if (input.name) { + const workflow = await findWorkflowByName(input.name) + if (!workflow) { + const available = (await loadWorkflows()) + .map(entry => entry.name) + .join(', ') + return { + error: `Unknown workflow '${input.name}'.${available ? ` Available: ${available}` : ''}`, + } + } + // Deliberately no scriptPath: the run gets its own session copy. Pointing + // it at the saved workflow would make the run write back over the user's + // file, and would resume from whatever that file says later rather than + // from what actually ran. + return { script: workflow.script } + } + + if (input.script) return { script: input.script } + + return { + error: 'Workflow requires one of `script`, `scriptPath`, or `name`.', + } } -const stub: any = new Proxy(__target, __handler) -export default stub -export const __stubMissing = true -// 兼容常见的命名导出 —— 没列在这里的也会通过 default Proxy 兜底 -export const createCachedMCState = stub -export const isCachedMicrocompactEnabled = stub -export const isModelSupportedForCacheEditing = stub -export const getCachedMCConfig = stub -export const markToolsSentToAPI = stub -export const resetCachedMCState = stub -export const checkProtectedNamespace = stub -export const getCoordinatorUserContext = stub diff --git a/src/tools/WorkflowTool/bundled/index.ts b/src/tools/WorkflowTool/bundled/index.ts deleted file mode 100644 index c170c49a..00000000 --- a/src/tools/WorkflowTool/bundled/index.ts +++ /dev/null @@ -1,34 +0,0 @@ -// @generated stub from scan-missing-imports -// 该文件自动生成,对应 ant-internal 的 feature() gated 模块。 -// 所有外部 build 的代码路径在 DCE 后都不会真的执行这里的代码,这只是 -// bun build resolver 的占位符。 -const __target = function noop() {} -const __handler: ProxyHandler = { - get(_t, prop) { - if (prop === '__esModule') return true - if (prop === 'default') return new Proxy(__target, __handler) - if (prop === Symbol.toPrimitive) return () => undefined - if (prop === Symbol.iterator) return function* () {} - if (prop === Symbol.asyncIterator) return async function* () {} - if (prop === 'then') return undefined - return new Proxy(__target, __handler) - }, - apply() { - return new Proxy(__target, __handler) - }, - construct() { - return new Proxy(__target, __handler) - }, -} -const stub: any = new Proxy(__target, __handler) -export default stub -export const __stubMissing = true -// 兼容常见的命名导出 —— 没列在这里的也会通过 default Proxy 兜底 -export const createCachedMCState = stub -export const isCachedMicrocompactEnabled = stub -export const isModelSupportedForCacheEditing = stub -export const getCachedMCConfig = stub -export const markToolsSentToAPI = stub -export const resetCachedMCState = stub -export const checkProtectedNamespace = stub -export const getCoordinatorUserContext = stub diff --git a/src/tools/WorkflowTool/constants.ts b/src/tools/WorkflowTool/constants.ts index 444a03a0..a575596b 100644 --- a/src/tools/WorkflowTool/constants.ts +++ b/src/tools/WorkflowTool/constants.ts @@ -1 +1 @@ -export const WORKFLOW_TOOL_NAME = 'workflow' +export const WORKFLOW_TOOL_NAME = 'Workflow' diff --git a/src/tools/WorkflowTool/createWorkflowCommand.ts b/src/tools/WorkflowTool/createWorkflowCommand.ts index c170c49a..814aa9cb 100644 --- a/src/tools/WorkflowTool/createWorkflowCommand.ts +++ b/src/tools/WorkflowTool/createWorkflowCommand.ts @@ -1,34 +1,67 @@ -// @generated stub from scan-missing-imports -// 该文件自动生成,对应 ant-internal 的 feature() gated 模块。 -// 所有外部 build 的代码路径在 DCE 后都不会真的执行这里的代码,这只是 -// bun build resolver 的占位符。 -const __target = function noop() {} -const __handler: ProxyHandler = { - get(_t, prop) { - if (prop === '__esModule') return true - if (prop === 'default') return new Proxy(__target, __handler) - if (prop === Symbol.toPrimitive) return () => undefined - if (prop === Symbol.iterator) return function* () {} - if (prop === Symbol.asyncIterator) return async function* () {} - if (prop === 'then') return undefined - return new Proxy(__target, __handler) - }, - apply() { - return new Proxy(__target, __handler) - }, - construct() { - return new Proxy(__target, __handler) - }, +import type { ContentBlockParam } from '@anthropic-ai/sdk/resources/index.mjs' +import type { Command } from '../../types/command.js' +import { loadWorkflows } from '../../utils/workflows/discovery.js' +import { areWorkflowsEnabled } from '../../utils/workflows/enabled.js' +import type { WorkflowDefinition } from '../../utils/workflows/types.js' +import { WORKFLOW_TOOL_NAME } from './constants.js' + +/** + * Turn every discovered workflow into a `/` command. + * + * The command is a prompt, not a direct tool call: whatever the user typed + * after the name has to become the `args` value, and only the model can decide + * whether "issues 1024, 1025" is a list of numbers or a sentence. The prompt + * pins everything else so the model's only job is that conversion. + */ +export async function getWorkflowCommands(cwd?: string): Promise { + if (!areWorkflowsEnabled()) return [] + const workflows = await loadWorkflows(cwd) + return workflows.map(workflow => createWorkflowCommand(workflow)) +} + +export function createWorkflowCommand(workflow: WorkflowDefinition): Command { + return { + type: 'prompt', + name: workflow.name, + description: workflow.description, + progressMessage: `running the ${workflow.name} workflow`, + contentLength: workflow.description.length + 200, + argNames: ['args'], + source: workflowCommandSource(workflow), + isEnabled: () => areWorkflowsEnabled(), + userFacingName: () => workflow.name, + async getPromptForCommand(args: string): Promise { + const argsLine = + args.trim() === '' + ? 'Pass no `args`.' + : `Convert the following invocation text into the \`args\` value and pass it: ${JSON.stringify(args.trim())}. ` + + 'If it is a list of items, pass a real JSON array, not a string.' + return [ + { + type: 'text', + text: + `Run the saved workflow "${workflow.name}" by calling ${WORKFLOW_TOOL_NAME} exactly once with ` + + `{ name: "${workflow.name}" }. ${argsLine}\n` + + 'Do not write a new script and do not do the work yourself — the saved workflow already ' + + 'contains the orchestration. After the tool returns, end your turn; the result arrives as a ' + + 'task notification.', + }, + ] + }, + } as Command +} + +function workflowCommandSource( + workflow: WorkflowDefinition, +): 'builtin' | 'userSettings' | 'projectSettings' | 'plugin' { + switch (workflow.source) { + case 'built-in': + return 'builtin' + case 'plugin': + return 'plugin' + case 'projectSettings': + return 'projectSettings' + default: + return 'userSettings' + } } -const stub: any = new Proxy(__target, __handler) -export default stub -export const __stubMissing = true -// 兼容常见的命名导出 —— 没列在这里的也会通过 default Proxy 兜底 -export const createCachedMCState = stub -export const isCachedMicrocompactEnabled = stub -export const isModelSupportedForCacheEditing = stub -export const getCachedMCConfig = stub -export const markToolsSentToAPI = stub -export const resetCachedMCState = stub -export const checkProtectedNamespace = stub -export const getCoordinatorUserContext = stub diff --git a/src/tools/WorkflowTool/launchWorkflow.ts b/src/tools/WorkflowTool/launchWorkflow.ts new file mode 100644 index 00000000..f758aab2 --- /dev/null +++ b/src/tools/WorkflowTool/launchWorkflow.ts @@ -0,0 +1,365 @@ +import { mkdir, writeFile } from 'fs/promises' +import { dirname } from 'path' +import type vm from 'vm' +import { + getCurrentTurnTokenBudget, + getTurnOutputTokens, +} from '../../bootstrap/state.js' +import type { CanUseToolFn } from '../../hooks/useCanUseTool.js' +import type { SetAppState } from '../../Task.js' +import type { ToolUseContext } from '../../Tool.js' +import { + completeWorkflowTask, + enqueueWorkflowNotification, + failWorkflowTask, + registerWorkflowTask, + updateWorkflowProgressBatch, + type LocalWorkflowTaskState, +} from '../../tasks/LocalWorkflowTask/LocalWorkflowTask.js' +import { logForDebugging } from '../../utils/debug.js' +import { registerTask, updateTaskState } from '../../utils/task/framework.js' +import { emitTaskProgress } from '../../utils/task/sdkProgress.js' +import { + WORKFLOW_PANEL_EMIT_INTERVAL_MS, + WORKFLOW_PROGRESS_BATCH_MS, +} from '../../utils/workflows/constants.js' +import { createWorkflowSharedCounters } from '../../utils/workflows/harness.js' +import { createRunJournal } from '../../utils/workflows/journal.js' +import { + getWorkflowTranscriptDir, + getWorkflowScriptPath, +} from '../../utils/workflows/paths.js' +import { executeWorkflowScript } from '../../utils/workflows/runtime.js' +import { createNestedWorkflowRunner } from './runNestedWorkflow.js' +import { + isDurableWorkflowEvent, + type WorkflowMeta, + type WorkflowProgressEvent, +} from '../../utils/workflows/types.js' + +export type LaunchWorkflowParams = { + taskId: string + workflowRunId: string + script: string + scriptPath?: string + args?: unknown + meta: WorkflowMeta + vmScript: vm.Script + toolUseContext: ToolUseContext + canUseTool: CanUseToolFn + toolUseId?: string + isResume: boolean + runNestedWorkflow?: (nameOrRef: unknown, args: unknown) => Promise +} + +export type LaunchedWorkflow = { + task: LocalWorkflowTaskState + transcriptDir: string + scriptPath: string +} + +/** + * Register the background task for a run and start executing its script. + * + * Returns as soon as the task exists so the tool call can hand the model a run + * id immediately; everything after that happens on the detached promise below. + */ +export function launchWorkflow(params: LaunchWorkflowParams): LaunchedWorkflow { + const { + taskId, + workflowRunId, + script, + args, + meta, + vmScript, + toolUseContext, + canUseTool, + toolUseId, + isResume, + } = params + + const setAppState: SetAppState = + toolUseContext.setAppStateForTasks ?? toolUseContext.setAppState + const transcriptDir = getWorkflowTranscriptDir(workflowRunId) + // A caller-supplied path is read-only: it is the user's file, and a resume + // is supposed to pick up their edits, not overwrite them. + const callerSuppliedPath = params.scriptPath !== undefined + const scriptPath = + params.scriptPath ?? getWorkflowScriptPath(workflowRunId, meta.name) + + // A resumed run replaces the settled task for the same run id, so the + // progress view shows one entry instead of a stack of dead ones. + if (isResume) { + const tasks = toolUseContext.getAppState().tasks + for (const [id, task] of Object.entries(tasks)) { + if ( + task.type === 'local_workflow' && + task.workflowRunId === workflowRunId && + task.status !== 'running' + ) { + setAppState(prev => { + const { [id]: _removed, ...rest } = prev.tasks + return { ...prev, tasks: rest } + }) + } + } + } + + const task = registerWorkflowTask({ + taskId, + script, + scriptPath, + args, + summary: meta.description, + workflowName: meta.name, + title: meta.title, + phases: meta.phases, + defaultModel: toolUseContext.options.mainLoopModel, + workflowRunId, + ownerAgentId: toolUseContext.agentId, + toolUseId, + }) + registerTask(task, setAppState) + + if (!callerSuppliedPath) void persistScript(scriptPath, script) + + const budgetTotal = getCurrentTurnTokenBudget() + const spentBeforeRun = getTurnOutputTokens() + const tokenBudget = { + total: budgetTotal, + getTurnSpent: () => getTurnOutputTokens() - spentBeforeRun, + } + + const batcher = createProgressBatcher({ + taskId, + toolUseId, + setAppState, + getTask: () => + toolUseContext.getAppState().tasks[taskId] as + | LocalWorkflowTaskState + | undefined, + fallbackDescription: task.description, + startTime: task.startTime, + summary: meta.description, + }) + + void (async () => { + const journal = createRunJournal(workflowRunId) + const journalSnapshot = isResume ? await journal.load() : undefined + const shared = createWorkflowSharedCounters() + const nestedContext = { + ...toolUseContext, + abortController: task.abortController ?? toolUseContext.abortController, + } + const onAgentController = ( + agentKey: string, + controller: AbortController | undefined, + ) => { + updateTaskState(taskId, setAppState, current => { + if (!current.agentControllers) return current + if (controller) current.agentControllers.set(agentKey, controller) + else current.agentControllers.delete(agentKey) + return current + }) + } + + const outcome = await executeWorkflowScript({ + vmScript, + toolUseContext: nestedContext, + canUseTool, + runId: workflowRunId, + workflowName: meta.name, + args, + seedPhaseTitles: meta.phases?.map(phase => phase.title), + tokenBudget, + journal, + journalSnapshot, + shared, + onProgress: event => batcher.push(event), + onAgentController, + runNestedWorkflow: + params.runNestedWorkflow ?? + createNestedWorkflowRunner({ + toolUseContext: nestedContext, + canUseTool, + runId: workflowRunId, + tokenBudget, + journal, + shared, + onProgress: event => batcher.push(event), + onAgentController, + }), + }) + + batcher.flush() + + const settled = toolUseContext.getAppState().tasks[taskId] as + | LocalWorkflowTaskState + | undefined + // A run the user stopped is already terminal; do not overwrite that. + if (settled && settled.status !== 'running') return + + const totalTokens = settled?.totalTokens ?? 0 + const totalToolCalls = settled?.totalToolCalls ?? 0 + const status = outcome.error ? 'failed' : 'completed' + + if (outcome.error) { + failWorkflowTask( + taskId, + outcome.error, + outcome.agentCount, + outcome.logs, + setAppState, + ) + } else { + completeWorkflowTask( + taskId, + outcome.result, + outcome.agentCount, + outcome.logs, + setAppState, + ) + } + + enqueueWorkflowNotification({ + taskId, + summary: meta.description, + status, + result: outcome.error ? undefined : outcome.result, + error: outcome.error, + failures: outcome.failures, + agentCount: outcome.agentCount, + totalTokens, + totalToolCalls, + durationMs: outcome.durationMs, + toolUseId, + transcriptDir, + scriptPath, + workflowRunId, + args, + setAppState, + }) + })().catch(error => { + batcher.cancel() + const message = error instanceof Error ? error.message : String(error) + logForDebugging(`Workflow ${workflowRunId} crashed: ${message}`) + failWorkflowTask(taskId, message, 0, [], setAppState) + enqueueWorkflowNotification({ + taskId, + summary: meta.description, + status: 'failed', + error: message, + agentCount: 0, + totalTokens: 0, + totalToolCalls: 0, + durationMs: Date.now() - task.startTime, + toolUseId, + transcriptDir, + scriptPath, + workflowRunId, + args, + setAppState, + }) + }) + + return { task, transcriptDir, scriptPath } +} + +/** + * Coalesce progress events before they touch AppState. + * + * A 40-agent run emits thousands of token-count updates; applying each one + * would re-render the whole task list per event. Batching on a short timer + * keeps the view live without making the UI the bottleneck. + */ +function createProgressBatcher(params: { + taskId: string + toolUseId?: string + setAppState: SetAppState + getTask: () => LocalWorkflowTaskState | undefined + fallbackDescription: string + startTime: number + summary?: string +}) { + let pending: WorkflowProgressEvent[] = [] + let timer: ReturnType | undefined + let lastPanelEmit = 0 + + const drain = (): void => { + timer = undefined + if (pending.length === 0) return + const batch = pending + pending = [] + updateWorkflowProgressBatch(params.taskId, batch, params.setAppState) + emitPanelProgress(batch) + } + + const emitPanelProgress = (batch: WorkflowProgressEvent[]): void => { + const durable = batch.filter(isDurableWorkflowEvent) + if (durable.length === 0) return + const task = params.getTask() + if (task?.type !== 'local_workflow' || task.status !== 'running') return + + const lastAgent = [...durable] + .reverse() + .find(event => event.type === 'workflow_agent') + // Token-count churn is throttled; anything else (an agent finishing, a new + // phase) goes out immediately so the panel is never visibly behind. + const onlyTokenChurn = durable.every( + event => event.type === 'workflow_agent' && event.state === 'progress', + ) + const now = Date.now() + if (onlyTokenChurn && now - lastPanelEmit < WORKFLOW_PANEL_EMIT_INTERVAL_MS) { + return + } + lastPanelEmit = now + + emitTaskProgress({ + taskId: params.taskId, + toolUseId: params.toolUseId, + description: + lastAgent?.type === 'workflow_agent' + ? lastAgent.phaseTitle + ? `${lastAgent.phaseTitle}: ${lastAgent.label}` + : lastAgent.label + : params.fallbackDescription, + startTime: params.startTime, + totalTokens: task.totalTokens, + toolUses: task.totalToolCalls, + lastToolName: + lastAgent?.type === 'workflow_agent' ? lastAgent.label : undefined, + summary: params.summary, + workflowProgress: task.workflowProgress.filter(isDurableWorkflowEvent), + }) + } + + return { + push(event: WorkflowProgressEvent): void { + pending.push(event) + if (!timer) { + timer = setTimeout(drain, WORKFLOW_PROGRESS_BATCH_MS) + timer.unref?.() + } + }, + flush(): void { + if (timer) clearTimeout(timer) + drain() + }, + cancel(): void { + if (timer) clearTimeout(timer) + timer = undefined + pending = [] + }, + } +} + +async function persistScript(path: string, script: string): Promise { + try { + await mkdir(dirname(path), { recursive: true }) + await writeFile(path, script, 'utf8') + } catch (error) { + logForDebugging( + `Failed to persist workflow script to ${path}: ${error instanceof Error ? error.message : String(error)}`, + ) + } +} diff --git a/src/tools/WorkflowTool/prompt.ts b/src/tools/WorkflowTool/prompt.ts new file mode 100644 index 00000000..df7a6741 --- /dev/null +++ b/src/tools/WorkflowTool/prompt.ts @@ -0,0 +1,161 @@ +import { + WORKFLOW_MAX_AGENTS, + WORKFLOW_MAX_FANOUT, + getWorkflowConcurrency, +} from '../../utils/workflows/constants.js' +import { describeWorkflowSizeGuideline } from '../../utils/workflows/enabled.js' + +/** + * The Workflow tool description. + * + * This is the whole product surface for the model: everything it knows about + * when to reach for a workflow, how to shape one, and which patterns produce + * trustworthy results comes from here. Sections that read as over-explained + * (the pipeline-vs-barrier argument, the loop-until-dry example) are load + * bearing — without them models default to a single `parallel()` barrier per + * stage and lose most of the wall-clock advantage. + */ +export function getWorkflowToolPrompt(): string { + return `Execute a workflow script that orchestrates multiple subagents deterministically. Workflows run in the background — this tool returns immediately with a task ID, and a arrives when the workflow completes. Use /workflows to watch live progress. + +A workflow structures work across many agents — to be comprehensive (decompose and cover in parallel), to be confident (independent perspectives and adversarial checks before committing), or to take on scale one context can't hold (migrations, audits, broad sweeps). The script is where you encode that structure: what fans out, what verifies, what synthesizes. + +ONLY call this tool when the user has explicitly opted into multi-agent orchestration. Workflows can spawn dozens of agents and consume a large amount of tokens; the user must request that scale, not have it inferred. Explicit opt-in means one of: +- The user included the keyword "ultracode" in their prompt (you'll see a system-reminder confirming it). +- Ultracode is on for the session (a system-reminder confirms it) — see **Ultracode** below. +- The user directly asked you to run a workflow or use multi-agent orchestration in their own words ("use a workflow", "run a workflow", "fan out agents", "orchestrate this with subagents"). The ask must be in the user's words — a task that would merely benefit from a workflow does not count. +- The user invoked a skill or slash command whose instructions tell you to call Workflow. +- The user asked you to run a specific named or saved workflow. + +For any other task — even one that would clearly benefit from parallelism — do NOT call this tool. Use the Agent tool for individual subagents, or briefly describe what a multi-agent workflow could do and how much it would roughly cost, and ask the user whether to run it. + +**Ultracode.** When a system-reminder confirms ultracode is on, that opt-in is standing: author and run a workflow for every substantive task by default. The goal is the most exhaustive, correct answer you can produce — token cost is not a constraint. For multi-phase work (understand → design → implement → review), that often means several workflows in sequence — one per phase — so you stay in the loop between them. The quality patterns below (adversarial verify, multi-modal sweep, completeness critic, loop-until-dry) are the tools; pick what fits the task. Lean toward orchestrating with workflows and adversarially verifying your findings — unless the work is trivial or already verified. Solo only on conversational turns or trivial mechanical edits. When a reminder says ultracode is off, revert to the opt-in rule above. + +When you do call it, the right move is often **hybrid**: scout inline first (list the files, find the channels, scope the diff) to discover the work-list, then call Workflow to pipeline over it. You don't need to know the shape before the *task* — only before the *orchestration step*. + +Common single-phase workflows you can chain across turns: +- **Understand** — parallel readers over relevant subsystems → structured map +- **Design** — judge panel of N independent approaches → scored synthesis +- **Review** — dimensions → find → adversarially verify (example below) +- **Research** — multi-modal sweep → deep-read → synthesize +- **Migrate** — discover sites → transform each (worktree isolation) → verify + +For larger work, run several in sequence — read each result before deciding the next phase. You stay in the loop; each workflow is one well-scoped fan-out. + +Pass the script inline via \`script\` — do not Write it to a file first. Every invocation automatically persists its script to a file under the session directory and returns the path in the tool result. To iterate on a workflow, edit that file with Write/Edit and re-invoke Workflow with \`{scriptPath: ""}\` instead of resending the full script. + +Every script must begin with \`export const meta = {...}\`: + export const meta = { + name: 'find-flaky-tests', + description: 'Find flaky tests and propose fixes', // one-line, shown in permission dialog + phases: [ // one entry per phase() call + { title: 'Scan', detail: 'grep test logs for retries' }, + { title: 'Fix', detail: 'one agent per flaky test' }, + ], + } + // script body starts here — use agent()/parallel()/pipeline()/phase()/log() + phase('Scan') + const flaky = await agent('grep CI logs for retry markers', {schema: FLAKY_SCHEMA}) + ... + +The \`meta\` object must be a PURE LITERAL — no variables, function calls, spreads, or template interpolation. Required fields: \`name\`, \`description\`. Optional: \`whenToUse\` (shown in the workflow list), \`phases\`. Use the SAME phase titles in meta.phases as in phase() calls — titles are matched exactly; a phase() call with no matching meta entry just gets its own progress group. Add \`model\` to a phase entry when that phase uses a specific model override. + +Script body hooks: +- agent(prompt: string, opts?: {label?: string, phase?: string, schema?: object, model?: string, effort?: string, isolation?: 'worktree', agentType?: string}): Promise — spawn a subagent. Without schema, returns its final text as a string. With schema (a JSON Schema), the subagent is forced to call a StructuredOutput tool and agent() returns the validated object — no parsing needed. Returns null if the user skips the agent mid-run or the subagent dies on a terminal API error after retries (filter with .filter(Boolean)). opts.label overrides the display label. opts.phase explicitly assigns this agent to a progress group (use this inside pipeline()/parallel() stages to avoid races on the global phase() state — same phase string → same group box). opts.model overrides the model for this agent call. Default to omitting it — the agent inherits the main-loop model, which is almost always correct. opts.effort overrides the reasoning effort for this agent call. opts.isolation: 'worktree' runs the agent in a fresh git worktree — EXPENSIVE, use ONLY when agents mutate files in parallel and would otherwise conflict; the worktree is auto-removed if unchanged. opts.agentType uses a custom subagent type (e.g. 'general-purpose', 'code-reviewer') instead of the default workflow subagent; composes with schema. +- pipeline(items, stage1, stage2, ...): Promise — run each item through all stages independently, NO barrier between stages. Item A can be in stage 3 while item B is still in stage 1. This is the DEFAULT for multi-stage work. Wall-clock = slowest single-item chain, not sum-of-slowest-per-stage. Every stage callback receives (prevResult, originalItem, index). A stage that throws drops that item to \`null\` and skips its remaining stages. +- parallel(thunks: Array<() => Promise>): Promise — run tasks concurrently. This is a BARRIER: awaits all thunks before returning. A thunk that throws resolves to \`null\` in the result array — the call itself never rejects, so \`.filter(Boolean)\` before using the results. Use ONLY when you genuinely need all results together. +- log(message: string): void — emit a progress message to the user +- phase(title: string): void — start a new phase; subsequent agent() calls are grouped under this title in the progress display +- args: any — the value passed as Workflow's \`args\` input, verbatim (undefined if not provided). Pass arrays/objects as actual JSON values in the tool call, NOT as a JSON-encoded string. +- budget: {total: number|null, spent(): number, remaining(): number} — the turn's token target. \`budget.total\` is null if no target was set. The target is a HARD ceiling: once \`spent()\` reaches \`total\`, further \`agent()\` calls throw. +- workflow(nameOrRef: string | {scriptPath: string}, args?: any): Promise — run a saved workflow inline as a sub-step and return whatever it returns. The child shares this run's concurrency cap, agent counter, abort signal, and token budget. Nesting is one level only. + +Subagents are told their final text IS the return value (not a human-facing message), so they return raw data. For structured output, use the schema option — validation happens at the tool-call layer so the model retries on mismatch. + +Scripts are plain JavaScript, NOT TypeScript — type annotations (\`: string[]\`), interfaces, and generics fail to parse. The script body runs in an async context — use await directly. Standard JS built-ins (JSON, Math, Array, etc.) are available — EXCEPT \`Date.now()\`/\`Math.random()\`/argless \`new Date()\`, which throw (they would break resume); pass timestamps in via \`args\`, stamp results after the workflow returns, and for randomness vary the agent prompt/label by index. No filesystem or Node.js API access. + +DEFAULT TO pipeline(). Only reach for a barrier (parallel between stages) when you genuinely need ALL prior-stage results together. + +A barrier is correct ONLY when stage N needs cross-item context from all of stage N-1: +- Dedup/merge across the full result set before expensive downstream work +- Early-exit if the total count is zero ("0 bugs found → skip verification entirely") +- Stage N's prompt references "the other findings" for comparison + +A barrier is NOT justified by: +- "I need to flatten/map/filter first" — do it inside a pipeline stage: pipeline(items, stageA, r => transform([r]).flat(), stageB) +- "The stages are conceptually separate" — that's what pipeline() models. Separate stages ≠ synchronized stages. +- "It's cleaner code" — barrier latency is real. + +The canonical multi-stage pattern — pipeline by default, each dimension verifies as soon as its review completes: + export const meta = { + name: 'review-changes', + description: 'Review changed files across dimensions, verify each finding', + phases: [{ title: 'Review' }, { title: 'Verify' }], + } + const DIMENSIONS = [{key: 'bugs', prompt: '...'}, {key: 'perf', prompt: '...'}] + const results = await pipeline( + DIMENSIONS, + d => agent(d.prompt, {label: \`review:\${d.key}\`, phase: 'Review', schema: FINDINGS_SCHEMA}), + review => parallel(review.findings.map(f => () => + agent(\`Adversarially verify: \${f.title}\`, {label: \`verify:\${f.file}\`, phase: 'Verify', schema: VERDICT_SCHEMA}) + .then(v => ({...f, verdict: v})) + )) + ) + const confirmed = results.flat().filter(Boolean).filter(f => f.verdict?.isReal) + return { confirmed } + +When a barrier IS correct — dedup across all findings before expensive verification: + const all = await parallel(DIMENSIONS.map(d => () => agent(d.prompt, {schema: FINDINGS_SCHEMA}))) + const deduped = dedupeByFileAndLine(all.filter(Boolean).flatMap(r => r.findings)) + const verified = await parallel(deduped.map(f => () => agent(verifyPrompt(f), {schema: VERDICT_SCHEMA}))) + +Loop-until-count pattern — accumulate to a target: + const bugs = [] + while (bugs.length < 10) { + const result = await agent("Find bugs in this codebase.", {schema: BUGS_SCHEMA}) + bugs.push(...result.bugs) + log(\`\${bugs.length}/10 found\`) + } + +Loop-until-budget pattern — guard on budget.total: with no target set, remaining() is Infinity and the loop would run straight to the agent cap. + while (budget.total && budget.remaining() > 50_000) { ... } + +Composing patterns — exhaustive review (find → dedup vs seen → diverse-lens panel → loop-until-dry): + const seen = new Set(), confirmed = [] + let dry = 0 + while (dry < 2) { + const found = (await parallel(FINDERS.map(f => () => + agent(f.prompt, {phase: 'Find', schema: BUGS})))).filter(Boolean).flatMap(r => r.bugs) + const fresh = found.filter(b => !seen.has(key(b))) + if (!fresh.length) { dry++; continue } + dry = 0; fresh.forEach(b => seen.add(key(b))) + const judged = await parallel(fresh.map(b => () => + parallel(['correctness','security','repro'].map(lens => () => + agent(\`Judge "\${b.desc}" via the \${lens} lens — real?\`, {phase: 'Verify', schema: VERDICT}))) + .then(vs => ({ b, real: vs.filter(Boolean).filter(v => v.real).length >= 2 })))) + confirmed.push(...judged.filter(v => v.real).map(v => v.b)) + } + return confirmed + // dedup vs \`seen\`, NOT \`confirmed\` — else judge-rejected findings reappear every round and it never converges. + +Quality patterns — common shapes; pick by task and compose freely: +- Adversarial verify: spawn N independent skeptics per finding, each prompted to REFUTE. Kill if ≥majority refute. +- Perspective-diverse verify: give each verifier a distinct lens (correctness, security, perf, does-it-reproduce) instead of N identical refuters. +- Judge panel: generate N independent attempts from different angles, score with parallel judges, synthesize from the winner while grafting the best ideas from runners-up. +- Loop-until-dry: for unknown-size discovery, keep spawning finders until K consecutive rounds return nothing new. +- Multi-modal sweep: parallel agents each searching a different way (by-container, by-content, by-entity, by-time). +- Completeness critic: a final agent that asks "what's missing — modality not run, claim unverified, source unread?" +- No silent caps: if a workflow bounds coverage (top-N, no-retry, sampling), \`log()\` what was dropped. + +Scale to what the user asked for. "find any bugs" → a few finders, single-vote verify. "thoroughly audit this" or "be comprehensive" → larger finder pool, 3–5 vote adversarial pass, synthesis stage. + +Use this tool for multi-step orchestration where control flow should be deterministic (loops, conditionals, fan-out) rather than model-driven. + +## Resume + +The tool result includes a runId. To resume after a pause, kill, or script edit, relaunch with Workflow({scriptPath, resumeFromRunId}) — the longest unchanged prefix of agent() calls returns cached results instantly; the first edited/new call and everything after it runs live. Same script + same args → 100% cache hit. Before diagnosing why a completed workflow returned an empty or unexpected result, Read /journal.jsonl — it records each agent's actual return value. + +Concurrent agent() calls are capped at ${getWorkflowConcurrency()} per workflow — excess calls queue and run as slots free up. Total agent count across a workflow's lifetime is capped at ${WORKFLOW_MAX_AGENTS}. A single parallel()/pipeline() call accepts at most ${WORKFLOW_MAX_FANOUT} items. + +${describeWorkflowSizeGuideline()}` +} diff --git a/src/tools/WorkflowTool/runNestedWorkflow.ts b/src/tools/WorkflowTool/runNestedWorkflow.ts new file mode 100644 index 00000000..bca5e6cd --- /dev/null +++ b/src/tools/WorkflowTool/runNestedWorkflow.ts @@ -0,0 +1,100 @@ +import { readFile } from 'fs/promises' +import type { CanUseToolFn } from '../../hooks/useCanUseTool.js' +import type { ToolUseContext } from '../../Tool.js' +import { findWorkflowByName } from '../../utils/workflows/discovery.js' +import type { WorkflowSharedCounters } from '../../utils/workflows/harness.js' +import type { WorkflowJournal } from '../../utils/workflows/journal.js' +import { + executeWorkflowScript, + prepareWorkflowScript, +} from '../../utils/workflows/runtime.js' +import type { + WorkflowProgressEvent, + WorkflowTokenBudget, +} from '../../utils/workflows/types.js' + +export type NestedWorkflowDeps = { + toolUseContext: ToolUseContext + canUseTool: CanUseToolFn + runId: string + tokenBudget?: WorkflowTokenBudget + journal?: WorkflowJournal + shared: WorkflowSharedCounters + onProgress: (event: WorkflowProgressEvent) => void + onAgentController: ( + agentKey: string, + controller: AbortController | undefined, + ) => void +} + +/** + * Build the `workflow(nameOrRef, args)` global. + * + * A child runs inside the parent's run: same run id, same journal, same + * concurrency pool and agent-index sequence (so its progress rows sit + * alongside the parent's instead of overwriting them), and the same token + * budget. Nesting is one level — a child's own `workflow()` throws, which is + * what keeps the agent cap meaningful. + */ +export function createNestedWorkflowRunner( + deps: NestedWorkflowDeps, +): (nameOrRef: unknown, args: unknown) => Promise { + return async (nameOrRef, args) => { + const script = await resolveNestedScript(nameOrRef) + const prepared = prepareWorkflowScript(script) + if (!prepared.ok) { + throw new Error(`workflow(): ${prepared.error}`) + } + + deps.onProgress({ + type: 'workflow_log', + message: `workflow(${prepared.meta.name}) started`, + }) + + const outcome = await executeWorkflowScript({ + vmScript: prepared.vmScript, + toolUseContext: deps.toolUseContext, + canUseTool: deps.canUseTool, + runId: deps.runId, + args, + seedPhaseTitles: prepared.meta.phases?.map(phase => phase.title), + tokenBudget: deps.tokenBudget, + journal: deps.journal, + onProgress: deps.onProgress, + onAgentController: deps.onAgentController, + shared: deps.shared, + // One level only: the child gets no `workflow()` of its own. + runNestedWorkflow: undefined, + }) + + if (outcome.error) { + throw new Error(`workflow(${prepared.meta.name}): ${outcome.error}`) + } + return outcome.result + } +} + +async function resolveNestedScript(nameOrRef: unknown): Promise { + if (typeof nameOrRef === 'string') { + const workflow = await findWorkflowByName(nameOrRef) + if (!workflow) throw new Error(`workflow(): unknown workflow '${nameOrRef}'`) + return workflow.script + } + if (nameOrRef !== null && typeof nameOrRef === 'object') { + const scriptPath = (nameOrRef as { scriptPath?: unknown }).scriptPath + if (typeof scriptPath === 'string') { + try { + return await readFile(scriptPath, 'utf8') + } catch (error) { + throw new Error( + `workflow(): failed to read ${scriptPath}: ${ + error instanceof Error ? error.message : String(error) + }`, + ) + } + } + } + throw new Error( + "workflow() expects a saved workflow name or { scriptPath: '...' }", + ) +} diff --git a/src/types/plugin.ts b/src/types/plugin.ts index e398314c..4127e906 100644 --- a/src/types/plugin.ts +++ b/src/types/plugin.ts @@ -63,6 +63,8 @@ export type LoadedPlugin = { skillsPaths?: string[] // Additional skill paths from manifest outputStylesPath?: string outputStylesPaths?: string[] // Additional output style paths from manifest + workflowsPath?: string + workflowsPaths?: string[] // Additional dynamic-workflow paths from manifest hooksConfig?: HooksSettings mcpServers?: Record lspServers?: Record diff --git a/src/utils/attachments.ts b/src/utils/attachments.ts index 953e0a32..6759e34a 100644 --- a/src/utils/attachments.ts +++ b/src/utils/attachments.ts @@ -200,6 +200,11 @@ const sessionTranscriptModule = feature('KAIROS') : null /* eslint-enable @typescript-eslint/no-require-imports */ import { hasUltrathinkKeyword, isUltrathinkEnabled } from './thinking.js' +import { + getWorkflowSizeGuideline, + isWorkflowKeywordTriggerEnabled, +} from './workflows/enabled.js' +import { hasWorkflowKeyword } from './workflows/keyword.js' import { tokenCountFromLastAPIResponse, tokenCountWithEstimation, @@ -674,6 +679,21 @@ export type Attachment = type: 'ultrathink_effort' level: 'high' } + | { + type: 'workflow_keyword_request' + } + | { + type: 'ultra_effort_enter' + /** `full` on the first turn of a session; `short` on every turn after. */ + reminderType: 'full' | 'short' + } + | { + type: 'ultra_effort_exit' + } + | { + type: 'workflow_size_guideline_change' + size: string + } | { type: 'deferred_tools_delta' addedNames: string[] @@ -825,6 +845,25 @@ export async function getAttachments( maybe('ultrathink_effort', () => Promise.resolve(getUltrathinkEffortAttachment(input)), ), + maybe('workflow_keyword_request', () => + Promise.resolve( + getWorkflowKeywordAttachment( + input, + toolUseContext.getAppState().suppressWorkflowKeyword === true, + ), + ), + ), + maybe('ultra_effort_enter', () => + Promise.resolve( + getUltracodeEffortAttachments( + messages, + toolUseContext.getAppState().ultracode === true, + ), + ), + ), + maybe('workflow_size_guideline_change', () => + Promise.resolve(getWorkflowSizeGuidelineAttachment(messages)), + ), maybe('deferred_tools_delta', () => Promise.resolve( getDeferredToolsDeltaAttachment( @@ -1431,6 +1470,99 @@ export function getDateChangeAttachments( return [{ type: 'date_change', newDate: currentDate }] } +/** + * Opt this turn into orchestration when the user typed `ultracode`. + * + * Only a prompt the human actually typed counts — a webhook payload or a + * relayed PR comment containing the word must not spend a hundred agents. + */ +function getWorkflowKeywordAttachment( + input: string | null, + suppressed: boolean, +): Attachment[] { + if (suppressed) return [] + if (!input || !isWorkflowKeywordTriggerEnabled()) return [] + if (!hasWorkflowKeyword(input)) return [] + logEvent('tengu_workflow_keyword', {}) + return [{ type: 'workflow_keyword_request' }] +} + +/** + * Announce ultracode turning on or off, once per transition. + * + * The model needs a standing instruction while it is on, but repeating the + * full paragraph every turn is pure token cost — so the transition gets the + * full text and later turns get a one-line reminder. + */ +function getUltracodeEffortAttachments( + messages: Message[] | undefined, + ultracodeActive: boolean, +): Attachment[] { + let lastState: 'enter' | 'exit' | 'none' = 'none' + for (let i = (messages?.length ?? 0) - 1; i >= 0; i--) { + const message = messages?.[i] + if (message?.type !== 'attachment') continue + if (message.attachment.type === 'ultra_effort_enter') { + lastState = 'enter' + break + } + if (message.attachment.type === 'ultra_effort_exit') { + lastState = 'exit' + break + } + } + + if (ultracodeActive) { + return [ + { + type: 'ultra_effort_enter', + reminderType: lastState === 'enter' ? 'short' : 'full', + }, + ] + } + return lastState === 'enter' ? [{ type: 'ultra_effort_exit' }] : [] +} + +/** + * Tell the model when the size guideline changed mid-session. + * + * The guideline is baked into the Workflow tool prompt, which the model may + * have cached from an earlier turn — without this, changing it in `/config` + * has no visible effect until the next session. + */ +function getWorkflowSizeGuidelineAttachment( + messages: Message[] | undefined, +): Attachment[] { + const current = getWorkflowSizeGuideline() + for (let i = (messages?.length ?? 0) - 1; i >= 0; i--) { + const message = messages?.[i] + if (message?.type !== 'attachment') continue + if (message.attachment.type !== 'workflow_size_guideline_change') continue + return message.attachment.size === current + ? [] + : [{ type: 'workflow_size_guideline_change', size: current }] + } + // Nothing announced yet: the tool prompt already carries the guideline on + // the first turn, so only a later change is worth a reminder. + return [] +} + +/** + * Test seam for the two workflow attachment producers. + * + * Both are pure given their inputs, but they sit behind `getAttachments()`, + * which needs a full ToolUseContext and a live session. Exposing them keeps + * the tests on the real functions instead of a re-implementation. + */ +export const getAttachmentsForTesting = { + workflowKeyword: (input: string | null, opts: { suppressed: boolean }) => + getWorkflowKeywordAttachment(input, opts.suppressed), + ultracodeEffort: (messages: Message[] | undefined, ultracodeActive: boolean) => + getUltracodeEffortAttachments(messages, ultracodeActive), + workflowSizeGuideline: (messages: Message[] | undefined) => + getWorkflowSizeGuidelineAttachment(messages), +} + function getUltrathinkEffortAttachment(input: string | null): Attachment[] { if (!isUltrathinkEnabled() || !input || !hasUltrathinkKeyword(input)) { return [] diff --git a/src/utils/config.ts b/src/utils/config.ts index 8c6741ed..8f275aec 100644 --- a/src/utils/config.ts +++ b/src/utils/config.ts @@ -383,6 +383,15 @@ export type GlobalConfig = { // Terminal progress bar configuration (OSC 9;4) terminalProgressBarEnabled: boolean + // How many subagents Claude should aim for when writing a dynamic workflow. + // Set from /config; a settings-file value overrides it and hides the row. + workflowSizeGuideline?: 'unrestricted' | 'small' | 'medium' | 'large' + + // Recorded the first time the user approves a workflow in auto permission + // mode. Auto mode is already "stop asking me about routine things", so the + // launch prompt asks once and then stays out of the way. + hasAcceptedWorkflowsInAutoMode?: boolean + // Terminal tab status indicator (OSC 21337). When on, emits a colored // dot + status text to the tab sidebar and drops the spinner prefix // from the title (the dot makes it redundant). @@ -647,6 +656,8 @@ export const GLOBAL_CONFIG_KEYS = [ 'autoInstallIdeExtension', 'fileCheckpointingEnabled', 'terminalProgressBarEnabled', + 'workflowSizeGuideline', + 'hasAcceptedWorkflowsInAutoMode', 'showStatusInTerminalTab', 'taskCompleteNotifEnabled', 'inputNeededNotifEnabled', diff --git a/src/utils/messages.ts b/src/utils/messages.ts index 2c1691ad..86102baa 100644 --- a/src/utils/messages.ts +++ b/src/utils/messages.ts @@ -1,3 +1,10 @@ +import { describeWorkflowSizeGuideline } from './workflows/enabled.js' +import { + ULTRACODE_ENTER_REMINDER, + ULTRACODE_EXIT_REMINDER, + ULTRACODE_STILL_ON_REMINDER, + WORKFLOW_KEYWORD_REMINDER, +} from './workflows/ultracode.js' import { feature } from 'bun:bundle' import type { BetaUsage as Usage } from '@anthropic-ai/sdk/resources/beta/messages/messages.mjs' import type { @@ -4319,6 +4326,43 @@ You have exited auto mode. The user may now want to interact more directly. You }), ]) } + case 'workflow_keyword_request': { + return wrapMessagesInSystemReminder([ + createUserMessage({ + content: WORKFLOW_KEYWORD_REMINDER, + isMeta: true, + }), + ]) + } + case 'ultra_effort_enter': { + return wrapMessagesInSystemReminder([ + createUserMessage({ + content: + attachment.reminderType === 'full' + ? ULTRACODE_ENTER_REMINDER + : ULTRACODE_STILL_ON_REMINDER, + isMeta: true, + }), + ]) + } + case 'ultra_effort_exit': { + return wrapMessagesInSystemReminder([ + createUserMessage({ + content: ULTRACODE_EXIT_REMINDER, + isMeta: true, + }), + ]) + } + case 'workflow_size_guideline_change': { + return wrapMessagesInSystemReminder([ + createUserMessage({ + content: describeWorkflowSizeGuideline( + attachment.size as Parameters[0], + ), + isMeta: true, + }), + ]) + } case 'deferred_tools_delta': { const parts: string[] = [] if (attachment.addedLines.length > 0) { diff --git a/src/utils/permissions/classifierDecision.ts b/src/utils/permissions/classifierDecision.ts index d329ebaf..7481233a 100644 --- a/src/utils/permissions/classifierDecision.ts +++ b/src/utils/permissions/classifierDecision.ts @@ -40,11 +40,9 @@ const VERIFY_PLAN_EXECUTION_TOOL_NAME = require('../../tools/VerifyPlanExecutionTool/constants.js') as typeof import('../../tools/VerifyPlanExecutionTool/constants.js') ).VERIFY_PLAN_EXECUTION_TOOL_NAME : null -const WORKFLOW_TOOL_NAME = feature('WORKFLOW_SCRIPTS') - ? ( - require('../../tools/WorkflowTool/constants.js') as typeof import('../../tools/WorkflowTool/constants.js') - ).WORKFLOW_TOOL_NAME - : null +const WORKFLOW_TOOL_NAME = ( + require('../../tools/WorkflowTool/constants.js') as typeof import('../../tools/WorkflowTool/constants.js') +).WORKFLOW_TOOL_NAME /* eslint-enable @typescript-eslint/no-require-imports */ /** diff --git a/src/utils/plugins/pluginLoader.ts b/src/utils/plugins/pluginLoader.ts index 0c25de14..b5f0e984 100644 --- a/src/utils/plugins/pluginLoader.ts +++ b/src/utils/plugins/pluginLoader.ts @@ -1591,6 +1591,31 @@ export async function createPluginFromPath( plugin.outputStylesPath = outputStylesPath } + // Step 4c-2: Register dynamic workflows shipped by the plugin. Namespaced by + // plugin name at command-build time, so two plugins can both ship `review`. + const workflowsPath = join(pluginPath, 'workflows') + if (await pathExists(workflowsPath)) { + plugin.workflowsPath = workflowsPath + } + if (manifest.workflows) { + const declared = Array.isArray(manifest.workflows) + ? manifest.workflows + : [manifest.workflows] + const validPaths = await validatePluginPaths( + declared, + pluginPath, + manifest.name, + source, + 'workflows', + 'Workflow', + 'specified in manifest but', + errors, + ) + if (validPaths.length > 0) { + plugin.workflowsPaths = validPaths + } + } + // Step 4e: Process additional output style paths from manifest if (manifest.outputStyles) { const outputStylePaths = Array.isArray(manifest.outputStyles) diff --git a/src/utils/plugins/schemas.ts b/src/utils/plugins/schemas.ts index b92fb80e..2bebc769 100644 --- a/src/utils/plugins/schemas.ts +++ b/src/utils/plugins/schemas.ts @@ -523,6 +523,29 @@ const PluginManifestOutputStylesSchema = lazySchema(() => }), ) +/** + * Schema for additional dynamic-workflow paths in a plugin manifest. + * + * Plugins ship workflows in `workflows/` at the plugin root; this field points + * at extra directories or files, matching how commands/agents/skills work. + */ +const PluginManifestWorkflowsSchema = lazySchema(() => + z.object({ + workflows: z.union([ + RelativePath().describe( + 'Path to additional workflows directory or file (in addition to those in the workflows/ directory, if it exists), relative to the plugin root', + ), + z + .array( + RelativePath().describe( + 'Path to additional workflows directory or file, relative to the plugin root', + ), + ) + .describe('List of paths to additional workflows directories or files'), + ]), + }), +) + // Helper validators for LSP config const nonEmptyString = lazySchema(() => z.string().min(1)) const fileExtension = lazySchema(() => @@ -889,6 +912,7 @@ export const PluginManifestSchema = lazySchema(() => ...PluginManifestAgentsSchema().partial().shape, ...PluginManifestSkillsSchema().partial().shape, ...PluginManifestOutputStylesSchema().partial().shape, + ...PluginManifestWorkflowsSchema().partial().shape, ...PluginManifestChannelsSchema().partial().shape, ...PluginManifestMcpServerSchema().partial().shape, ...PluginManifestLspServerSchema().partial().shape, diff --git a/src/utils/sessionStorage.ts b/src/utils/sessionStorage.ts index 3a46b3d2..a5a4826f 100644 --- a/src/utils/sessionStorage.ts +++ b/src/utils/sessionStorage.ts @@ -278,6 +278,23 @@ export type AgentMetadata = { * agent's activity and completion under the wrong card. Optional — * older metadata files lack this field. */ toolUseId?: string + /** + * Which workflow run and phase this agent belongs to. + * + * A workflow's shape — phases, and which agents ran in each — exists only in + * the live progress stream, so reopening a finished session had nothing to + * rebuild it from. This sidecar is written before the agent starts and + * outlives the process, which makes it the record. Absent for ordinary + * subagents and for workflow runs from before this was added. + */ + workflow?: { + runId: string + name: string + phaseIndex: number + phaseTitle?: string + /** The runtime's sequential agent number within the run. */ + agentIndex: number + } } /** diff --git a/src/utils/settings/types.ts b/src/utils/settings/types.ts index 0a5fa08b..d04a017a 100644 --- a/src/utils/settings/types.ts +++ b/src/utils/settings/types.ts @@ -460,6 +460,37 @@ export const SettingsSchema = lazySchema(() => .boolean() .optional() .describe('Disable all hooks and statusLine execution'), + // Dynamic workflows: the Workflow tool, /workflows, and saved workflow + // commands. Honored in managed settings as an org-wide kill switch. + disableWorkflows: z + .boolean() + .optional() + .describe( + 'Disable dynamic workflows: the Workflow tool, bundled workflow commands, ' + + 'and saved workflows in .claude/workflows.', + ), + enableWorkflows: z + .boolean() + .optional() + .describe( + 'Explicitly enable dynamic workflows. Only consulted when they are not ' + + 'disabled; `disableWorkflows` and CLAUDE_CODE_DISABLE_WORKFLOWS win.', + ), + workflowKeywordTriggerEnabled: z + .boolean() + .optional() + .describe( + 'Enable the "ultracode" keyword trigger: including the keyword in a prompt ' + + 'opts that turn into the Workflow tool. Set to false to disable the trigger. Default: true.', + ), + workflowSizeGuideline: z + .enum(['unrestricted', 'small', 'medium', 'large']) + .optional() + .describe( + 'How many subagents Claude should aim for when writing a dynamic workflow. ' + + 'small < 5, medium < 15, large < 50, unrestricted lets Claude size it to the task. ' + + 'Advice to the model, not a runtime cap.', + ), // Opt out of the cross-client `.agents/skills` convention (agentskills.io), // which is scanned alongside `.claude/skills` by default. disableAgentSkillsDirectory: z diff --git a/src/utils/task/framework.ts b/src/utils/task/framework.ts index 8413db2e..f0f72ebc 100644 --- a/src/utils/task/framework.ts +++ b/src/utils/task/framework.ts @@ -311,6 +311,8 @@ function getStatusText(status: TaskStatus): string { return 'was stopped' case 'running': return 'is running' + case 'paused': + return 'is paused' case 'pending': return 'is pending' } diff --git a/src/utils/workflows/autoModeConsent.ts b/src/utils/workflows/autoModeConsent.ts new file mode 100644 index 00000000..227c3148 --- /dev/null +++ b/src/utils/workflows/autoModeConsent.ts @@ -0,0 +1,30 @@ +import { getGlobalConfig, saveGlobalConfig } from '../config.js' + +/** + * Whether the user has already approved running workflows in auto mode. + * + * Auto mode means "stop asking me about routine things". Prompting on every + * workflow launch there would defeat the mode, so the launch prompt appears + * once and the answer is remembered for the machine. + */ +export function hasAcceptedWorkflowsInAutoMode(): boolean { + try { + return getGlobalConfig().hasAcceptedWorkflowsInAutoMode === true + } catch { + // Config access can be closed early in the process lifetime; a missing + // consent record just means we ask. + return false + } +} + +export function recordWorkflowAutoModeConsent(): void { + try { + if (getGlobalConfig().hasAcceptedWorkflowsInAutoMode === true) return + saveGlobalConfig(config => ({ + ...config, + hasAcceptedWorkflowsInAutoMode: true, + })) + } catch { + // Failing to persist consent only costs one extra prompt next time. + } +} diff --git a/src/utils/workflows/bundled/deepResearch.ts b/src/utils/workflows/bundled/deepResearch.ts new file mode 100644 index 00000000..e5cf75d9 --- /dev/null +++ b/src/utils/workflows/bundled/deepResearch.ts @@ -0,0 +1,172 @@ +/** + * `/deep-research` — the bundled workflow. + * + * Investigates one question across many sources: fan out searches on distinct + * angles, deep-read what they surface, then have independent verifiers vote on + * every claim before it reaches the report. The voting stage is the point — + * a single research pass reports whatever the first plausible page said. + */ +export const DEEP_RESEARCH_SCRIPT = `export const meta = { + name: 'deep-research', + description: 'Research a question across many sources and return a cited, cross-checked report', + whenToUse: 'Use for questions that need several independent sources weighed against each other.', + phases: [ + { title: 'Survey', detail: 'search the question from several angles' }, + { title: 'Read', detail: 'deep-read the most promising sources' }, + { title: 'Verify', detail: 'independent verifiers vote on each claim' }, + { title: 'Report', detail: 'synthesize a cited report' }, + ], +} + +const question = typeof args === 'string' ? args : (args && args.question) || '' +if (!question) return { error: 'deep-research needs a question. Run /deep-research .' } + +const ANGLE_SCHEMA = { + type: 'object', + required: ['angles'], + properties: { + angles: { + type: 'array', + items: { + type: 'object', + required: ['angle', 'query'], + properties: { angle: { type: 'string' }, query: { type: 'string' } }, + }, + }, + }, +} + +const SOURCE_SCHEMA = { + type: 'object', + required: ['sources'], + properties: { + sources: { + type: 'array', + items: { + type: 'object', + required: ['url', 'title', 'why'], + properties: { url: { type: 'string' }, title: { type: 'string' }, why: { type: 'string' } }, + }, + }, + }, +} + +const CLAIM_SCHEMA = { + type: 'object', + required: ['claims'], + properties: { + claims: { + type: 'array', + items: { + type: 'object', + required: ['claim', 'url'], + properties: { + claim: { type: 'string' }, + url: { type: 'string' }, + quote: { type: 'string' }, + }, + }, + }, + }, +} + +const VERDICT_SCHEMA = { + type: 'object', + required: ['verdict', 'reason'], + properties: { + verdict: { type: 'string', enum: ['supported', 'refuted', 'unverifiable'] }, + reason: { type: 'string' }, + }, +} + +phase('Survey') +const plan = await agent( + 'Break this research question into 4 distinct search angles that would surface DIFFERENT sources, ' + + 'not rephrasings of each other. Question: ' + question, + { label: 'plan angles', phase: 'Survey', schema: ANGLE_SCHEMA }, +) + +const angles = (plan && plan.angles ? plan.angles : []).slice(0, 4) +if (angles.length === 0) return { error: 'could not decompose the question into search angles' } +log(angles.length + ' angles: ' + angles.map(a => a.angle).join(', ')) + +const perAngle = await pipeline( + angles, + angle => + agent( + 'Use WebSearch to find the best sources for this angle on "' + question + '".\\n' + + 'Angle: ' + angle.angle + '\\nSuggested query: ' + angle.query + '\\n' + + 'Return the 3 most credible, most specific sources. Prefer primary sources over summaries.', + { label: angle.angle, phase: 'Survey', schema: SOURCE_SCHEMA }, + ), + found => + parallel( + (found && found.sources ? found.sources : []).slice(0, 3).map(source => () => + agent( + 'WebFetch ' + source.url + ' and extract every claim in it that bears on: ' + question + '\\n' + + 'Quote the sentence each claim comes from. Do not infer beyond the text. ' + + 'If the page does not load or does not address the question, return an empty claims array.', + { label: source.title, phase: 'Read', schema: CLAIM_SCHEMA }, + ), + ), + ), +) + +const claims = [] +const seen = new Set() +for (const batch of perAngle.flat().filter(Boolean)) { + for (const claim of batch.claims || []) { + const key = (claim.claim || '').toLowerCase().replace(/\\s+/g, ' ').trim() + if (!key || seen.has(key)) continue + seen.add(key) + claims.push(claim) + } +} +log(claims.length + ' distinct claims to verify') +if (claims.length === 0) return { question, claims: [], unverified: [], note: 'no claims found' } + +phase('Verify') +const judged = await pipeline(claims, (claim, _original, index) => + parallel( + ['does an independent source confirm this', 'does any source contradict this'].map( + (lens, lensIndex) => () => + agent( + 'Verify this claim independently of where it came from.\\n' + + 'Claim: ' + claim.claim + '\\nOriginally from: ' + claim.url + '\\n' + + 'Lens: ' + lens + '\\n' + + 'Search for and read at least one source OTHER than the original. ' + + 'Answer "unverifiable" if you cannot reach a source — do not guess, and do not ' + + 'treat a failed fetch as a refutation.', + { label: 'claim ' + (index + 1) + '.' + (lensIndex + 1), phase: 'Verify', schema: VERDICT_SCHEMA }, + ), + ), + ).then(votes => ({ claim, votes: votes.filter(Boolean) })), +) + +const supported = [] +const unverified = [] +for (const entry of judged.filter(Boolean)) { + const votes = entry.votes + const refuted = votes.filter(v => v.verdict === 'refuted').length + const confirms = votes.filter(v => v.verdict === 'supported').length + if (refuted > confirms) continue + if (confirms === 0) { + unverified.push({ claim: entry.claim.claim, url: entry.claim.url, reason: (votes[0] && votes[0].reason) || 'no verifier reached a source' }) + continue + } + supported.push({ claim: entry.claim.claim, url: entry.claim.url, quote: entry.claim.quote }) +} + +phase('Report') +const report = await agent( + 'Write a cited report answering: ' + question + '\\n\\n' + + 'Use ONLY these cross-checked claims, citing the URL after each statement:\\n' + + JSON.stringify(supported, null, 2) + + '\\n\\nThese claims could not be verified — list them at the end under "Unverified", do not treat them as fact:\\n' + + JSON.stringify(unverified, null, 2) + + '\\n\\nBe direct. Lead with the answer. Markdown, no preamble.', + { label: 'synthesize', phase: 'Report' }, +) + +return { question, report, supported, unverified } +` diff --git a/src/utils/workflows/bundled/index.ts b/src/utils/workflows/bundled/index.ts new file mode 100644 index 00000000..1b7af2d1 --- /dev/null +++ b/src/utils/workflows/bundled/index.ts @@ -0,0 +1,39 @@ +import { parseWorkflowScript } from '../meta.js' +import type { WorkflowDefinition } from '../types.js' +import { DEEP_RESEARCH_SCRIPT } from './deepResearch.js' + +const BUNDLED_SCRIPTS = [DEEP_RESEARCH_SCRIPT] + +let cached: WorkflowDefinition[] | null = null + +/** + * Workflows that ship with the CLI. + * + * Their metadata is parsed from the same source the runtime executes, so a + * bundled script whose `meta` drifts out of shape fails the parser here rather + * than at run time. + */ +export function getBundledWorkflows(): WorkflowDefinition[] { + if (cached) return cached + const definitions: WorkflowDefinition[] = [] + for (const script of BUNDLED_SCRIPTS) { + const parsed = parseWorkflowScript(script) + if ('error' in parsed) { + throw new Error(`Bundled workflow failed to parse: ${parsed.error}`) + } + definitions.push({ + source: 'built-in', + name: parsed.meta.name, + description: parsed.meta.description, + whenToUse: parsed.meta.whenToUse, + phases: parsed.meta.phases, + script, + }) + } + cached = definitions + return definitions +} + +export function resetBundledWorkflowsForTesting(): void { + cached = null +} diff --git a/src/utils/workflows/compile.ts b/src/utils/workflows/compile.ts new file mode 100644 index 00000000..98bc6cf6 --- /dev/null +++ b/src/utils/workflows/compile.ts @@ -0,0 +1,74 @@ +import vm from 'vm' +import { + WORKFLOW_DATE_BANNED_MESSAGE, + WORKFLOW_IMPORT_BANNED_MESSAGE, + WORKFLOW_RANDOM_BANNED_MESSAGE, +} from './constants.js' + +export type WorkflowCompileResult = + | { ok: true; vmScript: vm.Script } + | { ok: false; error: string } + +/** + * Remove the two sources of nondeterminism that would break resume. + * + * A resumed run replays cached agent results in start order. A script whose + * control flow depends on wall-clock time or randomness would take a different + * branch on replay and silently consume the wrong cache entries, so both are + * made to throw at the first call rather than shimmed into something plausible. + * + * Installed on the context (not prepended to the script) so `globalThis.Date` + * and an aliased `const r = Math.random` are covered too. + */ +export function installDeterminismGuards(context: vm.Context): void { + vm.runInContext( + `(() => { + const dateMessage = ${JSON.stringify(WORKFLOW_DATE_BANNED_MESSAGE)} + const randomMessage = ${JSON.stringify(WORKFLOW_RANDOM_BANNED_MESSAGE)} + const RealDate = Date + globalThis.Date = new Proxy(RealDate, { + construct(target, argv, newTarget) { + if (argv.length === 0) throw new Error(dateMessage) + return Reflect.construct(target, argv, newTarget === undefined ? target : newTarget) + }, + get(target, prop, receiver) { + if (prop === 'now') throw new Error(dateMessage) + return Reflect.get(target, prop, receiver) + }, + }) + Math.random = () => { + throw new Error(randomMessage) + } + })()`, + context, + { filename: 'workflow-guards.js' }, + ) +} + +/** + * Compile a script body into a `vm.Script` that evaluates to a promise. + * + * The body runs inside an async IIFE so a bare top-level `return` resolves the + * run and `await` works without the script declaring anything. Dynamic + * `import()` is refused at the module-resolution hook rather than by a lint + * pass, so there is no spelling of it that gets through. + */ +export function compileWorkflowScript( + scriptBody: string, +): WorkflowCompileResult { + const source = `(async () => {\n${scriptBody}\n})()` + try { + const vmScript = new vm.Script(source, { + filename: 'workflow.js', + importModuleDynamically: () => { + throw new Error(WORKFLOW_IMPORT_BANNED_MESSAGE) + }, + }) + return { ok: true, vmScript } + } catch (error) { + return { + ok: false, + error: `SyntaxError: ${error instanceof Error ? error.message : String(error)}`, + } + } +} diff --git a/src/utils/workflows/constants.ts b/src/utils/workflows/constants.ts new file mode 100644 index 00000000..5054f241 --- /dev/null +++ b/src/utils/workflows/constants.ts @@ -0,0 +1,94 @@ +import { cpus } from 'os' + +/** Hard cap on `agent()` calls per run — a backstop against runaway loops. */ +export const WORKFLOW_MAX_AGENTS = 1000 + +/** A single parallel()/pipeline() call may not fan out wider than this. */ +export const WORKFLOW_MAX_FANOUT = 4096 + +/** Scripts larger than this are rejected before parsing (512 KiB). */ +export const WORKFLOW_SCRIPT_MAX_BYTES = 524_288 + +/** + * Wall-clock budget for the *synchronous* portion of the script. Awaiting an + * agent does not consume it; a `while (true) {}` with no await does. + */ +export const WORKFLOW_SYNC_TIMEOUT_MS = 30_000 + +/** An agent that emits no progress for this long is reported as stalled. */ +export const WORKFLOW_AGENT_STALL_MS = 180_000 + +/** Prompt/result previews shown in the progress view are clipped to this. */ +export const WORKFLOW_PREVIEW_MAX_CHARS = 400 + +/** Labels are clipped to this when derived from the prompt. */ +export const WORKFLOW_LABEL_MAX_CHARS = 60 + +/** Upper bound on log lines carried in the final result. */ +export const WORKFLOW_MAX_COLLECTED_LOGS = 1000 + +/** Progress rows retained in task state; logs beyond this are dropped first. */ +export const WORKFLOW_MAX_PROGRESS_ROWS = 500 + +/** Progress events are coalesced on this interval before touching AppState. */ +export const WORKFLOW_PROGRESS_BATCH_MS = 16 + +/** Minimum spacing between task-panel progress emissions. */ +export const WORKFLOW_PANEL_EMIT_INTERVAL_MS = 10_000 + +/** Agent type used for every subagent a workflow script spawns. */ +export const WORKFLOW_SUBAGENT_TYPE = 'workflow-subagent' + +/** Run ids look like `wf_<8 hex>-<3 hex>`; agent ids append `-`. */ +export const WORKFLOW_RUN_ID_PATTERN = /^wf_[a-z0-9-]{6,}$/ + +/** + * Concurrency ceiling for in-flight agents. Bounded by CPU count so a fan-out + * of 500 items does not spawn 500 model streams on a laptop. + */ +export function getWorkflowConcurrency( + cpuCount: number = cpus().length, +): number { + return Math.min(16, Math.max(2, cpuCount - 2)) +} + +export const WORKFLOW_DATE_BANNED_MESSAGE = + 'Date.now() / new Date() are unavailable in workflow scripts (breaks resume). ' + + 'Stamp results after the workflow returns, or pass timestamps via args.' + +export const WORKFLOW_RANDOM_BANNED_MESSAGE = + 'Math.random() is unavailable in workflow scripts (breaks resume). ' + + 'For N independent samples, include the index in the agent label or prompt.' + +export const WORKFLOW_IMPORT_BANNED_MESSAGE = + 'import() is not available in workflow scripts.' + +export const WORKFLOW_AGENT_CAP_MESSAGE = + `Workflow agent() call cap reached (${WORKFLOW_MAX_AGENTS}). This usually means a loop using ` + + 'budget.remaining() never terminates because no token budget was set — remaining() returns ' + + 'Infinity when budget.total is null. Add a hard iteration cap to the loop, or pass a token budget.' + +/** System prompt for a workflow subagent that returns free text. */ +export const WORKFLOW_SUBAGENT_PROMPT = + 'You are a subagent spawned by a workflow orchestration script. Use the tools available to complete the task.\n' + + 'NOTE: You are running inside a workflow script. Your final text response is returned verbatim as a string ' + + 'to the calling script — it is your return value, not a message to a human. Output the literal result; do not ' + + 'output confirmations like "Done." Be concise — the script will parse your output.' + +/** System prompt for a workflow subagent forced through StructuredOutput. */ +export function workflowStructuredSubagentPrompt(toolName: string): string { + return ( + 'You are a subagent spawned by a workflow orchestration script. Use the tools available to complete the task.\n' + + `- After calling ${toolName} successfully, end your turn. No acknowledgment needed.` + ) +} + +/** Appended to a schema-bearing agent prompt so the model uses the tool. */ +export function workflowStructuredOutputNote(toolName: string): string { + return ( + `\nNOTE: You are running inside a workflow script. You MUST return your final answer by calling the ${toolName} ` + + "tool exactly once — the tool's input schema defines the required shape. Do your work, then call " + + `${toolName}; do NOT put your answer in a text response (the script reads ONLY the tool call). ` + + `If validation fails, read the error and call ${toolName} again with a corrected shape.` + ) +} diff --git a/src/utils/workflows/discovery.test.ts b/src/utils/workflows/discovery.test.ts new file mode 100644 index 00000000..f2cc2eeb --- /dev/null +++ b/src/utils/workflows/discovery.test.ts @@ -0,0 +1,100 @@ +import { afterEach, beforeEach, describe, expect, test } from 'bun:test' +import { mkdir, mkdtemp, rm, writeFile } from 'fs/promises' +import { tmpdir } from 'os' +import { join } from 'path' +import { loadWorkflows } from './discovery.js' +import { getUserWorkflowsDir } from './paths.js' + +function script(name: string, description: string): string { + return `export const meta = { name: '${name}', description: '${description}' }\nreturn null\n` +} + +describe('loadWorkflows', () => { + let root: string + let configDir: string + let projectDir: string + const originalConfigDir = process.env.CLAUDE_CONFIG_DIR + + beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), 'wf-discovery-')) + configDir = join(root, 'config') + projectDir = join(root, 'project') + // Redirect the personal workflows dir away from the developer's ~/.claude. + process.env.CLAUDE_CONFIG_DIR = configDir + await mkdir(join(configDir, 'workflows'), { recursive: true }) + await mkdir(join(projectDir, '.claude', 'workflows'), { recursive: true }) + // getProjectDirsUpToHome stops at the git root; without one it walks to + // home, so mark this temp dir as a repository. + await mkdir(join(projectDir, '.git'), { recursive: true }) + }) + + afterEach(async () => { + if (originalConfigDir === undefined) delete process.env.CLAUDE_CONFIG_DIR + else process.env.CLAUDE_CONFIG_DIR = originalConfigDir + await rm(root, { recursive: true, force: true }) + }) + + test('resolves the personal workflows dir under CLAUDE_CONFIG_DIR', () => { + expect(getUserWorkflowsDir()).toBe(join(configDir, 'workflows')) + }) + + test('includes bundled, personal, and project workflows', async () => { + await writeFile( + join(configDir, 'workflows', 'mine.js'), + script('mine', 'Personal one'), + ) + await writeFile( + join(projectDir, '.claude', 'workflows', 'ours.js'), + script('ours', 'Project one'), + ) + + const found = await loadWorkflows(projectDir) + const byName = new Map(found.map(entry => [entry.name, entry])) + expect(byName.get('deep-research')?.source).toBe('built-in') + expect(byName.get('mine')?.source).toBe('userSettings') + expect(byName.get('ours')?.source).toBe('projectSettings') + }) + + test('a project workflow shadows a personal one with the same name', async () => { + await writeFile( + join(configDir, 'workflows', 'review.js'), + script('review', 'Personal review'), + ) + await writeFile( + join(projectDir, '.claude', 'workflows', 'review.js'), + script('review', 'Project review'), + ) + + const found = await loadWorkflows(projectDir) + const review = found.filter(entry => entry.name === 'review') + expect(review).toHaveLength(1) + expect(review[0]?.description).toBe('Project review') + expect(review[0]?.source).toBe('projectSettings') + }) + + test('skips files that are not .js or whose meta is invalid', async () => { + await writeFile( + join(configDir, 'workflows', 'good.js'), + script('good', 'Fine'), + ) + await writeFile( + join(configDir, 'workflows', 'broken.js'), + 'const x = 1\nexport const meta = { name: "late", description: "b" }\n', + ) + await writeFile( + join(configDir, 'workflows', 'ignored.ts'), + script('ignored', 'Wrong extension'), + ) + + const names = (await loadWorkflows(projectDir)).map(entry => entry.name) + expect(names).toContain('good') + expect(names).not.toContain('late') + expect(names).not.toContain('ignored') + }) + + test('a missing workflows directory is not an error', async () => { + await rm(join(configDir, 'workflows'), { recursive: true, force: true }) + const names = (await loadWorkflows(projectDir)).map(entry => entry.name) + expect(names).toEqual(['deep-research']) + }) +}) diff --git a/src/utils/workflows/discovery.ts b/src/utils/workflows/discovery.ts new file mode 100644 index 00000000..002bdb39 --- /dev/null +++ b/src/utils/workflows/discovery.ts @@ -0,0 +1,127 @@ +import { readdir, readFile, stat } from 'fs/promises' +import { join } from 'path' +import { getOriginalCwd } from '../../bootstrap/state.js' +import { logForDebugging } from '../debug.js' +import { getProjectDirsUpToHome } from '../markdownConfigLoader.js' +import { getBundledWorkflows } from './bundled/index.js' +import { loadPluginWorkflows } from './pluginWorkflows.js' +import { WORKFLOW_SCRIPT_MAX_BYTES } from './constants.js' +import { parseWorkflowScript } from './meta.js' +import { getUserWorkflowsDir } from './paths.js' +import type { WorkflowDefinition, WorkflowSource } from './types.js' + +/** + * Every workflow runnable as a `/` command, most specific definition winning. + * + * Precedence, lowest to highest: bundled → plugin → personal + * (`~/.claude/workflows`) → project, and within project the directory closest + * to the working directory. + * A project workflow shadowing a personal one is deliberate — it is how a repo + * pins the version of a review its contributors run. + */ +export async function loadWorkflows( + cwd: string = getOriginalCwd(), +): Promise { + const projectDirs = safeProjectDirs(cwd) + + const [plugins, personal, ...projectResults] = await Promise.all([ + loadPluginWorkflows(), + loadWorkflowsFromDir(getUserWorkflowsDir(), 'userSettings'), + ...projectDirs.map(dir => loadWorkflowsFromDir(dir, 'projectSettings')), + ]) + + const byName = new Map() + for (const workflow of getBundledWorkflows()) byName.set(workflow.name, workflow) + // Plugin workflows are namespaced `plugin:name`, so they never collide with + // a personal or project workflow and never need to lose a precedence fight. + for (const workflow of plugins) byName.set(workflow.name, workflow) + for (const workflow of personal ?? []) byName.set(workflow.name, workflow) + // projectDirs is ordered most-specific first, so apply in reverse and let + // the closest directory overwrite the ones above it. + for (let i = projectResults.length - 1; i >= 0; i--) { + for (const workflow of projectResults[i] ?? []) { + byName.set(workflow.name, workflow) + } + } + + return [...byName.values()].sort((a, b) => a.name.localeCompare(b.name)) +} + +export async function findWorkflowByName( + name: string, + cwd?: string, +): Promise { + return (await loadWorkflows(cwd)).find(workflow => workflow.name === name) +} + +/** + * Read every `.js` workflow in one directory. + * + * Files are validated by parsing their `meta` literal, so a malformed script + * is skipped with a log line rather than breaking `/` autocomplete for the + * rest of the directory. Nothing here executes the script. + */ +export async function loadWorkflowsFromDir( + dir: string, + source: WorkflowSource, + namePrefix?: string, +): Promise { + let entries: Awaited> + try { + entries = await readdir(dir, { withFileTypes: true }) + } catch { + return [] + } + + const loaded = await Promise.all( + entries.map(async entry => { + if (!entry.isFile() && !entry.isSymbolicLink()) return null + if (!entry.name.endsWith('.js')) return null + const filePath = join(dir, entry.name) + try { + const stats = await stat(filePath) + if (stats.size > WORKFLOW_SCRIPT_MAX_BYTES) { + logForDebugging( + `Workflow ${filePath} exceeds ${WORKFLOW_SCRIPT_MAX_BYTES} bytes — skipping`, + ) + return null + } + const script = await readFile(filePath, 'utf8') + const parsed = parseWorkflowScript(script) + if ('error' in parsed) { + logForDebugging( + `Workflow ${filePath} has invalid meta: ${parsed.error} — skipping`, + ) + return null + } + return { + source, + name: namePrefix ? `${namePrefix}:${parsed.meta.name}` : parsed.meta.name, + description: parsed.meta.description, + whenToUse: parsed.meta.whenToUse, + phases: parsed.meta.phases, + script, + filePath, + } satisfies WorkflowDefinition + } catch (error) { + logForDebugging( + `Workflow ${filePath} could not be read: ${error instanceof Error ? error.message : String(error)}`, + ) + return null + } + }), + ) + + return loaded.filter((entry): entry is WorkflowDefinition => entry !== null) +} + +function safeProjectDirs(cwd: string): string[] { + try { + return getProjectDirsUpToHome('workflows', cwd) + } catch (error) { + logForDebugging( + `loadWorkflows: project-dir walk failed: ${error instanceof Error ? error.message : String(error)}`, + ) + return [] + } +} diff --git a/src/utils/workflows/enabled.test.ts b/src/utils/workflows/enabled.test.ts new file mode 100644 index 00000000..b4aba2e3 --- /dev/null +++ b/src/utils/workflows/enabled.test.ts @@ -0,0 +1,209 @@ +import { afterEach, beforeEach, describe, expect, test } from 'bun:test' +import { mkdtemp, rm, writeFile } from 'fs/promises' +import { mkdirSync } from 'fs' +import { tmpdir } from 'os' +import { join } from 'path' +import { getIsInteractive, setIsInteractive } from '../../bootstrap/state.js' +import { resetSettingsCache } from '../settings/settingsCache.js' +import { + areWorkflowsEnabled, + describeWorkflowSizeGuideline, + getLargeWorkflowWarning, + getWorkflowsDisabledReason, + getWorkflowSizeGuideline, + isWorkflowKeywordTriggerEnabled, +} from './enabled.js' + +let home: string +let configDir: string +const original = { + configDir: process.env.CLAUDE_CONFIG_DIR, + disable: process.env.CLAUDE_CODE_DISABLE_WORKFLOWS, + warnAgents: process.env.CLAUDE_CODE_WORKFLOW_SIZE_WARNING_AGENTS, + warnTokens: process.env.CLAUDE_CODE_WORKFLOW_SIZE_WARNING_TOKENS, +} + +async function writeSettings(value: Record): Promise { + await writeFile(join(configDir, 'settings.json'), JSON.stringify(value), 'utf8') + resetSettingsCache() +} + +describe('workflow gates', () => { + beforeEach(async () => { + home = await mkdtemp(join(tmpdir(), 'wf-enabled-')) + configDir = join(home, 'claude') + mkdirSync(configDir, { recursive: true }) + process.env.CLAUDE_CONFIG_DIR = configDir + delete process.env.CLAUDE_CODE_DISABLE_WORKFLOWS + delete process.env.CLAUDE_CODE_WORKFLOW_SIZE_WARNING_AGENTS + delete process.env.CLAUDE_CODE_WORKFLOW_SIZE_WARNING_TOKENS + resetSettingsCache() + }) + + afterEach(async () => { + for (const [key, value] of [ + ['CLAUDE_CONFIG_DIR', original.configDir], + ['CLAUDE_CODE_DISABLE_WORKFLOWS', original.disable], + ['CLAUDE_CODE_WORKFLOW_SIZE_WARNING_AGENTS', original.warnAgents], + ['CLAUDE_CODE_WORKFLOW_SIZE_WARNING_TOKENS', original.warnTokens], + ] as const) { + if (value === undefined) delete process.env[key] + else process.env[key] = value + } + resetSettingsCache() + await rm(home, { recursive: true, force: true }) + }) + + test('enabled by default', async () => { + await writeSettings({}) + expect(areWorkflowsEnabled()).toBe(true) + expect(getWorkflowsDisabledReason()).toBeNull() + }) + + test('disableWorkflows and enableWorkflows:false both turn it off', async () => { + await writeSettings({ disableWorkflows: true }) + expect(getWorkflowsDisabledReason()).toBe('settings') + await writeSettings({ enableWorkflows: false }) + expect(getWorkflowsDisabledReason()).toBe('settings') + await writeSettings({ enableWorkflows: true }) + expect(getWorkflowsDisabledReason()).toBeNull() + }) + + test('the env var wins over settings and is reported separately', async () => { + await writeSettings({ enableWorkflows: true }) + process.env.CLAUDE_CODE_DISABLE_WORKFLOWS = '1' + expect(getWorkflowsDisabledReason()).toBe('env') + }) + + test('the keyword trigger is on by default and off when set false', async () => { + const wasInteractive = getIsInteractive() + setIsInteractive(true) + try { + await writeSettings({}) + expect(isWorkflowKeywordTriggerEnabled()).toBe(true) + await writeSettings({ workflowKeywordTriggerEnabled: false }) + expect(isWorkflowKeywordTriggerEnabled()).toBe(false) + } finally { + setIsInteractive(wasInteractive) + } + }) + + test('a non-interactive session never honours the keyword', async () => { + await writeSettings({ workflowKeywordTriggerEnabled: true }) + const wasInteractive = getIsInteractive() + setIsInteractive(false) + try { + expect(isWorkflowKeywordTriggerEnabled()).toBe(false) + } finally { + setIsInteractive(wasInteractive) + } + }) + + test('an interactive session honours it', async () => { + await writeSettings({ workflowKeywordTriggerEnabled: true }) + const wasInteractive = getIsInteractive() + setIsInteractive(true) + try { + expect(isWorkflowKeywordTriggerEnabled()).toBe(true) + } finally { + setIsInteractive(wasInteractive) + } + }) + + test('disabling workflows also disables the keyword', async () => { + const wasInteractive = getIsInteractive() + setIsInteractive(true) + try { + await writeSettings({ + disableWorkflows: true, + workflowKeywordTriggerEnabled: true, + }) + expect(isWorkflowKeywordTriggerEnabled()).toBe(false) + } finally { + setIsInteractive(wasInteractive) + } + }) + + test('a settings-file guideline wins and shows in the prompt', async () => { + await writeSettings({ workflowSizeGuideline: 'small' }) + expect(getWorkflowSizeGuideline()).toBe('small') + expect(describeWorkflowSizeGuideline()).toContain('under 5 agents') + }) + + test('the default guideline is medium', async () => { + await writeSettings({}) + expect(getWorkflowSizeGuideline()).toBe('medium') + }) +}) + +describe('getLargeWorkflowWarning', () => { + beforeEach(async () => { + home = await mkdtemp(join(tmpdir(), 'wf-warn-')) + configDir = join(home, 'claude') + mkdirSync(configDir, { recursive: true }) + process.env.CLAUDE_CONFIG_DIR = configDir + delete process.env.CLAUDE_CODE_WORKFLOW_SIZE_WARNING_AGENTS + delete process.env.CLAUDE_CODE_WORKFLOW_SIZE_WARNING_TOKENS + await writeSettings({}) + }) + + afterEach(async () => { + if (original.configDir === undefined) delete process.env.CLAUDE_CONFIG_DIR + else process.env.CLAUDE_CONFIG_DIR = original.configDir + delete process.env.CLAUDE_CODE_WORKFLOW_SIZE_WARNING_AGENTS + resetSettingsCache() + await rm(home, { recursive: true, force: true }) + }) + + const base = { startedAgents: 4, totalTokens: 40_000, ultracodeActive: false } + + test('stays quiet for an ordinary run', () => { + expect( + getLargeWorkflowWarning({ ...base, scheduledAgents: 6 }), + ).toBeUndefined() + }) + + test('fires past the default 25-agent threshold', () => { + const warning = getLargeWorkflowWarning({ ...base, scheduledAgents: 26 }) + expect(warning?.axis).toBe('agents') + expect(warning?.agentCap).toBe(25) + }) + + test('projects tokens from the agents that have run so far', () => { + // 4 agents at 500k each → 40 scheduled projects to 20M, far over the cap. + const warning = getLargeWorkflowWarning({ + scheduledAgents: 40, + startedAgents: 4, + totalTokens: 2_000_000, + ultracodeActive: false, + }) + expect(warning?.axis).toBe('both') + expect(warning?.projectedTokens).toBe(20_000_000) + }) + + test('an explicit size guideline replaces the agent threshold', async () => { + await writeSettings({ workflowSizeGuideline: 'small' }) + const warning = getLargeWorkflowWarning({ ...base, scheduledAgents: 6 }) + expect(warning?.agentCap).toBe(5) + expect(warning?.capFromGuideline).toBe(true) + }) + + test('the env override beats the guideline', async () => { + await writeSettings({ workflowSizeGuideline: 'small' }) + process.env.CLAUDE_CODE_WORKFLOW_SIZE_WARNING_AGENTS = '100' + expect( + getLargeWorkflowWarning({ ...base, scheduledAgents: 40 }), + ).toBeUndefined() + }) + + test('ultracode suppresses it — that mode already opted in to scale', () => { + expect( + getLargeWorkflowWarning({ + scheduledAgents: 500, + startedAgents: 10, + totalTokens: 9_000_000, + ultracodeActive: true, + }), + ).toBeUndefined() + }) +}) diff --git a/src/utils/workflows/enabled.ts b/src/utils/workflows/enabled.ts new file mode 100644 index 00000000..58f7db1e --- /dev/null +++ b/src/utils/workflows/enabled.ts @@ -0,0 +1,237 @@ +import { getIsNonInteractiveSession } from '../../bootstrap/state.js' +import { getGlobalConfig } from '../config.js' +import { isEnvTruthy } from '../envUtils.js' +import { + getSettings_DEPRECATED, + getSettingsForSource, +} from '../settings/settings.js' + +export const WORKFLOW_SIZE_GUIDELINES = [ + 'unrestricted', + 'small', + 'medium', + 'large', +] as const + +export type WorkflowSizeGuideline = (typeof WORKFLOW_SIZE_GUIDELINES)[number] + +export const DEFAULT_WORKFLOW_SIZE_GUIDELINE: WorkflowSizeGuideline = 'medium' + +/** Agent count each guideline asks Claude to aim for. Advice, not a cap. */ +export const WORKFLOW_SIZE_AGENT_TARGETS: Record< + WorkflowSizeGuideline, + number | null +> = { + unrestricted: null, + small: 5, + medium: 15, + large: 50, +} + +export type WorkflowsDisabledReason = 'env' | 'managed' | 'settings' + +/** Agent count that triggers the "Large workflow" advisory, absent a guideline. */ +export const WORKFLOW_LARGE_AGENT_THRESHOLD = 25 +/** Projected-token count that triggers the same advisory. */ +export const WORKFLOW_LARGE_TOKEN_THRESHOLD = 1_500_000 +/** Token estimate per not-yet-started agent, before any real usage is known. */ +export const WORKFLOW_ASSUMED_TOKENS_PER_AGENT = 70_000 + +/** + * Why dynamic workflows are unavailable, or `null` when they are available. + * + * Managed settings are checked separately from user settings so the CLI can + * say *who* turned the feature off — a user who cannot re-enable it needs to + * know it was their organisation, not a toggle they flipped. + */ +export function getWorkflowsDisabledReason(): WorkflowsDisabledReason | null { + if (isEnvTruthy(process.env.CLAUDE_CODE_DISABLE_WORKFLOWS)) return 'env' + if (getSettingsForSource('policySettings')?.disableWorkflows === true) { + return 'managed' + } + const merged = getSettings_DEPRECATED() + if (merged.disableWorkflows === true) return 'settings' + // `enableWorkflows: false` is the /config toggle's off position. It is a + // separate key from `disableWorkflows` so an org can hard-disable while a + // user's toggle stays untouched underneath. + if (merged.enableWorkflows === false) return 'settings' + return null +} + +export function areWorkflowsEnabled(): boolean { + return getWorkflowsDisabledReason() === null +} + +/** + * Whether typing `ultracode` opts a turn into orchestration. + * + * On by default; the `/config` row writes `false` to turn it off. Independent + * of whether workflows themselves are enabled — a disabled feature has no + * keyword to suppress. + */ +export function isWorkflowKeywordTriggerEnabled(): boolean { + if (!areWorkflowsEnabled()) return false + // The keyword is an opt-in only in a prompt the user typed. A `-p` run, an + // SDK caller, a scheduled task, or a relayed PR comment can all contain the + // word "ultracode" without anyone having asked for a hundred agents. + if (getIsNonInteractiveSession()) return false + return getSettings_DEPRECATED().workflowKeywordTriggerEnabled ?? true +} + +export type LargeWorkflowWarning = { + axis: 'agents' | 'tokens' | 'both' + scheduledAgents: number + totalTokens: number + projectedTokens: number + agentCap: number + tokenCap: number + /** True when the agent cap came from the size guideline, not the default. */ + capFromGuideline: boolean +} + +/** + * Advisory shown when a run grows past what the user asked for. + * + * It never pauses or limits anything — the point is that a runaway script is + * visible before it has spent the tokens, so the user can stop it from + * `/workflows`. Suppressed under ultracode: turning that on already opts in + * to large runs, so the warning would fire on every single run. + */ +export function getLargeWorkflowWarning(params: { + scheduledAgents: number + startedAgents: number + totalTokens: number + ultracodeActive: boolean +}): LargeWorkflowWarning | undefined { + if (params.ultracodeActive) return undefined + + const guideline = getWorkflowSizeGuideline() + const guidelineCap = isWorkflowSizeGuidelineExplicit() + ? WORKFLOW_SIZE_AGENT_TARGETS[guideline] ?? undefined + : undefined + const agentCap = + positiveInt(process.env.CLAUDE_CODE_WORKFLOW_SIZE_WARNING_AGENTS) ?? + guidelineCap ?? + WORKFLOW_LARGE_AGENT_THRESHOLD + const tokenCap = + positiveInt(process.env.CLAUDE_CODE_WORKFLOW_SIZE_WARNING_TOKENS) ?? + WORKFLOW_LARGE_TOKEN_THRESHOLD + + const perAgent = + params.startedAgents > 0 + ? params.totalTokens / params.startedAgents + : WORKFLOW_ASSUMED_TOKENS_PER_AGENT + const projectedTokens = Math.max( + params.totalTokens, + Math.round(perAgent * params.scheduledAgents), + ) + + const overAgents = params.scheduledAgents > agentCap + const overTokens = params.totalTokens > tokenCap || projectedTokens > tokenCap + if (!overAgents && !overTokens) return undefined + + return { + axis: overAgents && overTokens ? 'both' : overAgents ? 'agents' : 'tokens', + scheduledAgents: params.scheduledAgents, + totalTokens: params.totalTokens, + projectedTokens, + agentCap, + tokenCap, + capFromGuideline: overAgents && guidelineCap !== undefined, + } +} + +function positiveInt(raw: string | undefined): number | undefined { + if (!raw) return undefined + const value = Number.parseInt(raw, 10) + return Number.isFinite(value) && value > 0 ? value : undefined +} + +export function describeWorkflowsDisabled( + reason: WorkflowsDisabledReason, +): string { + switch (reason) { + case 'env': + return 'Dynamic workflows are disabled by CLAUDE_CODE_DISABLE_WORKFLOWS.' + case 'managed': + return 'Dynamic workflows are disabled by managed settings (`disableWorkflows`).' + case 'settings': + return 'Dynamic workflows are disabled in settings (`disableWorkflows`). Turn them back on in /config.' + } +} + +/** + * The configured size guideline. + * + * A settings file wins over the interactive `/config` value so an organisation + * can pin the scale; `/config` writes to user settings, which is where the + * fallback lands. + */ +export function getWorkflowSizeGuideline(): WorkflowSizeGuideline { + const fromSettings = getWorkflowSizeGuidelineFromSettings() + if (fromSettings) return fromSettings + const fromConfig = readGlobalGuideline() + if (fromConfig) return fromConfig + return DEFAULT_WORKFLOW_SIZE_GUIDELINE +} + +/** + * The `/config` choice, or undefined. + * + * `getGlobalConfig()` throws before bootstrap opens config access, and this is + * read from the tool prompt — which the SDK can build early. A missing + * guideline is not worth failing a turn over. + */ +function readGlobalGuideline(): WorkflowSizeGuideline | undefined { + try { + const value = getGlobalConfig().workflowSizeGuideline + return isSizeGuideline(value) ? value : undefined + } catch { + return undefined + } +} + +/** + * The guideline a settings file pins, if any. + * + * A settings file outranks the `/config` choice so an organisation can fix the + * scale; `/config` hides its row entirely in that case rather than offering a + * control that would silently do nothing. + */ +export function getWorkflowSizeGuidelineFromSettings(): + | WorkflowSizeGuideline + | undefined { + const managed = getSettingsForSource('policySettings')?.workflowSizeGuideline + if (isSizeGuideline(managed)) return managed + const merged = getSettings_DEPRECATED().workflowSizeGuideline + if (isSizeGuideline(merged)) return merged + return undefined +} + +/** True when the guideline was chosen, rather than falling back to the default. */ +export function isWorkflowSizeGuidelineExplicit(): boolean { + if (getWorkflowSizeGuidelineFromSettings() !== undefined) return true + return readGlobalGuideline() !== undefined +} + +/** Sentence appended to the Workflow tool prompt so Claude sizes runs to taste. */ +export function describeWorkflowSizeGuideline( + guideline: WorkflowSizeGuideline = getWorkflowSizeGuideline(), +): string { + const target = WORKFLOW_SIZE_AGENT_TARGETS[guideline] + if (target === null) { + return 'This session has no workflow size guideline: size the workflow to the task.' + } + return ( + `This session has the workflow size guideline: ${guideline} — keep workflows under ${target} agents. ` + + "This is a guideline, not a hard limit — follow it unless the user's prompt calls for a different scale. " + + 'The user can raise or remove it with "Dynamic workflow size" in /config.' + ) +} + +function isSizeGuideline(value: unknown): value is WorkflowSizeGuideline { + return ( + typeof value === 'string' && + (WORKFLOW_SIZE_GUIDELINES as readonly string[]).includes(value) + ) +} diff --git a/src/utils/workflows/errors.ts b/src/utils/workflows/errors.ts new file mode 100644 index 00000000..2d014691 --- /dev/null +++ b/src/utils/workflows/errors.ts @@ -0,0 +1,39 @@ +/** + * Describe a thrown value that may have crossed the VM boundary. + * + * An `Error` constructed inside the workflow sandbox belongs to that realm, so + * `instanceof Error` is false in the host and the usual `err.message` narrowing + * never fires — the value would render as `{}`. Reading the fields + * defensively is the only way to keep the script author's own error text. + */ +export function describeThrown(value: unknown): string { + if (typeof value === 'string') return value + if (value === null || value === undefined) return String(value) + if (typeof value !== 'object') return String(value) + + const message = readString(value, 'message') + const name = readString(value, 'name') + if (message) return name && name !== 'Error' ? `${name}: ${message}` : message + if (name) return name + + try { + return JSON.stringify(value) ?? '[object]' + } catch { + return '[unprintable thrown value]' + } +} + +/** The `name` of a thrown value, used to recognise the runtime's own errors. */ +export function thrownName(value: unknown): string | undefined { + if (value === null || typeof value !== 'object') return undefined + return readString(value, 'name') +} + +function readString(target: object, key: string): string | undefined { + try { + const value = (target as Record)[key] + return typeof value === 'string' && value !== '' ? value : undefined + } catch { + return undefined + } +} diff --git a/src/utils/workflows/harness.ts b/src/utils/workflows/harness.ts new file mode 100644 index 00000000..998b05c3 --- /dev/null +++ b/src/utils/workflows/harness.ts @@ -0,0 +1,588 @@ +import type { CanUseToolFn } from '../../hooks/useCanUseTool.js' +import type { ToolUseContext } from '../../Tool.js' +import { logForDebugging } from '../debug.js' +import { describeThrown, thrownName } from './errors.js' +import { + WORKFLOW_AGENT_CAP_MESSAGE, + WORKFLOW_AGENT_STALL_MS, + WORKFLOW_LABEL_MAX_CHARS, + WORKFLOW_MAX_AGENTS, + WORKFLOW_MAX_FANOUT, + WORKFLOW_PREVIEW_MAX_CHARS, + getWorkflowConcurrency, +} from './constants.js' +import { + WorkflowJournal, + workflowCacheKey, + type WorkflowJournalSnapshot, +} from './journal.js' +import { createLimiter, type Limiter } from './limiter.js' +import { + createWorkflowAgentController, + runWorkflowAgent, + type WorkflowAgentRunParams, + type WorkflowAgentRunResult, +} from './runWorkflowAgent.js' +import type { + WorkflowAgentEvent, + WorkflowAgentOptions, + WorkflowProgressEvent, +} from './types.js' + +/** Thrown when a script's loop keeps calling `agent()` past the hard cap. */ +export class WorkflowAgentCapError extends Error { + constructor() { + super(WORKFLOW_AGENT_CAP_MESSAGE) + this.name = 'WorkflowAgentCapError' + } +} + +/** Thrown when the turn's token target is exhausted mid-run. */ +export class WorkflowBudgetExceededError extends Error { + constructor(spent: number, total: number) { + super( + `Workflow token budget exceeded (${spent.toLocaleString()} / ${total.toLocaleString()} output tokens). ` + + 'Stopping further agent() calls. In-flight agents will complete; their results are preserved.', + ) + this.name = 'WorkflowBudgetExceededError' + } +} + +export type WorkflowHarnessParams = { + toolUseContext: ToolUseContext + canUseTool: CanUseToolFn + runId: string + /** `meta.name`, recorded on each agent's sidecar metadata. */ + workflowName: string + emit: (event: WorkflowProgressEvent) => void + /** Titles from `meta.phases`, seeded so the phase list exists before agents run. */ + seedPhaseTitles?: string[] + tokenBudget?: { total: number | null; getTurnSpent: () => number } + journal?: WorkflowJournal + journalSnapshot?: WorkflowJournalSnapshot + /** Registers/unregisters a running agent so the UI can skip or retry it. */ + onAgentController: ( + agentKey: string, + controller: AbortController | undefined, + ) => void + abortSignal?: AbortSignal + /** Overridable so tests can drive the harness without a live model. */ + runAgentImpl?: ( + params: WorkflowAgentRunParams, + ) => Promise + /** + * Counters a nested `workflow()` child shares with its parent. + * + * Without this a child would get its own concurrency slot pool (doubling the + * agents actually in flight) and its own index sequence (whose progress rows + * would overwrite the parent's, since rows are keyed by index). + */ + shared?: WorkflowSharedCounters +} + +export type WorkflowSharedCounters = { + limiter: Limiter + nextAgentIndex: () => number + getAgentCount: () => number +} + +/** Counters for a top-level run; nested children are handed the same object. */ +export function createWorkflowSharedCounters(): WorkflowSharedCounters { + let agentCount = 0 + return { + limiter: createLimiter(getWorkflowConcurrency()), + nextAgentIndex: () => ++agentCount, + getAgentCount: () => agentCount, + } +} + +export type WorkflowHarness = { + agent: (prompt: unknown, opts?: unknown) => Promise + parallel: (thunks: unknown) => Promise + pipeline: (items: unknown, ...stages: unknown[]) => Promise + log: (message: unknown) => void + phase: (title: unknown) => void + getAgentCount: () => number + getFailures: () => string[] + recordFailure: (message: string) => void +} + +/** + * Build the globals a workflow script sees. + * + * Everything here is the host side of the VM boundary: values arriving from + * the script are untrusted (they can be cross-realm objects, proxies, or + * getters that throw), so each one is read defensively before use. + */ +export function createWorkflowHarness( + params: WorkflowHarnessParams, +): WorkflowHarness { + const { + toolUseContext, + canUseTool, + runId, + workflowName, + emit, + seedPhaseTitles, + tokenBudget, + journal, + journalSnapshot, + onAgentController, + abortSignal, + runAgentImpl = runWorkflowAgent, + shared = createWorkflowSharedCounters(), + } = params + + const runAgentLimiter = shared.limiter + const failures: string[] = [] + const phaseIndexByTitle = new Map() + + let phaseCounter = 0 + let currentPhase: string | undefined + let capReported = false + let budgetReported = false + /** Journal replay stops permanently at the first cache miss (see resume docs). */ + let cacheExhausted = false + let previousCacheKey = '' + + function resolvePhase(title: string, kind: 'meta' | 'script'): number { + const existing = phaseIndexByTitle.get(title) + if (existing !== undefined) return existing + const index = ++phaseCounter + phaseIndexByTitle.set(title, index) + emit({ type: 'workflow_phase', index, title, kind }) + return index + } + + for (const title of seedPhaseTitles ?? []) { + if (typeof title === 'string' && title !== '') resolvePhase(title, 'meta') + } + + function assertAgentCap(): void { + if (shared.getAgentCount() < WORKFLOW_MAX_AGENTS) return + if (!capReported) { + capReported = true + logForDebugging(`Workflow ${runId} hit the ${WORKFLOW_MAX_AGENTS}-agent cap`) + } + throw new WorkflowAgentCapError() + } + + function assertBudget(): void { + const total = tokenBudget?.total + if (total == null || total <= 0) return + const spent = tokenBudget.getTurnSpent() + if (spent < total) return + if (!budgetReported) { + budgetReported = true + logForDebugging(`Workflow ${runId} exhausted its ${total} token budget`) + } + throw new WorkflowBudgetExceededError(spent, total) + } + + function neverResolves(): Promise { + // An aborted run must not let the script continue past the await: the task + // is already terminal, so the cleanest stop is a promise that never settles. + return new Promise(() => {}) + } + + const agent = async ( + rawPrompt: unknown, + rawOpts?: unknown, + ): Promise => { + if (abortSignal?.aborted) return neverResolves() + const prompt = coerceToString(rawPrompt) + const opts = readAgentOptions(rawOpts) + + assertAgentCap() + assertBudget() + + const index = shared.nextAgentIndex() + const label = deriveLabel(prompt, opts.label) + const phaseTitle = opts.phase ?? currentPhase + const phaseIndex = + phaseTitle !== undefined ? resolvePhase(phaseTitle, 'script') : undefined + const model = opts.model ?? toolUseContext.options.mainLoopModel + const promptPreview = clip(prompt, WORKFLOW_PREVIEW_MAX_CHARS) + + if (opts.isolation === 'remote') { + throw new Error( + "agent({isolation:'remote'}) is not available in this build", + ) + } + + const cacheKey = journal + ? workflowCacheKey(prompt, opts, previousCacheKey) + : undefined + if (cacheKey !== undefined) previousCacheKey = cacheKey + + if (cacheKey !== undefined && !cacheExhausted) { + const cached = journalSnapshot?.results.get(cacheKey) + if (cached !== undefined) { + const now = Date.now() + emit({ + type: 'workflow_agent', + index, + label, + state: 'done', + phaseIndex, + phaseTitle, + model, + agentId: cached.agentId, + startedAt: now, + lastProgressAt: now, + cached: true, + promptPreview, + resultPreview: clip( + coerceToString(cached.result), + WORKFLOW_PREVIEW_MAX_CHARS, + ), + }) + return cached.result + } + // First miss: every later agent started after this one, so none of their + // cached results are valid any more. + cacheExhausted = true + } + + const queuedAt = Date.now() + const baseEvent: WorkflowAgentEvent = { + type: 'workflow_agent', + index, + label, + state: 'start', + phaseIndex, + phaseTitle, + model, + queuedAt, + lastProgressAt: queuedAt, + promptPreview, + ...(opts.agentType ? { agentType: opts.agentType } : {}), + ...(opts.isolation === 'worktree' ? { isolation: 'worktree' as const } : {}), + } + emit(baseEvent) + + const agentKey = `${runId}-${index}` + return runAgentLimiter(async () => { + if (abortSignal?.aborted) return neverResolves() + assertBudget() + + const controller = createWorkflowAgentController(abortSignal) + onAgentController(agentKey, controller) + const startedAt = Date.now() + let tokens = 0 + let toolCalls = 0 + let agentId: string | undefined + const stallMs = opts.stallMs ?? WORKFLOW_AGENT_STALL_MS + let lastProgressAt = startedAt + const stallTimer = setInterval(() => { + if (Date.now() - lastProgressAt < stallMs) return + emit({ + ...baseEvent, + state: 'progress', + startedAt, + lastProgressAt, + tokens, + toolCalls, + agentId, + }) + }, Math.max(5_000, Math.floor(stallMs / 2))) + stallTimer.unref?.() + + emit({ ...baseEvent, state: 'progress', startedAt, lastProgressAt }) + + try { + const result = await runAgentImpl({ + prompt, + opts, + toolUseContext, + canUseTool, + runId, + // Written to the agent's sidecar before it starts, so a finished + // run can be rebuilt from disk long after the progress stream is + // gone. Nothing else persists which phase an agent belonged to. + workflow: { + runId, + name: workflowName, + phaseIndex: phaseIndex ?? 0, + ...(phaseTitle ? { phaseTitle } : {}), + agentIndex: index, + }, + abortController: controller, + onAgentId: id => { + agentId = id + void journal + ?.append({ type: 'started', key: cacheKey ?? '', agentId: id }) + .catch(error => + logForDebugging(`workflow journal started-append failed: ${error}`), + ) + }, + onProgress: progress => { + tokens = progress.tokens + toolCalls = progress.toolCalls + lastProgressAt = Date.now() + emit({ + ...baseEvent, + state: 'progress', + startedAt, + lastProgressAt, + tokens, + toolCalls, + agentId, + ...(progress.lastToolName + ? { lastToolName: progress.lastToolName } + : {}), + }) + }, + }) + + emit({ + ...baseEvent, + state: 'done', + agentId: result.agentId, + startedAt, + lastProgressAt: Date.now(), + durationMs: Date.now() - startedAt, + tokens: result.tokens, + toolCalls: result.toolCalls, + resultPreview: clip( + coerceToString(result.value), + WORKFLOW_PREVIEW_MAX_CHARS, + ), + }) + + if (cacheKey !== undefined) { + void journal + ?.append({ + type: 'result', + key: cacheKey, + agentId: result.agentId, + result: result.value, + }) + .catch(error => + logForDebugging(`workflow journal result-append failed: ${error}`), + ) + } + return result.value + } catch (error) { + const skipped = + describeThrown(controller.signal.reason).includes('user-skip') + const message = skipped ? 'skipped by user' : describeThrown(error) + emit({ + ...baseEvent, + state: 'error', + agentId, + startedAt, + lastProgressAt: Date.now(), + durationMs: Date.now() - startedAt, + tokens, + toolCalls, + error: message, + ...(skipped ? { skipped: true } : {}), + }) + if (abortSignal?.aborted) return neverResolves() + // A skipped agent resolves to null so the surrounding pipeline keeps + // going; a real failure propagates so the script can react. + if (skipped) return null + throw error + } finally { + clearInterval(stallTimer) + onAgentController(agentKey, undefined) + } + }) + } + + const parallel = async (rawThunks: unknown): Promise => { + if (abortSignal?.aborted) return neverResolves() + const thunks = readArray(rawThunks, 'parallel() expects an array of functions') + if (thunks.length === 0) return [] + assertAgentCap() + assertBudget() + for (const thunk of thunks) { + if (typeof thunk !== 'function') { + throw new TypeError( + 'parallel() expects an array of functions, not promises. Wrap each call: () => agent(...)', + ) + } + } + const settled = await Promise.allSettled( + thunks.map(thunk => { + try { + return Promise.resolve((thunk as () => unknown)()) + } catch (error) { + return Promise.reject(error) + } + }), + ) + return collectSettled(settled, 'parallel') + } + + const pipeline = async ( + rawItems: unknown, + ...rawStages: unknown[] + ): Promise => { + if (abortSignal?.aborted) return neverResolves() + const items = readArray( + rawItems, + 'pipeline() expects an array as the first argument', + ) + if (items.length === 0) return [] + assertAgentCap() + assertBudget() + const stages = rawStages.flat() + for (const stage of stages) { + if (typeof stage !== 'function') { + throw new TypeError( + 'pipeline() stages must be functions: pipeline(items, item => ..., result => ...)', + ) + } + } + const settled = await Promise.allSettled( + items.map(async (item, index) => { + let value: unknown = item + for (const stage of stages) { + if (value === null) break + value = await (stage as ( + prev: unknown, + original: unknown, + index: number, + ) => unknown)(value, item, index) + } + return value + }), + ) + return collectSettled(settled, 'pipeline') + } + + /** + * Turn a settled batch into the script-visible array. + * + * A rejected slot becomes `null` rather than rejecting the whole call — one + * bad file in a 500-file migration should not discard the other 499. The + * failure is still recorded so it reaches the final report. + */ + function collectSettled( + settled: PromiseSettledResult[], + kind: 'parallel' | 'pipeline', + ): unknown[] { + let budgetDrops = 0 + const values = settled.map((entry, index) => { + if (entry.status === 'fulfilled') return entry.value + const reason = entry.reason + if (thrownName(reason) === 'WorkflowBudgetExceededError') { + budgetDrops++ + return null + } + const message = `${kind}[${index}] failed: ${describeThrown(reason)}` + failures.push(message) + emit({ type: 'workflow_log', message }) + return null + }) + if (budgetDrops > 0) { + failures.push( + `${kind}: ${budgetDrops} slot${budgetDrops === 1 ? '' : 's'} dropped — token budget exceeded`, + ) + } + return values + } + + const log = (rawMessage: unknown): void => { + emit({ type: 'workflow_log', message: coerceToString(rawMessage) }) + } + + const phase = (rawTitle: unknown): void => { + const title = coerceToString(rawTitle) + if (title === '') return + currentPhase = title + resolvePhase(title, 'script') + } + + return { + agent, + parallel, + pipeline, + log, + phase, + getAgentCount: shared.getAgentCount, + getFailures: () => failures, + recordFailure: message => failures.push(message), + } +} + +/** + * Copy the options object out of the VM before use. + * + * The script owns that object and could keep mutating it (or define getters + * that throw) while the agent runs; a flat snapshot of the known keys is the + * only version the runtime trusts. `schema` is passed through by reference + * because `createSyntheticOutputTool` caches compiled validators on identity. + */ +function readAgentOptions(raw: unknown): WorkflowAgentOptions { + if (raw === null || typeof raw !== 'object') return {} + const opts: WorkflowAgentOptions = {} + const label = readProp(raw, 'label') + if (label != null) opts.label = String(label) + const phase = readProp(raw, 'phase') + if (phase != null) opts.phase = String(phase) + const model = readProp(raw, 'model') + if (model != null) opts.model = String(model) + const effort = readProp(raw, 'effort') + if (effort != null) opts.effort = String(effort) + const agentType = readProp(raw, 'agentType') + if (agentType != null) opts.agentType = String(agentType) + const isolation = readProp(raw, 'isolation') + if (isolation === 'worktree' || isolation === 'remote') { + opts.isolation = isolation + } + const stallMs = readProp(raw, 'stallMs') + if (typeof stallMs === 'number' && Number.isFinite(stallMs)) { + opts.stallMs = stallMs + } + const schema = readProp(raw, 'schema') + if (schema != null && typeof schema === 'object') opts.schema = schema + return opts +} + +function readProp(target: unknown, key: string): unknown { + try { + return (target as Record)[key] + } catch { + return undefined + } +} + +function readArray(value: unknown, message: string): unknown[] { + if (!Array.isArray(value)) throw new TypeError(message) + if (value.length > WORKFLOW_MAX_FANOUT) { + throw new RangeError( + `A single parallel()/pipeline() call accepts at most ${WORKFLOW_MAX_FANOUT} items (got ${value.length})`, + ) + } + return [...value] +} + +function deriveLabel(prompt: string, explicit: string | undefined): string { + const source = explicit ?? prompt.slice(0, WORKFLOW_LABEL_MAX_CHARS) + return source.replace(/\s+/g, ' ').trim() || 'agent' +} + +function clip(value: string, max: number): string | undefined { + const trimmed = value.trim() + if (trimmed === '') return undefined + return trimmed.length > max ? `${trimmed.slice(0, max)}…` : trimmed +} + +/** Render any script value as a string without invoking its toString trap. */ +function coerceToString(value: unknown): string { + if (typeof value === 'string') return value + if (value === null) return 'null' + if (value === undefined) return 'undefined' + if (typeof value === 'function') return '[function]' + if (typeof value === 'object') { + try { + return JSON.stringify(value) ?? '[object]' + } catch { + return '[object]' + } + } + return String(value) +} diff --git a/src/utils/workflows/journal.ts b/src/utils/workflows/journal.ts new file mode 100644 index 00000000..3b5508bd --- /dev/null +++ b/src/utils/workflows/journal.ts @@ -0,0 +1,144 @@ +import { appendFile, mkdir, readFile } from 'fs/promises' +import { dirname } from 'path' +import { logForDebugging } from '../debug.js' +import { getWorkflowJournalPath } from './paths.js' + +/** + * One line per agent lifecycle transition, appended as the run progresses. + * + * `started` is written before the subagent is spawned and `result` after it + * returns. Resume needs both: an agent that has a `started` line but no + * `result` was in flight when the run stopped, which is what makes every agent + * after it re-run instead of replaying. + */ +export type WorkflowJournalEntry = + | { type: 'started'; key: string; agentId: string } + | { type: 'result'; key: string; agentId: string; result: unknown } + +export type WorkflowJournalSnapshot = { + /** key → the cached return value of a completed agent. */ + results: Map + /** key → agent ids that started but never produced a result. */ + started: Map +} + +/** + * Append-only record of a run's agent results, used to replay a resumed run. + * + * Writes are serialized through a promise chain: `agent()` calls resolve + * concurrently, and two overlapping appends to the same file can interleave + * partial lines. + */ +export class WorkflowJournal { + private readonly path: string + private writeChain: Promise = Promise.resolve() + private dirReady = false + + /** Takes an absolute path so tests never resolve against the real session dir. */ + constructor(path: string) { + this.path = path + } + + get filePath(): string { + return this.path + } + + /** + * Wait for every queued append to hit disk. + * + * Appends are fired without awaiting so an agent's result reaches the script + * immediately; the run must flush before it settles or the last result can + * be missing from a resume. + */ + async flush(): Promise { + await this.writeChain + } + + async append(entry: WorkflowJournalEntry): Promise { + const line = `${safeStringify(entry)}\n` + this.writeChain = this.writeChain.then(async () => { + if (!this.dirReady) { + await mkdir(dirname(this.path), { recursive: true }) + this.dirReady = true + } + await appendFile(this.path, line, 'utf8') + }) + return this.writeChain + } + + /** + * Read back a previous run's journal. A truncated or corrupt trailing line + * is skipped rather than failing the resume — the worst case is that one + * agent re-runs. + */ + async load(): Promise { + const snapshot: WorkflowJournalSnapshot = { + results: new Map(), + started: new Map(), + } + let raw: string + try { + raw = await readFile(this.path, 'utf8') + } catch { + return snapshot + } + for (const line of raw.split('\n')) { + if (line.trim() === '') continue + let entry: WorkflowJournalEntry + try { + entry = JSON.parse(line) as WorkflowJournalEntry + } catch { + logForDebugging(`workflow journal: skipping malformed line in ${this.path}`) + continue + } + if (entry.type === 'result') { + snapshot.results.set(entry.key, { + agentId: entry.agentId, + result: entry.result, + }) + snapshot.started.delete(entry.key) + } else if (entry.type === 'started') { + if (snapshot.results.has(entry.key)) continue + const existing = snapshot.started.get(entry.key) ?? [] + existing.push(entry.agentId) + snapshot.started.set(entry.key, existing) + } + } + return snapshot + } +} + +/** The journal for a run, resolved against the current session's directory. */ +export function createRunJournal(runId: string): WorkflowJournal { + return new WorkflowJournal(getWorkflowJournalPath(runId)) +} + +/** + * Cache key for one `agent()` call. + * + * Chained through the previous key so position in the call order is part of + * the identity: two identical `agent('review')` calls in a loop must not share + * a cache entry, and inserting a call ahead of them must invalidate both. + */ +export function workflowCacheKey( + prompt: string, + opts: unknown, + previousKey: string, +): string { + return `${previousKey}|${prompt}|${safeStringify(opts ?? null)}` +} + +function safeStringify(value: unknown): string { + const seen = new WeakSet() + return ( + JSON.stringify(value, (_key, val) => { + if (typeof val === 'bigint') return val.toString() + if (typeof val === 'function') return undefined + if (val !== null && typeof val === 'object') { + if (seen.has(val as object)) return '[Circular]' + seen.add(val as object) + } + return val + }) ?? 'null' + ) +} diff --git a/src/utils/workflows/keyword.test.ts b/src/utils/workflows/keyword.test.ts new file mode 100644 index 00000000..ae9fd9e0 --- /dev/null +++ b/src/utils/workflows/keyword.test.ts @@ -0,0 +1,51 @@ +import { describe, expect, test } from 'bun:test' +import { findKeywordRanges, hasWorkflowKeyword } from './keyword.js' + +describe('ultracode keyword matcher', () => { + test.each([ + ['bare', 'ultracode: audit every endpoint'], + ['mid-sentence', 'please ultracode this repo'], + ['capitalised', 'Ultracode the migration'], + ['after a comma', 'ok, ultracode, go'], + ['at the end', 'audit the routes ultracode'], + ])('fires on %s', (_label, text) => { + expect(hasWorkflowKeyword(text)).toBe(true) + }) + + test.each([ + ['inside backticks', 'run `ultracode` to see the keyword'], + ['inside double quotes', 'the word "ultracode" opts a turn in'], + ['inside single quotes', "the word 'ultracode' opts a turn in"], + ['inside a tag', ' is not a real tag'], + ['inside braces', 'config is { mode: ultracode }'], + ['inside brackets', 'see [ultracode] in the docs'], + ['inside parens', 'the flag (ultracode) exists'], + ['as a path segment', 'open docs/ultracode/readme.md'], + ['as a filename', 'edit ultracode.ts'], + ['as a flag', 'pass --ultracode to the CLI'], + ['hyphenated', 'the ultracode-runner package'], + ['as a question', 'what is ultracode?'], + ['as a substring', 'superultracoded is a different word'], + ])('does not fire %s', (_label, text) => { + expect(hasWorkflowKeyword(text)).toBe(false) + }) + + test('an apostrophe inside a word does not open a quoted span', () => { + expect(hasWorkflowKeyword("don't stop — ultracode this")).toBe(true) + }) + + test('reports the range so the composer can highlight it', () => { + const ranges = findKeywordRanges('please Ultracode this repo') + expect(ranges).toEqual([{ word: 'Ultracode', start: 7, end: 16 }]) + }) + + test('finds every standalone occurrence', () => { + expect(findKeywordRanges('ultracode now, ultracode later')).toHaveLength(2) + }) + + test('empty and missing input never fires', () => { + expect(hasWorkflowKeyword('')).toBe(false) + expect(hasWorkflowKeyword(null)).toBe(false) + expect(hasWorkflowKeyword(undefined)).toBe(false) + }) +}) diff --git a/src/utils/workflows/keyword.ts b/src/utils/workflows/keyword.ts new file mode 100644 index 00000000..3cfbad48 --- /dev/null +++ b/src/utils/workflows/keyword.ts @@ -0,0 +1,99 @@ +/** + * The `ultracode` keyword trigger. + * + * Typing `ultracode` in a prompt opts that turn into multi-agent orchestration. + * The matcher has to be conservative: the word appears in file paths, code + * fences, and quoted prose far more often than it appears as an instruction, + * and a false positive silently turns a one-line question into a run that + * spawns dozens of agents. + */ + +export const WORKFLOW_KEYWORD = 'ultracode' + +export type KeywordRange = { + word: string + start: number + end: number +} + +/** Spans a match must not fall inside — quoted text, code, and bracketed args. */ +const SPAN_DELIMITERS: Record = { + '`': '`', + '"': '"', + '<': '>', + '{': '}', + '[': ']', + '(': ')', + "'": "'", +} + +function isWordChar(char: string | undefined): boolean { + return char !== undefined && /[A-Za-z0-9_]/.test(char) +} + +/** + * Spans of `text` that should be treated as opaque. + * + * `<` only opens a span when it looks like a tag, and `'` only when it is not + * an apostrophe inside a word — otherwise "don't" would swallow the rest of + * the sentence and hide a real keyword behind it. + */ +function findOpaqueSpans(text: string): Array<{ start: number; end: number }> { + const spans: Array<{ start: number; end: number }> = [] + let open: string | null = null + let openIndex = 0 + + for (let i = 0; i < text.length; i++) { + const char = text[i]! + if (open !== null) { + if (char !== SPAN_DELIMITERS[open]) continue + if (open === "'" && isWordChar(text[i + 1])) continue + spans.push({ start: openIndex, end: i + 1 }) + open = null + continue + } + const opensTag = char === '<' && i + 1 < text.length && /[a-zA-Z/]/.test(text[i + 1]!) + const opensQuote = char === "'" && !isWordChar(text[i - 1]) + const opensOther = char !== '<' && char !== "'" && char in SPAN_DELIMITERS + if (opensTag || opensQuote || opensOther) { + open = char + openIndex = i + } + } + return spans +} + +/** + * Every place `keyword` appears as a standalone word the user meant. + * + * Returned as ranges, not a boolean, because the composer highlights the word + * so the user can see the turn was opted in before they send it. + */ +export function findKeywordRanges( + text: string, + keyword: string = WORKFLOW_KEYWORD, +): KeywordRange[] { + const opaque = findOpaqueSpans(text) + const matches: KeywordRange[] = [] + for (const match of text.matchAll(new RegExp(`\\b${keyword}\\b`, 'gi'))) { + if (match.index === undefined) continue + const start = match.index + const end = start + match[0].length + if (opaque.some(span => start >= span.start && start < span.end)) continue + const before = text[start - 1] + const after = text[end] + // Path- and flag-like neighbours: `--ultracode`, `foo/ultracode`, + // `ultracode-runner`, `ultracode?`. + if (before === '/' || before === '\\' || before === '-') continue + if (after === '/' || after === '\\' || after === '-' || after === '?') continue + // `ultracode.js` — a filename, not an instruction. + if (after === '.' && isWordChar(text[end + 1])) continue + matches.push({ word: match[0], start, end }) + } + return matches +} + +export function hasWorkflowKeyword(text: string | null | undefined): boolean { + if (!text) return false + return findKeywordRanges(text).length > 0 +} diff --git a/src/utils/workflows/keywordAttachment.test.ts b/src/utils/workflows/keywordAttachment.test.ts new file mode 100644 index 00000000..25ca3025 --- /dev/null +++ b/src/utils/workflows/keywordAttachment.test.ts @@ -0,0 +1,157 @@ +import { afterEach, beforeEach, describe, expect, test } from 'bun:test' +import { mkdirSync } from 'fs' +import { mkdtemp, rm, writeFile } from 'fs/promises' +import { tmpdir } from 'os' +import { join } from 'path' +import { getIsInteractive, setIsInteractive } from '../../bootstrap/state.js' +import type { AppState } from '../../state/AppState.js' +import type { ToolUseContext } from '../../Tool.js' +import { getAttachmentsForTesting } from '../attachments.js' +import { resetSettingsCache } from '../settings/settingsCache.js' + +let home: string +let configDir: string +let wasInteractive: boolean +const originalConfigDir = process.env.CLAUDE_CONFIG_DIR + +async function writeSettings(value: Record): Promise { + await writeFile(join(configDir, 'settings.json'), JSON.stringify(value), 'utf8') + resetSettingsCache() +} + +function contextWith(state: Partial): ToolUseContext { + return { + getAppState: () => state as AppState, + } as unknown as ToolUseContext +} + +/** + * The keyword path has three independent off-switches (the setting, the + * per-prompt `opt+w` dismissal, and the matcher itself). Each one has silently + * regressed in the official CLI at least once, and the failure mode is the + * same either way: a one-line prompt quietly becomes a run that spawns dozens + * of agents, or the keyword stops working with no error anywhere. + */ +describe('workflow keyword attachment', () => { + beforeEach(async () => { + home = await mkdtemp(join(tmpdir(), 'wf-keyword-attach-')) + configDir = join(home, 'claude') + mkdirSync(configDir, { recursive: true }) + process.env.CLAUDE_CONFIG_DIR = configDir + // The keyword is interactive-only, so these cases have to run as if the + // user were at the prompt. + wasInteractive = getIsInteractive() + setIsInteractive(true) + await writeSettings({}) + }) + + afterEach(async () => { + setIsInteractive(wasInteractive) + if (originalConfigDir === undefined) delete process.env.CLAUDE_CONFIG_DIR + else process.env.CLAUDE_CONFIG_DIR = originalConfigDir + resetSettingsCache() + await rm(home, { recursive: true, force: true }) + }) + + test('fires on a typed keyword', () => { + expect( + getAttachmentsForTesting.workflowKeyword('ultracode: audit the routes', { + suppressed: false, + }), + ).toEqual([{ type: 'workflow_keyword_request' }]) + }) + + test('opt+w suppression wins over a present keyword', () => { + expect( + getAttachmentsForTesting.workflowKeyword('ultracode: audit the routes', { + suppressed: true, + }), + ).toEqual([]) + }) + + test('the setting turns the trigger off entirely', async () => { + await writeSettings({ workflowKeywordTriggerEnabled: false }) + expect( + getAttachmentsForTesting.workflowKeyword('ultracode: audit', { + suppressed: false, + }), + ).toEqual([]) + }) + + test('a quoted mention of the word is not an opt-in', () => { + expect( + getAttachmentsForTesting.workflowKeyword( + 'what does the "ultracode" keyword do?', + { suppressed: false }, + ), + ).toEqual([]) + }) +}) + +describe('ultracode effort attachments', () => { + test('announces the transition once, then keeps it short', () => { + const enter = getAttachmentsForTesting.ultracodeEffort([], true) + expect(enter).toEqual([{ type: 'ultra_effort_enter', reminderType: 'full' }]) + + const messages = [ + { type: 'attachment', attachment: { type: 'ultra_effort_enter' } }, + ] as never + expect(getAttachmentsForTesting.ultracodeEffort(messages, true)).toEqual([ + { type: 'ultra_effort_enter', reminderType: 'short' }, + ]) + }) + + test('announces the exit only after it was announced on', () => { + expect(getAttachmentsForTesting.ultracodeEffort([], false)).toEqual([]) + + const messages = [ + { type: 'attachment', attachment: { type: 'ultra_effort_enter' } }, + ] as never + expect(getAttachmentsForTesting.ultracodeEffort(messages, false)).toEqual([ + { type: 'ultra_effort_exit' }, + ]) + }) + + test('does not repeat the exit once it has been announced', () => { + const messages = [ + { type: 'attachment', attachment: { type: 'ultra_effort_enter' } }, + { type: 'attachment', attachment: { type: 'ultra_effort_exit' } }, + ] as never + expect(getAttachmentsForTesting.ultracodeEffort(messages, false)).toEqual([]) + }) + + test('reads the flag from AppState, not from a parameter default', () => { + const context = contextWith({ ultracode: true }) + expect(context.getAppState().ultracode).toBe(true) + }) +}) + +describe('workflow size guideline reminder', () => { + test('stays silent on the first turn — the tool prompt already carries it', () => { + expect(getAttachmentsForTesting.workflowSizeGuideline([])).toEqual([]) + }) + + test('announces a change away from what was last announced', () => { + const messages = [ + { + type: 'attachment', + attachment: { type: 'workflow_size_guideline_change', size: 'small' }, + }, + ] as never + // The live guideline in this environment is the default, so a previously + // announced 'small' must produce a fresh announcement. + expect(getAttachmentsForTesting.workflowSizeGuideline(messages)).toEqual([ + { type: 'workflow_size_guideline_change', size: 'medium' }, + ]) + }) + + test('does not repeat an announcement that still holds', () => { + const messages = [ + { + type: 'attachment', + attachment: { type: 'workflow_size_guideline_change', size: 'medium' }, + }, + ] as never + expect(getAttachmentsForTesting.workflowSizeGuideline(messages)).toEqual([]) + }) +}) diff --git a/src/utils/workflows/limiter.ts b/src/utils/workflows/limiter.ts new file mode 100644 index 00000000..1e3833cb --- /dev/null +++ b/src/utils/workflows/limiter.ts @@ -0,0 +1,34 @@ +/** + * A minimal FIFO concurrency gate. + * + * `parallel()` and `pipeline()` hand the runtime as many items as the script + * asks for; this is what keeps only N model streams alive at a time. Queued + * work runs in submission order so a fan-out's progress view fills top-down + * instead of at random. + */ +export type Limiter = (task: () => Promise) => Promise + +export function createLimiter(concurrency: number): Limiter { + const max = Math.max(1, Math.floor(concurrency)) + const queue: Array<() => void> = [] + let active = 0 + + // Release hands the slot directly to the next waiter instead of decrementing + // and letting it re-check: a caller that arrives during the microtask gap + // would otherwise see a free slot that is already spoken for. + const release = (): void => { + const next = queue.shift() + if (next) next() + else active-- + } + + return async (task: () => Promise): Promise => { + if (active < max) active++ + else await new Promise(resolve => queue.push(resolve)) + try { + return await task() + } finally { + release() + } + } +} diff --git a/src/utils/workflows/meta.test.ts b/src/utils/workflows/meta.test.ts new file mode 100644 index 00000000..87400bf8 --- /dev/null +++ b/src/utils/workflows/meta.test.ts @@ -0,0 +1,112 @@ +import { describe, expect, test } from 'bun:test' +import { parseWorkflowScript, usesBannedNondeterminism } from './meta.js' + +describe('parseWorkflowScript', () => { + test('extracts meta and leaves the body executable', () => { + const parsed = parseWorkflowScript( + [ + "export const meta = { name: 'audit', description: 'Audit routes' }", + "const found = await agent('list files')", + 'return found', + ].join('\n'), + ) + expect(parsed).not.toHaveProperty('error') + if ('error' in parsed) throw new Error(parsed.error) + expect(parsed.meta.name).toBe('audit') + expect(parsed.meta.description).toBe('Audit routes') + expect(parsed.scriptBody).toBe( + "const found = await agent('list files')\nreturn found", + ) + }) + + test('keeps phases with their detail and model', () => { + const parsed = parseWorkflowScript( + [ + 'export const meta = {', + " name: 'review',", + " description: 'Review the diff',", + " whenToUse: 'before merging',", + ' phases: [', + " { title: 'Find', detail: 'one agent per file' },", + " { title: 'Verify', model: 'sonnet' },", + ' ],', + '}', + 'return []', + ].join('\n'), + ) + if ('error' in parsed) throw new Error(parsed.error) + expect(parsed.meta.whenToUse).toBe('before merging') + expect(parsed.meta.phases).toEqual([ + { title: 'Find', detail: 'one agent per file' }, + { title: 'Verify', model: 'sonnet' }, + ]) + }) + + test('rejects a script whose first statement is not the meta export', () => { + const parsed = parseWorkflowScript( + ["const x = 1", "export const meta = { name: 'a', description: 'b' }"].join( + '\n', + ), + ) + expect(parsed).toEqual({ + error: + '`export const meta = { name, description, phases }` must be the FIRST statement in the script', + }) + }) + + test.each([ + ['a call', "export const meta = { name: 'a', description: makeIt() }\n"], + ['an identifier', "export const meta = { name: 'a', description: other }\n"], + [ + 'interpolation', + 'export const meta = { name: `a${1}`, description: "b" }\n', + ], + [ + 'a spread', + "export const meta = { ...base, name: 'a', description: 'b' }\n", + ], + ])('rejects meta containing %s', (_label, script) => { + const parsed = parseWorkflowScript(script) + expect('error' in parsed && parsed.error).toContain('pure literal') + }) + + test('rejects a name that would not be a safe command', () => { + const parsed = parseWorkflowScript( + "export const meta = { name: '../escape', description: 'b' }\n", + ) + expect('error' in parsed && parsed.error).toContain('meta.name') + }) + + test('reports TypeScript syntax with the plain-JavaScript hint', () => { + const parsed = parseWorkflowScript( + [ + "export const meta = { name: 'a', description: 'b' }", + 'const files: string[] = []', + ].join('\n'), + ) + expect('error' in parsed && parsed.error).toContain('plain JavaScript') + }) + + test('rejects a script over the byte limit', () => { + const script = `export const meta = { name: 'a', description: 'b' }\n// ${'x'.repeat( + 600_000, + )}` + expect('error' in parseWorkflowScript(script)).toBe(true) + }) +}) + +describe('usesBannedNondeterminism', () => { + test.each([ + ['Date.now()', 'const t = Date.now()'], + ['new Date()', 'const t = new Date()'], + ['Math.random()', 'const r = Math.random()'], + ])('flags %s', (_label, body) => { + expect(usesBannedNondeterminism(body)).toBe(true) + }) + + test('allows a Date built from an explicit timestamp', () => { + expect(usesBannedNondeterminism('const t = new Date(args.stamp)')).toBe( + false, + ) + }) +}) diff --git a/src/utils/workflows/meta.ts b/src/utils/workflows/meta.ts new file mode 100644 index 00000000..fbfdcd60 --- /dev/null +++ b/src/utils/workflows/meta.ts @@ -0,0 +1,276 @@ +import { parse } from 'acorn' +import { simple as walkSimple } from 'acorn-walk' +import { WORKFLOW_SCRIPT_MAX_BYTES } from './constants.js' +import type { WorkflowMeta, WorkflowPhaseMeta } from './types.js' + +/* eslint-disable @typescript-eslint/no-explicit-any */ +type AnyNode = any +/* eslint-enable @typescript-eslint/no-explicit-any */ + +export type WorkflowParseResult = + | { meta: WorkflowMeta; scriptBody: string } + | { error: string } + +const PLAIN_JS_HINT = + 'Workflow scripts must be plain JavaScript — common causes are TypeScript syntax ' + + '(type annotations, interfaces, generics) and broken string quoting or escaping.' + +const CARET_WINDOW = 80 + +const ACORN_OPTIONS = { + ecmaVersion: 'latest', + sourceType: 'module', + allowAwaitOutsideFunction: true, + allowReturnOutsideFunction: true, +} as const + +/** + * Split a workflow script into its `meta` literal and the executable body. + * + * `meta` must be the very first statement and a pure literal: the discovery + * path reads it for every saved workflow at `/` autocomplete time, so it can + * never be allowed to run code or reference anything outside itself. + */ +export function parseWorkflowScript(script: string): WorkflowParseResult { + if (Buffer.byteLength(script, 'utf8') > WORKFLOW_SCRIPT_MAX_BYTES) { + return { error: `Script exceeds ${WORKFLOW_SCRIPT_MAX_BYTES} bytes` } + } + + let program: AnyNode + try { + program = parse(script, ACORN_OPTIONS as never) + } catch (error) { + return { error: formatParseError(error, script) } + } + + const first = program.body[0] + if (!first || first.type !== 'ExportNamedDeclaration' || !isMetaExport(first)) { + return { + error: + '`export const meta = { name, description, phases }` must be the FIRST statement in the script', + } + } + + const init = first.declaration.declarations[0].init + let literal: unknown + try { + literal = evaluateObjectLiteral(init) + } catch (error) { + return { + error: `meta must be a pure literal: ${ + error instanceof Error ? error.message : String(error) + }`, + } + } + + const validated = validateMeta(literal) + if ('error' in validated) return validated + + // Drop the trailing `;` and the newline that ended the export so line + // numbers in runtime stack traces line up with the body the user wrote. + const scriptBody = script.slice(first.end).replace(/^[;\s]*\n/, '').trimStart() + return { meta: validated.meta, scriptBody } +} + +/** + * True when the script reads `Date.now()`, `new Date()` or `Math.random()`. + * Those are unavailable at runtime (they would make a resume replay diverge), + * so the caller surfaces a targeted hint instead of a bare TypeError. + */ +export function usesBannedNondeterminism(script: string): boolean { + let found = false + try { + const program = parse(script, ACORN_OPTIONS as never) + walkSimple(program as never, { + MemberExpression(node: AnyNode) { + if ( + node.computed || + node.object.type !== 'Identifier' || + node.property.type !== 'Identifier' + ) { + return + } + const object = node.object.name + const property = node.property.name + if ( + (object === 'Date' && property === 'now') || + (object === 'Math' && property === 'random') + ) { + found = true + } + }, + NewExpression(node: AnyNode) { + if ( + node.callee.type === 'Identifier' && + node.callee.name === 'Date' && + node.arguments.length === 0 + ) { + found = true + } + }, + }) + } catch { + return false + } + return found +} + +function isMetaExport(node: AnyNode): boolean { + const declaration = node.declaration + if (!declaration || declaration.type !== 'VariableDeclaration') return false + if (declaration.kind !== 'const' || declaration.declarations.length !== 1) { + return false + } + const declarator = declaration.declarations[0] + return ( + declarator.id.type === 'Identifier' && + declarator.id.name === 'meta' && + declarator.init?.type === 'ObjectExpression' + ) +} + +/** + * Evaluate an ObjectExpression made only of literals. Anything that could + * observe or mutate the outside world (identifiers, calls, spread, getters) + * throws, which the caller reports as "meta must be a pure literal". + */ +function evaluateObjectLiteral(node: AnyNode): Record { + const result: Record = {} + for (const property of node.properties) { + if (property.type === 'SpreadElement') { + throw new Error('spread not allowed in meta') + } + if (property.kind !== 'init' || property.method) { + throw new Error('only plain key: value pairs are allowed in meta') + } + let key: string + if (property.computed) { + throw new Error('computed keys not allowed in meta') + } else if (property.key.type === 'Identifier') { + key = property.key.name + } else if (property.key.type === 'Literal') { + key = String(property.key.value) + } else { + throw new Error('unsupported key in meta') + } + result[key] = evaluateLiteral(property.value) + } + return result +} + +function evaluateLiteral(node: AnyNode): unknown { + switch (node.type) { + case 'Literal': + return node.value + case 'ArrayExpression': + return node.elements.map((element: AnyNode) => { + if (element === null) throw new Error('sparse arrays not allowed') + if (element.type === 'SpreadElement') { + throw new Error('spread not allowed in meta') + } + return evaluateLiteral(element) + }) + case 'ObjectExpression': + return evaluateObjectLiteral(node) + case 'TemplateLiteral': { + if (node.expressions.length > 0) { + throw new Error('template interpolation not allowed in meta') + } + return node.quasis + .map((quasi: AnyNode) => quasi.value.cooked ?? '') + .join('') + } + case 'UnaryExpression': { + if (node.operator === '-' && node.argument.type === 'Literal') { + return -(node.argument.value as number) + } + throw new Error(`unsupported expression: ${node.operator}`) + } + default: + throw new Error(`unsupported expression: ${node.type}`) + } +} + +function validateMeta( + value: unknown, +): { meta: WorkflowMeta } | { error: string } { + if (typeof value !== 'object' || value === null) { + return { error: 'meta must be an object literal' } + } + const raw = value as Record + const name = raw.name + if (typeof name !== 'string' || name.trim() === '') { + return { error: 'meta.name must be a non-empty string' } + } + if (!/^[a-zA-Z0-9][a-zA-Z0-9_-]*$/.test(name)) { + return { + error: `meta.name '${name}' must start with a letter or digit and contain only letters, digits, '-' and '_'`, + } + } + const description = raw.description + if (typeof description !== 'string' || description.trim() === '') { + return { error: 'meta.description must be a non-empty string' } + } + + const meta: WorkflowMeta = { name, description } + if (typeof raw.whenToUse === 'string') meta.whenToUse = raw.whenToUse + if (typeof raw.title === 'string') meta.title = raw.title + if (typeof raw.model === 'string') meta.model = raw.model + + if (raw.phases !== undefined) { + if (!Array.isArray(raw.phases)) { + return { error: 'meta.phases must be an array' } + } + const phases: WorkflowPhaseMeta[] = [] + for (const entry of raw.phases) { + if (typeof entry !== 'object' || entry === null) { + return { error: 'each meta.phases entry must be an object' } + } + const phase = entry as Record + if (typeof phase.title !== 'string' || phase.title.trim() === '') { + return { error: 'each meta.phases entry needs a non-empty title' } + } + const parsed: WorkflowPhaseMeta = { title: phase.title } + if (typeof phase.detail === 'string') parsed.detail = phase.detail + if (typeof phase.model === 'string') parsed.model = phase.model + phases.push(parsed) + } + meta.phases = phases + } + + return { meta } +} + +function formatParseError(error: unknown, script: string): string { + const message = error instanceof Error ? error.message : String(error) + const loc = hasLoc(error) ? error.loc : undefined + const line = loc ? script.split('\n')[loc.line - 1] : undefined + if (!loc || line === undefined) { + return `Script parse error: ${message}. ${PLAIN_JS_HINT}` + } + const column = Math.max(0, Math.min(loc.column, line.length)) + const start = Math.max( + 0, + Math.min(column - Math.floor(CARET_WINDOW / 2), line.length - CARET_WINDOW), + ) + const excerpt = line.slice(start, start + CARET_WINDOW) + const caret = `${' '.repeat(column - start)}^` + return `Script parse error: ${message}\n${excerpt}\n${caret}\n${PLAIN_JS_HINT}` +} + +function hasLoc( + error: unknown, +): error is { loc: { line: number; column: number } } { + if (typeof error !== 'object' || error === null || !('loc' in error)) { + return false + } + const loc = (error as { loc: unknown }).loc + return ( + typeof loc === 'object' && + loc !== null && + 'line' in loc && + typeof (loc as { line: unknown }).line === 'number' && + 'column' in loc && + typeof (loc as { column: unknown }).column === 'number' + ) +} diff --git a/src/utils/workflows/nested.test.ts b/src/utils/workflows/nested.test.ts new file mode 100644 index 00000000..99245579 --- /dev/null +++ b/src/utils/workflows/nested.test.ts @@ -0,0 +1,134 @@ +import { describe, expect, test } from 'bun:test' +import type { CanUseToolFn } from '../../hooks/useCanUseTool.js' +import type { ToolUseContext } from '../../Tool.js' +import { createWorkflowSharedCounters } from './harness.js' +import { executeWorkflowScript, prepareWorkflowScript } from './runtime.js' +import type { + WorkflowAgentRunParams, + WorkflowAgentRunResult, +} from './runWorkflowAgent.js' +import type { WorkflowProgressEvent } from './types.js' + +const CHILD = [ + "export const meta = { name: 'child', description: 'Child run' }", + "const out = await agent('child-' + (args ?? 'none'))", + 'return { child: out }', +].join('\n') + +async function runParent( + parentBody: string, + options: { childScript?: string } = {}, +) { + const parent = prepareWorkflowScript( + `export const meta = { name: 'parent', description: 'Parent run' }\n${parentBody}`, + ) + if (!parent.ok) throw new Error(parent.error) + + const events: WorkflowProgressEvent[] = [] + const shared = createWorkflowSharedCounters() + let seq = 0 + const runAgentImpl = async ( + params: WorkflowAgentRunParams, + ): Promise => ({ + agentId: `a${++seq}`, + value: `ran:${params.prompt}`, + tokens: 1, + toolCalls: 0, + }) + const toolUseContext = { + abortController: new AbortController(), + options: { mainLoopModel: 'test-model' }, + } as unknown as ToolUseContext + const canUseTool = (() => {}) as unknown as CanUseToolFn + + const outcome = await executeWorkflowScript({ + vmScript: parent.vmScript, + toolUseContext, + canUseTool, + runId: 'wf_nested00-abc', + shared, + onProgress: event => events.push(event), + onAgentController: () => {}, + runAgentImpl, + runNestedWorkflow: async (_nameOrRef, args) => { + const child = prepareWorkflowScript(options.childScript ?? CHILD) + if (!child.ok) throw new Error(child.error) + const result = await executeWorkflowScript({ + vmScript: child.vmScript, + toolUseContext, + canUseTool, + runId: 'wf_nested00-abc', + args, + shared, + onProgress: event => events.push(event), + onAgentController: () => {}, + runAgentImpl, + runNestedWorkflow: undefined, + }) + if (result.error) throw new Error(result.error) + return result.result + }, + }) + + return { outcome, events, shared } +} + +describe('nested workflow()', () => { + test('returns the child result and passes args through', async () => { + const { outcome } = await runParent( + "const child = await workflow('child', 'from-parent')\nreturn child", + ) + expect(outcome.error).toBeUndefined() + expect(outcome.result).toEqual({ child: 'ran:child-from-parent' }) + }) + + test('parent and child agents share one index sequence', async () => { + const { outcome, events, shared } = await runParent( + [ + "const a = await agent('parent-1')", + "const child = await workflow('child', 'x')", + "const b = await agent('parent-2')", + 'return [a, child, b]', + ].join('\n'), + ) + expect(outcome.error).toBeUndefined() + + const indexes = [ + ...new Set( + events + .filter(event => event.type === 'workflow_agent') + .map(event => event.index), + ), + ].sort((a, b) => a - b) + // Three agents in total across both scripts, no index reused — a reused + // index would overwrite the parent's row in the progress view. + expect(indexes).toEqual([1, 2, 3]) + expect(shared.getAgentCount()).toBe(3) + }) + + test("the child's failure surfaces as the parent's failure", async () => { + const { outcome } = await runParent( + "return await workflow('child')", + { childScript: "export const meta = { name: 'child', description: 'Boom' }\nthrow new Error('child exploded')\n" }, + ) + expect(outcome.error).toContain('child exploded') + }) + + test('workflow() is unavailable when the runner is not supplied', async () => { + const prepared = prepareWorkflowScript( + "export const meta = { name: 'p', description: 'p' }\nreturn await workflow('child')\n", + ) + if (!prepared.ok) throw new Error(prepared.error) + const outcome = await executeWorkflowScript({ + vmScript: prepared.vmScript, + toolUseContext: { + abortController: new AbortController(), + options: { mainLoopModel: 'test-model' }, + } as unknown as ToolUseContext, + canUseTool: (() => {}) as unknown as CanUseToolFn, + runId: 'wf_nested00-abc', + onAgentController: () => {}, + }) + expect(outcome.error).toBe('workflow() is not available in this run') + }) +}) diff --git a/src/utils/workflows/paths.ts b/src/utils/workflows/paths.ts new file mode 100644 index 00000000..0db79f78 --- /dev/null +++ b/src/utils/workflows/paths.ts @@ -0,0 +1,54 @@ +import { randomBytes } from 'crypto' +import { join } from 'path' +import { + getOriginalCwd, + getSessionId, + getSessionProjectDir, +} from '../../bootstrap/state.js' +import { getClaudeConfigHomeDir } from '../envUtils.js' +import { getProjectDir } from '../sessionStorage.js' + +/** `~/.claude/workflows` — personal workflows, available in every project. */ +export function getUserWorkflowsDir(): string { + return join(getClaudeConfigHomeDir(), 'workflows') +} + +/** + * A fresh run id. Shaped `wf_<8 hex>-<3 hex>` so it is short enough to read in + * the progress view and still collision-free within a session. + */ +export function createWorkflowRunId(): string { + const bytes = randomBytes(6).toString('hex') + return `wf_${bytes.slice(0, 8)}-${bytes.slice(8, 11)}` +} + +/** Subdirectory (relative to `subagents/`) holding this run's agent transcripts. */ +export function getWorkflowTranscriptSubdir(runId: string): string { + return join('workflows', runId) +} + +/** + * Absolute directory for a run's artifacts: the resume journal plus one + * `agent-.jsonl` transcript per subagent. Lives beside the session's other + * subagent transcripts so existing transcript tooling can read it unchanged. + */ +export function getWorkflowTranscriptDir(runId: string): string { + const projectDir = getSessionProjectDir() ?? getProjectDir(getOriginalCwd()) + return join( + projectDir, + getSessionId(), + 'subagents', + getWorkflowTranscriptSubdir(runId), + ) +} + +/** Where the executed script is persisted so a run can be read back or edited. */ +export function getWorkflowScriptPath(runId: string, name: string): string { + const projectDir = getSessionProjectDir() ?? getProjectDir(getOriginalCwd()) + const safeName = name.replace(/[^a-zA-Z0-9._-]/g, '-').slice(0, 64) || 'workflow' + return join(projectDir, getSessionId(), 'workflows', `${safeName}.${runId}.js`) +} + +export function getWorkflowJournalPath(runId: string): string { + return join(getWorkflowTranscriptDir(runId), 'journal.jsonl') +} diff --git a/src/utils/workflows/pluginWorkflows.ts b/src/utils/workflows/pluginWorkflows.ts new file mode 100644 index 00000000..e11225e4 --- /dev/null +++ b/src/utils/workflows/pluginWorkflows.ts @@ -0,0 +1,51 @@ +import { stat } from 'fs/promises' +import { dirname } from 'path' +import { logForDebugging } from '../debug.js' +import { loadAllPluginsCacheOnly } from '../plugins/pluginLoader.js' +import { loadWorkflowsFromDir } from './discovery.js' +import type { WorkflowDefinition } from './types.js' + +/** + * Dynamic workflows shipped by enabled plugins. + * + * Names are namespaced `plugin:workflow` — two plugins can both ship a + * `release-audit` and neither has to lose. Reads the plugin cache rather than + * re-scanning: this runs on every `/` autocomplete. + */ +export async function loadPluginWorkflows(): Promise { + let plugins: Awaited>['plugins'] + try { + plugins = (await loadAllPluginsCacheOnly()).plugins ?? [] + } catch (error) { + logForDebugging( + `loadPluginWorkflows: plugin cache unavailable: ${error instanceof Error ? error.message : String(error)}`, + ) + return [] + } + + const found: WorkflowDefinition[] = [] + for (const plugin of plugins) { + if (plugin.enabled === false) continue + const dirs = [ + ...(plugin.workflowsPath ? [plugin.workflowsPath] : []), + ...(plugin.workflowsPaths ?? []), + ] + for (const path of dirs) { + const dir = await resolveDirectory(path) + if (!dir) continue + found.push( + ...(await loadWorkflowsFromDir(dir, 'plugin', plugin.manifest.name)), + ) + } + } + return found +} + +/** A manifest entry may point at a single `.js` file rather than a directory. */ +async function resolveDirectory(path: string): Promise { + try { + return (await stat(path)).isDirectory() ? path : dirname(path) + } catch { + return undefined + } +} diff --git a/src/utils/workflows/reminders.test.ts b/src/utils/workflows/reminders.test.ts new file mode 100644 index 00000000..50149d05 --- /dev/null +++ b/src/utils/workflows/reminders.test.ts @@ -0,0 +1,60 @@ +import { describe, expect, test } from 'bun:test' +import { normalizeAttachmentForAPI } from '../messages.js' +import { + ULTRACODE_ENTER_REMINDER, + ULTRACODE_EXIT_REMINDER, + ULTRACODE_STILL_ON_REMINDER, + WORKFLOW_KEYWORD_REMINDER, +} from './ultracode.js' + +/** + * The reminders are the only channel that tells the model a turn was opted + * into orchestration. If one stops rendering, the Workflow tool's opt-in rule + * silently stops firing and nothing else in the system notices. + */ +function renderedText( + attachment: Parameters[0], +): string { + return normalizeAttachmentForAPI(attachment) + .map(message => + typeof message.message.content === 'string' + ? message.message.content + : message.message.content + .map(block => ('text' in block ? block.text : '')) + .join('\n'), + ) + .join('\n') +} + +describe('workflow reminders', () => { + test('the keyword reminder names the tool the model must call', () => { + const text = renderedText({ type: 'workflow_keyword_request' }) + expect(text).toContain(WORKFLOW_KEYWORD_REMINDER) + expect(text).toContain('Workflow tool') + expect(text).toContain('') + }) + + test('entering ultracode gets the full standing instruction', () => { + const text = renderedText({ + type: 'ultra_effort_enter', + reminderType: 'full', + }) + expect(text).toContain(ULTRACODE_ENTER_REMINDER) + expect(text).toContain('every substantive task') + }) + + test('staying in ultracode gets the short form, not the paragraph', () => { + const text = renderedText({ + type: 'ultra_effort_enter', + reminderType: 'short', + }) + expect(text).toContain(ULTRACODE_STILL_ON_REMINDER) + expect(text).not.toContain('token cost is not a constraint') + }) + + test('leaving ultracode restores the opt-in rule explicitly', () => { + const text = renderedText({ type: 'ultra_effort_exit' }) + expect(text).toContain(ULTRACODE_EXIT_REMINDER) + expect(text).toContain('opt-in rule applies again') + }) +}) diff --git a/src/utils/workflows/resume.test.ts b/src/utils/workflows/resume.test.ts new file mode 100644 index 00000000..67d715c5 --- /dev/null +++ b/src/utils/workflows/resume.test.ts @@ -0,0 +1,142 @@ +import { describe, expect, test } from 'bun:test' +import { mkdtemp, rm } from 'fs/promises' +import { tmpdir } from 'os' +import { join } from 'path' +import type { CanUseToolFn } from '../../hooks/useCanUseTool.js' +import type { ToolUseContext } from '../../Tool.js' +import { createWorkflowHarness } from './harness.js' +import { WorkflowJournal } from './journal.js' +import type { + WorkflowAgentRunParams, + WorkflowAgentRunResult, +} from './runWorkflowAgent.js' +import type { WorkflowProgressEvent } from './types.js' + +function makeHarness(options: { + journal?: WorkflowJournal + journalSnapshot?: Awaited> + onRun?: (params: WorkflowAgentRunParams) => void +}) { + const events: WorkflowProgressEvent[] = [] + let seq = 0 + const harness = createWorkflowHarness({ + toolUseContext: { + options: { mainLoopModel: 'test-model' }, + } as unknown as ToolUseContext, + canUseTool: (() => {}) as unknown as CanUseToolFn, + runId: 'wf_temp0000-aaa', + emit: event => events.push(event), + onAgentController: () => {}, + journal: options.journal, + journalSnapshot: options.journalSnapshot, + runAgentImpl: async ( + params: WorkflowAgentRunParams, + ): Promise => { + options.onRun?.(params) + return { + agentId: `live-${++seq}`, + value: `live:${params.prompt}`, + tokens: 1, + toolCalls: 0, + } + }, + }) + return { harness, events } +} + +describe('workflow resume', () => { + test('replays cached results and stops replaying at the first miss', async () => { + const dir = await mkdtemp(join(tmpdir(), 'wf-journal-')) + try { + const journal = new WorkflowJournal(join(dir, 'journal.jsonl')) + + // First run: three agents, all recorded. + const first = makeHarness({ journal }) + const a1 = await first.harness.agent('one') + const a2 = await first.harness.agent('two') + const a3 = await first.harness.agent('three') + expect([a1, a2, a3]).toEqual(['live:one', 'live:two', 'live:three']) + + await journal.flush() + const snapshot = await journal.load() + expect(snapshot.results.size).toBe(3) + + // Second run: the middle prompt changed, so it and everything after it + // must run live again even though the third call is byte-identical. + const liveRuns: string[] = [] + const second = makeHarness({ + journal, + journalSnapshot: snapshot, + onRun: params => liveRuns.push(params.prompt), + }) + const b1 = await second.harness.agent('one') + const b2 = await second.harness.agent('CHANGED') + const b3 = await second.harness.agent('three') + + expect(b1).toBe('live:one') + expect(liveRuns).toEqual(['CHANGED', 'three']) + expect(b2).toBe('live:CHANGED') + expect(b3).toBe('live:three') + + const cachedEvents = second.events.filter( + event => event.type === 'workflow_agent' && event.cached === true, + ) + expect(cachedEvents).toHaveLength(1) + } finally { + await rm(dir, { recursive: true, force: true }) + } + }) + + test('an unchanged script replays every agent from cache', async () => { + const dir = await mkdtemp(join(tmpdir(), 'wf-journal-')) + try { + const journal = new WorkflowJournal(join(dir, 'journal.jsonl')) + const first = makeHarness({ journal }) + await first.harness.agent('one') + await first.harness.agent('two') + + await journal.flush() + const snapshot = await journal.load() + const liveRuns: string[] = [] + const second = makeHarness({ + journal, + journalSnapshot: snapshot, + onRun: params => liveRuns.push(params.prompt), + }) + expect(await second.harness.agent('one')).toBe('live:one') + expect(await second.harness.agent('two')).toBe('live:two') + expect(liveRuns).toEqual([]) + expect(second.harness.getAgentCount()).toBe(2) + } finally { + await rm(dir, { recursive: true, force: true }) + } + }) + + test('an agent that started but never finished is not replayed', async () => { + const dir = await mkdtemp(join(tmpdir(), 'wf-journal-')) + try { + const path = join(dir, 'journal.jsonl') + const journal = new WorkflowJournal(path) + await journal.append({ + type: 'started', + key: '|one|null', + agentId: 'interrupted', + }) + + const snapshot = await journal.load() + expect(snapshot.results.size).toBe(0) + expect(snapshot.started.get('|one|null')).toEqual(['interrupted']) + + const liveRuns: string[] = [] + const { harness } = makeHarness({ + journal, + journalSnapshot: snapshot, + onRun: params => liveRuns.push(params.prompt), + }) + await harness.agent('one') + expect(liveRuns).toEqual(['one']) + } finally { + await rm(dir, { recursive: true, force: true }) + } + }) +}) diff --git a/src/utils/workflows/runWorkflowAgent.ts b/src/utils/workflows/runWorkflowAgent.ts new file mode 100644 index 00000000..7bfb2cf2 --- /dev/null +++ b/src/utils/workflows/runWorkflowAgent.ts @@ -0,0 +1,336 @@ +import type { CanUseToolFn } from '../../hooks/useCanUseTool.js' +import type { Tool, ToolUseContext } from '../../Tool.js' +import { toolMatchesName } from '../../Tool.js' +import { + createActivityDescriptionResolver, + createProgressTracker, + getTokenCountFromTracker, + updateProgressFromMessage, +} from '../../tasks/LocalAgentTask/LocalAgentTask.js' +import type { + AgentDefinition, + BuiltInAgentDefinition, +} from '../../tools/AgentTool/loadAgentsDir.js' +import { getLastToolUseName } from '../../tools/AgentTool/agentToolUtils.js' +import { runAgent } from '../../tools/AgentTool/runAgent.js' +import { + createSyntheticOutputTool, + SYNTHETIC_OUTPUT_TOOL_NAME, +} from '../../tools/SyntheticOutputTool/SyntheticOutputTool.js' +import { assembleToolPool } from '../../tools.js' +import type { Message } from '../../types/message.js' +import { createAbortController } from '../abortController.js' +import { runWithCwdOverride } from '../cwd.js' +import { logForDebugging } from '../debug.js' +import { createUserMessage, extractTextContent } from '../messages.js' +import { createAgentId } from '../uuid.js' +import type { AgentMetadata } from '../sessionStorage.js' +import { + createAgentWorktree, + hasWorktreeChanges, + removeAgentWorktree, +} from '../worktree.js' +import { + WORKFLOW_SUBAGENT_PROMPT, + WORKFLOW_SUBAGENT_TYPE, + workflowStructuredOutputNote, + workflowStructuredSubagentPrompt, +} from './constants.js' +import type { WorkflowAgentOptions } from './types.js' + +/** + * The agent definition every workflow subagent runs under. + * + * `permissionMode: 'acceptEdits'` is deliberate and independent of the + * session's mode: the script is the orchestrator and there is nobody to + * approve each of a hundred file edits. Tool calls still go through the + * session allowlist, so shell/MCP calls can prompt. + */ +function buildWorkflowAgentDefinition( + schemaToolName: string | undefined, +): BuiltInAgentDefinition { + return { + agentType: WORKFLOW_SUBAGENT_TYPE, + whenToUse: 'Internal subagent for workflow script orchestration.', + tools: ['*'], + source: 'built-in', + baseDir: 'built-in', + permissionMode: 'acceptEdits', + getSystemPrompt: () => + schemaToolName + ? workflowStructuredSubagentPrompt(schemaToolName) + : WORKFLOW_SUBAGENT_PROMPT, + } +} + + +export type WorkflowAgentRunParams = { + prompt: string + opts?: WorkflowAgentOptions + toolUseContext: ToolUseContext + canUseTool: CanUseToolFn + runId: string + /** Workflow/phase provenance stamped onto the agent's sidecar metadata. */ + workflow?: AgentMetadata['workflow'] + /** Per-agent controller so the user can skip or restart a single agent. */ + abortController: AbortController + /** Called once the agent id is known, before the first model request. */ + onAgentId: (agentId: string) => void + /** Called on every assistant message with the running token/tool totals. */ + onProgress: (progress: { + tokens: number + toolCalls: number + lastToolName?: string + }) => void +} + +export type WorkflowAgentRunResult = { + agentId: string + /** Structured output object when a schema was given, otherwise the final text. */ + value: unknown + tokens: number + toolCalls: number + worktreePath?: string +} + +/** + * Run one `agent()` call to completion and return its value to the script. + * + * With a `schema`, the agent is forced through `StructuredOutput` and the + * validated object is returned — validation happens at the tool-call layer so + * the model retries a bad shape itself instead of the script parsing prose. + */ +export async function runWorkflowAgent( + params: WorkflowAgentRunParams, +): Promise { + const { + prompt, + opts, + toolUseContext, + canUseTool, + runId, + workflow, + abortController, + onAgentId, + onProgress, + } = params + + const agentDefinition = resolveAgentDefinition(toolUseContext, opts) + const schemaTool = buildSchemaTool(opts?.schema) + if (schemaTool && 'error' in schemaTool) { + throw new Error(`agent({schema}): ${schemaTool.error}`) + } + + const appState = toolUseContext.getAppState() + const workerPermissionContext = { + ...appState.toolPermissionContext, + mode: agentDefinition.permissionMode ?? ('acceptEdits' as const), + } + const basePool = assembleToolPool(workerPermissionContext, appState.mcp.tools) + // A parent StructuredOutput (from --json-schema) carries a different schema; + // leaving it in would let the agent satisfy the wrong contract. + const availableTools: Tool[] = schemaTool + ? [ + ...basePool.filter( + tool => !toolMatchesName(tool, SYNTHETIC_OUTPUT_TOOL_NAME), + ), + schemaTool.tool, + ] + : [...basePool] + + const agentId = createAgentId() + onAgentId(agentId) + + const promptText = schemaTool + ? `${prompt}${workflowStructuredOutputNote(SYNTHETIC_OUTPUT_TOOL_NAME)}` + : prompt + + let worktree: Awaited> | null = null + if (opts?.isolation === 'worktree') { + worktree = await createAgentWorktree(`agent-${agentId.slice(0, 8)}`) + } + + const tracker = createProgressTracker() + const resolveActivity = createActivityDescriptionResolver( + toolUseContext.options.tools, + ) + const messages: Message[] = [] + let structuredOutput: unknown + + const runInCwd = (fn: () => T): T => + worktree ? runWithCwdOverride(worktree.worktreePath, fn) : fn() + + try { + const iterator = runInCwd(() => + runAgent({ + agentDefinition, + promptMessages: [createUserMessage({ content: promptText })], + toolUseContext: { + ...toolUseContext, + abortController, + agentId, + agentType: agentDefinition.agentType, + options: { + ...toolUseContext.options, + tools: availableTools, + ...(opts?.model ? { mainLoopModel: opts.model } : {}), + }, + }, + canUseTool, + isAsync: false, + canShowPermissionPrompts: true, + querySource: 'workflow_agent', + model: opts?.model, + availableTools, + description: opts?.label ?? prompt.slice(0, 60), + ...(workflow ? { workflow } : {}), + // Deliberately no `transcriptSubdir`. A workflow agent is an ordinary + // subagent run by the same runner, so its transcript belongs in the + // session's flat `subagents/` directory with every other one — that is + // the only place the reader looks, and routing these into + // `subagents/workflows//` is what made them unopenable in the + // UI. The run's journal still lives in the per-run directory. + ...(worktree ? { worktreePath: worktree.worktreePath } : {}), + override: { agentId, abortController }, + }), + ) + + for await (const message of iterator) { + messages.push(message) + if ( + message.type === 'attachment' && + message.attachment.type === 'structured_output' + ) { + structuredOutput = message.attachment.data + continue + } + if (message.type !== 'assistant') continue + updateProgressFromMessage( + tracker, + message, + resolveActivity, + toolUseContext.options.tools, + ) + onProgress({ + tokens: getTokenCountFromTracker(tracker), + toolCalls: tracker.toolUseCount, + lastToolName: getLastToolUseName(message), + }) + } + } finally { + if (worktree) { + await cleanupWorkflowWorktree(worktree).catch(error => + logForDebugging( + `Workflow worktree cleanup failed: ${error instanceof Error ? error.message : String(error)}`, + ), + ) + } + } + + const value = schemaTool + ? structuredOutput + : extractFinalText(messages) + + if (schemaTool && structuredOutput === undefined) { + throw new Error( + `agent({schema}): the subagent ended without calling ${SYNTHETIC_OUTPUT_TOOL_NAME}`, + ) + } + + return { + agentId, + value: value ?? null, + tokens: getTokenCountFromTracker(tracker), + toolCalls: tracker.toolUseCount, + ...(worktree ? { worktreePath: worktree.worktreePath } : {}), + } +} + +/** + * Resolve `opts.agentType` against the session's active agents. + * + * A named agent keeps its own system prompt and tool list — the workflow only + * supplies the prompt — so a script can reuse `code-reviewer` or a custom + * agent instead of the generic workflow subagent. + */ +function resolveAgentDefinition( + toolUseContext: ToolUseContext, + opts: WorkflowAgentOptions | undefined, +): AgentDefinition { + const requested = opts?.agentType + const schemaToolName = opts?.schema ? SYNTHETIC_OUTPUT_TOOL_NAME : undefined + if (requested == null) return buildWorkflowAgentDefinition(schemaToolName) + + const activeAgents = toolUseContext.options.agentDefinitions.activeAgents + const match = activeAgents.find(agent => agent.agentType === requested) + if (!match) { + const available = activeAgents.map(agent => agent.agentType).join(', ') + throw new Error( + `agent({agentType}): agent type '${requested}' not found. Available agents: ${available}`, + ) + } + if (!schemaToolName) return match + // Custom agent + schema: keep its prompt but append the StructuredOutput + // instruction so the two contracts don't fight. + const basePrompt = match.getSystemPrompt({ + toolUseContext: { options: toolUseContext.options }, + } as never) + return { + ...match, + getSystemPrompt: () => + `${basePrompt}\n${workflowStructuredOutputNote(schemaToolName)}`, + } as AgentDefinition +} + +function buildSchemaTool( + schema: unknown, +): { tool: Tool } | { error: string } | undefined { + if (schema == null) return undefined + if (typeof schema !== 'object' || Array.isArray(schema)) { + return { error: 'schema must be a JSON Schema object' } + } + return createSyntheticOutputTool(schema as Record) as + | { tool: Tool } + | { error: string } +} + +/** + * The agent's final text. Falls back to the last assistant message that had + * any text, so a run that ended on a bare tool_use block still returns + * something instead of an empty string. + */ +function extractFinalText(messages: Message[]): string { + for (let i = messages.length - 1; i >= 0; i--) { + const message = messages[i] + if (message?.type !== 'assistant') continue + const text = extractTextContent(message.message.content, '\n').trim() + if (text !== '') return text + } + return '' +} + +async function cleanupWorkflowWorktree( + worktree: Awaited>, +): Promise { + const { worktreePath, worktreeBranch, headCommit, gitRoot, hookBased } = + worktree + if (hookBased) return + if (!headCommit) return + if (await hasWorktreeChanges(worktreePath, headCommit)) return + await removeAgentWorktree(worktreePath, worktreeBranch, gitRoot) +} + +/** Exposed so the harness can create per-agent controllers consistently. */ +export function createWorkflowAgentController( + parent: AbortSignal | undefined, +): AbortController { + const controller = createAbortController() + if (parent) { + if (parent.aborted) controller.abort(parent.reason) + else + parent.addEventListener('abort', () => controller.abort(parent.reason), { + once: true, + }) + } + return controller +} diff --git a/src/utils/workflows/runtime.test.ts b/src/utils/workflows/runtime.test.ts new file mode 100644 index 00000000..79c5bbaf --- /dev/null +++ b/src/utils/workflows/runtime.test.ts @@ -0,0 +1,287 @@ +import { describe, expect, test } from 'bun:test' +import type { ToolUseContext } from '../../Tool.js' +import type { CanUseToolFn } from '../../hooks/useCanUseTool.js' +import { executeWorkflowScript, prepareWorkflowScript } from './runtime.js' +import type { + WorkflowAgentRunParams, + WorkflowAgentRunResult, +} from './runWorkflowAgent.js' +import type { WorkflowProgressEvent } from './types.js' + +/** + * Drive a whole script through the real VM sandbox with a scripted stand-in + * for the subagent. The harness/runtime seam is what these tests own; the + * model call itself is covered by the CLI smoke lane. + */ +async function run( + script: string, + options: { + agent?: (params: WorkflowAgentRunParams) => Promise + args?: unknown + abortController?: AbortController + } = {}, +) { + const prepared = prepareWorkflowScript(script) + if (!prepared.ok) throw new Error(prepared.error) + + const events: WorkflowProgressEvent[] = [] + const abortController = options.abortController ?? new AbortController() + let seq = 0 + + const outcome = await executeWorkflowScript({ + vmScript: prepared.vmScript, + toolUseContext: { + abortController, + options: { mainLoopModel: 'test-model' }, + } as unknown as ToolUseContext, + canUseTool: (() => {}) as unknown as CanUseToolFn, + runId: 'wf_test0000-abc', + workflowName: prepared.meta.name, + args: options.args, + seedPhaseTitles: prepared.meta.phases?.map(phase => phase.title), + onProgress: event => events.push(event), + onAgentController: () => {}, + runAgentImpl: async ( + params: WorkflowAgentRunParams, + ): Promise => { + const value = options.agent + ? await options.agent(params) + : `ran: ${params.prompt}` + return { + agentId: `agent-${++seq}`, + value, + tokens: 10, + toolCalls: 1, + } + }, + }) + return { outcome, events, meta: prepared.meta } +} + +const META = "export const meta = { name: 'demo', description: 'A demo' }\n" + +describe('executeWorkflowScript', () => { + test('returns the script result and counts the agents it spawned', async () => { + const { outcome } = await run( + `${META} + const a = await agent('first') + const b = await agent('second') + return { a, b }`, + ) + expect(outcome.error).toBeUndefined() + expect(outcome.result).toEqual({ a: 'ran: first', b: 'ran: second' }) + expect(outcome.agentCount).toBe(2) + }) + + test('pipeline threads each item through every stage', async () => { + const { outcome } = await run( + `${META} + const out = await pipeline( + ['a.ts', 'b.ts'], + file => agent('review ' + file, { label: file, phase: 'Review' }), + (review, original, index) => agent('verify ' + original + '#' + index, { phase: 'Verify' }), + ) + return out`, + { agent: async params => params.prompt }, + ) + expect(outcome.result).toEqual([ + 'verify a.ts#0', + 'verify b.ts#1', + ]) + expect(outcome.agentCount).toBe(4) + }) + + test('a failing slot becomes null instead of failing the whole batch', async () => { + const { outcome } = await run( + `${META} + const out = await parallel([ + () => agent('ok'), + () => agent('boom'), + ]) + return out`, + { + agent: async params => { + if (params.prompt === 'boom') throw new Error('agent exploded') + return 'fine' + }, + }, + ) + expect(outcome.result).toEqual(['fine', null]) + expect(outcome.failures).toEqual(['parallel[1] failed: agent exploded']) + }) + + test('parallel rejects promises passed instead of thunks', async () => { + const { outcome } = await run( + `${META} + return await parallel([agent('a')])`, + ) + expect(outcome.error).toContain('Wrap each call: () => agent(...)') + }) + + test('emits phase rows for meta.phases and for phase() calls', async () => { + const { events } = await run( + `export const meta = { name: 'demo', description: 'A demo', phases: [{ title: 'Scan' }] } + phase('Scan') + await agent('one') + phase('Fix') + await agent('two') + return null`, + ) + const phases = events.filter(event => event.type === 'workflow_phase') + expect(phases).toEqual([ + { type: 'workflow_phase', index: 1, title: 'Scan', kind: 'meta' }, + { type: 'workflow_phase', index: 2, title: 'Fix', kind: 'script' }, + ]) + const agentPhases = events + .filter(event => event.type === 'workflow_agent' && event.state === 'done') + .map(event => (event.type === 'workflow_agent' ? event.phaseTitle : null)) + expect(agentPhases).toEqual(['Scan', 'Fix']) + }) + + test('log() and console.log both reach the progress stream', async () => { + const { outcome, events } = await run( + `${META} + log('from log') + console.log('from console', { n: 1 }) + return 'done'`, + ) + expect(outcome.logs).toEqual(['from log', 'from console {"n":1}']) + expect( + events.filter(event => event.type === 'workflow_log'), + ).toHaveLength(2) + }) + + test('args arrive as structured data, not a JSON string', async () => { + const { outcome } = await run( + `${META} + return args.filter(n => n > 1)`, + { args: [1, 2, 3] }, + ) + expect(outcome.result).toEqual([2, 3]) + }) + + test('args is undefined when the caller passed none', async () => { + const { outcome } = await run(`${META} + return typeof args`) + expect(outcome.result).toBe('undefined') + }) + + test('budget.remaining is Infinity with no target', async () => { + const { outcome } = await run(`${META} + return { total: budget.total, infinite: budget.remaining() === Infinity }`) + expect(outcome.result).toEqual({ total: null, infinite: true }) + }) + + test('Date.now, new Date and Math.random are unavailable', async () => { + for (const expression of ['Date.now()', 'new Date()', 'Math.random()']) { + const { outcome } = await run(`${META} + return ${expression}`) + expect(outcome.error).toContain('unavailable in workflow scripts') + } + }) + + test('a Date built from an explicit timestamp still works', async () => { + const { outcome } = await run(`${META} + return new Date(0).toISOString()`) + expect(outcome.result).toBe('1970-01-01T00:00:00.000Z') + }) + + test('eval and new Function are blocked', async () => { + const { outcome } = await run(`${META} + return eval('1 + 1')`) + expect(outcome.error).toBeDefined() + }) + + test('script values cannot reach the host realm through a constructor', async () => { + const { outcome } = await run( + `${META} + const out = await parallel([() => agent('a')]) + const Ctor = out.constructor.constructor + return typeof Ctor('return process')().version`, + ) + // codeGeneration is disabled, so the Function constructor cannot compile — + // and out.constructor is the VM's Array, not the host's. + expect(outcome.error).toBeDefined() + }) + + test('import() is refused before the run starts', () => { + const prepared = prepareWorkflowScript(`${META} + const mod = await import('fs') + return mod`) + expect(prepared.ok).toBe(true) + }) + + test('a rejected agent propagates when awaited directly', async () => { + const { outcome } = await run( + `${META} + return await agent('boom')`, + { + agent: async () => { + throw new Error('agent exploded') + }, + }, + ) + expect(outcome.error).toBe('agent exploded') + }) + + test('a synchronous infinite loop is cut off by the sync timeout', async () => { + const prepared = prepareWorkflowScript(`${META} + while (true) {}`) + if (!prepared.ok) throw new Error(prepared.error) + const outcome = await executeWorkflowScript({ + vmScript: prepared.vmScript, + toolUseContext: { + abortController: new AbortController(), + options: { mainLoopModel: 'test-model' }, + } as unknown as ToolUseContext, + canUseTool: (() => {}) as unknown as CanUseToolFn, + runId: 'wf_test0000-abc', + onAgentController: () => {}, + syncTimeoutMs: 200, + }) + expect(outcome.error).toContain('timed out') + }) + + test('a workflow result that is a function is rejected', async () => { + const { outcome } = await run(`${META} + return () => 1`) + expect(outcome.error).toBe('workflow result cannot be a function') + }) +}) + +describe('workflow agent provenance', () => { + test('stamps every agent with the run, name and phase it belongs to', async () => { + // This is the only record of a run's shape once the process exits — the + // desktop rebuilds a finished run's phases from it. It shipped with + // `name` silently undefined because nothing type-checks src/. + const seen: Array = [] + await run( + `export const meta = { name: 'demo', description: 'A demo' } + phase('Scan') + await agent('one', { label: 'scan:one' }) + phase('Verify') + await agent('two', { label: 'verify:two' })`, + { + agent: async (params: WorkflowAgentRunParams) => { + seen.push(params.workflow) + return 'ok' + }, + }, + ) + + expect(seen).toHaveLength(2) + expect(seen[0]).toMatchObject({ + runId: 'wf_test0000-abc', + name: 'demo', + phaseTitle: 'Scan', + agentIndex: 1, + }) + expect(seen[1]).toMatchObject({ + name: 'demo', + phaseTitle: 'Verify', + agentIndex: 2, + }) + // Distinct phases must get distinct indices or the rebuild collapses them. + expect(seen[0]!.phaseIndex).not.toBe(seen[1]!.phaseIndex) + }) +}) diff --git a/src/utils/workflows/runtime.ts b/src/utils/workflows/runtime.ts new file mode 100644 index 00000000..565e9412 --- /dev/null +++ b/src/utils/workflows/runtime.ts @@ -0,0 +1,404 @@ +import vm from 'vm' +import type { CanUseToolFn } from '../../hooks/useCanUseTool.js' +import type { ToolUseContext } from '../../Tool.js' +import { logForDebugging } from '../debug.js' +import { compileWorkflowScript, installDeterminismGuards } from './compile.js' +import { + WORKFLOW_MAX_COLLECTED_LOGS, + WORKFLOW_SYNC_TIMEOUT_MS, +} from './constants.js' +import { describeThrown } from './errors.js' +import { + createWorkflowHarness, + createWorkflowSharedCounters, + type WorkflowHarness, + type WorkflowHarnessParams, + type WorkflowSharedCounters, +} from './harness.js' +import type { WorkflowJournal, WorkflowJournalSnapshot } from './journal.js' +import { parseWorkflowScript } from './meta.js' +import type { + WorkflowMeta, + WorkflowProgressEvent, + WorkflowRunOutcome, + WorkflowTokenBudget, +} from './types.js' + +export type WorkflowExecutionParams = { + vmScript: vm.Script + toolUseContext: ToolUseContext + canUseTool: CanUseToolFn + runId: string + /** `meta.name`, persisted with each agent so a finished run is identifiable. */ + workflowName: string + args?: unknown + seedPhaseTitles?: string[] + tokenBudget?: WorkflowTokenBudget + journal?: WorkflowJournal + journalSnapshot?: WorkflowJournalSnapshot + onProgress?: (event: WorkflowProgressEvent) => void + onAgentController: ( + agentKey: string, + controller: AbortController | undefined, + ) => void + /** Runs a saved workflow inline; `undefined` disables the `workflow()` global. */ + runNestedWorkflow?: (nameOrRef: unknown, args: unknown) => Promise + syncTimeoutMs?: number + /** Overridable so tests can drive a whole run without a live model. */ + runAgentImpl?: WorkflowHarnessParams['runAgentImpl'] + /** Concurrency and agent-index counters, shared with any nested run. */ + shared?: WorkflowSharedCounters +} + +/** + * Execute a compiled workflow script and return its result. + * + * The script runs in a `node:vm` context that holds nothing but the harness + * globals. Every value that crosses the boundary is rebuilt on the far side: + * handing VM code a live host object would expose `obj.constructor.constructor` + * — the host `Function` constructor — and with it the whole process. + */ +export async function executeWorkflowScript( + params: WorkflowExecutionParams, +): Promise { + const startedAt = Date.now() + const logs: string[] = [] + const abortSignal = params.toolUseContext.abortController?.signal + + const emit = (event: WorkflowProgressEvent): void => { + if ( + event.type === 'workflow_log' && + logs.length < WORKFLOW_MAX_COLLECTED_LOGS + ) { + logs.push(event.message) + } + params.onProgress?.(event) + } + + const harness = createWorkflowHarness({ + toolUseContext: params.toolUseContext, + canUseTool: params.canUseTool, + runId: params.runId, + workflowName: params.workflowName, + emit, + seedPhaseTitles: params.seedPhaseTitles, + tokenBudget: params.tokenBudget, + journal: params.journal, + journalSnapshot: params.journalSnapshot, + onAgentController: params.onAgentController, + abortSignal, + runAgentImpl: params.runAgentImpl, + shared: params.shared ?? createWorkflowSharedCounters(), + }) + + const sandbox = createWorkflowSandbox({ + harness, + tokenBudget: params.tokenBudget, + args: params.args, + emit, + abortSignal, + runNestedWorkflow: params.runNestedWorkflow, + }) + + let detachAbort: (() => void) | undefined + try { + const pending = params.vmScript.runInContext(sandbox.context, { + timeout: params.syncTimeoutMs ?? WORKFLOW_SYNC_TIMEOUT_MS, + }) + const settled = sandbox.awaitInVm(pending) as Promise<{ v: unknown }> + settled.catch(() => {}) + + let raced: { v: unknown } + if (abortSignal) { + raced = await Promise.race([ + settled, + new Promise((_resolve, reject) => { + const onAbort = () => reject(new Error('Workflow aborted')) + if (abortSignal.aborted) onAbort() + else { + abortSignal.addEventListener('abort', onAbort, { once: true }) + detachAbort = () => + abortSignal.removeEventListener('abort', onAbort) + } + }), + ]) + } else { + raced = await settled + } + + const result = sandbox.exportValue(raced.v) + await params.journal?.flush() + return { + result, + agentCount: harness.getAgentCount(), + logs, + failures: harness.getFailures(), + durationMs: Date.now() - startedAt, + } + } catch (error) { + const message = describeError(error) + logForDebugging(`Workflow ${params.runId} script error: ${message}`) + await params.journal?.flush().catch(() => {}) + return { + result: null, + agentCount: harness.getAgentCount(), + logs, + failures: harness.getFailures(), + durationMs: Date.now() - startedAt, + error: message, + } + } finally { + detachAbort?.() + sandbox.dispose() + } +} + +type SandboxParams = { + harness: WorkflowHarness + tokenBudget?: WorkflowTokenBudget + args?: unknown + emit: (event: WorkflowProgressEvent) => void + abortSignal?: AbortSignal + runNestedWorkflow?: (nameOrRef: unknown, args: unknown) => Promise +} + +type WorkflowSandbox = { + context: vm.Context + awaitInVm: (value: unknown) => Promise<{ v: unknown }> + exportValue: (value: unknown) => unknown + dispose: () => void +} + +function createWorkflowSandbox(params: SandboxParams): WorkflowSandbox { + const { harness, tokenBudget, emit, abortSignal, runNestedWorkflow } = params + + // codeGeneration off: `eval` and `new Function` inside the script would + // otherwise reconstruct anything the sandbox withholds. + const context = vm.createContext(Object.create(null) as object, { + codeGeneration: { strings: false, wasm: false }, + }) + installDeterminismGuards(context) + + const evalInVm = (source: string): T => + vm.runInContext(source, context, { filename: 'workflow-bridge.js' }) as T + + const awaitInVm = evalInVm<(value: unknown) => Promise<{ v: unknown }>>( + '(async v => ({ __proto__: null, v: await v }))', + ) + const parseInVm = evalInVm<(json: string) => unknown>('(json => JSON.parse(json))') + const stringifyInVm = evalInVm<(value: unknown) => string | undefined>( + '(value => JSON.stringify(value, (_k, v) => typeof v === "function" ? undefined : v))', + ) + const newArrayInVm = evalInVm<(length: number) => unknown[]>( + '(length => new Array(length))', + ) + const setIndexInVm = evalInVm< + (array: unknown, index: number, value: unknown) => void + >('((array, index, value) => { array[index] = value })') + // Host functions are never exposed directly: the script would reach the host + // realm through `fn.constructor`. Each one is re-wrapped by a VM arrow whose + // closure holds the host reference out of reach. + const wrapAsyncInVm = evalInVm< + (hostFn: (...args: unknown[]) => Promise) => unknown + >('(hostFn => async (...args) => hostFn(...args))') + const wrapSyncInVm = evalInVm<(hostFn: (...args: unknown[]) => void) => unknown>( + '(hostFn => (...args) => { hostFn(...args) })', + ) + + /** Rebuild a host value inside the VM realm. */ + const intake = (value: unknown): unknown => { + if (value === undefined) return undefined + if (value === null) return null + const primitive = typeof value + if (primitive === 'string' || primitive === 'number' || primitive === 'boolean') { + return value + } + let json: string | undefined + try { + json = JSON.stringify(value) + } catch { + json = undefined + } + if (json === undefined) return null + return parseInVm(json) + } + + /** Rebuild a VM array of already-VM elements without copying the elements. */ + const intakeArray = (values: unknown[]): unknown[] => { + const array = newArrayInVm(values.length) + for (let i = 0; i < values.length; i++) setIndexInVm(array, i, values[i]) + return array + } + + const timers = new Set() + const clearAllTimers = (): void => { + for (const timer of timers) clearTimeout(timer) + timers.clear() + } + abortSignal?.addEventListener('abort', clearAllTimers, { once: true }) + + const defineGlobal = (name: string, value: unknown): void => { + Object.defineProperty(context, name, { + value, + writable: true, + enumerable: true, + configurable: true, + }) + } + + defineGlobal( + 'agent', + wrapAsyncInVm(async (prompt, opts) => + intake(await harness.agent(prompt, opts)), + ), + ) + defineGlobal( + 'parallel', + wrapAsyncInVm(async thunks => intakeArray(await harness.parallel(thunks))), + ) + defineGlobal( + 'pipeline', + wrapAsyncInVm(async (items, ...stages) => + intakeArray(await harness.pipeline(items, ...stages)), + ), + ) + defineGlobal('log', wrapSyncInVm(message => harness.log(message))) + defineGlobal('phase', wrapSyncInVm(title => harness.phase(title))) + defineGlobal( + 'workflow', + wrapAsyncInVm(async (nameOrRef, args) => { + if (!runNestedWorkflow) { + throw new Error('workflow() is not available in this run') + } + return intake(await runNestedWorkflow(nameOrRef, args)) + }), + ) + + // budget is built inside the VM so `budget.spent.constructor` resolves to the + // VM's Function, not the host's. + const makeBudget = evalInVm< + ( + total: number | null, + spent: () => number, + remaining: () => number, + ) => unknown + >( + '((total, spent, remaining) => Object.freeze({ __proto__: null, total, spent: () => spent(), remaining: () => remaining() }))', + ) + defineGlobal( + 'budget', + makeBudget( + tokenBudget?.total ?? null, + () => tokenBudget?.getTurnSpent() ?? 0, + () => + tokenBudget?.total == null + ? Number.POSITIVE_INFINITY + : Math.max(0, tokenBudget.total - tokenBudget.getTurnSpent()), + ), + ) + + const makeConsole = evalInVm<(write: (line: string) => void) => unknown>( + `(write => { + const render = args => args.map(a => { + if (typeof a === 'string') return a + try { return JSON.stringify(a) ?? String(a) } catch { return '[object]' } + }).join(' ') + const emit = (...args) => { write(render(args)) } + return Object.freeze({ __proto__: null, log: emit, info: emit, warn: emit, error: emit, debug: emit }) + })`, + ) + defineGlobal( + 'console', + makeConsole(line => emit({ type: 'workflow_log', message: line })), + ) + + const makeTimers = evalInVm< + ( + set: (callback: unknown, ms: unknown) => unknown, + clear: (handle: unknown) => void, + ) => { setTimeout: unknown; clearTimeout: unknown } + >( + `((set, clear) => ({ + __proto__: null, + setTimeout: (callback, ms) => set(callback, ms), + clearTimeout: handle => { clear(handle) }, + }))`, + ) + const vmTimers = makeTimers( + (callback, ms) => { + if (typeof callback !== 'function') return 0 + const delay = typeof ms === 'number' && Number.isFinite(ms) ? ms : 0 + const timer = setTimeout(() => { + timers.delete(timer) + try { + ;(callback as () => void)() + } catch (error) { + logForDebugging(`workflow setTimeout callback threw: ${describeError(error)}`) + } + }, Math.max(0, delay)) + timer.unref?.() + timers.add(timer) + return timer + }, + handle => { + if (handle && typeof handle === 'object') { + clearTimeout(handle as NodeJS.Timeout) + timers.delete(handle as NodeJS.Timeout) + } + }, + ) + defineGlobal('setTimeout', vmTimers.setTimeout) + defineGlobal('clearTimeout', vmTimers.clearTimeout) + + if (params.args !== undefined) { + const json = JSON.stringify(params.args) + defineGlobal('args', json === undefined ? undefined : parseInVm(json)) + } else { + defineGlobal('args', undefined) + } + + return { + context, + awaitInVm, + exportValue(value: unknown): unknown { + if (typeof value === 'function') { + throw new Error('workflow result cannot be a function') + } + if (value === undefined) return null + const json = stringifyInVm(value) + if (json === undefined) return null + return JSON.parse(json) as unknown + }, + dispose: clearAllTimers, + } +} + +const describeError = describeThrown + +/** + * Parse + compile a script in one step. Used by the tool, the resume path, and + * the tests so all three reject the same scripts for the same reasons. + */ +export function prepareWorkflowScript( + script: string, +): PreparedWorkflowScript { + const parsed = parseWorkflowScript(script) + if ('error' in parsed) return { ok: false, error: parsed.error } + const compiled = compileWorkflowScript(parsed.scriptBody) + if (!compiled.ok) return { ok: false, error: compiled.error } + return { + ok: true, + meta: parsed.meta, + scriptBody: parsed.scriptBody, + vmScript: compiled.vmScript, + } +} + +export type PreparedWorkflowScript = + | { + ok: true + meta: WorkflowMeta + scriptBody: string + vmScript: vm.Script + } + | { ok: false; error: string } diff --git a/src/utils/workflows/save.test.ts b/src/utils/workflows/save.test.ts new file mode 100644 index 00000000..7e45b21d --- /dev/null +++ b/src/utils/workflows/save.test.ts @@ -0,0 +1,105 @@ +import { afterEach, beforeEach, describe, expect, test } from 'bun:test' +import { mkdirSync, symlinkSync } from 'fs' +import { mkdtemp, readFile, rm, writeFile } from 'fs/promises' +import { tmpdir } from 'os' +import { join } from 'path' +import { resolveProjectWorkflowsDir, saveWorkflowScript } from './save.js' + +let root: string +let configDir: string +let repo: string +const originalConfigDir = process.env.CLAUDE_CONFIG_DIR + +const SCRIPT = [ + "export const meta = { name: 'my-review', description: 'Review the diff' }", + "return await agent('review')", +].join('\n') + +describe('saveWorkflowScript', () => { + beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), 'wf-save-')) + configDir = join(root, 'claude') + repo = join(root, 'repo') + mkdirSync(configDir, { recursive: true }) + mkdirSync(join(repo, '.git'), { recursive: true }) + process.env.CLAUDE_CONFIG_DIR = configDir + }) + + afterEach(async () => { + if (originalConfigDir === undefined) delete process.env.CLAUDE_CONFIG_DIR + else process.env.CLAUDE_CONFIG_DIR = originalConfigDir + await rm(root, { recursive: true, force: true }) + }) + + test('saves to the personal directory under CLAUDE_CONFIG_DIR', async () => { + const result = await saveWorkflowScript({ script: SCRIPT, scope: 'user' }) + expect(result).toEqual({ + name: 'my-review', + filePath: join(configDir, 'workflows', 'my-review.js'), + }) + expect(await readFile(join(configDir, 'workflows', 'my-review.js'), 'utf8')).toBe( + SCRIPT, + ) + }) + + test('saves to the repository root when no .claude/workflows exists yet', async () => { + const result = await saveWorkflowScript({ + script: SCRIPT, + scope: 'project', + cwd: join(repo, 'packages', 'api'), + }) + expect('filePath' in result && result.filePath).toBe( + join(repo, '.claude', 'workflows', 'my-review.js'), + ) + }) + + test('prefers the closest existing .claude/workflows in a monorepo', async () => { + const pkg = join(repo, 'packages', 'api') + mkdirSync(join(pkg, '.claude', 'workflows'), { recursive: true }) + mkdirSync(join(repo, '.claude', 'workflows'), { recursive: true }) + expect(resolveProjectWorkflowsDir(pkg)).toBe( + join(pkg, '.claude', 'workflows'), + ) + + const result = await saveWorkflowScript({ + script: SCRIPT, + scope: 'project', + cwd: pkg, + }) + expect('filePath' in result && result.filePath).toBe( + join(pkg, '.claude', 'workflows', 'my-review.js'), + ) + }) + + test('rejects a script whose meta will not parse', async () => { + const result = await saveWorkflowScript({ + script: 'const x = 1\n', + scope: 'user', + }) + expect('error' in result && result.error).toContain('FIRST statement') + }) + + test('refuses to write through a symlinked target file', async () => { + const outside = join(root, 'outside.js') + await writeFile(outside, '// untouched\n', 'utf8') + mkdirSync(join(configDir, 'workflows'), { recursive: true }) + symlinkSync(outside, join(configDir, 'workflows', 'my-review.js')) + + const result = await saveWorkflowScript({ script: SCRIPT, scope: 'user' }) + expect('error' in result && result.error).toContain('symlink') + expect(await readFile(outside, 'utf8')).toBe('// untouched\n') + }) + + test('refuses when the project .claude directory is itself a symlink', async () => { + const elsewhere = join(root, 'elsewhere') + mkdirSync(elsewhere, { recursive: true }) + symlinkSync(elsewhere, join(repo, '.claude')) + + const result = await saveWorkflowScript({ + script: SCRIPT, + scope: 'project', + cwd: repo, + }) + expect('error' in result && result.error).toContain('symlink') + }) +}) diff --git a/src/utils/workflows/save.ts b/src/utils/workflows/save.ts new file mode 100644 index 00000000..4d63a06d --- /dev/null +++ b/src/utils/workflows/save.ts @@ -0,0 +1,98 @@ +import { lstat, mkdir, writeFile } from 'fs/promises' +import { join } from 'path' +import { getOriginalCwd } from '../../bootstrap/state.js' +import { getProjectDirsUpToHome } from '../markdownConfigLoader.js' +import { findGitRoot } from '../git.js' +import { parseWorkflowScript } from './meta.js' +import { getUserWorkflowsDir } from './paths.js' + +export type WorkflowSaveScope = 'user' | 'project' + +export type WorkflowSaveResult = + | { name: string; filePath: string } + | { error: string } + +/** + * Save a run's script as a `/name` command. + * + * Refuses to write through a symlink. For the project scope the check is + * wider than for the personal one: `.claude` there is repo-controlled, so a + * symlinked `.claude` or `.claude/workflows` could redirect the write out of + * the repository entirely. A `~/.claude` managed by a dotfiles tool is a + * normal setup, so only the target file itself is checked there. + */ +export async function saveWorkflowScript(params: { + script: string + scope: WorkflowSaveScope + cwd?: string +}): Promise { + const parsed = parseWorkflowScript(params.script) + if ('error' in parsed) return { error: parsed.error } + + const cwd = params.cwd ?? getOriginalCwd() + const dir = + params.scope === 'project' + ? resolveProjectWorkflowsDir(cwd) + : getUserWorkflowsDir() + + if (params.scope === 'project') { + for (const candidate of [ + join(dirOf(dir), '.claude'), + dir, + join(dir, `${parsed.meta.name}.js`), + ]) { + const link = await isSymlink(candidate) + if (link) return { error: `Refusing to write through a symlink: ${candidate}` } + } + } else if (await isSymlink(join(dir, `${parsed.meta.name}.js`))) { + return { + error: `Refusing to write through a symlink: ${join(dir, `${parsed.meta.name}.js`)}`, + } + } + + const filePath = join(dir, `${parsed.meta.name}.js`) + try { + await mkdir(dir, { recursive: true }) + await writeFile(filePath, params.script, 'utf8') + } catch (error) { + return { + error: error instanceof Error ? error.message : String(error), + } + } + return { name: parsed.meta.name, filePath } +} + +/** + * Where a project save lands in a monorepo. + * + * The closest existing `.claude/workflows` between cwd and the repo root wins, + * so a workflow saved while working in `packages/api` stays with that package + * instead of being hoisted to the root and shown to every other package. + */ +export function resolveProjectWorkflowsDir(cwd: string): string { + const existing = safeProjectDirs(cwd) + if (existing.length > 0) return existing[0]! + const root = findGitRoot(cwd) ?? cwd + return join(root, '.claude', 'workflows') +} + +function safeProjectDirs(cwd: string): string[] { + try { + return getProjectDirsUpToHome('workflows', cwd) + } catch { + return [] + } +} + +function dirOf(workflowsDir: string): string { + // `/.claude/workflows` → ``; the caller re-appends `.claude`. + return join(workflowsDir, '..', '..') +} + +async function isSymlink(path: string): Promise { + try { + return (await lstat(path)).isSymbolicLink() + } catch { + return false + } +} diff --git a/src/utils/workflows/types.ts b/src/utils/workflows/types.ts new file mode 100644 index 00000000..4b19ab46 --- /dev/null +++ b/src/utils/workflows/types.ts @@ -0,0 +1,147 @@ +/** + * Shared shapes for dynamic workflows. + * + * A dynamic workflow is a plain-JavaScript script that orchestrates subagents. + * The script never touches the filesystem or the network itself: it calls + * `agent()` / `parallel()` / `pipeline()` and the runtime spawns real subagents + * on its behalf. Everything the UI renders while a run is in flight travels as + * `WorkflowProgressEvent`s, so these shapes are the contract between the + * runtime, the Ink TUI, the local server, and the desktop app. + */ + +/** One entry of `meta.phases` — the phase list shown before a run is approved. */ +export type WorkflowPhaseMeta = { + title: string + detail?: string + model?: string +} + +/** + * The `export const meta = {...}` block every script must start with. + * Parsed as a pure literal (no identifiers, calls, or interpolation) so that + * listing saved workflows never executes untrusted code. + */ +export type WorkflowMeta = { + name: string + description: string + whenToUse?: string + title?: string + phases?: WorkflowPhaseMeta[] + model?: string +} + +/** Where a workflow definition was loaded from. */ +export type WorkflowSource = + | 'built-in' + | 'userSettings' + | 'projectSettings' + | 'plugin' + | 'inline' + +/** A discovered, runnable workflow definition. */ +export type WorkflowDefinition = { + source: WorkflowSource + name: string + description: string + whenToUse?: string + phases?: WorkflowPhaseMeta[] + script: string + filePath?: string +} + +export type WorkflowAgentRunState = 'start' | 'progress' | 'done' | 'error' + +/** Lifecycle of one `agent()` call. Keyed by `index` (1-based, run-scoped). */ +export type WorkflowAgentEvent = { + type: 'workflow_agent' + index: number + label: string + state: WorkflowAgentRunState + phaseIndex?: number + phaseTitle?: string + agentType?: string + isolation?: 'worktree' | 'remote' + model?: string + agentId?: string + queuedAt?: number + startedAt?: number + lastProgressAt?: number + durationMs?: number + tokens?: number + toolCalls?: number + /** Replayed from the resume journal instead of re-run. */ + cached?: boolean + /** Stopped by the user rather than failed on its own. */ + skipped?: boolean + /** Refused before spawning (permission rule or safety check). */ + blocked?: boolean + error?: string + promptPreview?: string + resultPreview?: string + lastToolName?: string +} + +/** A `phase()` call, or a seeded entry from `meta.phases`. */ +export type WorkflowPhaseEvent = { + type: 'workflow_phase' + index: number + title: string + /** `'meta'` for entries seeded from meta.phases, `'script'` for phase() calls. */ + kind?: 'meta' | 'script' +} + +/** A `log()` call, or a runtime note (a failed pipeline slot, a stall warning). */ +export type WorkflowLogEvent = { + type: 'workflow_log' + message: string +} + +export type WorkflowProgressEvent = + | WorkflowAgentEvent + | WorkflowPhaseEvent + | WorkflowLogEvent + +export function isWorkflowAgentEvent( + event: WorkflowProgressEvent, +): event is WorkflowAgentEvent { + return event.type === 'workflow_agent' +} + +export function isWorkflowPhaseEvent( + event: WorkflowProgressEvent, +): event is WorkflowPhaseEvent { + return event.type === 'workflow_phase' +} + +/** Progress rows worth persisting in the run summary — logs are transient. */ +export function isDurableWorkflowEvent(event: WorkflowProgressEvent): boolean { + return event.type !== 'workflow_log' +} + +/** Options a script may pass as the second argument to `agent()`. */ +export type WorkflowAgentOptions = { + label?: string + phase?: string + schema?: unknown + model?: string + effort?: string + isolation?: 'worktree' | 'remote' + agentType?: string + stallMs?: number +} + +/** Outcome of one script execution, before it is turned into a tool result. */ +export type WorkflowRunOutcome = { + result: unknown + agentCount: number + logs: string[] + failures: string[] + durationMs: number + error?: string +} + +/** The token allowance a run may spend, threaded in from the parent turn. */ +export type WorkflowTokenBudget = { + total: number | null + getTurnSpent: () => number +} diff --git a/src/utils/workflows/ultracode.test.ts b/src/utils/workflows/ultracode.test.ts new file mode 100644 index 00000000..4b21a5eb --- /dev/null +++ b/src/utils/workflows/ultracode.test.ts @@ -0,0 +1,98 @@ +import { afterEach, beforeEach, describe, expect, test } from 'bun:test' +import { mkdirSync } from 'fs' +import { mkdtemp, rm, writeFile } from 'fs/promises' +import { tmpdir } from 'os' +import { join } from 'path' +import { executeEffort } from '../../commands/effort/effort.js' +import { resetSettingsCache } from '../settings/settingsCache.js' +import { modelSupportsXHighEffort } from '../effort.js' +import { canEnableUltracode } from './ultracode.js' + +/** + * A model the capability table accepts in this environment. + * + * `modelSupportsXHighEffort` is gated on provider trust, so hardcoding a name + * makes the test depend on the developer's provider config rather than on + * ultracode's own logic. + */ +const XHIGH_MODEL = ['claude-opus-4-7', 'claude-sonnet-5', 'claude-fable-5'].find( + modelSupportsXHighEffort, +) + +let home: string +let configDir: string +const originalConfigDir = process.env.CLAUDE_CONFIG_DIR +const originalEffortEnv = process.env.CLAUDE_CODE_EFFORT_LEVEL + +async function writeSettings(value: Record): Promise { + await writeFile(join(configDir, 'settings.json'), JSON.stringify(value), 'utf8') + resetSettingsCache() +} + +describe('ultracode', () => { + beforeEach(async () => { + home = await mkdtemp(join(tmpdir(), 'wf-ultracode-')) + configDir = join(home, 'claude') + mkdirSync(configDir, { recursive: true }) + process.env.CLAUDE_CONFIG_DIR = configDir + delete process.env.CLAUDE_CODE_EFFORT_LEVEL + await writeSettings({}) + }) + + afterEach(async () => { + if (originalConfigDir === undefined) delete process.env.CLAUDE_CONFIG_DIR + else process.env.CLAUDE_CONFIG_DIR = originalConfigDir + if (originalEffortEnv === undefined) delete process.env.CLAUDE_CODE_EFFORT_LEVEL + else process.env.CLAUDE_CODE_EFFORT_LEVEL = originalEffortEnv + resetSettingsCache() + await rm(home, { recursive: true, force: true }) + }) + + test('requires an xhigh-capable model', () => { + const refused = canEnableUltracode('deepseek-v4-flash') + expect(refused.ok).toBe(false) + if (refused.ok) return + expect(refused.reason).toBe('model') + }) + + test('requires workflows to be enabled', async () => { + await writeSettings({ disableWorkflows: true }) + const refused = canEnableUltracode(XHIGH_MODEL ?? 'claude-opus-4-7') + expect(refused.ok).toBe(false) + if (refused.ok) return + // The workflow gate is checked before the model gate, so this reason holds + // whether or not the capability table trusts the model here. + expect(refused.reason).toBe('workflows-disabled') + }) + + test('/effort ultracode resolves to xhigh plus the session flag', () => { + if (!XHIGH_MODEL) { + // No xhigh-capable model is trusted in this environment; the refusal + // path is covered above and asserting here would test the provider + // config, not ultracode. + expect(canEnableUltracode('claude-opus-4-7').ok).toBe(false) + return + } + const result = executeEffort('ultracode', XHIGH_MODEL) + expect(result.effortUpdate).toEqual({ value: 'xhigh', ultracode: true }) + expect(result.message).toContain('this session only') + }) + + test('/effort ultracode explains itself when the model cannot do xhigh', () => { + const result = executeEffort('ultracode', 'deepseek-v4-flash') + expect(result.effortUpdate).toBeUndefined() + expect(result.message).toContain("doesn't support") + }) + + test('/effort ultracode is refused when workflows are off', async () => { + await writeSettings({ enableWorkflows: false }) + const result = executeEffort('ultracode', XHIGH_MODEL ?? 'claude-opus-4-7') + expect(result.effortUpdate).toBeUndefined() + expect(result.message).toContain('dynamic workflows enabled') + }) + + test('an ordinary level never sets the flag', () => { + const result = executeEffort('high', XHIGH_MODEL ?? 'claude-opus-4-7') + expect(result.effortUpdate?.ultracode).toBeUndefined() + }) +}) diff --git a/src/utils/workflows/ultracode.ts b/src/utils/workflows/ultracode.ts new file mode 100644 index 00000000..5d40d183 --- /dev/null +++ b/src/utils/workflows/ultracode.ts @@ -0,0 +1,62 @@ +/** + * Ultracode: xhigh effort plus standing dynamic-workflow orchestration. + * + * Not a sixth effort level — it is `xhigh` with a flag beside it. Keeping it + * out of `EffortLevel` matters: every model-capability check, settings write, + * and picker row already reasons about the five real levels, and a sixth would + * have to be special-cased in all of them to mean "xhigh, but also…". + */ + +import { modelSupportsXHighEffort } from '../effort.js' +import { areWorkflowsEnabled } from './enabled.js' + +export const ULTRACODE_EFFORT_ARG = 'ultracode' + +/** Effort ultracode resolves to once the flag is recorded separately. */ +export const ULTRACODE_EFFORT_LEVEL = 'xhigh' as const + +export const ULTRACODE_MENU_DESCRIPTION = + 'xhigh effort + dynamic workflows for maximum thoroughness' + +export const ULTRACODE_ENTER_REMINDER = + 'Ultracode is on: optimize for the most exhaustive, correct answer — not the fastest or cheapest. ' + + "Use the Workflow tool on every substantive task; token cost is not a constraint. See the Workflow tool's " + + '**Ultracode** section and quality patterns. Solo only on conversational/trivial turns.' + +export const ULTRACODE_STILL_ON_REMINDER = + 'Ultracode is still on — use the Workflow tool; see its Ultracode section.' + +export const ULTRACODE_EXIT_REMINDER = + "Ultracode is off — the Workflow tool's standard opt-in rule applies again." + +export const WORKFLOW_KEYWORD_REMINDER = + 'The user included the keyword "ultracode", opting this turn into multi-agent orchestration — ' + + 'use the Workflow tool to fulfill the request.' + +export type UltracodeRequest = + | { ok: true } + | { ok: false; reason: 'workflows-disabled' | 'model' } + +/** + * Whether ultracode can be turned on right now. + * + * Both preconditions are real: without workflows there is nothing to + * orchestrate, and without xhigh the "ultra" half of the level is a lie. + */ +export function canEnableUltracode(model: string): UltracodeRequest { + if (!areWorkflowsEnabled()) return { ok: false, reason: 'workflows-disabled' } + if (!modelSupportsXHighEffort(model)) return { ok: false, reason: 'model' } + return { ok: true } +} + +export function describeUltracodeRefusal( + reason: Exclude['reason'], + model: string, +): string { + switch (reason) { + case 'workflows-disabled': + return 'Ultracode needs dynamic workflows enabled (see /config).' + case 'model': + return `Ultracode runs at xhigh effort, which ${model} doesn't support — switch to an xhigh-capable model.` + } +}