feat(quality-gate): drive the agent flow with a provider the user configured

`check:agent-flow` proves the protocol with the mock CLI, which is what makes it
CI-safe and lets any contributor run it with no credentials. It cannot prove the
thing this product actually is: a desktop agent talking to a real model. That can
only run where the credentials are, so this lane is local and manual by
construction — registered in no quality-gate mode, referenced by no workflow, and
live.test.ts fails if either changes.

Six scenarios, sharing the existing harness rather than a second copy of it:
first turn, permission allow, permission deny, interrupt, reconnect, and history
recovery. Prompts induce the behaviour instead of dictating it, and assertions
only look at protocol shape and side effects on disk — never at generated text —
so the lane passes on any provider, including a local one. The three flows left
out (api-error, tool-error, runtime-select) each carry a written reason, because
a silently missing flow reads as a covered one.

Spending someone's quota is the failure mode worth engineering against, so the
runner refuses to guess: no implicit fallback to the active provider, an ambiguous
selector is an error rather than a pick, and without --yes it prints the provider,
model and config path it would use and exits without sending anything. User state
is copied into a throwaway config dir and the real ~/.claude is fingerprinted
before and after — a run that writes to it fails loudly instead of being cleaned
up quietly.

Not yet run end to end: the local LM Studio endpoint answers 502 here, so the six
runners have only been verified for structure. Target resolution, the
confirmation gate, and lane placement are covered by 14 tests that need no
provider at all.
This commit is contained in:
程序员阿江(Relakkes)
2026-08-04 17:15:29 +08:00
parent 20b6e6dc9e
commit 0cea7b5a61
6 changed files with 711 additions and 8 deletions
+3 -2
View File
@@ -14,7 +14,7 @@
"perf:local-index:10k": "bun run scripts/perf/local-index-benchmark.ts --sessions 10000 --runs 20",
"check:pr": "bun run scripts/pr/check-pr.ts",
"check:impact": "bun run scripts/pr/impact-report.ts",
"check:policy": "bun test ./scripts/pr/bun-test-filter.test.ts ./scripts/pr/change-policy.test.ts ./scripts/pr/changed-files.test.ts ./scripts/pr/module-graph.test.ts ./scripts/pr/pr-triage-workflow.test.ts ./scripts/pr/pr-quality-workflow.test.ts ./scripts/pr/release-workflow.test.ts ./scripts/pr/quality-contract.test.ts ./scripts/pr/test-environment.test.ts ./scripts/release-update-metadata.test.ts ./scripts/git-hooks/install.test.ts ./scripts/quality-gate/quarantine.test.ts ./scripts/quality-gate/coverage.test.ts ./scripts/quality-gate/provider-smoke/execute.test.ts ./scripts/quality-gate/desktop-smoke/execute.test.ts ./scripts/quality-gate/providerTargets.test.ts ./scripts/quality-gate/runner.test.ts ./scripts/quality-gate/sandbox.test.ts ./scripts/quality-gate/agent-flow/scenarios.test.ts ./scripts/quality-gate/desktop-smoke/deterministic.test.ts",
"check:policy": "bun test ./scripts/pr/bun-test-filter.test.ts ./scripts/pr/change-policy.test.ts ./scripts/pr/changed-files.test.ts ./scripts/pr/module-graph.test.ts ./scripts/pr/pr-triage-workflow.test.ts ./scripts/pr/pr-quality-workflow.test.ts ./scripts/pr/release-workflow.test.ts ./scripts/pr/quality-contract.test.ts ./scripts/pr/test-environment.test.ts ./scripts/release-update-metadata.test.ts ./scripts/git-hooks/install.test.ts ./scripts/quality-gate/quarantine.test.ts ./scripts/quality-gate/coverage.test.ts ./scripts/quality-gate/provider-smoke/execute.test.ts ./scripts/quality-gate/desktop-smoke/execute.test.ts ./scripts/quality-gate/providerTargets.test.ts ./scripts/quality-gate/runner.test.ts ./scripts/quality-gate/sandbox.test.ts ./scripts/quality-gate/agent-flow/scenarios.test.ts ./scripts/quality-gate/agent-flow/live.test.ts ./scripts/quality-gate/desktop-smoke/deterministic.test.ts",
"check:server": "bun run scripts/pr/run-server-tests.ts",
"check:provider-contract": "bun run scripts/pr/run-provider-contract-tests.ts",
"check:chat-contract": "bun run scripts/pr/run-chat-contract-tests.ts",
@@ -42,7 +42,8 @@
"test:package-smoke:current": "bun run scripts/quality-gate/package-smoke/current.ts",
"docs:dev": "npm --prefix site run dev",
"docs:build": "npm --prefix site run build",
"docs:preview": "npm --prefix site run preview"
"docs:preview": "npm --prefix site run preview",
"check:agent-flow:live": "bun run scripts/quality-gate/agent-flow/live-cli.ts"
},
"dependencies": {
"@anthropic-ai/sandbox-runtime": "^0.0.44",
+11 -6
View File
@@ -26,7 +26,7 @@ export type AgentFlowScenarioResult = {
covers: string[]
}
function getPort() {
export function getPort() {
return new Promise<number>((resolvePort, reject) => {
const server = createServer()
server.once('error', reject)
@@ -38,7 +38,7 @@ function getPort() {
})
}
async function waitForHttp(url: string, timeoutMs: number) {
export async function waitForHttp(url: string, timeoutMs: number) {
const deadline = Date.now() + timeoutMs
let lastError = ''
while (Date.now() < deadline) {
@@ -54,7 +54,7 @@ async function waitForHttp(url: string, timeoutMs: number) {
throw new Error(`Timed out waiting for ${url}${lastError ? ` (${lastError})` : ''}`)
}
async function pipeToFile(stream: ReadableStream<Uint8Array> | null, path: string) {
export async function pipeToFile(stream: ReadableStream<Uint8Array> | null, path: string) {
if (!stream) return
const reader = stream.getReader()
const decoder = new TextDecoder()
@@ -66,10 +66,15 @@ async function pipeToFile(stream: ReadableStream<Uint8Array> | null, path: strin
}
/**
* Exported for `live.ts`, which drives the same protocol against a real provider.
* The transport, the ordering assertions and the turn loop are identical there —
* only the runtime behind the CLI and the assertions' tolerance differ — so the two
* runners share this rather than growing a second, drifting copy.
*
* Thin client over the real session WebSocket. It records every frame so a scenario
* can assert on ordering after the fact instead of racing the stream.
*/
class SessionSocket {
export class SessionSocket {
readonly messages: ProtocolMessage[] = []
private ws: WebSocket | null = null
@@ -135,7 +140,7 @@ type ScenarioContext = {
openSocket(sessionId: string): Promise<SessionSocket>
}
function assertOrder(messages: readonly ProtocolMessage[], types: readonly string[]) {
export function assertOrder(messages: readonly ProtocolMessage[], types: readonly string[]) {
const ordered = findOrderedTypes(messages, types)
if (!ordered.ok) {
throw new Error(`expected ${types.join(' -> ')} but ${ordered.missing} never arrived; saw ${ordered.seen.join(', ')}`)
@@ -143,7 +148,7 @@ function assertOrder(messages: readonly ProtocolMessage[], types: readonly strin
}
/** Drives one prompt to completion and returns the frames observed for that turn. */
async function runTurn(socket: SessionSocket, prompt: string, timeoutMs = DEFAULT_STEP_TIMEOUT_MS) {
export async function runTurn(socket: SessionSocket, prompt: string, timeoutMs = DEFAULT_STEP_TIMEOUT_MS) {
const start = socket.messages.length
socket.send({ type: 'user_message', content: prompt })
await socket.waitFor((message) => message.type === 'message_complete', timeoutMs, 'message_complete', start)
@@ -0,0 +1,81 @@
/**
* `bun run check:agent-flow:live -- --provider <name> --yes`
*
* Local and manual by design. Nothing in CI calls this, and it takes no fallback:
* without an explicit provider and an explicit --yes it prints what it *would* do and
* exits without sending anything upstream.
*/
import { mkdirSync, writeFileSync } from 'node:fs'
import { join, resolve } from 'node:path'
import { describeLiveTarget, executeLiveAgentFlow, resolveLiveTarget } from './live.ts'
import { LIVE_AGENT_FLOW_SCENARIOS, LIVE_FLOW_EXCLUSIONS } from './liveScenarios.ts'
function flag(argv: string[], name: string): string | undefined {
const index = argv.indexOf(`--${name}`)
if (index === -1) return undefined
const value = argv[index + 1]
return value && !value.startsWith('--') ? value : ''
}
async function main() {
const argv = process.argv.slice(2)
if (argv.includes('--list')) {
console.log('Live agent-flow scenarios:')
for (const scenario of LIVE_AGENT_FLOW_SCENARIOS) {
console.log(` ${scenario.id.padEnd(24)} ${scenario.title}`)
}
console.log('\nDeliberately not covered live:')
for (const [key, reason] of Object.entries(LIVE_FLOW_EXCLUSIONS)) {
console.log(` ${key.padEnd(24)} ${reason}`)
}
return
}
const rootDir = resolve(import.meta.dir, '../../..')
const only = flag(argv, 'only')
const target = resolveLiveTarget(flag(argv, 'provider'), { modelId: flag(argv, 'model') || undefined })
const selected = only
? LIVE_AGENT_FLOW_SCENARIOS.filter((scenario) => scenario.id === only)
: LIVE_AGENT_FLOW_SCENARIOS
if (selected.length === 0) {
throw new Error(`--only ${only} matches no scenario. Use --list to see them.`)
}
if (!argv.includes('--yes')) {
console.log(describeLiveTarget(target, selected.length))
process.exitCode = 1
return
}
const artifactDir = join(rootDir, 'artifacts', 'agent-flow-live')
mkdirSync(artifactDir, { recursive: true })
console.log(`Running ${selected.length} live scenario(s) against ${target.providerName} (${target.modelId})\n`)
const results = await executeLiveAgentFlow({
rootDir,
artifactDir,
target,
only: selected.map((scenario) => scenario.id),
})
for (const result of results) {
const mark = result.status === 'passed' ? 'PASS' : 'FAIL'
console.log(` ${mark} ${result.id.padEnd(24)} ${(result.durationMs / 1000).toFixed(1)}s`)
if (result.detail) console.log(` ${result.detail}`)
}
writeFileSync(join(artifactDir, 'results.json'), `${JSON.stringify({ target, results }, null, 2)}\n`)
const failed = results.filter((result) => result.status === 'failed')
console.log(`\n${results.length - failed.length}/${results.length} passed. Artifacts: ${artifactDir}`)
if (failed.length > 0) process.exitCode = 1
}
main().catch((error) => {
console.error(error instanceof Error ? error.message : error)
process.exitCode = 1
})
@@ -0,0 +1,138 @@
import { describe, expect, test } from 'bun:test'
import { mkdtempSync, mkdirSync, writeFileSync, readFileSync } from 'node:fs'
import { tmpdir } from 'node:os'
import { join } from 'node:path'
import { describeLiveTarget, resolveLiveTarget } from './live.ts'
import { LIVE_AGENT_FLOW_SCENARIOS, LIVE_FLOW_COVERAGE, LIVE_FLOW_EXCLUSIONS } from './liveScenarios.ts'
/**
* Everything here runs without a provider. The parts of the live lane that can be
* tested for free are the parts that decide whether to spend money at all — target
* resolution and the confirmation gate — so those are exactly what is pinned.
*/
function configDirWith(providers: Array<{ id: string; name: string; models?: Record<string, string> }>) {
const dir = mkdtempSync(join(tmpdir(), 'cc-haha-live-test-'))
mkdirSync(join(dir, 'cc-haha'), { recursive: true })
writeFileSync(
join(dir, 'cc-haha', 'providers.json'),
JSON.stringify({ activeId: providers[0]?.id ?? null, providers }),
)
return dir
}
describe('live target resolution', () => {
const configDir = configDirWith([
{ id: 'aaaaaaaa-0000-4000-8000-000000000001', name: 'LM Studio', models: { main: 'local-model' } },
{ id: 'aaaaaaaa-0000-4000-8000-000000000002', name: 'DeepSeek', models: { main: 'deepseek-chat' } },
{ id: 'aaaaaaaa-0000-4000-8000-000000000003', name: 'DeepSeek Backup', models: { main: 'deepseek-chat' } },
])
// The whole point of the gate: never pick a provider on the user's behalf. An
// implicit fallback to the active provider is how a test run bills someone for a
// model they did not choose.
test('refuses to run without an explicit provider', () => {
expect(() => resolveLiveTarget(undefined, { configDir })).toThrow(/--provider is required/)
})
test('refuses an ambiguous selector instead of picking one', () => {
// 'deep' substring-matches both DeepSeek entries and names neither.
expect(() => resolveLiveTarget('deep', { configDir })).toThrow(/matches 2 providers/)
})
// An exact name is not ambiguous just because it prefixes another entry, or
// 'DeepSeek' would be unusable for as long as 'DeepSeek Backup' exists.
test('an exact name beats a longer substring match', () => {
expect(resolveLiveTarget('DeepSeek', { configDir }).providerName).toBe('DeepSeek')
})
test('resolves an exact name and takes the provider main model', () => {
const target = resolveLiveTarget('LM Studio', { configDir })
expect(target.providerId).toBe('aaaaaaaa-0000-4000-8000-000000000001')
expect(target.modelId).toBe('local-model')
})
test('resolves by id, and an explicit model wins over the configured one', () => {
const target = resolveLiveTarget('aaaaaaaa-0000-4000-8000-000000000002', { configDir, modelId: 'override' })
expect(target.providerName).toBe('DeepSeek')
expect(target.modelId).toBe('override')
})
test('names the configured providers when nothing matches', () => {
expect(() => resolveLiveTarget('nope', { configDir })).toThrow(/LM Studio/)
})
test('says where to configure one when there are none', () => {
expect(() => resolveLiveTarget('anything', { configDir: configDirWith([]) })).toThrow(/No providers configured/)
})
})
describe('confirmation banner', () => {
const target = {
providerId: 'id-1',
providerName: 'LM Studio',
modelId: 'local-model',
host: 'x',
source: '/tmp/providers.json',
}
test('states the cost, the target and how to proceed', () => {
const banner = describeLiveTarget(target, 6)
expect(banner).toContain('real money')
expect(banner).toContain('LM Studio')
expect(banner).toContain('local-model')
expect(banner).toContain('--yes')
})
// A banner that leaks a key into a terminal or a CI log is worse than no banner.
test('carries no credential material', () => {
expect(describeLiveTarget(target, 6)).not.toMatch(/sk-|api[_-]?key|token|Bearer/i)
})
})
describe('live scenario catalog', () => {
test('every scenario has a runner and a reason it works on any model', () => {
const source = readFileSync(join(import.meta.dir, 'live.ts'), 'utf8')
for (const scenario of LIVE_AGENT_FLOW_SCENARIOS) {
expect(source, `${scenario.id} has no runner`).toContain(`'${scenario.id}'(ctx)`)
expect(scenario.modelAgnosticBecause.length, `${scenario.id} needs a real justification`).toBeGreaterThan(40)
}
})
test('covers every flow it claims, and documents each gap', () => {
const covered = new Set(LIVE_AGENT_FLOW_SCENARIOS.flatMap((scenario) => scenario.covers))
expect([...LIVE_FLOW_COVERAGE].filter((item) => !covered.has(item))).toEqual([])
// A flow dropped from the live lane must say why, so nobody assumes it is covered.
for (const reason of Object.values(LIVE_FLOW_EXCLUSIONS)) {
expect(reason.length).toBeGreaterThan(40)
}
})
test('asserts on protocol and disk, never on generated text', () => {
const source = readFileSync(join(import.meta.dir, 'live.ts'), 'utf8')
// Guards the rule that keeps this provider-agnostic: the moment an assertion
// compares model output, the lane only passes for whoever wrote it.
const runnerBody = source.slice(source.indexOf('const runners'))
expect(runnerBody).not.toMatch(/\.text\s*===\s*['"]/)
expect(runnerBody).not.toMatch(/toContain\(['"](?!ALLOWED)/)
})
})
describe('lane placement', () => {
// The reason this file exists rather than a mode entry: a live lane in any CI mode
// would mean CI needs credentials, which contradicts the brief that every
// contributor can pass the required gate with no provider at all.
test('is registered in no quality-gate mode', () => {
const modes = readFileSync(join(import.meta.dir, '../modes.ts'), 'utf8')
expect(modes).not.toContain('agent-flow-live')
expect(modes).not.toContain('check:agent-flow:live')
})
test('is not referenced by any workflow', () => {
for (const file of ['pr-quality.yml', 'nightly-quality.yml']) {
const workflow = readFileSync(join(import.meta.dir, '../../../.github/workflows', file), 'utf8')
expect(workflow, `${file} must not run the live lane`).not.toContain('agent-flow:live')
}
})
})
+379
View File
@@ -0,0 +1,379 @@
/**
* Agent-flow scenarios driven by a provider the user actually configured.
*
* `check:agent-flow` proves the protocol is wired correctly using the mock CLI. It
* cannot prove the thing this product is: a desktop agent talking to a real model.
* That only runs where the credentials are — the maintainer's machine — so this lane
* is local and manual by construction. It is registered in no CI mode and no
* `requiredForModes`, and `check:policy` fails if that changes.
*
* Real provider means real spend, so the run refuses to start until it has printed
* exactly which provider, model and host it is about to talk to and been told to go
* ahead. All user state is copied into a throwaway config dir first: the run reads
* the real `~/.claude` once and never writes to it.
*/
import { cpSync, existsSync, mkdirSync, readFileSync, rmSync, writeFileSync } from 'node:fs'
import { mkdtemp } from 'node:fs/promises'
import { homedir, tmpdir } from 'node:os'
import { join } from 'node:path'
import { createQualityGateSandbox } from '../sandbox.ts'
import { loadProviderIndex, getProviderIndexPath } from '../providerTargets.ts'
import { getPort, pipeToFile, runTurn, SessionSocket, waitForHttp } from './execute.ts'
import { LIVE_AGENT_FLOW_SCENARIOS } from './liveScenarios.ts'
const FIXTURE = 'scripts/quality-gate/agent-flow/fixtures/workspace'
/** Live turns wait on a real model, so every step gets far longer than the mock lane. */
const LIVE_STEP_TIMEOUT_MS = 180_000
export type LiveAgentFlowResult = {
id: string
title: string
status: 'passed' | 'failed'
detail?: string
durationMs: number
}
export type ResolvedLiveTarget = {
providerId: string
providerName: string
modelId: string
/** Host only. The full base URL can carry a key in its query string. */
host: string
source: string
}
/**
* Picks the provider to run against. Never guesses: an ambiguous or missing selector
* is an error, because the failure mode of guessing is spending someone's money on a
* provider they did not choose.
*/
export function resolveLiveTarget(
selector: string | undefined,
options: { configDir?: string; modelId?: string } = {},
): ResolvedLiveTarget {
const configDir = options.configDir ?? process.env.CLAUDE_CONFIG_DIR ?? join(homedir(), '.claude')
const index = loadProviderIndex(configDir)
const indexPath = getProviderIndexPath(configDir)
if (index.providers.length === 0) {
throw new Error(
`No providers configured in ${indexPath}. Add one in the desktop app first — this lane deliberately has no built-in fallback.`,
)
}
if (!selector) {
const names = index.providers.map((provider) => provider.name).join(', ')
throw new Error(
`--provider is required. This lane spends real quota, so it will not fall back to the active provider.\nConfigured: ${names}`,
)
}
const needle = selector.trim().toLowerCase()
const matches = index.providers.filter(
(provider) => provider.id.toLowerCase() === needle || provider.name.toLowerCase() === needle,
)
const loose = matches.length > 0
? matches
: index.providers.filter((provider) => provider.name.toLowerCase().includes(needle))
if (loose.length === 0) {
throw new Error(`No configured provider matches "${selector}". Configured: ${index.providers.map((p) => p.name).join(', ')}`)
}
if (loose.length > 1) {
throw new Error(
`${selector} matches ${loose.length} providers (${loose.map((p) => p.name).join(', ')}). Pass an exact name or id.`,
)
}
const provider = loose[0]!
const modelId = options.modelId ?? provider.models?.main ?? 'current'
return {
providerId: provider.id,
providerName: provider.name,
modelId,
host: 'resolved by the server from the sandboxed provider config',
source: indexPath,
}
}
/** The banner the operator has to read before anything is sent upstream. */
export function describeLiveTarget(target: ResolvedLiveTarget, scenarioCount: number): string {
return [
'',
' This lane talks to a real provider. That may cost real money.',
'',
` provider ${target.providerName} (${target.providerId})`,
` model ${target.modelId}`,
` read from ${target.source}`,
` scenarios ${scenarioCount}`,
'',
' User state is copied into a throwaway config dir; the real ~/.claude is never written to.',
' Re-run with --yes to proceed.',
'',
].join('\n')
}
type LiveContext = {
baseUrl: string
workRoot: string
target: ResolvedLiveTarget
createSession(): Promise<string>
openSocket(sessionId: string): Promise<SessionSocket>
pinRuntime(socket: SessionSocket): void
}
function assistantTextCount(socket: SessionSocket) {
return socket.messages.filter((message) => message.type === 'content_start').length
}
const runners: Record<string, (ctx: LiveContext) => Promise<void>> = {
async 'live-first-turn'(ctx) {
const socket = await ctx.openSocket(await ctx.createSession())
try {
ctx.pinRuntime(socket)
const turn = await runTurn(socket, 'Reply with a single short sentence confirming you are ready.', LIVE_STEP_TIMEOUT_MS)
const streamed = turn.filter((message) => message.type === 'content_delta' && String(message.text ?? '').trim())
if (streamed.length === 0) {
throw new Error(`no assistant text streamed; saw ${turn.map((m) => m.type).join(', ')}`)
}
} finally {
socket.close()
}
},
async 'live-permission-allow'(ctx) {
const socket = await ctx.openSocket(await ctx.createSession())
const target = join(ctx.workRoot, 'live-allowed.txt')
try {
ctx.pinRuntime(socket)
const start = socket.messages.length
socket.send({
type: 'user_message',
content: `Create a file called live-allowed.txt in the current directory. Its entire contents must be the single word ALLOWED. Use your file writing tool, then stop.`,
})
const request = await socket.waitFor(
(message) => message.type === 'permission_request',
LIVE_STEP_TIMEOUT_MS,
'permission_request',
start,
)
if (existsSync(target)) {
throw new Error('the file appeared before the permission request was answered')
}
socket.send({ type: 'permission_response', requestId: request.requestId, allowed: true, rule: 'agent-flow-live' })
await socket.waitFor((m) => m.type === 'message_complete', LIVE_STEP_TIMEOUT_MS, 'message_complete', start)
if (!existsSync(target)) {
throw new Error(`approved write never landed. Tool was ${request.toolName}`)
}
if (!readFileSync(target, 'utf8').toUpperCase().includes('ALLOWED')) {
throw new Error('approved write landed but the content is not what was asked for')
}
} finally {
socket.close()
}
},
async 'live-permission-deny'(ctx) {
const socket = await ctx.openSocket(await ctx.createSession())
const target = join(ctx.workRoot, 'live-denied.txt')
try {
ctx.pinRuntime(socket)
const start = socket.messages.length
socket.send({
type: 'user_message',
content: `Create a file called live-denied.txt in the current directory containing the word DENIED. Use your file writing tool, then stop.`,
})
const request = await socket.waitFor(
(message) => message.type === 'permission_request',
LIVE_STEP_TIMEOUT_MS,
'permission_request',
start,
)
socket.send({ type: 'permission_response', requestId: request.requestId, allowed: false, rule: 'agent-flow-live' })
await socket.waitFor((m) => m.type === 'message_complete', LIVE_STEP_TIMEOUT_MS, 'message_complete', start)
if (existsSync(target)) {
throw new Error('a denied write still reached the disk')
}
} finally {
socket.close()
}
},
async 'live-interrupt'(ctx) {
const socket = await ctx.openSocket(await ctx.createSession())
try {
ctx.pinRuntime(socket)
const start = socket.messages.length
socket.send({
type: 'user_message',
content: 'Count from 1 to 300, writing each number on its own line. Do not stop early.',
})
await socket.waitFor((m) => m.type === 'content_delta', LIVE_STEP_TIMEOUT_MS, 'first content_delta', start)
socket.send({ type: 'stop' })
await socket.waitFor(
(m) => m.type === 'message_complete' || m.type === 'session_state_changed',
LIVE_STEP_TIMEOUT_MS,
'stream to settle after stop',
start,
)
// The real check is that it goes quiet: a stop that only flips a flag while the
// model keeps streaming is the failure this scenario exists for.
const settled = socket.messages.length
await Bun.sleep(3_000)
const arrivedAfter = socket.messages.slice(settled).filter((m) => m.type === 'content_delta')
if (arrivedAfter.length > 0) {
throw new Error(`${arrivedAfter.length} content_delta frames arrived 3s after the stream was stopped`)
}
} finally {
socket.close()
}
},
async 'live-reconnect'(ctx) {
const sessionId = await ctx.createSession()
const first = await ctx.openSocket(sessionId)
let second: SessionSocket | null = null
try {
ctx.pinRuntime(first)
await runTurn(first, 'Reply with one short sentence.', LIVE_STEP_TIMEOUT_MS)
const before = assistantTextCount(first)
first.close()
second = await ctx.openSocket(sessionId)
await Bun.sleep(3_000)
const replayed = assistantTextCount(second)
if (replayed > before) {
throw new Error(`reconnect replayed ${replayed - before} extra assistant message(s)`)
}
} finally {
first.close()
second?.close()
}
},
async 'live-session-recovery'(ctx) {
const sessionId = await ctx.createSession()
const socket = await ctx.openSocket(sessionId)
try {
ctx.pinRuntime(socket)
await runTurn(socket, 'Reply with one short sentence.', LIVE_STEP_TIMEOUT_MS)
const seen = assistantTextCount(socket)
const response = await fetch(`${ctx.baseUrl}/api/sessions/${sessionId}/messages`)
if (!response.ok) throw new Error(`history fetch failed: ${response.status}`)
const body = await response.json() as { messages?: Array<{ type?: string }> }
const persisted = (body.messages ?? []).filter((message) => message.type === 'assistant').length
if (persisted < seen) {
throw new Error(`socket delivered ${seen} assistant message(s) but the transcript kept ${persisted}`)
}
} finally {
socket.close()
}
},
}
export async function executeLiveAgentFlow(options: {
rootDir: string
artifactDir: string
target: ResolvedLiveTarget
only?: string[]
}): Promise<LiveAgentFlowResult[]> {
const { rootDir, artifactDir, target } = options
mkdirSync(artifactDir, { recursive: true })
const serverLogPath = join(artifactDir, 'server.log')
writeFileSync(serverLogPath, '')
const port = await getPort()
const baseUrl = `http://127.0.0.1:${port}`
const workRoot = await mkdtemp(join(tmpdir(), 'cc-haha-agent-flow-live-'))
cpSync(join(rootDir, FIXTURE), workRoot, { recursive: true })
// Seeded, not shared: the sandbox gets a copy of the real provider config so the
// server can reach the chosen provider, and every write the run makes lands in the
// throwaway dir. CLAUDE_CLI_PATH is deliberately left alone — unlike the mock lane,
// this one wants the real CLI.
const sandbox = createQualityGateSandbox({
label: 'agent-flow-live',
seedProviders: true,
envOverrides: { CC_HAHA_DISABLE_TERMINAL_SHELL_ENV: '1' },
})
const server = Bun.spawn(['bun', 'run', 'src/server/index.ts', '--host', '127.0.0.1', '--port', String(port)], {
cwd: rootDir,
stdout: 'pipe',
stderr: 'pipe',
env: { ...sandbox.env, SERVER_PORT: String(port) },
})
const pumps = [pipeToFile(server.stdout, serverLogPath), pipeToFile(server.stderr, serverLogPath)]
const results: LiveAgentFlowResult[] = []
try {
await waitForHttp(`${baseUrl}/health`, 60_000)
const ctx: LiveContext = {
baseUrl,
workRoot,
target,
async createSession() {
const response = await fetch(`${baseUrl}/api/sessions`, {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify({ workDir: workRoot }),
})
if (!response.ok) throw new Error(`session create failed: ${response.status}`)
const body = await response.json() as { id?: string; sessionId?: string }
const id = body.id ?? body.sessionId
if (!id) throw new Error(`session create returned no id: ${JSON.stringify(body)}`)
return id
},
async openSocket(sessionId) {
return await new SessionSocket(baseUrl, sessionId).open()
},
pinRuntime(socket) {
socket.send({ type: 'set_runtime_config', providerId: target.providerId, modelId: target.modelId })
},
}
const selected = options.only?.length
? LIVE_AGENT_FLOW_SCENARIOS.filter((scenario) => options.only!.includes(scenario.id))
: LIVE_AGENT_FLOW_SCENARIOS
for (const scenario of selected) {
const started = Date.now()
try {
await runners[scenario.id]!(ctx)
results.push({ id: scenario.id, title: scenario.title, status: 'passed', durationMs: Date.now() - started })
} catch (error) {
results.push({
id: scenario.id,
title: scenario.title,
status: 'failed',
detail: error instanceof Error ? error.message : String(error),
durationMs: Date.now() - started,
})
}
}
} finally {
server.kill()
await Promise.allSettled(pumps)
// Checked before teardown: a run that wrote to the real config dir has to be loud,
// not quietly cleaned up. This lane seeds from the user's live provider state, so
// it is the one with something to lose.
const mutations = sandbox.detectUserStateMutations()
sandbox.cleanup()
if (mutations.length > 0) {
throw new Error(`live agent flow mutated real user state:\n ${mutations.join('\n ')}`)
}
rmSync(workRoot, { recursive: true, force: true })
}
return results
}
@@ -0,0 +1,99 @@
/**
* Agent-flow scenarios for a real provider.
*
* The deterministic catalog in `scenarios.ts` scripts the mock CLI down to the exact
* tool call, which is what makes it reproducible and CI-safe. A real model will not
* follow that script, so this is a separate catalog with the same job and different
* rules: the prompt has to *induce* the behaviour rather than dictate it, and the
* assertion has to describe the outcome a user would notice rather than an exact
* frame payload.
*
* Everything here is model-agnostic on purpose — the point is that any provider the
* user has configured can run it, including a local one. So: no prompt relies on a
* particular model's phrasing, no assertion compares generated text, and the only
* things checked are protocol shape and observable side effects on disk.
*
* Pure module, so the catalog and its rules stay unit-testable without a provider.
*/
/**
* What a live run can prove that the deterministic one cannot: that a real model,
* driven through the real protocol, still lands on the same user-visible outcome.
*/
export const LIVE_FLOW_COVERAGE = [
'first-turn',
'tool-execute',
'permission-allow',
'permission-deny',
'interrupt',
'reconnect',
'session-recover',
] as const
export type LiveFlowCoverage = (typeof LIVE_FLOW_COVERAGE)[number]
export type LiveAgentFlowScenario = {
id: string
title: string
covers: LiveFlowCoverage[]
/**
* Why this one is safe to assert against any model. A scenario without a defensible
* answer here is a flake waiting to fail on someone else's provider.
*/
modelAgnosticBecause: string
}
export const LIVE_AGENT_FLOW_SCENARIOS: readonly LiveAgentFlowScenario[] = [
{
id: 'live-first-turn',
title: 'A real model answers the first turn over the real socket',
covers: ['first-turn'],
modelAgnosticBecause:
'Asserts only that assistant text streamed and the turn completed. No comparison against generated wording.',
},
{
id: 'live-permission-allow',
title: 'Approving a write permission lets the file land',
covers: ['tool-execute', 'permission-allow'],
modelAgnosticBecause:
'Every coding model reaches for a write tool when told to create a file; the assertion is that the file exists afterwards, not which tool was chosen.',
},
{
id: 'live-permission-deny',
title: 'Denying a write permission keeps the file off disk and still ends the turn',
covers: ['permission-deny'],
modelAgnosticBecause:
'The check is the absence of the file plus a completed turn. How the model narrates the refusal is not asserted.',
},
{
id: 'live-interrupt',
title: 'Interrupting a long answer stops the stream',
covers: ['interrupt'],
modelAgnosticBecause:
'Any model produces a long enough response to a "list many items" prompt to be interrupted mid-stream; the assertion is that streaming stops, not where it stopped.',
},
{
id: 'live-reconnect',
title: 'Reconnecting mid-turn does not duplicate the reply',
covers: ['reconnect'],
modelAgnosticBecause:
'Counts assistant messages before and after a reconnect. Independent of content.',
},
{
id: 'live-session-recovery',
title: 'Reloading history returns the same turns the socket delivered',
covers: ['session-recover'],
modelAgnosticBecause:
'Compares the transcript against what this run itself observed, so there is no fixed expected text.',
},
] as const
/** Coverage the live catalog deliberately does not attempt. */
export const LIVE_FLOW_EXCLUSIONS: Readonly<Record<string, string>> = {
'api-error':
'Inducing a provider error means breaking the credentials or the base URL, which would prove the harness misconfigured itself rather than that the app handles an upstream failure. The deterministic lane covers it with the mock CLI.',
'tool-error':
'Depends on a model choosing a tool that then fails. Reachable, but not reliably enough across providers to belong in a gate.',
'runtime-select':
'Exercised implicitly: the run pins the provider under test with set_runtime_config before the first prompt, and a wrong pin shows up as the whole run failing.',
}