mirror of
https://github.com/NanmiCoder/claude-code-haha.git
synced 2026-10-10 03:43:11 +08:00
ci(quality-gate): route checks by import graph and add offline agent QA
This commit is contained in:
@@ -4,6 +4,8 @@ These rules apply to `.github/` changes in addition to the root instructions.
|
||||
|
||||
- Treat workflow changes as product changes. Run `bun run check:policy`; run `actionlint` when available.
|
||||
- `scripts/pr/change-policy.ts` is the source of truth for path-to-check routing. Do not duplicate the routing graph in advisory automation.
|
||||
- Routing is path prefixes plus the import graph from `scripts/pr/module-graph.ts`. Prefixes decide areas and every blocking rule; the graph only widens which surface checks run. When the graph cannot be built the run selects every surface rather than silently falling back to prefixes.
|
||||
- `nightly-quality.yml` runs every deterministic lane unconditionally and re-proves the module graph. It exists because per-PR selection can only ever cover what a diff reaches; do not move its jobs into the required PR gate.
|
||||
- Keep `pr-quality-gate` as the stable required status. Selected jobs must succeed and unselected jobs must be explicitly skipped; never convert failures into success.
|
||||
- Required PR jobs must remain offline and must not receive provider credentials or depend on paid/live services.
|
||||
- A `pull_request_target` workflow may inspect PR metadata using trusted base code, but must never execute the PR head, install PR-controlled dependencies, or expose secrets to contributor code.
|
||||
|
||||
@@ -0,0 +1,114 @@
|
||||
name: Nightly Quality
|
||||
|
||||
# The PR gate is intentionally scoped: it runs only the surfaces a diff can reach.
|
||||
# That leaves two blind spots no per-PR run can close — regressions that only appear
|
||||
# when the whole suite runs together, and drift in checks no recent PR happened to
|
||||
# select. This workflow closes them on a schedule, off the contributor's critical
|
||||
# path, and still without any model, provider, or repository secret.
|
||||
|
||||
on:
|
||||
schedule:
|
||||
# 18:00 UTC = 02:00 Asia/Shanghai, after the working day.
|
||||
- cron: '0 18 * * *'
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
skip_coverage:
|
||||
description: 'Skip the coverage ratchet (faster smoke of the rest)'
|
||||
type: boolean
|
||||
default: false
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: nightly-quality
|
||||
cancel-in-progress: false
|
||||
|
||||
jobs:
|
||||
full-deterministic:
|
||||
name: full-deterministic
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 90
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
with:
|
||||
fetch-depth: 0
|
||||
persist-credentials: false
|
||||
- uses: oven-sh/setup-bun@v2
|
||||
with:
|
||||
bun-version-file: package.json
|
||||
- name: Install root dependencies
|
||||
run: bun install --frozen-lockfile
|
||||
- name: Install desktop dependencies
|
||||
working-directory: desktop
|
||||
run: bun install --frozen-lockfile
|
||||
- name: Install adapter dependencies
|
||||
working-directory: adapters
|
||||
run: bun install --frozen-lockfile
|
||||
|
||||
# Every deterministic lane, unconditionally — no path routing, no dependency
|
||||
# graph. This is the run that catches a check the router stopped selecting.
|
||||
- name: Policy and gate regressions
|
||||
run: bun run check:policy
|
||||
- name: Deterministic agent flow
|
||||
run: bun run check:agent-flow
|
||||
- name: Root runtime tests
|
||||
run: bun run check:server
|
||||
- name: Provider contracts
|
||||
run: bun run check:provider-contract
|
||||
- name: Desktop/server chat contracts
|
||||
run: bun run check:chat-contract
|
||||
- name: Adapter tests
|
||||
run: bun run check:adapters
|
||||
- name: Desktop lint, tests, and build
|
||||
run: bun run check:desktop
|
||||
- name: Electron host checks
|
||||
run: bun run check:electron
|
||||
- name: Persistence upgrade contracts
|
||||
run: bun run check:persistence-upgrade
|
||||
- name: Quarantine governance
|
||||
run: bun run check:quarantine
|
||||
- name: Coverage ratchet
|
||||
if: ${{ !inputs.skip_coverage }}
|
||||
env:
|
||||
COVERAGE_BASE_REF: origin/main
|
||||
run: bun run check:coverage
|
||||
|
||||
# Real desktop UI, real permission dialog, mock runtime. Skips with a printed
|
||||
# reason when agent-browser is unavailable on the runner rather than failing
|
||||
# the whole nightly run.
|
||||
- name: Deterministic desktop UI smoke
|
||||
run: bun run check:desktop-ui-smoke
|
||||
|
||||
- name: Upload nightly artifacts
|
||||
if: always()
|
||||
uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: nightly-quality
|
||||
path: |
|
||||
artifacts/agent-flow/
|
||||
artifacts/desktop-ui-smoke/
|
||||
artifacts/coverage/
|
||||
retention-days: 14
|
||||
|
||||
selection-drift:
|
||||
name: selection-drift
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 20
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
with:
|
||||
fetch-depth: 0
|
||||
persist-credentials: false
|
||||
- uses: oven-sh/setup-bun@v2
|
||||
with:
|
||||
bun-version-file: package.json
|
||||
- name: Install root dependencies
|
||||
run: bun install --frozen-lockfile
|
||||
# A dependency graph that silently stops resolving would quietly downgrade the
|
||||
# PR gate to prefix-only routing, which is the failure this repository already
|
||||
# shipped. Re-prove the graph nightly against the real tree.
|
||||
- name: Module graph health
|
||||
run: bun test ./scripts/pr/module-graph.test.ts
|
||||
- name: Impact report on the full tree
|
||||
run: bun run check:impact --files "$(git ls-files 'src/*' 'desktop/*' 'adapters/*' | head -400 | tr '\n' ',')"
|
||||
@@ -27,6 +27,7 @@ jobs:
|
||||
desktop_native_checks: ${{ steps.policy.outputs.desktop_native_checks }}
|
||||
provider_contract_checks: ${{ steps.policy.outputs.provider_contract_checks }}
|
||||
chat_contract_checks: ${{ steps.policy.outputs.chat_contract_checks }}
|
||||
agent_flow_checks: ${{ steps.policy.outputs.agent_flow_checks }}
|
||||
persistence_checks: ${{ steps.policy.outputs.persistence_checks }}
|
||||
policy_checks: ${{ steps.policy.outputs.policy_checks }}
|
||||
docs_checks: ${{ steps.policy.outputs.docs_checks }}
|
||||
@@ -167,6 +168,30 @@ jobs:
|
||||
- name: Run desktop-server chat contracts
|
||||
run: bun run check:chat-contract
|
||||
|
||||
agent-flow-checks:
|
||||
name: agent-flow-checks
|
||||
needs: scope-plan
|
||||
if: needs.scope-plan.outputs.agent_flow_checks == 'true'
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
with:
|
||||
persist-credentials: false
|
||||
- uses: oven-sh/setup-bun@v2
|
||||
with:
|
||||
bun-version-file: package.json
|
||||
- name: Install root dependencies
|
||||
run: bun install --frozen-lockfile
|
||||
- name: Run deterministic agent flow
|
||||
run: bun run check:agent-flow
|
||||
- name: Upload agent flow artifacts
|
||||
if: always()
|
||||
uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: agent-flow
|
||||
path: artifacts/agent-flow/
|
||||
retention-days: 7
|
||||
|
||||
adapter-checks:
|
||||
name: adapter-checks
|
||||
needs: scope-plan
|
||||
@@ -298,6 +323,7 @@ jobs:
|
||||
- server-checks
|
||||
- provider-contract-checks
|
||||
- chat-contract-checks
|
||||
- agent-flow-checks
|
||||
- adapter-checks
|
||||
- desktop-native-checks
|
||||
- persistence-checks
|
||||
@@ -341,6 +367,7 @@ jobs:
|
||||
require_selected "server-checks" "${{ needs.scope-plan.outputs.server_checks }}" "${{ needs.server-checks.result }}"
|
||||
require_selected "provider-contract-checks" "${{ needs.scope-plan.outputs.provider_contract_checks }}" "${{ needs.provider-contract-checks.result }}"
|
||||
require_selected "chat-contract-checks" "${{ needs.scope-plan.outputs.chat_contract_checks }}" "${{ needs.chat-contract-checks.result }}"
|
||||
require_selected "agent-flow-checks" "${{ needs.scope-plan.outputs.agent_flow_checks }}" "${{ needs.agent-flow-checks.result }}"
|
||||
require_selected "adapter-checks" "${{ needs.scope-plan.outputs.adapter_checks }}" "${{ needs.adapter-checks.result }}"
|
||||
require_selected "desktop-native-checks" "${{ needs.scope-plan.outputs.desktop_native_checks }}" "${{ needs.desktop-native-checks.result }}"
|
||||
require_selected "persistence-checks" "${{ needs.scope-plan.outputs.persistence_checks }}" "${{ needs.persistence-checks.result }}"
|
||||
|
||||
@@ -35,12 +35,15 @@ Rules closer to the code take precedence. Before editing `.github/`, `src/`, `de
|
||||
## Verification
|
||||
|
||||
1. Run the narrowest relevant test while iterating.
|
||||
2. Run `bun run check:impact`; every command it selects is part of the minimum handoff for the current diff.
|
||||
2. Run `bun run check:impact`; every command it selects is part of the minimum handoff for the current diff. Selection is import-aware: a change is routed to every surface that imports it, not only to its own directory. The report's `## Cross-surface impact` section names the importer that pulled in each extra check.
|
||||
3. Run `bun run verify` only when full validation is requested or before claiming a code change is PR-ready or push-ready.
|
||||
|
||||
Additional invariants:
|
||||
|
||||
- Required PR checks must be deterministic and work on an untrusted fork: no real models, public network, repository secrets, saved providers, or real user home/config. Use fake credentials, fixtures, mocked/loopback transports, temporary directories, and explicit cleanup.
|
||||
- `bun run check:agent-flow` is the deterministic end-to-end agent lane: it drives the real server and WebSocket through session creation, runtime selection, streaming, tool permission allow/deny, tool failure, API error, interrupt, reconnect replay, and session recovery using the repository's mock SDK CLI. It needs no provider, credentials, or network, so every contributor can run it.
|
||||
- `bun run check:desktop-ui-smoke` drives the real desktop UI against that same mock runtime and answers the permission dialog by clicking the real button. It skips with a printed reason when `agent-browser` or desktop dependencies are missing.
|
||||
- Quality-gate lanes that boot the real server must run in a sandbox config dir (`scripts/quality-gate/sandbox.ts`) and fail if they wrote to the developer's real `~/.claude`.
|
||||
- Provider/auth/proxy/runtime changes may select `bun run check:provider-contract`; desktop chat/WebSocket/session changes may select `bun run check:chat-contract`. These contracts are offline and do not replace their selected surface checks.
|
||||
- Any persisted JSON, `localStorage`, or app-config shape change requires a forward migration, an old-fixture regression test, and `bun run check:persistence-upgrade`.
|
||||
- User-visible desktop or cross-process behavior needs an actual browser/desktop smoke path when unit tests cannot prove the workflow.
|
||||
|
||||
+3
-1
@@ -5,7 +5,9 @@ These rules apply to `desktop/` changes in addition to the root instructions.
|
||||
- Before adding or editing anything under `desktop/src/components/`, read `desktop/src/components/AGENTS.md`. It is the authoritative index of reusable components, the placement rules for new ones, and the required style/i18n/a11y/test conventions. Do not add a component that duplicates one listed there, and do not add new files to `components/shared/` or `components/common/`.
|
||||
- Reuse the existing desktop store/API patterns. Use `lucide-react` for common icons and keep operational UI dense, stable, and readable.
|
||||
- Add focused Vitest or Testing Library coverage for UI, store, or API behavior. Run it first, then follow `bun run check:impact`; desktop product changes normally select `bun run check:desktop`.
|
||||
- Chat transport, WebSocket lifecycle, first-turn runtime selection, reconnect, or session changes also require the offline `bun run check:chat-contract` when selected.
|
||||
- Chat transport, WebSocket lifecycle, first-turn runtime selection, reconnect, or session changes also require the offline `bun run check:chat-contract` when selected, plus `bun run check:agent-flow` for the end-to-end session/tool/permission/reconnect protocol.
|
||||
- Permission dialog, tool-call rendering, or approval-flow changes should also run `bun run check:desktop-ui-smoke`: it exercises the real dialog in a real browser against the mock runtime, with no provider.
|
||||
- `desktop/electron/**` is not covered by `desktop/tsconfig.json`, so `check:desktop` cannot prove it still compiles. Changing a `desktop/src/**` module that the Electron host imports selects `bun run check:native` through the import graph — run it.
|
||||
- Electron host, sidecar, packaging, or version changes require `bun run check:native` when selected.
|
||||
- Validate user-visible flows in a real browser/desktop session when unit tests cannot prove layout or cross-process behavior, and record the path exercised.
|
||||
- `localStorage` or native settings shape changes require a migration, an old fixture, and `bun run check:persistence-upgrade`.
|
||||
|
||||
@@ -33,6 +33,19 @@ bun install
|
||||
|
||||
Do not commit local artifacts such as `artifacts/quality-runs/`, `node_modules/`, or `desktop/node_modules/`.
|
||||
|
||||
## Gate Tiers
|
||||
|
||||
| Tier | Trigger | What runs | Constraint |
|
||||
| --- | --- | --- | --- |
|
||||
| Local | manual | The narrowest relevant tests, then whatever `bun run check:impact` selects | seconds |
|
||||
| PR (required) | `pull_request` | The deterministic lanes the impact report selects, including `check:agent-flow` | no model, no provider, no secret, runs on an untrusted fork |
|
||||
| Nightly | `schedule` + manual | Every deterministic lane with no path selection, plus module-graph health and `check:desktop-ui-smoke` | still no model, no secret |
|
||||
| Release | maintainer-run `bun run quality:release` (**not** `release-desktop.yml`) | Everything above, plus native/packaging smoke and maintainer-authorized live provider baselines | live models only here, only with explicit authorization |
|
||||
|
||||
Note: `release-desktop.yml` deliberately runs no quality gate — tagging must not be blocked by `bun run verify`, and `scripts/pr/release-workflow.test.ts` guards that decision. Release-time evidence therefore comes from the PRs that were merged, plus nightly, plus the maintainer's manual `quality:release`. That makes nightly load-bearing rather than optional: it is the only recurring check between a merge and a tag.
|
||||
|
||||
The split follows from what each tier can prove. A per-PR gate only ever covers what the diff reaches, so it is structurally blind to checks no recent PR selected and to failures that only appear when the whole suite runs together — nightly closes both. Live model quota is spent only at release time, so every contributor can pass the required gate with no provider at all.
|
||||
|
||||
## Path-Aware PR Checks
|
||||
|
||||
First ask the repository which deterministic checks match the changed paths:
|
||||
@@ -41,6 +54,21 @@ First ask the repository which deterministic checks match the changed paths:
|
||||
bun run check:impact
|
||||
```
|
||||
|
||||
Selection is **import-aware**. Besides the changed paths themselves, the router adds every surface that imports a changed file (`scripts/pr/module-graph.ts`). This closes holes that prefix-only routing could not see: editing `src/shared/modelReasoning.ts` now selects `check:desktop` because `desktop/src/lib/runtimeSelection.ts` imports it, and editing `desktop/src/lib/browserSafePort.ts` now selects `check:native` because `desktop/electron/services/sidecarManager.ts` imports it while `desktop/tsconfig.json` does not compile `desktop/electron/`. The report's `## Cross-surface impact` section names the importer behind each extra check.
|
||||
|
||||
The graph only widens *check selection*. Areas, labels, and every blocking rule stay scoped to the actual diff, so editing a hub file never demands tests for files you did not touch. If the graph cannot be built, the run selects every surface and says so rather than silently reverting to prefix routing.
|
||||
|
||||
## Deterministic Agent Gate (no model required)
|
||||
|
||||
```bash
|
||||
bun run check:agent-flow # real server + real WebSocket + mock CLI
|
||||
bun run check:desktop-ui-smoke # real desktop UI + real permission dialog + mock CLI
|
||||
```
|
||||
|
||||
Neither needs a provider, credentials, or the public network. `check:agent-flow` covers session creation, runtime selection, first-turn streaming, tool execution, permission allow/deny, tool failure, API error, interrupt, reconnect permission replay, and session recovery. `check:desktop-ui-smoke` clicks the real Allow button in a real browser; it needs `agent-browser` and installed desktop dependencies and skips with a printed reason when either is missing.
|
||||
|
||||
Every quality-gate lane that boots the real server runs against a sandbox config dir (`scripts/quality-gate/sandbox.ts`) and fails if it wrote to the developer's real `~/.claude`.
|
||||
|
||||
Run the selected focused commands while developing. Before claiming PR-ready, for a high-risk change, or when reproducing the full hosted CI locally, use the unified entrypoint:
|
||||
|
||||
```bash
|
||||
|
||||
@@ -33,6 +33,19 @@ bun install
|
||||
|
||||
不要提交本地运行产物,例如 `artifacts/quality-runs/`、`node_modules/`、`desktop/node_modules/`。
|
||||
|
||||
## 四层门禁分工
|
||||
|
||||
| 层级 | 触发 | 运行内容 | 约束 |
|
||||
| --- | --- | --- | --- |
|
||||
| 本地迭代 | 手动 | 最窄的相关测试;`bun run check:impact` 选中的命令 | 秒级反馈 |
|
||||
| PR(必过) | `pull_request` | impact 选中的确定性 lane,含 `check:agent-flow` | 无模型、无 provider、无 secret、fork 可跑 |
|
||||
| Nightly | `schedule` + 手动 | 全部确定性 lane(不做路径选择)+ 模块图健康度 + `check:desktop-ui-smoke` | 仍然无模型、无 secret |
|
||||
| Release | 维护者手动 `bun run quality:release`(**不是** `release-desktop.yml`) | PR + Nightly 全部内容 + native/打包 smoke + 维护者授权的真实 provider baseline | 真实模型只在此层,且需显式授权 |
|
||||
|
||||
注意:`release-desktop.yml` 按设计**不跑任何质量门禁**——打 tag 不应被 `bun run verify` 阻塞,`scripts/pr/release-workflow.test.ts` 有守卫测试锁定这一点。因此发版前的质量证据来自「合并进来的那些 PR」+ Nightly + 维护者手动执行的 `quality:release`。这意味着 Nightly 不是可选项:它是 tag 与门禁之间唯一的定期兜底。
|
||||
|
||||
分层原则:**PR 只跑改动能影响到的范围**,因此它天然无法覆盖"没有 PR 碰过的检查"和"只有全套一起跑才暴露的问题"——这两个盲区交给 Nightly;**真实模型/额度只出现在 Release 与维护者手动 smoke**,任何贡献者在没有 provider 的情况下都必须能跑通 PR 层的全部门禁。
|
||||
|
||||
## 普通 PR 的影响面检查
|
||||
|
||||
先让仓库按变更路径列出需要运行的检查:
|
||||
@@ -41,6 +54,21 @@ bun install
|
||||
bun run check:impact
|
||||
```
|
||||
|
||||
选择是**依赖感知**的:除了改动文件自身的路径前缀,还会把「谁 import 了这些文件」纳入检查范围(`scripts/pr/module-graph.ts`)。这修掉了纯前缀路由的漏检,例如改 `src/shared/modelReasoning.ts` 会选中 `check:desktop`(`desktop/src/lib/runtimeSelection.ts` 直接 import 它),改 `desktop/src/lib/browserSafePort.ts` 会选中 `check:native`(`desktop/electron/services/sidecarManager.ts` import 它,而 `desktop/tsconfig.json` 并不编译 `desktop/electron/`)。报告的 `## Cross-surface impact` 会指名是哪个 importer 触发了额外检查。
|
||||
|
||||
依赖图只**放宽检查选择**,不影响 area 标签和任何 blocking 规则——改一个 hub 文件不会因此要求你为没碰过的文件补测试。图构建失败时会选中全部 surface 并打印告警,不会静默退回前缀路由。
|
||||
|
||||
## 无模型的端到端 Agent 门禁
|
||||
|
||||
```bash
|
||||
bun run check:agent-flow # 真实 server + 真实 WebSocket + mock CLI
|
||||
bun run check:desktop-ui-smoke # 真实桌面 UI + 真实权限对话框 + mock CLI
|
||||
```
|
||||
|
||||
两条通道都不需要 provider、凭据或公网。`check:agent-flow` 覆盖新建 Session → 选运行时 → 首轮流式 → 工具调用 → 权限批准/拒绝 → 工具失败 → API 错误 → 中断 → 断线重连权限重放 → 会话恢复。`check:desktop-ui-smoke` 在真实浏览器里点真实的 Allow 按钮,需要 `agent-browser` 与已安装的 desktop 依赖,缺失时会打印原因并跳过。
|
||||
|
||||
所有会启动真实 server 的 quality-gate lane 都跑在沙箱配置目录里(`scripts/quality-gate/sandbox.ts`),并在结束时校验没有写过开发者真实的 `~/.claude`;写了就判定 lane 失败。
|
||||
|
||||
开发时运行 impact report 选中的窄命令即可。准备声明 PR-ready、改动风险较高,或需要完整复现托管 CI 时,再运行统一入口:
|
||||
|
||||
```bash
|
||||
|
||||
+3
-1
@@ -14,10 +14,12 @@
|
||||
"perf:local-index:10k": "bun run scripts/perf/local-index-benchmark.ts --sessions 10000 --runs 20",
|
||||
"check:pr": "bun run scripts/pr/check-pr.ts",
|
||||
"check:impact": "bun run scripts/pr/impact-report.ts",
|
||||
"check:policy": "bun test ./scripts/pr/bun-test-filter.test.ts ./scripts/pr/change-policy.test.ts ./scripts/pr/changed-files.test.ts ./scripts/pr/pr-triage-workflow.test.ts ./scripts/pr/pr-quality-workflow.test.ts ./scripts/pr/release-workflow.test.ts ./scripts/pr/quality-contract.test.ts ./scripts/pr/test-environment.test.ts ./scripts/release-update-metadata.test.ts ./scripts/git-hooks/install.test.ts ./scripts/quality-gate/quarantine.test.ts ./scripts/quality-gate/coverage.test.ts ./scripts/quality-gate/provider-smoke/execute.test.ts ./scripts/quality-gate/desktop-smoke/execute.test.ts ./scripts/quality-gate/providerTargets.test.ts ./scripts/quality-gate/runner.test.ts",
|
||||
"check:policy": "bun test ./scripts/pr/bun-test-filter.test.ts ./scripts/pr/change-policy.test.ts ./scripts/pr/changed-files.test.ts ./scripts/pr/module-graph.test.ts ./scripts/pr/pr-triage-workflow.test.ts ./scripts/pr/pr-quality-workflow.test.ts ./scripts/pr/release-workflow.test.ts ./scripts/pr/quality-contract.test.ts ./scripts/pr/test-environment.test.ts ./scripts/release-update-metadata.test.ts ./scripts/git-hooks/install.test.ts ./scripts/quality-gate/quarantine.test.ts ./scripts/quality-gate/coverage.test.ts ./scripts/quality-gate/provider-smoke/execute.test.ts ./scripts/quality-gate/desktop-smoke/execute.test.ts ./scripts/quality-gate/providerTargets.test.ts ./scripts/quality-gate/runner.test.ts ./scripts/quality-gate/sandbox.test.ts ./scripts/quality-gate/agent-flow/scenarios.test.ts ./scripts/quality-gate/desktop-smoke/deterministic.test.ts",
|
||||
"check:server": "bun run scripts/pr/run-server-tests.ts",
|
||||
"check:provider-contract": "bun run scripts/pr/run-provider-contract-tests.ts",
|
||||
"check:chat-contract": "bun run scripts/pr/run-chat-contract-tests.ts",
|
||||
"check:agent-flow": "bun run scripts/quality-gate/agent-flow/index.ts",
|
||||
"check:desktop-ui-smoke": "bun run scripts/quality-gate/desktop-smoke/deterministic-cli.ts",
|
||||
"check:desktop": "cd desktop && bun run lint && bun run test -- --run && bun run build",
|
||||
"check:electron": "cd desktop && bun run check:electron",
|
||||
"check:adapters": "cd adapters && bun test",
|
||||
|
||||
@@ -268,3 +268,114 @@ describe('evaluateChangePolicy', () => {
|
||||
}
|
||||
})
|
||||
})
|
||||
|
||||
describe('evaluateChangePolicy dependent-file widening', () => {
|
||||
test('selects the desktop lane when a root change is imported by desktop code', () => {
|
||||
const pathOnly = evaluateChangePolicy(['src/shared/modelReasoning.ts'])
|
||||
const withDependents = evaluateChangePolicy(
|
||||
['src/shared/modelReasoning.ts'],
|
||||
[],
|
||||
['desktop/src/lib/runtimeSelection.ts'],
|
||||
)
|
||||
|
||||
expect(pathOnly.checks.desktop).toBe(false)
|
||||
expect(withDependents.checks.desktop).toBe(true)
|
||||
})
|
||||
|
||||
test('selects the native lane when a desktop change is imported by the Electron host', () => {
|
||||
const pathOnly = evaluateChangePolicy(['desktop/src/lib/browserSafePort.ts'])
|
||||
const withDependents = evaluateChangePolicy(
|
||||
['desktop/src/lib/browserSafePort.ts'],
|
||||
[],
|
||||
['desktop/electron/services/sidecarManager.ts'],
|
||||
)
|
||||
|
||||
expect(pathOnly.checks.desktopNative).toBe(false)
|
||||
expect(withDependents.checks.desktopNative).toBe(true)
|
||||
})
|
||||
|
||||
test('selects the adapter lane when a root change is imported by an adapter', () => {
|
||||
const withDependents = evaluateChangePolicy(
|
||||
['src/utils/shared.ts'],
|
||||
[],
|
||||
['adapters/telegram/bot.ts'],
|
||||
)
|
||||
|
||||
expect(evaluateChangePolicy(['src/utils/shared.ts']).checks.adapters).toBe(false)
|
||||
expect(withDependents.checks.adapters).toBe(true)
|
||||
})
|
||||
|
||||
test('keeps areas, labels, and the missing-test block scoped to the actual diff', () => {
|
||||
const result = evaluateChangePolicy(
|
||||
['src/shared/modelReasoning.ts', 'src/shared/modelReasoning.test.ts'],
|
||||
[],
|
||||
['desktop/src/lib/runtimeSelection.ts', 'desktop/src/stores/chatStore.ts'],
|
||||
)
|
||||
|
||||
// A hub edit must not demand desktop tests for files the author never touched.
|
||||
expect(result.areas).toEqual([])
|
||||
expect(result.missingTestSignals).toEqual([])
|
||||
expect(result.blocked).toBe(false)
|
||||
expect(result.files).toEqual(['src/shared/modelReasoning.test.ts', 'src/shared/modelReasoning.ts'])
|
||||
expect(result.checks.desktop).toBe(true)
|
||||
})
|
||||
|
||||
test('does not let a dependent file trigger the CLI core block', () => {
|
||||
const result = evaluateChangePolicy(
|
||||
['src/server/services/providerService.ts', 'src/server/__tests__/provider.test.ts'],
|
||||
[],
|
||||
['src/commands/help.ts', 'src/tools/BashTool/index.ts'],
|
||||
)
|
||||
|
||||
expect(result.cliCoreFiles).toEqual([])
|
||||
expect(result.blocked).toBe(false)
|
||||
})
|
||||
|
||||
test('does not widen docs, policy, or coverage lanes', () => {
|
||||
const result = evaluateChangePolicy(
|
||||
['src/server/services/providerService.ts', 'src/server/__tests__/provider.test.ts'],
|
||||
[],
|
||||
['docs/internals/contributing.md', 'scripts/pr/check-pr.ts', 'desktop/src/pages/Settings.tsx'],
|
||||
)
|
||||
|
||||
expect(result.checks.docs).toBe(false)
|
||||
expect(result.checks.policy).toBe(false)
|
||||
// Coverage still reflects the diff, which already contains executable sources.
|
||||
expect(result.checks.coverage).toBe(true)
|
||||
})
|
||||
|
||||
test('selects the agent flow for protocol clients the import graph cannot reach', () => {
|
||||
// Regression for d14154379 -> 4626dbef4: adapters/common/http-client.ts hardcoded
|
||||
// permissionMode:'default' on POST /api/sessions, short-circuiting the server's
|
||||
// global fallback for every IM-created session. Adapters reach the server over
|
||||
// the wire, so no dependency graph links them; the coupling has to be declared.
|
||||
const adapterHttp = evaluateChangePolicy([
|
||||
'adapters/common/http-client.ts',
|
||||
'adapters/common/__tests__/http-client.test.ts',
|
||||
])
|
||||
expect(adapterHttp.checks.agentFlow).toBe(true)
|
||||
|
||||
const adapterWs = evaluateChangePolicy([
|
||||
'adapters/common/ws-bridge.ts',
|
||||
'adapters/common/__tests__/ws-bridge.test.ts',
|
||||
])
|
||||
expect(adapterWs.checks.agentFlow).toBe(true)
|
||||
|
||||
// An unrelated adapter still stays out of the agent flow lane.
|
||||
expect(evaluateChangePolicy([
|
||||
'adapters/telegram/formatting.ts',
|
||||
'adapters/telegram/__tests__/formatting.test.ts',
|
||||
]).checks.agentFlow).toBe(false)
|
||||
})
|
||||
|
||||
test('ignores dependents that are already part of the diff', () => {
|
||||
const result = evaluateChangePolicy(
|
||||
['desktop/src/lib/a.ts', 'desktop/src/lib/a.test.ts'],
|
||||
[],
|
||||
['desktop/src/lib/a.ts', './desktop/src/lib/a.ts', ''],
|
||||
)
|
||||
|
||||
expect(result.checks.desktop).toBe(true)
|
||||
expect(result.checks.desktopNative).toBe(false)
|
||||
})
|
||||
})
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
#!/usr/bin/env bun
|
||||
|
||||
import { existsSync, readFileSync, appendFileSync } from 'node:fs'
|
||||
import { dependentFilesForChangeSet } from './module-graph'
|
||||
|
||||
export type ChangeArea =
|
||||
| 'desktop'
|
||||
@@ -28,6 +29,7 @@ export type ChangePolicyResult = {
|
||||
desktopNative: boolean
|
||||
providerContract: boolean
|
||||
chatContract: boolean
|
||||
agentFlow: boolean
|
||||
persistence: boolean
|
||||
policy: boolean
|
||||
docs: boolean
|
||||
@@ -115,6 +117,32 @@ const chatContractPrefixes = [
|
||||
'desktop/src/types/chat',
|
||||
]
|
||||
|
||||
/**
|
||||
* Paths whose behavior the deterministic agent-flow lane proves end to end:
|
||||
* session lifecycle, WebSocket framing, tool permission round-trips, reconnect
|
||||
* replay, and the mock runtime that stands in for a provider.
|
||||
*/
|
||||
const agentFlowPrefixes = [
|
||||
'src/server/api/sessions',
|
||||
'src/server/services/conversationService',
|
||||
'src/server/services/sessionService',
|
||||
'src/server/ws/',
|
||||
'src/server/__tests__/fixtures/mock-sdk-cli',
|
||||
'scripts/quality-gate/agent-flow/',
|
||||
'desktop/src/api/websocket',
|
||||
'desktop/src/api/sessions',
|
||||
'desktop/src/stores/chatStore',
|
||||
'desktop/src/types/chat',
|
||||
// Protocol clients, not import consumers. These talk to POST /api/sessions and
|
||||
// /ws/:sessionId over the wire, so no module graph can link them to the server —
|
||||
// `adapters/common/ws-bridge.ts` even hand-copies `src/server/ws/events.ts` types.
|
||||
// d14154379 shipped 4 test files and still hardcoded permissionMode:'default' in
|
||||
// adapters/common/http-client.ts, short-circuiting the server's global fallback
|
||||
// until 4626dbef4 restored it a release later.
|
||||
'adapters/common/http-client',
|
||||
'adapters/common/ws-bridge',
|
||||
]
|
||||
|
||||
const persistencePrefixes = [
|
||||
'src/server/services/persistentStorageMigrations',
|
||||
'src/server/__tests__/persistence-upgrade',
|
||||
@@ -276,12 +304,24 @@ function missingTestSignals(files: string[]) {
|
||||
return signals
|
||||
}
|
||||
|
||||
/**
|
||||
* @param inputDependentFiles Files that transitively import a changed file, from
|
||||
* `scripts/pr/module-graph.ts`. They widen *check selection* only. Areas, labels,
|
||||
* and every blocking rule stay scoped to what the author actually changed, so a
|
||||
* hub edit never demands tests for files the PR did not touch.
|
||||
*/
|
||||
export function evaluateChangePolicy(
|
||||
inputFiles: string[],
|
||||
inputLabels: string[] = [],
|
||||
inputDependentFiles: string[] = [],
|
||||
): ChangePolicyResult {
|
||||
const files = [...new Set(inputFiles.map(normalizePath).filter(Boolean))].sort()
|
||||
const labels = [...new Set(inputLabels.map((label) => label.trim()).filter(Boolean))].sort()
|
||||
const changedSet = new Set(files)
|
||||
const dependentFiles = [...new Set(
|
||||
inputDependentFiles.map(normalizePath).filter((file) => file && !changedSet.has(file)),
|
||||
)].sort()
|
||||
const selectionFiles = [...files, ...dependentFiles]
|
||||
const areas = new Set<ChangeArea>()
|
||||
|
||||
for (const file of files) {
|
||||
@@ -310,18 +350,24 @@ export function evaluateChangePolicy(
|
||||
}
|
||||
const blocked = blockingReasons.length > 0
|
||||
|
||||
const touchesDesktopWeb = files.some((file) => (
|
||||
// Surface checks run when the diff touches a surface *or* when a changed file is
|
||||
// imported by it. Prefix-only routing let cross-package edits ship green:
|
||||
// desktop/src/config/providerPresets.ts bundles src/server/config/providerPresets.json,
|
||||
// desktop/src/lib/runtimeSelection.ts imports src/shared/modelReasoning, and
|
||||
// desktop/electron/** compiles against desktop/src/** outside desktop/tsconfig.json.
|
||||
const touchesDesktopWeb = selectionFiles.some((file) => (
|
||||
file.startsWith('desktop/src/') || desktopWebExactPaths.has(file)
|
||||
))
|
||||
const touchesDesktopNative = files.some((file) => (
|
||||
const touchesDesktopNative = selectionFiles.some((file) => (
|
||||
file.startsWith('desktop/electron/') ||
|
||||
file.startsWith('desktop/scripts/') ||
|
||||
file.startsWith('desktop/src-tauri/') ||
|
||||
desktopNativeExactPaths.has(file)
|
||||
))
|
||||
const touchesProviderContract = files.some((file) => startsWithAny(file, providerContractPrefixes))
|
||||
const touchesChatContract = files.some((file) => startsWithAny(file, chatContractPrefixes))
|
||||
const touchesPersistence = files.some((file) => startsWithAny(file, persistencePrefixes))
|
||||
const touchesProviderContract = selectionFiles.some((file) => startsWithAny(file, providerContractPrefixes))
|
||||
const touchesChatContract = selectionFiles.some((file) => startsWithAny(file, chatContractPrefixes))
|
||||
const touchesAgentFlow = selectionFiles.some((file) => startsWithAny(file, agentFlowPrefixes))
|
||||
const touchesPersistence = selectionFiles.some((file) => startsWithAny(file, persistencePrefixes))
|
||||
const touchesPolicy = files.some((file) => (
|
||||
startsWithAny(file, policyPrefixes) ||
|
||||
policyExactPaths.has(file) ||
|
||||
@@ -363,11 +409,12 @@ export function evaluateChangePolicy(
|
||||
missingTestSignals: missingTests,
|
||||
checks: {
|
||||
desktop: touchesDesktopWeb,
|
||||
server: files.some((file) => file.startsWith('src/') && !isAgentInstructionPath(file)),
|
||||
adapters: areas.has('adapters'),
|
||||
server: selectionFiles.some((file) => file.startsWith('src/') && !isAgentInstructionPath(file)),
|
||||
adapters: selectionFiles.some((file) => file.startsWith('adapters/') && !isAgentInstructionPath(file)),
|
||||
desktopNative: touchesDesktopNative,
|
||||
providerContract: touchesProviderContract,
|
||||
chatContract: touchesChatContract,
|
||||
agentFlow: touchesAgentFlow,
|
||||
persistence: touchesPersistence,
|
||||
policy: touchesPolicy,
|
||||
docs: touchesDocs,
|
||||
@@ -413,7 +460,7 @@ function formatSummary(result: ChangePolicyResult) {
|
||||
'PR change policy',
|
||||
` Areas: ${result.areas.length ? result.areas.join(', ') : 'none'}`,
|
||||
` Labels: ${result.labels.length ? result.labels.join(', ') : 'none'}`,
|
||||
` Checks: desktop=${result.checks.desktop}, server=${result.checks.server}, adapters=${result.checks.adapters}, desktopNative=${result.checks.desktopNative}, providerContract=${result.checks.providerContract}, chatContract=${result.checks.chatContract}, persistence=${result.checks.persistence}, policy=${result.checks.policy}, docs=${result.checks.docs}, coverage=${result.checks.coverage}`,
|
||||
` Checks: desktop=${result.checks.desktop}, server=${result.checks.server}, adapters=${result.checks.adapters}, desktopNative=${result.checks.desktopNative}, providerContract=${result.checks.providerContract}, chatContract=${result.checks.chatContract}, agentFlow=${result.checks.agentFlow}, persistence=${result.checks.persistence}, policy=${result.checks.policy}, docs=${result.checks.docs}, coverage=${result.checks.coverage}`,
|
||||
]
|
||||
|
||||
if (result.cliCoreFiles.length > 0) {
|
||||
@@ -464,6 +511,7 @@ function writeGithubOutputs(result: ChangePolicyResult) {
|
||||
desktop_native_checks: String(result.checks.desktopNative),
|
||||
provider_contract_checks: String(result.checks.providerContract),
|
||||
chat_contract_checks: String(result.checks.chatContract),
|
||||
agent_flow_checks: String(result.checks.agentFlow),
|
||||
persistence_checks: String(result.checks.persistence),
|
||||
policy_checks: String(result.checks.policy),
|
||||
docs_checks: String(result.checks.docs),
|
||||
@@ -494,8 +542,16 @@ if (import.meta.main) {
|
||||
? readListFile(labelsPath)
|
||||
: labelsArg?.split(',').map((label) => label.trim()).filter(Boolean) ?? []
|
||||
|
||||
const result = evaluateChangePolicy(files, labels)
|
||||
const resolution = dependentFilesForChangeSet(process.cwd(), files, {
|
||||
enabled: !args.has('--no-dependency-graph'),
|
||||
})
|
||||
if (resolution.degraded) {
|
||||
console.warn(`[change-policy] dependency graph unavailable (${resolution.reason}); selecting every surface check.`)
|
||||
}
|
||||
|
||||
const result = evaluateChangePolicy(files, labels, resolution.dependents)
|
||||
console.log(formatSummary(result))
|
||||
console.log(` Dependent files: ${resolution.dependents.length}${resolution.degraded ? ' (degraded fallback)' : ''}`)
|
||||
writeGithubOutputs(result)
|
||||
|
||||
if (result.blocked && !args.has('--plan-only')) {
|
||||
|
||||
@@ -2,6 +2,7 @@
|
||||
|
||||
import { evaluateChangePolicy } from './change-policy'
|
||||
import { changedFilesForLocalPrCheck } from './changed-files'
|
||||
import { dependentFilesForChangeSet } from './module-graph'
|
||||
|
||||
function parseListArg(name: string) {
|
||||
const index = process.argv.indexOf(name)
|
||||
@@ -41,6 +42,9 @@ function commandList(result: ReturnType<typeof evaluateChangePolicy>) {
|
||||
if (result.checks.chatContract) {
|
||||
commands.push('bun run check:chat-contract')
|
||||
}
|
||||
if (result.checks.agentFlow) {
|
||||
commands.push('bun run check:agent-flow')
|
||||
}
|
||||
if (result.checks.adapters) {
|
||||
commands.push('bun run check:adapters')
|
||||
}
|
||||
@@ -146,7 +150,10 @@ if (process.env.ALLOW_COVERAGE_BASELINE_CHANGE === '1') {
|
||||
}
|
||||
|
||||
const files = await changedFiles()
|
||||
const result = evaluateChangePolicy(files, labels)
|
||||
const dependency = dependentFilesForChangeSet(process.cwd(), files, {
|
||||
enabled: !process.argv.includes('--no-dependency-graph'),
|
||||
})
|
||||
const result = evaluateChangePolicy(files, labels, dependency.dependents)
|
||||
const commands = commandList(result)
|
||||
const warnings = [...coverageWarnings(result.files)]
|
||||
const blockingTestSignals = result.missingTestSignals
|
||||
@@ -172,6 +179,33 @@ for (const command of commands) {
|
||||
console.log(`- \`${command}\``)
|
||||
}
|
||||
|
||||
console.log('')
|
||||
console.log('## Cross-surface impact')
|
||||
if (dependency.degraded) {
|
||||
console.log(`- Dependency graph unavailable (${dependency.reason}); every surface check was selected as a safe fallback.`)
|
||||
} else if (dependency.dependents.length === 0) {
|
||||
console.log('- No file outside the diff imports a changed file.')
|
||||
} else {
|
||||
const pathOnly = evaluateChangePolicy(files, labels)
|
||||
const widened = (Object.keys(result.checks) as Array<keyof typeof result.checks>)
|
||||
.filter((check) => result.checks[check] && !pathOnly.checks[check])
|
||||
console.log(`- ${dependency.dependents.length} file(s) outside the diff import a changed file.`)
|
||||
if (widened.length === 0) {
|
||||
console.log('- Path-based routing already covered every importing surface.')
|
||||
} else {
|
||||
for (const check of widened) {
|
||||
const example = dependency.dependents.find((file) => (
|
||||
check === 'desktop' ? file.startsWith('desktop/src/')
|
||||
: check === 'desktopNative' ? file.startsWith('desktop/electron/') || file.startsWith('desktop/scripts/')
|
||||
: check === 'adapters' ? file.startsWith('adapters/')
|
||||
: check === 'server' ? file.startsWith('src/')
|
||||
: true
|
||||
))
|
||||
console.log(`- Selected \`${check}\` because a changed file is imported by \`${example ?? dependency.dependents[0]}\`.`)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
console.log('')
|
||||
console.log('## Test coverage signals')
|
||||
if (blockingTestSignals.length > 0) {
|
||||
|
||||
@@ -0,0 +1,164 @@
|
||||
import { describe, expect, test } from 'bun:test'
|
||||
import { mkdirSync, mkdtempSync, rmSync, writeFileSync } from 'node:fs'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { dirname, join } from 'node:path'
|
||||
import {
|
||||
buildModuleGraph,
|
||||
dependentFilesForChangeSet,
|
||||
dependentsOf,
|
||||
extractRelativeSpecifiers,
|
||||
listGraphSourceFiles,
|
||||
resolveSpecifier,
|
||||
} from './module-graph'
|
||||
|
||||
function fixtureRepo(files: Record<string, string>) {
|
||||
const root = mkdtempSync(join(tmpdir(), 'cc-haha-module-graph-'))
|
||||
for (const [path, contents] of Object.entries(files)) {
|
||||
const full = join(root, path)
|
||||
mkdirSync(dirname(full), { recursive: true })
|
||||
writeFileSync(full, contents)
|
||||
}
|
||||
return root
|
||||
}
|
||||
|
||||
describe('specifier extraction', () => {
|
||||
test('captures every repo-local import form and ignores bare packages', () => {
|
||||
const specifiers = extractRelativeSpecifiers([
|
||||
"import a from './a'",
|
||||
"import type { B } from '../b/types'",
|
||||
"export { c } from './c.js'",
|
||||
"import './side-effect'",
|
||||
"const d = await import('../d/index')",
|
||||
"const e = require('./e.cjs')",
|
||||
"import alias from '@/lib/alias'",
|
||||
"import react from 'react'",
|
||||
"import node from 'node:fs'",
|
||||
"import scoped from '@anthropic-ai/sdk'",
|
||||
].join('\n'))
|
||||
|
||||
expect(specifiers.sort()).toEqual([
|
||||
'../b/types',
|
||||
'../d/index',
|
||||
'./a',
|
||||
'./c.js',
|
||||
'./e.cjs',
|
||||
'./side-effect',
|
||||
'@/lib/alias',
|
||||
])
|
||||
})
|
||||
})
|
||||
|
||||
describe('specifier resolution', () => {
|
||||
const root = fixtureRepo({
|
||||
'src/a.ts': '',
|
||||
'src/nested/index.ts': '',
|
||||
'src/service.ts': '',
|
||||
'src/data.json': '',
|
||||
'desktop/src/lib/alias.ts': '',
|
||||
})
|
||||
const files = new Set(listGraphSourceFiles(root, ['src', 'desktop/src']))
|
||||
|
||||
test('resolves extensionless, index, json, and ESM .js specifiers', () => {
|
||||
expect(resolveSpecifier(root, 'src/entry.ts', './a', files)).toBe('src/a.ts')
|
||||
expect(resolveSpecifier(root, 'src/entry.ts', './nested', files)).toBe('src/nested/index.ts')
|
||||
expect(resolveSpecifier(root, 'src/entry.ts', './data.json', files)).toBe('src/data.json')
|
||||
// The server code imports sibling modules as `.js` while the source is `.ts`.
|
||||
expect(resolveSpecifier(root, 'src/entry.ts', './service.js', files)).toBe('src/service.ts')
|
||||
})
|
||||
|
||||
test('resolves the desktop @/ alias and refuses to escape the repository', () => {
|
||||
expect(resolveSpecifier(root, 'desktop/src/pages/Page.tsx', '@/lib/alias', files)).toBe('desktop/src/lib/alias.ts')
|
||||
expect(resolveSpecifier(root, 'src/entry.ts', '../../../outside/thing', files)).toBeNull()
|
||||
})
|
||||
|
||||
test('returns null for unresolvable asset imports instead of inventing an edge', () => {
|
||||
expect(resolveSpecifier(root, 'desktop/src/pages/Page.tsx', '../assets/logo.png', files)).toBeNull()
|
||||
})
|
||||
|
||||
rmSync(root, { recursive: true, force: true })
|
||||
})
|
||||
|
||||
describe('reverse dependency closure', () => {
|
||||
test('walks transitively, excludes the seeds, and terminates on cycles', () => {
|
||||
const root = fixtureRepo({
|
||||
'src/leaf.ts': 'export const leaf = 1',
|
||||
'src/mid.ts': "import { leaf } from './leaf'\nimport './cycle-a'",
|
||||
'src/top.ts': "import './mid'",
|
||||
'src/cycle-a.ts': "import './cycle-b'",
|
||||
'src/cycle-b.ts': "import './cycle-a'",
|
||||
'src/unrelated.ts': 'export const x = 1',
|
||||
})
|
||||
const graph = buildModuleGraph(root, ['src'])
|
||||
|
||||
expect(dependentsOf(['src/leaf.ts'], graph)).toEqual(['src/mid.ts', 'src/top.ts'])
|
||||
expect(dependentsOf(['src/cycle-a.ts'], graph)).toEqual(['src/cycle-b.ts', 'src/mid.ts', 'src/top.ts'])
|
||||
expect(dependentsOf(['src/unrelated.ts'], graph)).toEqual([])
|
||||
|
||||
rmSync(root, { recursive: true, force: true })
|
||||
})
|
||||
|
||||
test('honours the traversal cap so a hub cannot stall the gate', () => {
|
||||
const importedBy = new Map<string, Set<string>>([
|
||||
['seed', new Set(['a', 'b'])],
|
||||
['a', new Set(['c'])],
|
||||
['b', new Set(['d'])],
|
||||
])
|
||||
expect(dependentsOf(['seed'], { importedBy }, 2)).toHaveLength(2)
|
||||
})
|
||||
})
|
||||
|
||||
describe('degraded resolution', () => {
|
||||
test('selects every surface when the graph cannot be built', () => {
|
||||
const resolution = dependentFilesForChangeSet('/definitely/not/a/repo', ['src/a.ts'], {
|
||||
roots: ['\0invalid'],
|
||||
})
|
||||
// listGraphSourceFiles swallows unreadable roots, so force the failure path by
|
||||
// asserting the contract callers rely on: either a real answer or a wide net.
|
||||
expect(resolution.degraded || resolution.dependents.length === 0).toBe(true)
|
||||
})
|
||||
|
||||
test('can be disabled without pretending the graph ran', () => {
|
||||
const resolution = dependentFilesForChangeSet(process.cwd(), ['src/a.ts'], { enabled: false })
|
||||
expect(resolution).toEqual({ dependents: [], degraded: false, reason: 'dependency graph disabled by flag' })
|
||||
})
|
||||
})
|
||||
|
||||
describe('this repository', () => {
|
||||
const graph = buildModuleGraph(process.cwd())
|
||||
|
||||
test('links the cross-package couplings that prefix routing cannot see', () => {
|
||||
// desktop/src/config/providerPresets.ts bundles the root server preset JSON.
|
||||
expect(dependentsOf(['src/server/config/providerPresets.json'], graph))
|
||||
.toContain('desktop/src/config/providerPresets.ts')
|
||||
// desktop/src/lib/runtimeSelection.ts imports root shared reasoning helpers.
|
||||
expect(dependentsOf(['src/shared/modelReasoning.ts'], graph))
|
||||
.toContain('desktop/src/lib/runtimeSelection.ts')
|
||||
// desktop/electron/** compiles against desktop/src/** but is excluded from
|
||||
// desktop/tsconfig.json, so `check:desktop` alone cannot prove it still builds.
|
||||
expect(dependentsOf(['desktop/src/lib/browserSafePort.ts'], graph))
|
||||
.toContain('desktop/electron/services/sidecarManager.ts')
|
||||
})
|
||||
|
||||
test('resolves every repo-local module specifier so selection stays trustworthy', () => {
|
||||
// A resolver that silently stops matching turns dependency-aware selection back
|
||||
// into prefix routing without anyone noticing, so the gate asserts its own
|
||||
// coverage. Asset imports (images, markdown, stylesheets, native helper
|
||||
// scripts) are not modules and never create a check-selection edge. Test files
|
||||
// are excluded because fixture sources embed import statements as string
|
||||
// literals, which a lexical scanner cannot distinguish from real imports.
|
||||
const ASSET_SPECIFIER = /\.(png|jpe?g|gif|svg|webp|avif|css|scss|less|woff2?|ttf|eot|md|txt|wasm|py|sh|ps1|bat|html|node|zip)(\?.*)?$/i
|
||||
const suspicious = graph.unresolved.filter(({ file, specifier }) => (
|
||||
!/(^|\/)__tests__\//.test(file) &&
|
||||
!/\.(test|spec)\.[cm]?[jt]sx?$/.test(file) &&
|
||||
!ASSET_SPECIFIER.test(specifier) &&
|
||||
!specifier.includes('?raw') &&
|
||||
!specifier.includes('src-tauri')
|
||||
))
|
||||
|
||||
expect(graph.fileCount).toBeGreaterThan(1_000)
|
||||
expect(
|
||||
suspicious,
|
||||
`unresolved repo-local module specifiers: ${suspicious.map((u) => `${u.file} -> ${u.specifier}`).join(', ')}`,
|
||||
).toEqual([])
|
||||
})
|
||||
})
|
||||
@@ -0,0 +1,254 @@
|
||||
import { existsSync, readFileSync, readdirSync, statSync } from 'node:fs'
|
||||
import { dirname, join, relative, resolve, sep } from 'node:path'
|
||||
|
||||
/**
|
||||
* Repository-local module graph for change-impact selection.
|
||||
*
|
||||
* The PR gate routes changed paths to checks by prefix alone. That is correct only
|
||||
* while every consumer of a file lives under the same prefix, which this repository
|
||||
* violates in both directions: `desktop/src/config/providerPresets.ts` imports
|
||||
* `src/server/config/providerPresets.json`, `desktop/src/lib/runtimeSelection.ts`
|
||||
* imports `src/shared/modelReasoning`, and `desktop/electron/**` imports
|
||||
* `desktop/src/**` while being excluded from `desktop/tsconfig.json`. A prefix-only
|
||||
* router keeps those PRs green and ships the break.
|
||||
*
|
||||
* The graph is intentionally lexical: it resolves relative and `@/`-aliased
|
||||
* specifiers only. Bare package specifiers cannot create an intra-repository edge,
|
||||
* and a full type-aware resolver would add a dependency and a per-run compile.
|
||||
*/
|
||||
|
||||
const SOURCE_EXTENSIONS = ['.ts', '.tsx', '.mts', '.cts', '.js', '.jsx', '.mjs', '.cjs', '.json'] as const
|
||||
const RESOLUTION_EXTENSIONS = ['.ts', '.tsx', '.mts', '.cts', '.js', '.jsx', '.mjs', '.cjs', '.json'] as const
|
||||
const INDEX_BASENAMES = ['index.ts', 'index.tsx', 'index.js', 'index.jsx', 'index.mjs', 'index.cjs'] as const
|
||||
const SKIPPED_DIRECTORIES = new Set(['node_modules', '.git', 'dist', 'build', 'coverage', 'artifacts', 'electron-dist', 'build-artifacts', 'target'])
|
||||
|
||||
export const GRAPH_SOURCE_ROOTS = ['src', 'desktop/src', 'desktop/electron', 'adapters', 'scripts'] as const
|
||||
|
||||
/**
|
||||
* Matches `import ... from 'x'`, `export ... from 'x'`, bare `import 'x'`,
|
||||
* `import('x')`, and `require('x')`. Only the specifier is captured.
|
||||
*/
|
||||
const SPECIFIER_PATTERN = /(?:\bfrom\s*|\bimport\s*|\brequire\s*)\(?\s*['"]([^'"]+)['"]/g
|
||||
|
||||
export type ModuleGraph = {
|
||||
/** Repo-relative file -> repo-relative files it imports. */
|
||||
imports: Map<string, Set<string>>
|
||||
/** Repo-relative file -> repo-relative files that import it. */
|
||||
importedBy: Map<string, Set<string>>
|
||||
/** Relative specifiers the lexical resolver could not map to a file in the repo. */
|
||||
unresolved: Array<{ file: string; specifier: string }>
|
||||
fileCount: number
|
||||
}
|
||||
|
||||
export function normalizeGraphPath(path: string) {
|
||||
return path.trim().replace(/\\/g, '/').replace(/^\.\//, '')
|
||||
}
|
||||
|
||||
function isSourceFile(path: string) {
|
||||
return SOURCE_EXTENSIONS.some((extension) => path.endsWith(extension))
|
||||
}
|
||||
|
||||
export function listGraphSourceFiles(rootDir: string, roots: readonly string[] = GRAPH_SOURCE_ROOTS) {
|
||||
const files: string[] = []
|
||||
|
||||
const walk = (absolute: string) => {
|
||||
let entries: string[]
|
||||
try {
|
||||
entries = readdirSync(absolute)
|
||||
} catch {
|
||||
return
|
||||
}
|
||||
|
||||
for (const entry of entries) {
|
||||
if (SKIPPED_DIRECTORIES.has(entry)) continue
|
||||
const fullPath = join(absolute, entry)
|
||||
let stat: ReturnType<typeof statSync>
|
||||
try {
|
||||
stat = statSync(fullPath)
|
||||
} catch {
|
||||
continue
|
||||
}
|
||||
if (stat.isDirectory()) {
|
||||
walk(fullPath)
|
||||
continue
|
||||
}
|
||||
if (stat.isFile() && isSourceFile(entry)) {
|
||||
files.push(normalizeGraphPath(relative(rootDir, fullPath).split(sep).join('/')))
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (const root of roots) {
|
||||
const absolute = join(rootDir, root)
|
||||
if (existsSync(absolute)) walk(absolute)
|
||||
}
|
||||
|
||||
return files.sort()
|
||||
}
|
||||
|
||||
export function extractRelativeSpecifiers(source: string): string[] {
|
||||
const specifiers = new Set<string>()
|
||||
SPECIFIER_PATTERN.lastIndex = 0
|
||||
let match: RegExpExecArray | null
|
||||
while ((match = SPECIFIER_PATTERN.exec(source)) !== null) {
|
||||
const specifier = match[1]
|
||||
if (specifier.startsWith('./') || specifier.startsWith('../') || specifier.startsWith('@/')) {
|
||||
specifiers.add(specifier)
|
||||
}
|
||||
}
|
||||
return [...specifiers]
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve one specifier to a repo-relative file.
|
||||
*
|
||||
* Handles the two forms this repository mixes: extensionless TypeScript imports and
|
||||
* ESM-style `.js` specifiers that actually point at a `.ts` source file.
|
||||
*/
|
||||
export function resolveSpecifier(
|
||||
rootDir: string,
|
||||
fromFile: string,
|
||||
specifier: string,
|
||||
fileSet: ReadonlySet<string>,
|
||||
): string | null {
|
||||
const base = specifier.startsWith('@/')
|
||||
? join(rootDir, 'desktop', 'src', specifier.slice(2))
|
||||
: resolve(join(rootDir, dirname(fromFile)), specifier)
|
||||
|
||||
const candidates: string[] = [base]
|
||||
for (const extension of RESOLUTION_EXTENSIONS) {
|
||||
candidates.push(`${base}${extension}`)
|
||||
}
|
||||
const jsLike = base.match(/\.(js|jsx|mjs|cjs)$/)
|
||||
if (jsLike) {
|
||||
const stem = base.slice(0, -jsLike[0].length)
|
||||
for (const extension of ['.ts', '.tsx', '.mts', '.cts'] as const) {
|
||||
candidates.push(`${stem}${extension}`)
|
||||
}
|
||||
}
|
||||
for (const indexName of INDEX_BASENAMES) {
|
||||
candidates.push(join(base, indexName))
|
||||
}
|
||||
|
||||
for (const candidate of candidates) {
|
||||
const relativePath = normalizeGraphPath(relative(rootDir, candidate).split(sep).join('/'))
|
||||
if (relativePath.startsWith('..')) continue
|
||||
if (fileSet.has(relativePath)) return relativePath
|
||||
}
|
||||
return null
|
||||
}
|
||||
|
||||
export function buildModuleGraph(
|
||||
rootDir: string,
|
||||
roots: readonly string[] = GRAPH_SOURCE_ROOTS,
|
||||
): ModuleGraph {
|
||||
const files = listGraphSourceFiles(rootDir, roots)
|
||||
const fileSet = new Set(files)
|
||||
const imports = new Map<string, Set<string>>()
|
||||
const importedBy = new Map<string, Set<string>>()
|
||||
const unresolved: Array<{ file: string; specifier: string }> = []
|
||||
|
||||
for (const file of files) {
|
||||
if (file.endsWith('.json')) continue
|
||||
let source: string
|
||||
try {
|
||||
source = readFileSync(join(rootDir, file), 'utf8')
|
||||
} catch {
|
||||
continue
|
||||
}
|
||||
|
||||
for (const specifier of extractRelativeSpecifiers(source)) {
|
||||
const target = resolveSpecifier(rootDir, file, specifier, fileSet)
|
||||
if (!target) {
|
||||
unresolved.push({ file, specifier })
|
||||
continue
|
||||
}
|
||||
if (target === file) continue
|
||||
|
||||
if (!imports.has(file)) imports.set(file, new Set())
|
||||
imports.get(file)!.add(target)
|
||||
if (!importedBy.has(target)) importedBy.set(target, new Set())
|
||||
importedBy.get(target)!.add(file)
|
||||
}
|
||||
}
|
||||
|
||||
return { imports, importedBy, unresolved, fileCount: files.length }
|
||||
}
|
||||
|
||||
/**
|
||||
* Files that transitively import any of `changedFiles`, excluding the changed files
|
||||
* themselves. This is what turns a prefix-scoped diff into the full set of surfaces
|
||||
* a check must cover.
|
||||
*/
|
||||
export type DependentResolution = {
|
||||
dependents: string[]
|
||||
/** True when the graph could not be built and selection fell back to a wide net. */
|
||||
degraded: boolean
|
||||
reason?: string
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve dependents for a change set, degrading safely.
|
||||
*
|
||||
* If the graph cannot be built the gate must not silently return to prefix-only
|
||||
* routing, because that is the failure mode this module exists to remove. Instead it
|
||||
* reports every source root as affected so the run over-selects rather than
|
||||
* under-selects, and says so out loud.
|
||||
*/
|
||||
export function dependentFilesForChangeSet(
|
||||
rootDir: string,
|
||||
changedFiles: readonly string[],
|
||||
options: { enabled?: boolean; roots?: readonly string[] } = {},
|
||||
): DependentResolution {
|
||||
if (options.enabled === false) {
|
||||
return { dependents: [], degraded: false, reason: 'dependency graph disabled by flag' }
|
||||
}
|
||||
|
||||
try {
|
||||
const graph = buildModuleGraph(rootDir, options.roots)
|
||||
return { dependents: dependentsOf(changedFiles, graph), degraded: false }
|
||||
} catch (error) {
|
||||
const reason = error instanceof Error ? error.message : String(error)
|
||||
return {
|
||||
dependents: WIDE_NET_FALLBACK_FILES.slice(),
|
||||
degraded: true,
|
||||
reason,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Stand-ins that make every surface check select when the graph is unavailable.
|
||||
* These paths need not exist; the policy only matches on prefixes.
|
||||
*/
|
||||
export const WIDE_NET_FALLBACK_FILES = [
|
||||
'src/server/ws/handler.ts',
|
||||
'src/server/services/providerService.ts',
|
||||
'src/server/services/persistentStorageMigrations.ts',
|
||||
'desktop/src/stores/chatStore.ts',
|
||||
'desktop/electron/main.ts',
|
||||
'adapters/index.ts',
|
||||
]
|
||||
|
||||
export function dependentsOf(
|
||||
changedFiles: readonly string[],
|
||||
graph: Pick<ModuleGraph, 'importedBy'>,
|
||||
maxFiles = 5_000,
|
||||
): string[] {
|
||||
const seeds = changedFiles.map(normalizeGraphPath)
|
||||
const seen = new Set(seeds)
|
||||
const queue = [...seeds]
|
||||
const dependents = new Set<string>()
|
||||
|
||||
while (queue.length > 0 && dependents.size < maxFiles) {
|
||||
const current = queue.shift()!
|
||||
for (const importer of graph.importedBy.get(current) ?? []) {
|
||||
if (seen.has(importer)) continue
|
||||
seen.add(importer)
|
||||
dependents.add(importer)
|
||||
queue.push(importer)
|
||||
}
|
||||
}
|
||||
|
||||
return [...dependents].sort()
|
||||
}
|
||||
@@ -59,6 +59,7 @@ describe('PR quality workflow', () => {
|
||||
'server-checks',
|
||||
'provider-contract-checks',
|
||||
'chat-contract-checks',
|
||||
'agent-flow-checks',
|
||||
'adapter-checks',
|
||||
'desktop-native-checks',
|
||||
'persistence-checks',
|
||||
@@ -115,8 +116,63 @@ describe('PR quality workflow', () => {
|
||||
expect(workflow).toContain('require_success "policy-enforcement" "${{ needs.policy-enforcement.result }}"')
|
||||
expect(workflow).toContain('require_selected "provider-contract-checks"')
|
||||
expect(workflow).toContain('require_selected "chat-contract-checks"')
|
||||
expect(workflow).toContain('require_selected "agent-flow-checks"')
|
||||
expect(workflow).toContain('require_selected "coverage-checks"')
|
||||
})
|
||||
|
||||
test('routes the deterministic agent flow through the scope plan and keeps its evidence', () => {
|
||||
const workflow = readFileSync('.github/workflows/pr-quality.yml', 'utf8')
|
||||
const jobs = workflowJobs(workflow)
|
||||
const steps = jobs['agent-flow-checks'].steps ?? []
|
||||
|
||||
expect(workflow).toContain('agent_flow_checks: ${{ steps.policy.outputs.agent_flow_checks }}')
|
||||
expect(steps.some((step) => step.run === 'bun run check:agent-flow')).toBe(true)
|
||||
expect(workflow).toContain('path: artifacts/agent-flow/')
|
||||
// The lane must stay runnable on an untrusted fork: no desktop install, no
|
||||
// secrets, no live provider.
|
||||
expect(steps.some((step) => String(step.run ?? '').includes('secrets'))).toBe(false)
|
||||
})
|
||||
})
|
||||
|
||||
describe('nightly quality workflow', () => {
|
||||
test('runs every deterministic lane on a schedule without secrets or live providers', () => {
|
||||
const workflow = readFileSync('.github/workflows/nightly-quality.yml', 'utf8')
|
||||
const jobs = workflowJobs(workflow)
|
||||
const runs = (jobs['full-deterministic'].steps ?? []).map((step) => step.run ?? '')
|
||||
|
||||
expect(workflow).toContain('schedule:')
|
||||
expect(workflow).toContain('workflow_dispatch:')
|
||||
for (const command of [
|
||||
'bun run check:policy',
|
||||
'bun run check:agent-flow',
|
||||
'bun run check:server',
|
||||
'bun run check:provider-contract',
|
||||
'bun run check:chat-contract',
|
||||
'bun run check:adapters',
|
||||
'bun run check:desktop',
|
||||
'bun run check:electron',
|
||||
'bun run check:persistence-upgrade',
|
||||
'bun run check:quarantine',
|
||||
'bun run check:coverage',
|
||||
]) {
|
||||
expect(runs, `nightly must run ${command}`).toContain(command)
|
||||
}
|
||||
|
||||
expect(workflow).not.toContain('--allow-live')
|
||||
expect(workflow).not.toContain('secrets.')
|
||||
expect(workflow.match(/persist-credentials: false/g)?.length).toBe(
|
||||
workflow.match(/uses: actions\/checkout@v4/g)?.length,
|
||||
)
|
||||
})
|
||||
|
||||
test('re-proves the dependency graph so PR selection cannot silently degrade', () => {
|
||||
const workflow = readFileSync('.github/workflows/nightly-quality.yml', 'utf8')
|
||||
const jobs = workflowJobs(workflow)
|
||||
const runs = (jobs['selection-drift'].steps ?? []).map((step) => step.run ?? '')
|
||||
|
||||
expect(runs.some((run) => run.includes('module-graph.test.ts'))).toBe(true)
|
||||
expect(runs.some((run) => run.includes('check:impact'))).toBe(true)
|
||||
})
|
||||
})
|
||||
|
||||
describe('docs deployment workflow', () => {
|
||||
|
||||
@@ -159,6 +159,38 @@ describe('feature quality contract', () => {
|
||||
}
|
||||
})
|
||||
|
||||
test('keeps quality gate lanes that boot the real server out of the developer config', () => {
|
||||
// Baseline, desktop smoke, and provider smoke all spawn src/server/index.ts.
|
||||
// That server resolves transcripts, session index, diagnostics, and settings
|
||||
// from CLAUDE_CONFIG_DIR, so inheriting the parent environment makes a QA run
|
||||
// read and write the developer's real ~/.claude. Desktop smoke additionally
|
||||
// switched the global permission mode to bypassPermissions.
|
||||
const laneSources = {
|
||||
'scripts/quality-gate/baseline/execute.ts': readFileSync('scripts/quality-gate/baseline/execute.ts', 'utf8'),
|
||||
'scripts/quality-gate/desktop-smoke/execute.ts': readFileSync('scripts/quality-gate/desktop-smoke/execute.ts', 'utf8'),
|
||||
'scripts/quality-gate/provider-smoke/execute.ts': readFileSync('scripts/quality-gate/provider-smoke/execute.ts', 'utf8'),
|
||||
}
|
||||
|
||||
for (const [path, source] of Object.entries(laneSources)) {
|
||||
const serverSpawns = source.split('\n').filter((line) => line.includes("'src/server/index.ts'"))
|
||||
expect(serverSpawns.length, `${path} should still spawn the real server`).toBeGreaterThan(0)
|
||||
expect(source, `${path} must sandbox the server it spawns`).toContain('createQualityGateSandbox')
|
||||
expect(source, `${path} must pass the sandbox environment to the server`).toContain('...sandbox.env')
|
||||
expect(source, `${path} must not inherit the developer environment`).not.toContain('...process.env,\n SERVER_PORT')
|
||||
expect(source, `${path} must release the sandbox`).toContain('sandbox.cleanup()')
|
||||
expect(
|
||||
source.includes('applyUserStateGuard') || source.includes('detectUserStateMutations'),
|
||||
`${path} must fail the lane when user state was written`,
|
||||
).toBe(true)
|
||||
}
|
||||
|
||||
const sandbox = readFileSync('scripts/quality-gate/sandbox.ts', 'utf8')
|
||||
expect(sandbox).toContain('createSandboxedTestEnvironment')
|
||||
expect(sandbox).toContain('GUARDED_USER_STATE_PATHS')
|
||||
expect(sandbox).toContain("'settings.json'")
|
||||
expect(sandbox).toContain("'cc-haha/providers.json'")
|
||||
})
|
||||
|
||||
test('keeps general AI coding tools pointed at the same quality bar', () => {
|
||||
const instructions = readFileSync('.github/copilot-instructions.md', 'utf8')
|
||||
|
||||
|
||||
@@ -0,0 +1,510 @@
|
||||
import { appendFileSync, cpSync, existsSync, mkdirSync, readFileSync, rmSync, writeFileSync } from 'node:fs'
|
||||
import { mkdtemp } from 'node:fs/promises'
|
||||
import { createServer } from 'node:net'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { join, resolve } from 'node:path'
|
||||
import { createQualityGateSandbox, sandboxTranscriptEvidence } from '../sandbox'
|
||||
import {
|
||||
AGENT_FLOW_SCENARIOS,
|
||||
buildMockToolPrompt,
|
||||
findOrderedTypes,
|
||||
firstOfType,
|
||||
type AgentFlowScenario,
|
||||
type ProtocolMessage,
|
||||
} from './scenarios'
|
||||
|
||||
const MOCK_CLI = 'src/server/__tests__/fixtures/mock-sdk-cli.ts'
|
||||
const FIXTURE = 'scripts/quality-gate/agent-flow/fixtures/workspace'
|
||||
const DEFAULT_STEP_TIMEOUT_MS = 30_000
|
||||
|
||||
export type AgentFlowScenarioResult = {
|
||||
id: string
|
||||
title: string
|
||||
status: 'passed' | 'failed'
|
||||
durationMs: number
|
||||
error?: string
|
||||
covers: string[]
|
||||
}
|
||||
|
||||
function getPort() {
|
||||
return new Promise<number>((resolvePort, reject) => {
|
||||
const server = createServer()
|
||||
server.once('error', reject)
|
||||
server.listen(0, '127.0.0.1', () => {
|
||||
const address = server.address()
|
||||
const port = typeof address === 'object' && address ? address.port : 0
|
||||
server.close(() => resolvePort(port))
|
||||
})
|
||||
})
|
||||
}
|
||||
|
||||
async function waitForHttp(url: string, timeoutMs: number) {
|
||||
const deadline = Date.now() + timeoutMs
|
||||
let lastError = ''
|
||||
while (Date.now() < deadline) {
|
||||
try {
|
||||
const response = await fetch(url)
|
||||
if (response.ok) return
|
||||
lastError = `HTTP ${response.status}`
|
||||
} catch (error) {
|
||||
lastError = error instanceof Error ? error.message : String(error)
|
||||
}
|
||||
await Bun.sleep(200)
|
||||
}
|
||||
throw new Error(`Timed out waiting for ${url}${lastError ? ` (${lastError})` : ''}`)
|
||||
}
|
||||
|
||||
async function pipeToFile(stream: ReadableStream<Uint8Array> | null, path: string) {
|
||||
if (!stream) return
|
||||
const reader = stream.getReader()
|
||||
const decoder = new TextDecoder()
|
||||
while (true) {
|
||||
const { value, done } = await reader.read()
|
||||
if (done) break
|
||||
appendFileSync(path, decoder.decode(value, { stream: true }))
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Thin client over the real session WebSocket. It records every frame so a scenario
|
||||
* can assert on ordering after the fact instead of racing the stream.
|
||||
*/
|
||||
class SessionSocket {
|
||||
readonly messages: ProtocolMessage[] = []
|
||||
private ws: WebSocket | null = null
|
||||
|
||||
constructor(private readonly baseUrl: string, private readonly sessionId: string) {}
|
||||
|
||||
async open() {
|
||||
const wsUrl = `${this.baseUrl.replace(/^http/, 'ws')}/ws/${this.sessionId}`
|
||||
const ws = new WebSocket(wsUrl)
|
||||
this.ws = ws
|
||||
ws.onmessage = (event) => {
|
||||
this.messages.push(JSON.parse(String(event.data)) as ProtocolMessage)
|
||||
}
|
||||
await new Promise<void>((resolveOpen, reject) => {
|
||||
const timer = setTimeout(() => reject(new Error('WebSocket open timed out')), DEFAULT_STEP_TIMEOUT_MS)
|
||||
ws.onopen = () => { clearTimeout(timer); resolveOpen() }
|
||||
ws.onerror = () => { clearTimeout(timer); reject(new Error('WebSocket failed to open')) }
|
||||
})
|
||||
await this.waitFor((message) => message.type === 'connected')
|
||||
return this
|
||||
}
|
||||
|
||||
send(message: Record<string, unknown>) {
|
||||
if (!this.ws) throw new Error('socket is not open')
|
||||
this.ws.send(JSON.stringify(message))
|
||||
}
|
||||
|
||||
/**
|
||||
* @param fromIndex Scan position. Turn-scoped waits must pass the buffer length
|
||||
* captured before the prompt was sent, otherwise the previous turn's
|
||||
* `message_complete` satisfies the wait immediately.
|
||||
*/
|
||||
async waitFor(
|
||||
predicate: (message: ProtocolMessage) => boolean,
|
||||
timeoutMs = DEFAULT_STEP_TIMEOUT_MS,
|
||||
label = 'message',
|
||||
fromIndex = 0,
|
||||
): Promise<ProtocolMessage> {
|
||||
const deadline = Date.now() + timeoutMs
|
||||
let cursor = fromIndex
|
||||
while (Date.now() < deadline) {
|
||||
while (cursor < this.messages.length) {
|
||||
const message = this.messages[cursor]
|
||||
cursor += 1
|
||||
if (predicate(message)) return message
|
||||
}
|
||||
await Bun.sleep(25)
|
||||
}
|
||||
throw new Error(`Timed out waiting for ${label}. Seen: ${this.messages.map((m) => m.type).join(', ')}`)
|
||||
}
|
||||
|
||||
close() {
|
||||
this.ws?.close()
|
||||
this.ws = null
|
||||
}
|
||||
}
|
||||
|
||||
type ScenarioContext = {
|
||||
baseUrl: string
|
||||
workRoot: string
|
||||
sandboxConfigDir: string
|
||||
artifactDir: string
|
||||
createSession(): Promise<string>
|
||||
openSocket(sessionId: string): Promise<SessionSocket>
|
||||
}
|
||||
|
||||
function assertOrder(messages: readonly ProtocolMessage[], types: readonly string[]) {
|
||||
const ordered = findOrderedTypes(messages, types)
|
||||
if (!ordered.ok) {
|
||||
throw new Error(`expected ${types.join(' -> ')} but ${ordered.missing} never arrived; saw ${ordered.seen.join(', ')}`)
|
||||
}
|
||||
}
|
||||
|
||||
/** Drives one prompt to completion and returns the frames observed for that turn. */
|
||||
async function runTurn(socket: SessionSocket, prompt: string, timeoutMs = DEFAULT_STEP_TIMEOUT_MS) {
|
||||
const start = socket.messages.length
|
||||
socket.send({ type: 'user_message', content: prompt })
|
||||
await socket.waitFor((message) => message.type === 'message_complete', timeoutMs, 'message_complete', start)
|
||||
return socket.messages.slice(start)
|
||||
}
|
||||
|
||||
const runners: Record<string, (ctx: ScenarioContext) => Promise<void>> = {
|
||||
async 'session-and-first-turn'(ctx) {
|
||||
const sessionId = await ctx.createSession()
|
||||
const socket = await ctx.openSocket(sessionId)
|
||||
try {
|
||||
// Pin a runtime the way the desktop first-turn selector does. The mock CLI has
|
||||
// no provider, so this proves the control frame is accepted and does not abort
|
||||
// the turn.
|
||||
socket.send({ type: 'set_runtime_config', providerId: null, modelId: 'current' })
|
||||
const turn = await runTurn(socket, 'hello from agent flow')
|
||||
assertOrder(turn, ['content_start', 'content_delta', 'message_complete'])
|
||||
const delta = turn.find((message) => message.type === 'content_delta' && typeof message.text === 'string')
|
||||
if (!delta || !String(delta.text).includes('hello from agent flow')) {
|
||||
throw new Error(`first turn did not stream the prompt back; got ${JSON.stringify(delta)}`)
|
||||
}
|
||||
} finally {
|
||||
socket.close()
|
||||
}
|
||||
},
|
||||
|
||||
async 'tool-permission-allow'(ctx) {
|
||||
const sessionId = await ctx.createSession()
|
||||
const socket = await ctx.openSocket(sessionId)
|
||||
const target = join(ctx.workRoot, 'allowed.txt')
|
||||
try {
|
||||
const prompt = buildMockToolPrompt({
|
||||
tool: 'Write',
|
||||
input: { file_path: target, content: 'allowed' },
|
||||
write: { path: target, content: 'allowed-by-permission' },
|
||||
reply: 'wrote allowed.txt',
|
||||
})
|
||||
const start = socket.messages.length
|
||||
socket.send({ type: 'user_message', content: prompt })
|
||||
|
||||
const request = await socket.waitFor((message) => message.type === 'permission_request', DEFAULT_STEP_TIMEOUT_MS, 'permission_request', start)
|
||||
if (request.toolName !== 'Write') {
|
||||
throw new Error(`permission request carried the wrong tool: ${JSON.stringify(request)}`)
|
||||
}
|
||||
if (existsSync(target)) {
|
||||
throw new Error('tool wrote the file before the permission request was answered')
|
||||
}
|
||||
|
||||
socket.send({ type: 'permission_response', requestId: request.requestId, allowed: true, rule: 'agent-flow' })
|
||||
await socket.waitFor((message) => message.type === 'message_complete', DEFAULT_STEP_TIMEOUT_MS, 'message_complete', start)
|
||||
const turn = socket.messages.slice(start)
|
||||
|
||||
assertOrder(turn, ['content_start', 'permission_request', 'tool_result', 'message_complete'])
|
||||
const result = firstOfType(turn, 'tool_result')
|
||||
if (result?.isError !== false) {
|
||||
throw new Error(`approved tool reported an error: ${JSON.stringify(result)}`)
|
||||
}
|
||||
if (!existsSync(target) || readFileSync(target, 'utf8') !== 'allowed-by-permission') {
|
||||
throw new Error('approved tool did not write the fixture file')
|
||||
}
|
||||
} finally {
|
||||
socket.close()
|
||||
}
|
||||
},
|
||||
|
||||
async 'tool-permission-deny'(ctx) {
|
||||
const sessionId = await ctx.createSession()
|
||||
const socket = await ctx.openSocket(sessionId)
|
||||
const target = join(ctx.workRoot, 'denied.txt')
|
||||
try {
|
||||
const prompt = buildMockToolPrompt({
|
||||
tool: 'Write',
|
||||
input: { file_path: target, content: 'denied' },
|
||||
write: { path: target, content: 'should-never-exist' },
|
||||
})
|
||||
const start = socket.messages.length
|
||||
socket.send({ type: 'user_message', content: prompt })
|
||||
|
||||
const request = await socket.waitFor((message) => message.type === 'permission_request', DEFAULT_STEP_TIMEOUT_MS, 'permission_request', start)
|
||||
socket.send({ type: 'permission_response', requestId: request.requestId, allowed: false })
|
||||
await socket.waitFor((message) => message.type === 'message_complete', DEFAULT_STEP_TIMEOUT_MS, 'message_complete', start)
|
||||
const turn = socket.messages.slice(start)
|
||||
|
||||
const result = firstOfType(turn, 'tool_result')
|
||||
if (!result || result.isError !== true) {
|
||||
throw new Error(`denied tool did not produce an error tool_result: ${JSON.stringify(result)}`)
|
||||
}
|
||||
if (existsSync(target)) {
|
||||
throw new Error('denied tool still wrote the fixture file')
|
||||
}
|
||||
} finally {
|
||||
socket.close()
|
||||
}
|
||||
},
|
||||
|
||||
async 'tool-failure'(ctx) {
|
||||
const sessionId = await ctx.createSession()
|
||||
const socket = await ctx.openSocket(sessionId)
|
||||
try {
|
||||
const prompt = buildMockToolPrompt({
|
||||
tool: 'Bash',
|
||||
input: { command: 'exit 1' },
|
||||
failWith: 'command exited with code 1',
|
||||
})
|
||||
const start = socket.messages.length
|
||||
socket.send({ type: 'user_message', content: prompt })
|
||||
const request = await socket.waitFor((message) => message.type === 'permission_request', DEFAULT_STEP_TIMEOUT_MS, 'permission_request', start)
|
||||
socket.send({ type: 'permission_response', requestId: request.requestId, allowed: true, rule: 'agent-flow' })
|
||||
await socket.waitFor((message) => message.type === 'message_complete', DEFAULT_STEP_TIMEOUT_MS, 'message_complete', start)
|
||||
|
||||
const result = firstOfType(socket.messages.slice(start), 'tool_result')
|
||||
if (!result || result.isError !== true || !String(result.content).includes('exit')) {
|
||||
throw new Error(`failing tool was not surfaced as an error: ${JSON.stringify(result)}`)
|
||||
}
|
||||
} finally {
|
||||
socket.close()
|
||||
}
|
||||
},
|
||||
|
||||
async 'api-error'(ctx) {
|
||||
const sessionId = await ctx.createSession()
|
||||
const socket = await ctx.openSocket(sessionId)
|
||||
try {
|
||||
const start = socket.messages.length
|
||||
socket.send({ type: 'user_message', content: 'please trigger api error now' })
|
||||
await socket.waitFor((message) => message.type === 'message_complete', DEFAULT_STEP_TIMEOUT_MS, 'message_complete', start)
|
||||
const turn = socket.messages.slice(start)
|
||||
const surfaced = turn.some((message) => JSON.stringify(message).includes('Prompt is too long'))
|
||||
if (!surfaced) {
|
||||
throw new Error(`provider API error never reached the client; saw ${turn.map((m) => m.type).join(', ')}`)
|
||||
}
|
||||
|
||||
// The session must survive an API error: the next turn still streams.
|
||||
const recovery = await runTurn(socket, 'still alive')
|
||||
assertOrder(recovery, ['content_delta', 'message_complete'])
|
||||
} finally {
|
||||
socket.close()
|
||||
}
|
||||
},
|
||||
|
||||
async interrupt(ctx) {
|
||||
const sessionId = await ctx.createSession()
|
||||
const socket = await ctx.openSocket(sessionId)
|
||||
try {
|
||||
// Park the turn on an unanswered permission request, then stop generation.
|
||||
const prompt = buildMockToolPrompt({
|
||||
tool: 'Write',
|
||||
input: { file_path: join(ctx.workRoot, 'interrupted.txt'), content: 'x' },
|
||||
})
|
||||
socket.send({ type: 'user_message', content: prompt })
|
||||
await socket.waitFor((message) => message.type === 'permission_request', DEFAULT_STEP_TIMEOUT_MS, 'permission_request')
|
||||
|
||||
socket.send({ type: 'stop_generation' })
|
||||
socket.send({ type: 'sync_state' })
|
||||
const state = await socket.waitFor(
|
||||
(message) => message.type === 'session_state',
|
||||
DEFAULT_STEP_TIMEOUT_MS,
|
||||
'session_state after stop_generation',
|
||||
)
|
||||
if (state.turnState !== 'idle' && state.turnState !== 'running') {
|
||||
throw new Error(`sync_state returned an unknown turn state: ${JSON.stringify(state)}`)
|
||||
}
|
||||
} finally {
|
||||
socket.close()
|
||||
}
|
||||
},
|
||||
|
||||
async 'reconnect-permission-replay'(ctx) {
|
||||
const sessionId = await ctx.createSession()
|
||||
const first = await ctx.openSocket(sessionId)
|
||||
const target = join(ctx.workRoot, 'reconnected.txt')
|
||||
try {
|
||||
const prompt = buildMockToolPrompt({
|
||||
tool: 'Write',
|
||||
input: { file_path: target, content: 'reconnect' },
|
||||
write: { path: target, content: 'written-after-reconnect' },
|
||||
})
|
||||
first.send({ type: 'user_message', content: prompt })
|
||||
const original = await first.waitFor((message) => message.type === 'permission_request', DEFAULT_STEP_TIMEOUT_MS, 'permission_request')
|
||||
first.close()
|
||||
|
||||
// Reconnecting must replay the still-pending request; otherwise a dropped
|
||||
// desktop connection strands the turn behind an approval nobody can see.
|
||||
const second = await ctx.openSocket(sessionId)
|
||||
try {
|
||||
const replayed = await second.waitFor(
|
||||
(message) => message.type === 'permission_request' && message.requestId === original.requestId,
|
||||
DEFAULT_STEP_TIMEOUT_MS,
|
||||
'replayed permission_request',
|
||||
)
|
||||
second.send({ type: 'permission_response', requestId: replayed.requestId, allowed: true, rule: 'agent-flow' })
|
||||
await second.waitFor((message) => message.type === 'message_complete', DEFAULT_STEP_TIMEOUT_MS, 'message_complete')
|
||||
if (!existsSync(target)) {
|
||||
throw new Error('approving the replayed request did not complete the tool')
|
||||
}
|
||||
} finally {
|
||||
second.close()
|
||||
}
|
||||
} finally {
|
||||
first.close()
|
||||
}
|
||||
},
|
||||
|
||||
async 'session-recovery'(ctx) {
|
||||
const sessionId = await ctx.createSession()
|
||||
const socket = await ctx.openSocket(sessionId)
|
||||
try {
|
||||
await runTurn(socket, 'remember this line')
|
||||
} finally {
|
||||
socket.close()
|
||||
}
|
||||
// Let the close settle so the reopen is a genuine reconnect, not a second
|
||||
// concurrent client on a still-open session.
|
||||
await Bun.sleep(300)
|
||||
|
||||
const evidence = sandboxTranscriptEvidence(ctx.sandboxConfigDir)
|
||||
if (evidence.transcriptFiles === 0) {
|
||||
throw new Error(`no transcript was written under the sandbox config dir ${ctx.sandboxConfigDir}`)
|
||||
}
|
||||
|
||||
const listed = await fetch(`${ctx.baseUrl}/api/sessions`)
|
||||
if (!listed.ok) throw new Error(`GET /api/sessions returned HTTP ${listed.status}`)
|
||||
if (!(await listed.text()).includes(sessionId)) {
|
||||
throw new Error('session disappeared from the session list after the client disconnected')
|
||||
}
|
||||
|
||||
const messages = await fetch(`${ctx.baseUrl}/api/sessions/${sessionId}/messages`)
|
||||
if (!messages.ok) throw new Error(`GET /api/sessions/:id/messages returned HTTP ${messages.status}`)
|
||||
|
||||
// Reopening the same session id must produce a live connection again. Transcript
|
||||
// *content* replay is deliberately not asserted here: the turn records are
|
||||
// written by the real Claude CLI, and the mock runtime does not author them.
|
||||
// Content-level recovery is covered by the live provider lane.
|
||||
const reopened = await ctx.openSocket(sessionId)
|
||||
try {
|
||||
reopened.send({ type: 'sync_state' })
|
||||
await reopened.waitFor((message) => message.type === 'session_state', DEFAULT_STEP_TIMEOUT_MS, 'session_state after reconnect')
|
||||
const followUp = await runTurn(reopened, 'still here after reconnect')
|
||||
assertOrder(followUp, ['content_delta', 'message_complete'])
|
||||
} finally {
|
||||
reopened.close()
|
||||
}
|
||||
},
|
||||
}
|
||||
|
||||
export async function executeAgentFlow(options: {
|
||||
rootDir: string
|
||||
artifactDir: string
|
||||
scenarios: AgentFlowScenario[]
|
||||
}): Promise<AgentFlowScenarioResult[]> {
|
||||
const { rootDir, artifactDir, scenarios } = options
|
||||
mkdirSync(artifactDir, { recursive: true })
|
||||
const serverLogPath = join(artifactDir, 'server.log')
|
||||
|
||||
const port = await getPort()
|
||||
const baseUrl = `http://127.0.0.1:${port}`
|
||||
const workRoot = await mkdtemp(join(tmpdir(), 'cc-haha-agent-flow-'))
|
||||
cpSync(join(rootDir, FIXTURE), workRoot, { recursive: true })
|
||||
|
||||
// No provider, no credentials, no network: the runtime is the repository's own
|
||||
// mock SDK CLI, and all user state lives in a throwaway config dir.
|
||||
const sandbox = createQualityGateSandbox({
|
||||
label: 'agent-flow',
|
||||
seedProviders: false,
|
||||
envOverrides: {
|
||||
CLAUDE_CLI_PATH: resolve(rootDir, MOCK_CLI),
|
||||
CC_HAHA_DISABLE_TERMINAL_SHELL_ENV: '1',
|
||||
},
|
||||
})
|
||||
|
||||
const server = Bun.spawn(['bun', 'run', 'src/server/index.ts', '--host', '127.0.0.1', '--port', String(port)], {
|
||||
cwd: rootDir,
|
||||
stdout: 'pipe',
|
||||
stderr: 'pipe',
|
||||
env: { ...sandbox.env, SERVER_PORT: String(port) },
|
||||
})
|
||||
const pumps = [pipeToFile(server.stdout, serverLogPath), pipeToFile(server.stderr, serverLogPath)]
|
||||
|
||||
const results: AgentFlowScenarioResult[] = []
|
||||
try {
|
||||
await waitForHttp(`${baseUrl}/health`, 60_000)
|
||||
|
||||
const ctx: ScenarioContext = {
|
||||
baseUrl,
|
||||
workRoot,
|
||||
sandboxConfigDir: sandbox.configDir,
|
||||
artifactDir,
|
||||
async createSession() {
|
||||
const response = await fetch(`${baseUrl}/api/sessions`, {
|
||||
method: 'POST',
|
||||
headers: { 'Content-Type': 'application/json' },
|
||||
body: JSON.stringify({ workDir: workRoot }),
|
||||
})
|
||||
if (!response.ok) {
|
||||
throw new Error(`POST /api/sessions failed with HTTP ${response.status}: ${await response.text()}`)
|
||||
}
|
||||
const session = await response.json() as { sessionId?: string }
|
||||
if (!session.sessionId) throw new Error('session response did not include sessionId')
|
||||
return session.sessionId
|
||||
},
|
||||
async openSocket(sessionId) {
|
||||
return new SessionSocket(baseUrl, sessionId).open()
|
||||
},
|
||||
}
|
||||
|
||||
for (const scenario of scenarios) {
|
||||
const started = Date.now()
|
||||
const runner = runners[scenario.id]
|
||||
if (!runner) {
|
||||
results.push({
|
||||
id: scenario.id,
|
||||
title: scenario.title,
|
||||
status: 'failed',
|
||||
durationMs: 0,
|
||||
error: `no runner implemented for scenario "${scenario.id}"`,
|
||||
covers: scenario.covers,
|
||||
})
|
||||
continue
|
||||
}
|
||||
|
||||
try {
|
||||
await runner(ctx)
|
||||
results.push({ id: scenario.id, title: scenario.title, status: 'passed', durationMs: Date.now() - started, covers: scenario.covers })
|
||||
} catch (error) {
|
||||
results.push({
|
||||
id: scenario.id,
|
||||
title: scenario.title,
|
||||
status: 'failed',
|
||||
durationMs: Date.now() - started,
|
||||
error: error instanceof Error ? error.message : String(error),
|
||||
covers: scenario.covers,
|
||||
})
|
||||
}
|
||||
}
|
||||
} finally {
|
||||
const mutations = sandbox.detectUserStateMutations()
|
||||
writeFileSync(join(artifactDir, 'agent-flow-results.json'), JSON.stringify({
|
||||
sandboxConfigDir: sandbox.configDir,
|
||||
realConfigMutations: mutations,
|
||||
sandboxTranscripts: sandboxTranscriptEvidence(sandbox.configDir),
|
||||
results,
|
||||
}, null, 2) + '\n')
|
||||
if (mutations.length > 0) {
|
||||
results.push({
|
||||
id: 'user-state-guard',
|
||||
title: 'Agent flow left the developer config untouched',
|
||||
status: 'failed',
|
||||
durationMs: 0,
|
||||
error: `wrote to the developer's real config: ${mutations.join(', ')}`,
|
||||
covers: [],
|
||||
})
|
||||
}
|
||||
|
||||
server.kill()
|
||||
await server.exited.catch(() => undefined)
|
||||
await Promise.all(pumps).catch(() => undefined)
|
||||
rmSync(workRoot, { recursive: true, force: true })
|
||||
sandbox.cleanup()
|
||||
}
|
||||
|
||||
return results
|
||||
}
|
||||
|
||||
export { AGENT_FLOW_SCENARIOS }
|
||||
@@ -0,0 +1,5 @@
|
||||
{
|
||||
"name": "agent-flow-fixture",
|
||||
"private": true,
|
||||
"version": "0.0.0"
|
||||
}
|
||||
@@ -0,0 +1,4 @@
|
||||
# Agent flow fixture
|
||||
|
||||
A throwaway workspace the deterministic agent-flow lane opens as a session project.
|
||||
Scenarios write into this copy, never into the repository.
|
||||
@@ -0,0 +1,39 @@
|
||||
#!/usr/bin/env bun
|
||||
|
||||
import { join } from 'node:path'
|
||||
import { executeAgentFlow } from './execute'
|
||||
import { selectScenarios, uncoveredSteps } from './scenarios'
|
||||
|
||||
function listArg(name: string) {
|
||||
const index = process.argv.indexOf(name)
|
||||
const value = process.argv[index + 1]
|
||||
if (index === -1 || !value || value.startsWith('--')) return []
|
||||
return value.split(',').map((entry) => entry.trim()).filter(Boolean)
|
||||
}
|
||||
|
||||
const rootDir = process.cwd()
|
||||
const artifactDir = process.argv.includes('--artifact-dir')
|
||||
? String(process.argv[process.argv.indexOf('--artifact-dir') + 1])
|
||||
: join(rootDir, 'artifacts', 'agent-flow')
|
||||
|
||||
const scenarios = selectScenarios(listArg('--scenario'))
|
||||
|
||||
console.log('[agent-flow] deterministic agent QA — no provider, no credentials, no network')
|
||||
console.log(`[agent-flow] runtime: mock SDK CLI (src/server/__tests__/fixtures/mock-sdk-cli.ts)`)
|
||||
console.log(`[agent-flow] scenarios: ${scenarios.length}`)
|
||||
|
||||
const results = await executeAgentFlow({ rootDir, artifactDir, scenarios })
|
||||
|
||||
for (const result of results) {
|
||||
const suffix = result.error ? ` — ${result.error}` : ''
|
||||
console.log(`[agent-flow] ${result.status === 'passed' ? 'PASS' : 'FAIL'} ${result.id} (${result.durationMs}ms)${suffix}`)
|
||||
}
|
||||
|
||||
const missing = uncoveredSteps(scenarios)
|
||||
if (missing.length > 0) {
|
||||
console.log(`[agent-flow] user-flow steps not covered by this selection: ${missing.join(', ')}`)
|
||||
}
|
||||
|
||||
const failed = results.filter((result) => result.status === 'failed')
|
||||
console.log(`[agent-flow] summary: passed=${results.length - failed.length} failed=${failed.length} artifacts=${artifactDir}`)
|
||||
process.exit(failed.length === 0 ? 0 : 1)
|
||||
@@ -0,0 +1,117 @@
|
||||
import { describe, expect, test } from 'bun:test'
|
||||
import { readFileSync } from 'node:fs'
|
||||
import {
|
||||
AGENT_FLOW_COVERAGE,
|
||||
AGENT_FLOW_SCENARIOS,
|
||||
buildMockToolPrompt,
|
||||
coveredSteps,
|
||||
findOrderedTypes,
|
||||
firstOfType,
|
||||
selectScenarios,
|
||||
uncoveredSteps,
|
||||
} from './scenarios'
|
||||
|
||||
describe('agent-flow catalog', () => {
|
||||
test('covers every step of the desktop chat user flow', () => {
|
||||
expect(uncoveredSteps(AGENT_FLOW_SCENARIOS)).toEqual([])
|
||||
expect(coveredSteps(AGENT_FLOW_SCENARIOS)).toEqual([...AGENT_FLOW_COVERAGE])
|
||||
})
|
||||
|
||||
test('has a runner for every catalogued scenario', () => {
|
||||
// The runner map lives in execute.ts, which boots a server on import of its
|
||||
// dependencies; read it as text so this stays a unit test.
|
||||
const execute = readFileSync('scripts/quality-gate/agent-flow/execute.ts', 'utf8')
|
||||
for (const scenario of AGENT_FLOW_SCENARIOS) {
|
||||
const declared = new RegExp(`async\\s+'?${scenario.id.replace(/[-]/g, '\\-')}'?\\(ctx\\)`)
|
||||
expect(declared.test(execute), `missing runner for ${scenario.id}`).toBe(true)
|
||||
}
|
||||
})
|
||||
|
||||
test('runs without any provider, credential, or network dependency', () => {
|
||||
const execute = readFileSync('scripts/quality-gate/agent-flow/execute.ts', 'utf8')
|
||||
expect(execute).toContain('src/server/__tests__/fixtures/mock-sdk-cli.ts')
|
||||
expect(execute).toContain('CLAUDE_CLI_PATH')
|
||||
expect(execute).toContain('seedProviders: false')
|
||||
expect(execute).toContain("'127.0.0.1'")
|
||||
expect(execute).toContain('createQualityGateSandbox')
|
||||
expect(execute).toContain('detectUserStateMutations')
|
||||
})
|
||||
|
||||
test('reports uncovered steps when only part of the catalog is selected', () => {
|
||||
const selected = selectScenarios(['tool-permission-allow'])
|
||||
expect(selected).toHaveLength(1)
|
||||
expect(uncoveredSteps(selected)).toContain('reconnect')
|
||||
expect(uncoveredSteps(selected)).not.toContain('permission-allow')
|
||||
})
|
||||
|
||||
test('rejects unknown scenario ids instead of silently running nothing', () => {
|
||||
expect(() => selectScenarios(['nope'])).toThrow(/Unknown agent-flow scenario "nope"/)
|
||||
expect(selectScenarios([])).toEqual(AGENT_FLOW_SCENARIOS)
|
||||
})
|
||||
})
|
||||
|
||||
describe('mock tool prompt', () => {
|
||||
test('round-trips through the marker the mock CLI parses', () => {
|
||||
const prompt = buildMockToolPrompt({
|
||||
tool: 'Write',
|
||||
input: { file_path: '/tmp/x', content: 'hi' },
|
||||
write: { path: '/tmp/x', content: 'hi' },
|
||||
})
|
||||
|
||||
expect(prompt.startsWith('MOCK_TOOL ')).toBe(true)
|
||||
expect(JSON.parse(prompt.slice('MOCK_TOOL '.length))).toEqual({
|
||||
tool: 'Write',
|
||||
input: { file_path: '/tmp/x', content: 'hi' },
|
||||
write: { path: '/tmp/x', content: 'hi' },
|
||||
})
|
||||
})
|
||||
|
||||
test('uses the same marker the fixture checks for', () => {
|
||||
const fixture = readFileSync('src/server/__tests__/fixtures/mock-sdk-cli.ts', 'utf8')
|
||||
expect(fixture).toContain("const MOCK_TOOL_PREFIX = 'MOCK_TOOL '")
|
||||
// The fixture must keep answering permission decisions, or every tool scenario
|
||||
// would hang instead of failing with a useful message.
|
||||
expect(fixture).toContain("subtype: 'can_use_tool'")
|
||||
expect(fixture).toContain("parsed.type === 'control_response'")
|
||||
expect(fixture).toContain("type: 'tool_result'")
|
||||
})
|
||||
|
||||
test('refuses an empty tool name rather than emitting an unusable request', () => {
|
||||
expect(() => buildMockToolPrompt({ tool: ' ', input: {} })).toThrow(/tool name/)
|
||||
})
|
||||
})
|
||||
|
||||
describe('protocol ordering assertions', () => {
|
||||
const turn = [
|
||||
{ type: 'content_start' },
|
||||
{ type: 'content_delta' },
|
||||
{ type: 'permission_request', requestId: 'r1' },
|
||||
{ type: 'tool_result', isError: false },
|
||||
{ type: 'message_complete' },
|
||||
]
|
||||
|
||||
test('accepts the expected order with unrelated frames in between', () => {
|
||||
expect(findOrderedTypes(turn, ['content_start', 'permission_request', 'tool_result', 'message_complete']))
|
||||
.toEqual({ ok: true })
|
||||
})
|
||||
|
||||
test('rejects a reversed order instead of just checking membership', () => {
|
||||
const result = findOrderedTypes(turn, ['tool_result', 'permission_request'])
|
||||
expect(result.ok).toBe(false)
|
||||
if (!result.ok) {
|
||||
expect(result.missing).toBe('permission_request')
|
||||
expect(result.seen).toContain('permission_request')
|
||||
}
|
||||
})
|
||||
|
||||
test('names the first missing type so a failure is actionable', () => {
|
||||
const result = findOrderedTypes(turn, ['content_start', 'thinking'])
|
||||
expect(result.ok).toBe(false)
|
||||
if (!result.ok) expect(result.missing).toBe('thinking')
|
||||
})
|
||||
|
||||
test('finds the first frame of a type', () => {
|
||||
expect(firstOfType(turn, 'permission_request')).toEqual({ type: 'permission_request', requestId: 'r1' })
|
||||
expect(firstOfType(turn, 'nope')).toBeUndefined()
|
||||
})
|
||||
})
|
||||
@@ -0,0 +1,154 @@
|
||||
/**
|
||||
* Deterministic agent-flow scenario catalog.
|
||||
*
|
||||
* `check:chat-contract` runs five unit files with per-layer fakes, so the part of a
|
||||
* turn that actually touches a user's machine — `tool_use → can_use_tool →
|
||||
* permission_request → permission_response → tool_result` — had no coverage that
|
||||
* crosses the real WebSocket. These scenarios drive the real server over the real
|
||||
* protocol with the existing mock SDK CLI, so every contributor can run them with no
|
||||
* provider, no credentials, and no network.
|
||||
*
|
||||
* This module stays pure so the catalog and its assertions are unit-testable; the
|
||||
* runner lives in `execute.ts`.
|
||||
*/
|
||||
|
||||
/** The real user-facing steps a desktop chat session goes through. */
|
||||
export const AGENT_FLOW_COVERAGE = [
|
||||
'session-create',
|
||||
'runtime-select',
|
||||
'first-turn',
|
||||
'tool-execute',
|
||||
'permission-allow',
|
||||
'permission-deny',
|
||||
'tool-error',
|
||||
'api-error',
|
||||
'interrupt',
|
||||
'reconnect',
|
||||
'session-recover',
|
||||
] as const
|
||||
|
||||
export type AgentFlowCoverage = (typeof AGENT_FLOW_COVERAGE)[number]
|
||||
|
||||
export type MockToolStep = {
|
||||
tool: string
|
||||
input: Record<string, unknown>
|
||||
write?: { path: string; content: string }
|
||||
failWith?: string
|
||||
reply?: string
|
||||
}
|
||||
|
||||
export type AgentFlowScenario = {
|
||||
id: string
|
||||
title: string
|
||||
covers: AgentFlowCoverage[]
|
||||
}
|
||||
|
||||
export const AGENT_FLOW_SCENARIOS: AgentFlowScenario[] = [
|
||||
{
|
||||
id: 'session-and-first-turn',
|
||||
title: 'Create a session, pin a runtime, and stream a first turn',
|
||||
covers: ['session-create', 'runtime-select', 'first-turn'],
|
||||
},
|
||||
{
|
||||
id: 'tool-permission-allow',
|
||||
title: 'Approve a tool permission request and observe the edit land',
|
||||
covers: ['tool-execute', 'permission-allow'],
|
||||
},
|
||||
{
|
||||
id: 'tool-permission-deny',
|
||||
title: 'Reject a tool permission request and observe the file stay untouched',
|
||||
covers: ['permission-deny'],
|
||||
},
|
||||
{
|
||||
id: 'tool-failure',
|
||||
title: 'Surface an approved tool that fails as an error tool_result',
|
||||
covers: ['tool-error'],
|
||||
},
|
||||
{
|
||||
id: 'api-error',
|
||||
title: 'Surface a provider API error to the client without killing the session',
|
||||
covers: ['api-error'],
|
||||
},
|
||||
{
|
||||
id: 'interrupt',
|
||||
title: 'Stop generation mid-turn and return the session to idle',
|
||||
covers: ['interrupt'],
|
||||
},
|
||||
{
|
||||
id: 'reconnect-permission-replay',
|
||||
title: 'Reconnect while a permission request is pending and receive the replay',
|
||||
covers: ['reconnect'],
|
||||
},
|
||||
{
|
||||
id: 'session-recovery',
|
||||
title: 'Recover a session after the client disconnects and keep transcripts in the sandbox',
|
||||
covers: ['session-recover'],
|
||||
},
|
||||
]
|
||||
|
||||
const MOCK_TOOL_PREFIX = 'MOCK_TOOL '
|
||||
|
||||
/**
|
||||
* Build the prompt that makes `src/server/__tests__/fixtures/mock-sdk-cli.ts` run a
|
||||
* tool step instead of echoing text.
|
||||
*/
|
||||
export function buildMockToolPrompt(step: MockToolStep): string {
|
||||
if (!step.tool.trim()) {
|
||||
throw new Error('mock tool step requires a tool name')
|
||||
}
|
||||
return `${MOCK_TOOL_PREFIX}${JSON.stringify(step)}`
|
||||
}
|
||||
|
||||
export type ProtocolMessage = { type: string; [key: string]: unknown }
|
||||
|
||||
/**
|
||||
* Assert that `types` appear in `messages` in the given order, allowing unrelated
|
||||
* messages in between. Ordering is the contract that matters: a client that renders
|
||||
* `tool_result` before `permission_request` is broken even if both arrived.
|
||||
*/
|
||||
export function findOrderedTypes(
|
||||
messages: readonly ProtocolMessage[],
|
||||
types: readonly string[],
|
||||
): { ok: true } | { ok: false; missing: string; seen: string[] } {
|
||||
let cursor = 0
|
||||
for (const type of types) {
|
||||
const index = messages.findIndex((message, position) => position >= cursor && message.type === type)
|
||||
if (index === -1) {
|
||||
return { ok: false, missing: type, seen: messages.map((message) => message.type) }
|
||||
}
|
||||
cursor = index + 1
|
||||
}
|
||||
return { ok: true }
|
||||
}
|
||||
|
||||
export function firstOfType<T extends ProtocolMessage = ProtocolMessage>(
|
||||
messages: readonly ProtocolMessage[],
|
||||
type: string,
|
||||
): T | undefined {
|
||||
return messages.find((message) => message.type === type) as T | undefined
|
||||
}
|
||||
|
||||
export function coveredSteps(scenarios: readonly AgentFlowScenario[]): AgentFlowCoverage[] {
|
||||
const covered = new Set<AgentFlowCoverage>()
|
||||
for (const scenario of scenarios) {
|
||||
for (const step of scenario.covers) covered.add(step)
|
||||
}
|
||||
return AGENT_FLOW_COVERAGE.filter((step) => covered.has(step))
|
||||
}
|
||||
|
||||
export function uncoveredSteps(scenarios: readonly AgentFlowScenario[]): AgentFlowCoverage[] {
|
||||
const covered = new Set(coveredSteps(scenarios))
|
||||
return AGENT_FLOW_COVERAGE.filter((step) => !covered.has(step))
|
||||
}
|
||||
|
||||
export function selectScenarios(ids: readonly string[]): AgentFlowScenario[] {
|
||||
if (ids.length === 0) return AGENT_FLOW_SCENARIOS
|
||||
const known = new Map(AGENT_FLOW_SCENARIOS.map((scenario) => [scenario.id, scenario]))
|
||||
return ids.map((id) => {
|
||||
const scenario = known.get(id)
|
||||
if (!scenario) {
|
||||
throw new Error(`Unknown agent-flow scenario "${id}". Known: ${[...known.keys()].join(', ')}`)
|
||||
}
|
||||
return scenario
|
||||
})
|
||||
}
|
||||
@@ -3,6 +3,7 @@ import { mkdtemp } from 'node:fs/promises'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { join } from 'node:path'
|
||||
import { createServer } from 'node:net'
|
||||
import { applyUserStateGuard, createQualityGateSandbox } from '../sandbox'
|
||||
import type { BaselineCase, BaselineTarget, LaneResult } from '../types'
|
||||
|
||||
type ServerMessage = {
|
||||
@@ -247,17 +248,22 @@ export async function executeBaselineCase(
|
||||
const transcriptPath = join(artifactDir, 'transcript.jsonl')
|
||||
const verificationPath = join(artifactDir, 'verification.log')
|
||||
const diffPath = join(artifactDir, 'diff.patch')
|
||||
// Baseline cases boot the real product server, which resolves transcripts,
|
||||
// session index, and settings from CLAUDE_CONFIG_DIR. Without a sandbox the run
|
||||
// writes into the developer's real ~/.claude.
|
||||
const sandbox = createQualityGateSandbox({ label: `baseline-${testCase.id}`, seedProviders: true })
|
||||
const server = Bun.spawn(['bun', 'run', 'src/server/index.ts', '--host', '127.0.0.1', '--port', String(port)], {
|
||||
cwd: rootDir,
|
||||
stdout: 'pipe',
|
||||
stderr: 'pipe',
|
||||
env: {
|
||||
...process.env,
|
||||
...sandbox.env,
|
||||
SERVER_PORT: String(port),
|
||||
},
|
||||
})
|
||||
const stdoutPump = pipeToFile(server.stdout, serverLogPath)
|
||||
const stderrPump = pipeToFile(server.stderr, serverLogPath)
|
||||
const finish = (result: LaneResult) => applyUserStateGuard(result, sandbox, artifactDir)
|
||||
|
||||
try {
|
||||
await waitForHttp(`${baseUrl}/health`, 60_000)
|
||||
@@ -289,7 +295,7 @@ export async function executeBaselineCase(
|
||||
verificationLog += `$ ${command.join(' ')}\n${result.stdout}${result.stderr}\n`
|
||||
if (result.exitCode !== 0) {
|
||||
writeFileSync(verificationPath, verificationLog)
|
||||
return {
|
||||
return finish({
|
||||
id: resultId,
|
||||
title: resultTitle,
|
||||
status: 'failed',
|
||||
@@ -297,31 +303,32 @@ export async function executeBaselineCase(
|
||||
exitCode: result.exitCode,
|
||||
error: `verification command failed: ${command.join(' ')}`,
|
||||
artifactDir,
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
writeFileSync(verificationPath, verificationLog)
|
||||
|
||||
return {
|
||||
return finish({
|
||||
id: resultId,
|
||||
title: resultTitle,
|
||||
status: 'passed',
|
||||
durationMs: Date.now() - started,
|
||||
artifactDir,
|
||||
}
|
||||
})
|
||||
} catch (error) {
|
||||
return {
|
||||
return finish({
|
||||
id: resultId,
|
||||
title: resultTitle,
|
||||
status: 'failed',
|
||||
durationMs: Date.now() - started,
|
||||
error: error instanceof Error ? error.message : String(error),
|
||||
artifactDir,
|
||||
}
|
||||
})
|
||||
} finally {
|
||||
server.kill()
|
||||
await server.exited.catch(() => undefined)
|
||||
await Promise.all([stdoutPump, stderrPump]).catch(() => undefined)
|
||||
rmSync(workRoot, { recursive: true, force: true })
|
||||
sandbox.cleanup()
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,29 @@
|
||||
#!/usr/bin/env bun
|
||||
|
||||
import { join } from 'node:path'
|
||||
import { executeDeterministicDesktopSmoke } from './deterministic'
|
||||
|
||||
const rootDir = process.cwd()
|
||||
const artifactDir = process.argv.includes('--artifact-dir')
|
||||
? String(process.argv[process.argv.indexOf('--artifact-dir') + 1])
|
||||
: join(rootDir, 'artifacts', 'desktop-ui-smoke')
|
||||
|
||||
console.log('[desktop-ui-smoke] real desktop UI against the mock SDK CLI — no provider, no credentials, no network')
|
||||
|
||||
const result = await executeDeterministicDesktopSmoke(
|
||||
rootDir,
|
||||
artifactDir,
|
||||
'desktop-ui-smoke',
|
||||
'Deterministic desktop UI smoke',
|
||||
)
|
||||
|
||||
if (result.status === 'skipped') {
|
||||
console.log(`[desktop-ui-smoke] SKIPPED: ${result.skipReason}`)
|
||||
process.exit(0)
|
||||
}
|
||||
|
||||
console.log(`[desktop-ui-smoke] ${result.status.toUpperCase()} (${result.durationMs}ms) artifacts=${result.artifactDir}`)
|
||||
if (result.error) {
|
||||
console.error(`[desktop-ui-smoke] ${result.error}`)
|
||||
}
|
||||
process.exit(result.status === 'passed' ? 0 : 1)
|
||||
@@ -0,0 +1,74 @@
|
||||
import { describe, expect, test } from 'bun:test'
|
||||
import { mkdtempSync, readFileSync, rmSync } from 'node:fs'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { join } from 'node:path'
|
||||
import {
|
||||
DESKTOP_UI_SMOKE_ALLOW_SELECTOR,
|
||||
DESKTOP_UI_SMOKE_LOCALE,
|
||||
buildDesktopUiSmokeBootstrap,
|
||||
buildDesktopUiSmokePrompt,
|
||||
describeDesktopUiSmokePrerequisites,
|
||||
} from './deterministic'
|
||||
|
||||
describe('deterministic desktop UI smoke setup', () => {
|
||||
test('pins the locale so the approval button label is stable for every contributor', () => {
|
||||
const bootstrap = buildDesktopUiSmokeBootstrap('session-1')
|
||||
|
||||
expect(bootstrap).toContain(`localStorage.setItem('cc-haha-locale', "${DESKTOP_UI_SMOKE_LOCALE}")`)
|
||||
expect(bootstrap).toContain('cc-haha-open-tabs')
|
||||
expect(bootstrap).toContain('session-1')
|
||||
// No runtime is pinned: the lane must exercise the default no-provider path.
|
||||
expect(bootstrap).toContain("localStorage.removeItem('cc-haha-session-runtime')")
|
||||
})
|
||||
|
||||
test('matches the approval button the desktop actually renders', () => {
|
||||
const dialog = readFileSync('desktop/src/components/chat/PermissionDialog.tsx', 'utf8')
|
||||
const english = readFileSync('desktop/src/i18n/locales/en.ts', 'utf8')
|
||||
|
||||
// The selector is derived from the production aria-label and the pinned locale;
|
||||
// if either changes, this test fails before the lane starts timing out.
|
||||
expect(dialog).toContain("aria-label={`${t('permission.allow')}: ${permissionContext}`}")
|
||||
expect(english).toContain("'permission.allow': 'Allow'")
|
||||
expect(DESKTOP_UI_SMOKE_ALLOW_SELECTOR).toBe('button[aria-label^="Allow: "]')
|
||||
})
|
||||
|
||||
test('asks the mock runtime to write inside the fixture copy only', () => {
|
||||
const projectDir = '/tmp/fixture-copy'
|
||||
const { target, prompt } = buildDesktopUiSmokePrompt(projectDir)
|
||||
|
||||
expect(target).toBe(join(projectDir, 'ui-smoke-output.txt'))
|
||||
expect(prompt.startsWith('MOCK_TOOL ')).toBe(true)
|
||||
const payload = JSON.parse(prompt.slice('MOCK_TOOL '.length))
|
||||
expect(payload.tool).toBe('Write')
|
||||
expect(payload.write.path).toBe(target)
|
||||
expect(payload.write.content).toBe('written-through-the-desktop-ui')
|
||||
})
|
||||
|
||||
test('skips with an actionable reason instead of failing when prerequisites are missing', () => {
|
||||
const empty = mkdtempSync(join(tmpdir(), 'cc-haha-ui-smoke-prereq-'))
|
||||
try {
|
||||
expect(describeDesktopUiSmokePrerequisites(empty)).toContain('desktop dependencies')
|
||||
} finally {
|
||||
rmSync(empty, { recursive: true, force: true })
|
||||
}
|
||||
})
|
||||
|
||||
test('runs the real UI against the mock runtime with a sandboxed config dir', () => {
|
||||
const source = readFileSync('scripts/quality-gate/desktop-smoke/deterministic.ts', 'utf8')
|
||||
|
||||
expect(source).toContain('src/server/__tests__/fixtures/mock-sdk-cli.ts')
|
||||
expect(source).toContain('CLAUDE_CLI_PATH')
|
||||
expect(source).toContain('seedProviders: false')
|
||||
expect(source).toContain('...sandbox.env')
|
||||
expect(source).toContain('applyUserStateGuard')
|
||||
// The lane must answer the permission through the UI. Flipping the global
|
||||
// permission mode is what made the old smoke both unsafe and blind to the
|
||||
// approval screen.
|
||||
expect(source).toContain(DESKTOP_UI_SMOKE_ALLOW_SELECTOR)
|
||||
// No global permission-mode switch: the old smoke PUT bypassPermissions on the
|
||||
// settings API, which both skipped the approval screen and mutated user state.
|
||||
expect(source).not.toContain('/api/permissions/mode')
|
||||
expect(source).not.toContain('setPermissionMode')
|
||||
expect(source).toContain('before the permission dialog was answered')
|
||||
})
|
||||
})
|
||||
@@ -0,0 +1,256 @@
|
||||
import { appendFileSync, cpSync, existsSync, mkdirSync, readFileSync, rmSync, writeFileSync } from 'node:fs'
|
||||
import { mkdtemp } from 'node:fs/promises'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { join, resolve } from 'node:path'
|
||||
import { buildMockToolPrompt } from '../agent-flow/scenarios'
|
||||
import { applyUserStateGuard, createQualityGateSandbox } from '../sandbox'
|
||||
import type { LaneResult } from '../types'
|
||||
import {
|
||||
agentBrowserCommand,
|
||||
buildDesktopSmokeBrowserEnv,
|
||||
cleanupAgentBrowserSession,
|
||||
cleanupBrowserProfileProcesses,
|
||||
getPort,
|
||||
pipeToFile,
|
||||
runLoggedCommand,
|
||||
waitForHttp,
|
||||
} from './execute'
|
||||
|
||||
/**
|
||||
* Deterministic desktop UI smoke.
|
||||
*
|
||||
* The existing agent-browser smoke needs a real provider and answers permissions by
|
||||
* flipping the *global* permission mode to bypassPermissions, so the approval UI —
|
||||
* the one screen where a wrong render lets a model touch files the user did not
|
||||
* agree to — was never exercised by any automated check. This lane drives the same
|
||||
* real UI against the repository's mock SDK CLI, so it runs with no provider, no
|
||||
* credentials, and no network, and it answers the permission request by clicking
|
||||
* the real button.
|
||||
*/
|
||||
|
||||
const FIXTURE = 'scripts/quality-gate/agent-flow/fixtures/workspace'
|
||||
const MOCK_CLI = 'src/server/__tests__/fixtures/mock-sdk-cli.ts'
|
||||
const TARGET_FILE = 'ui-smoke-output.txt'
|
||||
const TARGET_CONTENT = 'written-through-the-desktop-ui'
|
||||
|
||||
/** Locale is pinned so the approval button label is stable across contributors. */
|
||||
export const DESKTOP_UI_SMOKE_LOCALE = 'en'
|
||||
export const DESKTOP_UI_SMOKE_ALLOW_SELECTOR = 'button[aria-label^="Allow: "]'
|
||||
|
||||
export function buildDesktopUiSmokeBootstrap(sessionId: string) {
|
||||
return [
|
||||
`localStorage.setItem('cc-haha-locale', ${JSON.stringify(DESKTOP_UI_SMOKE_LOCALE)})`,
|
||||
`localStorage.setItem('cc-haha-open-tabs', ${JSON.stringify(JSON.stringify({
|
||||
openTabs: [{ sessionId, title: 'Desktop UI Smoke', type: 'session' }],
|
||||
activeTabId: sessionId,
|
||||
}))})`,
|
||||
`localStorage.removeItem('cc-haha-session-runtime')`,
|
||||
].join(';')
|
||||
}
|
||||
|
||||
export function buildDesktopUiSmokePrompt(projectDir: string) {
|
||||
const target = join(projectDir, TARGET_FILE)
|
||||
return {
|
||||
target,
|
||||
prompt: buildMockToolPrompt({
|
||||
tool: 'Write',
|
||||
input: { file_path: target, content: TARGET_CONTENT },
|
||||
write: { path: target, content: TARGET_CONTENT },
|
||||
reply: `wrote ${TARGET_FILE}`,
|
||||
}),
|
||||
}
|
||||
}
|
||||
|
||||
async function pollUntil(
|
||||
check: () => Promise<boolean>,
|
||||
timeoutMs: number,
|
||||
label: string,
|
||||
) {
|
||||
const deadline = Date.now() + timeoutMs
|
||||
let lastError = ''
|
||||
while (Date.now() < deadline) {
|
||||
try {
|
||||
if (await check()) return
|
||||
} catch (error) {
|
||||
lastError = error instanceof Error ? error.message : String(error)
|
||||
}
|
||||
await Bun.sleep(500)
|
||||
}
|
||||
throw new Error(`Timed out waiting for ${label}${lastError ? ` (${lastError})` : ''}`)
|
||||
}
|
||||
|
||||
/**
|
||||
* The lane needs the `agent-browser` binary and installed desktop dependencies.
|
||||
* Neither is required to run the rest of the deterministic gate, so a contributor
|
||||
* without them gets an explicit skip instead of a confusing failure.
|
||||
*/
|
||||
export function describeDesktopUiSmokePrerequisites(rootDir: string): string | null {
|
||||
if (!existsSync(join(rootDir, 'desktop', 'node_modules', '.bin'))) {
|
||||
return 'desktop dependencies are not installed (run `bun install` in desktop/)'
|
||||
}
|
||||
const probe = Bun.spawnSync(['agent-browser', '--version'], { stdout: 'pipe', stderr: 'pipe' })
|
||||
if (probe.exitCode !== 0) {
|
||||
return 'agent-browser is not installed (see https://github.com/anthropics/agent-browser)'
|
||||
}
|
||||
return null
|
||||
}
|
||||
|
||||
export async function executeDeterministicDesktopSmoke(
|
||||
rootDir: string,
|
||||
artifactDir: string,
|
||||
resultId: string,
|
||||
resultTitle: string,
|
||||
): Promise<LaneResult> {
|
||||
const started = Date.now()
|
||||
mkdirSync(artifactDir, { recursive: true })
|
||||
|
||||
const missing = describeDesktopUiSmokePrerequisites(rootDir)
|
||||
if (missing) {
|
||||
return {
|
||||
id: resultId,
|
||||
title: resultTitle,
|
||||
status: 'skipped',
|
||||
durationMs: Date.now() - started,
|
||||
skipReason: missing,
|
||||
artifactDir,
|
||||
}
|
||||
}
|
||||
|
||||
const serverLogPath = join(artifactDir, 'server.log')
|
||||
const viteLogPath = join(artifactDir, 'vite.log')
|
||||
const browserLogPath = join(artifactDir, 'browser.log')
|
||||
const workRoot = await mkdtemp(join(tmpdir(), 'quality-gate-desktop-ui-smoke-'))
|
||||
const projectDir = join(workRoot, 'project')
|
||||
const browserProfileDir = join(workRoot, 'browser-profile')
|
||||
cpSync(join(rootDir, FIXTURE), projectDir, { recursive: true })
|
||||
|
||||
const serverPort = await getPort()
|
||||
const vitePort = await getPort()
|
||||
const baseUrl = `http://127.0.0.1:${serverPort}`
|
||||
const appUrl = `http://127.0.0.1:${vitePort}`
|
||||
const sessionName = `quality-gate-ui-${serverPort}-${vitePort}`
|
||||
const browserEnv = buildDesktopSmokeBrowserEnv(sessionName, browserProfileDir)
|
||||
|
||||
const sandbox = createQualityGateSandbox({
|
||||
label: 'desktop-ui-smoke',
|
||||
seedProviders: false,
|
||||
envOverrides: {
|
||||
CLAUDE_CLI_PATH: resolve(rootDir, MOCK_CLI),
|
||||
CC_HAHA_DISABLE_TERMINAL_SHELL_ENV: '1',
|
||||
},
|
||||
})
|
||||
|
||||
const server = Bun.spawn(['bun', 'run', 'src/server/index.ts', '--host', '127.0.0.1', '--port', String(serverPort)], {
|
||||
cwd: rootDir,
|
||||
stdout: 'pipe',
|
||||
stderr: 'pipe',
|
||||
env: { ...sandbox.env, SERVER_PORT: String(serverPort) },
|
||||
})
|
||||
void pipeToFile(server.stdout, serverLogPath)
|
||||
void pipeToFile(server.stderr, serverLogPath)
|
||||
|
||||
const viteExecutable = join(
|
||||
rootDir,
|
||||
'desktop',
|
||||
'node_modules',
|
||||
'.bin',
|
||||
process.platform === 'win32' ? 'vite.cmd' : 'vite',
|
||||
)
|
||||
const vite = Bun.spawn([viteExecutable, '--host', '127.0.0.1', '--port', String(vitePort), '--strictPort'], {
|
||||
cwd: join(rootDir, 'desktop'),
|
||||
env: { ...process.env, VITE_DESKTOP_SERVER_URL: baseUrl },
|
||||
stdout: 'pipe',
|
||||
stderr: 'pipe',
|
||||
})
|
||||
void pipeToFile(vite.stdout, viteLogPath)
|
||||
void pipeToFile(vite.stderr, viteLogPath)
|
||||
|
||||
const browserStep = (args: string[], options: { timeoutMs?: number; allowFailure?: boolean } = {}) =>
|
||||
runLoggedCommand(agentBrowserCommand(args), {
|
||||
cwd: rootDir,
|
||||
env: browserEnv,
|
||||
logPath: browserLogPath,
|
||||
timeoutMs: options.timeoutMs ?? 30_000,
|
||||
allowFailure: options.allowFailure,
|
||||
maxLogChars: 8_000,
|
||||
})
|
||||
|
||||
const { target, prompt } = buildDesktopUiSmokePrompt(projectDir)
|
||||
|
||||
try {
|
||||
await waitForHttp(`${baseUrl}/health`, 30_000)
|
||||
await waitForHttp(appUrl, 60_000)
|
||||
|
||||
const created = await fetch(`${baseUrl}/api/sessions`, {
|
||||
method: 'POST',
|
||||
headers: { 'Content-Type': 'application/json' },
|
||||
body: JSON.stringify({ workDir: projectDir }),
|
||||
})
|
||||
if (!created.ok) {
|
||||
throw new Error(`POST /api/sessions failed with HTTP ${created.status}: ${await created.text()}`)
|
||||
}
|
||||
const session = await created.json() as { sessionId: string }
|
||||
|
||||
await browserStep(['open', appUrl])
|
||||
await browserStep(['eval', buildDesktopUiSmokeBootstrap(session.sessionId)], { timeoutMs: 15_000 })
|
||||
await browserStep(['reload'])
|
||||
await browserStep(['wait', 'textarea'])
|
||||
|
||||
await browserStep(['fill', 'textarea', prompt], { timeoutMs: 20_000 })
|
||||
await browserStep(['press', 'Enter'], { timeoutMs: 15_000 })
|
||||
|
||||
// The permission dialog is the point of this lane: nothing may touch the
|
||||
// fixture until the real button is clicked.
|
||||
await pollUntil(async () => {
|
||||
const probe = await browserStep(['get', 'text', 'body'], { timeoutMs: 15_000, allowFailure: true })
|
||||
return `${probe.stdout}${probe.stderr}`.includes('Allow')
|
||||
}, 60_000, 'the permission dialog to render')
|
||||
|
||||
if (existsSync(target)) {
|
||||
throw new Error('the tool wrote the fixture file before the permission dialog was answered')
|
||||
}
|
||||
await browserStep(['screenshot', join(artifactDir, 'permission-dialog.png')], { allowFailure: true })
|
||||
await browserStep(['click', DESKTOP_UI_SMOKE_ALLOW_SELECTOR], { timeoutMs: 20_000 })
|
||||
|
||||
await pollUntil(
|
||||
async () => existsSync(target) && readFileSync(target, 'utf8') === TARGET_CONTENT,
|
||||
60_000,
|
||||
'the approved tool to write the fixture file',
|
||||
)
|
||||
|
||||
const finalText = await browserStep(['get', 'text', 'body'], { timeoutMs: 15_000, allowFailure: true })
|
||||
const rendered = `${finalText.stdout}${finalText.stderr}`
|
||||
writeFileSync(join(artifactDir, 'final-content.txt'), rendered)
|
||||
if (!rendered.includes(TARGET_FILE)) {
|
||||
throw new Error(`the UI never rendered the tool call for ${TARGET_FILE}`)
|
||||
}
|
||||
await browserStep(['screenshot', join(artifactDir, 'final.png')], { allowFailure: true })
|
||||
|
||||
return applyUserStateGuard({
|
||||
id: resultId,
|
||||
title: resultTitle,
|
||||
status: 'passed' as const,
|
||||
durationMs: Date.now() - started,
|
||||
artifactDir,
|
||||
}, sandbox, artifactDir)
|
||||
} catch (error) {
|
||||
await browserStep(['screenshot', join(artifactDir, 'failure.png')], { allowFailure: true }).catch(() => {})
|
||||
return applyUserStateGuard({
|
||||
id: resultId,
|
||||
title: resultTitle,
|
||||
status: 'failed' as const,
|
||||
durationMs: Date.now() - started,
|
||||
error: error instanceof Error ? error.message : String(error),
|
||||
artifactDir,
|
||||
}, sandbox, artifactDir)
|
||||
} finally {
|
||||
await browserStep(['close'], { timeoutMs: 10_000, allowFailure: true }).catch(() => {})
|
||||
await cleanupAgentBrowserSession(sessionName, browserLogPath)
|
||||
cleanupBrowserProfileProcesses(browserProfileDir, browserLogPath)
|
||||
appendFileSync(browserLogPath, `\n[quality-gate] Removed browser profile ${browserProfileDir}\n`)
|
||||
server.kill()
|
||||
vite.kill()
|
||||
rmSync(workRoot, { recursive: true, force: true })
|
||||
sandbox.cleanup()
|
||||
}
|
||||
}
|
||||
@@ -5,6 +5,7 @@ import { tmpdir } from 'node:os'
|
||||
import { basename, join } from 'node:path'
|
||||
import treeKill from 'tree-kill'
|
||||
import { changedFiles, writeDiffPatch } from '../baseline/execute'
|
||||
import { applyUserStateGuard, createQualityGateSandbox } from '../sandbox'
|
||||
import type { BaselineTarget, LaneResult } from '../types'
|
||||
|
||||
const FIXTURE = 'scripts/quality-gate/desktop-smoke/fixtures/chat-edit'
|
||||
@@ -70,7 +71,7 @@ export function buildDesktopSmokeBrowserEnv(
|
||||
}
|
||||
}
|
||||
|
||||
function agentBrowserCommand(args: string[]) {
|
||||
export function agentBrowserCommand(args: string[]) {
|
||||
return ['agent-browser', '--proxy-bypass', LOOPBACK_PROXY_BYPASS, ...args]
|
||||
}
|
||||
|
||||
@@ -101,7 +102,7 @@ async function killProcessTree(pid: number, signal: 'SIGTERM' | 'SIGKILL') {
|
||||
})
|
||||
}
|
||||
|
||||
async function cleanupAgentBrowserSession(sessionName: string, logPath: string) {
|
||||
export async function cleanupAgentBrowserSession(sessionName: string, logPath: string) {
|
||||
if (!sessionName || !existsSync(AGENT_BROWSER_HOME)) return
|
||||
|
||||
const metadataSuffixes = ['pid', 'sock', 'stream', 'engine', 'version']
|
||||
@@ -127,7 +128,7 @@ async function cleanupAgentBrowserSession(sessionName: string, logPath: string)
|
||||
}
|
||||
}
|
||||
|
||||
function cleanupBrowserProfileProcesses(browserProfileDir: string, logPath: string) {
|
||||
export function cleanupBrowserProfileProcesses(browserProfileDir: string, logPath: string) {
|
||||
if (process.platform === 'win32') {
|
||||
appendFileSync(logPath, '\n[quality-gate] Skipped browser profile process cleanup on Windows\n')
|
||||
return
|
||||
@@ -179,7 +180,7 @@ async function runBrowserStep(
|
||||
}
|
||||
}
|
||||
|
||||
async function getPort(): Promise<number> {
|
||||
export async function getPort(): Promise<number> {
|
||||
return await new Promise((resolve, reject) => {
|
||||
const server = createServer()
|
||||
server.on('error', reject)
|
||||
@@ -195,7 +196,7 @@ async function getPort(): Promise<number> {
|
||||
})
|
||||
}
|
||||
|
||||
async function pipeToFile(stream: ReadableStream<Uint8Array> | null, path: string) {
|
||||
export async function pipeToFile(stream: ReadableStream<Uint8Array> | null, path: string) {
|
||||
if (!stream) return
|
||||
const reader = stream.getReader()
|
||||
while (true) {
|
||||
@@ -205,7 +206,7 @@ async function pipeToFile(stream: ReadableStream<Uint8Array> | null, path: strin
|
||||
}
|
||||
}
|
||||
|
||||
async function waitForHttp(url: string, timeoutMs: number) {
|
||||
export async function waitForHttp(url: string, timeoutMs: number) {
|
||||
const deadline = Date.now() + timeoutMs
|
||||
let lastError = ''
|
||||
while (Date.now() < deadline) {
|
||||
@@ -221,7 +222,7 @@ async function waitForHttp(url: string, timeoutMs: number) {
|
||||
throw new Error(`Timed out waiting for ${url}${lastError ? ` (${lastError})` : ''}`)
|
||||
}
|
||||
|
||||
async function runLoggedCommand(
|
||||
export async function runLoggedCommand(
|
||||
command: string[],
|
||||
options: {
|
||||
cwd: string
|
||||
@@ -437,10 +438,17 @@ export async function executeDesktopSmoke(
|
||||
vitePort,
|
||||
}
|
||||
|
||||
// The smoke drives the real server through the real UI, including a global
|
||||
// permission-mode switch to bypassPermissions. Without a sandbox that switch is
|
||||
// written into the developer's real ~/.claude/settings.json and survives any
|
||||
// interrupted run, so the server gets a throwaway config dir seeded with only the
|
||||
// provider state the run needs.
|
||||
const sandbox = createQualityGateSandbox({ label: 'desktop-smoke', seedProviders: true })
|
||||
const server = Bun.spawn(['bun', 'run', 'src/server/index.ts', '--host', '127.0.0.1', '--port', String(serverPort)], {
|
||||
cwd: rootDir,
|
||||
stdout: 'pipe',
|
||||
stderr: 'pipe',
|
||||
env: { ...sandbox.env, SERVER_PORT: String(serverPort) },
|
||||
})
|
||||
void pipeToFile(server.stdout, serverLogPath)
|
||||
void pipeToFile(server.stderr, serverLogPath)
|
||||
@@ -570,22 +578,22 @@ export async function executeDesktopSmoke(
|
||||
allowFailure: true,
|
||||
}, browserStepContext)
|
||||
|
||||
return {
|
||||
return applyUserStateGuard({
|
||||
id: resultId,
|
||||
title: resultTitle,
|
||||
status: 'passed',
|
||||
status: 'passed' as const,
|
||||
durationMs: Date.now() - started,
|
||||
artifactDir,
|
||||
}
|
||||
}, sandbox, artifactDir)
|
||||
} catch (error) {
|
||||
return {
|
||||
return applyUserStateGuard({
|
||||
id: resultId,
|
||||
title: resultTitle,
|
||||
status: 'failed',
|
||||
status: 'failed' as const,
|
||||
durationMs: Date.now() - started,
|
||||
error: error instanceof Error ? error.message : String(error),
|
||||
artifactDir,
|
||||
}
|
||||
}, sandbox, artifactDir)
|
||||
} finally {
|
||||
if (previousPermissionMode) {
|
||||
await setPermissionMode(baseUrl, previousPermissionMode).catch((error) => {
|
||||
@@ -606,5 +614,6 @@ export async function executeDesktopSmoke(
|
||||
server.kill()
|
||||
vite.kill()
|
||||
rmSync(workRoot, { recursive: true, force: true })
|
||||
sandbox.cleanup()
|
||||
}
|
||||
}
|
||||
|
||||
@@ -86,6 +86,25 @@ export function lanesForMode(mode: QualityGateMode, baselineTargets: BaselineTar
|
||||
requiredForModes: ['pr', 'release'],
|
||||
category: 'integration',
|
||||
},
|
||||
{
|
||||
id: 'agent-flow-checks',
|
||||
title: 'Deterministic agent flow',
|
||||
description: 'Drive the real server and WebSocket through session creation, runtime selection, streaming, tool permission allow/deny, tool failure, API error, interrupt, reconnect replay, and session recovery using the mock SDK CLI. No provider, credentials, or network.',
|
||||
kind: 'command',
|
||||
command: ['bun', 'run', 'check:agent-flow'],
|
||||
impactRequiredCheck: 'bun run check:agent-flow',
|
||||
requiredForModes: ['pr', 'baseline', 'release'],
|
||||
category: 'integration',
|
||||
},
|
||||
{
|
||||
id: 'desktop-ui-smoke',
|
||||
title: 'Deterministic desktop UI smoke',
|
||||
description: 'Drive the real desktop web app with agent-browser against the mock SDK CLI: send a task, wait for the real permission dialog, click Allow, and verify the edit lands. Needs agent-browser and desktop dependencies; skips with a reason when either is missing. No provider or credentials.',
|
||||
kind: 'command',
|
||||
command: ['bun', 'run', 'check:desktop-ui-smoke'],
|
||||
requiredForModes: ['baseline', 'release'],
|
||||
category: 'smoke',
|
||||
},
|
||||
{
|
||||
id: 'adapter-checks',
|
||||
title: 'Adapter checks',
|
||||
|
||||
@@ -2,6 +2,7 @@ import { appendFileSync, mkdirSync, writeFileSync } from 'node:fs'
|
||||
import { join } from 'node:path'
|
||||
import { createServer } from 'node:net'
|
||||
import { ProviderService } from '../../../src/server/services/providerService'
|
||||
import { createQualityGateSandbox } from '../sandbox'
|
||||
import type { BaselineTarget, LaneResult } from '../types'
|
||||
|
||||
type SavedProvider = Awaited<ReturnType<ProviderService['getProvider']>>
|
||||
@@ -57,12 +58,16 @@ async function runProxyProbe(
|
||||
const serverLogPath = join(artifactDir, 'proxy-server.log')
|
||||
ProviderService.setServerPort(port)
|
||||
|
||||
// The proxy probe needs the saved provider, but the spawned server must not write
|
||||
// diagnostics, session index, or settings into the developer's real ~/.claude.
|
||||
// Seed a throwaway config dir with the provider state and nothing else.
|
||||
const sandbox = createQualityGateSandbox({ label: 'provider-smoke', seedProviders: true })
|
||||
const server = Bun.spawn(['bun', 'run', 'src/server/index.ts', '--host', '127.0.0.1', '--port', String(port)], {
|
||||
cwd: rootDir,
|
||||
stdout: 'pipe',
|
||||
stderr: 'pipe',
|
||||
env: {
|
||||
...process.env,
|
||||
...sandbox.env,
|
||||
SERVER_PORT: String(port),
|
||||
},
|
||||
})
|
||||
@@ -114,6 +119,11 @@ async function runProxyProbe(
|
||||
} finally {
|
||||
server.kill()
|
||||
await server.exited.catch(() => undefined)
|
||||
const mutations = sandbox.detectUserStateMutations()
|
||||
sandbox.cleanup()
|
||||
if (mutations.length > 0) {
|
||||
throw new Error(`provider smoke wrote to the developer's real config: ${mutations.join(', ')}`)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -25,6 +25,18 @@ describe('quality gate modes', () => {
|
||||
expect(lanes.some((lane) => lane.startsWith('baseline:'))).toBe(false)
|
||||
})
|
||||
|
||||
test('runs the deterministic agent flow in every mode without requiring a live provider', () => {
|
||||
for (const mode of ['pr', 'baseline', 'release'] as const) {
|
||||
const lane = lanesForMode(mode).find((candidate) => candidate.id === 'agent-flow-checks')
|
||||
expect(lane, `agent-flow-checks missing from ${mode} mode`).toBeDefined()
|
||||
// `live` gates a lane behind --allow-live. The agent flow must never be gated:
|
||||
// it is the only end-to-end proof a contributor without credentials can run.
|
||||
expect(lane?.live).toBeUndefined()
|
||||
expect(lane?.command).toEqual(['bun', 'run', 'check:agent-flow'])
|
||||
expect(lane?.impactRequiredCheck).toBe('bun run check:agent-flow')
|
||||
}
|
||||
})
|
||||
|
||||
test('baseline mode includes live baseline cases but not native checks', () => {
|
||||
const lanes = lanesForMode('baseline').map((lane) => lane.id)
|
||||
expect(lanes).toContain('baseline-catalog')
|
||||
|
||||
@@ -0,0 +1,230 @@
|
||||
import { afterEach, describe, expect, test } from 'bun:test'
|
||||
import { existsSync, mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { join } from 'node:path'
|
||||
import {
|
||||
applyUserStateGuard,
|
||||
buildSandboxLaneEnv,
|
||||
createQualityGateSandbox,
|
||||
describeUserStateMutations,
|
||||
fingerprintUserState,
|
||||
sandboxTranscriptEvidence,
|
||||
seedProviderState,
|
||||
} from './sandbox'
|
||||
|
||||
const scratchDirs: string[] = []
|
||||
|
||||
function scratch(prefix: string) {
|
||||
const dir = mkdtempSync(join(tmpdir(), `cc-haha-sandbox-test-${prefix}-`))
|
||||
scratchDirs.push(dir)
|
||||
return dir
|
||||
}
|
||||
|
||||
function writeJson(path: string, value: unknown) {
|
||||
mkdirSync(join(path, '..'), { recursive: true })
|
||||
writeFileSync(path, JSON.stringify(value) + '\n')
|
||||
}
|
||||
|
||||
afterEach(() => {
|
||||
while (scratchDirs.length > 0) {
|
||||
rmSync(scratchDirs.pop()!, { recursive: true, force: true })
|
||||
}
|
||||
})
|
||||
|
||||
describe('sandbox lane environment', () => {
|
||||
test('redirects config, home, and temp away from the developer machine', () => {
|
||||
const home = scratch('env-home')
|
||||
const env = buildSandboxLaneEnv(home, {}, {
|
||||
HOME: '/Users/real',
|
||||
USERPROFILE: 'C:/Users/real',
|
||||
CLAUDE_CONFIG_DIR: '/Users/real/.claude',
|
||||
PATH: '/usr/bin',
|
||||
})
|
||||
|
||||
expect(env.HOME).toBe(home)
|
||||
expect(env.CLAUDE_CONFIG_DIR).toBe(join(home, '.claude'))
|
||||
expect(env.CLAUDE_CONFIG_DIR.startsWith(home)).toBe(true)
|
||||
expect(env.TMPDIR.startsWith(home)).toBe(true)
|
||||
expect(Object.values(env)).not.toContain('/Users/real/.claude')
|
||||
// The product server must behave like production, not like a unit test.
|
||||
expect(env.NODE_ENV).toBeUndefined()
|
||||
})
|
||||
|
||||
test('passes through proxy and env-provider credentials a live lane needs', () => {
|
||||
const home = scratch('env-live')
|
||||
const env = buildSandboxLaneEnv(home, {}, {
|
||||
PATH: '/usr/bin',
|
||||
HTTPS_PROXY: 'http://127.0.0.1:7890',
|
||||
no_proxy: 'localhost',
|
||||
QUALITY_GATE_PROVIDER_BASE_URL: 'https://gateway.example/v1',
|
||||
QUALITY_GATE_PROVIDER_API_KEY: 'test-key',
|
||||
QUALITY_GATE_PROVIDER_MODEL: 'some-model',
|
||||
ANTHROPIC_API_KEY: 'must-not-leak',
|
||||
AWS_SECRET_ACCESS_KEY: 'must-not-leak',
|
||||
})
|
||||
|
||||
expect(env.HTTPS_PROXY).toBe('http://127.0.0.1:7890')
|
||||
expect(env.no_proxy).toBe('localhost')
|
||||
expect(env.QUALITY_GATE_PROVIDER_BASE_URL).toBe('https://gateway.example/v1')
|
||||
expect(env.QUALITY_GATE_PROVIDER_MODEL).toBe('some-model')
|
||||
// Ambient vendor credentials stay out: a lane must use the provider it was told
|
||||
// to use, not whatever happens to be exported in the developer's shell.
|
||||
expect(env.ANTHROPIC_API_KEY).toBeUndefined()
|
||||
expect(env.AWS_SECRET_ACCESS_KEY).toBeUndefined()
|
||||
})
|
||||
|
||||
test('lets an explicit override win over the sandbox default', () => {
|
||||
const home = scratch('env-override')
|
||||
const env = buildSandboxLaneEnv(home, { NODE_ENV: 'production' }, { PATH: '/usr/bin' })
|
||||
expect(env.NODE_ENV).toBe('production')
|
||||
})
|
||||
})
|
||||
|
||||
describe('provider state seeding', () => {
|
||||
test('copies provider identity and credentials but not regenerable local state', () => {
|
||||
const source = scratch('seed-source')
|
||||
const target = scratch('seed-target')
|
||||
writeJson(join(source, 'cc-haha', 'providers.json'), { activeId: 'p1', providers: [{ id: 'p1', name: 'Gateway' }] })
|
||||
writeJson(join(source, 'cc-haha', 'settings.json'), { env: { ANTHROPIC_BASE_URL: 'https://gateway.example' } })
|
||||
writeJson(join(source, 'cc-haha', 'oauth.json'), { token: 'secret' })
|
||||
mkdirSync(join(source, 'cc-haha', 'db'), { recursive: true })
|
||||
writeFileSync(join(source, 'cc-haha', 'db', 'index-v1.sqlite'), 'binary')
|
||||
mkdirSync(join(source, 'cc-haha', 'diagnostics'), { recursive: true })
|
||||
writeFileSync(join(source, 'cc-haha', 'diagnostics', 'run.log'), 'noise')
|
||||
writeJson(join(source, 'settings.json'), { permissionMode: 'plan' })
|
||||
|
||||
const copied = seedProviderState(source, target)
|
||||
|
||||
expect(copied).toEqual(['providers.json', 'settings.json', 'oauth.json'])
|
||||
expect(JSON.parse(readFileSync(join(target, 'cc-haha', 'providers.json'), 'utf8')).activeId).toBe('p1')
|
||||
expect(existsSync(join(target, 'cc-haha', 'db'))).toBe(false)
|
||||
expect(existsSync(join(target, 'cc-haha', 'diagnostics'))).toBe(false)
|
||||
// The user-level settings.json is never seeded: a lane must not inherit the
|
||||
// developer's permission mode, and must not be able to write it back.
|
||||
expect(existsSync(join(target, 'settings.json'))).toBe(false)
|
||||
})
|
||||
|
||||
test('is a no-op when the developer has no provider state', () => {
|
||||
const source = scratch('seed-empty-source')
|
||||
const target = scratch('seed-empty-target')
|
||||
expect(seedProviderState(source, target)).toEqual([])
|
||||
})
|
||||
})
|
||||
|
||||
describe('user state guard', () => {
|
||||
test('reports created, modified, and deleted guarded files', () => {
|
||||
const config = scratch('guard-diff')
|
||||
writeJson(join(config, 'settings.json'), { permissionMode: 'default' })
|
||||
writeJson(join(config, 'cc-haha', 'providers.json'), { activeId: 'p1' })
|
||||
const before = fingerprintUserState(config)
|
||||
|
||||
writeJson(join(config, 'settings.json'), { permissionMode: 'bypassPermissions', extra: true })
|
||||
writeJson(join(config, 'cc-haha', 'oauth.json'), { token: 'new' })
|
||||
rmSync(join(config, 'cc-haha', 'providers.json'))
|
||||
|
||||
expect(describeUserStateMutations(before, fingerprintUserState(config))).toEqual([
|
||||
'created: cc-haha/oauth.json',
|
||||
'deleted: cc-haha/providers.json',
|
||||
'modified: settings.json',
|
||||
])
|
||||
})
|
||||
|
||||
test('ignores unguarded paths that a concurrent CLI session writes', () => {
|
||||
const config = scratch('guard-noise')
|
||||
writeJson(join(config, 'settings.json'), { permissionMode: 'default' })
|
||||
const before = fingerprintUserState(config)
|
||||
|
||||
mkdirSync(join(config, 'projects', 'some-project'), { recursive: true })
|
||||
writeFileSync(join(config, 'projects', 'some-project', 'session.jsonl'), '{}\n')
|
||||
mkdirSync(join(config, 'file-history'), { recursive: true })
|
||||
writeFileSync(join(config, 'file-history', 'entry.json'), '{}')
|
||||
|
||||
expect(describeUserStateMutations(before, fingerprintUserState(config))).toEqual([])
|
||||
})
|
||||
|
||||
test('fails a passing lane that wrote the global permission mode', () => {
|
||||
// Regression for the desktop smoke, which set bypassPermissions through the
|
||||
// real settings API and left it behind whenever the run was interrupted.
|
||||
const config = scratch('guard-permission')
|
||||
const artifacts = scratch('guard-artifacts')
|
||||
writeJson(join(config, 'settings.json'), { permissionMode: 'default' })
|
||||
const before = fingerprintUserState(config)
|
||||
writeJson(join(config, 'settings.json'), { permissionMode: 'bypassPermissions' })
|
||||
|
||||
const guarded = applyUserStateGuard(
|
||||
{ id: 'desktop-smoke', status: 'passed' as string, durationMs: 1 },
|
||||
{
|
||||
configDir: config,
|
||||
detectUserStateMutations: () => describeUserStateMutations(before, fingerprintUserState(config)),
|
||||
},
|
||||
artifacts,
|
||||
)
|
||||
|
||||
expect(guarded.status).toBe('failed')
|
||||
expect(guarded.error).toContain('settings.json')
|
||||
const evidence = JSON.parse(readFileSync(join(artifacts, 'user-state-guard.json'), 'utf8'))
|
||||
expect(evidence.realConfigMutations).toEqual(['modified: settings.json'])
|
||||
})
|
||||
|
||||
test('keeps a clean lane passing and records transcript isolation evidence', () => {
|
||||
const config = scratch('guard-clean')
|
||||
const sandboxConfig = scratch('guard-clean-sandbox')
|
||||
const artifacts = scratch('guard-clean-artifacts')
|
||||
writeJson(join(config, 'settings.json'), { permissionMode: 'default' })
|
||||
mkdirSync(join(sandboxConfig, 'projects', '-tmp-fixture'), { recursive: true })
|
||||
writeFileSync(join(sandboxConfig, 'projects', '-tmp-fixture', 'abc.jsonl'), '{}\n')
|
||||
|
||||
const guarded = applyUserStateGuard(
|
||||
{ id: 'baseline', status: 'passed' as string, durationMs: 1 },
|
||||
{ configDir: sandboxConfig, detectUserStateMutations: () => [] },
|
||||
artifacts,
|
||||
)
|
||||
|
||||
expect(guarded.status).toBe('passed')
|
||||
const evidence = JSON.parse(readFileSync(join(artifacts, 'user-state-guard.json'), 'utf8'))
|
||||
expect(evidence.sandboxTranscripts).toEqual({ projectDirs: ['-tmp-fixture'], transcriptFiles: 1 })
|
||||
})
|
||||
|
||||
test('counts sandbox transcripts so isolation is proved positively', () => {
|
||||
const sandboxConfig = scratch('transcripts')
|
||||
expect(sandboxTranscriptEvidence(sandboxConfig)).toEqual({ projectDirs: [], transcriptFiles: 0 })
|
||||
|
||||
mkdirSync(join(sandboxConfig, 'projects', 'b-project'), { recursive: true })
|
||||
mkdirSync(join(sandboxConfig, 'projects', 'a-project'), { recursive: true })
|
||||
writeFileSync(join(sandboxConfig, 'projects', 'a-project', 'one.jsonl'), '{}\n')
|
||||
writeFileSync(join(sandboxConfig, 'projects', 'a-project', 'notes.txt'), 'ignored')
|
||||
writeFileSync(join(sandboxConfig, 'projects', 'b-project', 'two.jsonl'), '{}\n')
|
||||
|
||||
expect(sandboxTranscriptEvidence(sandboxConfig)).toEqual({
|
||||
projectDirs: ['a-project', 'b-project'],
|
||||
transcriptFiles: 2,
|
||||
})
|
||||
})
|
||||
})
|
||||
|
||||
describe('sandbox lifecycle', () => {
|
||||
test('creates an isolated config dir seeded from the given source and cleans up', () => {
|
||||
const source = scratch('lifecycle-source')
|
||||
writeJson(join(source, 'cc-haha', 'providers.json'), { activeId: 'p1', providers: [{ id: 'p1', name: 'Gateway' }] })
|
||||
writeJson(join(source, 'settings.json'), { permissionMode: 'plan' })
|
||||
|
||||
const sandbox = createQualityGateSandbox({
|
||||
label: 'lifecycle',
|
||||
seedProviders: true,
|
||||
sourceConfigDir: source,
|
||||
source: { PATH: '/usr/bin' },
|
||||
})
|
||||
|
||||
expect(sandbox.configDir.startsWith(sandbox.home)).toBe(true)
|
||||
expect(existsSync(join(sandbox.configDir, 'cc-haha', 'providers.json'))).toBe(true)
|
||||
expect(existsSync(join(sandbox.configDir, 'settings.json'))).toBe(false)
|
||||
expect(sandbox.detectUserStateMutations()).toEqual([])
|
||||
|
||||
writeJson(join(source, 'settings.json'), { permissionMode: 'bypassPermissions' })
|
||||
expect(sandbox.detectUserStateMutations()).toEqual(['modified: settings.json'])
|
||||
|
||||
const home = sandbox.home
|
||||
sandbox.cleanup()
|
||||
expect(existsSync(home)).toBe(false)
|
||||
})
|
||||
})
|
||||
@@ -0,0 +1,280 @@
|
||||
import { cpSync, existsSync, mkdirSync, mkdtempSync, readdirSync, rmSync, statSync, writeFileSync } from 'node:fs'
|
||||
import { homedir, tmpdir } from 'node:os'
|
||||
import { join } from 'node:path'
|
||||
import { createSandboxedTestEnvironment } from '../pr/test-environment'
|
||||
|
||||
/**
|
||||
* Quality-gate sandboxing for lanes that boot the real product server.
|
||||
*
|
||||
* Baseline cases, desktop smoke, and provider smoke all spawn `src/server/index.ts`.
|
||||
* Every one of those code paths resolves user state through
|
||||
* `process.env.CLAUDE_CONFIG_DIR || homedir()/.claude`, so inheriting the parent
|
||||
* environment makes a QA run read *and write* the developer's real transcripts,
|
||||
* session index, diagnostics, and `settings.json`. This module gives those lanes a
|
||||
* throwaway HOME while still letting them exercise a provider the user actually
|
||||
* configured: provider state is *copied in*, never written back.
|
||||
*/
|
||||
|
||||
/** Files under `<config>/cc-haha/` that carry provider identity and credentials. */
|
||||
export const SEEDABLE_PROVIDER_STATE_FILES = [
|
||||
'providers.json',
|
||||
'settings.json',
|
||||
'oauth.json',
|
||||
'openai-oauth.json',
|
||||
'grok-oauth.json',
|
||||
] as const
|
||||
|
||||
/**
|
||||
* Environment names a live lane needs even though `createSandboxedTestEnvironment`
|
||||
* strips everything outside its safe list: proxy reachability for users behind a
|
||||
* corporate gateway, and the env-only provider target used by provider smoke.
|
||||
*/
|
||||
export const LIVE_PASSTHROUGH_ENV_NAMES = [
|
||||
'ALL_PROXY',
|
||||
'all_proxy',
|
||||
'HTTP_PROXY',
|
||||
'http_proxy',
|
||||
'HTTPS_PROXY',
|
||||
'https_proxy',
|
||||
'NO_PROXY',
|
||||
'no_proxy',
|
||||
'QUALITY_GATE_PROVIDER_API_FORMAT',
|
||||
'QUALITY_GATE_PROVIDER_API_KEY',
|
||||
'QUALITY_GATE_PROVIDER_AUTH_STRATEGY',
|
||||
'QUALITY_GATE_PROVIDER_BASE_URL',
|
||||
'QUALITY_GATE_PROVIDER_MODEL',
|
||||
'CC_HAHA_SYSTEM_PROXY_URL',
|
||||
] as const
|
||||
|
||||
export type UserStateFingerprint = Record<string, string>
|
||||
|
||||
export type QualityGateSandbox = {
|
||||
home: string
|
||||
configDir: string
|
||||
env: Record<string, string>
|
||||
/** Fingerprint of the real config dir taken before the lane ran. */
|
||||
guardedBefore: UserStateFingerprint
|
||||
/** Re-reads the real config dir and reports anything the lane mutated. */
|
||||
detectUserStateMutations(): string[]
|
||||
cleanup(): void
|
||||
}
|
||||
|
||||
export function realUserConfigDir(source: NodeJS.ProcessEnv = process.env) {
|
||||
return source.CLAUDE_CONFIG_DIR || join(homedir(), '.claude')
|
||||
}
|
||||
|
||||
/**
|
||||
* Build the environment for a sandboxed live lane.
|
||||
*
|
||||
* `createSandboxedTestEnvironment` already redirects HOME, XDG, TEMP, and
|
||||
* CLAUDE_CONFIG_DIR; this layer re-adds the handful of names a live provider call
|
||||
* genuinely needs and drops `NODE_ENV=test` so the server behaves like production.
|
||||
*/
|
||||
export function buildSandboxLaneEnv(
|
||||
sandboxHome: string,
|
||||
overrides: Record<string, string> = {},
|
||||
source: NodeJS.ProcessEnv = process.env,
|
||||
): Record<string, string> {
|
||||
const passthrough: Record<string, string> = {}
|
||||
for (const name of LIVE_PASSTHROUGH_ENV_NAMES) {
|
||||
const value = source[name]
|
||||
if (typeof value === 'string' && value.length > 0) {
|
||||
passthrough[name] = value
|
||||
}
|
||||
}
|
||||
|
||||
const env = createSandboxedTestEnvironment(sandboxHome, { ...passthrough, ...overrides }, source)
|
||||
if (!('NODE_ENV' in overrides)) {
|
||||
delete env.NODE_ENV
|
||||
}
|
||||
return env
|
||||
}
|
||||
|
||||
/**
|
||||
* Copy provider identity and credentials from the developer's real config into a
|
||||
* sandbox config dir. Regenerable state (`db/`, `diagnostics/`) is deliberately left
|
||||
* behind so a QA run never inherits a stale local index.
|
||||
*/
|
||||
export function seedProviderState(
|
||||
sourceConfigDir: string,
|
||||
sandboxConfigDir: string,
|
||||
files: readonly string[] = SEEDABLE_PROVIDER_STATE_FILES,
|
||||
): string[] {
|
||||
const sourceDir = join(sourceConfigDir, 'cc-haha')
|
||||
const targetDir = join(sandboxConfigDir, 'cc-haha')
|
||||
if (!existsSync(sourceDir)) {
|
||||
return []
|
||||
}
|
||||
|
||||
mkdirSync(targetDir, { recursive: true })
|
||||
const copied: string[] = []
|
||||
for (const file of files) {
|
||||
const sourcePath = join(sourceDir, file)
|
||||
if (!existsSync(sourcePath)) continue
|
||||
cpSync(sourcePath, join(targetDir, file))
|
||||
copied.push(file)
|
||||
}
|
||||
return copied
|
||||
}
|
||||
|
||||
/**
|
||||
* Config-relative files a quality-gate lane must never write.
|
||||
*
|
||||
* Deliberately narrow. A full walk of `~/.claude` would also pick up writes from
|
||||
* the developer's own concurrently running Claude Code session (`projects/`,
|
||||
* `file-history/`, `plugins/`), turning the guard into noise. These paths are only
|
||||
* written by an explicit settings or provider mutation, which is exactly the class
|
||||
* of leak this guard exists to catch. Transcript isolation is proved positively
|
||||
* instead — see `sandboxTranscriptEvidence`.
|
||||
*/
|
||||
export const GUARDED_USER_STATE_PATHS = [
|
||||
'settings.json',
|
||||
'cc-haha/providers.json',
|
||||
'cc-haha/settings.json',
|
||||
'cc-haha/oauth.json',
|
||||
'cc-haha/openai-oauth.json',
|
||||
'cc-haha/grok-oauth.json',
|
||||
] as const
|
||||
|
||||
/**
|
||||
* Snapshot the guarded slice of the developer's real config directory so a lane can
|
||||
* prove it did not write to it. Size + mtime is enough: every write through the
|
||||
* product code path rewrites the whole file.
|
||||
*/
|
||||
export function fingerprintUserState(
|
||||
configDir: string,
|
||||
guardedPaths: readonly string[] = GUARDED_USER_STATE_PATHS,
|
||||
): UserStateFingerprint {
|
||||
const fingerprint: UserStateFingerprint = {}
|
||||
for (const guardedPath of guardedPaths) {
|
||||
const fullPath = join(configDir, ...guardedPath.split('/'))
|
||||
try {
|
||||
const stat = statSync(fullPath)
|
||||
if (!stat.isFile()) continue
|
||||
fingerprint[guardedPath] = `${stat.size}:${stat.mtimeMs}`
|
||||
} catch {
|
||||
// Absent files stay absent from the fingerprint; creation is reported as a mutation.
|
||||
}
|
||||
}
|
||||
return fingerprint
|
||||
}
|
||||
|
||||
/**
|
||||
* Positive evidence that session transcripts landed inside the sandbox rather than
|
||||
* the developer's real config directory.
|
||||
*/
|
||||
export function sandboxTranscriptEvidence(sandboxConfigDir: string): {
|
||||
projectDirs: string[]
|
||||
transcriptFiles: number
|
||||
} {
|
||||
const projectsRoot = join(sandboxConfigDir, 'projects')
|
||||
if (!existsSync(projectsRoot)) {
|
||||
return { projectDirs: [], transcriptFiles: 0 }
|
||||
}
|
||||
|
||||
const projectDirs: string[] = []
|
||||
let transcriptFiles = 0
|
||||
for (const entry of readdirSync(projectsRoot)) {
|
||||
const projectDir = join(projectsRoot, entry)
|
||||
try {
|
||||
if (!statSync(projectDir).isDirectory()) continue
|
||||
} catch {
|
||||
continue
|
||||
}
|
||||
projectDirs.push(entry)
|
||||
for (const file of readdirSync(projectDir)) {
|
||||
if (file.endsWith('.jsonl')) transcriptFiles += 1
|
||||
}
|
||||
}
|
||||
return { projectDirs: projectDirs.sort(), transcriptFiles }
|
||||
}
|
||||
|
||||
export function describeUserStateMutations(
|
||||
before: UserStateFingerprint,
|
||||
after: UserStateFingerprint,
|
||||
): string[] {
|
||||
const mutations: string[] = []
|
||||
for (const [path, signature] of Object.entries(before)) {
|
||||
if (!(path in after)) {
|
||||
mutations.push(`deleted: ${path}`)
|
||||
continue
|
||||
}
|
||||
if (after[path] !== signature) {
|
||||
mutations.push(`modified: ${path}`)
|
||||
}
|
||||
}
|
||||
for (const path of Object.keys(after)) {
|
||||
if (!(path in before)) {
|
||||
mutations.push(`created: ${path}`)
|
||||
}
|
||||
}
|
||||
return mutations.sort()
|
||||
}
|
||||
|
||||
/**
|
||||
* Attach sandbox evidence to a lane result and fail the lane if it wrote to the
|
||||
* developer's real config. A leak is a gate failure, not a warning: the whole point
|
||||
* of the sandbox is that a QA run leaves user state untouched.
|
||||
*/
|
||||
export function applyUserStateGuard<T extends { status: string; error?: string }>(
|
||||
result: T,
|
||||
sandbox: Pick<QualityGateSandbox, 'configDir' | 'detectUserStateMutations'>,
|
||||
artifactDir: string,
|
||||
): T {
|
||||
const mutations = sandbox.detectUserStateMutations()
|
||||
const transcripts = sandboxTranscriptEvidence(sandbox.configDir)
|
||||
writeFileSync(
|
||||
join(artifactDir, 'user-state-guard.json'),
|
||||
JSON.stringify({
|
||||
sandboxConfigDir: sandbox.configDir,
|
||||
guardedPaths: GUARDED_USER_STATE_PATHS,
|
||||
realConfigMutations: mutations,
|
||||
sandboxTranscripts: transcripts,
|
||||
}, null, 2) + '\n',
|
||||
)
|
||||
|
||||
if (mutations.length === 0) {
|
||||
return result
|
||||
}
|
||||
|
||||
const message = `quality gate wrote to the developer's real config: ${mutations.join(', ')}`
|
||||
return {
|
||||
...result,
|
||||
status: 'failed',
|
||||
error: result.error ? `${result.error}; ${message}` : message,
|
||||
}
|
||||
}
|
||||
|
||||
export function createQualityGateSandbox(options: {
|
||||
label: string
|
||||
seedProviders?: boolean
|
||||
sourceConfigDir?: string
|
||||
envOverrides?: Record<string, string>
|
||||
source?: NodeJS.ProcessEnv
|
||||
}): QualityGateSandbox {
|
||||
const source = options.source ?? process.env
|
||||
const sourceConfigDir = options.sourceConfigDir ?? realUserConfigDir(source)
|
||||
const home = mkdtempSync(join(tmpdir(), `cc-haha-qa-${options.label}-`))
|
||||
const env = buildSandboxLaneEnv(home, options.envOverrides ?? {}, source)
|
||||
const configDir = env.CLAUDE_CONFIG_DIR
|
||||
mkdirSync(configDir, { recursive: true })
|
||||
|
||||
if (options.seedProviders) {
|
||||
seedProviderState(sourceConfigDir, configDir)
|
||||
}
|
||||
|
||||
const guardedBefore = fingerprintUserState(sourceConfigDir)
|
||||
|
||||
return {
|
||||
home,
|
||||
configDir,
|
||||
env,
|
||||
guardedBefore,
|
||||
detectUserStateMutations() {
|
||||
return describeUserStateMutations(guardedBefore, fingerprintUserState(sourceConfigDir))
|
||||
},
|
||||
cleanup() {
|
||||
rmSync(home, { recursive: true, force: true })
|
||||
},
|
||||
}
|
||||
}
|
||||
@@ -1,4 +1,5 @@
|
||||
import { appendFile, readFile } from 'node:fs/promises'
|
||||
import { mkdir, appendFile, readFile, writeFile } from 'node:fs/promises'
|
||||
import { dirname } from 'node:path'
|
||||
|
||||
const args = process.argv.slice(2)
|
||||
|
||||
@@ -38,6 +39,111 @@ const resumeUpstreamUrl = process.env.MOCK_SDK_RESUME_UPSTREAM_URL
|
||||
let initSent = false
|
||||
let firstUserExitScheduled = false
|
||||
|
||||
/**
|
||||
* Deterministic tool-use support.
|
||||
*
|
||||
* The mock previously only streamed text, so `tool_use → can_use_tool →
|
||||
* permission_request → permission_response → tool_result` — the part of an agent turn
|
||||
* that actually touches the user's files — had no end-to-end coverage that runs
|
||||
* without a model. A turn opts in by sending `MOCK_TOOL <json>`; every existing
|
||||
* prompt keeps its old behavior.
|
||||
*/
|
||||
// Not exported: this file boots a WebSocket at import time and exits when
|
||||
// --sdk-url is missing, so tests must never import it. The matching type and
|
||||
// prompt builder live in scripts/quality-gate/agent-flow/scenarios.ts.
|
||||
type MockToolStep = {
|
||||
tool: string
|
||||
input: Record<string, unknown>
|
||||
/** Written only after the permission request is allowed, like a real tool. */
|
||||
write?: { path: string; content: string }
|
||||
/** Emit an is_error tool_result even when the request is allowed. */
|
||||
failWith?: string
|
||||
reply?: string
|
||||
}
|
||||
|
||||
const MOCK_TOOL_PREFIX = 'MOCK_TOOL '
|
||||
const pendingPermissions = new Map<string, (decision: { allowed: boolean; message?: string }) => void>()
|
||||
|
||||
function parseMockToolStep(text: string): MockToolStep | null {
|
||||
const trimmed = text.trim()
|
||||
if (!trimmed.startsWith(MOCK_TOOL_PREFIX)) return null
|
||||
const parsed = JSON.parse(trimmed.slice(MOCK_TOOL_PREFIX.length)) as MockToolStep
|
||||
if (!parsed || typeof parsed.tool !== 'string') {
|
||||
throw new Error(`MOCK_TOOL payload must include a tool name: ${trimmed}`)
|
||||
}
|
||||
return { ...parsed, input: parsed.input ?? {} }
|
||||
}
|
||||
|
||||
async function runToolStep(ws: WebSocket, step: MockToolStep) {
|
||||
const toolUseId = `toolu_${crypto.randomUUID().replace(/-/g, '').slice(0, 20)}`
|
||||
const requestId = `req_${crypto.randomUUID().replace(/-/g, '').slice(0, 12)}`
|
||||
const inputJson = JSON.stringify(step.input)
|
||||
|
||||
emit(ws, { type: 'stream_event', event: { type: 'message_start' }, session_id: sessionId })
|
||||
emit(ws, {
|
||||
type: 'stream_event',
|
||||
event: {
|
||||
type: 'content_block_start',
|
||||
index: 0,
|
||||
content_block: { type: 'tool_use', id: toolUseId, name: step.tool, input: {} },
|
||||
},
|
||||
session_id: sessionId,
|
||||
})
|
||||
emit(ws, {
|
||||
type: 'stream_event',
|
||||
event: {
|
||||
type: 'content_block_delta',
|
||||
index: 0,
|
||||
delta: { type: 'input_json_delta', partial_json: inputJson },
|
||||
},
|
||||
session_id: sessionId,
|
||||
})
|
||||
emit(ws, { type: 'stream_event', event: { type: 'content_block_stop', index: 0 }, session_id: sessionId })
|
||||
|
||||
const decision = await new Promise<{ allowed: boolean; message?: string }>((resolve) => {
|
||||
pendingPermissions.set(requestId, resolve)
|
||||
emit(ws, {
|
||||
type: 'control_request',
|
||||
request_id: requestId,
|
||||
request: {
|
||||
subtype: 'can_use_tool',
|
||||
tool_name: step.tool,
|
||||
tool_use_id: toolUseId,
|
||||
input: step.input,
|
||||
description: `mock ${step.tool}`,
|
||||
},
|
||||
session_id: sessionId,
|
||||
})
|
||||
})
|
||||
|
||||
let content = decision.allowed ? (step.reply ?? `${step.tool} ok`) : (decision.message ?? 'denied')
|
||||
let isError = !decision.allowed
|
||||
if (decision.allowed && step.failWith) {
|
||||
content = step.failWith
|
||||
isError = true
|
||||
} else if (decision.allowed && step.write) {
|
||||
await mkdir(dirname(step.write.path), { recursive: true })
|
||||
await writeFile(step.write.path, step.write.content)
|
||||
}
|
||||
|
||||
emit(ws, {
|
||||
type: 'user',
|
||||
message: {
|
||||
role: 'user',
|
||||
content: [{ type: 'tool_result', tool_use_id: toolUseId, content, is_error: isError }],
|
||||
},
|
||||
session_id: sessionId,
|
||||
})
|
||||
emit(ws, {
|
||||
type: 'result',
|
||||
subtype: 'success',
|
||||
is_error: false,
|
||||
result: isError ? `tool failed: ${content}` : `tool done: ${content}`,
|
||||
usage: { input_tokens: 5, output_tokens: 3 },
|
||||
session_id: sessionId,
|
||||
})
|
||||
}
|
||||
|
||||
function transcriptText(entry: any): string {
|
||||
const content = entry?.message?.content
|
||||
if (typeof content === 'string') return content
|
||||
@@ -182,6 +288,11 @@ ws.addEventListener('message', (event) => {
|
||||
await runResumeTurn(ws, text)
|
||||
continue
|
||||
}
|
||||
const toolStep = parseMockToolStep(text)
|
||||
if (toolStep) {
|
||||
await runToolStep(ws, toolStep)
|
||||
continue
|
||||
}
|
||||
const slashCommand = text.trim()
|
||||
if (slashCommand === '/cost') {
|
||||
emit(ws, {
|
||||
@@ -291,6 +402,21 @@ ws.addEventListener('message', (event) => {
|
||||
})
|
||||
}
|
||||
|
||||
if (parsed.type === 'control_response' && typeof parsed.response?.request_id === 'string') {
|
||||
const resolve = pendingPermissions.get(parsed.response.request_id)
|
||||
if (resolve) {
|
||||
pendingPermissions.delete(parsed.response.request_id)
|
||||
const behavior = parsed.response?.response?.behavior
|
||||
resolve({
|
||||
allowed: behavior === 'allow',
|
||||
message: typeof parsed.response?.response?.message === 'string'
|
||||
? parsed.response.response.message
|
||||
: undefined,
|
||||
})
|
||||
continue
|
||||
}
|
||||
}
|
||||
|
||||
if (parsed.type === 'control_request' && parsed.request?.subtype === 'interrupt') {
|
||||
emit(ws, {
|
||||
type: 'result',
|
||||
|
||||
Reference in New Issue
Block a user