diff --git a/bun.lock b/bun.lock index 7e4ee14a..92876a58 100644 --- a/bun.lock +++ b/bun.lock @@ -55,6 +55,7 @@ "react": "^19.2.4", "react-reconciler": "^0.33.0", "semver": "^7.7.4", + "sharp": "0.34.5", "shell-quote": "^1.8.3", "signal-exit": "^4.1.0", "stack-utils": "^2.0.6", @@ -172,6 +173,8 @@ "@commander-js/extra-typings": ["@commander-js/extra-typings@14.0.0", "https://registry.npmmirror.com/@commander-js/extra-typings/-/extra-typings-14.0.0.tgz", { "peerDependencies": { "commander": "~14.0.0" } }, "sha512-hIn0ncNaJRLkZrxBIp5AsW/eXEHNKYQBh0aPdoUqNgD+Io3NIykQqpKFyKcuasZhicGaEZJX/JBSIkZ4e5x8Dg=="], + "@emnapi/runtime": ["@emnapi/runtime@1.11.3", "https://registry.npmmirror.com/@emnapi/runtime/-/runtime-1.11.3.tgz", { "dependencies": { "tslib": "^2.4.0" } }, "sha512-Xz4Tpyki7XyrpbUK1jR1AhdAdaXyhhY4lZ3neLodmhpuWfy2PAQN5B46sAiU4liOXGLkHypn/qU+jvfWSCYYLA=="], + "@growthbook/growthbook": ["@growthbook/growthbook@1.6.5", "https://registry.npmmirror.com/@growthbook/growthbook/-/growthbook-1.6.5.tgz", { "dependencies": { "dom-mutator": "^0.6.0" } }, "sha512-mUaMsgeUTpRIUOTn33EUXHRK6j7pxBjwqH4WpQyq+pukjd1AIzWlEa6w7i6bInJUcweGgP2beXZmaP6b6UPn7A=="], "@hono/node-server": ["@hono/node-server@1.19.12", "https://registry.npmmirror.com/@hono/node-server/-/node-server-1.19.12.tgz", { "peerDependencies": { "hono": "^4" } }, "sha512-txsUW4SQ1iilgE0l9/e9VQWmELXifEFvmdA1j6WFh/aFPj99hIntrSsq/if0UWyGVkmrRPKA1wCeP+UCr1B9Uw=="], @@ -180,6 +183,56 @@ "@iconify/utils": ["@iconify/utils@3.1.0", "https://registry.npmmirror.com/@iconify/utils/-/utils-3.1.0.tgz", { "dependencies": { "@antfu/install-pkg": "^1.1.0", "@iconify/types": "^2.0.0", "mlly": "^1.8.0" } }, "sha512-Zlzem1ZXhI1iHeeERabLNzBHdOa4VhQbqAcOQaMKuTuyZCpwKbC2R4Dd0Zo3g9EAc+Y4fiarO8HIHRAth7+skw=="], + "@img/colour": ["@img/colour@1.1.0", "https://registry.npmmirror.com/@img/colour/-/colour-1.1.0.tgz", {}, "sha512-Td76q7j57o/tLVdgS746cYARfSyxk8iEfRxewL9h4OMzYhbW4TAcppl0mT4eyqXddh6L/jwoM75mo7ixa/pCeQ=="], + + "@img/sharp-darwin-arm64": ["@img/sharp-darwin-arm64@0.34.5", "https://registry.npmmirror.com/@img/sharp-darwin-arm64/-/sharp-darwin-arm64-0.34.5.tgz", { "optionalDependencies": { "@img/sharp-libvips-darwin-arm64": "1.2.4" }, "os": "darwin", "cpu": "arm64" }, "sha512-imtQ3WMJXbMY4fxb/Ndp6HBTNVtWCUI0WdobyheGf5+ad6xX8VIDO8u2xE4qc/fr08CKG/7dDseFtn6M6g/r3w=="], + + "@img/sharp-darwin-x64": ["@img/sharp-darwin-x64@0.34.5", "https://registry.npmmirror.com/@img/sharp-darwin-x64/-/sharp-darwin-x64-0.34.5.tgz", { "optionalDependencies": { "@img/sharp-libvips-darwin-x64": "1.2.4" }, "os": "darwin", "cpu": "x64" }, "sha512-YNEFAF/4KQ/PeW0N+r+aVVsoIY0/qxxikF2SWdp+NRkmMB7y9LBZAVqQ4yhGCm/H3H270OSykqmQMKLBhBJDEw=="], + + "@img/sharp-libvips-darwin-arm64": ["@img/sharp-libvips-darwin-arm64@1.2.4", "https://registry.npmmirror.com/@img/sharp-libvips-darwin-arm64/-/sharp-libvips-darwin-arm64-1.2.4.tgz", { "os": "darwin", "cpu": "arm64" }, "sha512-zqjjo7RatFfFoP0MkQ51jfuFZBnVE2pRiaydKJ1G/rHZvnsrHAOcQALIi9sA5co5xenQdTugCvtb1cuf78Vf4g=="], + + "@img/sharp-libvips-darwin-x64": ["@img/sharp-libvips-darwin-x64@1.2.4", "https://registry.npmmirror.com/@img/sharp-libvips-darwin-x64/-/sharp-libvips-darwin-x64-1.2.4.tgz", { "os": "darwin", "cpu": "x64" }, "sha512-1IOd5xfVhlGwX+zXv2N93k0yMONvUlANylbJw1eTah8K/Jtpi15KC+WSiaX/nBmbm2HxRM1gZ0nSdjSsrZbGKg=="], + + "@img/sharp-libvips-linux-arm": ["@img/sharp-libvips-linux-arm@1.2.4", "https://registry.npmmirror.com/@img/sharp-libvips-linux-arm/-/sharp-libvips-linux-arm-1.2.4.tgz", { "os": "linux", "cpu": "arm" }, "sha512-bFI7xcKFELdiNCVov8e44Ia4u2byA+l3XtsAj+Q8tfCwO6BQ8iDojYdvoPMqsKDkuoOo+X6HZA0s0q11ANMQ8A=="], + + "@img/sharp-libvips-linux-arm64": ["@img/sharp-libvips-linux-arm64@1.2.4", "https://registry.npmmirror.com/@img/sharp-libvips-linux-arm64/-/sharp-libvips-linux-arm64-1.2.4.tgz", { "os": "linux", "cpu": "arm64" }, "sha512-excjX8DfsIcJ10x1Kzr4RcWe1edC9PquDRRPx3YVCvQv+U5p7Yin2s32ftzikXojb1PIFc/9Mt28/y+iRklkrw=="], + + "@img/sharp-libvips-linux-ppc64": ["@img/sharp-libvips-linux-ppc64@1.2.4", "https://registry.npmmirror.com/@img/sharp-libvips-linux-ppc64/-/sharp-libvips-linux-ppc64-1.2.4.tgz", { "os": "linux", "cpu": "ppc64" }, "sha512-FMuvGijLDYG6lW+b/UvyilUWu5Ayu+3r2d1S8notiGCIyYU/76eig1UfMmkZ7vwgOrzKzlQbFSuQfgm7GYUPpA=="], + + "@img/sharp-libvips-linux-riscv64": ["@img/sharp-libvips-linux-riscv64@1.2.4", "https://registry.npmmirror.com/@img/sharp-libvips-linux-riscv64/-/sharp-libvips-linux-riscv64-1.2.4.tgz", { "os": "linux", "cpu": "none" }, "sha512-oVDbcR4zUC0ce82teubSm+x6ETixtKZBh/qbREIOcI3cULzDyb18Sr/Wcyx7NRQeQzOiHTNbZFF1UwPS2scyGA=="], + + "@img/sharp-libvips-linux-s390x": ["@img/sharp-libvips-linux-s390x@1.2.4", "https://registry.npmmirror.com/@img/sharp-libvips-linux-s390x/-/sharp-libvips-linux-s390x-1.2.4.tgz", { "os": "linux", "cpu": "s390x" }, "sha512-qmp9VrzgPgMoGZyPvrQHqk02uyjA0/QrTO26Tqk6l4ZV0MPWIW6LTkqOIov+J1yEu7MbFQaDpwdwJKhbJvuRxQ=="], + + "@img/sharp-libvips-linux-x64": ["@img/sharp-libvips-linux-x64@1.2.4", "https://registry.npmmirror.com/@img/sharp-libvips-linux-x64/-/sharp-libvips-linux-x64-1.2.4.tgz", { "os": "linux", "cpu": "x64" }, "sha512-tJxiiLsmHc9Ax1bz3oaOYBURTXGIRDODBqhveVHonrHJ9/+k89qbLl0bcJns+e4t4rvaNBxaEZsFtSfAdquPrw=="], + + "@img/sharp-libvips-linuxmusl-arm64": ["@img/sharp-libvips-linuxmusl-arm64@1.2.4", "https://registry.npmmirror.com/@img/sharp-libvips-linuxmusl-arm64/-/sharp-libvips-linuxmusl-arm64-1.2.4.tgz", { "os": "linux", "cpu": "arm64" }, "sha512-FVQHuwx1IIuNow9QAbYUzJ+En8KcVm9Lk5+uGUQJHaZmMECZmOlix9HnH7n1TRkXMS0pGxIJokIVB9SuqZGGXw=="], + + "@img/sharp-libvips-linuxmusl-x64": ["@img/sharp-libvips-linuxmusl-x64@1.2.4", "https://registry.npmmirror.com/@img/sharp-libvips-linuxmusl-x64/-/sharp-libvips-linuxmusl-x64-1.2.4.tgz", { "os": "linux", "cpu": "x64" }, "sha512-+LpyBk7L44ZIXwz/VYfglaX/okxezESc6UxDSoyo2Ks6Jxc4Y7sGjpgU9s4PMgqgjj1gZCylTieNamqA1MF7Dg=="], + + "@img/sharp-linux-arm": ["@img/sharp-linux-arm@0.34.5", "https://registry.npmmirror.com/@img/sharp-linux-arm/-/sharp-linux-arm-0.34.5.tgz", { "optionalDependencies": { "@img/sharp-libvips-linux-arm": "1.2.4" }, "os": "linux", "cpu": "arm" }, "sha512-9dLqsvwtg1uuXBGZKsxem9595+ujv0sJ6Vi8wcTANSFpwV/GONat5eCkzQo/1O6zRIkh0m/8+5BjrRr7jDUSZw=="], + + "@img/sharp-linux-arm64": ["@img/sharp-linux-arm64@0.34.5", "https://registry.npmmirror.com/@img/sharp-linux-arm64/-/sharp-linux-arm64-0.34.5.tgz", { "optionalDependencies": { "@img/sharp-libvips-linux-arm64": "1.2.4" }, "os": "linux", "cpu": "arm64" }, "sha512-bKQzaJRY/bkPOXyKx5EVup7qkaojECG6NLYswgktOZjaXecSAeCWiZwwiFf3/Y+O1HrauiE3FVsGxFg8c24rZg=="], + + "@img/sharp-linux-ppc64": ["@img/sharp-linux-ppc64@0.34.5", "https://registry.npmmirror.com/@img/sharp-linux-ppc64/-/sharp-linux-ppc64-0.34.5.tgz", { "optionalDependencies": { "@img/sharp-libvips-linux-ppc64": "1.2.4" }, "os": "linux", "cpu": "ppc64" }, "sha512-7zznwNaqW6YtsfrGGDA6BRkISKAAE1Jo0QdpNYXNMHu2+0dTrPflTLNkpc8l7MUP5M16ZJcUvysVWWrMefZquA=="], + + "@img/sharp-linux-riscv64": ["@img/sharp-linux-riscv64@0.34.5", "https://registry.npmmirror.com/@img/sharp-linux-riscv64/-/sharp-linux-riscv64-0.34.5.tgz", { "optionalDependencies": { "@img/sharp-libvips-linux-riscv64": "1.2.4" }, "os": "linux", "cpu": "none" }, "sha512-51gJuLPTKa7piYPaVs8GmByo7/U7/7TZOq+cnXJIHZKavIRHAP77e3N2HEl3dgiqdD/w0yUfiJnII77PuDDFdw=="], + + "@img/sharp-linux-s390x": ["@img/sharp-linux-s390x@0.34.5", "https://registry.npmmirror.com/@img/sharp-linux-s390x/-/sharp-linux-s390x-0.34.5.tgz", { "optionalDependencies": { "@img/sharp-libvips-linux-s390x": "1.2.4" }, "os": "linux", "cpu": "s390x" }, "sha512-nQtCk0PdKfho3eC5MrbQoigJ2gd1CgddUMkabUj+rBevs8tZ2cULOx46E7oyX+04WGfABgIwmMC0VqieTiR4jg=="], + + "@img/sharp-linux-x64": ["@img/sharp-linux-x64@0.34.5", "https://registry.npmmirror.com/@img/sharp-linux-x64/-/sharp-linux-x64-0.34.5.tgz", { "optionalDependencies": { "@img/sharp-libvips-linux-x64": "1.2.4" }, "os": "linux", "cpu": "x64" }, "sha512-MEzd8HPKxVxVenwAa+JRPwEC7QFjoPWuS5NZnBt6B3pu7EG2Ge0id1oLHZpPJdn3OQK+BQDiw9zStiHBTJQQQQ=="], + + "@img/sharp-linuxmusl-arm64": ["@img/sharp-linuxmusl-arm64@0.34.5", "https://registry.npmmirror.com/@img/sharp-linuxmusl-arm64/-/sharp-linuxmusl-arm64-0.34.5.tgz", { "optionalDependencies": { "@img/sharp-libvips-linuxmusl-arm64": "1.2.4" }, "os": "linux", "cpu": "arm64" }, "sha512-fprJR6GtRsMt6Kyfq44IsChVZeGN97gTD331weR1ex1c1rypDEABN6Tm2xa1wE6lYb5DdEnk03NZPqA7Id21yg=="], + + "@img/sharp-linuxmusl-x64": ["@img/sharp-linuxmusl-x64@0.34.5", "https://registry.npmmirror.com/@img/sharp-linuxmusl-x64/-/sharp-linuxmusl-x64-0.34.5.tgz", { "optionalDependencies": { "@img/sharp-libvips-linuxmusl-x64": "1.2.4" }, "os": "linux", "cpu": "x64" }, "sha512-Jg8wNT1MUzIvhBFxViqrEhWDGzqymo3sV7z7ZsaWbZNDLXRJZoRGrjulp60YYtV4wfY8VIKcWidjojlLcWrd8Q=="], + + "@img/sharp-wasm32": ["@img/sharp-wasm32@0.34.5", "https://registry.npmmirror.com/@img/sharp-wasm32/-/sharp-wasm32-0.34.5.tgz", { "dependencies": { "@emnapi/runtime": "^1.7.0" }, "cpu": "none" }, "sha512-OdWTEiVkY2PHwqkbBI8frFxQQFekHaSSkUIJkwzclWZe64O1X4UlUjqqqLaPbUpMOQk6FBu/HtlGXNblIs0huw=="], + + "@img/sharp-win32-arm64": ["@img/sharp-win32-arm64@0.34.5", "https://registry.npmmirror.com/@img/sharp-win32-arm64/-/sharp-win32-arm64-0.34.5.tgz", { "os": "win32", "cpu": "arm64" }, "sha512-WQ3AgWCWYSb2yt+IG8mnC6Jdk9Whs7O0gxphblsLvdhSpSTtmu69ZG1Gkb6NuvxsNACwiPV6cNSZNzt0KPsw7g=="], + + "@img/sharp-win32-ia32": ["@img/sharp-win32-ia32@0.34.5", "https://registry.npmmirror.com/@img/sharp-win32-ia32/-/sharp-win32-ia32-0.34.5.tgz", { "os": "win32", "cpu": "ia32" }, "sha512-FV9m/7NmeCmSHDD5j4+4pNI8Cp3aW+JvLoXcTUo0IqyjSfAZJ8dIUmijx1qaJsIiU+Hosw6xM5KijAWRJCSgNg=="], + + "@img/sharp-win32-x64": ["@img/sharp-win32-x64@0.34.5", "https://registry.npmmirror.com/@img/sharp-win32-x64/-/sharp-win32-x64-0.34.5.tgz", { "os": "win32", "cpu": "x64" }, "sha512-+29YMsqY2/9eFEiW93eqWnuLcWcufowXewwSNIT6UwZdUUCrM3oFjMWH/Z6/TMmb4hlFenmfAVbpWeup2jryCw=="], + "@mermaid-js/parser": ["@mermaid-js/parser@1.1.0", "https://registry.npmmirror.com/@mermaid-js/parser/-/parser-1.1.0.tgz", { "dependencies": { "langium": "^4.0.0" } }, "sha512-gxK9ZX2+Fex5zu8LhRQoMeMPEHbc73UKZ0FQ54YrQtUxE1VVhMwzeNtKRPAu5aXks4FasbMe4xB4bWrmq6Jlxw=="], "@mixmark-io/domino": ["@mixmark-io/domino@2.2.0", "https://registry.npmmirror.com/@mixmark-io/domino/-/domino-2.2.0.tgz", {}, "sha512-Y28PR25bHXUg88kCV7nivXrP2Nj2RueZ3/l/jdx6J9f8J4nsEGcgX0Qe6lt7Pa+J79+kPiJU3LguR6O/6zrLOw=="], @@ -546,6 +599,8 @@ "depd": ["depd@2.0.0", "https://registry.npmmirror.com/depd/-/depd-2.0.0.tgz", {}, "sha512-g7nH6P6dyDioJogAAGprGpCtVImJhpPk/roCzdb3fIh61/s/nPsfR6onyMwkCAR/OlC3yBC0lESvUoQEAssIrw=="], + "detect-libc": ["detect-libc@2.1.2", "https://registry.npmmirror.com/detect-libc/-/detect-libc-2.1.2.tgz", {}, "sha512-Btj2BOOO83o3WyH59e8MgXsxEQVcarkUOpEYrubB0urwnN10yQ364rsiByU11nZlqWYZm05i/of7io4mzihBtQ=="], + "diff": ["diff@8.0.4", "https://registry.npmmirror.com/diff/-/diff-8.0.4.tgz", {}, "sha512-DPi0FmjiSU5EvQV0++GFDOJ9ASQUVFh5kD+OzOnYdi7n3Wpm9hWWGfB/O2blfHcMVTL5WkQXSnRiK9makhrcnw=="], "dijkstrajs": ["dijkstrajs@1.0.3", "https://registry.npmmirror.com/dijkstrajs/-/dijkstrajs-1.0.3.tgz", {}, "sha512-qiSlmBq9+BCdCA/L46dw8Uy93mloxsPSbwnm5yrKn2vMPiy8KyAskTF6zuV/j5BMsmOGZDPs7KjU+mjb670kfA=="], @@ -870,6 +925,8 @@ "setprototypeof": ["setprototypeof@1.2.0", "https://registry.npmmirror.com/setprototypeof/-/setprototypeof-1.2.0.tgz", {}, "sha512-E5LDX7Wrp85Kil5bhZv46j8jOeboKq5JMmYM3gVGdGH8xFpPWXUMsNrlODCrkoxMEeNi/XZIwuRvY4XNwYMJpw=="], + "sharp": ["sharp@0.34.5", "https://registry.npmmirror.com/sharp/-/sharp-0.34.5.tgz", { "dependencies": { "@img/colour": "^1.0.0", "detect-libc": "^2.1.2", "semver": "^7.7.3" }, "optionalDependencies": { "@img/sharp-darwin-arm64": "0.34.5", "@img/sharp-darwin-x64": "0.34.5", "@img/sharp-libvips-darwin-arm64": "1.2.4", "@img/sharp-libvips-darwin-x64": "1.2.4", "@img/sharp-libvips-linux-arm": "1.2.4", "@img/sharp-libvips-linux-arm64": "1.2.4", "@img/sharp-libvips-linux-ppc64": "1.2.4", "@img/sharp-libvips-linux-riscv64": "1.2.4", "@img/sharp-libvips-linux-s390x": "1.2.4", "@img/sharp-libvips-linux-x64": "1.2.4", "@img/sharp-libvips-linuxmusl-arm64": "1.2.4", "@img/sharp-libvips-linuxmusl-x64": "1.2.4", "@img/sharp-linux-arm": "0.34.5", "@img/sharp-linux-arm64": "0.34.5", "@img/sharp-linux-ppc64": "0.34.5", "@img/sharp-linux-riscv64": "0.34.5", "@img/sharp-linux-s390x": "0.34.5", "@img/sharp-linux-x64": "0.34.5", "@img/sharp-linuxmusl-arm64": "0.34.5", "@img/sharp-linuxmusl-x64": "0.34.5", "@img/sharp-wasm32": "0.34.5", "@img/sharp-win32-arm64": "0.34.5", "@img/sharp-win32-ia32": "0.34.5", "@img/sharp-win32-x64": "0.34.5" } }, "sha512-Ou9I5Ft9WNcCbXrU9cMgPBcCK8LiwLqcbywW3t4oDV37n1pzpuNLsYiAV8eODnjbtQlSDwZ2cUEeQz4E54Hltg=="], + "shebang-command": ["shebang-command@2.0.0", "https://registry.npmmirror.com/shebang-command/-/shebang-command-2.0.0.tgz", { "dependencies": { "shebang-regex": "^3.0.0" } }, "sha512-kHxr2zZpYtdmrN1qDjrrX/Z1rR1kG8Dx+gkpK1G4eXmvXswmcE1hTWBWYUzlraYw1/yZp6YuDY77YtvbN0dmDA=="], "shebang-regex": ["shebang-regex@3.0.0", "https://registry.npmmirror.com/shebang-regex/-/shebang-regex-3.0.0.tgz", {}, "sha512-7++dFhtcx3353uBaq8DDR4NuxBetBzC7ZQOhmTQInHEd6bSrXdiEyzCvG07Z44UYdLShWUyXt5M/yhz8ekcb1A=="], diff --git a/desktop/bun.lock b/desktop/bun.lock index cb6223d8..a49e026c 100644 --- a/desktop/bun.lock +++ b/desktop/bun.lock @@ -32,6 +32,7 @@ "react-diff-viewer-continued": "^4.2.0", "react-dom": "^18.3.1", "react-shiki": "^0.9.2", + "sharp": "0.34.5", "shiki": "^4.0.2", "zustand": "^5.0.3", }, @@ -257,6 +258,56 @@ "@iconify/utils": ["@iconify/utils@3.1.0", "https://registry.npmmirror.com/@iconify/utils/-/utils-3.1.0.tgz", { "dependencies": { "@antfu/install-pkg": "^1.1.0", "@iconify/types": "^2.0.0", "mlly": "^1.8.0" } }, "sha512-Zlzem1ZXhI1iHeeERabLNzBHdOa4VhQbqAcOQaMKuTuyZCpwKbC2R4Dd0Zo3g9EAc+Y4fiarO8HIHRAth7+skw=="], + "@img/colour": ["@img/colour@1.1.0", "https://registry.npmmirror.com/@img/colour/-/colour-1.1.0.tgz", {}, "sha512-Td76q7j57o/tLVdgS746cYARfSyxk8iEfRxewL9h4OMzYhbW4TAcppl0mT4eyqXddh6L/jwoM75mo7ixa/pCeQ=="], + + "@img/sharp-darwin-arm64": ["@img/sharp-darwin-arm64@0.34.5", "https://registry.npmmirror.com/@img/sharp-darwin-arm64/-/sharp-darwin-arm64-0.34.5.tgz", { "optionalDependencies": { "@img/sharp-libvips-darwin-arm64": "1.2.4" }, "os": "darwin", "cpu": "arm64" }, "sha512-imtQ3WMJXbMY4fxb/Ndp6HBTNVtWCUI0WdobyheGf5+ad6xX8VIDO8u2xE4qc/fr08CKG/7dDseFtn6M6g/r3w=="], + + "@img/sharp-darwin-x64": ["@img/sharp-darwin-x64@0.34.5", "https://registry.npmmirror.com/@img/sharp-darwin-x64/-/sharp-darwin-x64-0.34.5.tgz", { "optionalDependencies": { "@img/sharp-libvips-darwin-x64": "1.2.4" }, "os": "darwin", "cpu": "x64" }, "sha512-YNEFAF/4KQ/PeW0N+r+aVVsoIY0/qxxikF2SWdp+NRkmMB7y9LBZAVqQ4yhGCm/H3H270OSykqmQMKLBhBJDEw=="], + + "@img/sharp-libvips-darwin-arm64": ["@img/sharp-libvips-darwin-arm64@1.2.4", "https://registry.npmmirror.com/@img/sharp-libvips-darwin-arm64/-/sharp-libvips-darwin-arm64-1.2.4.tgz", { "os": "darwin", "cpu": "arm64" }, "sha512-zqjjo7RatFfFoP0MkQ51jfuFZBnVE2pRiaydKJ1G/rHZvnsrHAOcQALIi9sA5co5xenQdTugCvtb1cuf78Vf4g=="], + + "@img/sharp-libvips-darwin-x64": ["@img/sharp-libvips-darwin-x64@1.2.4", "https://registry.npmmirror.com/@img/sharp-libvips-darwin-x64/-/sharp-libvips-darwin-x64-1.2.4.tgz", { "os": "darwin", "cpu": "x64" }, "sha512-1IOd5xfVhlGwX+zXv2N93k0yMONvUlANylbJw1eTah8K/Jtpi15KC+WSiaX/nBmbm2HxRM1gZ0nSdjSsrZbGKg=="], + + "@img/sharp-libvips-linux-arm": ["@img/sharp-libvips-linux-arm@1.2.4", "https://registry.npmmirror.com/@img/sharp-libvips-linux-arm/-/sharp-libvips-linux-arm-1.2.4.tgz", { "os": "linux", "cpu": "arm" }, "sha512-bFI7xcKFELdiNCVov8e44Ia4u2byA+l3XtsAj+Q8tfCwO6BQ8iDojYdvoPMqsKDkuoOo+X6HZA0s0q11ANMQ8A=="], + + "@img/sharp-libvips-linux-arm64": ["@img/sharp-libvips-linux-arm64@1.2.4", "https://registry.npmmirror.com/@img/sharp-libvips-linux-arm64/-/sharp-libvips-linux-arm64-1.2.4.tgz", { "os": "linux", "cpu": "arm64" }, "sha512-excjX8DfsIcJ10x1Kzr4RcWe1edC9PquDRRPx3YVCvQv+U5p7Yin2s32ftzikXojb1PIFc/9Mt28/y+iRklkrw=="], + + "@img/sharp-libvips-linux-ppc64": ["@img/sharp-libvips-linux-ppc64@1.2.4", "https://registry.npmmirror.com/@img/sharp-libvips-linux-ppc64/-/sharp-libvips-linux-ppc64-1.2.4.tgz", { "os": "linux", "cpu": "ppc64" }, "sha512-FMuvGijLDYG6lW+b/UvyilUWu5Ayu+3r2d1S8notiGCIyYU/76eig1UfMmkZ7vwgOrzKzlQbFSuQfgm7GYUPpA=="], + + "@img/sharp-libvips-linux-riscv64": ["@img/sharp-libvips-linux-riscv64@1.2.4", "https://registry.npmmirror.com/@img/sharp-libvips-linux-riscv64/-/sharp-libvips-linux-riscv64-1.2.4.tgz", { "os": "linux", "cpu": "none" }, "sha512-oVDbcR4zUC0ce82teubSm+x6ETixtKZBh/qbREIOcI3cULzDyb18Sr/Wcyx7NRQeQzOiHTNbZFF1UwPS2scyGA=="], + + "@img/sharp-libvips-linux-s390x": ["@img/sharp-libvips-linux-s390x@1.2.4", "https://registry.npmmirror.com/@img/sharp-libvips-linux-s390x/-/sharp-libvips-linux-s390x-1.2.4.tgz", { "os": "linux", "cpu": "s390x" }, "sha512-qmp9VrzgPgMoGZyPvrQHqk02uyjA0/QrTO26Tqk6l4ZV0MPWIW6LTkqOIov+J1yEu7MbFQaDpwdwJKhbJvuRxQ=="], + + "@img/sharp-libvips-linux-x64": ["@img/sharp-libvips-linux-x64@1.2.4", "https://registry.npmmirror.com/@img/sharp-libvips-linux-x64/-/sharp-libvips-linux-x64-1.2.4.tgz", { "os": "linux", "cpu": "x64" }, "sha512-tJxiiLsmHc9Ax1bz3oaOYBURTXGIRDODBqhveVHonrHJ9/+k89qbLl0bcJns+e4t4rvaNBxaEZsFtSfAdquPrw=="], + + "@img/sharp-libvips-linuxmusl-arm64": ["@img/sharp-libvips-linuxmusl-arm64@1.2.4", "https://registry.npmmirror.com/@img/sharp-libvips-linuxmusl-arm64/-/sharp-libvips-linuxmusl-arm64-1.2.4.tgz", { "os": "linux", "cpu": "arm64" }, "sha512-FVQHuwx1IIuNow9QAbYUzJ+En8KcVm9Lk5+uGUQJHaZmMECZmOlix9HnH7n1TRkXMS0pGxIJokIVB9SuqZGGXw=="], + + "@img/sharp-libvips-linuxmusl-x64": ["@img/sharp-libvips-linuxmusl-x64@1.2.4", "https://registry.npmmirror.com/@img/sharp-libvips-linuxmusl-x64/-/sharp-libvips-linuxmusl-x64-1.2.4.tgz", { "os": "linux", "cpu": "x64" }, "sha512-+LpyBk7L44ZIXwz/VYfglaX/okxezESc6UxDSoyo2Ks6Jxc4Y7sGjpgU9s4PMgqgjj1gZCylTieNamqA1MF7Dg=="], + + "@img/sharp-linux-arm": ["@img/sharp-linux-arm@0.34.5", "https://registry.npmmirror.com/@img/sharp-linux-arm/-/sharp-linux-arm-0.34.5.tgz", { "optionalDependencies": { "@img/sharp-libvips-linux-arm": "1.2.4" }, "os": "linux", "cpu": "arm" }, "sha512-9dLqsvwtg1uuXBGZKsxem9595+ujv0sJ6Vi8wcTANSFpwV/GONat5eCkzQo/1O6zRIkh0m/8+5BjrRr7jDUSZw=="], + + "@img/sharp-linux-arm64": ["@img/sharp-linux-arm64@0.34.5", "https://registry.npmmirror.com/@img/sharp-linux-arm64/-/sharp-linux-arm64-0.34.5.tgz", { "optionalDependencies": { "@img/sharp-libvips-linux-arm64": "1.2.4" }, "os": "linux", "cpu": "arm64" }, "sha512-bKQzaJRY/bkPOXyKx5EVup7qkaojECG6NLYswgktOZjaXecSAeCWiZwwiFf3/Y+O1HrauiE3FVsGxFg8c24rZg=="], + + "@img/sharp-linux-ppc64": ["@img/sharp-linux-ppc64@0.34.5", "https://registry.npmmirror.com/@img/sharp-linux-ppc64/-/sharp-linux-ppc64-0.34.5.tgz", { "optionalDependencies": { "@img/sharp-libvips-linux-ppc64": "1.2.4" }, "os": "linux", "cpu": "ppc64" }, "sha512-7zznwNaqW6YtsfrGGDA6BRkISKAAE1Jo0QdpNYXNMHu2+0dTrPflTLNkpc8l7MUP5M16ZJcUvysVWWrMefZquA=="], + + "@img/sharp-linux-riscv64": ["@img/sharp-linux-riscv64@0.34.5", "https://registry.npmmirror.com/@img/sharp-linux-riscv64/-/sharp-linux-riscv64-0.34.5.tgz", { "optionalDependencies": { "@img/sharp-libvips-linux-riscv64": "1.2.4" }, "os": "linux", "cpu": "none" }, "sha512-51gJuLPTKa7piYPaVs8GmByo7/U7/7TZOq+cnXJIHZKavIRHAP77e3N2HEl3dgiqdD/w0yUfiJnII77PuDDFdw=="], + + "@img/sharp-linux-s390x": ["@img/sharp-linux-s390x@0.34.5", "https://registry.npmmirror.com/@img/sharp-linux-s390x/-/sharp-linux-s390x-0.34.5.tgz", { "optionalDependencies": { "@img/sharp-libvips-linux-s390x": "1.2.4" }, "os": "linux", "cpu": "s390x" }, "sha512-nQtCk0PdKfho3eC5MrbQoigJ2gd1CgddUMkabUj+rBevs8tZ2cULOx46E7oyX+04WGfABgIwmMC0VqieTiR4jg=="], + + "@img/sharp-linux-x64": ["@img/sharp-linux-x64@0.34.5", "https://registry.npmmirror.com/@img/sharp-linux-x64/-/sharp-linux-x64-0.34.5.tgz", { "optionalDependencies": { "@img/sharp-libvips-linux-x64": "1.2.4" }, "os": "linux", "cpu": "x64" }, "sha512-MEzd8HPKxVxVenwAa+JRPwEC7QFjoPWuS5NZnBt6B3pu7EG2Ge0id1oLHZpPJdn3OQK+BQDiw9zStiHBTJQQQQ=="], + + "@img/sharp-linuxmusl-arm64": ["@img/sharp-linuxmusl-arm64@0.34.5", "https://registry.npmmirror.com/@img/sharp-linuxmusl-arm64/-/sharp-linuxmusl-arm64-0.34.5.tgz", { "optionalDependencies": { "@img/sharp-libvips-linuxmusl-arm64": "1.2.4" }, "os": "linux", "cpu": "arm64" }, "sha512-fprJR6GtRsMt6Kyfq44IsChVZeGN97gTD331weR1ex1c1rypDEABN6Tm2xa1wE6lYb5DdEnk03NZPqA7Id21yg=="], + + "@img/sharp-linuxmusl-x64": ["@img/sharp-linuxmusl-x64@0.34.5", "https://registry.npmmirror.com/@img/sharp-linuxmusl-x64/-/sharp-linuxmusl-x64-0.34.5.tgz", { "optionalDependencies": { "@img/sharp-libvips-linuxmusl-x64": "1.2.4" }, "os": "linux", "cpu": "x64" }, "sha512-Jg8wNT1MUzIvhBFxViqrEhWDGzqymo3sV7z7ZsaWbZNDLXRJZoRGrjulp60YYtV4wfY8VIKcWidjojlLcWrd8Q=="], + + "@img/sharp-wasm32": ["@img/sharp-wasm32@0.34.5", "https://registry.npmmirror.com/@img/sharp-wasm32/-/sharp-wasm32-0.34.5.tgz", { "dependencies": { "@emnapi/runtime": "^1.7.0" }, "cpu": "none" }, "sha512-OdWTEiVkY2PHwqkbBI8frFxQQFekHaSSkUIJkwzclWZe64O1X4UlUjqqqLaPbUpMOQk6FBu/HtlGXNblIs0huw=="], + + "@img/sharp-win32-arm64": ["@img/sharp-win32-arm64@0.34.5", "https://registry.npmmirror.com/@img/sharp-win32-arm64/-/sharp-win32-arm64-0.34.5.tgz", { "os": "win32", "cpu": "arm64" }, "sha512-WQ3AgWCWYSb2yt+IG8mnC6Jdk9Whs7O0gxphblsLvdhSpSTtmu69ZG1Gkb6NuvxsNACwiPV6cNSZNzt0KPsw7g=="], + + "@img/sharp-win32-ia32": ["@img/sharp-win32-ia32@0.34.5", "https://registry.npmmirror.com/@img/sharp-win32-ia32/-/sharp-win32-ia32-0.34.5.tgz", { "os": "win32", "cpu": "ia32" }, "sha512-FV9m/7NmeCmSHDD5j4+4pNI8Cp3aW+JvLoXcTUo0IqyjSfAZJ8dIUmijx1qaJsIiU+Hosw6xM5KijAWRJCSgNg=="], + + "@img/sharp-win32-x64": ["@img/sharp-win32-x64@0.34.5", "https://registry.npmmirror.com/@img/sharp-win32-x64/-/sharp-win32-x64-0.34.5.tgz", { "os": "win32", "cpu": "x64" }, "sha512-+29YMsqY2/9eFEiW93eqWnuLcWcufowXewwSNIT6UwZdUUCrM3oFjMWH/Z6/TMmb4hlFenmfAVbpWeup2jryCw=="], + "@isaacs/cliui": ["@isaacs/cliui@8.0.2", "https://registry.npmmirror.com/@isaacs/cliui/-/cliui-8.0.2.tgz", { "dependencies": { "string-width": "^5.1.2", "string-width-cjs": "npm:string-width@^4.2.0", "strip-ansi": "^7.0.1", "strip-ansi-cjs": "npm:strip-ansi@^6.0.1", "wrap-ansi": "^8.1.0", "wrap-ansi-cjs": "npm:wrap-ansi@^7.0.0" } }, "sha512-O8jcjabXaleOG9DQ0+ARXWZBTfnP4WNAqzuiJK7ll44AmxGKv/J2M4TPjxjY3znBCfvBXFzucm1twdyFybFqEA=="], "@isaacs/fs-minipass": ["@isaacs/fs-minipass@4.0.1", "https://registry.npmmirror.com/@isaacs/fs-minipass/-/fs-minipass-4.0.1.tgz", { "dependencies": { "minipass": "^7.0.4" } }, "sha512-wgm9Ehl2jpeqP3zw/7mo3kRHFp5MEDhqAdwy1fTGkHAwnkGOVsgpvQhL8B5n1qlb01jV3n/bI0ZfZp5lWA1k4w=="], @@ -1479,6 +1530,8 @@ "set-blocking": ["set-blocking@2.0.0", "https://registry.npmmirror.com/set-blocking/-/set-blocking-2.0.0.tgz", {}, "sha512-KiKBS8AnWGEyLzofFfmvKwpdPzqiy16LvQfK3yv/fVH7Bj13/wl3JSR1J+rfgRE9q7xUJK4qvgS8raSOeLUehw=="], + "sharp": ["sharp@0.34.5", "https://registry.npmmirror.com/sharp/-/sharp-0.34.5.tgz", { "dependencies": { "@img/colour": "^1.0.0", "detect-libc": "^2.1.2", "semver": "^7.7.3" }, "optionalDependencies": { "@img/sharp-darwin-arm64": "0.34.5", "@img/sharp-darwin-x64": "0.34.5", "@img/sharp-libvips-darwin-arm64": "1.2.4", "@img/sharp-libvips-darwin-x64": "1.2.4", "@img/sharp-libvips-linux-arm": "1.2.4", "@img/sharp-libvips-linux-arm64": "1.2.4", "@img/sharp-libvips-linux-ppc64": "1.2.4", "@img/sharp-libvips-linux-riscv64": "1.2.4", "@img/sharp-libvips-linux-s390x": "1.2.4", "@img/sharp-libvips-linux-x64": "1.2.4", "@img/sharp-libvips-linuxmusl-arm64": "1.2.4", "@img/sharp-libvips-linuxmusl-x64": "1.2.4", "@img/sharp-linux-arm": "0.34.5", "@img/sharp-linux-arm64": "0.34.5", "@img/sharp-linux-ppc64": "0.34.5", "@img/sharp-linux-riscv64": "0.34.5", "@img/sharp-linux-s390x": "0.34.5", "@img/sharp-linux-x64": "0.34.5", "@img/sharp-linuxmusl-arm64": "0.34.5", "@img/sharp-linuxmusl-x64": "0.34.5", "@img/sharp-wasm32": "0.34.5", "@img/sharp-win32-arm64": "0.34.5", "@img/sharp-win32-ia32": "0.34.5", "@img/sharp-win32-x64": "0.34.5" } }, "sha512-Ou9I5Ft9WNcCbXrU9cMgPBcCK8LiwLqcbywW3t4oDV37n1pzpuNLsYiAV8eODnjbtQlSDwZ2cUEeQz4E54Hltg=="], + "shebang-command": ["shebang-command@2.0.0", "https://registry.npmmirror.com/shebang-command/-/shebang-command-2.0.0.tgz", { "dependencies": { "shebang-regex": "^3.0.0" } }, "sha512-kHxr2zZpYtdmrN1qDjrrX/Z1rR1kG8Dx+gkpK1G4eXmvXswmcE1hTWBWYUzlraYw1/yZp6YuDY77YtvbN0dmDA=="], "shebang-regex": ["shebang-regex@3.0.0", "https://registry.npmmirror.com/shebang-regex/-/shebang-regex-3.0.0.tgz", {}, "sha512-7++dFhtcx3353uBaq8DDR4NuxBetBzC7ZQOhmTQInHEd6bSrXdiEyzCvG07Z44UYdLShWUyXt5M/yhz8ekcb1A=="], diff --git a/desktop/package.json b/desktop/package.json index 7270cfa1..261b1864 100644 --- a/desktop/package.json +++ b/desktop/package.json @@ -17,6 +17,8 @@ "asarUnpack": [ "dist/**", "node_modules/node-pty/**", + "node_modules/sharp/**", + "node_modules/@img/**", "src-tauri/binaries/**" ], "artifactName": "Claude-Code-Haha-${version}-${os}-${arch}.${ext}", @@ -131,6 +133,7 @@ "react-diff-viewer-continued": "^4.2.0", "react-dom": "^18.3.1", "react-shiki": "^0.9.2", + "sharp": "0.34.5", "shiki": "^4.0.2", "zustand": "^5.0.3" }, diff --git a/desktop/scripts/build-sidecars.ts b/desktop/scripts/build-sidecars.ts index e90b64c8..849f9e89 100644 --- a/desktop/scripts/build-sidecars.ts +++ b/desktop/scripts/build-sidecars.ts @@ -206,10 +206,9 @@ async function compileExecutable({ minify: { whitespace: true, identifiers: true, syntax: true }, sourcemap: 'none', target: 'bun', - // 可选 npm 包:开 telemetry / 用 sharp 图像 / 用 Bedrock/Vertex 等 - // 替代 provider 时才需要,全部不在顶层 package.json 里。标 external - // 让 bun build 跳过解析;运行时 import 在没装时自然失败,由 try/catch - // 或 feature() gate 兜底。 + // 运行时可选 npm 包保持 external,避免将原生模块嵌入 sidecar。 + // sharp 由 desktop/package.json 声明并随 Electron 应用分发;其他 + // provider/telemetry 模块在未安装时由 try/catch 或 feature() gate 兜底。 external: [ // OpenTelemetry exporters(开 OTEL_* env 时才加载) '@opentelemetry/exporter-trace-otlp-grpc', diff --git a/desktop/src/__tests__/generalSettings.test.tsx b/desktop/src/__tests__/generalSettings.test.tsx index 7ec0013e..38265339 100644 --- a/desktop/src/__tests__/generalSettings.test.tsx +++ b/desktop/src/__tests__/generalSettings.test.tsx @@ -2051,6 +2051,62 @@ describe('Settings > Providers tab', () => { expect(within(dialog).getByText('Requests will be translated via the local proxy')).toBeInTheDocument() }) + it('uses the proxy in settings JSON and connection tests when nested tool media is unsupported', async () => { + providerStoreState.testConfig = vi.fn().mockResolvedValue({ + connectivity: { success: true, latencyMs: 1 }, + proxy: { success: true, latencyMs: 1 }, + }) + providerStoreState.presets = [{ + id: 'custom', + name: 'Custom', + baseUrl: 'https://api.example.com/anthropic', + apiFormat: 'anthropic', + defaultModels: { + main: 'model-main', + haiku: '', + sonnet: '', + opus: '', + }, + needsApiKey: true, + websiteUrl: '', + }] + + render() + fireEvent.click(screen.getByRole('button', { name: /Add Provider/i })) + + const dialog = screen.getByRole('dialog') + const mediaSupport = within(dialog).getByLabelText('Preserve nested tool result media') + expect(mediaSupport).toBeChecked() + fireEvent.click(mediaSupport) + + const settingsTextarea = await waitFor(() => { + const textarea = dialog.querySelector('textarea') as HTMLTextAreaElement + const settings = JSON.parse(textarea.value) as { + env?: { + ANTHROPIC_API_KEY?: string + ANTHROPIC_AUTH_TOKEN?: string + ANTHROPIC_BASE_URL?: string + } + } + expect(settings.env?.ANTHROPIC_BASE_URL).toMatch(/\/proxy$/) + expect(settings.env?.ANTHROPIC_API_KEY).toBe('proxy-managed') + expect(settings.env?.ANTHROPIC_AUTH_TOKEN).toBeUndefined() + return textarea + }) + expect(settingsTextarea.value).not.toContain('"ANTHROPIC_BASE_URL": "https://api.example.com/anthropic"') + + fireEvent.change(within(dialog).getByPlaceholderText('sk-...'), { target: { value: 'sk-test' } }) + fireEvent.click(within(dialog).getByRole('button', { name: /Test Connection/i })) + + await waitFor(() => { + expect(providerStoreState.testConfig).toHaveBeenCalledWith(expect.objectContaining({ + baseUrl: 'https://api.example.com/anthropic', + apiFormat: 'anthropic', + supportsNestedToolResultMedia: false, + })) + }) + }) + it('localizes the main model placeholder in the provider form', () => { useSettingsStore.setState({ locale: 'zh' }) providerStoreState.presets = [ diff --git a/desktop/src/i18n/locales/en.ts b/desktop/src/i18n/locales/en.ts index e2faadee..7630d7de 100644 --- a/desktop/src/i18n/locales/en.ts +++ b/desktop/src/i18n/locales/en.ts @@ -797,6 +797,9 @@ Row 9, all 8 cells: continuing from straight down, turning left through lower-le 'settings.providers.toolSearchConfirmEnable': 'Enable anyway', 'settings.providers.disableExperimentalBetas': 'Disable experimental beta headers', 'settings.providers.disableExperimentalBetasDesc': 'Sets CLAUDE_CODE_DISABLE_EXPERIMENTAL_BETAS=1 to avoid beta API shapes that third-party gateways may reject. GPT and o-series models still receive the reasoning effort selected for the Session; other direct Anthropic-compatible models may fall back to the gateway default.', + 'settings.providers.supportsNestedToolResultMedia': 'Preserve nested tool result media', + 'settings.providers.nestedToolResultMediaDesc': 'Keeps images and files inside tool results when the Anthropic-compatible endpoint supports nested media. Turn this off for third-party endpoints that cannot display nested media; images and files are then lifted into standalone content blocks.', + 'settings.providers.nestedToolResultMediaUnsupported': 'Only Anthropic Messages providers can configure nested tool result media.', 'settings.providers.imageGenerationEnabled': 'Enable image generation', 'settings.providers.imageGenerationEnabledDesc': 'Let chat generate images through an OpenAI-compatible Images API. Credentials stay in provider settings, not in the skill.', 'settings.providers.imageGenerationModel': 'Image model', diff --git a/desktop/src/i18n/locales/jp.ts b/desktop/src/i18n/locales/jp.ts index 07a8d0cf..008d0e6b 100644 --- a/desktop/src/i18n/locales/jp.ts +++ b/desktop/src/i18n/locales/jp.ts @@ -799,6 +799,9 @@ export const jp: Record = { 'settings.providers.toolSearchConfirmEnable': '有効にする', 'settings.providers.disableExperimentalBetas': '実験的な Beta ヘッダーを無効化', 'settings.providers.disableExperimentalBetasDesc': 'このプロバイダーに CLAUDE_CODE_DISABLE_EXPERIMENTAL_BETAS=1 を設定し、サードパーティゲートウェイが拒否する可能性のある beta API 形式を避けます。GPT および o シリーズのモデルには Session で選択した推論強度が引き続き転送されます。その他の Anthropic 互換モデルへの直接接続では、ゲートウェイの既定値に戻る場合があります。', + 'settings.providers.supportsNestedToolResultMedia': 'ツール結果のメディアを保持', + 'settings.providers.nestedToolResultMediaDesc': 'Anthropic 互換エンドポイントがネストされたメディアに対応している場合、画像とファイルを tool result 内に保持します。ネストされたメディアを表示できないサードパーティエンドポイントでは、このオプションをオフにすると画像とファイルが独立したコンテンツブロックに引き上げられます。', + 'settings.providers.nestedToolResultMediaUnsupported': 'ツール結果のネストされたメディアを設定できるのは Anthropic Messages プロバイダーのみです。', 'settings.providers.imageGenerationEnabled': '画像生成を有効にする', 'settings.providers.imageGenerationEnabledDesc': 'OpenAI 互換 Images API を通じてチャットから画像を生成します。認証情報は Skill ではなくプロバイダー設定に保存されます。', 'settings.providers.imageGenerationModel': '画像モデル', diff --git a/desktop/src/i18n/locales/kr.ts b/desktop/src/i18n/locales/kr.ts index e64510a0..5e751195 100644 --- a/desktop/src/i18n/locales/kr.ts +++ b/desktop/src/i18n/locales/kr.ts @@ -799,6 +799,9 @@ export const kr: Record = { 'settings.providers.toolSearchConfirmEnable': '계속 사용', 'settings.providers.disableExperimentalBetas': '실험적 Beta 헤더 비활성화', 'settings.providers.disableExperimentalBetasDesc': '이 공급자에 CLAUDE_CODE_DISABLE_EXPERIMENTAL_BETAS=1을 설정해 타사 게이트웨이가 거부할 수 있는 beta API 형식을 피합니다. GPT 및 o 시리즈 모델에는 Session에서 선택한 추론 강도가 계속 전달됩니다. 다른 Anthropic 호환 모델에 직접 연결할 때는 게이트웨이 기본값으로 돌아갈 수 있습니다.', + 'settings.providers.supportsNestedToolResultMedia': '도구 결과의 미디어 보존', + 'settings.providers.nestedToolResultMediaDesc': 'Anthropic 호환 엔드포인트가 중첩 미디어를 지원하면 이미지와 파일을 tool result 안에 유지합니다. 중첩 미디어를 표시할 수 없는 타사 엔드포인트에서는 이 옵션을 끄면 이미지와 파일이 독립 콘텐츠 블록으로 올라갑니다.', + 'settings.providers.nestedToolResultMediaUnsupported': 'Anthropic Messages 공급자만 도구 결과의 중첩 미디어를 설정할 수 있습니다.', 'settings.providers.imageGenerationEnabled': '이미지 생성 사용', 'settings.providers.imageGenerationEnabledDesc': 'OpenAI 호환 Images API를 통해 채팅에서 이미지를 생성합니다. 인증 정보는 Skill이 아니라 공급자 설정에 저장됩니다.', 'settings.providers.imageGenerationModel': '이미지 모델', diff --git a/desktop/src/i18n/locales/zh-TW.ts b/desktop/src/i18n/locales/zh-TW.ts index 767fcf2c..67b1fd27 100644 --- a/desktop/src/i18n/locales/zh-TW.ts +++ b/desktop/src/i18n/locales/zh-TW.ts @@ -798,6 +798,9 @@ export const zh: Record = { 'settings.providers.toolSearchConfirmEnable': '仍然啟用', 'settings.providers.disableExperimentalBetas': '關閉實驗性 Beta 標頭', 'settings.providers.disableExperimentalBetasDesc': '為此服務商設定 CLAUDE_CODE_DISABLE_EXPERIMENTAL_BETAS=1,避免第三方通道拒絕 beta API 形態。GPT 與 o 系列模型仍會轉送 Session 中選擇的推理強度;其他直連 Anthropic 相容模型可能退回通道預設值。', + 'settings.providers.supportsNestedToolResultMedia': '保留工具結果中的媒體', + 'settings.providers.nestedToolResultMediaDesc': '當 Anthropic 相容端點支援巢狀媒體時,將圖片和檔案保留在 tool result 內。若第三方端點無法顯示巢狀媒體,請關閉此選項,圖片和檔案將提升為獨立的內容區塊。', + 'settings.providers.nestedToolResultMediaUnsupported': '只有 Anthropic Messages 服務商可以設定工具結果中的巢狀媒體。', 'settings.providers.imageGenerationEnabled': '啟用圖片生成', 'settings.providers.imageGenerationEnabledDesc': '允許聊天透過 OpenAI 相容的 Images API 生成圖片。憑證保存在服務商設定中,不寫入 Skill。', 'settings.providers.imageGenerationModel': '圖片模型', diff --git a/desktop/src/i18n/locales/zh.ts b/desktop/src/i18n/locales/zh.ts index 5b61a306..3b3e8ecb 100644 --- a/desktop/src/i18n/locales/zh.ts +++ b/desktop/src/i18n/locales/zh.ts @@ -798,6 +798,9 @@ export const zh: Record = { 'settings.providers.toolSearchConfirmEnable': '仍然启用', 'settings.providers.disableExperimentalBetas': '关闭实验性 Beta 头', 'settings.providers.disableExperimentalBetasDesc': '为此服务商设置 CLAUDE_CODE_DISABLE_EXPERIMENTAL_BETAS=1,避免第三方通道拒绝 beta API 形态。GPT 和 o 系列模型仍会转发 Session 中选择的推理强度;其他直连 Anthropic 兼容模型可能回退为通道默认值。', + 'settings.providers.supportsNestedToolResultMedia': '保留工具结果中的媒体', + 'settings.providers.nestedToolResultMediaDesc': '当 Anthropic 兼容端点支持嵌套媒体时,将图片和文件保留在 tool result 内。若第三方端点无法显示嵌套媒体,请关闭此选项,图片和文件将提升为独立的内容块。', + 'settings.providers.nestedToolResultMediaUnsupported': '仅 Anthropic Messages 服务商可以配置工具结果中的嵌套媒体。', 'settings.providers.imageGenerationEnabled': '启用图片生成', 'settings.providers.imageGenerationEnabledDesc': '允许聊天通过 OpenAI 兼容的 Images API 生图。凭证保存在服务商设置中,不写进 Skill。', 'settings.providers.imageGenerationModel': '生图模型', diff --git a/desktop/src/pages/settings/ProviderSettings.tsx b/desktop/src/pages/settings/ProviderSettings.tsx index 57cc7cbc..22d99cbf 100644 --- a/desktop/src/pages/settings/ProviderSettings.tsx +++ b/desktop/src/pages/settings/ProviderSettings.tsx @@ -571,12 +571,12 @@ function getProviderAuthValue(apiKey: string, preset: ProviderPreset): string { } function buildSettingsJsonAuthEnv( - apiFormat: ApiFormat, + needsProxy: boolean, authStrategy: ProviderAuthStrategy, apiKey: string, preset: ProviderPreset, ): Record { - if (apiFormat !== 'anthropic') { + if (needsProxy) { return { ANTHROPIC_API_KEY: 'proxy-managed' } } @@ -903,6 +903,13 @@ function updateSettingsJsonModels( } } +function providerNeedsProxy( + apiFormat: ApiFormat, + supportsNestedToolResultMedia: boolean, +): boolean { + return apiFormat !== 'anthropic' || !supportsNestedToolResultMedia +} + function updateSettingsJsonProviderConnection( raw: string, apiFormat: ApiFormat, @@ -913,6 +920,7 @@ function updateSettingsJsonProviderConnection( proxyBaseUrl: string, toolSearchEnabled = false, disableExperimentalBetas = false, + supportsNestedToolResultMedia = true, ): string { try { const parsed = JSON.parse(raw || '{}') as { env?: Record } @@ -924,8 +932,13 @@ function updateSettingsJsonProviderConnection( delete env.ANTHROPIC_AUTH_TOKEN applyToolSearchEnv(env, apiFormat, toolSearchEnabled) applyDisableExperimentalBetasEnv(env, disableExperimentalBetas) - env.ANTHROPIC_BASE_URL = apiFormat !== 'anthropic' ? proxyBaseUrl : baseUrl - Object.assign(env, buildSettingsJsonAuthEnv(apiFormat, authStrategy, apiKey, preset)) + env.ANTHROPIC_BASE_URL = providerNeedsProxy(apiFormat, supportsNestedToolResultMedia) ? proxyBaseUrl : baseUrl + Object.assign(env, buildSettingsJsonAuthEnv( + providerNeedsProxy(apiFormat, supportsNestedToolResultMedia), + authStrategy, + apiKey, + preset, + )) parsed.env = env return JSON.stringify(parsed, null, 2) } catch { @@ -1011,6 +1024,7 @@ function ProviderFormModal({ open, onClose, mode, provider, presets }: ProviderF const [toolSearchEnabled, setToolSearchEnabled] = useState(provider?.toolSearchEnabled ?? false) const [toolSearchConfirmOpen, setToolSearchConfirmOpen] = useState(false) const [disableExperimentalBetas, setDisableExperimentalBetas] = useState(provider?.disableExperimentalBetas ?? false) + const [supportsNestedToolResultMedia, setSupportsNestedToolResultMedia] = useState(provider?.supportsNestedToolResultMedia ?? true) const [imageGeneration, setImageGeneration] = useState({ enabled: Boolean(provider?.imageGeneration), model: provider?.imageGeneration?.model ?? '', @@ -1043,6 +1057,7 @@ function ProviderFormModal({ open, onClose, mode, provider, presets }: ProviderF autoCompactWindow, toolSearchEnabled, disableExperimentalBetas, + supportsNestedToolResultMedia, } const providerSettingsRef = useRef(currentProviderSettings) providerSettingsRef.current = currentProviderSettings @@ -1071,8 +1086,9 @@ function ProviderFormModal({ open, onClose, mode, provider, presets }: ProviderF autoCompactWindow, toolSearchEnabled, disableExperimentalBetas, + supportsNestedToolResultMedia, } = providerSettingsRef.current - const needsProxy = apiFormat !== 'anthropic' + const needsProxy = providerNeedsProxy(apiFormat, supportsNestedToolResultMedia) const autoCompactWindowEnv = autoCompactWindow.trim() const modelContextWindows = buildModelContextWindows(models, modelContextInputs) const normalizedModels = normalizeModelMapping(models) @@ -1087,7 +1103,7 @@ function ProviderFormModal({ open, onClose, mode, provider, presets }: ProviderF ? { [MODEL_CONTEXT_WINDOWS_ENV_KEY]: JSON.stringify(modelContextWindows) } : {}), ANTHROPIC_BASE_URL: needsProxy ? providerProxyBaseUrl : baseUrl, - ...buildSettingsJsonAuthEnv(apiFormat, authStrategy, apiKey, selectedPreset), + ...buildSettingsJsonAuthEnv(needsProxy, authStrategy, apiKey, selectedPreset), ANTHROPIC_MODEL: runtimeModels.main, ...(runtimeModels.fable ? { ANTHROPIC_DEFAULT_FABLE_MODEL: runtimeModels.fable } : {}), ANTHROPIC_DEFAULT_HAIKU_MODEL: runtimeModels.haiku, @@ -1148,6 +1164,7 @@ function ProviderFormModal({ open, onClose, mode, provider, presets }: ProviderF setToolSearchEnabled(false) setToolSearchConfirmOpen(false) setDisableExperimentalBetas(false) + setSupportsNestedToolResultMedia(true) setShowContextSettings(false) setTestResult(null) } @@ -1238,6 +1255,10 @@ function ProviderFormModal({ open, onClose, mode, provider, presets }: ProviderF ] satisfies Array<{ value: ProviderAuthStrategy; label: string; description: string; icon: ReactNode }> const selectedAuthStrategyLabel = authStrategyItems.find((item) => item.value === authStrategy)?.label ?? t('settings.providers.authStrategyAuthToken') const toolSearchUnsupported = apiFormat !== 'anthropic' + const nestedToolResultMediaUnsupported = apiFormat !== 'anthropic' + const nestedToolResultMediaDescription = nestedToolResultMediaUnsupported + ? t('settings.providers.nestedToolResultMediaUnsupported') + : t('settings.providers.nestedToolResultMediaDesc') const toolSearchDescription = toolSearchUnsupported ? t('settings.providers.toolSearchUnsupported') : t('settings.providers.toolSearchDesc') @@ -1261,20 +1282,37 @@ function ProviderFormModal({ open, onClose, mode, provider, presets }: ProviderF } const handleBaseUrlChange = (value: string) => { setBaseUrl(value) - setSettingsJson((current) => updateSettingsJsonProviderConnection(current, apiFormat, authStrategy, apiKey, selectedPreset, value, providerProxyBaseUrl, toolSearchEnabled, disableExperimentalBetas)) + setSettingsJson((current) => updateSettingsJsonProviderConnection(current, apiFormat, authStrategy, apiKey, selectedPreset, value, providerProxyBaseUrl, toolSearchEnabled, disableExperimentalBetas, supportsNestedToolResultMedia)) } const handleApiKeyChange = (value: string) => { setApiKey(value) - setSettingsJson((current) => updateSettingsJsonProviderConnection(current, apiFormat, authStrategy, value, selectedPreset, baseUrl, providerProxyBaseUrl, toolSearchEnabled, disableExperimentalBetas)) + setSettingsJson((current) => updateSettingsJsonProviderConnection(current, apiFormat, authStrategy, value, selectedPreset, baseUrl, providerProxyBaseUrl, toolSearchEnabled, disableExperimentalBetas, supportsNestedToolResultMedia)) } const handleApiFormatChange = (value: ApiFormat) => { setApiFormat(value) - setSettingsJson((current) => updateSettingsJsonProviderConnection(current, value, authStrategy, apiKey, selectedPreset, baseUrl, providerProxyBaseUrl, toolSearchEnabled, disableExperimentalBetas)) + setSettingsJson((current) => updateSettingsJsonProviderConnection(current, value, authStrategy, apiKey, selectedPreset, baseUrl, providerProxyBaseUrl, toolSearchEnabled, disableExperimentalBetas, supportsNestedToolResultMedia)) } const handleAuthStrategyChange = (value: ProviderAuthStrategy) => { setAuthStrategy(value) - setSettingsJson((current) => updateSettingsJsonProviderConnection(current, apiFormat, value, apiKey, selectedPreset, baseUrl, providerProxyBaseUrl, toolSearchEnabled, disableExperimentalBetas)) + setSettingsJson((current) => updateSettingsJsonProviderConnection(current, apiFormat, value, apiKey, selectedPreset, baseUrl, providerProxyBaseUrl, toolSearchEnabled, disableExperimentalBetas, supportsNestedToolResultMedia)) } + const handleNestedToolResultMediaToggle = (enabled: boolean) => { + if (nestedToolResultMediaUnsupported) return + setSupportsNestedToolResultMedia(enabled) + setSettingsJson((current) => updateSettingsJsonProviderConnection( + current, + apiFormat, + authStrategy, + apiKey, + selectedPreset, + baseUrl, + providerProxyBaseUrl, + toolSearchEnabled, + disableExperimentalBetas, + enabled, + )) + } + const handleToolSearchToggle = (enabled: boolean) => { if (toolSearchUnsupported) return if (enabled) { @@ -1293,6 +1331,7 @@ function ProviderFormModal({ open, onClose, mode, provider, presets }: ProviderF setDisableExperimentalBetas(disabled) setSettingsJson((current) => updateSettingsJsonDisableExperimentalBetas(current, disabled)) } + const handleModelChange = (slot: ModelSlot, value: string) => { const hasMarker = hasModel1mMarker(value) const nextModels = { ...models, [slot]: stripModel1mMarker(value) } @@ -1457,6 +1496,7 @@ function ProviderFormModal({ open, onClose, mode, provider, presets }: ProviderF ...(Object.keys(parsedModelContextWindows).length > 0 && { modelContextWindows: parsedModelContextWindows }), toolSearchEnabled, ...(disableExperimentalBetas && { disableExperimentalBetas }), + supportsNestedToolResultMedia, ...(storedImageGeneration !== undefined && { imageGeneration: storedImageGeneration }), notes: notes.trim() || undefined, }) @@ -1474,6 +1514,7 @@ function ProviderFormModal({ open, onClose, mode, provider, presets }: ProviderF : null, toolSearchEnabled, disableExperimentalBetas, + supportsNestedToolResultMedia, imageGeneration: storedImageGeneration ?? null, notes: notes.trim() || undefined, } @@ -1503,7 +1544,8 @@ function ProviderFormModal({ open, onClose, mode, provider, presets }: ProviderF const savedConfigUnchanged = mode === 'edit' && provider && !apiKey.trim() && baseUrl.trim() === provider.baseUrl.trim() && apiFormat === provider.apiFormat && - authStrategy === provider.authStrategy + authStrategy === provider.authStrategy && + supportsNestedToolResultMedia === (provider.supportsNestedToolResultMedia ?? true) if (savedConfigUnchanged && provider) { result = await useProviderStore.getState().testProvider(provider.id, { modelId: models.main.trim(), @@ -1516,6 +1558,7 @@ function ProviderFormModal({ open, onClose, mode, provider, presets }: ProviderF modelId: models.main.trim(), authStrategy, apiFormat, + supportsNestedToolResultMedia, }) } setTestResult(result) @@ -1698,6 +1741,26 @@ function ProviderFormModal({ open, onClose, mode, provider, presets }: ProviderF + + handleNestedToolResultMediaToggle(e.target.checked)} + className={SETTINGS_CHECKBOX_INPUT_CLASS} + /> + + + + {t('settings.providers.supportsNestedToolResultMedia')} + + + {nestedToolResultMediaDescription} + + + + {t('settings.providers.apiKey')} diff --git a/desktop/src/types/provider.ts b/desktop/src/types/provider.ts index 7893f701..935dff4e 100644 --- a/desktop/src/types/provider.ts +++ b/desktop/src/types/provider.ts @@ -49,6 +49,7 @@ export type SavedProvider = { modelContextWindows?: ModelContextWindows toolSearchEnabled?: boolean disableExperimentalBetas?: boolean + supportsNestedToolResultMedia?: boolean imageGeneration?: ImageGenerationConfig notes?: string } @@ -67,6 +68,7 @@ export type CreateProviderInput = { modelContextWindows?: ModelContextWindows toolSearchEnabled?: boolean disableExperimentalBetas?: boolean + supportsNestedToolResultMedia?: boolean imageGeneration?: ImageGenerationConfig notes?: string } @@ -84,6 +86,7 @@ export type UpdateProviderInput = { modelContextWindows?: ModelContextWindows | null toolSearchEnabled?: boolean disableExperimentalBetas?: boolean + supportsNestedToolResultMedia?: boolean imageGeneration?: ImageGenerationConfig | null notes?: string } @@ -94,6 +97,7 @@ export type TestProviderConfigInput = { modelId: string authStrategy?: ProviderAuthStrategy apiFormat?: ApiFormat + supportsNestedToolResultMedia?: boolean } export type ProviderTestStepResult = { @@ -107,7 +111,7 @@ export type ProviderTestStepResult = { export type ProviderTestResult = { /** Step 1: Basic connectivity */ connectivity: ProviderTestStepResult - /** Step 2: Proxy pipeline (only for openai_* formats) */ + /** Step 2: Proxy pipeline when the provider requires local request handling */ proxy?: ProviderTestStepResult } diff --git a/package.json b/package.json index cd1d42b1..57bbd7fd 100644 --- a/package.json +++ b/package.json @@ -98,6 +98,7 @@ "react": "^19.2.4", "react-reconciler": "^0.33.0", "semver": "^7.7.4", + "sharp": "0.34.5", "shell-quote": "^1.8.3", "signal-exit": "^4.1.0", "stack-utils": "^2.0.6", diff --git a/src/server/__tests__/anthropicMediaHoist.test.ts b/src/server/__tests__/anthropicMediaHoist.test.ts new file mode 100644 index 00000000..a9a04d6a --- /dev/null +++ b/src/server/__tests__/anthropicMediaHoist.test.ts @@ -0,0 +1,726 @@ +import { describe, expect, test } from 'bun:test' +import type { AnthropicRequest } from '../proxy/transform/types.js' +import { hoistToolResultMediaForCompatibility } from '../proxy/transform/anthropicMediaHoist.js' + +function makeRequest(messages: AnthropicRequest['messages']): AnthropicRequest { + return { model: 'test-model', max_tokens: 100, messages } +} + +describe('hoistToolResultMediaForCompatibility', () => { + test('lifts nested images to the end of the user message and keeps tool results contiguous', () => { + const body = makeRequest([{ + role: 'user', + content: [ + { + type: 'tool_result', + tool_use_id: 'tool-a', + content: [ + { type: 'text', text: 'before' }, + { type: 'image', source: { type: 'base64', media_type: 'image/png', data: 'a' } }, + { type: 'text', text: 'after' }, + ], + }, + { + type: 'tool_result', + tool_use_id: 'tool-b', + content: [{ type: 'image', source: { type: 'base64', media_type: 'image/jpeg', data: 'b' } }], + }, + ], + }]) + + const result = hoistToolResultMediaForCompatibility(body) + + expect(result.messages[0]).toEqual({ + role: 'user', + content: [ + { + type: 'tool_result', + tool_use_id: 'tool-a', + content: [ + { type: 'text', text: 'before' }, + { type: 'text', text: 'after' }, + ], + }, + { + type: 'tool_result', + tool_use_id: 'tool-b', + content: [{ type: 'text', text: 'Media result attached after this tool result.' }], + }, + { type: 'text', text: '[Image content for tool call tool-a]' }, + { type: 'image', source: { type: 'base64', media_type: 'image/png', data: 'a' } }, + { type: 'text', text: '[Image content for tool call tool-b]' }, + { type: 'image', source: { type: 'base64', media_type: 'image/jpeg', data: 'b' } }, + ], + }) + }) + + test('lifts documents alongside images', () => { + const body = makeRequest([{ + role: 'user', + content: [ + { + type: 'tool_result', + tool_use_id: 'tool-doc', + content: [ + { type: 'document', title: 'report.pdf', source: { type: 'url', url: 'https://example.test/report.pdf' } }, + ], + }, + ], + }]) + + const result = hoistToolResultMediaForCompatibility(body) + + expect(result.messages[0]).toEqual({ + role: 'user', + content: [ + { type: 'tool_result', tool_use_id: 'tool-doc', content: [{ type: 'text', text: 'Media result attached after this tool result.' }] }, + { type: 'text', text: '[Document content for tool call tool-doc]' }, + { type: 'document', title: 'report.pdf', source: { type: 'url', url: 'https://example.test/report.pdf' } }, + ], + }) + }) + + test('keeps lifted media ahead of trailing user text', () => { + const body = makeRequest([{ + role: 'user', + content: [ + { + type: 'tool_result', + tool_use_id: 'tool-shot', + content: [{ type: 'image', source: { type: 'base64', media_type: 'image/png', data: 'a' } }], + }, + { type: 'text', text: 'Please look at the top-left corner' }, + ], + }]) + + const result = hoistToolResultMediaForCompatibility(body) + + expect(result.messages[0]).toEqual({ + role: 'user', + content: [ + { type: 'tool_result', tool_use_id: 'tool-shot', content: [{ type: 'text', text: 'Media result attached after this tool result.' }] }, + { type: 'text', text: '[Image content for tool call tool-shot]' }, + { type: 'image', source: { type: 'base64', media_type: 'image/png', data: 'a' } }, + { type: 'text', text: 'Please look at the top-left corner' }, + ], + }) + }) + + test('preserves references for later messages that need no transform', () => { + const body = makeRequest([ + { + role: 'user', + content: [{ + type: 'tool_result', + tool_use_id: 'tool-image', + content: [{ type: 'image', source: { type: 'base64', media_type: 'image/png', data: 'a' } }], + }], + }, + { role: 'assistant', content: [{ type: 'text', text: 'done' }] }, + { role: 'user', content: [{ type: 'text', text: 'unchanged' }] }, + ]) + + const result = hoistToolResultMediaForCompatibility(body) + + expect(result.messages[2]).toBe(body.messages[2]) + }) + + test('describes mixed images and documents in the media marker', () => { + const body = makeRequest([{ + role: 'user', + content: [{ + type: 'tool_result', + tool_use_id: 'tool-mixed-media', + content: [ + { type: 'image', source: { type: 'base64', media_type: 'image/png', data: 'a' } }, + { type: 'document', source: { type: 'url', url: 'https://example.test/report.pdf' } }, + ], + }], + }]) + + const result = hoistToolResultMediaForCompatibility(body) + const content = result.messages[0]?.content + + expect(Array.isArray(content) ? content[1] : undefined).toEqual({ + type: 'text', + text: '[Media content for tool call tool-mixed-media: 1 image, 1 document]', + }) + }) + + test('leaves string tool results and media-free results untouched', () => { + const body = makeRequest([{ + role: 'user', + content: [ + { type: 'tool_result', tool_use_id: 'tool-text', content: 'plain result' }, + { + type: 'tool_result', + tool_use_id: 'tool-mixed', + content: [{ type: 'text', text: 'text only' }], + }, + ], + }]) + + const result = hoistToolResultMediaForCompatibility(body) + + expect(result).toBe(body) + expect(result.messages).toEqual(body.messages) + }) + + test('leaves assistant messages and plain text user messages untouched', () => { + const body = makeRequest([ + { role: 'user', content: 'hello' }, + { + role: 'assistant', + content: [{ type: 'text', text: 'hi' }], + }, + ]) + + expect(hoistToolResultMediaForCompatibility(body)).toBe(body) + }) + + test('skips hoisting when the last assistant turn has an unresolved server tool call', () => { + const body = makeRequest([ + { + role: 'assistant', + content: [ + { type: 'text', text: 'searching' }, + { type: 'server_tool_use', id: 'st_1', name: 'web_search', input: { query: 'x' } }, + ], + }, + { + role: 'user', + content: [{ + type: 'tool_result', + tool_use_id: 'tool-a', + content: [{ type: 'image', source: { type: 'base64', media_type: 'image/png', data: 'a' } }], + }], + }, + ]) + + expect(hoistToolResultMediaForCompatibility(body)).toBe(body) + }) + + test('hoists when a server tool call already has its result in the same turn', () => { + const body = makeRequest([ + { + role: 'assistant', + content: [ + { type: 'server_tool_use', id: 'st_1', name: 'web_search', input: { query: 'x' } }, + { type: 'web_search_tool_result', tool_use_id: 'st_1', content: { type: 'web_search_result', query: 'x' } }, + ], + }, + { + role: 'user', + content: [{ + type: 'tool_result', + tool_use_id: 'tool-a', + content: [{ type: 'image', source: { type: 'base64', media_type: 'image/png', data: 'a' } }], + }], + }, + ] as unknown as AnthropicRequest['messages']) + + const result = hoistToolResultMediaForCompatibility(body) + expect(result).not.toBe(body) + expect(result.messages[1]).not.toEqual(body.messages[1]) + }) + + test('hoists when a deferred server tool result arrived in the next assistant turn', () => { + // Mixed server/client execution: the API returns the server_tool_use + // without a result, the client returns only its own tool_result, and the + // server tool result leads the next assistant response. + const body = makeRequest([ + { + role: 'assistant', + content: [ + { type: 'server_tool_use', id: 'st_1', name: 'web_fetch', input: { url: 'https://example.test' } }, + { type: 'tool_use', id: 't_1', name: 'run_command', input: { command: 'uname' } }, + ], + }, + { + role: 'user', + content: [{ type: 'tool_result', tool_use_id: 't_1', content: 'Linux' }], + }, + { + role: 'assistant', + content: [ + { type: 'web_fetch_tool_result', tool_use_id: 'st_1', content: { type: 'web_fetch_result', url: 'https://example.test' } }, + { type: 'text', text: 'fetched' }, + ], + }, + { + role: 'user', + content: [{ + type: 'tool_result', + tool_use_id: 'tool-a', + content: [{ type: 'image', source: { type: 'base64', media_type: 'image/png', data: 'a' } }], + }], + }, + ] as unknown as AnthropicRequest['messages']) + + const result = hoistToolResultMediaForCompatibility(body) + expect(result).not.toBe(body) + expect(result.messages[3]).not.toEqual(body.messages[3]) + }) + + test('lifts history media while leaving the unresolved continuation message untouched', () => { + // A completed earlier turn with tool-result media stays convertible even + // when the current turn continues an unresolved server tool — only the + // continuation user message may not gain non-tool_result blocks. + const body = makeRequest([ + { role: 'user', content: 'check this' }, + { role: 'assistant', content: [{ type: 'tool_use', id: 't_1', name: 'run_command', input: { command: 'ls' } }] }, + { + role: 'user', + content: [{ + type: 'tool_result', + tool_use_id: 't_1', + content: [{ type: 'image', source: { type: 'base64', media_type: 'image/png', data: 'a' } }], + }], + }, + { + role: 'assistant', + content: [ + { type: 'server_tool_use', id: 'st_1', name: 'web_search', input: { query: 'x' } }, + { type: 'tool_use', id: 't_2', name: 'run_command', input: { command: 'uname' } }, + ], + }, + { + role: 'user', + content: [{ + type: 'tool_result', + tool_use_id: 't_2', + content: [{ type: 'image', source: { type: 'base64', media_type: 'image/png', data: 'b' } }], + }], + }, + ] as unknown as AnthropicRequest['messages']) + + const result = hoistToolResultMediaForCompatibility(body) + expect(result).not.toBe(body) + // History turn: media lifted out of the tool_result. + expect(result.messages[2]).not.toEqual(body.messages[2]) + expect(result.messages[2]).toEqual({ + role: 'user', + content: [ + { type: 'tool_result', tool_use_id: 't_1', content: [{ type: 'text', text: 'Media result attached after this tool result.' }] }, + { type: 'text', text: '[Image content for tool call t_1]' }, + { type: 'image', source: { type: 'base64', media_type: 'image/png', data: 'a' } }, + ], + }) + // Continuation turn: left as-is. + expect(result.messages[4]).toEqual(body.messages[4]) + }) + + test('skips hoisting when the last assistant turn has an unresolved mcp_tool_use', () => { + const body = makeRequest([ + { + role: 'assistant', + content: [ + { type: 'mcp_tool_use', id: 'mcp_1', name: 'slack', input: { action: 'list' } }, + { type: 'tool_use', id: 't_1', name: 'run_command', input: { command: 'ls' } }, + ], + }, + { + role: 'user', + content: [{ + type: 'tool_result', + tool_use_id: 't_1', + content: [{ type: 'image', source: { type: 'base64', media_type: 'image/png', data: 'a' } }], + }], + }, + ] as unknown as AnthropicRequest['messages']) + + expect(hoistToolResultMediaForCompatibility(body)).toBe(body) + }) + + test('hoists when an mcp_tool_use already has its result in the same turn', () => { + const body = makeRequest([ + { + role: 'assistant', + content: [ + { type: 'mcp_tool_use', id: 'mcp_1', name: 'slack', input: { action: 'list' } }, + { type: 'mcp_tool_result', tool_use_id: 'mcp_1', content: { type: 'mcp_result', data: 'ok' } }, + ], + }, + { + role: 'user', + content: [{ + type: 'tool_result', + tool_use_id: 'tool-a', + content: [{ type: 'image', source: { type: 'base64', media_type: 'image/png', data: 'a' } }], + }], + }, + ] as unknown as AnthropicRequest['messages']) + + const result = hoistToolResultMediaForCompatibility(body) + expect(result).not.toBe(body) + expect(result.messages[1]).not.toEqual(body.messages[1]) + }) + + test('keeps plain-text documents inside the tool result as text', () => { + const body = makeRequest([{ + role: 'user', + content: [ + { + type: 'tool_result', + tool_use_id: 'tool-doc', + content: [ + { type: 'document', title: 'notes.txt', source: { type: 'text', media_type: 'text/plain', data: 'actual tool result' } }, + { type: 'image', source: { type: 'base64', media_type: 'image/png', data: 'a' } }, + ], + }, + ], + }]) + + const result = hoistToolResultMediaForCompatibility(body) + + // The plain-text document stays inside the tool result as text (its + // provenance and title preserved); only the image lifts. + expect(result.messages[0]).toEqual({ + role: 'user', + content: [ + { + type: 'tool_result', + tool_use_id: 'tool-doc', + content: [{ type: 'text', text: '[Document: notes.txt]\nactual tool result' }], + }, + { type: 'text', text: '[Image content for tool call tool-doc]' }, + { type: 'image', source: { type: 'base64', media_type: 'image/png', data: 'a' } }, + ], + }) + }) + + test('degrades a text-only document without lifting any media', () => { + const body = makeRequest([{ + role: 'user', + content: [ + { + type: 'tool_result', + tool_use_id: 'tool-text-doc', + content: [ + { type: 'document', title: 'notes.txt', source: { type: 'text', media_type: 'text/plain', data: 'result text' } }, + ], + }, + ], + }]) + + const result = hoistToolResultMediaForCompatibility(body) + + expect(result.messages[0]).toEqual({ + role: 'user', + content: [ + { + type: 'tool_result', + tool_use_id: 'tool-text-doc', + content: [{ type: 'text', text: '[Document: notes.txt]\nresult text' }], + }, + ], + }) + // No user-level media blocks were added. + expect(result.messages[0].content).toHaveLength(1) + }) + + test('keeps text-only custom-content documents inside the tool result', () => { + const body = makeRequest([ + { + role: 'user', + content: [ + { + type: 'tool_result', + tool_use_id: 'tool-cdoc-string', + content: [ + { type: 'document', title: 'raw', source: { type: 'content', content: 'plain tool output' } }, + ], + }, + ], + }, + { + role: 'user', + content: [ + { + type: 'tool_result', + tool_use_id: 'tool-cdoc-blocks', + content: [ + { + type: 'document', + title: 'cited', + source: { + type: 'content', + content: [ + { type: 'text', text: 'Bearer ' }, + { type: 'text', text: 'abc123' }, + ], + }, + }, + ], + }, + ], + }, + ]) + + const result = hoistToolResultMediaForCompatibility(body) + + expect(result.messages[0]).toEqual({ + role: 'user', + content: [ + { + type: 'tool_result', + tool_use_id: 'tool-cdoc-string', + content: [ + { type: 'text', text: '[Document: raw]' }, + { type: 'text', text: 'plain tool output' }, + ], + }, + ], + }) + expect(result.messages[1]).toEqual({ + role: 'user', + content: [ + { + type: 'tool_result', + tool_use_id: 'tool-cdoc-blocks', + content: [ + { type: 'text', text: '[Document: cited]' }, + // Original text block boundaries are preserved — no separators + // are injected into the tool output. + { type: 'text', text: 'Bearer ' }, + { type: 'text', text: 'abc123' }, + ], + }, + ], + }) + // No user-level media blocks were added. + expect(result.messages[0].content).toHaveLength(1) + expect(result.messages[1].content).toHaveLength(1) + }) + + test('preserves cache_control when degrading plain-text documents', () => { + const body = makeRequest([{ + role: 'user', + content: [ + { + type: 'tool_result', + tool_use_id: 'tool-cache', + content: [ + { type: 'document', title: 'cached.txt', source: { type: 'text', media_type: 'text/plain', data: 'large cached output' }, cache_control: { type: 'ephemeral' } }, + { type: 'document', title: 'cited', source: { type: 'content', content: 'quoted' }, cache_control: { type: 'ephemeral' } }, + ], + }, + ], + }]) + + const result = hoistToolResultMediaForCompatibility(body) + + expect(result.messages[0]).toEqual({ + role: 'user', + content: [ + { + type: 'tool_result', + tool_use_id: 'tool-cache', + content: [ + { type: 'text', text: '[Document: cached.txt]\nlarge cached output', cache_control: { type: 'ephemeral' } }, + // The document-level breakpoint sits on the *last* degraded block + // (the end of the cached prefix), not on the synthetic title. + { type: 'text', text: '[Document: cited]' }, + { type: 'text', text: 'quoted', cache_control: { type: 'ephemeral' } }, + ], + }, + ], + }) + }) + + test('keeps model-visible title and context when degrading documents', () => { + const body = makeRequest([{ + role: 'user', + content: [ + { + type: 'tool_result', + tool_use_id: 'tool-prov', + content: [ + { + type: 'document', + title: 'Auth specification', + context: 'The examples use production credentials', + source: { type: 'text', media_type: 'text/plain', data: 'Bearer abc123' }, + }, + { + type: 'document', + title: 'Policy', + context: 'Applies to tenant A', + source: { type: 'content', content: [{ type: 'text', text: 'quoted' }] }, + }, + ], + }, + ], + }] as unknown as Parameters[0]) + + const result = hoistToolResultMediaForCompatibility(body) + + expect(result.messages[0]).toEqual({ + role: 'user', + content: [ + { + type: 'tool_result', + tool_use_id: 'tool-prov', + content: [ + { type: 'text', text: '[Document: Auth specification]\n[Document context: The examples use production credentials]\nBearer abc123' }, + { type: 'text', text: '[Document: Policy]' }, + { type: 'text', text: '[Document context: Applies to tenant A]' }, + { type: 'text', text: 'quoted' }, + ], + }, + ], + }) + }) + + test('keeps nested cache_control and citations when degrading custom-content documents', () => { + const body = makeRequest([{ + role: 'user', + content: [ + { + type: 'tool_result', + tool_use_id: 'tool-nested', + content: [ + { + type: 'document', + title: 'spec', + source: { + type: 'content', + content: [ + { type: 'text', text: 'cached chunk', cache_control: { type: 'ephemeral' } }, + { type: 'text', text: 'cited chunk', citations: [{ type: 'char_location', cited_text: 'x' }] }, + ], + }, + }, + ], + }, + ], + }] as unknown as Parameters[0]) + + const result = hoistToolResultMediaForCompatibility(body) + + expect(result.messages[0]).toEqual({ + role: 'user', + content: [ + { + type: 'tool_result', + tool_use_id: 'tool-nested', + content: [ + { type: 'text', text: '[Document: spec]' }, + { type: 'text', text: 'cached chunk', cache_control: { type: 'ephemeral' } }, + { type: 'text', text: 'cited chunk', citations: [{ type: 'char_location', cited_text: 'x' }] }, + ], + }, + ], + }) + }) + + test('does not overwrite an inner cache_control with the document-level breakpoint', () => { + const body = makeRequest([{ + role: 'user', + content: [ + { + type: 'tool_result', + tool_use_id: 'tool-outer', + content: [ + { + type: 'document', + title: 'mixed', + source: { + type: 'content', + content: [ + { type: 'text', text: 'a' }, + // The last degraded block already carries an inner marker — + // the document-level breakpoint must not clobber it. + { type: 'text', text: 'b', cache_control: { type: 'ephemeral' } }, + ], + }, + cache_control: { type: 'ephemeral' }, + }, + ], + }, + ], + }] as unknown as Parameters[0]) + + const result = hoistToolResultMediaForCompatibility(body) + + expect(result.messages[0]).toEqual({ + role: 'user', + content: [ + { + type: 'tool_result', + tool_use_id: 'tool-outer', + content: [ + { type: 'text', text: '[Document: mixed]' }, + { type: 'text', text: 'a' }, + { type: 'text', text: 'b', cache_control: { type: 'ephemeral' } }, + ], + }, + ], + }) + }) + + test('lifts custom-content documents that carry inline images', () => { + const body = makeRequest([{ + role: 'user', + content: [ + { + type: 'tool_result', + tool_use_id: 'tool-cdoc-img', + content: [ + { + type: 'document', + title: 'cited', + source: { + type: 'content', + content: [ + { type: 'text', text: 'before' }, + { type: 'image', source: { type: 'base64', media_type: 'image/png', data: 'a' } }, + ], + }, + }, + ], + }, + ], + }]) + + const result = hoistToolResultMediaForCompatibility(body) + + expect(result.messages[0]).toEqual({ + role: 'user', + content: [ + { type: 'tool_result', tool_use_id: 'tool-cdoc-img', content: [{ type: 'text', text: 'Media result attached after this tool result.' }] }, + { type: 'text', text: '[Document content for tool call tool-cdoc-img]' }, + { + type: 'document', + title: 'cited', + source: { + type: 'content', + content: [ + { type: 'text', text: 'before' }, + { type: 'image', source: { type: 'base64', media_type: 'image/png', data: 'a' } }, + ], + }, + }, + ], + }) + }) + + test('hoists when typed client tools (bash, text_editor) are declared but no server call is pending', () => { + const body = makeRequest([{ + role: 'user', + content: [{ + type: 'tool_result', + tool_use_id: 'tool-a', + content: [{ type: 'image', source: { type: 'base64', media_type: 'image/png', data: 'a' } }], + }], + }]) as AnthropicRequest & { tools: Array> } + body.tools = [ + { name: 'bash', type: 'bash_20250124', input_schema: { type: 'object' } }, + { name: 'str_replace_editor', type: 'text_editor_20250728', input_schema: { type: 'object' } }, + ] + + const result = hoistToolResultMediaForCompatibility(body) + expect(result).not.toBe(body) + expect(result.messages[0]).not.toEqual(body.messages[0]) + }) +}) diff --git a/src/server/__tests__/providers.test.ts b/src/server/__tests__/providers.test.ts index 68fc871a..94b31185 100644 --- a/src/server/__tests__/providers.test.ts +++ b/src/server/__tests__/providers.test.ts @@ -11,6 +11,7 @@ import { handleProvidersApi } from '../api/providers.js' import { handleProxyRequest } from '../proxy/handler.js' import { clearTraceCaptureStateForTests, + drainTraceCaptureForTests, setTraceAppendBeforeWriteHookForTests, traceCaptureService, } from '../services/traceCaptureService.js' @@ -32,6 +33,7 @@ async function setup() { } async function teardown() { + await drainTraceCaptureForTests() clearTraceCaptureStateForTests() if (originalConfigDir !== undefined) { process.env.CLAUDE_CONFIG_DIR = originalConfigDir @@ -43,6 +45,17 @@ async function teardown() { } else { delete process.env.HOME } + // The background trace projection may still hold a handle briefly (first + // index builds are slower); retry the removal instead of failing the test + // on Windows. + for (let attempt = 0; attempt < 60; attempt++) { + try { + await fs.rm(tmpDir, { recursive: true, force: true }) + return + } catch { + await new Promise(resolve => setTimeout(resolve, 25)) + } + } await fs.rm(tmpDir, { recursive: true, force: true }) } @@ -838,6 +851,34 @@ describe('ProviderService', () => { expect(env.ANTHROPIC_MODEL).toBe('model-main') }) + test('editing an existing provider persists supportsNestedToolResultMedia and reroutes it through the proxy', async () => { + const svc = new ProviderService() + const added = await svc.addProvider(sampleInput()) + await svc.activateProvider(added.id) + + // Default: nested media preserved, direct connection. + let settings = await readSettings() + let env = settings.env as Record + expect(env.ANTHROPIC_BASE_URL).toBe('https://api.example.com') + + const updated = await svc.updateProvider(added.id, { supportsNestedToolResultMedia: false }) + + expect(updated.supportsNestedToolResultMedia).toBe(false) + + settings = await readSettings() + env = settings.env as Record + expect(env.ANTHROPIC_BASE_URL).toContain('127.0.0.1') + expect(env.ANTHROPIC_API_KEY).toBe('proxy-managed') + + // Editing back to nested media restores the direct connection. + const reverted = await svc.updateProvider(added.id, { supportsNestedToolResultMedia: true }) + expect(reverted.supportsNestedToolResultMedia).toBe(true) + + settings = await readSettings() + env = settings.env as Record + expect(env.ANTHROPIC_BASE_URL).toBe('https://api.example.com') + }) + test('updating active provider should override and clear auto compact window', async () => { const svc = new ProviderService() const added = await svc.addProvider(sampleInput({ autoCompactWindow: 64000 })) @@ -1527,6 +1568,21 @@ describe('ProviderService', () => { expect(active!.apiFormat).toBe('anthropic') }) + test('should resolve preset default auth for a no-key proxy provider', async () => { + const svc = new ProviderService() + const provider = await svc.addProvider(sampleInput({ + presetId: 'lmstudio', + apiKey: '', + apiFormat: 'anthropic', + supportsNestedToolResultMedia: false, + })) + + const config = await svc.getProviderForProxy(provider.id) + + expect(config?.apiKey).toBe('lmstudio') + expect(config?.authStrategy).toBe('auth_token_empty_api_key') + }) + test('should return null when ChatGPT Official is the active provider', async () => { const svc = new ProviderService() await svc.activateProvider('openai-official') @@ -2370,6 +2426,40 @@ describe('ProviderService', () => { } }) + test('tests the proxy path for Anthropic providers that require media hoisting', async () => { + const originalFetch = globalThis.fetch + const calls: Array<{ body: Record }> = [] + globalThis.fetch = mock(async (_url: string | URL | Request, init?: RequestInit) => { + calls.push({ body: JSON.parse(String(init?.body)) as Record }) + return new Response(JSON.stringify({ + type: 'message', + model: 'model-main', + content: [{ type: 'text', text: 'ok' }], + }), { + status: 200, + headers: { 'Content-Type': 'application/json' }, + }) + }) as typeof fetch + + try { + const svc = new ProviderService() + const result = await svc.testProviderConfig({ + baseUrl: 'https://api.example.com/anthropic', + apiKey: 'sk-api', + modelId: 'model-main', + authStrategy: 'api_key', + apiFormat: 'anthropic', + supportsNestedToolResultMedia: false, + }) + + expect(result.connectivity.success).toBe(true) + expect(result.proxy?.success).toBe(true) + expect(calls).toHaveLength(2) + } finally { + globalThis.fetch = originalFetch + } + }) + test('normalizes context-window suffixes for Anthropic-compatible connectivity tests', async () => { const originalFetch = globalThis.fetch const calls: Array<{ body: Record }> = [] diff --git a/src/server/__tests__/proxy-anthropic-compat.test.ts b/src/server/__tests__/proxy-anthropic-compat.test.ts new file mode 100644 index 00000000..b8c46846 --- /dev/null +++ b/src/server/__tests__/proxy-anthropic-compat.test.ts @@ -0,0 +1,1556 @@ +import { afterEach, beforeEach, describe, expect, mock, test } from 'bun:test' +import * as fs from 'fs/promises' +import * as os from 'os' +import * as path from 'path' +import { handleProxyRequest } from '../proxy/handler.js' +import { ProviderService } from '../services/providerService.js' +import { resetSettingsCache } from '../../utils/settings/settingsCache.js' +import { clearTraceCaptureStateForTests, drainTraceCaptureForTests } from '../../services/api/traceCapture.js' + +let tmpDir: string +let originalConfigDir: string | undefined + +async function setup() { + tmpDir = await fs.mkdtemp(path.join(os.tmpdir(), 'proxy-anthropic-test-')) + originalConfigDir = process.env.CLAUDE_CONFIG_DIR + process.env.CLAUDE_CONFIG_DIR = tmpDir + resetSettingsCache() +} + +async function teardown() { + if (originalConfigDir !== undefined) { + process.env.CLAUDE_CONFIG_DIR = originalConfigDir + } else { + delete process.env.CLAUDE_CONFIG_DIR + } + resetSettingsCache() + // Wait for in-flight appends/projections, then close cached trace index + // handles so the temp dir is not locked on Windows. + await drainTraceCaptureForTests() + clearTraceCaptureStateForTests() + // The background trace projection may still hold a handle briefly; retry + // the removal instead of failing the test on Windows. + for (let attempt = 0; attempt < 10; attempt++) { + try { + await fs.rm(tmpDir, { recursive: true, force: true }) + return + } catch (err) { + if ((err as { code?: string }).code !== 'EBUSY') throw err + await new Promise(resolve => setTimeout(resolve, 50)) + } + } +} + +describe('proxy anthropic-compatible path', () => { + beforeEach(setup) + afterEach(teardown) + + test('hoists nested media, sends auth headers, and forwards protocol headers', async () => { + const svc = new ProviderService() + const provider = await svc.addProvider({ + presetId: 'custom', + name: 'Anthropic Compat', + baseUrl: 'https://relay.example.com', + apiKey: 'sk-relay', + apiFormat: 'anthropic', + authStrategy: 'api_key', + supportsNestedToolResultMedia: false, + models: { + main: 'model-main', + haiku: 'model-main', + sonnet: 'model-main', + opus: 'model-main', + }, + }) + + const originalFetch = globalThis.fetch + const captured: Array<{ url: string; init: RequestInit }> = [] + globalThis.fetch = mock(async (url: string | URL | Request, init?: RequestInit) => { + captured.push({ url: String(url), init: init ?? {} }) + return new Response(JSON.stringify({ + id: 'msg_relay', + type: 'message', + role: 'assistant', + model: 'model-main', + content: [{ type: 'text', text: 'ok' }], + stop_reason: 'end_turn', + usage: { input_tokens: 1, output_tokens: 1 }, + }), { + status: 200, + headers: { 'Content-Type': 'application/json' }, + }) + }) as typeof fetch + + try { + const body = { + model: 'model-main', + max_tokens: 64, + messages: [{ + role: 'user', + content: [{ + type: 'tool_result', + tool_use_id: 'tc_1', + content: [{ type: 'image', source: { type: 'base64', media_type: 'image/png', data: 'abc' } }], + }], + }], + } + const req = new Request( + `http://localhost:3456/proxy/providers/${provider.id}/v1/messages`, + { + method: 'POST', + headers: { + 'Content-Type': 'application/json', + 'anthropic-version': '2023-06-01', + 'anthropic-beta': 'some-beta-header', + }, + body: JSON.stringify(body), + }, + ) + + const res = await handleProxyRequest(req, new URL(req.url)) + expect(res.status).toBe(200) + + expect(captured).toHaveLength(1) + const upstream = captured[0]! + expect(upstream.url).toBe('https://relay.example.com/v1/messages') + + const headers = upstream.init.headers as Record + expect(headers['x-api-key']).toBe('sk-relay') + expect(headers['anthropic-version']).toBe('2023-06-01') + expect(headers['anthropic-beta']).toBe('some-beta-header') + expect(headers.Authorization).toBeUndefined() + + const sentBody = JSON.parse(String(upstream.init.body)) as { + messages: Array<{ content: Array<{ type: string }> }> + } + const types = sentBody.messages[0]!.content.map(block => block.type) + expect(types).toEqual(['tool_result', 'text', 'image']) + } finally { + globalThis.fetch = originalFetch + } + }) + + test('forwards custom headers and drops empty-key auth headers', async () => { + const svc = new ProviderService() + const provider = await svc.addProvider({ + presetId: 'custom', + name: 'Anthropic Compat', + baseUrl: 'https://relay.example.com', + apiKey: '', + apiFormat: 'anthropic', + authStrategy: 'api_key', + supportsNestedToolResultMedia: false, + models: { + main: 'model-main', + haiku: 'model-main', + sonnet: 'model-main', + opus: 'model-main', + }, + }) + + const originalFetch = globalThis.fetch + const captured: Array<{ init: RequestInit }> = [] + globalThis.fetch = mock(async (_url: string | URL | Request, init?: RequestInit) => { + captured.push({ init: init ?? {} }) + return new Response(JSON.stringify({ + id: 'msg_relay', + type: 'message', + role: 'assistant', + model: 'model-main', + content: [{ type: 'text', text: 'ok' }], + stop_reason: 'end_turn', + usage: { input_tokens: 1, output_tokens: 1 }, + }), { status: 200, headers: { 'Content-Type': 'application/json' } }) + }) as typeof fetch + + try { + const req = new Request( + `http://localhost:3456/proxy/providers/${provider.id}/v1/messages`, + { + method: 'POST', + headers: { + 'Content-Type': 'application/json', + 'anthropic-version': '2023-06-01', + 'x-gateway-token': 'gw-42', + 'x-tenant-id': 'tenant-a', + 'x-app': 'cli', + 'x-claude-code-session-id': 'session-secret', + 'x-claude-remote-container-id': 'container-secret', + 'x-claude-remote-session-id': 'remote-secret', + 'x-client-app': 'sdk-client', + 'User-Agent': 'claude-cli/2.1.220.693 (external, sdk, client-app/example/1.0)', + connection: 'x-internal-hop', + 'x-internal-hop': 'secret', + }, + body: JSON.stringify({ + model: 'model-main', + max_tokens: 64, + messages: [{ role: 'user', content: 'hi' }], + }), + }, + ) + + const res = await handleProxyRequest(req, new URL(req.url)) + expect(res.status).toBe(200) + + const headers = captured[0]!.init.headers as Record + expect(headers['x-gateway-token']).toBe('gw-42') + expect(headers['x-tenant-id']).toBe('tenant-a') + expect(headers['x-app']).toBeUndefined() + expect(headers['x-claude-code-session-id']).toBeUndefined() + expect(headers['x-claude-remote-container-id']).toBeUndefined() + expect(headers['x-claude-remote-session-id']).toBeUndefined() + expect(headers['x-client-app']).toBeUndefined() + expect(headers['user-agent']).toBeUndefined() + // Empty key sends no auth header at all, and hop-by-hop headers — both + // the fixed set and the ones named by `Connection` — do not leak to the + // upstream. + expect(headers['x-api-key']).toBeUndefined() + expect(headers.Authorization).toBeUndefined() + expect(headers.connection).toBeUndefined() + expect(headers['x-internal-hop']).toBeUndefined() + } finally { + globalThis.fetch = originalFetch + } + }) + + test('redacts forwarded custom header values in trace capture', async () => { + const svc = new ProviderService() + const provider = await svc.addProvider({ + presetId: 'custom', + name: 'Anthropic Compat', + baseUrl: 'https://relay.example.com', + apiKey: 'sk-relay', + apiFormat: 'anthropic', + authStrategy: 'api_key', + supportsNestedToolResultMedia: false, + models: { + main: 'model-main', + haiku: 'model-main', + sonnet: 'model-main', + opus: 'model-main', + }, + }) + + const originalFetch = globalThis.fetch + const captured: Headers[] = [] + globalThis.fetch = mock(async (_url: string | URL | Request, init?: RequestInit) => { + captured.push(new Headers(init?.headers)) + return new Response(JSON.stringify({ + type: 'message', + role: 'assistant', + model: 'model-main', + content: [{ type: 'text', text: 'ok' }], + stop_reason: 'end_turn', + usage: { input_tokens: 1, output_tokens: 1 }, + }), { status: 200, headers: { 'Content-Type': 'application/json' } }) + }) as typeof fetch + + try { + const sessionId = 'session-custom-header-trace' + const req = new Request( + `http://localhost:3456/proxy/providers/${provider.id}/v1/messages`, + { + method: 'POST', + headers: { + 'Content-Type': 'application/json', + 'anthropic-version': '2023-06-01', + 'x-claude-code-session-id': sessionId, + 'x-relay-credential': 'opaque-value-42', + 'anthropic-relay-credential': 'opaque-value-43', + }, + body: JSON.stringify({ + model: 'model-main', + max_tokens: 64, + messages: [{ role: 'user', content: 'hi' }], + }), + }, + ) + + const res = await handleProxyRequest(req, new URL(req.url)) + expect(res.status).toBe(200) + expect(captured[0]?.get('x-relay-credential')).toBe('opaque-value-42') + expect(captured[0]?.get('anthropic-relay-credential')).toBe('opaque-value-43') + + await drainTraceCaptureForTests() + const tracePath = path.join(tmpDir, 'cc-haha', 'traces', `${sessionId}.jsonl`) + const traceRaw = await fs.readFile(tracePath, 'utf-8') + expect(traceRaw).toContain('x-relay-credential') + expect(traceRaw).toContain('anthropic-relay-credential') + expect(traceRaw).toContain('[redacted]') + expect(traceRaw).not.toContain('opaque-value-42') + expect(traceRaw).not.toContain('opaque-value-43') + } finally { + globalThis.fetch = originalFetch + } + }) + + test('keeps a provider-specific user agent', async () => { + const svc = new ProviderService() + const provider = await svc.addProvider({ + presetId: 'custom', + name: 'Anthropic Compat', + baseUrl: 'https://relay.example.com', + apiKey: '', + apiFormat: 'anthropic', + authStrategy: 'api_key', + supportsNestedToolResultMedia: false, + models: { + main: 'model-main', + haiku: 'model-main', + sonnet: 'model-main', + opus: 'model-main', + }, + }) + + const originalFetch = globalThis.fetch + const captured: Headers[] = [] + globalThis.fetch = mock(async (_url: string | URL | Request, init?: RequestInit) => { + captured.push(new Headers(init?.headers)) + return new Response(JSON.stringify({ + type: 'message', + role: 'assistant', + model: 'model-main', + content: [{ type: 'text', text: 'ok' }], + stop_reason: 'end_turn', + usage: { input_tokens: 1, output_tokens: 1 }, + }), { status: 200, headers: { 'Content-Type': 'application/json' } }) + }) as typeof fetch + + try { + const req = new Request( + `http://localhost:3456/proxy/providers/${provider.id}/v1/messages`, + { + method: 'POST', + headers: { + 'Content-Type': 'application/json', + 'anthropic-version': '2023-06-01', + 'User-Agent': 'third-party-gateway/1.0', + 'x-app': 'provider-routing-client', + }, + body: JSON.stringify({ + model: 'model-main', + max_tokens: 64, + messages: [{ role: 'user', content: 'hi' }], + }), + }, + ) + + const res = await handleProxyRequest(req, new URL(req.url)) + expect(res.status).toBe(200) + expect(captured[0]?.get('user-agent')).toBe('third-party-gateway/1.0') + expect(captured[0]?.get('x-app')).toBe('provider-routing-client') + } finally { + globalThis.fetch = originalFetch + } + }) + + test('uses preset default Bearer auth for a saved no-key provider', async () => { + const svc = new ProviderService() + const provider = await svc.addProvider({ + presetId: 'lmstudio', + name: 'LM Studio', + baseUrl: 'http://localhost:1234', + apiKey: '', + apiFormat: 'anthropic', + supportsNestedToolResultMedia: false, + models: { + main: 'model-main', + haiku: 'model-main', + sonnet: 'model-main', + opus: 'model-main', + }, + }) + + const originalFetch = globalThis.fetch + const captured: Array<{ init: RequestInit }> = [] + globalThis.fetch = mock(async (_url: string | URL | Request, init?: RequestInit) => { + captured.push({ init: init ?? {} }) + return new Response(JSON.stringify({ + id: 'msg_relay', + type: 'message', + role: 'assistant', + model: 'model-main', + content: [{ type: 'text', text: 'ok' }], + stop_reason: 'end_turn', + usage: { input_tokens: 1, output_tokens: 1 }, + }), { status: 200, headers: { 'Content-Type': 'application/json' } }) + }) as typeof fetch + + try { + const req = new Request( + `http://localhost:3456/proxy/providers/${provider.id}/v1/messages`, + { + method: 'POST', + headers: { 'Content-Type': 'application/json', 'anthropic-version': '2023-06-01' }, + body: JSON.stringify({ + model: 'model-main', + max_tokens: 64, + messages: [{ role: 'user', content: 'hi' }], + }), + }, + ) + + const res = await handleProxyRequest(req, new URL(req.url)) + expect(res.status).toBe(200) + + const headers = captured[0]!.init.headers as Record + expect(headers.Authorization).toBe('Bearer lmstudio') + expect(headers['x-api-key']).toBeUndefined() + } finally { + globalThis.fetch = originalFetch + } + }) + + test('uses Bearer auth for auth_token strategy', async () => { + const svc = new ProviderService() + const provider = await svc.addProvider({ + presetId: 'custom', + name: 'Anthropic Compat', + baseUrl: 'https://relay.example.com', + apiKey: 'sk-relay', + apiFormat: 'anthropic', + authStrategy: 'auth_token', + supportsNestedToolResultMedia: false, + models: { + main: 'model-main', + haiku: 'model-main', + sonnet: 'model-main', + opus: 'model-main', + }, + }) + + const originalFetch = globalThis.fetch + const captured: Array<{ init: RequestInit }> = [] + globalThis.fetch = mock(async (_url: string | URL | Request, init?: RequestInit) => { + captured.push({ init: init ?? {} }) + return new Response(JSON.stringify({ + id: 'msg_relay', + type: 'message', + role: 'assistant', + model: 'model-main', + content: [{ type: 'text', text: 'ok' }], + stop_reason: 'end_turn', + usage: { input_tokens: 1, output_tokens: 1 }, + }), { status: 200, headers: { 'Content-Type': 'application/json' } }) + }) as typeof fetch + + try { + const req = new Request( + `http://localhost:3456/proxy/providers/${provider.id}/v1/messages`, + { + method: 'POST', + headers: { 'Content-Type': 'application/json', 'anthropic-version': '2023-06-01' }, + body: JSON.stringify({ + model: 'model-main', + max_tokens: 64, + messages: [{ role: 'user', content: 'hi' }], + }), + }, + ) + + const res = await handleProxyRequest(req, new URL(req.url)) + expect(res.status).toBe(200) + + const headers = captured[0]!.init.headers as Record + expect(headers.Authorization).toBe('Bearer sk-relay') + expect(headers['x-api-key']).toBeUndefined() + } finally { + globalThis.fetch = originalFetch + } + }) + + test('passes upstream error bodies and headers through unchanged', async () => { + const svc = new ProviderService() + const provider = await svc.addProvider({ + presetId: 'custom', + name: 'Anthropic Compat', + baseUrl: 'https://relay.example.com', + apiKey: 'sk-relay', + apiFormat: 'anthropic', + supportsNestedToolResultMedia: false, + models: { + main: 'model-main', + haiku: 'model-main', + sonnet: 'model-main', + opus: 'model-main', + }, + }) + + const originalFetch = globalThis.fetch + globalThis.fetch = mock(async () => { + return new Response(JSON.stringify({ + type: 'error', + error: { type: 'rate_limit_error', message: 'slow down' }, + }), { + status: 429, + headers: { 'Content-Type': 'application/json', 'retry-after': '42' }, + }) + }) as typeof fetch + + try { + const req = new Request( + `http://localhost:3456/proxy/providers/${provider.id}/v1/messages`, + { + method: 'POST', + headers: { 'Content-Type': 'application/json' }, + body: JSON.stringify({ + model: 'model-main', + max_tokens: 64, + messages: [{ role: 'user', content: 'hi' }], + }), + }, + ) + + const res = await handleProxyRequest(req, new URL(req.url)) + expect(res.status).toBe(429) + expect(res.headers.get('retry-after')).toBe('42') + const body = await res.json() as { error: { type: string } } + expect(body.error.type).toBe('rate_limit_error') + } finally { + globalThis.fetch = originalFetch + } + }) + + test('passes success bodies and entity headers through byte-for-byte', async () => { + const svc = new ProviderService() + const provider = await svc.addProvider({ + presetId: 'custom', + name: 'Anthropic Compat', + baseUrl: 'https://relay.example.com', + apiKey: 'sk-relay', + apiFormat: 'anthropic', + supportsNestedToolResultMedia: false, + models: { + main: 'model-main', + haiku: 'model-main', + sonnet: 'model-main', + opus: 'model-main', + }, + }) + + const originalFetch = globalThis.fetch + const upstreamBody = JSON.stringify({ + id: 'msg_relay', + type: 'message', + role: 'assistant', + model: 'model-main', + content: [{ type: 'text', text: 'ok' }], + stop_reason: 'end_turn', + usage: { input_tokens: 1, output_tokens: 1 }, + }) + globalThis.fetch = mock(async () => { + return new Response(upstreamBody, { + status: 200, + headers: { + 'Content-Type': 'application/json', + 'Content-Length': String(upstreamBody.length), + 'x-request-id': 'req_abc', + 'etag': '"v1"', + }, + }) + }) as typeof fetch + + try { + const req = new Request( + `http://localhost:3456/proxy/providers/${provider.id}/v1/messages`, + { + method: 'POST', + headers: { 'Content-Type': 'application/json' }, + body: JSON.stringify({ + model: 'model-main', + max_tokens: 64, + messages: [{ role: 'user', content: 'hi' }], + }), + }, + ) + + const res = await handleProxyRequest(req, new URL(req.url)) + expect(res.status).toBe(200) + expect(res.headers.get('content-length')).toBe(String(upstreamBody.length)) + expect(res.headers.get('x-request-id')).toBe('req_abc') + expect(res.headers.get('etag')).toBe('"v1"') + expect(await res.text()).toBe(upstreamBody) + } finally { + globalThis.fetch = originalFetch + } + }) + + test('forwards compressed upstream bodies byte-for-byte with their encoding headers', async () => { + const svc = new ProviderService() + const provider = await svc.addProvider({ + presetId: 'custom', + name: 'Anthropic Compat', + baseUrl: 'https://relay.example.com', + apiKey: 'sk-relay', + apiFormat: 'anthropic', + supportsNestedToolResultMedia: false, + models: { + main: 'model-main', + haiku: 'model-main', + sonnet: 'model-main', + opus: 'model-main', + }, + }) + + const originalFetch = globalThis.fetch + const upstreamBody = JSON.stringify({ + id: 'msg_relay', + type: 'message', + role: 'assistant', + model: 'model-main', + content: [{ type: 'text', text: 'ok' }], + stop_reason: 'end_turn', + usage: { input_tokens: 1, output_tokens: 1 }, + }) + const gzipBody = Bun.gzipSync(Buffer.from(upstreamBody)) + globalThis.fetch = mock(async () => { + return new Response(gzipBody, { + status: 200, + headers: { + 'Content-Type': 'application/json', + 'Content-Encoding': 'gzip', + 'Content-Length': String(gzipBody.length), + 'etag': '"gzip-v1"', + }, + }) + }) as typeof fetch + + try { + const sessionId = 'session-gzip-trace' + const req = new Request( + `http://localhost:3456/proxy/providers/${provider.id}/v1/messages`, + { + method: 'POST', + headers: { + 'Content-Type': 'application/json', + 'x-claude-code-session-id': sessionId, + }, + body: JSON.stringify({ + model: 'model-main', + max_tokens: 64, + messages: [{ role: 'user', content: 'hi' }], + }), + }, + ) + + const res = await handleProxyRequest(req, new URL(req.url)) + expect(res.status).toBe(200) + expect(res.headers.get('content-encoding')).toBe('gzip') + expect(res.headers.get('content-length')).toBe(String(gzipBody.length)) + expect(res.headers.get('etag')).toBe('"gzip-v1"') + const forwarded = new Uint8Array(await res.arrayBuffer()) + expect(forwarded).toEqual(new Uint8Array(gzipBody)) + + // The trace stores readable text: the captured gzip bytes are + // decompressed for storage instead of being UTF-8-decoded as garbage. + // (The body is JSON-stringified inside the trace record, so the marker + // appears unescaped only in decompressed plain text.) + const tracePath = path.join(tmpDir, 'cc-haha', 'traces', `${sessionId}.jsonl`) + let traceRaw = '' + for (let attempt = 0; attempt < 200; attempt++) { + try { + traceRaw = await fs.readFile(tracePath, 'utf-8') + if (traceRaw.includes('upstream_fetch_completed')) break + } catch { + // not written yet + } + await new Promise(resolve => setTimeout(resolve, 25)) + } + expect(traceRaw).toContain('msg_relay') + } finally { + globalThis.fetch = originalFetch + } + }) + + test('forwards compressed SSE byte-for-byte while the trace stores plain text', async () => { + const svc = new ProviderService() + const provider = await svc.addProvider({ + presetId: 'custom', + name: 'Anthropic Compat', + baseUrl: 'https://relay.example.com', + apiKey: 'sk-relay', + apiFormat: 'anthropic', + supportsNestedToolResultMedia: false, + models: { + main: 'model-main', + haiku: 'model-main', + sonnet: 'model-main', + opus: 'model-main', + }, + }) + + const originalFetch = globalThis.fetch + const sseText = [ + 'event: message_start', + 'data: {"type":"message_start","message":{"id":"msg_relay_stream","type":"message","role":"assistant","content":[],"model":"model-main","stop_reason":null,"usage":{"input_tokens":1,"output_tokens":1}}}', + '', + 'event: content_block_delta', + 'data: {"type":"content_block_delta","index":0,"delta":{"type":"text_delta","text":"hello"}}', + '', + 'event: message_stop', + 'data: {"type":"message_stop"}', + '', + '', + ].join('\n') + const gzipBody = Bun.gzipSync(Buffer.from(sseText)) + globalThis.fetch = mock(async () => { + return new Response(gzipBody, { + status: 200, + headers: { + 'Content-Type': 'text/event-stream', + 'Content-Encoding': 'gzip', + 'Content-Length': String(gzipBody.length), + }, + }) + }) as typeof fetch + + try { + const sessionId = 'session-gzip-stream-trace' + const req = new Request( + `http://localhost:3456/proxy/providers/${provider.id}/v1/messages`, + { + method: 'POST', + headers: { + 'Content-Type': 'application/json', + 'x-claude-code-session-id': sessionId, + }, + body: JSON.stringify({ + model: 'model-main', + max_tokens: 64, + stream: true, + messages: [{ role: 'user', content: 'hi' }], + }), + }, + ) + + const res = await handleProxyRequest(req, new URL(req.url)) + expect(res.status).toBe(200) + expect(res.headers.get('content-encoding')).toBe('gzip') + expect(res.headers.get('content-length')).toBe(String(gzipBody.length)) + // The client receives the raw compressed bytes unchanged (it decodes + // them itself); only the trace snapshot is decompressed. + const forwarded = new Uint8Array(await res.arrayBuffer()) + expect(forwarded).toEqual(new Uint8Array(gzipBody)) + + const tracePath = path.join(tmpDir, 'cc-haha', 'traces', `${sessionId}.jsonl`) + let traceRaw = '' + for (let attempt = 0; attempt < 200; attempt++) { + try { + traceRaw = await fs.readFile(tracePath, 'utf-8') + if (traceRaw.includes('upstream_fetch_completed')) break + } catch { + // not written yet + } + await new Promise(resolve => setTimeout(resolve, 25)) + } + expect(traceRaw).toContain('msg_relay_stream') + // The snapshot holds readable SSE text, not gzip bytes decoded as UTF-8. + expect(traceRaw).not.toContain('\uFFFD') + } finally { + globalThis.fetch = originalFetch + } + }) + + test('keeps compressed stream trace readable when the decoded size exceeds the capture cap', async () => { + const svc = new ProviderService() + const provider = await svc.addProvider({ + presetId: 'custom', + name: 'Anthropic Compat', + baseUrl: 'https://relay.example.com', + apiKey: 'sk-relay', + apiFormat: 'anthropic', + supportsNestedToolResultMedia: false, + models: { + main: 'model-main', + haiku: 'model-main', + sonnet: 'model-main', + opus: 'model-main', + }, + }) + + const originalFetch = globalThis.fetch + // A compact gzip body that decompresses well past the 1 MiB trace cap: + // the cap applies to decoded output, so the trace still holds a readable + // plain-text prefix instead of an unterminated gzip member. + const chunk = 'x'.repeat(1024) + const sseText = [ + 'event: message_start', + 'data: {"type":"message_start","message":{"id":"msg_relay_big","type":"message","role":"assistant","content":[],"model":"model-main","stop_reason":null,"usage":{"input_tokens":1,"output_tokens":1}}}', + '', + ...Array.from({ length: 1200 }, () => `data: ${chunk}`), + '', + 'event: message_stop', + 'data: {"type":"message_stop"}', + '', + '', + ].join('\n') + expect(sseText.length).toBeGreaterThan(1024 * 1024) + const gzipBody = Bun.gzipSync(Buffer.from(sseText)) + expect(gzipBody.length).toBeLessThan(1024 * 1024) + globalThis.fetch = mock(async () => { + return new Response(gzipBody, { + status: 200, + headers: { + 'Content-Type': 'text/event-stream', + 'Content-Encoding': 'gzip', + 'Content-Length': String(gzipBody.length), + }, + }) + }) as typeof fetch + + try { + const sessionId = 'session-gzip-big-trace' + const req = new Request( + `http://localhost:3456/proxy/providers/${provider.id}/v1/messages`, + { + method: 'POST', + headers: { + 'Content-Type': 'application/json', + 'x-claude-code-session-id': sessionId, + }, + body: JSON.stringify({ + model: 'model-main', + max_tokens: 64, + stream: true, + messages: [{ role: 'user', content: 'hi' }], + }), + }, + ) + + const res = await handleProxyRequest(req, new URL(req.url)) + expect(res.status).toBe(200) + const forwarded = new Uint8Array(await res.arrayBuffer()) + expect(forwarded).toEqual(new Uint8Array(gzipBody)) + + const tracePath = path.join(tmpDir, 'cc-haha', 'traces', `${sessionId}.jsonl`) + let traceRaw = '' + for (let attempt = 0; attempt < 200; attempt++) { + try { + traceRaw = await fs.readFile(tracePath, 'utf-8') + if (traceRaw.includes('upstream_fetch_completed')) break + } catch { + // not written yet + } + await new Promise(resolve => setTimeout(resolve, 25)) + } + expect(traceRaw).toContain('msg_relay_big') + expect(traceRaw).not.toContain('\uFFFD') + } finally { + globalThis.fetch = originalFetch + } + }) + + test('keeps the plain-text prefix when the client cancels a compressed stream mid-way', async () => { + const svc = new ProviderService() + const provider = await svc.addProvider({ + presetId: 'custom', + name: 'Anthropic Compat', + baseUrl: 'https://relay.example.com', + apiKey: 'sk-relay', + apiFormat: 'anthropic', + supportsNestedToolResultMedia: false, + models: { + main: 'model-main', + haiku: 'model-main', + sonnet: 'model-main', + opus: 'model-main', + }, + }) + + const originalFetch = globalThis.fetch + // A gzip member made of two deflate blocks (flush between parts), so a + // mid-way cancel leaves a complete first block the trace can decode. The + // first part carries high-entropy filler so its compressed size exceeds + // the zlib writable high-water mark: writing it returns false, exercising + // the backpressure wait path instead of the trivial 128-byte case. + const { createGzip } = await import('node:zlib') + const { randomBytes } = await import('node:crypto') + const filler = randomBytes(48 * 1024).toString('base64') + const gzip = createGzip() + const gzipChunks: Buffer[] = [] + gzip.on('data', (chunk: Buffer) => gzipChunks.push(chunk)) + const part1 = 'event: message_start\ndata: {"type":"message_start","message":{"id":"msg_relay_partial","type":"message","role":"assistant","content":[],"model":"model-main","stop_reason":null,"usage":{"input_tokens":1,"output_tokens":1}}}\n\ndata: ' + filler + '\n\n' + const part2 = 'event: content_block_delta\ndata: {"type":"content_block_delta","index":0,"delta":{"type":"text_delta","text":"hello"}}\n\nevent: message_stop\ndata: {"type":"message_stop"}\n\n' + gzip.write(Buffer.from(part1)) + await new Promise(resolve => gzip.flush(() => resolve())) + gzip.write(Buffer.from(part2)) + gzip.end() + await new Promise(resolve => gzip.on('end', () => resolve())) + const gzipBody = Buffer.concat(gzipChunks) + // The first chunk must exceed zlib's default writableHighWaterMark + // (16 KiB) so decompressor.write() returns false. + expect(gzipBody.length).toBeGreaterThan(32 * 1024) + + // The second chunk stays pending behind a gate until the client cancels: + // the test must prove the trace wrapper stops reading mid-member, not + // that both chunks happened to arrive before the cancel. + let releasePart2: (() => void) | null = null + let part2Released = false + let upstreamCancelled = false + let upstreamBody: ReadableStream | null = null + globalThis.fetch = mock(async () => { + // Chunked body so the client can stop reading mid-member. + upstreamBody = new ReadableStream({ + start(controller) { + controller.enqueue(gzipBody.subarray(0, 32 * 1024)) + void new Promise(resolve => { + releasePart2 = resolve + }).then(() => { + part2Released = true + controller.enqueue(gzipBody.subarray(32 * 1024)) + controller.close() + }) + }, + cancel() { + upstreamCancelled = true + }, + }) + return new Response(upstreamBody, { + status: 200, + headers: { + 'Content-Type': 'text/event-stream', + 'Content-Encoding': 'gzip', + 'Content-Length': String(gzipBody.length), + }, + }) + }) as typeof fetch + + try { + const sessionId = 'session-gzip-cancel-trace' + const req = new Request( + `http://localhost:3456/proxy/providers/${provider.id}/v1/messages`, + { + method: 'POST', + headers: { + 'Content-Type': 'application/json', + 'x-claude-code-session-id': sessionId, + }, + body: JSON.stringify({ + model: 'model-main', + max_tokens: 64, + stream: true, + messages: [{ role: 'user', content: 'hi' }], + }), + }, + ) + + const res = await handleProxyRequest(req, new URL(req.url)) + expect(res.status).toBe(200) + // The client stops reading after the first chunk of the gzip member: + // the trace copy must still hold the decoded SSE prefix already + // received, not an unterminated gzip member decoded as garbage. + const reader = res.body!.getReader() + await reader.read() + await reader.cancel('client stopped generation').catch(() => undefined) + + // The cancel must propagate upstream while part 2 is still pending: + // the trace wrapper reads lazily, so it cannot have consumed part 2. + expect(upstreamCancelled).toBe(true) + expect(part2Released).toBe(false) + // The read loop must finish and release the upstream reader even when + // the cancel lands while the decompressor is backpressured (write() + // returned false and the waiter is parked on drain/finish/close). + for (let attempt = 0; attempt < 100 && upstreamBody?.locked; attempt++) { + await new Promise(resolve => setTimeout(resolve, 5)) + } + expect(upstreamBody?.locked).toBe(false) + + const tracePath = path.join(tmpDir, 'cc-haha', 'traces', `${sessionId}.jsonl`) + let traceRaw = '' + let lastSize = 0 + for (let attempt = 0; attempt < 200; attempt++) { + try { + traceRaw = await fs.readFile(tracePath, 'utf-8') + if (traceRaw.includes('cancelled') && traceRaw.length === lastSize) break + lastSize = traceRaw.length + } catch { + // not written yet + } + await new Promise(resolve => setTimeout(resolve, 25)) + } + expect(traceRaw).toContain('msg_relay_partial') + expect(traceRaw).toContain('cancelled') + expect(traceRaw).not.toContain('content_block_delta') + expect(traceRaw).not.toContain('\uFFFD') + } finally { + globalThis.fetch = originalFetch + } + }) + + test('marks the trace unavailable for stacked content encodings instead of storing garbage', async () => { + const svc = new ProviderService() + const provider = await svc.addProvider({ + presetId: 'custom', + name: 'Anthropic Compat', + baseUrl: 'https://relay.example.com', + apiKey: 'sk-relay', + apiFormat: 'anthropic', + supportsNestedToolResultMedia: false, + models: { + main: 'model-main', + haiku: 'model-main', + sonnet: 'model-main', + opus: 'model-main', + }, + }) + + const originalFetch = globalThis.fetch + const sseText = [ + 'event: message_start', + 'data: {"type":"message_start","message":{"id":"msg_relay_stacked","type":"message","role":"assistant","content":[],"model":"model-main","stop_reason":null,"usage":{"input_tokens":1,"output_tokens":1}}}', + '', + 'event: message_stop', + 'data: {"type":"message_stop"}', + '', + '', + ].join('\n') + // Stacked encodings: Content-Encoding lists codes in application order, + // so 'gzip, deflate' means gzip was applied first, then deflate — the + // wire bytes are deflate(gzip(body)). The streaming branch only checks + // that more than one codec is present, so the exact stacking does not + // change the assertion; the fixture stays semantically correct for any + // future buffered decode test. + const doubleEncoded = Bun.deflateSync(Bun.gzipSync(Buffer.from(sseText))) + globalThis.fetch = mock(async () => { + return new Response(new Uint8Array(doubleEncoded), { + status: 200, + headers: { + 'Content-Type': 'text/event-stream', + 'Content-Encoding': 'gzip, deflate', + 'Content-Length': String(doubleEncoded.length), + }, + }) + }) as typeof fetch + + try { + const sessionId = 'session-gzip-stacked-trace' + const req = new Request( + `http://localhost:3456/proxy/providers/${provider.id}/v1/messages`, + { + method: 'POST', + headers: { + 'Content-Type': 'application/json', + 'x-claude-code-session-id': sessionId, + }, + body: JSON.stringify({ + model: 'model-main', + max_tokens: 64, + stream: true, + messages: [{ role: 'user', content: 'hi' }], + }), + }, + ) + + const res = await handleProxyRequest(req, new URL(req.url)) + expect(res.status).toBe(200) + // The client still receives the raw double-encoded bytes unchanged. + const forwarded = new Uint8Array(await res.arrayBuffer()) + expect(forwarded).toEqual(new Uint8Array(doubleEncoded)) + + const tracePath = path.join(tmpDir, 'cc-haha', 'traces', `${sessionId}.jsonl`) + let traceRaw = '' + for (let attempt = 0; attempt < 200; attempt++) { + try { + traceRaw = await fs.readFile(tracePath, 'utf-8') + if (traceRaw.includes('upstream_fetch_completed')) break + } catch { + // not written yet + } + await new Promise(resolve => setTimeout(resolve, 25)) + } + // The streaming branch only unwinds a single known codec; the trace is + // marked unavailable instead of storing compressed bytes as UTF-8. + expect(traceRaw).toContain('trace body unavailable') + expect(traceRaw).not.toContain('msg_relay_stacked') + expect(traceRaw).not.toContain('\uFFFD') + } finally { + globalThis.fetch = originalFetch + } + }) + + test('marks the buffered trace unavailable for unsupported content encodings', async () => { + const svc = new ProviderService() + const provider = await svc.addProvider({ + presetId: 'custom', + name: 'Anthropic Compat', + baseUrl: 'https://relay.example.com', + apiKey: 'sk-relay', + apiFormat: 'anthropic', + supportsNestedToolResultMedia: false, + models: { + main: 'model-main', + haiku: 'model-main', + sonnet: 'model-main', + opus: 'model-main', + }, + }) + + const originalFetch = globalThis.fetch + const upstreamBody = JSON.stringify({ + id: 'msg_relay_br', + type: 'message', + role: 'assistant', + model: 'model-main', + content: [{ type: 'text', text: 'ok' }], + stop_reason: 'end_turn', + usage: { input_tokens: 1, output_tokens: 1 }, + }) + // Brotli is advertised by Bun's default Accept-Encoding and reachable via + // header passthrough, but the trace decoder has no brotli codec — the + // buffered path must mark the trace unavailable instead of storing the + // compressed bytes as UTF-8 garbage. The body bytes themselves are + // irrelevant here: the header alone drives the unavailable marker. + const brBody = Bun.gzipSync(Buffer.from(upstreamBody)) + globalThis.fetch = mock(async () => { + return new Response(brBody, { + status: 200, + headers: { + 'Content-Type': 'application/json', + 'Content-Encoding': 'br', + 'Content-Length': String(brBody.length), + }, + }) + }) as typeof fetch + + try { + const sessionId = 'session-br-trace' + const req = new Request( + `http://localhost:3456/proxy/providers/${provider.id}/v1/messages`, + { + method: 'POST', + headers: { + 'Content-Type': 'application/json', + 'x-claude-code-session-id': sessionId, + }, + body: JSON.stringify({ + model: 'model-main', + max_tokens: 64, + messages: [{ role: 'user', content: 'hi' }], + }), + }, + ) + + const res = await handleProxyRequest(req, new URL(req.url)) + expect(res.status).toBe(200) + + const tracePath = path.join(tmpDir, 'cc-haha', 'traces', `${sessionId}.jsonl`) + let traceRaw = '' + for (let attempt = 0; attempt < 200; attempt++) { + try { + traceRaw = await fs.readFile(tracePath, 'utf-8') + if (traceRaw.includes('upstream_fetch_completed')) break + } catch { + // not written yet + } + await new Promise(resolve => setTimeout(resolve, 25)) + } + expect(traceRaw).toContain('trace body unavailable') + expect(traceRaw).not.toContain('\uFFFD') + } finally { + globalThis.fetch = originalFetch + } + }) + + test('marks the buffered trace unavailable when a known codec fails to decompress', async () => { + const svc = new ProviderService() + const provider = await svc.addProvider({ + presetId: 'custom', + name: 'Anthropic Compat', + baseUrl: 'https://relay.example.com', + apiKey: 'sk-relay', + apiFormat: 'anthropic', + supportsNestedToolResultMedia: false, + models: { + main: 'model-main', + haiku: 'model-main', + sonnet: 'model-main', + opus: 'model-main', + }, + }) + + const originalFetch = globalThis.fetch + // A supported codec in the header, but the body is not a valid gzip + // member: the buffered path must mark the trace unavailable instead of + // falling back to decoding the compressed bytes as UTF-8 garbage. + const corruptGzip = new Uint8Array([0x1f, 0x8b, 0x08, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x03, 0xde, 0xad, 0xbe, 0xef]) + globalThis.fetch = mock(async () => { + return new Response(corruptGzip, { + status: 200, + headers: { + 'Content-Type': 'application/json', + 'Content-Encoding': 'gzip', + 'Content-Length': String(corruptGzip.length), + }, + }) + }) as typeof fetch + + try { + const sessionId = 'session-gzip-corrupt-trace' + const req = new Request( + `http://localhost:3456/proxy/providers/${provider.id}/v1/messages`, + { + method: 'POST', + headers: { + 'Content-Type': 'application/json', + 'x-claude-code-session-id': sessionId, + }, + body: JSON.stringify({ + model: 'model-main', + max_tokens: 64, + messages: [{ role: 'user', content: 'hi' }], + }), + }, + ) + + const res = await handleProxyRequest(req, new URL(req.url)) + expect(res.status).toBe(200) + // The raw bytes still pass through unchanged; only the trace is marked. + const forwarded = new Uint8Array(await res.arrayBuffer()) + expect(forwarded).toEqual(corruptGzip) + + const tracePath = path.join(tmpDir, 'cc-haha', 'traces', `${sessionId}.jsonl`) + let traceRaw = '' + for (let attempt = 0; attempt < 200; attempt++) { + try { + traceRaw = await fs.readFile(tracePath, 'utf-8') + if (traceRaw.includes('upstream_fetch_completed')) break + } catch { + // not written yet + } + await new Promise(resolve => setTimeout(resolve, 25)) + } + expect(traceRaw).toContain('trace body unavailable') + expect(traceRaw).not.toContain('\uFFFD') + } finally { + globalThis.fetch = originalFetch + } + }) + + test('marks the streaming trace unavailable when a known codec fails to decompress', async () => { + const svc = new ProviderService() + const provider = await svc.addProvider({ + presetId: 'custom', + name: 'Anthropic Compat', + baseUrl: 'https://relay.example.com', + apiKey: 'sk-relay', + apiFormat: 'anthropic', + supportsNestedToolResultMedia: false, + models: { + main: 'model-main', + haiku: 'model-main', + sonnet: 'model-main', + opus: 'model-main', + }, + }) + + const originalFetch = globalThis.fetch + // The header names a supported codec, but the streamed body is not a + // valid gzip member: the streaming branch must record the failure instead + // of storing a partial/empty decode as a successful trace. + const corruptGzip = new Uint8Array([0x1f, 0x8b, 0x08, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x03, 0xde, 0xad, 0xbe, 0xef]) + globalThis.fetch = mock(async () => { + return new Response(corruptGzip, { + status: 200, + headers: { + 'Content-Type': 'text/event-stream', + 'Content-Encoding': 'gzip', + 'Content-Length': String(corruptGzip.length), + }, + }) + }) as typeof fetch + + try { + const sessionId = 'session-gzip-corrupt-stream-trace' + const req = new Request( + `http://localhost:3456/proxy/providers/${provider.id}/v1/messages`, + { + method: 'POST', + headers: { + 'Content-Type': 'application/json', + 'x-claude-code-session-id': sessionId, + }, + body: JSON.stringify({ + model: 'model-main', + max_tokens: 64, + stream: true, + messages: [{ role: 'user', content: 'hi' }], + }), + }, + ) + + const res = await handleProxyRequest(req, new URL(req.url)) + expect(res.status).toBe(200) + // The client still receives the raw bytes unchanged. + const forwarded = new Uint8Array(await res.arrayBuffer()) + expect(forwarded).toEqual(corruptGzip) + + const tracePath = path.join(tmpDir, 'cc-haha', 'traces', `${sessionId}.jsonl`) + let traceRaw = '' + for (let attempt = 0; attempt < 200; attempt++) { + try { + traceRaw = await fs.readFile(tracePath, 'utf-8') + if (traceRaw.includes('upstream_fetch_completed')) break + } catch { + // not written yet + } + await new Promise(resolve => setTimeout(resolve, 25)) + } + expect(traceRaw).toContain('trace body unavailable') + expect(traceRaw).not.toContain('\uFFFD') + } finally { + globalThis.fetch = originalFetch + } + }) + + test('strips hop-by-hop headers from upstream success responses', async () => { + const svc = new ProviderService() + const provider = await svc.addProvider({ + presetId: 'custom', + name: 'Anthropic Compat', + baseUrl: 'https://relay.example.com', + apiKey: 'sk-relay', + apiFormat: 'anthropic', + supportsNestedToolResultMedia: false, + models: { + main: 'model-main', + haiku: 'model-main', + sonnet: 'model-main', + opus: 'model-main', + }, + }) + + const originalFetch = globalThis.fetch + globalThis.fetch = mock(async () => { + return new Response(JSON.stringify({ + id: 'msg_relay', + type: 'message', + role: 'assistant', + model: 'model-main', + content: [{ type: 'text', text: 'ok' }], + stop_reason: 'end_turn', + usage: { input_tokens: 1, output_tokens: 1 }, + }), { + status: 200, + headers: { + 'Content-Type': 'application/json', + connection: 'x-internal-hop', + 'x-internal-hop': 'secret', + 'x-request-id': 'req_abc', + }, + }) + }) as typeof fetch + + try { + const req = new Request( + `http://localhost:3456/proxy/providers/${provider.id}/v1/messages`, + { + method: 'POST', + headers: { 'Content-Type': 'application/json' }, + body: JSON.stringify({ + model: 'model-main', + max_tokens: 64, + messages: [{ role: 'user', content: 'hi' }], + }), + }, + ) + + const res = await handleProxyRequest(req, new URL(req.url)) + expect(res.status).toBe(200) + expect(res.headers.get('x-request-id')).toBe('req_abc') + expect(res.headers.get('connection')).toBeNull() + expect(res.headers.get('x-internal-hop')).toBeNull() + } finally { + globalThis.fetch = originalFetch + } + }) + + test('returns a structured 502 when the upstream error body fails to read', async () => { + const svc = new ProviderService() + const provider = await svc.addProvider({ + presetId: 'custom', + name: 'Anthropic Compat', + baseUrl: 'https://relay.example.com', + apiKey: 'sk-relay', + apiFormat: 'anthropic', + supportsNestedToolResultMedia: false, + models: { + main: 'model-main', + haiku: 'model-main', + sonnet: 'model-main', + opus: 'model-main', + }, + }) + + const originalFetch = globalThis.fetch + globalThis.fetch = mock(async () => { + // Status and headers arrive, then the body stream errors mid-read. + const stream = new ReadableStream({ + start(controller) { + controller.enqueue(new TextEncoder().encode('partial')) + controller.error('connection reset') + }, + }) + return new Response(stream, { + status: 429, + headers: { 'Content-Type': 'application/json', 'retry-after': '42' }, + }) + }) as typeof fetch + + try { + const req = new Request( + `http://localhost:3456/proxy/providers/${provider.id}/v1/messages`, + { + method: 'POST', + headers: { 'Content-Type': 'application/json' }, + body: JSON.stringify({ + model: 'model-main', + max_tokens: 64, + messages: [{ role: 'user', content: 'hi' }], + }), + }, + ) + + const res = await handleProxyRequest(req, new URL(req.url)) + expect(res.status).toBe(502) + const body = await res.json() as { error: { type: string } } + expect(body.error.type).toBe('api_error') + } finally { + globalThis.fetch = originalFetch + } + }) + + test('closes the pending trace when the upstream error body fails to read', async () => { + const svc = new ProviderService() + const provider = await svc.addProvider({ + presetId: 'custom', + name: 'Anthropic Compat', + baseUrl: 'https://relay.example.com', + apiKey: 'sk-relay', + apiFormat: 'anthropic', + supportsNestedToolResultMedia: false, + models: { + main: 'model-main', + haiku: 'model-main', + sonnet: 'model-main', + opus: 'model-main', + }, + }) + + const originalFetch = globalThis.fetch + globalThis.fetch = mock(async () => { + const stream = new ReadableStream({ + start(controller) { + controller.enqueue(new TextEncoder().encode('partial')) + controller.error('connection reset') + }, + }) + return new Response(stream, { + status: 429, + headers: { 'Content-Type': 'application/json' }, + }) + }) as typeof fetch + + try { + const sessionId = 'session-trace-body-read-fail' + const req = new Request( + `http://localhost:3456/proxy/providers/${provider.id}/v1/messages`, + { + method: 'POST', + headers: { + 'Content-Type': 'application/json', + 'x-claude-code-session-id': sessionId, + }, + body: JSON.stringify({ + model: 'model-main', + max_tokens: 64, + messages: [{ role: 'user', content: 'hi' }], + }), + }, + ) + + const res = await handleProxyRequest(req, new URL(req.url)) + expect(res.status).toBe(502) + + // Wait for the background trace write to land, then read the raw jsonl: + // the call opened as `pending` must be closed with the same id as an + // error, and no second call with a different id may appear. + const tracePath = path.join(tmpDir, 'cc-haha', 'traces', `${sessionId}.jsonl`) + let lines: string[] = [] + let lastSize = -1 + let stableReads = 0 + for (let attempt = 0; attempt < 200; attempt++) { + try { + const raw = await fs.readFile(tracePath, 'utf-8') + lines = raw.trim().split('\n').filter(Boolean) + // Wait for the failed call entry AND for the file to stop growing, + // so the background trace write (including the sqlite projection) + // has fully landed before the teardown removes the temp dir. + if (lines.some(line => line.includes('upstream_fetch_failed')) && raw.length === lastSize) { + stableReads += 1 + if (stableReads >= 3) break + } else { + stableReads = 0 + } + lastSize = raw.length + } catch { + // not written yet + } + await new Promise(resolve => setTimeout(resolve, 25)) + } + + const calls = lines + .map(line => JSON.parse(line) as { type: string; record?: { id: string; status: string } }) + .filter(entry => entry.type === 'call' && entry.record) + .map(entry => entry.record!) + expect(calls.length).toBeGreaterThan(0) + const callIds = new Set(calls.map(call => call.id)) + expect(callIds.size).toBe(1) + expect(calls[calls.length - 1]!.status).toBe('error') + } finally { + globalThis.fetch = originalFetch + } + }) + + test('returns 400 when anthropic provider supports nested media (proxy not needed)', async () => { + const svc = new ProviderService() + const provider = await svc.addProvider({ + presetId: 'custom', + name: 'Anthropic Compat', + baseUrl: 'https://relay.example.com', + apiKey: 'sk-relay', + apiFormat: 'anthropic', + models: { + main: 'model-main', + haiku: 'model-main', + sonnet: 'model-main', + opus: 'model-main', + }, + }) + + const originalFetch = globalThis.fetch + globalThis.fetch = mock(async () => { + throw new Error('fetch should not be called') + }) as typeof fetch + + try { + const req = new Request( + `http://localhost:3456/proxy/providers/${provider.id}/v1/messages`, + { + method: 'POST', + headers: { 'Content-Type': 'application/json' }, + body: JSON.stringify({ + model: 'model-main', + max_tokens: 64, + messages: [{ role: 'user', content: 'hi' }], + }), + }, + ) + + const res = await handleProxyRequest(req, new URL(req.url)) + expect(res.status).toBe(400) + } finally { + globalThis.fetch = originalFetch + } + }) +}) diff --git a/src/server/__tests__/proxy-network-settings.test.ts b/src/server/__tests__/proxy-network-settings.test.ts index 409f206c..9bd928aa 100644 --- a/src/server/__tests__/proxy-network-settings.test.ts +++ b/src/server/__tests__/proxy-network-settings.test.ts @@ -6,6 +6,7 @@ import { handleProxyRequest, withStreamIdleTimeout } from '../proxy/handler.js' import { ProviderService } from '../services/providerService.js' import { clearTraceCaptureStateForTests, + drainTraceCaptureForTests, traceCaptureService, } from '../services/traceCaptureService.js' import { resetSettingsCache } from '../../utils/settings/settingsCache.js' @@ -28,6 +29,7 @@ async function teardown() { delete process.env.CLAUDE_CONFIG_DIR } resetSettingsCache() + await drainTraceCaptureForTests() clearTraceCaptureStateForTests() await fs.rm(tmpDir, { recursive: true, force: true }) } diff --git a/src/server/__tests__/proxy-transform.test.ts b/src/server/__tests__/proxy-transform.test.ts index e763475b..d74591d8 100644 --- a/src/server/__tests__/proxy-transform.test.ts +++ b/src/server/__tests__/proxy-transform.test.ts @@ -297,40 +297,18 @@ describe('anthropicToOpenaiChat', () => { expect(result.messages[0].content).toBe('Sunny, 72°F') }) - test('preserves text-only tool_result arrays as strings', () => { + test('lifts tool_result images into a user message after the tool message', () => { const req: AnthropicRequest = { - model: 'gpt-4o', + model: 'gpt-4', max_tokens: 100, messages: [{ role: 'user', content: [{ type: 'tool_result', - tool_use_id: 'computer_0', - content: [ - { type: 'text', text: 'first' }, - { type: 'text', text: 'second' }, - ], - }], - }], - } - - const result = anthropicToOpenaiChat(req) - - expect(result.messages[0].content).toBe('first\nsecond') - }) - - test('preserves image-only tool_result content', () => { - const req: AnthropicRequest = { - model: 'gpt-4o', - max_tokens: 100, - messages: [{ - role: 'user', - content: [{ - type: 'tool_result', - tool_use_id: 'computer_1', + tool_use_id: 'tc_image', content: [{ type: 'image', - source: { type: 'base64', media_type: 'image/jpeg', data: '/9j/AA==' }, + source: { type: 'base64', media_type: 'image/png', data: 'abc123' }, }], }], }], @@ -338,32 +316,360 @@ describe('anthropicToOpenaiChat', () => { const result = anthropicToOpenaiChat(req) - expect(result.messages[0]).toEqual({ - role: 'tool', - tool_call_id: 'computer_1', - content: [{ - type: 'image_url', - image_url: { url: 'data:image/jpeg;base64,/9j/AA==' }, - }], - }) + expect(result.messages).toEqual([ + { + role: 'tool', + tool_call_id: 'tc_image', + content: 'Media result attached after this tool result.', + }, + { + role: 'user', + content: [ + { type: 'text', text: '[Media content for tool call tc_image]' }, + { + type: 'image_url', + image_url: { url: 'data:image/png;base64,abc123' }, + }, + ], + }, + ]) }) - test('preserves mixed tool_result content in order', () => { + test('keeps tool messages text-only and groups lifted media per tool call', () => { const req: AnthropicRequest = { - model: 'gpt-4o', + model: 'gpt-4', + max_tokens: 100, + messages: [{ + role: 'user', + content: [ + { type: 'tool_result', tool_use_id: 'tc_1', content: [ + { type: 'text', text: 'first' }, + { type: 'image', source: { type: 'base64', media_type: 'image/png', data: 'abc' } }, + { type: 'text', text: 'last' }, + ] }, + { type: 'tool_result', tool_use_id: 'tc_2', content: [ + { type: 'image', source: { type: 'base64', media_type: 'image/jpeg', data: 'def' } }, + ] }, + ], + }], + } + + const result = anthropicToOpenaiChat(req) + + expect(result.messages).toEqual([ + // Adjacent text blocks are joined without a separator — the wire shape + // carries no newline between them. + { role: 'tool', tool_call_id: 'tc_1', content: 'firstlast' }, + { role: 'tool', tool_call_id: 'tc_2', content: 'Media result attached after this tool result.' }, + { role: 'user', content: [ + { type: 'text', text: '[Media content for tool call tc_1]' }, + { type: 'image_url', image_url: { url: 'data:image/png;base64,abc' } }, + { type: 'text', text: '[Media content for tool call tc_2]' }, + { type: 'image_url', image_url: { url: 'data:image/jpeg;base64,def' } }, + ] }, + ]) + }) + + test('collapses pure-text user messages to a plain string content', () => { + const req: AnthropicRequest = { + model: 'gpt-4', + max_tokens: 100, + messages: [{ + role: 'user', + content: [{ type: 'text', text: 'hello' }], + }], + } + + // OpenAI-compatible endpoints that only implement string `content` keep + // working; the multipart array form is reserved for media messages. + expect(anthropicToOpenaiChat(req).messages).toEqual([ + { role: 'user', content: 'hello' }, + ]) + }) + + test('collapses multiple text-only blocks to a plain string', () => { + const req: AnthropicRequest = { + model: 'gpt-4', + max_tokens: 100, + messages: [{ + role: 'user', + content: [ + { type: 'text', text: 'hello ' }, + { type: 'text', text: 'world' }, + ], + }], + } + + expect(anthropicToOpenaiChat(req).messages).toEqual([ + { role: 'user', content: 'hello world' }, + ]) + }) + + test('keeps ordinary mixed user content in one message', () => { + const req: AnthropicRequest = { + model: 'gpt-4', + max_tokens: 100, + messages: [{ + role: 'user', + content: [ + { type: 'text', text: 'before' }, + { type: 'image', source: { type: 'base64', media_type: 'image/png', data: 'abc' } }, + { type: 'text', text: 'after' }, + ], + }], + } + + expect(anthropicToOpenaiChat(req).messages).toEqual([{ + role: 'user', + content: [ + { type: 'text', text: 'before' }, + { type: 'image_url', image_url: { url: 'data:image/png;base64,abc' } }, + { type: 'text', text: 'after' }, + ], + }]) + }) + + test('keeps tool-result media ahead of trailing user text', () => { + const req: AnthropicRequest = { + model: 'gpt-4', + max_tokens: 100, + messages: [{ + role: 'user', + content: [ + { + type: 'tool_result', + tool_use_id: 'tc_1', + content: [{ type: 'image', source: { type: 'base64', media_type: 'image/png', data: 'abc' } }], + }, + { type: 'text', text: 'Please look at the top-left corner of the image' }, + ], + }], + } + + expect(anthropicToOpenaiChat(req).messages).toEqual([ + { role: 'tool', tool_call_id: 'tc_1', content: 'Media result attached after this tool result.' }, + { role: 'user', content: [ + { type: 'text', text: '[Media content for tool call tc_1]' }, + { type: 'image_url', image_url: { url: 'data:image/png;base64,abc' } }, + ] }, + { role: 'user', content: 'Please look at the top-left corner of the image' }, + ]) + }) + + test('keeps image URL sources in lifted tool-result media', () => { + const req: AnthropicRequest = { + model: 'gpt-4', max_tokens: 100, messages: [{ role: 'user', content: [{ type: 'tool_result', - tool_use_id: 'computer_2', + tool_use_id: 'tc_url', + content: [{ + type: 'image', + source: { type: 'url', url: 'https://example.test/shot.png' }, + }], + }], + }], + } + + expect(anthropicToOpenaiChat(req).messages).toEqual([ + { role: 'tool', tool_call_id: 'tc_url', content: 'Media result attached after this tool result.' }, + { role: 'user', content: [ + { type: 'text', text: '[Media content for tool call tc_url]' }, + { type: 'image_url', image_url: { url: 'https://example.test/shot.png' } }, + ] }, + ]) + }) + + test('keeps tool-result documents visible as text references in the tool message', () => { + const req: AnthropicRequest = { + model: 'gpt-4', + max_tokens: 100, + messages: [{ + role: 'user', + content: [{ + type: 'tool_result', + tool_use_id: 'tc_doc', + content: [ + { type: 'document', title: 'report.pdf', source: { type: 'url', url: 'https://example.test/report.pdf' } }, + { type: 'document', title: 'notes.txt', source: { type: 'base64', media_type: 'text/plain', data: 'aGVsbG8=' } }, + { type: 'document', title: 'readme', source: { type: 'text', media_type: 'text/plain', data: 'hello world' } }, + { type: 'document', title: 'secret.pdf', source: { type: 'file', file_id: 'file_123' } }, + ], + }], + }], + } + + // Textual document representations stay inside the tool message (external + // tool output belongs behind the tool role); only real media lifts. + expect(anthropicToOpenaiChat(req).messages).toEqual([ + { + role: 'tool', + tool_call_id: 'tc_doc', + content: '[Document: report.pdf](https://example.test/report.pdf)\n[Document: notes.txt]\nhello\n[Document: readme]\nhello world\n[Document: secret.pdf omitted — file-based source]', + }, + ]) + }) + + test('keeps tool-result search results as text', () => { + const req: AnthropicRequest = { + model: 'gpt-4', + max_tokens: 100, + messages: [{ + role: 'user', + content: [{ + type: 'tool_result', + tool_use_id: 'tc_search', content: [ - { type: 'text', text: 'before' }, { - type: 'image', - source: { type: 'base64', media_type: 'image/jpeg', data: '/9j/AA==' }, + type: 'search_result', + title: 'Result title', + content: [{ type: 'text', text: 'Snippet body' }], + source: 'https://example.test/result', + }, + ], + }], + }], + } + + expect(anthropicToOpenaiChat(req).messages).toEqual([ + { role: 'tool', tool_call_id: 'tc_search', content: 'Result title — Snippet body — https://example.test/result' }, + ]) + }) + + test('degrades file-based image sources to a visible text notice', () => { + const req: AnthropicRequest = { + model: 'gpt-4', + max_tokens: 100, + messages: [{ + role: 'user', + content: [{ + type: 'tool_result', + tool_use_id: 'tc_file', + content: [ + { type: 'image', source: { type: 'file', file_id: 'file-123' } }, + ], + }], + }], + } + + expect(anthropicToOpenaiChat(req).messages).toEqual([ + { role: 'tool', tool_call_id: 'tc_file', content: '\n[Image omitted: file-based image source is not supported by this endpoint.]\n' }, + ]) + }) + + test('keeps top-level user search results as text', () => { + const req: AnthropicRequest = { + model: 'gpt-4', + max_tokens: 100, + messages: [{ + role: 'user', + content: [ + { + type: 'search_result', + title: 'Top result', + content: [{ type: 'text', text: 'Top snippet' }], + source: 'https://example.test/top', + }, + ], + }], + } + + expect(anthropicToOpenaiChat(req).messages).toEqual([ + { role: 'user', content: 'Top result — Top snippet — https://example.test/top' }, + ]) + }) + + test('flattens custom-content documents as text', () => { + const req: AnthropicRequest = { + model: 'gpt-4', + max_tokens: 100, + messages: [{ + role: 'user', + content: [{ + type: 'tool_result', + tool_use_id: 'tc_cdoc', + content: [ + { + type: 'document', + title: 'cited doc', + source: { + type: 'content', + content: [{ type: 'text', text: 'quoted passage' }], + }, + }, + ], + }], + }], + } + + expect(anthropicToOpenaiChat(req).messages).toEqual([ + { role: 'tool', tool_call_id: 'tc_cdoc', content: '[Document: cited doc]\nquoted passage' }, + ]) + }) + + test('keeps inline images of custom-content documents in tool-result documents', () => { + const req: AnthropicRequest = { + model: 'gpt-4', + max_tokens: 100, + messages: [{ + role: 'user', + content: [{ + type: 'tool_result', + tool_use_id: 'tc_cdoc', + content: [{ + type: 'document', + title: 'cited', + source: { + type: 'content', + content: [ + { type: 'text', text: 'before' }, + { type: 'image', source: { type: 'base64', media_type: 'image/png', data: 'abc' } }, + { type: 'text', text: 'after' }, + ], + }, + }], + }], + }], + } + + expect(anthropicToOpenaiChat(req).messages).toEqual([ + // The document's text blocks keep their boundaries without a separator; + // the synthetic title prefix carries its own newline. + { role: 'tool', tool_call_id: 'tc_cdoc', content: '[Document: cited]\nbeforeafter' }, + { + role: 'user', + content: [ + { type: 'text', text: '[Media content for tool call tc_cdoc]' }, + { type: 'image_url', image_url: { url: 'data:image/png;base64,abc' } }, + ], + }, + ]) + }) + + test('keeps model-visible title and context when degrading documents', () => { + const req: AnthropicRequest = { + model: 'gpt-4', + max_tokens: 100, + messages: [{ + role: 'user', + content: [{ + type: 'tool_result', + tool_use_id: 'tc_prov', + content: [ + { + type: 'document', + title: 'Auth specification', + context: 'The examples use production credentials', + source: { type: 'text', media_type: 'text/plain', data: 'Bearer abc123' }, + }, + { + type: 'document', + title: 'Policy', + context: 'Applies to tenant A', + source: { type: 'content', content: [{ type: 'text', text: 'quoted' }] }, }, - { type: 'text', text: 'after' }, ], }], }], @@ -371,43 +677,160 @@ describe('anthropicToOpenaiChat', () => { const result = anthropicToOpenaiChat(req) - expect(result.messages[0].content).toEqual([ - { type: 'text', text: 'before' }, - { type: 'image_url', image_url: { url: 'data:image/jpeg;base64,/9j/AA==' } }, - { type: 'text', text: 'after' }, + expect(result.messages).toEqual([ + { + role: 'tool', + tool_call_id: 'tc_prov', + content: '[Document: Auth specification]\n[Document context: The examples use production credentials]\nBearer abc123\n[Document: Policy]\n[Document context: Applies to tenant A]\nquoted', + }, ]) }) - test('replaces nested tool_result images in text-only mode without leaking base64', () => { + test('keeps model-visible title and context for top-level user documents', () => { const req: AnthropicRequest = { - model: 'deepseek-v4-pro', + model: 'gpt-4', + max_tokens: 100, + messages: [{ + role: 'user', + content: [ + { + type: 'document', + title: 'Auth specification', + context: 'The examples use production credentials', + source: { type: 'text', media_type: 'text/plain', data: 'Bearer abc123' }, + }, + { + type: 'document', + title: 'Contract', + context: 'Applies to tenant A', + source: { type: 'url', url: 'https://example.test/contract.pdf' }, + }, + ], + }], + } + + const result = anthropicToOpenaiChat(req) + + expect(result.messages).toEqual([ + { + role: 'user', + content: [ + { type: 'text', text: '[Document: Auth specification]\n[Document context: The examples use production credentials]\nBearer abc123' }, + // The URL reference already carries the title as its label, so only + // the context gets a synthetic prefix. + { type: 'text', text: '[Document context: Applies to tenant A]\n[Document: Contract](https://example.test/contract.pdf)' }, + ], + }, + ]) + }) + + test('keeps inline images of custom-content documents in top-level user content', () => { + const req: AnthropicRequest = { + model: 'gpt-4', max_tokens: 100, messages: [{ role: 'user', content: [{ - type: 'tool_result', - tool_use_id: 'computer_3', - content: [ - { type: 'text', text: 'before' }, - { - type: 'image', - source: { type: 'base64', media_type: 'image/jpeg', data: 'private-screenshot-data' }, - }, - { type: 'text', text: 'after' }, - ], + type: 'document', + title: 'cited', + source: { + type: 'content', + content: [ + { type: 'text', text: 'before' }, + { type: 'image', source: { type: 'url', url: 'https://example.test/img.png' } }, + ], + }, }], }], } - const result = anthropicToOpenaiChat(req, { imageContentMode: 'text_only' }) - - expect(result.messages[0].content).toBe( - 'before\n[Image omitted: this OpenAI-compatible chat endpoint only supports text content.]\nafter', - ) - expect(JSON.stringify(result)).not.toContain('private-screenshot-data') - expect(JSON.stringify(result)).not.toContain('image_url') + expect(anthropicToOpenaiChat(req).messages).toEqual([ + { + role: 'user', + content: [ + { type: 'text', text: '[Document: cited]\n' }, + { type: 'text', text: 'before' }, + { type: 'image_url', image_url: { url: 'https://example.test/img.png' } }, + ], + }, + ]) }) + test('keeps the interleaved text/image order of custom-content documents', () => { + const req: AnthropicRequest = { + model: 'gpt-4', + max_tokens: 100, + messages: [{ + role: 'user', + content: [{ + type: 'document', + title: 'cited', + source: { + type: 'content', + content: [ + { type: 'text', text: 'before' }, + { type: 'image', source: { type: 'base64', media_type: 'image/png', data: 'a' } }, + { type: 'text', text: 'after' }, + ], + }, + }], + }], + } + + expect(anthropicToOpenaiChat(req).messages).toEqual([ + { + role: 'user', + content: [ + { type: 'text', text: '[Document: cited]\n' }, + { type: 'text', text: 'before' }, + { type: 'image_url', image_url: { url: 'data:image/png;base64,a' } }, + { type: 'text', text: 'after' }, + ], + }, + ]) + }) + + test('degrades top-level file-based user images to a visible text notice', () => { + const req: AnthropicRequest = { + model: 'gpt-4', + max_tokens: 100, + messages: [{ + role: 'user', + content: [ + { type: 'text', text: 'look at this' }, + { type: 'image', source: { type: 'file', file_id: 'file-9' } }, + ], + }], + } + + expect(anthropicToOpenaiChat(req).messages).toEqual([ + { role: 'user', content: [ + { type: 'text', text: 'look at this' }, + { type: 'text', text: '\n[Image omitted: file-based image source is not supported by this endpoint.]\n' }, + ] }, + ]) + }) + + test('keeps user text after tool results after tool messages', () => { + const req: AnthropicRequest = { + model: 'gpt-4', + max_tokens: 100, + messages: [{ + role: 'user', + content: [ + { type: 'text', text: 'before' }, + { type: 'tool_result', tool_use_id: 'tc_1', content: 'result' }, + { type: 'text', text: 'after' }, + ], + }], + } + + expect(anthropicToOpenaiChat(req).messages).toEqual([ + { role: 'user', content: 'before' }, + { role: 'tool', tool_call_id: 'tc_1', content: 'result' }, + { role: 'user', content: 'after' }, + ]) + }) test('image content conversion', () => { const req: AnthropicRequest = { model: 'gpt-4', @@ -439,11 +862,236 @@ describe('anthropicToOpenaiChat', () => { } const result = anthropicToOpenaiChat(req, { imageContentMode: 'text_only' }) expect(result.messages[0].content).toBe( - 'What is in this screenshot?\n[Image omitted: this OpenAI-compatible chat endpoint only supports text content.]', + 'What is in this screenshot?\n[Image omitted: this OpenAI-compatible chat endpoint only supports text content.]\n', ) expect(JSON.stringify(result)).not.toContain('image_url') expect(JSON.stringify(result)).not.toContain('abc123') }) + + test('text-only mode emits one omission notice per image-only tool result', () => { + const req: AnthropicRequest = { + model: 'deepseek-v4-pro', + max_tokens: 100, + messages: [{ + role: 'user', + content: [{ + type: 'tool_result', + tool_use_id: 'tc_img', + content: [{ + type: 'image', + source: { type: 'base64', media_type: 'image/png', data: 'abc123' }, + }], + }], + }], + } + const result = anthropicToOpenaiChat(req, { imageContentMode: 'text_only' }) + expect(result.messages).toEqual([ + { role: 'tool', tool_call_id: 'tc_img', content: '\n[Image omitted: this OpenAI-compatible chat endpoint only supports text content.]\n' }, + ]) + expect(JSON.stringify(result)).not.toContain('image_url') + expect(JSON.stringify(result)).not.toContain('abc123') + }) + + test('text-only mode keeps plain-text documents instead of omitting them', () => { + const req: AnthropicRequest = { + model: 'deepseek-v4-pro', + max_tokens: 100, + messages: [{ + role: 'user', + content: [{ + type: 'tool_result', + tool_use_id: 'tc_doc', + content: [ + { type: 'document', title: 'notes.txt', source: { type: 'text', media_type: 'text/plain', data: 'actual tool result' } }, + { + type: 'document', + title: 'cited', + source: { type: 'content', content: [{ type: 'text', text: 'quoted passage' }] }, + }, + ], + }], + }], + } + const result = anthropicToOpenaiChat(req, { imageContentMode: 'text_only' }) + expect(result.messages).toEqual([ + { role: 'tool', tool_call_id: 'tc_doc', content: '[Document: notes.txt]\nactual tool result\n[Document: cited]\nquoted passage' }, + ]) + expect(JSON.stringify(result)).not.toContain('image_url') + }) + + test('text-only mode degrades inline document images to a text notice', () => { + const req: AnthropicRequest = { + model: 'deepseek-v4-pro', + max_tokens: 100, + messages: [{ + role: 'user', + content: [{ + type: 'document', + title: 'cited', + source: { + type: 'content', + content: [ + { type: 'text', text: 'before' }, + { type: 'image', source: { type: 'base64', media_type: 'image/png', data: 'abc123' } }, + { type: 'text', text: 'after' }, + ], + }, + }], + }], + } + const result = anthropicToOpenaiChat(req, { imageContentMode: 'text_only' }) + expect(result.messages).toEqual([ + { + role: 'user', + content: '[Document: cited]\nbefore\n[Image omitted: this OpenAI-compatible chat endpoint only supports text content.]\nafter', + }, + ]) + expect(JSON.stringify(result)).not.toContain('image_url') + expect(JSON.stringify(result)).not.toContain('abc123') + }) + + test('text-only mode keeps explicit boundaries around top-level documents', () => { + const req: AnthropicRequest = { + model: 'deepseek-v4-pro', + max_tokens: 100, + messages: [{ + role: 'user', + content: [ + { type: 'text', text: 'before' }, + { type: 'document', title: 'spec', source: { type: 'text', media_type: 'text/plain', data: 'DOC' } }, + { type: 'text', text: 'after' }, + ], + }], + } + const result = anthropicToOpenaiChat(req, { imageContentMode: 'text_only' }) + // Raw text blocks keep their exact bytes, but a document is a discrete + // block: its degraded text must not glue itself to the surrounding text. + expect(result.messages).toEqual([ + { role: 'user', content: 'before\n[Document: spec]\nDOC\nafter' }, + ]) + }) + + test('text-only mode keeps explicit boundaries around top-level search results', () => { + const req: AnthropicRequest = { + model: 'deepseek-v4-pro', + max_tokens: 100, + messages: [{ + role: 'user', + content: [ + { type: 'text', text: 'before' }, + { + type: 'search_result', + title: 'Result title', + content: [{ type: 'text', text: 'Snippet body' }], + source: 'https://example.test/result', + }, + { type: 'text', text: 'after' }, + ], + }], + } + const result = anthropicToOpenaiChat(req, { imageContentMode: 'text_only' }) + expect(result.messages).toEqual([ + { role: 'user', content: 'before\nResult title — Snippet body — https://example.test/result\nafter' }, + ]) + }) + + test('keeps tool-result search results separate from adjacent text', () => { + const req: AnthropicRequest = { + model: 'gpt-4', + max_tokens: 100, + messages: [{ + role: 'user', + content: [{ + type: 'tool_result', + tool_use_id: 'tc_search', + content: [ + { type: 'text', text: 'before' }, + { + type: 'search_result', + title: 'Result title', + content: [{ type: 'text', text: 'Snippet body' }], + source: 'https://example.test/result', + }, + { type: 'text', text: 'after' }, + ], + }], + }], + } + + expect(anthropicToOpenaiChat(req).messages).toEqual([ + { role: 'tool', tool_call_id: 'tc_search', content: 'before\nResult title — Snippet body — https://example.test/result\nafter' }, + ]) + }) + + test('text-only mode does not duplicate line boundaries the text already provides', () => { + const req: AnthropicRequest = { + model: 'deepseek-v4-pro', + max_tokens: 100, + messages: [{ + role: 'user', + content: [ + { type: 'text', text: 'before\n' }, + { type: 'document', title: 'spec', source: { type: 'text', media_type: 'text/plain', data: 'DOC' } }, + { type: 'text', text: '\nafter' }, + ], + }], + } + const result = anthropicToOpenaiChat(req, { imageContentMode: 'text_only' }) + // The serializer must not rewrite bytes the prompt already carries: a + // single boundary newline, no extra blank line. + expect(result.messages).toEqual([ + { role: 'user', content: 'before\n[Document: spec]\nDOC\nafter' }, + ]) + }) + + test('does not duplicate line boundaries after tool-result documents', () => { + const req: AnthropicRequest = { + model: 'gpt-4', + max_tokens: 100, + messages: [{ + role: 'user', + content: [{ + type: 'tool_result', + tool_use_id: 'tc_doc', + content: [ + { type: 'document', title: 'spec', source: { type: 'text', media_type: 'text/plain', data: 'DOC' } }, + { type: 'text', text: '\nafter' }, + ], + }], + }], + } + + expect(anthropicToOpenaiChat(req).messages).toEqual([ + { role: 'tool', tool_call_id: 'tc_doc', content: '[Document: spec]\nDOC\nafter' }, + ]) + }) + + test('does not duplicate line boundaries after tool-result search results', () => { + const req: AnthropicRequest = { + model: 'gpt-4', + max_tokens: 100, + messages: [{ + role: 'user', + content: [{ + type: 'tool_result', + tool_use_id: 'tc_search', + content: [ + { + type: 'search_result', + title: 'Result title', + content: [{ type: 'text', text: 'Snippet body' }], + source: 'https://example.test/result', + }, + { type: 'text', text: '\nafter' }, + ], + }], + }], + } + + expect(anthropicToOpenaiChat(req).messages).toEqual([ + { role: 'tool', tool_call_id: 'tc_search', content: 'Result title — Snippet body — https://example.test/result\nafter' }, + ]) + }) }) // ─── openaiChatToAnthropic ────────────────────────────────────── @@ -683,6 +1331,25 @@ describe('anthropicToOpenaiResponses', () => { }]) }) + test('normalizes empty tool_result arrays to an empty string', () => { + const req: AnthropicRequest = { + model: 'gpt-4o', + max_tokens: 100, + messages: [{ + role: 'user', + content: [{ type: 'tool_result', tool_use_id: 'tc_empty', content: [] }], + }], + } + + const result = anthropicToOpenaiResponses(req) + + expect(result.input).toEqual([{ + type: 'function_call_output', + call_id: 'tc_empty', + output: '', + }]) + }) + test('preserves text-only tool_result arrays as strings', () => { const req: AnthropicRequest = { model: 'gpt-4o', @@ -705,7 +1372,9 @@ describe('anthropicToOpenaiResponses', () => { expect(result.input).toEqual([{ type: 'function_call_output', call_id: 'tc_2', - output: 'first\nsecond', + // Adjacent text blocks are joined without a separator — the wire shape + // carries no newline between them. + output: 'firstsecond', }]) }) @@ -772,6 +1441,223 @@ describe('anthropicToOpenaiResponses', () => { }]) }) + test('preserves image URL sources in tool_result content', () => { + const req: AnthropicRequest = { + model: 'gpt-4o', + max_tokens: 100, + messages: [{ + role: 'user', + content: [{ + type: 'tool_result', + tool_use_id: 'read_url', + content: [{ + type: 'image', + source: { type: 'url', url: 'https://example.test/shot.png' }, + }], + }], + }], + } + + const result = anthropicToOpenaiResponses(req) + + expect(result.input).toEqual([{ + type: 'function_call_output', + call_id: 'read_url', + output: [{ + type: 'input_image', + image_url: 'https://example.test/shot.png', + }], + }]) + }) + + test('maps document blocks to input_file in tool_result content', () => { + const req: AnthropicRequest = { + model: 'gpt-4o', + max_tokens: 100, + messages: [{ + role: 'user', + content: [{ + type: 'tool_result', + tool_use_id: 'read_doc', + content: [ + { + type: 'document', + title: 'report.pdf', + source: { type: 'base64', media_type: 'application/pdf', data: 'pdf-data' }, + }, + { + type: 'document', + source: { type: 'url', url: 'https://example.test/report.pdf' }, + }, + ], + }], + }], + } + + const result = anthropicToOpenaiResponses(req) + + expect(result.input).toEqual([{ + type: 'function_call_output', + call_id: 'read_doc', + output: [ + // The titled base64 document keeps its model-visible title as a + // synthetic prefix ahead of the input_file part. + { type: 'input_text', text: '[Document: report.pdf]\n' }, + { type: 'input_file', file_data: 'data:application/pdf;base64,pdf-data', filename: 'report.pdf' }, + { type: 'input_file', file_url: 'https://example.test/report.pdf' }, + ], + }]) + }) + + test('maps search_result blocks to input_text in tool_result content', () => { + const req: AnthropicRequest = { + model: 'gpt-4o', + max_tokens: 100, + messages: [{ + role: 'user', + content: [{ + type: 'tool_result', + tool_use_id: 'read_search', + content: [ + { + type: 'search_result', + title: 'Result title', + content: [{ type: 'text', text: 'Snippet body' }], + source: 'https://example.test/result', + }, + ], + }], + }], + } + + const result = anthropicToOpenaiResponses(req) + + expect(result.input).toEqual([{ + type: 'function_call_output', + call_id: 'read_search', + // A search_result is not a text block: the output keeps its array shape + // instead of flattening into a joined string. + output: [{ type: 'input_text', text: 'Result title — Snippet body — https://example.test/result' }], + }]) + }) + + test('keeps text documents as input_text instead of base64 in tool_result content', () => { + const req: AnthropicRequest = { + model: 'gpt-4o', + max_tokens: 100, + messages: [{ + role: 'user', + content: [{ + type: 'tool_result', + tool_use_id: 'read_textdoc', + content: [ + { type: 'document', title: 'notes.txt', source: { type: 'text', media_type: 'text/plain', data: 'hello world' } }, + ], + }], + }], + } + + const result = anthropicToOpenaiResponses(req) + + expect(result.input).toEqual([{ + type: 'function_call_output', + call_id: 'read_textdoc', + // A document is not a text block: the output keeps its array shape + // instead of flattening into a joined string. The model-visible title + // stays as a synthetic prefix. + output: [{ type: 'input_text', text: '[Document: notes.txt]\nhello world' }], + }]) + }) + + test('keeps inline images of custom-content documents in tool_result content', () => { + const req: AnthropicRequest = { + model: 'gpt-4o', + max_tokens: 100, + messages: [{ + role: 'user', + content: [{ + type: 'tool_result', + tool_use_id: 'read_cdoc', + content: [{ + type: 'document', + title: 'cited', + source: { + type: 'content', + content: [ + { type: 'text', text: 'before' }, + { type: 'image', source: { type: 'base64', media_type: 'image/png', data: 'abc' } }, + { type: 'text', text: 'after' }, + ], + }, + }], + }], + }], + } + + const result = anthropicToOpenaiResponses(req) + + expect(result.input).toEqual([{ + type: 'function_call_output', + call_id: 'read_cdoc', + output: [ + { type: 'input_text', text: '[Document: cited]\n' }, + { type: 'input_text', text: 'before' }, + { type: 'input_image', image_url: 'data:image/png;base64,abc' }, + { type: 'input_text', text: 'after' }, + ], + }]) + }) + + test('degrades file-based image sources to a visible text notice', () => { + const req: AnthropicRequest = { + model: 'gpt-4o', + max_tokens: 100, + messages: [{ + role: 'user', + content: [{ + type: 'tool_result', + tool_use_id: 'read_file', + content: [ + { type: 'image', source: { type: 'file', file_id: 'file-123' } }, + ], + }], + }], + } + + const result = anthropicToOpenaiResponses(req) + + expect(result.input).toEqual([{ + type: 'function_call_output', + call_id: 'read_file', + output: [{ type: 'input_text', text: '[Image omitted: file-based image source is not supported by this endpoint.]' }], + }]) + }) + + test('preserves image URL sources in ordinary message content', () => { + const req: AnthropicRequest = { + model: 'gpt-4o', + max_tokens: 100, + messages: [{ + role: 'user', + content: [{ + type: 'image', + source: { type: 'url', url: 'https://example.test/photo.png' }, + }], + }], + } + + const result = anthropicToOpenaiResponses(req) + + expect(result.input).toEqual([{ + type: 'message', + role: 'user', + content: [{ + type: 'input_image', + image_url: 'https://example.test/photo.png', + }], + }]) + }) + test('preserves message and tool_result item order', () => { const req: AnthropicRequest = { model: 'gpt-4o', diff --git a/src/server/proxy/handler.ts b/src/server/proxy/handler.ts index 21a954ab..fc758c6b 100644 --- a/src/server/proxy/handler.ts +++ b/src/server/proxy/handler.ts @@ -9,10 +9,14 @@ * Original work by Jason Young, MIT License */ +import { createGunzip, createInflate } from 'node:zlib' + import { ProviderService } from '../services/providerService.js' +import type { ProviderAuthStrategy } from '../types/provider.js' import { resolvePromptCacheKey } from './promptCacheKey.js' import { anthropicToOpenaiChat } from './transform/anthropicToOpenaiChat.js' import { anthropicToOpenaiResponses } from './transform/anthropicToOpenaiResponses.js' +import { hoistToolResultMediaForCompatibility } from './transform/anthropicMediaHoist.js' import { openaiChatToAnthropic } from './transform/openaiChatToAnthropic.js' import { openaiResponsesToAnthropic } from './transform/openaiResponsesToAnthropic.js' import { openaiChatStreamToAnthropic } from './streaming/openaiChatStreamToAnthropic.js' @@ -38,7 +42,8 @@ import { resolveModelReasoningProfile } from '../../shared/modelReasoning.js' const providerService = new ProviderService() type ProxyFetchOptions = ReturnType -type UpstreamRequestInit = RequestInit & ProxyFetchOptions +// `decompress` is a Bun fetch option absent from the DOM RequestInit type. +type UpstreamRequestInit = RequestInit & ProxyFetchOptions & { decompress?: boolean } type ProxyTraceContext = { sessionId: string provider: TraceProviderInfo @@ -47,6 +52,12 @@ type ProxyTraceContext = { const TRACE_RECORDED_ERROR_MARKER = Symbol('cc-haha-trace-recorded-error') +// Per-context dedup for failures that rethrow a value that cannot carry a +// marker (stream errors may be any value, e.g. a string from +// `controller.error('...')`). The marker above still covers Error objects for +// paths that only see object throws. +const recordedTraceErrorContexts = new WeakSet() + function markTraceErrorRecorded(error: unknown): void { if (error && typeof error === 'object') { try { @@ -148,6 +159,9 @@ export function withStreamIdleTimeout( } catch (err) { clearIdleTimer() if (!timedOut) controller.error(err) + } finally { + reader?.releaseLock() + reader = null } }, cancel(reason) { @@ -190,22 +204,8 @@ export async function handleProxyRequest(req: Request, url: URL): Promise { + switch (authStrategy) { + case 'auth_token': + case 'auth_token_empty_api_key': + return apiKey ? { Authorization: `Bearer ${apiKey}` } : {} + case 'dual_same_token': + return apiKey ? { 'x-api-key': apiKey, Authorization: `Bearer ${apiKey}` } : {} + case 'dual_dummy': + return { 'x-api-key': 'dummy', Authorization: 'Bearer dummy' } + case 'api_key': + default: + return apiKey ? { 'x-api-key': apiKey } : {} + } +} + +// Connection-management headers that must not be forwarded between hops (they +// describe the client↔proxy hop, not the proxy↔upstream hop). Per RFC 9110 the +// `Connection` header may also name additional connection-specific headers, +// which are added to the deny set dynamically. +const HOP_BY_HOP_HEADERS = new Set([ + 'connection', + 'keep-alive', + 'proxy-authenticate', + 'proxy-authorization', + 'te', + 'trailer', + 'transfer-encoding', + 'upgrade', + 'proxy-connection', +]) + +const INTERNAL_CLIENT_HEADERS = new Set([ + 'x-claude-code-session-id', + 'x-claude-remote-container-id', + 'x-claude-remote-session-id', + 'x-client-app', +]) + +function isInternalClientHeader(name: string, value: string): boolean { + if (INTERNAL_CLIENT_HEADERS.has(name)) return true + if (name === 'x-app') return value === 'cli' + return name === 'user-agent' && /^claude-cli\/[^\s]+\s+\(/i.test(value) +} + +function hopByHopDenySet(headers: Headers): Set { + const deny = new Set(HOP_BY_HOP_HEADERS) + const connection = headers.get('connection') + if (connection) { + for (const token of connection.split(',')) { + const trimmed = token.trim().toLowerCase() + if (trimmed) deny.add(trimmed) + } + } + return deny +} + +/** + * Copy a response's entity headers minus hop-by-hop headers. The upstream's + * `Connection`-scoped headers describe the proxy↔upstream hop and must not + * leak into the proxy↔client hop. + */ +/** + * Parse a `Content-Encoding` header into the codecs to unwind, in decoding + * order. `Content-Encoding` lists encodings in application order, so decoding + * unwinds them in reverse; `identity` is a no-op. + */ +function parseContentEncodings(contentEncoding: string | undefined): string[] { + return (contentEncoding ?? '') + .split(',') + .map(encoding => encoding.trim().toLowerCase()) + .filter(encoding => encoding !== '' && encoding !== 'identity') + .reverse() +} + +/** Codecs the trace decoder can unwind. Anything else (br, zstd, stacked + * combinations) is marked unavailable instead of decoding raw bytes as UTF-8. */ +const SUPPORTED_TRACE_CODECS = new Set(['gzip', 'x-gzip', 'deflate']) + +/** + * Decode captured upstream bytes for trace storage. The passthrough keeps the + * raw bytes (`decompress: false`) so Content-Encoding/Length stay valid for + * the client, but the trace should store readable text — decompress a copy + * when the upstream compressed the body. + * + * Unknown encodings (for example `br`) and failed decompression of a known + * codec are both marked unavailable — trace capture must never fail the + * request, but it must not store bytes decoded as UTF-8 either. + */ +function decodeTraceBytes(bytes: Uint8Array, contentEncoding: string | undefined): string { + const encodings = parseContentEncodings(contentEncoding) + // An unknown codec (br, zstd, …) or a stacked combination cannot be + // unwound — mark the trace unavailable instead of storing compressed bytes + // decoded as UTF-8, matching the streaming branch. + if (encodings.some(codec => !SUPPORTED_TRACE_CODECS.has(codec))) { + return '[trace body unavailable: unsupported content encoding]' + } + // Copy into an ArrayBuffer-backed view: Bun's sync decompressors require + // Uint8Array (a view over a non-shared buffer). + let data: Uint8Array = new Uint8Array(bytes) + for (const encoding of encodings) { + try { + if (encoding === 'gzip' || encoding === 'x-gzip') { + data = Bun.gunzipSync(data) + } else if (encoding === 'deflate') { + data = Bun.inflateSync(data) + } else { + return '[trace body unavailable: unsupported content encoding]' + } + } catch { + // The header names a codec the decoder supports, but the body is not + // valid for it (corrupt member, truncated stream). Decoding whatever + // was unwound so far as UTF-8 would store binary garbage — mark the + // trace unavailable instead. + return '[trace body unavailable: decompression failed]' + } + } + return new TextDecoder().decode(data) +} + +function decodeTraceResponseBody(bytes: ArrayBuffer, headers: Headers | undefined): string { + return decodeTraceBytes(new Uint8Array(bytes), headers?.get('content-encoding') ?? undefined) +} + +function stripHopByHopHeaders(headers: Headers): Headers { + const deny = hopByHopDenySet(headers) + const stripped = new Headers() + for (const [name, value] of headers.entries()) { + if (!deny.has(name.toLowerCase())) stripped.set(name, value) + } + return stripped +} + +/** + * Forward an Anthropic Messages request to an anthropic-format upstream after + * lifting media out of nested tool results (provider opted out of nested media). + * The wire format stays Anthropic; only the media placement changes. Protocol + * headers and error responses are passed through so SDK classification and + * retry behavior are preserved. + */ +async function handleAnthropicCompatible( + body: AnthropicRequest, + baseUrl: string, + apiKey: string, + authStrategy: ProviderAuthStrategy, + incomingHeaders: Headers, + isStream: boolean, + networkSettings: NetworkSettings, + traceContext: ProxyTraceContext | null, +): Promise { + const transformed = hoistToolResultMediaForCompatibility(body) + const url = `${baseUrl}/v1/messages` + const proxyOptions = getNetworkProxyFetchOptions(networkSettings, url) + + const headers: Record = { + 'Content-Type': 'application/json', + ...buildAnthropicAuthHeaders(apiKey, authStrategy), + } + // Preserve protocol and custom headers from the incoming request + // (anthropic-version is required; anthropic-beta and custom headers such as + // those injected via ANTHROPIC_CUSTOM_HEADERS carry real semantics for the + // upstream endpoint). Hop-by-hop and auth headers are not forwarded. + const deny = hopByHopDenySet(incomingHeaders) + for (const [name, value] of incomingHeaders.entries()) { + const lower = name.toLowerCase() + if (deny.has(lower) || isInternalClientHeader(lower, value)) continue + if (lower === 'content-type' || lower === 'content-length') continue + if (lower === 'x-api-key' || lower === 'authorization') continue + if (value) headers[name] = value + } + + const traceHeaders = Object.fromEntries( + Object.entries(headers).map(([name, value]) => { + const lower = name.toLowerCase() + return [ + name, + lower === 'content-type' || lower === 'anthropic-version' || lower === 'anthropic-beta' + ? value + : '[redacted]', + ] + }), + ) + + const startedAtMs = Date.now() + const startedAt = new Date(startedAtMs).toISOString() + const traceCallId = traceContext + ? startProxyTraceCall({ + context: traceContext, + model: body.model, + upstreamUrl: url, + upstreamRequest: transformed, + requestHeaders: traceHeaders, + startedAt, + }) + : undefined + + // Close the pending trace started above when the upstream call fails, so the + // caller's unified error handling does not record a second trace for the + // same request. + const recordTraceError = (err: unknown): void => { + if (!traceContext) return + recordProxyTraceInBackground({ + callId: traceCallId, + context: traceContext, + model: body.model, + upstreamUrl: url, + upstreamRequest: transformed, + requestHeaders: traceHeaders, + startedAt, + startedAtMs, + error: err, + }) + markTraceErrorRecorded(err) + recordedTraceErrorContexts.add(traceContext) + } + + let upstream: Response + try { + upstream = await fetchUpstreamWithTimeout(url, { + method: 'POST', + headers, + body: JSON.stringify(transformed), + // Keep the raw bytes: Bun decompresses by default, which would leave a + // decompressed body behind the upstream Content-Encoding/Length headers + // when forwarding the response unchanged. + decompress: false, + ...proxyOptions, + }, networkSettings.aiRequestTimeoutMs, isStream) + } catch (err) { + recordTraceError(err) + console.error('[Proxy] Upstream anthropic request failed:', err) + return Response.json( + { + type: 'error', + error: { + type: 'api_error', + message: err instanceof Error ? err.message : String(err), + }, + }, + { status: 502 }, + ) + } + + try { + if (!upstream.ok) { + // Pass the upstream error body and headers through unchanged so the SDK + // keeps error classification (authentication_error, rate_limit_error, …), + // request_id, and retry-after semantics. A body read failure here closes + // the pending trace and surfaces to the caller's unified error handling + // (structured 502) like any other upstream failure. + const errBody = await upstream.arrayBuffer() + if (traceContext) { + recordProxyTraceInBackground({ + callId: traceCallId, + context: traceContext, + model: body.model, + upstreamUrl: url, + upstreamRequest: transformed, + requestHeaders: traceHeaders, + startedAt, + startedAtMs, + responseStatus: upstream.status, + upstreamResponseBody: decodeTraceResponseBody(errBody, upstream.headers), + responseHeaders: upstream.headers, + }) + } + return new Response(errBody, { + status: upstream.status, + headers: stripHopByHopHeaders(upstream.headers), + }) + } + + if (isStream) { + if (!upstream.body) { + if (traceContext) { + recordProxyTraceInBackground({ + callId: traceCallId, + context: traceContext, + model: body.model, + upstreamUrl: url, + upstreamRequest: transformed, + requestHeaders: traceHeaders, + startedAt, + startedAtMs, + error: new Error('Upstream returned no body for stream'), + }) + } + return Response.json( + { type: 'error', error: { type: 'api_error', message: 'Upstream returned no body for stream' } }, + { status: 502 }, + ) + } + // Keep SSE framing headers while passing through request/rate-limit + // metadata from the upstream (request_id, ratelimit-*, custom headers). + const responseHeaders = stripHopByHopHeaders(upstream.headers) + responseHeaders.set('Content-Type', 'text/event-stream') + responseHeaders.set('Cache-Control', 'no-cache') + responseHeaders.set('Connection', 'keep-alive') + const anthropicStream = withStreamIdleTimeout(upstream.body, networkSettings.aiRequestTimeoutMs) + const tracedStream = traceContext + ? captureTraceStream(anthropicStream, async (bodySnapshot, error) => { + await recordProxyTrace({ + callId: traceCallId, + context: traceContext, + model: body.model, + upstreamUrl: url, + upstreamRequest: transformed, + requestHeaders: traceHeaders, + startedAt, + startedAtMs, + responseStatus: 200, + responseBodySnapshot: bodySnapshot, + responseHeaders: upstream.headers, + ...(error ? { error } : {}), + }) + }, upstream.headers.get('content-encoding') ?? undefined) + : anthropicStream + return new Response(tracedStream, { + status: 200, + headers: responseHeaders, + }) + } + + // Byte-for-byte passthrough: re-serializing the body would invalidate + // Content-Length/ETag entity headers from the upstream. + const responseBody = await upstream.arrayBuffer() + if (traceContext) { + recordProxyTraceInBackground({ + callId: traceCallId, + context: traceContext, + model: body.model, + upstreamUrl: url, + upstreamRequest: transformed, + requestHeaders: traceHeaders, + startedAt, + startedAtMs, + responseStatus: upstream.status, + upstreamResponseBody: decodeTraceResponseBody(responseBody, upstream.headers), + responseHeaders: upstream.headers, + }) + } + return new Response(responseBody, { + status: upstream.status, + headers: stripHopByHopHeaders(upstream.headers), + }) + } catch (err) { + // A body read failure closes the pending trace with the original call id + // (so no trace stays pending and no second trace is created), then + // rethrows so the caller returns the same structured 502 as any other + // upstream failure. + recordTraceError(err) + throw err + } +} + async function handleOpenaiChat( body: AnthropicRequest, baseUrl: string, @@ -786,18 +1171,25 @@ async function recordProxyTrace({ function captureTraceStream( stream: ReadableStream, onComplete: (snapshot: TraceBodySnapshot, error?: unknown) => Promise, + contentEncoding?: string, ): ReadableStream { - const decoder = new TextDecoder() - let captured = '' + // The Anthropic passthrough forwards raw upstream bytes (`decompress: + // false`), so a compressed SSE body would otherwise be stored as binary + // garbage. Decompress the *trace copy* while it streams — the capture cap + // applies to decoded output, so truncation, client cancellation, or an + // upstream error leave a readable plain-text prefix instead of an + // unterminated gzip member, and a highly compressible body cannot blow past + // the memory cap before it is counted. + const chunks: Uint8Array[] = [] let bytes = 0 let truncated = false let finalized = false let reader: ReadableStreamDefaultReader | null = null - const captureChunk = (chunk: Uint8Array) => { + const captureDecoded = (chunk: Uint8Array) => { bytes += chunk.byteLength if (bytes <= TRACE_STREAM_CAPTURE_BYTES) { - captured += decoder.decode(chunk, { stream: true }) + chunks.push(chunk) } else { truncated = true } @@ -806,11 +1198,108 @@ function captureTraceStream( const finalize = async (error?: unknown) => { if (finalized) return finalized = true - captured += decoder.decode() - const snapshot = createTraceBodySnapshot(captured, { alreadyTruncated: truncated }) + const joined = new Uint8Array(chunks.reduce((total, chunk) => total + chunk.byteLength, 0)) + let offset = 0 + for (const chunk of chunks) { + joined.set(chunk, offset) + offset += chunk.byteLength + } + const snapshot = createTraceBodySnapshot( + unsupportedEncoding + ? '[trace body unavailable: unsupported content encoding]' + : unexpectedDecompressionFailure + ? '[trace body unavailable: decompression failed]' + : new TextDecoder().decode(joined), + { alreadyTruncated: truncated }, + ) await onComplete(snapshot, error).catch(() => {}) } + // The streaming branch decodes a *single* known codec (gzip/x-gzip or + // deflate). Stacked or unknown encodings cannot be unwound here — mark the + // trace unavailable instead of storing compressed bytes decoded as UTF-8. + // The buffered path still unwinds every codec via decodeTraceBytes. + const encodings = parseContentEncodings(contentEncoding) + const singleKnownCodec = encodings.length === 1 && SUPPORTED_TRACE_CODECS.has(encodings[0]!) + const unsupportedEncoding = encodings.length > 0 && !singleKnownCodec + const decompressor = singleKnownCodec + ? encodings[0] === 'deflate' + ? createInflate() + : createGunzip() + : null + // node:zlib streams honor the Writable backpressure contract: write() + // returns false when the writable buffer is full and the caller must wait + // for 'drain' before writing more. The trace copy is a side channel, but it + // still must not buffer an unbounded amount of compressed input. + let decompressorFailed = false + let decompressorEnded = false + // Explicitly ended by design (client cancel, capture cap, upstream read + // error): an end() on an unterminated gzip member then errors as expected + // and the decoded plain-text prefix is kept. Only a body that fails to + // decompress while ending normally marks the trace unavailable — an error + // from zlib is delivered asynchronously, so "ended before the error" is not + // a reliable signal, but the ending path itself is. + let activelyEnded = false + let unexpectedDecompressionFailure = false + let decompressEnded: Promise = Promise.resolve() + if (decompressor) { + decompressor.on('data', captureDecoded) + // Partial data is already captured; an error mid-stream must not surface + // beyond the trace copy. + decompressor.on('error', () => { + if (!activelyEnded) { + unexpectedDecompressionFailure = true + } + decompressorFailed = true + }) + decompressEnded = new Promise(resolve => { + decompressor.on('end', resolve) + decompressor.on('error', resolve) + }) + } + + // Resolve when the decompressor is ready for more input. An errored stream + // never drains and rejects further writes, so failure also resolves — the + // loop must stop feeding it afterwards. An end() from the cancel/cap path + // may finish through 'finish'/'close' without ever emitting 'drain', so the + // waiter settles on any of the terminal events or the read loop would hang + // with the upstream reader lock never released. + const waitForDecompressorDrain = (): Promise => { + if (!decompressor || decompressorFailed || decompressorEnded) return Promise.resolve() + return new Promise(resolve => { + const settle = () => { + decompressor.off('drain', settle) + decompressor.off('error', settle) + decompressor.off('finish', settle) + decompressor.off('close', settle) + resolve() + } + decompressor.once('drain', settle) + decompressor.once('error', settle) + decompressor.once('finish', settle) + decompressor.once('close', settle) + }) + } + + // End the decompressor gracefully instead of destroying it: destroy() + // would drop data already written but not yet flushed as 'data', so a + // cancelled or errored stream would lose the decoded prefix. end() flushes + // what was received; an unterminated gzip member then errors, which + // resolves decompressEnded through the error branch. + const finishDecompressor = async () => { + if (!decompressor) return + if (!decompressorEnded) { + decompressorEnded = true + try { + decompressor.end() + } catch { + decompressor.destroy() + return + } + } + await decompressEnded + } + return new ReadableStream({ async start(controller) { reader = stream.getReader() @@ -818,13 +1307,33 @@ function captureTraceStream( while (true) { const { done, value } = await reader.read() if (done) break - captureChunk(value) controller.enqueue(value) + if (decompressor) { + if (!decompressorFailed && !decompressorEnded) { + if (truncated) { + // The decoded trace already exceeded the capture cap: stop + // feeding the decompressor so the trace side work cannot + // grow without bound or delay upstream cancellation. end() + // flushes the bytes already accepted, preserving the + // captured plain-text prefix. + decompressorEnded = true + activelyEnded = true + decompressor.end() + } else if (!decompressor.write(value)) { + await waitForDecompressorDrain() + } + } + } else if (!unsupportedEncoding) { + captureDecoded(value) + } } controller.close() + await finishDecompressor() void finalize() } catch (err) { controller.error(err) + activelyEnded = true + await finishDecompressor() void finalize(err) } finally { reader?.releaseLock() @@ -835,6 +1344,8 @@ function captureTraceStream( const error = reason instanceof Error ? reason : new Error(reason ? `Stream cancelled: ${String(reason)}` : 'Stream cancelled') + activelyEnded = true + await finishDecompressor() void finalize(error) await reader?.cancel(reason).catch(() => undefined) }, diff --git a/src/server/proxy/transform/anthropicMediaHoist.ts b/src/server/proxy/transform/anthropicMediaHoist.ts new file mode 100644 index 00000000..8f9820dc --- /dev/null +++ b/src/server/proxy/transform/anthropicMediaHoist.ts @@ -0,0 +1,234 @@ +/** + * Anthropic Messages compatibility transform for third-party endpoints that + * drop media nested inside `tool_result`. + * + * Only used when a provider explicitly configures + * `supportsNestedToolResultMedia: false`. The transform keeps every + * `tool_result` contiguous (Anthropic requires tool results to precede other + * content in a user turn) and lifts the images/documents out of each + * `tool_result`, placing them right after the last `tool_result` and before + * any trailing user text, so media stays ahead of the text that follows it. + * Each group is preceded by a marker naming the owning tool call, and groups + * keep their original order. Nested text stays in the `tool_result`. + * + * Note: media interleaved between text blocks inside one `tool_result` cannot + * keep its exact position in the Anthropic wire shape after lifting — this is + * inherent to the compatibility mode the provider opted into. + */ + +import type { + AnthropicRequest, + AnthropicMessage, + AnthropicContentBlock, + AnthropicDocumentContentTextBlock, + AnthropicDocumentSource, +} from './types.js' + +/** + * Server-executed tools appear in the transcript as `server_tool_use` blocks + * (built-in tools such as web_search) and `mcp_tool_use` blocks (the MCP + * connector), which share the same continuation semantics. Their results + * arrive as a `_tool_result` block (for example `web_search_tool_result` + * or `mcp_tool_result`) paired by `tool_use_id`. The API attaches it to the + * same assistant turn when the tool ran directly, or to the following + * assistant response when the call was mixed with client tools — never as a + * client `tool_result` in a user message. + */ +function isServerLikeToolUse( + block: AnthropicContentBlock, +): block is AnthropicContentBlock & { id: string } { + const candidate = block as { type?: unknown; id?: unknown } + return typeof candidate.id === 'string' + && (candidate.type === 'server_tool_use' || candidate.type === 'mcp_tool_use') +} + +function isServerToolResultBlock( + block: AnthropicContentBlock, +): block is AnthropicContentBlock & { tool_use_id: string } { + const candidate = block as { type?: unknown; tool_use_id?: unknown } + return typeof candidate.tool_use_id === 'string' + && typeof candidate.type === 'string' + && candidate.type.endsWith('_tool_result') +} + +/** + * True when an assistant turn contains a server-executed tool call (`server_tool_use` + * or `mcp_tool_use`) whose result has not arrived in the same turn. The user + * message that continues such a turn may only contain tool_result blocks, so + * media lifting would produce an invalid request. A deferred server tool that + * ran after the client returned its own tool_results arrives in the next + * assistant response and is not repeated, so the turn that follows it carries + * no pending server call at all and lifts normally. + */ +function isUnresolvedServerToolTurn(msg: AnthropicMessage): boolean { + if (typeof msg.content === 'string') return false + const serverIds = msg.content.filter(isServerLikeToolUse).map(block => block.id) + if (serverIds.length === 0) return false + const resolved = new Set( + msg.content.filter(isServerToolResultBlock).map(block => block.tool_use_id), + ) + return serverIds.some(id => !resolved.has(id)) +} + +function isTextOnlyDocumentContent( + source: Extract, +): source is Extract & { + content: string | AnthropicDocumentContentTextBlock[] +} { + return typeof source.content === 'string' + || source.content.every((block): block is AnthropicDocumentContentTextBlock => block.type === 'text') +} + +/** + * Model-visible document metadata (title/context) as synthetic prefix text. + * Both fields are visible to the model in the Anthropic protocol, so + * degradation keeps them instead of dropping them silently. Returns an empty + * string when neither is set, otherwise a newline-terminated prefix so the + * document body follows on its own line. + */ +function documentProvenanceText(document: { title?: string; context?: string }): string { + const lines = [ + ...(document.title ? [`[Document: ${document.title}]`] : []), + ...(document.context ? [`[Document context: ${document.context}]`] : []), + ] + return lines.length > 0 ? `${lines.join('\n')}\n` : '' +} + +function mediaMarker( + toolUseId: string, + media: Extract[], +): AnthropicContentBlock { + const images = media.filter(block => block.type === 'image').length + const documents = media.length - images + if (images > 0 && documents > 0) { + const imageLabel = images === 1 ? 'image' : 'images' + const documentLabel = documents === 1 ? 'document' : 'documents' + return { + type: 'text', + text: `[Media content for tool call ${toolUseId}: ${images} ${imageLabel}, ${documents} ${documentLabel}]`, + } + } + const label = images > 0 ? 'Image' : 'Document' + const count = media.length > 1 ? ` (${media.length})` : '' + return { + type: 'text', + text: `[${label} content for tool call ${toolUseId}${count}]`, + } +} + +export function hoistToolResultMediaForCompatibility( + body: AnthropicRequest, +): AnthropicRequest { + let changed = false + const messages = body.messages.map((msg, index) => { + if (msg.role !== 'user' || typeof msg.content === 'string') return msg + + // Anthropic requires a user message that continues an unresolved + // server-executed tool (`server_tool_use` or `mcp_tool_use`) to contain + // only tool_result blocks, so media lifting would risk a 400 for that one + // message. Earlier turns that already completed are not restricted and + // keep their media lifted. Merely declaring server-side tools in `tools` + // does not trigger this. + for (let i = index - 1; i >= 0; i--) { + const prev = body.messages[i] + if (prev.role !== 'assistant') continue + if (isUnresolvedServerToolTurn(prev)) return msg + break + } + + let messageChanged = false + const hoisted: AnthropicContentBlock[] = [] + const content = msg.content.map(block => { + if (block.type !== 'tool_result') return block + const inner = block.content + if (typeof inner === 'string' || !Array.isArray(inner)) return block + + const media: Extract[] = [] + const retained: AnthropicContentBlock[] = [] + let degraded = false + for (const part of inner) { + if (part.type === 'image') { + media.push(part) + } else if (part.type === 'document' && part.source.type === 'text') { + // Plain-text documents degrade to text inside the tool result, + // keeping their provenance instead of lifting them as user-level + // media. Title and context are model-visible metadata in the + // Anthropic protocol, so they stay visible as a prefix without + // altering the data; cache_control survives the degradation. + const provenance = documentProvenanceText(part) + const text = `${provenance}${part.source.data}` + retained.push({ type: 'text', text, ...(part.cache_control !== undefined ? { cache_control: part.cache_control } : {}) }) + degraded = true + } else if (part.type === 'document' && part.source.type === 'content' && isTextOnlyDocumentContent(part.source)) { + // Text-only custom-content documents degrade to text inside the + // tool result. The original text blocks keep their boundaries (no + // injected separators) and their own cache_control/citations; + // title/context become separate provenance blocks. The + // document-level cache_control attaches to the last degraded block + // — Anthropic prompt caching treats the marked block as the end of + // the cached prefix, so the breakpoint must sit after the + // document's content, not on the synthetic title — unless that + // block already carries an inner marker, which is preserved. + const degradedBlocks: AnthropicDocumentContentTextBlock[] = [] + if (part.title) degradedBlocks.push({ type: 'text', text: `[Document: ${part.title}]` }) + if (part.context) degradedBlocks.push({ type: 'text', text: `[Document context: ${part.context}]` }) + if (typeof part.source.content === 'string') { + degradedBlocks.push({ type: 'text', text: part.source.content }) + } else { + for (const block of part.source.content) { + degradedBlocks.push({ + type: 'text', + text: block.text, + ...(block.cache_control !== undefined ? { cache_control: block.cache_control } : {}), + ...(block.citations !== undefined ? { citations: block.citations } : {}), + }) + } + } + if (part.cache_control !== undefined && degradedBlocks.length > 0) { + const last = degradedBlocks.length - 1 + if (degradedBlocks[last].cache_control === undefined) { + degradedBlocks[last] = { ...degradedBlocks[last], cache_control: part.cache_control } + } + } + retained.push(...degradedBlocks) + degraded = true + } else if (part.type === 'document') { + media.push(part) + } else { + retained.push(part) + } + } + if (media.length === 0) { + if (!degraded) return block + messageChanged = true + return { ...block, content: retained } + } + + messageChanged = true + hoisted.push(mediaMarker(block.tool_use_id, media), ...media) + if (retained.length === 0) { + retained.push({ type: 'text', text: 'Media result attached after this tool result.' }) + } + return { ...block, content: retained } + }) + + if (!messageChanged) return msg + changed = true + + // Insert the lifted media right after the last tool_result so it stays + // ahead of trailing user text, matching the original content order. + const lastToolResultIndex = content.findLastIndex(block => block.type === 'tool_result') + const insertAt = lastToolResultIndex >= 0 ? lastToolResultIndex + 1 : content.length + return { + ...msg, + content: [ + ...content.slice(0, insertAt), + ...hoisted, + ...content.slice(insertAt), + ], + } + }) + + if (!changed) return body + return { ...body, messages } +} diff --git a/src/server/proxy/transform/anthropicToOpenaiChat.ts b/src/server/proxy/transform/anthropicToOpenaiChat.ts index c6c88e61..3b9e3640 100644 --- a/src/server/proxy/transform/anthropicToOpenaiChat.ts +++ b/src/server/proxy/transform/anthropicToOpenaiChat.ts @@ -19,6 +19,11 @@ import { normalizeOpenAIReasoningEffort } from './effort.js' type OpenAIChatImageContentMode = 'vision' | 'text_only' +// Synthetic text parts (degraded documents, search results) carry an internal +// marker so serializers can preserve their boundaries. The marker is removed +// before content parts reach the wire. +type UserTextPart = OpenAIChatContentPart & { synthetic?: boolean } + type OpenAIChatTransformOptions = { roundTripReasoningContent?: boolean passThinkingToggle?: boolean @@ -26,7 +31,13 @@ type OpenAIChatTransformOptions = { imageContentMode?: OpenAIChatImageContentMode } -const OMITTED_IMAGE_TEXT = '[Image omitted: this OpenAI-compatible chat endpoint only supports text content.]' +// Synthetic degradation text carries its own separators: the parts are +// joined without a separator, so a notice must not glue itself to the +// surrounding user text. +const OMITTED_IMAGE_TEXT = '\n[Image omitted: this OpenAI-compatible chat endpoint only supports text content.]\n' +const FILE_IMAGE_OMITTED_TEXT = '\n[Image omitted: file-based image source is not supported by this endpoint.]\n' +const MEDIA_RESULT_ATTACHED_TEXT = 'Media result attached after this tool result.' +const DOCUMENT_TEXT_INLINE_LIMIT = 2000 /** * Convert Anthropic Messages request to OpenAI Chat Completions request. @@ -149,85 +160,416 @@ function convertMessage( } } +/** + * Convert an Anthropic user message to OpenAI Chat messages. + * + * OpenAI Chat's official schema only allows text in `role: "tool"` messages, so + * tool-result media cannot stay inside the tool message. The conversion keeps + * the Anthropic content order as closely as the protocol allows: + * + * - ordinary user content (text/image/document) accumulates in one user message; + * - every tool_result becomes a text-only tool message; + * - tool-result media is lifted into one user message after the tool messages, + * grouped per tool call with a `tool_use_id` marker, preserving group order; + * - text after the last tool_result is emitted after the lifted media, so media + * stays ahead of the text that follows it (matching Anthropic's ordering). + */ function convertUserMessage( blocks: AnthropicContentBlock[], output: OpenAIChatMessage[], imageContentMode: OpenAIChatImageContentMode, ): void { - // Separate tool_result blocks from other content - const contentParts: OpenAIChatContentPart[] = [] - const textOnlyParts: string[] = [] + const leadingUserParts: Array = [] + const trailingUserParts: Array = [] + const toolMessages: OpenAIChatMessage[] = [] + const mediaGroups: Array<{ toolUseId: string; parts: OpenAIChatContentPart[] }> = [] + let sawToolResult = false for (const block of blocks) { if (block.type === 'text') { - if (imageContentMode === 'text_only') { - textOnlyParts.push(block.text) + const target = sawToolResult ? trailingUserParts : leadingUserParts + target.push({ type: 'text', text: block.text }) + continue + } + + if (block.type === 'image') { + const target = sawToolResult ? trailingUserParts : leadingUserParts + if (block.source.type === 'file') { + // Files API references cannot be forwarded to OpenAI-compatible endpoints. + target.push({ type: 'text', text: FILE_IMAGE_OMITTED_TEXT, synthetic: true }) } else { - contentParts.push({ type: 'text', text: block.text }) + target.push(imageContentMode === 'text_only' + ? { type: 'text', text: OMITTED_IMAGE_TEXT } + : { type: 'image_url', image_url: toImageUrl(block) }) } - } else if (block.type === 'image') { - if (imageContentMode === 'text_only') { - textOnlyParts.push(OMITTED_IMAGE_TEXT) + continue + } + + if (block.type === 'document') { + // Documents degrade to text where possible — plain text sources keep + // their data, custom-content documents keep their inline text, and the + // rest becomes a text reference — so text-only endpoints still receive + // the content. Only inline images are omitted in text_only mode. + const target = sawToolResult ? trailingUserParts : leadingUserParts + if (block.source.type === 'text') { + // Anthropic text documents carry plain text in `data` — not base64. + // Title/context are model-visible metadata, kept as a synthetic prefix. + target.push(syntheticText(`${documentProvenanceText(block)}${block.source.data}`, imageContentMode)) + } else if (block.source.type === 'content') { + const degraded = documentContentToParts(block) + if (imageContentMode === 'text_only') { + // Collapse the document's degraded parts into one synthetic part: + // the internal text keeps its exact bytes, while the document as a + // whole gets an explicit boundary against adjacent raw text. + target.push(syntheticText( + degraded.map(part => part.type === 'image_url' ? OMITTED_IMAGE_TEXT : part.text).join(''), + imageContentMode, + )) + } else { + target.push(...degraded) + } } else { - const url = `data:${block.source.media_type};base64,${block.source.data}` - contentParts.push({ type: 'image_url', image_url: { url } }) + const reference = documentToTextReference(block) + target.push({ ...reference, synthetic: true }) } - } else if (block.type === 'tool_result') { - // tool_result → separate tool message - output.push({ - role: 'tool', - tool_call_id: block.tool_use_id, - content: convertToolResultContent(block.content, imageContentMode), - }) + continue + } + + if (block.type === 'search_result') { + // Top-level user search results carry their content as text; keep them + // visible instead of dropping the block. + const target = sawToolResult ? trailingUserParts : leadingUserParts + const text = searchResultToText(block) + if (text) target.push(syntheticText(text, imageContentMode)) + continue + } + + if (block.type !== 'tool_result') continue + + sawToolResult = true + const { resultText, mediaParts } = toolResultToParts(block, imageContentMode) + toolMessages.push({ + role: 'tool', + tool_call_id: block.tool_use_id, + content: resultText, + }) + if (mediaParts.length > 0 && imageContentMode !== 'text_only') { + mediaGroups.push({ toolUseId: block.tool_use_id, parts: mediaParts }) } } - if (imageContentMode === 'text_only') { - const content = textOnlyParts.filter(Boolean).join('\n') - if (content) { - output.push({ - role: 'user', - content, + if (leadingUserParts.length > 0) { + output.push(createUserMessage(leadingUserParts, imageContentMode)) + } + output.push(...toolMessages) + if (mediaGroups.length > 0) { + const mediaContent: OpenAIChatContentPart[] = [] + for (const group of mediaGroups) { + mediaContent.push({ + type: 'text', + text: `[Media content for tool call ${group.toolUseId}]`, }) + mediaContent.push(...group.parts) } - } else if (contentParts.length > 0) { - output.push({ - role: 'user', - content: contentParts.length === 1 && contentParts[0].type === 'text' - ? contentParts[0].text - : contentParts, - }) + output.push({ role: 'user', content: mediaContent }) + } + if (trailingUserParts.length > 0) { + output.push(createUserMessage(trailingUserParts, imageContentMode)) } } -function convertToolResultContent( - content: string | AnthropicContentBlock[], +function createUserMessage( + parts: Array, imageContentMode: OpenAIChatImageContentMode, -): string | OpenAIChatContentPart[] { - if (typeof content === 'string') return content +): OpenAIChatMessage { + // Collapse a single text block to a plain string: many OpenAI-compatible + // endpoints only implement string `content` and reject the multipart array + // form, so the array is reserved for messages that actually carry media or + // multiple blocks. Joining multiple blocks would inject separators the + // original prompt never had, so they keep their array shape. In text_only + // mode the parts must collapse to a string — raw text blocks keep their + // exact bytes (no separator), while synthetic parts get an explicit + // boundary against the surrounding text. + const textParts = parts.filter( + (part): part is Extract => part.type === 'text', + ) + const wireParts: OpenAIChatContentPart[] = parts.map(part => part.type === 'text' + ? { type: 'text', text: part.text } + : part) + const content = imageContentMode === 'text_only' + ? joinUserTextParts(textParts) + : textParts.length === parts.length + && (parts.length === 1 || textParts.every(part => !part.synthetic)) + ? textParts.map(part => part.text).join('') + : wireParts + return { role: 'user', content } +} - const parts: OpenAIChatContentPart[] = [] - for (const block of content) { - if (block.type === 'text') { - parts.push({ type: 'text', text: block.text }) - } else if (block.type === 'image') { - if (imageContentMode === 'text_only') { - parts.push({ type: 'text', text: OMITTED_IMAGE_TEXT }) - } else { - parts.push({ - type: 'image_url', - image_url: { url: `data:${block.source.media_type};base64,${block.source.data}` }, - }) +/** + * A synthetic text part (degraded document or search result) carries an + * internal marker so serializers can preserve boundaries without sending the + * marker to the upstream endpoint. + */ +function syntheticText(text: string, _imageContentMode: OpenAIChatImageContentMode): UserTextPart { + return { type: 'text', text, synthetic: true } +} + +/** + * Whether two adjacent text fragments need a line boundary inserted between + * them: the boundary must exist, but a fragment that already ends (left) or + * starts (right) with a line break provides it. Checking both '\r' and '\n' + * covers CRLF and lone-CR line breaks as well. + */ +function needsLineBoundary(left: string, right: string): boolean { + return ( + !left.endsWith('\n') && !left.endsWith('\r') && + !right.startsWith('\n') && !right.startsWith('\r') + ) +} + +/** + * Join text parts into one string for text_only mode. Raw text blocks keep + * their exact bytes — no separator is injected between them. A synthetic part + * (degraded document, search result) gets a line boundary on each side so a + * structured block cannot glue itself to the surrounding user text — but only + * when the adjacent text does not already provide one, so the serializer never + * rewrites bytes the prompt already carries. + */ +function joinUserTextParts(parts: Array>): string { + let content = '' + for (let i = 0; i < parts.length; i++) { + const part = parts[i]! + if (i > 0 && (parts[i - 1]!.synthetic || part.synthetic) && needsLineBoundary(content, part.text)) { + content += '\n' + } + content += part.text + } + return content +} + +function toolResultToParts( + block: Extract, + imageContentMode: OpenAIChatImageContentMode, +): { resultText: string; mediaParts: OpenAIChatContentPart[] } { + const content = typeof block.content === 'string' ? [{ type: 'text' as const, text: block.content }] : block.content + const textParts: string[] = [] + const mediaParts: OpenAIChatContentPart[] = [] + // A document is a discrete block: keep a newline on *both* sides of it. + // This flag marks that the previous block was a document, so the next text + // block gets the trailing separator. Text blocks *inside* one document keep + // their boundaries — the final join uses no separator, so no text is + // rewritten. + let pendingDocumentBoundary = false + + for (const resultBlock of content) { + if (resultBlock.type === 'text') { + if (pendingDocumentBoundary) { + // The previous block's boundary newline is only needed when this + // text does not already start with one. + if (needsLineBoundary('', resultBlock.text)) { + textParts.push('\n') + } + pendingDocumentBoundary = false } + textParts.push(resultBlock.text) + } else if (resultBlock.type === 'search_result') { + // Search results carry their content as text; keep it visible instead + // of dropping it. Like a document, a search result is a discrete block: + // keep a newline on both sides so it cannot glue itself to the + // surrounding tool output — unless the adjacent text already provides + // the boundary. + const text = searchResultToText(resultBlock) + if (text) { + if (pendingDocumentBoundary) { + if (needsLineBoundary('', text)) { + textParts.push('\n') + } + pendingDocumentBoundary = false + } else if (textParts.length > 0 && !textParts[textParts.length - 1].endsWith('\n')) { + textParts.push('\n') + } + textParts.push(text) + pendingDocumentBoundary = true + } + } else if (resultBlock.type === 'image') { + if (resultBlock.source.type === 'file') { + // Files API references cannot be forwarded to OpenAI-compatible + // endpoints; the degraded notice carries its own separators. + pendingDocumentBoundary = false + textParts.push(FILE_IMAGE_OMITTED_TEXT) + } else if (imageContentMode === 'text_only') { + // The degraded notice carries its own separators. + pendingDocumentBoundary = false + textParts.push(OMITTED_IMAGE_TEXT) + } else { + mediaParts.push({ type: 'image_url', image_url: toImageUrl(resultBlock) }) + } + } else if (resultBlock.type === 'document') { + // Documents degrade to text where possible. Plain text sources and text + // references stay in the tool message (external tool output belongs + // behind the tool role); only real media lifts with the other + // tool-result images. + if (textParts.length > 0 && !textParts[textParts.length - 1].endsWith('\n')) { + textParts.push('\n') + } + if (resultBlock.source.type === 'text') { + // Anthropic text documents carry plain text in `data` — not base64. + // Title/context are model-visible metadata, kept as a synthetic prefix. + textParts.push(`${documentProvenanceText(resultBlock)}${resultBlock.source.data}`) + } else if (resultBlock.source.type === 'content') { + // Custom-content documents carry inline text and image blocks + // (citations/RAG); keep the text in the tool message and lift the + // images with the other tool-result media. Text/image order within + // each group is preserved. + for (const part of documentContentToParts(resultBlock)) { + if (part.type === 'image_url') { + if (imageContentMode === 'text_only') textParts.push(OMITTED_IMAGE_TEXT) + else mediaParts.push(part) + } else { + textParts.push(part.text) + } + } + } else { + textParts.push(documentToTextReference(resultBlock).text) + } + pendingDocumentBoundary = true } } - if (parts.every((part) => part.type === 'text')) { - return parts.map((part) => part.text).join('\n') + // Adjacent text blocks carry no separator in the Anthropic wire shape, so + // joining with '\n' would inject separators the tool output never had. + // Concatenate without a separator to keep the text unchanged. + const resultText = textParts.join('') + return { + resultText: resultText || (mediaParts.length > 0 ? MEDIA_RESULT_ATTACHED_TEXT : ''), + mediaParts, + } +} + +/** + * Documents cannot be represented in OpenAI Chat's content part schema, so they + * are kept visible as a text reference instead of being silently dropped. + * Endpoints that need the actual file content should use the Responses/Azure + * paths, which map documents to input_file. + */ +function documentToTextReference( + block: Extract, +): { type: 'text'; text: string } { + const source = block.source + // The reference text already carries the title (as its label), so only the + // model-visible context needs a synthetic prefix here — it would otherwise + // be silently dropped for URL/base64/file sources. + const contextPrefix = block.context ? `[Document context: ${block.context}]\n` : '' + if (source.type === 'url') { + const label = block.title ?? source.url + return { + type: 'text', + text: `${contextPrefix}[Document: ${label}](${source.url})`, + } + } + if (source.type === 'file') { + return { + type: 'text', + text: `${contextPrefix}[Document: ${block.title ?? 'file'} omitted — file-based source]`, + } + } + if (source.type === 'base64' && source.media_type.startsWith('text/')) { + // Text documents encoded as base64 carry readable content; inline a + // bounded excerpt instead of leaving only a placeholder. + const text = Buffer.from(source.data, 'base64').toString('utf8') + const label = block.title ?? source.media_type + const excerpt = text.slice(0, DOCUMENT_TEXT_INLINE_LIMIT) + const suffix = text.length > DOCUMENT_TEXT_INLINE_LIMIT ? '\n[Document content truncated]' : '' + return { type: 'text', text: `${contextPrefix}[Document: ${label}]\n${excerpt}${suffix}` } + } + if (source.type === 'content') { + // Callers flatten custom-content documents before this point; keep a + // visible reference for safety. + return { type: 'text', text: `${contextPrefix}[Document: ${block.title ?? 'document'}]` } + } + const label = block.title ?? source.media_type + return { + type: 'text', + text: `${contextPrefix}[Document: ${label}]`, + } +} + +/** + * Flatten a custom-content document (source.type === 'content') into ordered + * content parts, keeping the inline text/image sequence intact. Base64 and URL + * image sources convert to Chat image_url parts so the media survives, while + * file-based sources degrade to a text notice. + */ +function documentContentToParts( + block: Extract, +): OpenAIChatContentPart[] { + const source = block.source + const parts: OpenAIChatContentPart[] = [] + if (source.type !== 'content') return parts + // Title/context are synthesized metadata, so they may carry their own + // separator (unlike the document's text blocks, which are never rewritten). + const provenance = documentProvenanceText(block) + if (provenance) parts.push({ type: 'text', text: provenance }) + if (typeof source.content === 'string') { + parts.push({ type: 'text', text: source.content }) + } else { + for (const part of source.content) { + if (part.type === 'text') { + parts.push({ type: 'text', text: part.text }) + } else if (part.source.type === 'file') { + // Files API references cannot be forwarded to OpenAI-compatible + // endpoints; the degraded notice carries its own separators. + parts.push({ type: 'text', text: '\n[Image omitted from document content]\n' }) + } else { + parts.push({ type: 'image_url', image_url: toImageUrl(part) }) + } + } } return parts } +/** + * Model-visible document metadata (title/context) as synthetic prefix text. + * Both fields are visible to the model in the Anthropic protocol, so + * degradation keeps them instead of dropping them silently. Returns an empty + * string when neither is set, otherwise a newline-terminated prefix so the + * document body follows on its own line. + */ +function documentProvenanceText(document: { title?: string; context?: string }): string { + const lines = [ + ...(document.title ? [`[Document: ${document.title}]`] : []), + ...(document.context ? [`[Document context: ${document.context}]`] : []), + ] + return lines.length > 0 ? `${lines.join('\n')}\n` : '' +} + +function toImageUrl(block: Extract): { url: string } { + const source = block.source + if (source.type === 'file') { + // Unreachable: callers degrade file-based sources to a text notice first. + throw new Error('file-based image source cannot be converted to a URL') + } + return source.type === 'url' + ? { url: source.url } + : { url: `data:${source.media_type};base64,${source.data}` } +} + +/** + * Flatten an Anthropic search_result block into its visible text (title, + * body, source URL), matching the official schema where `source` is a string. + */ +function searchResultToText(block: Extract): string { + const contentText = Array.isArray(block.content) + ? block.content.filter((part): part is { type: 'text'; text: string } => part.type === 'text').map(part => part.text) + : [] + return [ + block.title, + ...contentText, + block.source, + ].filter((part): part is string => typeof part === 'string' && part.length > 0) + .join(' — ') +} + function convertAssistantMessage( blocks: AnthropicContentBlock[], output: OpenAIChatMessage[], diff --git a/src/server/proxy/transform/anthropicToOpenaiResponses.ts b/src/server/proxy/transform/anthropicToOpenaiResponses.ts index 7edf6aa0..a401ec0e 100644 --- a/src/server/proxy/transform/anthropicToOpenaiResponses.ts +++ b/src/server/proxy/transform/anthropicToOpenaiResponses.ts @@ -119,17 +119,108 @@ export function anthropicToOpenaiResponses( return result } +/** + * Model-visible document metadata (title/context) as synthetic prefix text. + * Both fields are visible to the model in the Anthropic protocol, so + * degradation keeps them instead of dropping them silently. Returns an empty + * string when neither is set, otherwise a newline-terminated prefix so the + * document body follows on its own line. + */ +function documentProvenanceText(document: { title?: string; context?: string }): string { + const lines = [ + ...(document.title ? [`[Document: ${document.title}]`] : []), + ...(document.context ? [`[Document context: ${document.context}]`] : []), + ] + return lines.length > 0 ? `${lines.join('\n')}\n` : '' +} + function convertContentBlock( - block: Extract, -): OpenAIResponsesInputContentPart { + block: Extract, +): OpenAIResponsesInputContentPart[] { if (block.type === 'text') { - return { type: 'input_text', text: block.text } + return [{ type: 'input_text', text: block.text }] } - return { - type: 'input_image', - image_url: `data:${block.source.media_type};base64,${block.source.data}`, + if (block.type === 'search_result') { + // Search results carry their content as text; keep it visible instead of + // dropping the block. `source` is a URL string per the official schema. + const contentText = Array.isArray(block.content) + ? block.content.filter((part): part is { type: 'text'; text: string } => part.type === 'text').map(part => part.text) + : [] + const text = [ + block.title, + ...contentText, + block.source, + ].filter((part): part is string => typeof part === 'string' && part.length > 0) + .join(' — ') + return [{ type: 'input_text', text }] } + + if (block.type === 'document') { + const source = block.source + if (source.type === 'text') { + // Anthropic text documents carry plain text in `data` — not base64. + // Title/context are model-visible metadata, kept as a synthetic prefix. + return [{ type: 'input_text', text: `${documentProvenanceText(block)}${source.data}` }] + } + if (source.type === 'content') { + // Custom-content documents carry inline text and image blocks + // (citations/RAG). Keep both visible. Title/context are synthesized + // metadata, so they may carry their own separator (unlike the + // document's text blocks, which are never rewritten). + const parts: OpenAIResponsesInputContentPart[] = [] + const provenance = documentProvenanceText(block) + if (provenance) parts.push({ type: 'input_text', text: provenance }) + if (typeof source.content === 'string') { + parts.push({ type: 'input_text', text: source.content }) + } else { + for (const part of source.content) { + if (part.type === 'text') { + parts.push({ type: 'input_text', text: part.text }) + } else { + parts.push(...convertContentBlock(part)) + } + } + } + return parts + } + if (source.type === 'url') { + // The input_file carries the title as filename, so the synthetic + // provenance prefix keeps the model-visible title/context text. + const provenance = documentProvenanceText(block) + return [ + ...(provenance ? [{ type: 'input_text' as const, text: provenance }] : []), + { type: 'input_file', file_url: source.url, ...(block.title ? { filename: block.title } : {}) }, + ] + } + if (source.type === 'file') { + // Files API references cannot be forwarded to a third-party endpoint. + // The omission notice already carries the title, so only context needs + // a synthetic prefix. + const contextPrefix = block.context ? `[Document context: ${block.context}]\n` : '' + return [{ type: 'input_text', text: `${contextPrefix}[Document: ${block.title ?? 'file'} omitted — file-based source]` }] + } + const base64Provenance = documentProvenanceText(block) + return [ + ...(base64Provenance ? [{ type: 'input_text' as const, text: base64Provenance }] : []), + { + type: 'input_file', + file_data: `data:${source.media_type};base64,${source.data}`, + ...(block.title ? { filename: block.title } : {}), + }, + ] + } + + const source = block.source + if (source.type === 'file') { + return [{ type: 'input_text', text: '[Image omitted: file-based image source is not supported by this endpoint.]' }] + } + return [{ + type: 'input_image', + image_url: source.type === 'url' + ? source.url + : `data:${source.media_type};base64,${source.data}`, + }] } function convertMessageToInputItems( @@ -168,8 +259,8 @@ function convertMessageToInputItems( } for (const block of content) { - if (block.type === 'text' || block.type === 'image') { - contentParts.push(convertContentBlock(block)) + if (block.type === 'text' || block.type === 'image' || block.type === 'document' || block.type === 'search_result') { + contentParts.push(...convertContentBlock(block)) } else if (block.type === 'tool_use') { // Flush any accumulated content first flushContentParts() @@ -184,15 +275,24 @@ function convertMessageToInputItems( // Flush any accumulated content first flushContentParts() // Lift to function_call_output item + const sourceBlocks = Array.isArray(block.content) ? block.content : [] const resultContent = typeof block.content === 'string' ? block.content - : Array.isArray(block.content) - ? block.content.filter((part): part is Extract => ( - part.type === 'text' || part.type === 'image' - )).map(convertContentBlock) - : '' - const resultOutput = Array.isArray(resultContent) && resultContent.every((part) => part.type === 'input_text') - ? resultContent.map((part) => part.text).join('\n') + : sourceBlocks + .filter((part): part is Extract => ( + part.type === 'text' || part.type === 'image' || part.type === 'document' || part.type === 'search_result' + )) + .flatMap(convertContentBlock) + // Adjacent *text* blocks carry no separator in the Anthropic wire shape, + // so joining with '\n' would inject separators the tool output never + // had. Concatenate without a separator to keep the text unchanged, and + // only when every source block was text — anything degraded from another + // block type keeps its array shape so block boundaries stay visible. + const onlyTextBlocks = sourceBlocks.every(part => part.type === 'text') + const resultOutput = onlyTextBlocks + && Array.isArray(resultContent) + && resultContent.every((part): part is Extract => part.type === 'input_text') + ? resultContent.map((part) => part.text).join('') : resultContent output.push({ type: 'function_call_output', diff --git a/src/server/proxy/transform/types.ts b/src/server/proxy/transform/types.ts index c5040564..aab88f42 100644 --- a/src/server/proxy/transform/types.ts +++ b/src/server/proxy/transform/types.ts @@ -127,6 +127,7 @@ export type OpenAIChatStreamChunk = { export type OpenAIResponsesInputContentPart = | { type: 'input_text'; text: string } | { type: 'input_image'; image_url: string } + | { type: 'input_file'; file_url?: string; file_data?: string; filename?: string } export type OpenAIResponsesInputItem = | { type: 'message'; role: 'user' | 'assistant' | 'system'; content: string | OpenAIResponsesInputContentPart[] } @@ -180,10 +181,41 @@ export type OpenAIResponsesResponse = { // ─── Anthropic Types (subset used by transforms) ─────────── +export type AnthropicImageSource = + | { type: 'base64'; media_type: string; data: string } + | { type: 'url'; url: string } + | { type: 'file'; file_id: string } + +/** + * A text block inside a custom-content document (`source.type: 'content'`). + * Mirrors the Anthropic `TextBlockParam` fields the wire protocol allows + * (cache_control, citations) so degradation keeps them instead of dropping + * them silently. + */ +export type AnthropicDocumentContentTextBlock = { + type: 'text' + text: string + cache_control?: unknown + citations?: unknown +} + +export type AnthropicDocumentSource = + | { type: 'base64'; media_type: string; data: string } + | { type: 'url'; url: string } + | { type: 'text'; media_type: string; data: string } + | { type: 'file'; file_id: string } + | { + type: 'content' + content: string | Array + } + export type AnthropicContentBlock = | { type: 'text'; text: string; cache_control?: unknown } - | { type: 'image'; source: { type: 'base64'; media_type: string; data: string }; cache_control?: unknown } + | { type: 'image'; source: AnthropicImageSource; cache_control?: unknown } + | { type: 'document'; source: AnthropicDocumentSource; title?: string; context?: string; citations?: unknown; cache_control?: unknown } + | { type: 'search_result'; source: string; title: string; content: Array<{ type: 'text'; text: string }>; citations?: unknown; cache_control?: unknown } | { type: 'tool_use'; id: string; name: string; input: Record; cache_control?: unknown } + | { type: 'server_tool_use'; id: string; name: string; input: unknown; cache_control?: unknown } | { type: 'tool_result'; tool_use_id: string; content: string | AnthropicContentBlock[]; is_error?: boolean; cache_control?: unknown } | { type: 'thinking'; thinking: string; signature?: string } | { type: 'redacted_thinking'; data: string } diff --git a/src/server/services/providerRuntimeEnv.ts b/src/server/services/providerRuntimeEnv.ts index 1d2a053b..e4a6a9fa 100644 --- a/src/server/services/providerRuntimeEnv.ts +++ b/src/server/services/providerRuntimeEnv.ts @@ -215,6 +215,7 @@ export function normalizeSavedProvider(provider: SavedProvider): SavedProvider { disableExperimentalBetas: rawDisableExperimentalBetas, imageGeneration: rawImageGeneration, model1mSupport: rawModel1mSupport, + supportsNestedToolResultMedia: rawSupportsNestedToolResultMedia, ...rest } = provider const rawProvider = provider as SavedProvider & Record @@ -226,6 +227,9 @@ export function normalizeSavedProvider(provider: SavedProvider): SavedProvider { runtimeKind: provider.runtimeKind ?? 'anthropic_compatible', models: normalizeModelMapping(provider.models), toolSearchEnabled: normalizeToolSearchEnabled(rawProvider.toolSearchEnabled), + ...(typeof rawSupportsNestedToolResultMedia === 'boolean' + ? { supportsNestedToolResultMedia: rawSupportsNestedToolResultMedia } + : {}), ...(normalizeDisableExperimentalBetas(rawDisableExperimentalBetas) ? { disableExperimentalBetas: true } : {}), ...(model1mSupport !== undefined ? { model1mSupport } : {}), ...(imageGeneration !== undefined ? { imageGeneration } : {}), @@ -367,6 +371,16 @@ function getProviderCapabilityEnv( } } +export function resolveProviderApiKey( + provider: SavedProvider, + presetDefaultEnv: Record, +): string { + return provider.apiKey + || presetDefaultEnv.ANTHROPIC_AUTH_TOKEN + || presetDefaultEnv.ANTHROPIC_API_KEY + || '' +} + export function buildProviderAuthEnv( provider: SavedProvider, presetDefaultEnv: Record, @@ -377,7 +391,7 @@ export function buildProviderAuthEnv( } const strategy = provider.authStrategy ?? getPresetAuthStrategy(provider.presetId) - const key = provider.apiKey || presetDefaultEnv.ANTHROPIC_AUTH_TOKEN || presetDefaultEnv.ANTHROPIC_API_KEY || '' + const key = resolveProviderApiKey(provider, presetDefaultEnv) switch (strategy) { case 'api_key': @@ -405,6 +419,13 @@ export function getManagedEnvKeys(): string[] { return [...keys] } +export function providerNeedsProxy( + apiFormat: ApiFormat, + supportsNestedToolResultMedia?: boolean, +): boolean { + return apiFormat !== 'anthropic' || supportsNestedToolResultMedia === false +} + export function buildProviderManagedEnv( provider: SavedProvider, options?: { proxyPath?: string; serverPort?: number }, @@ -417,7 +438,10 @@ export function buildProviderManagedEnv( } const apiFormat: ApiFormat = provider.apiFormat ?? 'anthropic' - const needsProxy = apiFormat !== 'anthropic' + // Anthropic-format providers normally connect directly to the upstream. When + // the provider opts out of nested tool-result media, route through the proxy + // so images/documents are lifted out of tool_result before forwarding. + const needsProxy = providerNeedsProxy(apiFormat, provider.supportsNestedToolResultMedia) const proxyPath = options?.proxyPath ?? '/proxy' const serverPort = options?.serverPort ?? 3456 const baseUrl = needsProxy @@ -505,7 +529,12 @@ export function activeProviderNeedsProxy(configDir: string): boolean { const provider = index.providers.find((entry) => entry.id === index.activeId) if (!provider) return false - return (provider.apiFormat ?? 'anthropic') !== 'anthropic' + // Keep in sync with buildProviderManagedEnv: anthropic-format providers + // that opt out of nested tool-result media also route through the proxy. + return providerNeedsProxy( + provider.apiFormat ?? 'anthropic', + provider.supportsNestedToolResultMedia, + ) } catch { return false } diff --git a/src/server/services/providerService.ts b/src/server/services/providerService.ts index 9923ec6b..7a5ab91c 100644 --- a/src/server/services/providerService.ts +++ b/src/server/services/providerService.ts @@ -14,6 +14,7 @@ import { readRecoverableJsonFile } from './recoverableJsonFile.js' import { ManagedSettingsService } from './managedSettingsService.js' import { anthropicToOpenaiChat } from '../proxy/transform/anthropicToOpenaiChat.js' import { anthropicToOpenaiResponses } from '../proxy/transform/anthropicToOpenaiResponses.js' +import { hoistToolResultMediaForCompatibility } from '../proxy/transform/anthropicMediaHoist.js' import { openaiChatToAnthropic } from '../proxy/transform/openaiChatToAnthropic.js' import { openaiResponsesToAnthropic } from '../proxy/transform/openaiResponsesToAnthropic.js' import type { AnthropicRequest } from '../proxy/transform/types.js' @@ -41,6 +42,8 @@ import { normalizeImageGeneration, normalizeModelMapping, normalizeProvidersIndex, + providerNeedsProxy, + resolveProviderApiKey, } from './providerRuntimeEnv.js' import { getNetworkProxyFetchOptions, @@ -125,6 +128,7 @@ function buildSavedProvider(input: CreateProviderInput): SavedProvider { ...(input.modelContextWindows !== undefined && { modelContextWindows: input.modelContextWindows }), toolSearchEnabled: input.toolSearchEnabled ?? false, ...(input.disableExperimentalBetas === true && { disableExperimentalBetas: true }), + ...(input.supportsNestedToolResultMedia !== undefined && { supportsNestedToolResultMedia: input.supportsNestedToolResultMedia }), ...(imageGeneration !== undefined && { imageGeneration }), ...(input.notes !== undefined && { notes: input.notes }), } @@ -298,6 +302,7 @@ export class ProviderService { ...(typeof input.autoCompactWindow === 'number' && { autoCompactWindow: input.autoCompactWindow }), ...(input.modelContextWindows !== undefined && input.modelContextWindows !== null && { modelContextWindows: input.modelContextWindows }), ...(input.toolSearchEnabled !== undefined && { toolSearchEnabled: input.toolSearchEnabled }), + ...(input.supportsNestedToolResultMedia !== undefined && { supportsNestedToolResultMedia: input.supportsNestedToolResultMedia }), ...(input.disableExperimentalBetas === true && { disableExperimentalBetas: true }), ...(imageGeneration !== undefined && imageGeneration !== null && { imageGeneration }), ...(input.notes !== undefined && { notes: input.notes }), @@ -521,7 +526,10 @@ export class ProviderService { const provider = index.providers.find(p => p.id === index.activeId) if (provider) { const presetDefaultEnv = getPresetDefaultEnv(provider.presetId) - const needsProxy = provider.apiFormat != null && provider.apiFormat !== 'anthropic' + const needsProxy = providerNeedsProxy( + provider.apiFormat ?? 'anthropic', + provider.supportsNestedToolResultMedia, + ) const authEnv = buildProviderAuthEnv(provider, presetDefaultEnv, needsProxy) if (Object.values(authEnv).some(value => value.length > 0)) { return { hasAuth: true, source: 'cc-haha-provider', activeProvider: provider.name } @@ -567,20 +575,27 @@ export class ProviderService { baseUrl: string apiKey: string apiFormat: ApiFormat + supportsNestedToolResultMedia: boolean + authStrategy: ProviderAuthStrategy } | null> { - if (providerId) { - if (isOpenAIOfficialProviderId(providerId) || isGrokOfficialProviderId(providerId)) { - return null - } - const provider = await this.getProvider(providerId) + const toProxyConfig = (provider: SavedProvider) => { + const presetDefaultEnv = getPresetDefaultEnv(provider.presetId) return { id: provider.id, name: provider.name, baseUrl: provider.baseUrl, - apiKey: provider.apiKey, + apiKey: resolveProviderApiKey(provider, presetDefaultEnv), apiFormat: provider.apiFormat ?? 'anthropic', + supportsNestedToolResultMedia: provider.supportsNestedToolResultMedia ?? true, + authStrategy: provider.authStrategy ?? getPresetAuthStrategy(provider.presetId), } } + if (providerId) { + if (isOpenAIOfficialProviderId(providerId) || isGrokOfficialProviderId(providerId)) { + return null + } + return toProxyConfig(await this.getProvider(providerId)) + } const index = await this.readIndex() if (!index.activeId) return null @@ -588,20 +603,15 @@ export class ProviderService { return null } const provider = await this.getProvider(index.activeId).catch(() => null) - if (!provider) return null - return { - id: provider.id, - name: provider.name, - baseUrl: provider.baseUrl, - apiKey: provider.apiKey, - apiFormat: provider.apiFormat ?? 'anthropic', - } + return provider ? toProxyConfig(provider) : null } async getActiveProviderForProxy(): Promise<{ baseUrl: string apiKey: string apiFormat: ApiFormat + supportsNestedToolResultMedia: boolean + authStrategy: ProviderAuthStrategy } | null> { return this.getProviderForProxy() } @@ -618,9 +628,7 @@ export class ProviderService { const apiFormat = provider.apiFormat ?? 'anthropic' const authStrategy = provider.authStrategy ?? getPresetAuthStrategy(provider.presetId) const presetDefaultEnv = getPresetDefaultEnv(provider.presetId) - const apiKey = provider.apiKey - || presetDefaultEnv.ANTHROPIC_AUTH_TOKEN - || presetDefaultEnv.ANTHROPIC_API_KEY + const apiKey = resolveProviderApiKey(provider, presetDefaultEnv) || (authStrategy === 'dual_dummy' ? 'dummy' : '') if (!baseUrl || !apiKey) { @@ -632,6 +640,7 @@ export class ProviderService { modelId, authStrategy, apiFormat, + supportsNestedToolResultMedia: provider.supportsNestedToolResultMedia, }) } @@ -651,14 +660,20 @@ export class ProviderService { return { connectivity: step1 } } - // For native Anthropic format, no proxy pipeline to test - if (format === 'anthropic') { + if (!providerNeedsProxy(format, input.supportsNestedToolResultMedia)) { return { connectivity: step1 } } // ── Step 2: Full proxy pipeline ────────────────────────── // Anthropic request → transform → upstream → transform back → validate - const step2 = await this.testProxyPipeline(base, input.apiKey, modelId, format, networkSettings) + const step2 = await this.testProxyPipeline( + base, + input.apiKey, + modelId, + format, + authStrategy, + networkSettings, + ) return { connectivity: step1, proxy: step2 } } @@ -716,7 +731,8 @@ export class ProviderService { base: string, apiKey: string, modelId: string, - format: 'openai_chat' | 'openai_responses', + format: ApiFormat, + authStrategy: ProviderAuthStrategy, networkSettings: NetworkSettings, ): Promise { const start = Date.now() @@ -728,22 +744,32 @@ export class ProviderService { messages: [{ role: 'user', content: 'Say "ok" and nothing else.' }], } - // Transform to OpenAI format let upstreamUrl: string let transformedBody: unknown + let headers: Record if (format === 'openai_chat') { transformedBody = anthropicToOpenaiChat(anthropicReq) upstreamUrl = `${base}/v1/chat/completions` - } else { + headers = { 'Content-Type': 'application/json', Authorization: `Bearer ${apiKey}` } + } else if (format === 'openai_responses') { transformedBody = anthropicToOpenaiResponses(anthropicReq) upstreamUrl = `${base}/v1/responses` + headers = { 'Content-Type': 'application/json', Authorization: `Bearer ${apiKey}` } + } else { + transformedBody = hoistToolResultMediaForCompatibility(anthropicReq) + upstreamUrl = `${base}/v1/messages` + headers = { + 'Content-Type': 'application/json', + 'anthropic-version': '2023-06-01', + ...buildAnthropicAuthHeaders(apiKey, authStrategy), + } } const proxyOptions = getNetworkProxyFetchOptions(networkSettings, upstreamUrl) // Call upstream with transformed request const response = await fetch(upstreamUrl, { method: 'POST', - headers: { 'Content-Type': 'application/json', Authorization: `Bearer ${apiKey}` }, + headers, body: JSON.stringify(transformedBody), signal: AbortSignal.timeout(networkSettings.aiRequestTimeoutMs), ...proxyOptions, @@ -760,7 +786,9 @@ export class ProviderService { const responseBody = await response.json() const anthropicRes = format === 'openai_chat' ? openaiChatToAnthropic(responseBody, modelId) - : openaiResponsesToAnthropic(responseBody, modelId) + : format === 'openai_responses' + ? openaiResponsesToAnthropic(responseBody, modelId) + : responseBody const latencyMs = Date.now() - start diff --git a/src/server/services/traceCaptureService.ts b/src/server/services/traceCaptureService.ts index 044d6def..49873910 100644 --- a/src/server/services/traceCaptureService.ts +++ b/src/server/services/traceCaptureService.ts @@ -1,6 +1,7 @@ export { captureResponseTraceSnapshot, clearTraceCaptureStateForTests, + drainTraceCaptureForTests, createTraceCallId, createTraceBodySnapshot, getTraceCaptureDiagnosticsForTests, diff --git a/src/server/types/provider.ts b/src/server/types/provider.ts index 6bb4b001..daecc5a6 100644 --- a/src/server/types/provider.ts +++ b/src/server/types/provider.ts @@ -66,6 +66,7 @@ export const ModelContextWindowsSchema = z.record( ) export const ToolSearchEnabledSchema = z.boolean() export const DisableExperimentalBetasSchema = z.boolean() +export const SupportsNestedToolResultMediaSchema = z.boolean() export const ImageGenerationConfigSchema = z.object({ model: z.string().trim().min(1), @@ -88,6 +89,7 @@ export const SavedProviderSchema = z.object({ modelContextWindows: ModelContextWindowsSchema.optional(), toolSearchEnabled: ToolSearchEnabledSchema.optional(), disableExperimentalBetas: DisableExperimentalBetasSchema.optional(), + supportsNestedToolResultMedia: SupportsNestedToolResultMediaSchema.optional(), imageGeneration: ImageGenerationConfigSchema.optional(), notes: z.string().optional(), }) @@ -113,6 +115,7 @@ export const CreateProviderSchema = z.object({ modelContextWindows: ModelContextWindowsSchema.optional(), toolSearchEnabled: ToolSearchEnabledSchema.optional(), disableExperimentalBetas: DisableExperimentalBetasSchema.optional(), + supportsNestedToolResultMedia: SupportsNestedToolResultMediaSchema.optional(), imageGeneration: ImageGenerationConfigSchema.optional(), notes: z.string().optional(), }) @@ -130,6 +133,7 @@ export const UpdateProviderSchema = z.object({ modelContextWindows: ModelContextWindowsSchema.nullable().optional(), toolSearchEnabled: ToolSearchEnabledSchema.optional(), disableExperimentalBetas: DisableExperimentalBetasSchema.optional(), + supportsNestedToolResultMedia: SupportsNestedToolResultMediaSchema.optional(), imageGeneration: ImageGenerationConfigSchema.nullable().optional(), notes: z.string().optional(), }) @@ -140,6 +144,7 @@ export const TestProviderSchema = z.object({ modelId: z.string().min(1), authStrategy: ProviderAuthStrategySchema.optional(), apiFormat: ApiFormatSchema.default('anthropic'), + supportsNestedToolResultMedia: SupportsNestedToolResultMediaSchema.optional(), }) export const ReorderProvidersSchema = z.object({ @@ -170,6 +175,6 @@ export interface ProviderTestStepResult { export interface ProviderTestResult { /** Step 1: Basic connectivity — API reachable, key valid, model exists */ connectivity: ProviderTestStepResult - /** Step 2: Proxy pipeline — full Anthropic→OpenAI→Anthropic round-trip (only for openai_* formats) */ + /** Step 2: Proxy pipeline when the provider requires local request handling */ proxy?: ProviderTestStepResult } diff --git a/src/services/api/azureOpenAI.test.ts b/src/services/api/azureOpenAI.test.ts new file mode 100644 index 00000000..cc229b13 --- /dev/null +++ b/src/services/api/azureOpenAI.test.ts @@ -0,0 +1,485 @@ +import { expect, test } from "bun:test" +import { + buildAzureOpenAIInput, + parseAzureOpenAIResponse, + resolveAzureOpenAIEndpoint, + resolveAzureOpenAIDeployment, +} from "../src/services/api/azureOpenAI.js" + +test("resolveAzureOpenAIEndpoint appends responses path and api-version", () => { + const prevBase = process.env.AZURE_OPENAI_BASE_URL + const prevVersion = process.env.AZURE_OPENAI_API_VERSION + process.env.AZURE_OPENAI_BASE_URL = + "https://example.cognitiveservices.azure.com/" + process.env.AZURE_OPENAI_API_VERSION = "2025-04-01-preview" + + const url = resolveAzureOpenAIEndpoint() + expect(url).toContain("/openai/responses") + expect(url).toContain("api-version=2025-04-01-preview") + + process.env.AZURE_OPENAI_BASE_URL = prevBase + process.env.AZURE_OPENAI_API_VERSION = prevVersion +}) + +test("resolveAzureOpenAIEndpoint normalizes existing Azure OpenAI paths", () => { + const prevBase = process.env.AZURE_OPENAI_BASE_URL + const prevVersion = process.env.AZURE_OPENAI_API_VERSION + process.env.AZURE_OPENAI_API_VERSION = "2025-04-01-preview" + + process.env.AZURE_OPENAI_BASE_URL = + "https://example.cognitiveservices.azure.com/openai/v1/?foo=bar" + let url = new URL(resolveAzureOpenAIEndpoint()) + expect(url.pathname).toBe("/openai/responses") + expect(url.searchParams.get("foo")).toBe("bar") + expect(url.searchParams.get("api-version")).toBe("2025-04-01-preview") + + process.env.AZURE_OPENAI_BASE_URL = + "https://example.cognitiveservices.azure.com/openai/responses?api-version=custom" + url = new URL(resolveAzureOpenAIEndpoint()) + expect(url.pathname).toBe("/openai/responses") + expect(url.searchParams.get("api-version")).toBe("2025-04-01-preview") + + process.env.AZURE_OPENAI_BASE_URL = prevBase + process.env.AZURE_OPENAI_API_VERSION = prevVersion +}) + +test("buildAzureOpenAIInput maps tool_use and tool_result", () => { + const input = buildAzureOpenAIInput([ + { + type: "assistant", + message: { + content: [ + { type: "text", text: "Running tool" }, + { + type: "tool_use", + id: "tool_1", + name: "my_tool", + input: { foo: "bar" }, + }, + ], + }, + }, + { + type: "user", + message: { + content: [ + { + type: "tool_result", + tool_use_id: "tool_1", + content: [{ type: "text", text: "ok" }], + }, + ], + }, + }, + ]) + + expect(input.some(msg => msg.type === "function_call")).toBe(true) + expect(input.some(msg => msg.type === "function_call_output")).toBe(true) +}) + +test("parseAzureOpenAIResponse uses call_id to pair tool results", () => { + const result = parseAzureOpenAIResponse({ + id: "resp_1", + output: [{ + type: "function_call", + id: "item_1", + call_id: "call_1", + name: "my_tool", + arguments: "{}", + }], + }) + + expect(result.content).toContainEqual({ + type: "tool_use", + id: "call_1", + name: "my_tool", + input: {}, + }) +}) + +test("parseAzureOpenAIResponse uses id for a legacy tool_call item", () => { + const result = parseAzureOpenAIResponse({ + id: "resp_1", + output: [{ + type: "tool_call", + id: "legacy_call_1", + function: { + name: "my_tool", + arguments: "{}", + }, + }], + }) + + expect(result.content).toContainEqual({ + type: "tool_use", + id: "legacy_call_1", + name: "my_tool", + input: {}, + }) +}) + +test("parseAzureOpenAIResponse uses a legacy tool_call_id when call_id is absent", () => { + const result = parseAzureOpenAIResponse({ + id: "resp_1", + output: [{ + type: "function_call", + tool_call_id: "legacy_call_1", + name: "my_tool", + arguments: "{}", + }], + }) + + expect(result.content).toContainEqual({ + type: "tool_use", + id: "legacy_call_1", + name: "my_tool", + input: {}, + }) +}) + +test("parseAzureOpenAIResponse prefers call_id over a legacy tool_call_id", () => { + const result = parseAzureOpenAIResponse({ + id: "resp_1", + output: [{ + type: "function_call", + call_id: "call_1", + tool_call_id: "legacy_call_1", + name: "my_tool", + arguments: "{}", + }], + }) + + expect(result.content).toContainEqual({ + type: "tool_use", + id: "call_1", + name: "my_tool", + input: {}, + }) +}) + +test("parseAzureOpenAIResponse rejects an output item id without a call id", () => { + expect(() => parseAzureOpenAIResponse({ + id: "resp_1", + output: [{ + type: "function_call", + id: "item_1", + name: "my_tool", + arguments: "{}", + }], + })).toThrow("missing call_id") +}) + +test("parseAzureOpenAIResponse rejects function calls without an association id", () => { + expect(() => parseAzureOpenAIResponse({ + id: "resp_1", + output: [{ + type: "function_call", + name: "my_tool", + arguments: "{}", + }], + })).toThrow("missing call_id") +}) + +test("buildAzureOpenAIInput rejects tool history without association ids", () => { + expect(() => buildAzureOpenAIInput([{ + type: "assistant", + message: { + content: [{ type: "tool_use", name: "my_tool", input: {} }], + }, + }])).toThrow("tool_use missing call_id") + + expect(() => buildAzureOpenAIInput([{ + type: "user", + message: { + content: [{ type: "tool_result", content: "ok" }], + }, + }])).toThrow("tool_result missing call_id") +}) + +test("buildAzureOpenAIInput preserves user and tool-result images", () => { + const input = buildAzureOpenAIInput([ + { + type: "user", + message: { + content: [ + { type: "text", text: "Describe this" }, + { + type: "image", + source: { type: "base64", media_type: "image/png", data: "abc123" }, + }, + ], + }, + }, + { + type: "user", + message: { + content: [ + { + type: "tool_result", + tool_use_id: "tool_1", + content: [ + { + type: "image", + source: { type: "base64", media_type: "image/jpeg", data: "xyz789" }, + }, + ], + }, + ], + }, + }, + ]) + + expect(input[0]).toEqual({ + type: "message", + role: "user", + content: [ + { type: "input_text", text: "Describe this" }, + { type: "input_image", image_url: "data:image/png;base64,abc123" }, + ], + }) + expect(input[1]).toEqual({ + type: "function_call_output", + call_id: "tool_1", + output: [ + { type: "input_image", image_url: "data:image/jpeg;base64,xyz789" }, + ], + }) + expect(input).toHaveLength(2) +}) + +test("buildAzureOpenAIInput maps document URLs and base64 data to input_file", () => { + const input = buildAzureOpenAIInput([{ + type: "user", + message: { + content: [ + { + type: "tool_result", + tool_use_id: "tool_1", + content: [ + { + type: "document", + title: "report.pdf", + source: { type: "base64", media_type: "application/pdf", data: "pdf-data" }, + }, + { + type: "document", + source: { type: "url", url: "https://example.test/report.pdf" }, + }, + ], + }, + ], + }, + }]) + + expect(input).toEqual([{ + type: "function_call_output", + call_id: "tool_1", + output: [ + { type: "input_text", text: "[Document: report.pdf]\n" }, + { type: "input_file", file_data: "data:application/pdf;base64,pdf-data", filename: "report.pdf" }, + { type: "input_text", text: "[Document: https://example.test/report.pdf](https://example.test/report.pdf)" }, + ], + }]) +}) +test("buildAzureOpenAIInput maps ordinary user documents", () => { + const input = buildAzureOpenAIInput([{ + type: "user", + message: { + content: [{ + type: "document", + title: "notes.txt", + source: { type: "base64", media_type: "text/plain", data: "notes" }, + }], + }, + }]) + + expect(input).toEqual([{ + type: "message", + role: "user", + content: [ + { type: "input_text", text: "[Document: notes.txt]\n" }, + { + type: "input_file", + file_data: "data:text/plain;base64,notes", + filename: "notes.txt", + }, + ], + }]) +}) + +test("resolveAzureOpenAIDeployment throws when codex mapping is missing", () => { + const prevBase = process.env.AZURE_OPENAI_BASE_URL + const prevEnv = process.env.AZURE_OPENAI_CODEX_DEPLOYMENT + process.env.AZURE_OPENAI_BASE_URL = + "https://example.cognitiveservices.azure.com/" + delete process.env.AZURE_OPENAI_CODEX_DEPLOYMENT + + expect(() => resolveAzureOpenAIDeployment("gpt-5.2-codex")).toThrow() + expect(() => resolveAzureOpenAIDeployment("gpt-5.3-codex")).toThrow() + expect(() => resolveAzureOpenAIDeployment("gpt-5.4-codex")).toThrow() + + process.env.AZURE_OPENAI_BASE_URL = prevBase + process.env.AZURE_OPENAI_CODEX_DEPLOYMENT = prevEnv +}) + +test("resolveAzureOpenAIDeployment uses env default even if name matches", () => { + const prevBase = process.env.AZURE_OPENAI_BASE_URL + const prevEnv = process.env.AZURE_OPENAI_CODEX_DEPLOYMENT + process.env.AZURE_OPENAI_BASE_URL = + "https://example.cognitiveservices.azure.com/" + process.env.AZURE_OPENAI_CODEX_DEPLOYMENT = "gpt-5.2-codex" + + const resolved = resolveAzureOpenAIDeployment("gpt-5.2-codex") + expect(resolved).toBe("gpt-5.2-codex") + + process.env.AZURE_OPENAI_BASE_URL = prevBase + process.env.AZURE_OPENAI_CODEX_DEPLOYMENT = prevEnv +}) + +test("buildAzureOpenAIInput maps image URL sources to input_image", () => { + const input = buildAzureOpenAIInput([ + { + type: "user", + message: { + content: [ + { + type: "image", + source: { type: "url", url: "https://example.test/screenshot.png" }, + }, + ], + }, + }, + { + type: "user", + message: { + content: [ + { + type: "tool_result", + tool_use_id: "tool_1", + content: [ + { + type: "image", + source: { type: "url", url: "https://example.test/tool-shot.png" }, + }, + ], + }, + ], + }, + }, + ]) + + expect(input[0]).toEqual({ + type: "message", + role: "user", + content: [{ type: "input_image", image_url: "https://example.test/screenshot.png" }], + }) + expect(input[1]).toEqual({ + type: "function_call_output", + call_id: "tool_1", + output: [{ type: "input_image", image_url: "https://example.test/tool-shot.png" }], + }) +}) + +test("buildAzureOpenAIInput keeps top-level text-degraded media", () => { + const input = buildAzureOpenAIInput([{ + type: "user", + message: { + content: [ + { + type: "search_result", + title: "Top result", + content: [{ type: "text", text: "Top snippet" }], + source: "https://example.test/top", + }, + { + type: "document", + title: "readme.txt", + source: { type: "text", media_type: "text/plain", data: "plain text" }, + }, + { + type: "document", + title: "report.pdf", + source: { type: "url", url: "https://example.test/report.pdf" }, + }, + { + type: "image", + source: { type: "file", file_id: "file_9" }, + }, + ], + }, + }]) + + expect(input).toEqual([{ + type: "message", + role: "user", + // Non-text source blocks keep the array shape so block boundaries stay + // visible instead of being flattened with injected newlines. + content: [ + { type: "input_text", text: "Top result — Top snippet — https://example.test/top" }, + { type: "input_text", text: "[Document: readme.txt]\nplain text" }, + { type: "input_text", text: "[Document: report.pdf](https://example.test/report.pdf)" }, + { type: "input_text", text: "[Image omitted: file-based image source is not supported by this endpoint.]" }, + ], + }]) +}) + +test("buildAzureOpenAIInput keeps model-visible title and context when degrading documents", () => { + const input = buildAzureOpenAIInput([{ + type: "user", + message: { + content: [{ + type: "tool_result", + tool_use_id: "tool_1", + content: [{ + type: "document", + title: "Auth specification", + context: "The examples use production credentials", + source: { type: "text", media_type: "text/plain", data: "Bearer abc123" }, + }], + }], + }, + }]) + + expect(input).toEqual([{ + type: "function_call_output", + call_id: "tool_1", + output: [ + { type: "input_text", text: "[Document: Auth specification]\n[Document context: The examples use production credentials]\nBearer abc123" }, + ], + }]) +}) + +test("buildAzureOpenAIInput keeps inline images of custom-content documents", () => { + const input = buildAzureOpenAIInput([{ + type: "user", + message: { + content: [{ + type: "tool_result", + tool_use_id: "tool_1", + content: [{ + type: "document", + title: "cited", + source: { + type: "content", + content: [ + { type: "text", text: "before" }, + { type: "image", source: { type: "base64", media_type: "image/png", data: "abc" } }, + { type: "text", text: "after" }, + ], + }, + }], + }], + }, + }]) + + expect(input).toEqual([{ + type: "function_call_output", + call_id: "tool_1", + output: [ + { type: "input_text", text: "[Document: cited]\n" }, + { type: "input_text", text: "before" }, + { type: "input_image", image_url: "data:image/png;base64,abc" }, + { type: "input_text", text: "after" }, + ], + }]) +}) diff --git a/src/services/api/azureOpenAI.ts b/src/services/api/azureOpenAI.ts index cf48c8a7..687bbf6c 100644 --- a/src/services/api/azureOpenAI.ts +++ b/src/services/api/azureOpenAI.ts @@ -1,5 +1,4 @@ import type { BetaContentBlock, BetaUsage } from '@anthropic-ai/sdk/resources/beta/messages/messages.mjs' -import { randomUUID } from 'crypto' import type { Tools, ToolPermissionContext } from 'src/Tool.js' import { toolMatchesName } from 'src/Tool.js' import { TOOL_SEARCH_TOOL_NAME } from 'src/tools/ToolSearchTool/prompt.js' @@ -14,21 +13,28 @@ import type { AgentDefinition } from 'src/tools/AgentTool/loadAgentsDir.js' const DEFAULT_API_VERSION = '2025-04-01-preview' -type OpenAIToolCall = { - id: string - type: 'function' - function: { - name: string - arguments: string - } +function requireAssociationId(id: string | undefined, field: string): string { + if (!id) throw new Error(`${field} missing call_id`) + return id } -type OpenAIMessage = { - role: 'system' | 'user' | 'assistant' | 'tool' - content?: string | null - tool_calls?: OpenAIToolCall[] - tool_call_id?: string -} +type OpenAIContentPart = + | { type: 'input_text'; text: string } + | { type: 'input_image'; image_url: string } + | { type: 'input_file'; file_url?: string; file_data?: string; filename?: string } + +type OpenAIInputItem = + | { + type: 'message' + role: 'user' | 'assistant' + content: string | OpenAIContentPart[] + } + | { type: 'function_call'; call_id: string; name: string; arguments: string } + | { + type: 'function_call_output' + call_id: string + output: string | OpenAIContentPart[] + } type OpenAIResponseOutputItem = { type?: string @@ -199,8 +205,144 @@ function contentBlocksToText(content: unknown): string { .join('\n') } -export function buildAzureOpenAIInput(messages: Array<{ type: string; message: { content: unknown } }>): OpenAIMessage[] { - const inputs: OpenAIMessage[] = [] +/** + * Model-visible document metadata (title/context) as synthetic prefix text. + * Both fields are visible to the model in the Anthropic protocol, so + * degradation keeps them instead of dropping them silently. Returns an empty + * string when neither is set, otherwise a newline-terminated prefix so the + * document body follows on its own line. + */ +function documentProvenanceText(document: { title?: string; context?: string }): string { + const lines = [ + ...(document.title ? [`[Document: ${document.title}]`] : []), + ...(document.context ? [`[Document context: ${document.context}]`] : []), + ] + return lines.length > 0 ? `${lines.join('\n')}\n` : '' +} + +function contentBlocksToOpenAIContent( + content: unknown, +): string | OpenAIContentPart[] { + if (typeof content === 'string') return content + if (!Array.isArray(content)) return '' + + const parts: OpenAIContentPart[] = [] + // Only adjacent *text* blocks collapse into one string (joined without a + // separator — the wire shape carries no newline between them). Anything + // degraded from another block type keeps its array shape so block + // boundaries stay visible. + let allOriginalText = true + for (const block of content) { + if (!block || typeof block !== 'object' || !('type' in block)) continue + const typed = block as { + type?: string + text?: string + source?: { + type?: string + media_type?: string + data?: string + url?: string + file_id?: string + content?: unknown + } + title?: string + context?: string + content?: unknown + } + if (typed.type === 'text' && typeof typed.text === 'string') { + parts.push({ type: 'input_text', text: typed.text }) + } else if (typed.type === 'search_result') { + // `source` is a URL string per the official Anthropic schema. + allOriginalText = false + const contentText = Array.isArray(typed.content) + ? typed.content + .filter((part): part is { type: 'text'; text: string } => + typeof part === 'object' && part !== null && 'type' in part && part.type === 'text' && typeof part.text === 'string') + .map(part => part.text) + : [] + const text = [ + typed.title, + ...contentText, + typeof typed.source === 'string' ? typed.source : undefined, + ].filter((part): part is string => typeof part === 'string' && part.length > 0) + .join(' — ') + if (text) parts.push({ type: 'input_text', text }) + } else if (typed.type === 'image' && typed.source) { + allOriginalText = false + const source = typed.source + if (typeof source.url === 'string') { + parts.push({ type: 'input_image', image_url: source.url }) + } else if (source.type === 'file') { + parts.push({ type: 'input_text', text: '[Image omitted: file-based image source is not supported by this endpoint.]' }) + } else if (typeof source.media_type === 'string' && typeof source.data === 'string') { + parts.push({ + type: 'input_image', + image_url: `data:${source.media_type};base64,${source.data}`, + }) + } + } else if (typed.type === 'document' && typed.source) { + allOriginalText = false + const source = typed.source + if (source.type === 'text' && typeof source.data === 'string') { + // Anthropic text documents carry plain text in `data` — not base64. + // Title/context are model-visible metadata, kept as a synthetic prefix. + parts.push({ type: 'input_text', text: `${documentProvenanceText(typed)}${source.data}` }) + } else if (source.type === 'content') { + // Custom-content documents carry inline text and image blocks + // (citations/RAG). Keep both visible. Title/context are synthesized + // metadata, so they may carry their own separator (unlike the + // document's text blocks, which are never rewritten). + const provenance = documentProvenanceText(typed) + if (provenance) parts.push({ type: 'input_text', text: provenance }) + if (typeof source.content === 'string') { + if (source.content) parts.push({ type: 'input_text', text: source.content }) + } else if (Array.isArray(source.content)) { + for (const part of source.content) { + const media = contentBlocksToOpenAIContent([part]) + if (Array.isArray(media)) parts.push(...media) + else if (media) parts.push({ type: 'input_text', text: media }) + } + } + } else if (typeof source.media_type === 'string' && typeof source.data === 'string') { + // The input_file carries the title as filename, so the synthetic + // provenance prefix keeps the model-visible title/context text. + const provenance = documentProvenanceText(typed) + if (provenance) parts.push({ type: 'input_text', text: provenance }) + parts.push({ + type: 'input_file', + file_data: `data:${source.media_type};base64,${source.data}`, + ...(typed.title ? { filename: typed.title } : {}), + }) + } else if (typeof source.url === 'string') { + // Azure Responses (2025-04-01-preview) does not accept file_url the + // way the OpenAI public API does. Keep the document visible with a + // text reference instead of silently dropping it. The reference text + // carries the title as its label, so only context needs a prefix. + const contextPrefix = typed.context ? `[Document context: ${typed.context}]\n` : '' + parts.push({ + type: 'input_text', + text: `${contextPrefix}[Document: ${typed.title ?? source.url}](${source.url})`, + }) + } else if (source.type === 'file') { + const contextPrefix = typed.context ? `[Document context: ${typed.context}]\n` : '' + parts.push({ + type: 'input_text', + text: `${contextPrefix}[Document: ${typed.title ?? 'file'} omitted — file-based source]`, + }) + } + } + } + + if (allOriginalText && parts.every(part => part.type === 'input_text')) { + return parts.map(part => part.text).join('') + } + return parts +} + +export function buildAzureOpenAIInput( + messages: Array<{ type: string; message: { content: unknown } }>, +): OpenAIInputItem[] { + const inputs: OpenAIInputItem[] = [] for (const msg of messages) { if (msg.type !== 'user' && msg.type !== 'assistant') continue @@ -209,13 +351,30 @@ export function buildAzureOpenAIInput(messages: Array<{ type: string; message: { if (!Array.isArray(content)) { const text = contentBlocksToText(content) if (text.trim().length > 0) { - inputs.push({ role: msg.type, content: text }) + inputs.push({ type: 'message', role: msg.type, content: text }) } continue } - const textParts: string[] = [] - const toolCalls: OpenAIToolCall[] = [] + const contentParts: OpenAIContentPart[] = [] + // Only messages made of adjacent text blocks collapse to a string; any + // non-text block keeps the array shape so block boundaries stay visible. + let hasNonTextBlock = false + const flushMessage = (): void => { + if (contentParts.length === 0) return + const messageContent = !hasNonTextBlock && contentParts.every( + part => part.type === 'input_text', + ) + ? contentParts.map(part => part.text).join('') + : [...contentParts] + inputs.push({ + type: 'message', + role: msg.type, + content: messageContent, + }) + contentParts.length = 0 + hasNonTextBlock = false + } for (const block of content) { if (!block || typeof block !== 'object' || !('type' in block)) continue @@ -230,52 +389,37 @@ export function buildAzureOpenAIInput(messages: Array<{ type: string; message: { } if (typed.type === 'text' && typeof typed.text === 'string') { - textParts.push(typed.text) - } - - if (typed.type === 'tool_use' && typed.name) { - const args = - typeof typed.input === 'string' - ? typed.input - : JSON.stringify(typed.input ?? {}) - toolCalls.push({ - id: typed.id ?? randomUUID(), - type: 'function', - function: { - name: typed.name, - arguments: args, - }, - }) - } - - if (typed.type === 'tool_result' && msg.type === 'user') { - const resultText = contentBlocksToText(typed.content) + contentParts.push({ type: 'input_text', text: typed.text }) + } else if ((typed.type === 'image' || typed.type === 'document' || typed.type === 'search_result') && msg.type === 'user') { + hasNonTextBlock = true + const media = contentBlocksToOpenAIContent([typed]) + if (Array.isArray(media)) { + contentParts.push(...media) + } else if (media) { + contentParts.push({ type: 'input_text', text: media }) + } + } else if (typed.type === 'tool_use' && typed.name) { + flushMessage() inputs.push({ - role: 'tool', - tool_call_id: typed.tool_use_id ?? randomUUID(), - content: resultText, + type: 'function_call', + call_id: requireAssociationId(typed.id, 'tool_use'), + name: typed.name, + arguments: + typeof typed.input === 'string' + ? typed.input + : JSON.stringify(typed.input ?? {}), }) - } - } - - if (msg.type === 'assistant') { - const contentText = textParts.join('\n') - if (contentText || toolCalls.length > 0) { + } else if (typed.type === 'tool_result' && msg.type === 'user') { + flushMessage() inputs.push({ - role: 'assistant', - content: contentText.length > 0 ? contentText : null, - ...(toolCalls.length > 0 && { tool_calls: toolCalls }), + type: 'function_call_output', + call_id: requireAssociationId(typed.tool_use_id, 'tool_result'), + output: contentBlocksToOpenAIContent(typed.content), }) } - continue } - if (msg.type === 'user') { - const contentText = textParts.join('\n') - if (contentText.length > 0) { - inputs.push({ role: 'user', content: contentText }) - } - } + flushMessage() } return inputs @@ -303,7 +447,10 @@ function mapOutputItemToBlocks(item: OpenAIResponseOutputItem): BetaContentBlock typeof rawArgs === 'string' ? safeParseJSON(rawArgs) : rawArgs blocks.push({ type: 'tool_use', - id: item.id ?? item.call_id ?? item.tool_call_id ?? randomUUID(), + id: requireAssociationId( + item.call_id ?? item.tool_call_id ?? (item.type === 'tool_call' ? item.id : undefined), + 'function_call', + ), name, input: parsed ?? {}, } as BetaContentBlock) diff --git a/src/services/api/traceCapture.ts b/src/services/api/traceCapture.ts index 1e575804..c15e9bf0 100644 --- a/src/services/api/traceCapture.ts +++ b/src/services/api/traceCapture.ts @@ -485,6 +485,20 @@ function isTraceRecord(value: unknown): value is TraceJsonRecord { return typeof value === 'object' && value !== null && !Array.isArray(value) } +/** + * Wait for all in-flight trace appends (including their index projections) to + * finish. Test teardown should drain before clearing state: a background + * projection still running after `clearTraceCaptureStateForTests` would + * re-open the index database and hold a file handle past the temp dir + * removal on Windows. + */ +export async function drainTraceCaptureForTests(): Promise { + for (let attempt = 0; attempt < 200; attempt++) { + const pending = [...traceWriteQueues.values()] + if (pending.length === 0) return + await Promise.allSettled(pending) + } +} export function clearTraceCaptureStateForTests(): void { traceWriteQueues.clear() traceReadCache.clear() diff --git a/src/services/mcp/client.test.ts b/src/services/mcp/client.test.ts new file mode 100644 index 00000000..c952f726 --- /dev/null +++ b/src/services/mcp/client.test.ts @@ -0,0 +1,380 @@ +import { describe, expect, test } from 'bun:test' +import { transformMCPResult } from './client.js' + +const ONE_PX_PNG = + 'iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mNkYPhfDwAChwGA60e6kgAAAABJRU5ErkJggg==' + +describe('transformMCPResult media handling', () => { + test('does not duplicate structuredContent when content also carries the serialized JSON', async () => { + const result = { + content: [{ type: 'text', text: '{"users":[{"id":1,"name":"a"}]}' }], + structuredContent: { users: [{ id: 1, name: 'a' }] }, + } + + const transformed = await transformMCPResult(result, 'test-tool', 'test-server') + + expect(transformed.type).toBe('contentArray') + expect(transformed.content).toEqual([ + { type: 'text', text: '{"users":[{"id":1,"name":"a"}]}' }, + ]) + expect(JSON.stringify(transformed.content)).not.toContain('Structured content') + }) + + test('keeps images in content and appends structuredContent when not already serialized', async () => { + const result = { + content: [{ type: 'image', data: ONE_PX_PNG, mimeType: 'image/png' }], + structuredContent: { pages: 2 }, + } + + const transformed = await transformMCPResult(result, 'test-tool', 'test-server') + + expect(transformed.type).toBe('contentArray') + expect(transformed.content).toHaveLength(2) + const imageBlock = transformed.content[0] as { type: string; source?: { type: string } } + expect(imageBlock.type).toBe('image') + expect(imageBlock.source?.type).toBe('base64') + expect(transformed.content[1]).toEqual({ + type: 'text', + text: 'Structured content:\n{"pages":2}', + }) + expect(transformed.schema).toBe('{pages: number}') + }) + + test('deduplicates pretty-printed JSON that is semantically equal to structuredContent', async () => { + const result = { + content: [{ + type: 'text', + text: '{\n "title": "Dune",\n "author": "Frank Herbert"\n}', + }], + structuredContent: { title: 'Dune', author: 'Frank Herbert' }, + } + + const transformed = await transformMCPResult(result, 'test-tool', 'test-server') + + expect(transformed.type).toBe('contentArray') + expect(transformed.content).toEqual([ + { type: 'text', text: '{\n "title": "Dune",\n "author": "Frank Herbert"\n}' }, + ]) + expect(JSON.stringify(transformed.content)).not.toContain('Structured content') + }) + + test('appends structuredContent when content text is not JSON-equivalent', async () => { + const result = { + content: [{ type: 'text', text: 'A human-readable summary' }], + structuredContent: { count: 5 }, + } + + const transformed = await transformMCPResult(result, 'test-tool', 'test-server') + + expect(transformed.content).toEqual([ + { type: 'text', text: 'A human-readable summary' }, + { type: 'text', text: 'Structured content:\n{"count":5}' }, + ]) + }) + + test('falls back to structuredContent branch when content is absent', async () => { + const result = { structuredContent: { count: 3 } } + + const transformed = await transformMCPResult(result, 'test-tool', 'test-server') + + expect(transformed.type).toBe('structuredContent') + expect(transformed.content).toBe('{"count":3}') + expect(transformed.schema).toBe('{count: number}') + }) + + test('returns contentArray for content-only results', async () => { + const result = { content: [{ type: 'text', text: 'plain result' }] } + + const transformed = await transformMCPResult(result, 'test-tool', 'test-server') + + expect(transformed.type).toBe('contentArray') + expect(transformed.content).toEqual([{ type: 'text', text: 'plain result' }]) + }) +}) + +describe('persistTextFromImageContent', () => { + test('keeps media blocks in order with bounded text summaries and persists full text', async () => { + const fs = await import('fs/promises') + const os = await import('os') + const path = await import('path') + const { persistTextFromImageContent } = await import('./client.js') + const tmpDir = await fs.mkdtemp(path.join(os.tmpdir(), 'mcp-persist-test-')) + const prevConfigDir = process.env.CLAUDE_CONFIG_DIR + process.env.CLAUDE_CONFIG_DIR = tmpDir + + try { + const result = await persistTextFromImageContent([ + { type: 'text', text: 'Screenshot A' }, + { type: 'image', source: { type: 'base64', media_type: 'image/png', data: 'a' } }, + { type: 'text', text: 'Screenshot B' }, + { type: 'image', source: { type: 'base64', media_type: 'image/jpeg', data: 'b' } }, + ], 'test-server', 'test-tool') + + expect(result).not.toBeNull() + const blocks = result as Array<{ type: string }> + expect(blocks.map(block => block.type)).toEqual(['text', 'image', 'text', 'image', 'text']) + expect((blocks[0] as { text: string }).text).toBe('Screenshot A') + expect((blocks[2] as { text: string }).text).toBe('Screenshot B') + expect((blocks[4] as { text: string }).text).toContain('Binary content (text/plain') + expect((blocks[4] as { text: string }).text).toContain('saved to') + } finally { + if (prevConfigDir !== undefined) { + process.env.CLAUDE_CONFIG_DIR = prevConfigDir + } else { + delete process.env.CLAUDE_CONFIG_DIR + } + await fs.rm(tmpDir, { recursive: true, force: true }) + } + }) + + test('bounds text summaries by a shared budget instead of per block', async () => { + const fs = await import('fs/promises') + const os = await import('os') + const path = await import('path') + const { persistTextFromImageContent } = await import('./client.js') + const tmpDir = await fs.mkdtemp(path.join(os.tmpdir(), 'mcp-persist-budget-test-')) + const prevConfigDir = process.env.CLAUDE_CONFIG_DIR + process.env.CLAUDE_CONFIG_DIR = tmpDir + + try { + // Many small text blocks that individually stay under any per-block cap. + const blocks = [ + { type: 'image' as const, source: { type: 'base64' as const, media_type: 'image/png' as const, data: 'a' } }, + ...Array.from({ length: 50 }, (_, i) => ({ + type: 'text' as const, + text: `caption ${i}: ${'x'.repeat(100)}`, + })), + ] + + const result = await persistTextFromImageContent(blocks, 'test-server', 'test-tool') + + expect(result).not.toBeNull() + const kept = result as Array<{ type: string; text?: string }> + const keptText = kept.filter(block => block.type === 'text').map(block => block.text ?? '') + const summaryChars = keptText + .filter(text => !text.includes('saved to')) + .join('').length + // Summaries share one budget instead of 50 × 100 chars, and the total + // never exceeds the hard cap even when short blocks stack up. + expect(summaryChars).toBeLessThan(50 * 100) + expect(summaryChars).toBeLessThanOrEqual(2000) + expect(summaryChars).toBeGreaterThan(0) + expect(keptText.some(text => text.includes('saved to'))).toBe(true) + } finally { + if (prevConfigDir !== undefined) { + process.env.CLAUDE_CONFIG_DIR = prevConfigDir + } else { + delete process.env.CLAUDE_CONFIG_DIR + } + await fs.rm(tmpDir, { recursive: true, force: true }) + } + }) + + test('bounds the total even for many 199-char short blocks', async () => { + const fs = await import('fs/promises') + const os = await import('os') + const path = await import('path') + const { persistTextFromImageContent } = await import('./client.js') + const tmpDir = await fs.mkdtemp(path.join(os.tmpdir(), 'mcp-persist-hardcap-test-')) + const prevConfigDir = process.env.CLAUDE_CONFIG_DIR + process.env.CLAUDE_CONFIG_DIR = tmpDir + + try { + const blocks = Array.from({ length: 11 }, (_, i) => ({ + type: 'text' as const, + text: `block ${i}: ${'x'.repeat(189)}`, + })) + + const result = await persistTextFromImageContent(blocks, 'test-server', 'test-tool') + + expect(result).not.toBeNull() + const kept = result as Array<{ type: string; text?: string }> + const keptText = kept.filter(block => block.type === 'text').map(block => block.text ?? '') + const summaryChars = keptText.filter(text => !text.includes('saved to')).join('').length + expect(summaryChars).toBeLessThanOrEqual(2000) + expect(keptText.some(text => text.includes('saved to'))).toBe(true) + } finally { + if (prevConfigDir !== undefined) { + process.env.CLAUDE_CONFIG_DIR = prevConfigDir + } else { + delete process.env.CLAUDE_CONFIG_DIR + } + await fs.rm(tmpDir, { recursive: true, force: true }) + } + }) + + test('keeps the caption after an image even when short logs exhaust the budget', async () => { + const fs = await import('fs/promises') + const os = await import('os') + const path = await import('path') + const { persistTextFromImageContent } = await import('./client.js') + const tmpDir = await fs.mkdtemp(path.join(os.tmpdir(), 'mcp-persist-caption-priority-test-')) + const prevConfigDir = process.env.CLAUDE_CONFIG_DIR + process.env.CLAUDE_CONFIG_DIR = tmpDir + + try { + const blocks = [ + ...Array.from({ length: 11 }, (_, i) => ({ + type: 'text' as const, + text: `log ${i}: ${'x'.repeat(180)}`, + })), + { type: 'image' as const, source: { type: 'base64' as const, media_type: 'image/png' as const, data: 'a' } }, + { type: 'text' as const, text: '这是上图的说明' }, + ] + + const result = await persistTextFromImageContent(blocks, 'test-server', 'test-tool') + + expect(result).not.toBeNull() + const kept = result as Array<{ type: string; text?: string }> + const keptText = kept.filter(block => block.type === 'text').map(block => block.text ?? '') + // The caption after the image survives in full while the log lines are + // truncated, and the total stays within the hard budget. + expect(keptText).toContain('这是上图的说明') + const summaryChars = keptText.filter(text => !text.includes('saved to')).join('').length + expect(summaryChars).toBeLessThanOrEqual(2000) + } finally { + if (prevConfigDir !== undefined) { + process.env.CLAUDE_CONFIG_DIR = prevConfigDir + } else { + delete process.env.CLAUDE_CONFIG_DIR + } + await fs.rm(tmpDir, { recursive: true, force: true }) + } + }) + + test('keeps every caption when several follow media blocks', async () => { + const fs = await import('fs/promises') + const os = await import('os') + const path = await import('path') + const { persistTextFromImageContent } = await import('./client.js') + const tmpDir = await fs.mkdtemp(path.join(os.tmpdir(), 'mcp-persist-multi-caption-test-')) + const prevConfigDir = process.env.CLAUDE_CONFIG_DIR + process.env.CLAUDE_CONFIG_DIR = tmpDir + + try { + const blocks = [ + { type: 'text' as const, text: `ordinary log ${'x'.repeat(2000)}` }, + { type: 'image' as const, source: { type: 'base64' as const, media_type: 'image/png' as const, data: 'a' } }, + { type: 'text' as const, text: `caption one ${'y'.repeat(140)}` }, + { type: 'image' as const, source: { type: 'base64' as const, media_type: 'image/png' as const, data: 'b' } }, + { type: 'text' as const, text: `caption two ${'z'.repeat(140)}` }, + { type: 'image' as const, source: { type: 'base64' as const, media_type: 'image/png' as const, data: 'c' } }, + { type: 'text' as const, text: `caption three ${'w'.repeat(140)}` }, + ] + + const result = await persistTextFromImageContent(blocks, 'test-server', 'test-tool') + + expect(result).not.toBeNull() + const kept = result as Array<{ type: string; text?: string }> + const keptText = kept.filter(block => block.type === 'text').map(block => block.text ?? '') + // All three captions survive; the ordinary log is truncated to what the + // captions leave of the shared budget, and the total stays capped. + expect(keptText).toContain(`caption one ${'y'.repeat(140)}`) + expect(keptText).toContain(`caption two ${'z'.repeat(140)}`) + expect(keptText).toContain(`caption three ${'w'.repeat(140)}`) + const summaryChars = keptText.filter(text => !text.includes('saved to')).join('').length + expect(summaryChars).toBeLessThanOrEqual(2000) + } finally { + if (prevConfigDir !== undefined) { + process.env.CLAUDE_CONFIG_DIR = prevConfigDir + } else { + delete process.env.CLAUDE_CONFIG_DIR + } + await fs.rm(tmpDir, { recursive: true, force: true }) + } + }) + + test('keeps short captions whole when a long block dominates the budget', async () => { + const fs = await import('fs/promises') + const os = await import('os') + const path = await import('path') + const { persistTextFromImageContent } = await import('./client.js') + const tmpDir = await fs.mkdtemp(path.join(os.tmpdir(), 'mcp-persist-caption-test-')) + const prevConfigDir = process.env.CLAUDE_CONFIG_DIR + process.env.CLAUDE_CONFIG_DIR = tmpDir + + try { + const blocks = [ + { type: 'text' as const, text: `log line ${'x'.repeat(3000)}` }, + { type: 'image' as const, source: { type: 'base64' as const, media_type: 'image/png' as const, data: 'a' } }, + { type: 'text' as const, text: '上图是登录页面' }, + { type: 'image' as const, source: { type: 'base64' as const, media_type: 'image/jpeg' as const, data: 'b' } }, + { type: 'text' as const, text: '上图是支付失败弹窗' }, + ] + + const result = await persistTextFromImageContent(blocks, 'test-server', 'test-tool') + + expect(result).not.toBeNull() + const kept = result as Array<{ type: string; text?: string }> + const keptText = kept.filter(block => block.type === 'text').map(block => block.text ?? '') + // The long log block is truncated to the shared budget, but the short + // captions after the images survive in full — they are not starved. + expect(keptText).toContain('上图是登录页面') + expect(keptText).toContain('上图是支付失败弹窗') + const logSummary = keptText.find(text => text.startsWith('log line')) + expect(logSummary?.length ?? 0).toBeLessThan(3000) + expect(keptText.some(text => text.includes('saved to'))).toBe(true) + } finally { + if (prevConfigDir !== undefined) { + process.env.CLAUDE_CONFIG_DIR = prevConfigDir + } else { + delete process.env.CLAUDE_CONFIG_DIR + } + await fs.rm(tmpDir, { recursive: true, force: true }) + } + }) + + test('parallel invocations of the same server/tool persist to distinct files', async () => { + const fs = await import('fs/promises') + const os = await import('os') + const path = await import('path') + const { persistTextFromImageContent } = await import('./client.js') + const tmpDir = await fs.mkdtemp(path.join(os.tmpdir(), 'mcp-persist-collision-test-')) + const prevConfigDir = process.env.CLAUDE_CONFIG_DIR + process.env.CLAUDE_CONFIG_DIR = tmpDir + + const originalNow = Date.now + Date.now = () => 123456789 + + try { + // Same server, same tool, frozen clock: the persistence id must still + // be unique per invocation — persistToolResult treats an existing file + // ('wx') as a replay of the same invocation, so a timestamp-only id + // would make the second call report its own preview while the file + // holds the first call's content. + const [resultA, resultB] = await Promise.all([ + persistTextFromImageContent([ + { type: 'text' as const, text: 'AAAA'.repeat(500) }, + { type: 'image' as const, source: { type: 'base64' as const, media_type: 'image/png' as const, data: 'a' } }, + ], 'test-server', 'test-tool'), + persistTextFromImageContent([ + { type: 'text' as const, text: 'BBBB'.repeat(500) }, + { type: 'image' as const, source: { type: 'base64' as const, media_type: 'image/jpeg' as const, data: 'b' } }, + ], 'test-server', 'test-tool'), + ]) + + expect(resultA).not.toBeNull() + expect(resultB).not.toBeNull() + const savedPathA = (resultA as Array<{ type: string; text: string }>) + .find(block => block.type === 'text' && block.text.includes('saved to'))?.text.match(/saved to (.+)/)?.[1] + const savedPathB = (resultB as Array<{ type: string; text: string }>) + .find(block => block.type === 'text' && block.text.includes('saved to'))?.text.match(/saved to (.+)/)?.[1] + expect(savedPathA).toBeDefined() + expect(savedPathB).toBeDefined() + expect(savedPathA).not.toBe(savedPathB) + // Each file holds its own call's content, not the other call's. + expect(await fs.readFile(savedPathA!, 'utf-8')).toContain('AAAA') + expect(await fs.readFile(savedPathB!, 'utf-8')).toContain('BBBB') + expect(await fs.readFile(savedPathA!, 'utf-8')).not.toContain('BBBB') + expect(await fs.readFile(savedPathB!, 'utf-8')).not.toContain('AAAA') + } finally { + Date.now = originalNow + if (prevConfigDir !== undefined) { + process.env.CLAUDE_CONFIG_DIR = prevConfigDir + } else { + delete process.env.CLAUDE_CONFIG_DIR + } + await fs.rm(tmpDir, { recursive: true, force: true }) + } + }) +}) diff --git a/src/services/mcp/client.ts b/src/services/mcp/client.ts index 298f2ce9..4792953e 100644 --- a/src/services/mcp/client.ts +++ b/src/services/mcp/client.ts @@ -2630,6 +2630,37 @@ export type TransformedMCPResult = { schema?: string } +/** + * True when the given text block carries the same data as the structured + * content value — either byte-identical serialization or a JSON-equivalent + * parse (servers commonly pretty-print the JSON in TextContent). + */ +function jsonTextMatches(text: string, structuredContent: unknown): boolean { + const trimmed = text.trim() + if (trimmed === jsonStringify(structuredContent)) return true + try { + return jsonDeepEqual(JSON.parse(trimmed), structuredContent) + } catch { + return false + } +} + +function jsonDeepEqual(a: unknown, b: unknown): boolean { + if (a === b) return true + if (typeof a !== 'object' || typeof b !== 'object' || a === null || b === null) return false + if (Array.isArray(a) !== Array.isArray(b)) return false + if (Array.isArray(a)) { + const bArr = b as unknown[] + return a.length === bArr.length && a.every((item, i) => jsonDeepEqual(item, bArr[i])) + } + const aObj = a as Record + const bObj = b as Record + const aKeys = Object.keys(aObj).sort() + const bKeys = Object.keys(bObj).sort() + if (aKeys.length !== bKeys.length) return false + return aKeys.every((key, i) => key === bKeys[i] && jsonDeepEqual(aObj[key], bObj[key])) +} + /** * Generates a compact, jq-friendly type signature for a value. * e.g. "{title: string, items: [{id: number, name: string}]}" @@ -2665,6 +2696,44 @@ export async function transformMCPResult( } } + if ('content' in result && Array.isArray(result.content)) { + const transformedContent = ( + await Promise.all( + result.content.map(item => transformResultContent(item, name)), + ) + ).flat() + // The MCP spec encourages servers to also serialize structured content + // into a TextContent. When `content` already carries that serialized + // JSON, appending it again would duplicate the same data; when it does + // not (e.g. content is only a preview image), the structured data would + // otherwise be lost. Deduplicate, then merge when actually missing. + const structuredContent = + 'structuredContent' in result ? result.structuredContent : undefined + const serialized = + structuredContent !== undefined ? jsonStringify(structuredContent) : undefined + // Servers often pretty-print the serialized JSON in TextContent while + // structuredContent carries the object, so compare JSON semantics rather + // than byte equality before deciding the data is already present. + const alreadySerialized = + serialized !== undefined && + transformedContent.some( + block => block.type === 'text' && jsonTextMatches(block.text, structuredContent), + ) + if (serialized !== undefined && !alreadySerialized) { + transformedContent.push({ + type: 'text', + text: `Structured content:\n${serialized}`, + }) + } + return { + content: transformedContent, + type: 'contentArray', + schema: structuredContent !== undefined + ? inferCompactSchema(structuredContent) + : inferCompactSchema(transformedContent), + } + } + if ( 'structuredContent' in result && result.structuredContent !== undefined @@ -2675,19 +2744,6 @@ export async function transformMCPResult( schema: inferCompactSchema(result.structuredContent), } } - - if ('content' in result && Array.isArray(result.content)) { - const transformedContent = ( - await Promise.all( - result.content.map(item => transformResultContent(item, name)), - ) - ).flat() - return { - content: transformedContent, - type: 'contentArray', - schema: inferCompactSchema(transformedContent), - } - } } const errorMessage = `MCP server "${name}" tool "${tool}": unexpected response format` @@ -2710,6 +2766,111 @@ function contentContainsImages(content: MCPToolResult): boolean { return content.some(block => block.type === 'image') } +/** + * Persist the text blocks of an image-containing large result to a file and + * return the non-text media blocks plus a file reference, keeping original + * block order. Text blocks stay in the context as summaries bounded by a + * hard budget: the caption following a media block is kept whole from a + * reserved share, all other text is truncated against the shared budget. This + * keeps image captions and ordering visible without letting many small blocks + * re-inflate the context; the full text lives in the file. + * Returns null when there is no text to persist or persistence fails (caller + * falls back to truncation). + */ +export async function persistTextFromImageContent( + content: MCPToolResult, + name: string, + tool: string, +): Promise { + if (!Array.isArray(content)) return null + const fullText: string[] = [] + const kept: ContentBlockParam[] = [] + + // Text summaries share one hard budget. The short caption following a media + // block is kept whole and gets priority over ordinary text, so neither a + // long block nor a run of short log lines can starve the captions that + // explain the images; ordinary text is truncated against whatever the + // captions leave of the budget. Blank text blocks do not cancel caption + // priority. + const captionDemand = collectCaptionDemand(content) + const captionBudget = Math.min(MCP_TEXT_SUMMARY_BUDGET, captionDemand) + let remainingCaptionBudget = captionBudget + let remainingSummaryBudget = MCP_TEXT_SUMMARY_BUDGET - captionBudget + let awaitingCaption = false + + for (const block of content) { + if (block.type === 'text') { + fullText.push(block.text) + if (block.text.trim().length === 0) continue + if ( + awaitingCaption + && block.text.length <= MCP_TEXT_SUMMARY_LIMIT + && remainingCaptionBudget >= block.text.length + ) { + kept.push({ type: 'text', text: block.text }) + remainingCaptionBudget -= block.text.length + } else if (remainingSummaryBudget > 0) { + const summary = block.text.slice(0, Math.min(remainingSummaryBudget, block.text.length)) + kept.push({ type: 'text', text: summary }) + remainingSummaryBudget -= summary.length + } + awaitingCaption = false + } else { + // Keep every non-text media block (images, documents, resources, …). + kept.push(block) + awaitingCaption = true + } + } + const textContent = fullText.join('\n') + if (textContent.trim().length === 0) return null + + // persistToolResult treats an existing file ('wx') as a replay of the same + // invocation, so the id must be unique per call — parallel invocations of + // the same server/tool would otherwise share a timestamp-based file and + // read each other's content. A UUID provides the invocation identity. + const persistId = `mcp-${normalizeNameForMCP(name)}-${normalizeNameForMCP(tool)}-${crypto.randomUUID()}` + const persistResult = await persistToolResult(textContent, persistId) + if (isPersistError(persistResult)) return null + + kept.push({ + type: 'text', + text: getBinaryBlobSavedMessage( + persistResult.filepath, + 'text/plain', + persistResult.originalSize, + `[MCP output from ${name}] `, + ), + }) + return kept +} + +const MCP_TEXT_SUMMARY_LIMIT = 200 +const MCP_TEXT_SUMMARY_BUDGET = 2000 + +/** + * Total length of the captions in the result: the first non-blank short text + * block after each media block. Captions get priority over ordinary text in + * the summary budget. + */ +function collectCaptionDemand(content: ContentBlockParam[]): number { + let demand = 0 + let awaitingCaption = false + for (const block of content) { + if (block.type === 'text') { + if (block.text.trim().length > 0) { + if (awaitingCaption && block.text.length <= MCP_TEXT_SUMMARY_LIMIT) { + demand += block.text.length + } + awaitingCaption = false + } + // Blank blocks do not cancel caption priority. + } else { + awaitingCaption = true + } + } + return demand +} + export async function processMCPResult( result: unknown, tool: string, // Tool name for validation (e.g., "search") @@ -2747,8 +2908,20 @@ export async function processMCPResult( } // If content contains images, fall back to truncation - persisting images as JSON - // defeats the image compression logic and makes them non-viewable + // defeats the image compression logic and makes them non-viewable. + // Large text in the same array (e.g. appended structuredContent) would be + // truncated along with the whole result; persist that text instead so the + // data survives while the images stay visible in the context. if (contentContainsImages(content)) { + const persistedText = await persistTextFromImageContent(content, name, tool) + if (persistedText) { + logEvent('tengu_mcp_large_result_handled', { + outcome: 'persisted', + reason: 'text_persisted_images_kept', + sizeEstimateTokens, + } as AnalyticsMetadata_I_VERIFIED_THIS_IS_NOT_CODE_OR_FILEPATHS) + return persistedText + } logEvent('tengu_mcp_large_result_handled', { outcome: 'truncated', reason: 'contains_images', diff --git a/src/services/tools/toolExecution.ts b/src/services/tools/toolExecution.ts index 40ef38ec..3c95f98e 100644 --- a/src/services/tools/toolExecution.ts +++ b/src/services/tools/toolExecution.ts @@ -1414,7 +1414,7 @@ async function checkPermissionsAndCallTool( ) : await processToolResultBlock(tool, toolUseResult, toolUseID) - // Build content blocks - tool result first, then optional feedback + // Build content blocks - tool result first, then optional feedback. const contentBlocks: ContentBlockParam[] = [toolResultBlock] // Add accept feedback if user provided feedback when approving // (acceptFeedback only exists on PermissionAllowDecision, which is guaranteed here) diff --git a/src/utils/messages.test.ts b/src/utils/messages.test.ts index 69cc4ee9..4cd3cde0 100644 --- a/src/utils/messages.test.ts +++ b/src/utils/messages.test.ts @@ -120,6 +120,68 @@ describe('normalizeMessagesForAPI assistant fragment indexing', () => { }) }) +describe('normalizeMessagesForAPI tool-result media', () => { + test('preserves nested images from restored messages at the API boundary', () => { + const image = { + type: 'image' as const, + source: { + type: 'base64' as const, + media_type: 'image/png' as const, + data: 'AAECAwQ=', + }, + } + const message = createUserMessage({ + content: [ + { + type: 'tool_result', + tool_use_id: 'read-1', + content: [image], + }, + ], + }) + + const [normalized] = normalizeMessagesForAPI([message]) + + expect(normalized?.type).toBe('user') + if (normalized?.type === 'user') { + expect(normalized.message.content).toEqual([ + { + type: 'tool_result', + tool_use_id: 'read-1', + content: [image], + }, + ]) + } + }) + + test('keeps parallel tool results contiguous and preserves their ownership', () => { + const imageA = { + type: 'image' as const, + source: { type: 'base64' as const, media_type: 'image/png' as const, data: 'A' }, + } + const imageB = { + type: 'image' as const, + source: { type: 'base64' as const, media_type: 'image/png' as const, data: 'B' }, + } + const message = createUserMessage({ + content: [ + { type: 'tool_result', tool_use_id: 'tool-a', content: [imageA] }, + { type: 'tool_result', tool_use_id: 'tool-b', content: [imageB] }, + ], + }) + + const [normalized] = normalizeMessagesForAPI([message]) + + expect(normalized?.type).toBe('user') + if (normalized?.type === 'user') { + expect(normalized.message.content).toEqual([ + { type: 'tool_result', tool_use_id: 'tool-a', content: [imageA] }, + { type: 'tool_result', tool_use_id: 'tool-b', content: [imageB] }, + ]) + } + }) +}) + describe('stripSignatureBlocksAfterModelChange', () => { test('removes protected thinking from history produced by another model', () => { const previous = assistant('response-a', [ diff --git a/src/utils/messages.ts b/src/utils/messages.ts index cbdcf1c4..988166f2 100644 --- a/src/utils/messages.ts +++ b/src/utils/messages.ts @@ -2415,7 +2415,9 @@ export function normalizeMessagesForAPI( const smooshed = checkStatsigFeatureGate_CACHED_MAY_BE_STALE( 'tengu_chair_sermon', ) - ? smooshSystemReminderSiblings(mergeAdjacentUserMessages(withNonEmpty)) + ? smooshSystemReminderSiblings( + mergeAdjacentUserMessages(withNonEmpty), + ) : withNonEmpty // Unconditional — catches transcripts persisted before smooshIntoToolResult diff --git a/src/utils/toolResultStorage.hoistImages.test.ts b/src/utils/toolResultStorage.hoistImages.test.ts new file mode 100644 index 00000000..f1e952a9 --- /dev/null +++ b/src/utils/toolResultStorage.hoistImages.test.ts @@ -0,0 +1,91 @@ +import type { ToolResultBlockParam } from '@anthropic-ai/sdk/resources/index.mjs' +import { describe, expect, test } from 'bun:test' +import { processToolResultBlock } from './toolResultStorage.js' + +function makeTool() { + return { + name: 'test-tool', + maxResultSizeChars: 100_000, + mapToolResultToToolResultBlockParam: (result: ToolResultBlockParam) => result, + } +} + +describe('tool result media preservation', () => { + test('keeps mixed media ordering and tool ownership inside tool_result', async () => { + const result: ToolResultBlockParam = { + type: 'tool_result', + tool_use_id: 'tool-a', + content: [ + { type: 'text', text: 'before' }, + { type: 'image', source: { type: 'base64', media_type: 'image/png', data: 'a' } }, + { type: 'text', text: 'after' }, + ], + } + + const processed = await processToolResultBlock(makeTool(), result, 'tool-a') + + expect(processed).toEqual(result) + expect((processed.content as Array<{ type: string }>).map(block => block.type)).toEqual([ + 'text', + 'image', + 'text', + ]) + }) + + test('keeps parallel tool results as separate blocks with stable ids', async () => { + const results: ToolResultBlockParam[] = [ + { + type: 'tool_result', + tool_use_id: 'tool-a', + content: [{ type: 'image', source: { type: 'base64', media_type: 'image/png', data: 'a' } }], + }, + { + type: 'tool_result', + tool_use_id: 'tool-b', + content: [{ type: 'image', source: { type: 'base64', media_type: 'image/jpeg', data: 'b' } }], + }, + ] + + const processed = await Promise.all( + results.map(result => processToolResultBlock(makeTool(), result, result.tool_use_id)), + ) + + expect(processed.map(result => result.tool_use_id)).toEqual(['tool-a', 'tool-b']) + expect( + processed.map(result => (result.content as Array<{ type: string }>)[0].type), + ).toEqual(['image', 'image']) + }) + + test('keeps error tool results in their original shape', async () => { + const result: ToolResultBlockParam = { + type: 'tool_result', + tool_use_id: 'tool-error', + is_error: true, + content: [{ type: 'image', source: { type: 'base64', media_type: 'image/png', data: 'error' } }], + } + + const processed = await processToolResultBlock(makeTool(), result, 'tool-error') + + expect(processed.is_error).toBe(true) + expect(processed.content).toHaveLength(1) + expect((processed.content as Array<{ type: string }>)[0].type).toBe('image') + }) + + test('keeps document blocks alongside images', async () => { + const result: ToolResultBlockParam = { + type: 'tool_result', + tool_use_id: 'tool-document', + content: [ + { type: 'document', source: { type: 'base64', media_type: 'application/pdf', data: 'pdf' } }, + { type: 'image', source: { type: 'base64', media_type: 'image/png', data: 'image' } }, + ], + } + + const processed = await processToolResultBlock(makeTool(), result, 'tool-document') + + expect((processed.content as Array<{ type: string }>).map(block => block.type)).toEqual([ + 'document', + 'image', + ]) + }) +}) diff --git a/tests/azureOpenAI.test.ts b/tests/azureOpenAI.test.ts deleted file mode 100644 index dede1fc1..00000000 --- a/tests/azureOpenAI.test.ts +++ /dev/null @@ -1,124 +0,0 @@ -import { expect, test } from "bun:test" -import { - buildAzureOpenAIInput, - parseAzureOpenAIResponse, - resolveAzureOpenAIEndpoint, - resolveAzureOpenAIDeployment, -} from "../src/services/api/azureOpenAI.js" - -test("resolveAzureOpenAIEndpoint appends responses path and api-version", () => { - const prevBase = process.env.AZURE_OPENAI_BASE_URL - const prevVersion = process.env.AZURE_OPENAI_API_VERSION - process.env.AZURE_OPENAI_BASE_URL = - "https://example.cognitiveservices.azure.com/" - process.env.AZURE_OPENAI_API_VERSION = "2025-04-01-preview" - - const url = resolveAzureOpenAIEndpoint() - expect(url).toContain("/openai/responses") - expect(url).toContain("api-version=2025-04-01-preview") - - process.env.AZURE_OPENAI_BASE_URL = prevBase - process.env.AZURE_OPENAI_API_VERSION = prevVersion -}) - -test("resolveAzureOpenAIEndpoint normalizes existing Azure OpenAI paths", () => { - const prevBase = process.env.AZURE_OPENAI_BASE_URL - const prevVersion = process.env.AZURE_OPENAI_API_VERSION - process.env.AZURE_OPENAI_API_VERSION = "2025-04-01-preview" - - process.env.AZURE_OPENAI_BASE_URL = - "https://example.cognitiveservices.azure.com/openai/v1/?foo=bar" - let url = new URL(resolveAzureOpenAIEndpoint()) - expect(url.pathname).toBe("/openai/responses") - expect(url.searchParams.get("foo")).toBe("bar") - expect(url.searchParams.get("api-version")).toBe("2025-04-01-preview") - - process.env.AZURE_OPENAI_BASE_URL = - "https://example.cognitiveservices.azure.com/openai/responses?api-version=custom" - url = new URL(resolveAzureOpenAIEndpoint()) - expect(url.pathname).toBe("/openai/responses") - expect(url.searchParams.get("api-version")).toBe("2025-04-01-preview") - - process.env.AZURE_OPENAI_BASE_URL = prevBase - process.env.AZURE_OPENAI_API_VERSION = prevVersion -}) - -test("buildAzureOpenAIInput maps tool_use and tool_result", () => { - const input = buildAzureOpenAIInput([ - { - type: "assistant", - message: { - content: [ - { type: "text", text: "Running tool" }, - { - type: "tool_use", - id: "tool_1", - name: "my_tool", - input: { foo: "bar" }, - }, - ], - }, - }, - { - type: "user", - message: { - content: [ - { - type: "tool_result", - tool_use_id: "tool_1", - content: [{ type: "text", text: "ok" }], - }, - ], - }, - }, - ]) - - expect(input.some(msg => msg.role === "assistant")).toBe(true) - expect(input.some(msg => msg.role === "tool")).toBe(true) -}) - -test("parseAzureOpenAIResponse derives tool stop reason from function calls", () => { - const result = parseAzureOpenAIResponse({ - id: "resp-1", - output: [ - { - type: "function_call", - id: "call-1", - name: "my_tool", - arguments: "{\"foo\":\"bar\"}", - }, - ], - }) - - expect(result.stopReason).toBe("tool_use") - expect(result.content[0]?.type).toBe("tool_use") -}) - -test("resolveAzureOpenAIDeployment throws when codex mapping is missing", () => { - const prevBase = process.env.AZURE_OPENAI_BASE_URL - const prevEnv = process.env.AZURE_OPENAI_CODEX_DEPLOYMENT - process.env.AZURE_OPENAI_BASE_URL = - "https://example.cognitiveservices.azure.com/" - delete process.env.AZURE_OPENAI_CODEX_DEPLOYMENT - - expect(() => resolveAzureOpenAIDeployment("gpt-5.2-codex")).toThrow() - expect(() => resolveAzureOpenAIDeployment("gpt-5.3-codex")).toThrow() - expect(() => resolveAzureOpenAIDeployment("gpt-5.4-codex")).toThrow() - - process.env.AZURE_OPENAI_BASE_URL = prevBase - process.env.AZURE_OPENAI_CODEX_DEPLOYMENT = prevEnv -}) - -test("resolveAzureOpenAIDeployment uses env default even if name matches", () => { - const prevBase = process.env.AZURE_OPENAI_BASE_URL - const prevEnv = process.env.AZURE_OPENAI_CODEX_DEPLOYMENT - process.env.AZURE_OPENAI_BASE_URL = - "https://example.cognitiveservices.azure.com/" - process.env.AZURE_OPENAI_CODEX_DEPLOYMENT = "gpt-5.2-codex" - - const resolved = resolveAzureOpenAIDeployment("gpt-5.2-codex") - expect(resolved).toBe("gpt-5.2-codex") - - process.env.AZURE_OPENAI_BASE_URL = prevBase - process.env.AZURE_OPENAI_CODEX_DEPLOYMENT = prevEnv -})