From f4679f218d1a7b1475f72cd763601751ca094ad5 Mon Sep 17 00:00:00 2001 From: Owie6789 <151057755+Owie6789@users.noreply.github.com> Date: Wed, 23 Sep 2026 12:25:30 +0000 Subject: [PATCH 01/60] Move the SDK, Claude CLI and Codex binary pins together Three pins decide which agent runtime ships: the Claude Agent SDK version, the Claude CLI version the download scripts fetch for bundling, and the Codex version. They were set independently, so the SDK and the bundled binary could drift apart. Move all three in one change: SDK 0.2.45 to 0.3.270, CLI 2.1.45 to 2.1.270, Codex 0.137.0 to 0.154.0. 0.3.270 bundles CLI 2.1.270: its manifest.json lists linux-x64 with checksum 3a624a5a7cd79bbad4d32bd7db36f1197ecf458bc5bf1e2aed81834a01ad3ef0, which is the sha256 of the binary that install places in node_modules/@anthropic-ai/claude-agent-sdk-linux-x64/claude. The download script's offline fallback carries the same version as the package.json script so a retry after a failed manifest lookup cannot silently fetch 2.1.45. The lockfile gains the SDK's eight per-platform native packages, @anthropic-ai/sdk 0.128.0 and its transitive dependencies, and loses the fifteen @img/sharp-* entries that only the old SDK pulled in. sharp is now unreachable transitively while scripts/generate-icon.mjs still imports it, so declaring it as a devDependency is a separate change. Co-authored-by: arena-agent <297053741+arena-agent@users.noreply.github.com> --- bun.lock | 64 +++++++++++++++--------------- package.json | 10 ++--- scripts/download-claude-binary.mjs | 2 +- 3 files changed, 38 insertions(+), 38 deletions(-) diff --git a/bun.lock b/bun.lock index df2f609b..293b2f4c 100644 --- a/bun.lock +++ b/bun.lock @@ -7,7 +7,7 @@ "dependencies": { "@ai-sdk/google": "^3.0.65", "@ai-sdk/react": "^3.0.14", - "@anthropic-ai/claude-agent-sdk": "0.2.45", + "@anthropic-ai/claude-agent-sdk": "0.3.270", "@effect/platform-node": "4.0.0-rc.112", "@effect/platform-node-shared": "4.0.0-rc.112", "@git-diff-view/react": "^0.0.35", @@ -171,7 +171,25 @@ "@antfu/install-pkg": ["@antfu/install-pkg@1.1.0", "", { "dependencies": { "package-manager-detector": "^1.3.0", "tinyexec": "^1.0.1" } }, "sha512-MGQsmw10ZyI+EJo45CdSER4zEb+p31LpDAFp2Z3gkSd1yqVZGi0Ebx++YTEMonJy4oChEMLsxZ64j8FH6sSqtQ=="], - "@anthropic-ai/claude-agent-sdk": ["@anthropic-ai/claude-agent-sdk@0.2.45", "", { "optionalDependencies": { "@img/sharp-darwin-arm64": "^0.33.5", "@img/sharp-darwin-x64": "^0.33.5", "@img/sharp-linux-arm": "^0.33.5", "@img/sharp-linux-arm64": "^0.33.5", "@img/sharp-linux-x64": "^0.33.5", "@img/sharp-linuxmusl-arm64": "^0.33.5", "@img/sharp-linuxmusl-x64": "^0.33.5", "@img/sharp-win32-x64": "^0.33.5" }, "peerDependencies": { "zod": "^4.0.0" } }, "sha512-AKH2hKoJNyjLf9ThAttKqbmCjUFg7qs/8+LR/UTVX20fCLn359YH9WrQc6dAiAfi8RYNA+mWwrNYCAq+Sdo5Ag=="], + "@anthropic-ai/claude-agent-sdk": ["@anthropic-ai/claude-agent-sdk@0.3.270", "", { "optionalDependencies": { "@anthropic-ai/claude-agent-sdk-darwin-arm64": "0.3.270", "@anthropic-ai/claude-agent-sdk-darwin-x64": "0.3.270", "@anthropic-ai/claude-agent-sdk-linux-arm64": "0.3.270", "@anthropic-ai/claude-agent-sdk-linux-arm64-musl": "0.3.270", "@anthropic-ai/claude-agent-sdk-linux-x64": "0.3.270", "@anthropic-ai/claude-agent-sdk-linux-x64-musl": "0.3.270", "@anthropic-ai/claude-agent-sdk-win32-arm64": "0.3.270", "@anthropic-ai/claude-agent-sdk-win32-x64": "0.3.270" }, "peerDependencies": { "@anthropic-ai/sdk": ">=0.93.0", "@modelcontextprotocol/sdk": "^1.29.0", "zod": "^4.0.0" } }, "sha512-sSfcm5Nhb+WHeBCxqeHRRQMUKPmFTL+zgv5xcRUVaFMLttfNEbn3IZJE+fLJJmy4h3J8zdc5sXdSa1JxyB8ppQ=="], + + "@anthropic-ai/claude-agent-sdk-darwin-arm64": ["@anthropic-ai/claude-agent-sdk-darwin-arm64@0.3.270", "", { "os": "darwin", "cpu": "arm64" }, "sha512-nk7BP+i559rheYz9DIwAfevd4DulQXP0mXPP+MeO2fGuIGFmzhE/c0JRm9YswXv5HdaYJvSzjGIB7dVA01NehA=="], + + "@anthropic-ai/claude-agent-sdk-darwin-x64": ["@anthropic-ai/claude-agent-sdk-darwin-x64@0.3.270", "", { "os": "darwin", "cpu": "x64" }, "sha512-89Uql8Oalm52ojdZZeNLU24LKrU+WG9QR7d6YP9ly4aY0YvQUJDaTosbDkijQngUETbPBFdKuVp6fNM8x0Zt3Q=="], + + "@anthropic-ai/claude-agent-sdk-linux-arm64": ["@anthropic-ai/claude-agent-sdk-linux-arm64@0.3.270", "", { "os": "linux", "cpu": "arm64" }, "sha512-iHPYqwetyeO4tZPzXyKZz0hUh2fLpwu/+biGTxxynikG3XrYknovZ/znGDA3TxjSFurqHf5IIDA+SOjh9OPs0A=="], + + "@anthropic-ai/claude-agent-sdk-linux-arm64-musl": ["@anthropic-ai/claude-agent-sdk-linux-arm64-musl@0.3.270", "", { "os": "linux", "cpu": "arm64" }, "sha512-2BlLk2MAohWG2h43RKcjCA4ooMfBxzKf4yyYfOVv1DtYr8zPU876MHCT1VXB2BaemjKA0pdYJJBzH6pncwx6MQ=="], + + "@anthropic-ai/claude-agent-sdk-linux-x64": ["@anthropic-ai/claude-agent-sdk-linux-x64@0.3.270", "", { "os": "linux", "cpu": "x64" }, "sha512-ADaqz2viyAd0GUxdupYLX/K0YJb46xckNpEeWxyLK/9+26b/R5stbaGDyL29fIzyq1ymnNUOaCgEWEemgX0kEA=="], + + "@anthropic-ai/claude-agent-sdk-linux-x64-musl": ["@anthropic-ai/claude-agent-sdk-linux-x64-musl@0.3.270", "", { "os": "linux", "cpu": "x64" }, "sha512-mzH3lnbzrbDGrTf75jLEmkbvkKRLLgmjLaWvf3QuUsgcw+aU69aOY0mW33oOrsuq5zg330uI3B4e68f4LbxNIA=="], + + "@anthropic-ai/claude-agent-sdk-win32-arm64": ["@anthropic-ai/claude-agent-sdk-win32-arm64@0.3.270", "", { "os": "win32", "cpu": "arm64" }, "sha512-Pexeu26cLZByhs6VlrawNYAEu+QE2YptvwNkXsmpLRm7Q/C/M0N/BZBmuUcikXrQQ9cuVzSjuuTTpZt64mvtLA=="], + + "@anthropic-ai/claude-agent-sdk-win32-x64": ["@anthropic-ai/claude-agent-sdk-win32-x64@0.3.270", "", { "os": "win32", "cpu": "x64" }, "sha512-9UyfFcUYsyUZqSe/xX9nIJ1Og6i8FxhlQ35BDi79Ik5He87XFxEIGUvJuMcl7Mq2e3panyhekELFE+9H79xKdw=="], + + "@anthropic-ai/sdk": ["@anthropic-ai/sdk@0.128.0", "", { "dependencies": { "json-schema-to-ts": "^3.1.1", "standardwebhooks": "^1.0.0" }, "peerDependencies": { "zod": "^3.25.0 || ^4.0.0" }, "optionalPeers": ["zod"], "bin": { "anthropic-ai-sdk": "bin/cli" } }, "sha512-tl5cBFZC1jVFrTuQyvQObA4WmuNgcYVeXfriqOumgtLAHUm4gPUj18YnszBW60v0PrjDB8nBDOC6wd9CmHQFlA=="], "@apm-js-collab/code-transformer": ["@apm-js-collab/code-transformer@0.8.2", "", {}, "sha512-YRjJjNq5KFSjDUoqu5pFUWrrsvGOxl6c3bu+uMFc9HNNptZ2rNU/TI2nLw4jnhQNtka972Ee2m3uqbvDQtPeCA=="], @@ -211,6 +229,8 @@ "@babel/plugin-transform-react-jsx-source": ["@babel/plugin-transform-react-jsx-source@7.27.1", "", { "dependencies": { "@babel/helper-plugin-utils": "^7.27.1" }, "peerDependencies": { "@babel/core": "^7.0.0-0" } }, "sha512-zbwoTsBruTeKB9hSq73ha66iFeJHuaFkUbwvqElnygoNbj/jHRsSeokowZFN3CZ64IvEqcmmkVe89OPXc7ldAw=="], + "@babel/runtime": ["@babel/runtime@7.29.7", "", {}, "sha512-Nq8OhGWiZIZGV6hLHoyAKLLcJihP/xFeBMGJoUrxTX2psI8dCifzLhZISFb+VWS3wFMRDmCGw5R+dOySCqPLhw=="], + "@babel/template": ["@babel/template@7.28.6", "", { "dependencies": { "@babel/code-frame": "^7.28.6", "@babel/parser": "^7.28.6", "@babel/types": "^7.28.6" } }, "sha512-YA6Ma2KsCdGb+WC6UpBVFJGXL58MDA6oyONbjyF/+5sBgxY/dwkhLogbMT2GXXyU84/IhRw/2D1Os1B/giz+BQ=="], "@babel/traverse": ["@babel/traverse@7.28.6", "", { "dependencies": { "@babel/code-frame": "^7.28.6", "@babel/generator": "^7.28.6", "@babel/helper-globals": "^7.28.0", "@babel/parser": "^7.28.6", "@babel/template": "^7.28.6", "@babel/types": "^7.28.6", "debug": "^4.3.1" } }, "sha512-fgWX62k02qtjqdSNTAGxmKYY/7FSL9WAS1o2Hu5+I5m9T0yxZzr4cnrfXQ/MX0rIifthCSs6FKTlzYbJcPtMNg=="], @@ -353,36 +373,6 @@ "@iconify/utils": ["@iconify/utils@3.1.0", "", { "dependencies": { "@antfu/install-pkg": "^1.1.0", "@iconify/types": "^2.0.0", "mlly": "^1.8.0" } }, "sha512-Zlzem1ZXhI1iHeeERabLNzBHdOa4VhQbqAcOQaMKuTuyZCpwKbC2R4Dd0Zo3g9EAc+Y4fiarO8HIHRAth7+skw=="], - "@img/sharp-darwin-arm64": ["@img/sharp-darwin-arm64@0.33.5", "", { "optionalDependencies": { "@img/sharp-libvips-darwin-arm64": "1.0.4" }, "os": "darwin", "cpu": "arm64" }, "sha512-UT4p+iz/2H4twwAoLCqfA9UH5pI6DggwKEGuaPy7nCVQ8ZsiY5PIcrRvD1DzuY3qYL07NtIQcWnBSY/heikIFQ=="], - - "@img/sharp-darwin-x64": ["@img/sharp-darwin-x64@0.33.5", "", { "optionalDependencies": { "@img/sharp-libvips-darwin-x64": "1.0.4" }, "os": "darwin", "cpu": "x64" }, "sha512-fyHac4jIc1ANYGRDxtiqelIbdWkIuQaI84Mv45KvGRRxSAa7o7d1ZKAOBaYbnepLC1WqxfpimdeWfvqqSGwR2Q=="], - - "@img/sharp-libvips-darwin-arm64": ["@img/sharp-libvips-darwin-arm64@1.0.4", "", { "os": "darwin", "cpu": "arm64" }, "sha512-XblONe153h0O2zuFfTAbQYAX2JhYmDHeWikp1LM9Hul9gVPjFY427k6dFEcOL72O01QxQsWi761svJ/ev9xEDg=="], - - "@img/sharp-libvips-darwin-x64": ["@img/sharp-libvips-darwin-x64@1.0.4", "", { "os": "darwin", "cpu": "x64" }, "sha512-xnGR8YuZYfJGmWPvmlunFaWJsb9T/AO2ykoP3Fz/0X5XV2aoYBPkX6xqCQvUTKKiLddarLaxpzNe+b1hjeWHAQ=="], - - "@img/sharp-libvips-linux-arm": ["@img/sharp-libvips-linux-arm@1.0.5", "", { "os": "linux", "cpu": "arm" }, "sha512-gvcC4ACAOPRNATg/ov8/MnbxFDJqf/pDePbBnuBDcjsI8PssmjoKMAz4LtLaVi+OnSb5FK/yIOamqDwGmXW32g=="], - - "@img/sharp-libvips-linux-arm64": ["@img/sharp-libvips-linux-arm64@1.0.4", "", { "os": "linux", "cpu": "arm64" }, "sha512-9B+taZ8DlyyqzZQnoeIvDVR/2F4EbMepXMc/NdVbkzsJbzkUjhXv/70GQJ7tdLA4YJgNP25zukcxpX2/SueNrA=="], - - "@img/sharp-libvips-linux-x64": ["@img/sharp-libvips-linux-x64@1.0.4", "", { "os": "linux", "cpu": "x64" }, "sha512-MmWmQ3iPFZr0Iev+BAgVMb3ZyC4KeFc3jFxnNbEPas60e1cIfevbtuyf9nDGIzOaW9PdnDciJm+wFFaTlj5xYw=="], - - "@img/sharp-libvips-linuxmusl-arm64": ["@img/sharp-libvips-linuxmusl-arm64@1.0.4", "", { "os": "linux", "cpu": "arm64" }, "sha512-9Ti+BbTYDcsbp4wfYib8Ctm1ilkugkA/uscUn6UXK1ldpC1JjiXbLfFZtRlBhjPZ5o1NCLiDbg8fhUPKStHoTA=="], - - "@img/sharp-libvips-linuxmusl-x64": ["@img/sharp-libvips-linuxmusl-x64@1.0.4", "", { "os": "linux", "cpu": "x64" }, "sha512-viYN1KX9m+/hGkJtvYYp+CCLgnJXwiQB39damAO7WMdKWlIhmYTfHjwSbQeUK/20vY154mwezd9HflVFM1wVSw=="], - - "@img/sharp-linux-arm": ["@img/sharp-linux-arm@0.33.5", "", { "optionalDependencies": { "@img/sharp-libvips-linux-arm": "1.0.5" }, "os": "linux", "cpu": "arm" }, "sha512-JTS1eldqZbJxjvKaAkxhZmBqPRGmxgu+qFKSInv8moZ2AmT5Yib3EQ1c6gp493HvrvV8QgdOXdyaIBrhvFhBMQ=="], - - "@img/sharp-linux-arm64": ["@img/sharp-linux-arm64@0.33.5", "", { "optionalDependencies": { "@img/sharp-libvips-linux-arm64": "1.0.4" }, "os": "linux", "cpu": "arm64" }, "sha512-JMVv+AMRyGOHtO1RFBiJy/MBsgz0x4AWrT6QoEVVTyh1E39TrCUpTRI7mx9VksGX4awWASxqCYLCV4wBZHAYxA=="], - - "@img/sharp-linux-x64": ["@img/sharp-linux-x64@0.33.5", "", { "optionalDependencies": { "@img/sharp-libvips-linux-x64": "1.0.4" }, "os": "linux", "cpu": "x64" }, "sha512-opC+Ok5pRNAzuvq1AG0ar+1owsu842/Ab+4qvU879ippJBHvyY5n2mxF1izXqkPYlGuP/M556uh53jRLJmzTWA=="], - - "@img/sharp-linuxmusl-arm64": ["@img/sharp-linuxmusl-arm64@0.33.5", "", { "optionalDependencies": { "@img/sharp-libvips-linuxmusl-arm64": "1.0.4" }, "os": "linux", "cpu": "arm64" }, "sha512-XrHMZwGQGvJg2V/oRSUfSAfjfPxO+4DkiRh6p2AFjLQztWUuY/o8Mq0eMQVIY7HJ1CDQUJlxGGZRw1a5bqmd1g=="], - - "@img/sharp-linuxmusl-x64": ["@img/sharp-linuxmusl-x64@0.33.5", "", { "optionalDependencies": { "@img/sharp-libvips-linuxmusl-x64": "1.0.4" }, "os": "linux", "cpu": "x64" }, "sha512-WT+d/cgqKkkKySYmqoZ8y3pxx7lx9vVejxW/W4DOFMYVSkErR+w7mf2u8m/y4+xHe7yY9DAXQMWQhpnMuFfScw=="], - - "@img/sharp-win32-x64": ["@img/sharp-win32-x64@0.33.5", "", { "os": "win32", "cpu": "x64" }, "sha512-MpY/o8/8kj+EcnxwvrP4aTJSWw/aZ7JIGR4aBeZkZw5B7/Jn+tY9/VNwtcoGmdT7GfggGIU4kygOMSbYnOrAbg=="], - "@isaacs/balanced-match": ["@isaacs/balanced-match@4.0.1", "", {}, "sha512-yzMTt9lEb8Gv7zRioUilSglI0c0smZ9k5D65677DLWLtWJaXIS3CqcGyUFByYKlnUj6TkjLVs54fBl6+TiGQDQ=="], "@isaacs/brace-expansion": ["@isaacs/brace-expansion@5.0.0", "", { "dependencies": { "@isaacs/balanced-match": "^4.0.1" } }, "sha512-ZT55BDLV0yv0RBm2czMiZ+SqCGO7AvmOM3G/w2xhVPH+te0aKgFjmBvGlL1dH+ql2tgGO3MVrbb3jCKyvpgnxA=="], @@ -741,6 +731,8 @@ "@sindresorhus/is": ["@sindresorhus/is@4.6.0", "", {}, "sha512-t09vSN3MdfsyCHoFcTRCH/iUtG7OJ0CsjzB8cjAmKc/va/kIgeDI/TxsigdncE/4be734m0cvIYwNaV4i2XqAw=="], + "@stablelib/base64": ["@stablelib/base64@1.0.1", "", {}, "sha512-1bnPQqSxSuc3Ii6MhBysoWCg58j97aUjuCSZrGSmDxNqtytIi0k8utUenAwTZN4V5mXXYGsVUI9zeBqy+jBOSQ=="], + "@standard-schema/spec": ["@standard-schema/spec@1.1.0", "", {}, "sha512-l2aFy5jALhniG5HgqrD6jXLi/rUWrKvqN/qJx6yoJsgKhblVd+iqqU4RCXavm/jPityDo5TCvKMnpjKnOriy0w=="], "@szmarczak/http-timer": ["@szmarczak/http-timer@4.0.6", "", { "dependencies": { "defer-to-connect": "^2.0.0" } }, "sha512-4BAffykYOgO+5nzBWYwE3W90sBgLJoUPRWWcL8wlyiM8IB8ipJz3UMJ9KXQd1RKQXpKp8Tutn80HZtWsu2u76w=="], @@ -1417,6 +1409,8 @@ "fast-json-stable-stringify": ["fast-json-stable-stringify@2.1.0", "", {}, "sha512-lhd/wF+Lk98HZoTCtlVraHtfh5XYijIjalXck7saUtuanSDyLMxnHhSXEDJqHxD7msR8D0uCmqlkwjCV8xvwHw=="], + "fast-sha256": ["fast-sha256@1.3.0", "", {}, "sha512-n11RGP/lrWEFI/bWdygLxhI+pVeo1ZYIVwvvPkW7azl/rOy+F3HYRZ2K5zeE9mmkhQppyv9sQFx0JM9UabnpPQ=="], + "fast-uri": ["fast-uri@3.1.0", "", {}, "sha512-iPeeDKJSWf4IEOasVVrknXpaBV0IApz/gp7S2bb7Z4Lljbl2MGJRqInZiUrQwV16cpzw/D3S5j5Julj/gT52AA=="], "fastq": ["fastq@1.20.1", "", { "dependencies": { "reusify": "^1.0.4" } }, "sha512-GGToxJ/w1x32s/D2EKND7kTil4n8OVk/9mycTc4VDza13lOvpUZTGX3mFSCtV9ksdGBVzvsyAVLM6mHFThxXxw=="], @@ -1639,6 +1633,8 @@ "json-schema": ["json-schema@0.4.0", "", {}, "sha512-es94M3nTIfsEPisRafak+HDLfHXnKBhV3vU5eqPcS3flIWqcxJWgXHXiey3YrpaNsanY5ei1VoYEbOzijuq9BA=="], + "json-schema-to-ts": ["json-schema-to-ts@3.1.1", "", { "dependencies": { "@babel/runtime": "^7.18.3", "ts-algebra": "^2.0.0" } }, "sha512-+DWg8jCJG2TEnpy7kOm/7/AxaYoaRbjVB4LFZLySZlWn8exGs3A4OLJR966cVvU26N7X9TWxl+Jsw7dzAqKT6g=="], + "json-schema-traverse": ["json-schema-traverse@1.0.0", "", {}, "sha512-NM8/P9n3XjXhIZn1lLhkFaACTOURQXjWhV4BA/RnOv8xvgqtqpAX9IO4mRQxSx1Rlo4tqzeqb0sOlruaOy3dug=="], "json-schema-typed": ["json-schema-typed@8.0.2", "", {}, "sha512-fQhoXdcvc3V28x7C7BMs4P5+kNlgUURe2jmUT1T//oBRMDrqy1QPelJimwZGo7Hg9VPV3EQV5Bnq4hbFy2vetA=="], @@ -2215,6 +2211,8 @@ "stackback": ["stackback@0.0.2", "", {}, "sha512-1XMJE5fQo1jGH6Y/7ebnwPOBEkIEnT4QF32d5R1+VXdXveM0IBMJt8zfaxX1P3QhVwrYe+576+jkANtSS2mBbw=="], + "standardwebhooks": ["standardwebhooks@1.1.1", "", { "dependencies": { "@stablelib/base64": "^1.0.0", "fast-sha256": "^1.3.0" } }, "sha512-bCbX9ZEyFkWPsRz7Bl3NuQUJohmwGSev/yhr7vhaGPlc4AfIrspIRa6cPTBuI1ItmrTDJ4d/S2hCsfe4+vQGnQ=="], + "stat-mode": ["stat-mode@1.0.0", "", {}, "sha512-jH9EhtKIjuXZ2cWxmXS8ZP80XyC3iasQxMDV8jzhNJpfDb7VbQLVW4Wvsxz9QZvzV+G4YoSfBUVKDOyxLzi/sg=="], "state-local": ["state-local@1.0.7", "", {}, "sha512-HTEHMNieakEnoe33shBYcZ7NX83ACUjCu8c40iOGEZsngj9zRnkqS9j1pqQPXwobB0ZcVTk27REb7COQ0UR59w=="], @@ -2305,6 +2303,8 @@ "truncate-utf8-bytes": ["truncate-utf8-bytes@1.0.2", "", { "dependencies": { "utf8-byte-length": "^1.0.1" } }, "sha512-95Pu1QXQvruGEhv62XCMO3Mm90GscOCClvrIUwCM0PYOXK3kaF3l3sIHxx71ThJfcbM2O5Au6SO3AWCSEfW4mQ=="], + "ts-algebra": ["ts-algebra@2.0.0", "", {}, "sha512-FPAhNPFMrkwz76P7cdjdmiShwMynZYN6SgOujD1urY4oNm80Ou9oMdmbR45LotcKOXoy7wSmHkRFE6Mxbrhefw=="], + "ts-dedent": ["ts-dedent@2.2.0", "", {}, "sha512-q5W7tVM71e2xjHZTlgfTDoPF/SmqKG5hddq9SzR49CH2hayqRKJtQ4mtRlSxKaJlR/+9rEM+mnBHf7I2/BQcpQ=="], "ts-interface-checker": ["ts-interface-checker@0.1.13", "", {}, "sha512-Y/arvbn+rrz3JCKl9C4kVNfTfSm2/mEp5FSz5EsZSANGPSlQrpRI5M4PKF+mJnE52jOO90PnPSc3Ur3bTQw0gA=="], diff --git a/package.json b/package.json index 63b70cc2..7cb8454a 100644 --- a/package.json +++ b/package.json @@ -18,10 +18,10 @@ "package:linux": "electron-builder --linux", "dist": "electron-builder", "dist:manifest": "node scripts/generate-update-manifest.mjs", - "claude:download": "node scripts/download-claude-binary.mjs --version=2.1.45", - "claude:download:all": "node scripts/download-claude-binary.mjs --version=2.1.45 --all", - "codex:download": "node scripts/download-codex-binary.mjs --version=0.137.0", - "codex:download:all": "node scripts/download-codex-binary.mjs --version=0.137.0 --all", + "claude:download": "node scripts/download-claude-binary.mjs --version=2.1.270", + "claude:download:all": "node scripts/download-claude-binary.mjs --version=2.1.270 --all", + "codex:download": "node scripts/download-codex-binary.mjs --version=0.154.0", + "codex:download:all": "node scripts/download-codex-binary.mjs --version=0.154.0 --all", "release:dev": "rm -rf release && bun run claude:download && bun run codex:download && bun run build && bun run package:mac && rm -rf node_modules && bun i", "icon:generate": "node scripts/generate-icon.mjs", @@ -47,7 +47,7 @@ "dependencies": { "@ai-sdk/google": "^3.0.65", "@ai-sdk/react": "^3.0.14", - "@anthropic-ai/claude-agent-sdk": "0.2.45", + "@anthropic-ai/claude-agent-sdk": "0.3.270", "@effect/platform-node": "4.0.0-rc.112", "@effect/platform-node-shared": "4.0.0-rc.112", "@git-diff-view/react": "^0.0.35", diff --git a/scripts/download-claude-binary.mjs b/scripts/download-claude-binary.mjs index a2df70cf..30b123f4 100644 --- a/scripts/download-claude-binary.mjs +++ b/scripts/download-claude-binary.mjs @@ -149,7 +149,7 @@ async function getLatestVersion() { } // Fallback to known version (should be updated periodically) - return "2.1.45" + return "2.1.270" } /** From 7f992599671ed2c186ab9dbf994a773c46650668 Mon Sep 17 00:00:00 2001 From: Owie6789 <151057755+Owie6789@users.noreply.github.com> Date: Wed, 23 Sep 2026 12:42:21 +0000 Subject: [PATCH 02/60] Classify the whole 0.3.270 stream dialect and map the two new events At 0.2.45 the SDK could not type its own stream boundary: SDKRateLimitEvent was never declared and the assistant, user and stream-event payloads resolved to any because @anthropic-ai/sdk was not in the tree. The local read contracts in types.ts were therefore guesses. At 0.3.270 that package is a dependency and all 39 message members are declared, so the read contracts are now derived from the SDK types instead of spelled out beside them: each local block union keeps the shapes the translator reads and adds everything else the SDK can send through Exclude, which keeps a fixture small while a real payload still satisfies the contract. Two fields had to widen to stay supertypes: a tool result's content is optional and can hold a document, a search result or a browser state rather than only text and image, and usage cache counters can be null for a tier that did not apply. ClaudeStreamMessage covers the rest of the dialect by exclusion from SDKMessage rather than by a hand-written import list, so a member a future release adds joins the union on the bump and fails the classification guards until someone decides what it means. Those guards now cover 11 top-level types and 28 system subtypes, each handled or internal with a named reason in the comment above them. Two events gain a consumer. The api_retry system subtype maps to the existing retry-notification chunk, which both chat transports already toast, so a backoff reads as a retry instead of a stalled stream. prompt_suggestion maps to a new prompt-suggestion chunk carrying the suggestion and the session id, trimmed and bounded to 2000 characters because the string is provider-authored and crosses IPC into the composer. Co-authored-by: arena-agent <297053741+arena-agent@users.noreply.github.com> --- src/main/lib/claude/transform.test.ts | 69 +++++++++++- src/main/lib/claude/transform.ts | 106 ++++++++++++++++--- src/main/lib/claude/types.ts | 144 +++++++++++++++++--------- 3 files changed, 259 insertions(+), 60 deletions(-) diff --git a/src/main/lib/claude/transform.test.ts b/src/main/lib/claude/transform.test.ts index 374df6e6..481e126b 100644 --- a/src/main/lib/claude/transform.test.ts +++ b/src/main/lib/claude/transform.test.ts @@ -6,7 +6,7 @@ */ import { afterEach, describe, expect, it, vi } from "vitest" import { createTransformer, INTERNAL_STREAM_MESSAGE_TYPES } from "./transform" -import type { ClaudeStreamMessage } from "./types" +import type { ClaudeStreamMessage, UIMessageChunk } from "./types" const U1 = "00000000-0000-4000-8000-000000000001" const U2 = "00000000-0000-4000-8000-000000000002" @@ -22,6 +22,16 @@ function translate(...messages: ClaudeStreamMessage[]): string[] { return out } +/** The whole chunk, for the two mappings whose payload is the point. */ +function translateChunks(...messages: ClaudeStreamMessage[]): UIMessageChunk[] { + const transform = createTransformer() + const out: UIMessageChunk[] = [] + for (const msg of messages) { + for (const chunk of transform(msg)) out.push(chunk) + } + return out +} + afterEach(() => { vi.restoreAllMocks() }) @@ -301,8 +311,65 @@ describe("claude transform", () => { // runtime surface a new SDK member must join. expect([...INTERNAL_STREAM_MESSAGE_TYPES].sort()).toEqual([ "auth_status", + "conversation_reset", + "rate_limit_event", "tool_progress", "tool_use_summary", ]) }) + + it("turns an api_retry into one readable retry notification", () => { + const chunks = translateChunks({ + type: "system", + subtype: "api_retry", + attempt: 2, + max_retries: 5, + retry_delay_ms: 4200, + error_status: 429, + error: "rate_limit", + uuid: U1, + session_id: "sess-1", + }) + expect(chunks.map((chunk) => chunk.type)).toEqual(["start", "start-step", "retry-notification"]) + const retry = chunks[2] + expect(retry?.type === "retry-notification" && retry.message).toBe( + "Claude API retry: rate limit (HTTP 429), attempt 2 of 5, waiting 4s", + ) + }) + + it("leaves the status out of a retry that carries no HTTP code", () => { + const chunks = translateChunks({ + type: "system", + subtype: "api_retry", + attempt: 1, + max_retries: 3, + retry_delay_ms: 400, + error_status: null, + error: "overloaded", + uuid: U2, + session_id: "sess-1", + }) + const retry = chunks[2] + expect(retry?.type === "retry-notification" && retry.message).toBe( + "Claude API retry: overloaded, attempt 1 of 3, waiting 1s", + ) + }) + + it("maps a prompt suggestion to one bounded composer chunk", () => { + const chunks = translateChunks({ + type: "prompt_suggestion", + suggestion: ` ${"x".repeat(2500)} `, + uuid: U3, + session_id: "sess-9", + }) + const suggestion = chunks.find((chunk) => chunk.type === "prompt-suggestion") + expect(suggestion?.type === "prompt-suggestion" && suggestion.suggestion).toHaveLength(2000) + expect(suggestion?.type === "prompt-suggestion" && suggestion.sessionId).toBe("sess-9") + }) + + it("drops an empty prompt suggestion instead of showing an empty row", () => { + expect( + translate({ type: "prompt_suggestion", suggestion: " ", uuid: U4, session_id: "sess-9" }), + ).toEqual(["start", "start-step"]) + }) }) diff --git a/src/main/lib/claude/transform.ts b/src/main/lib/claude/transform.ts index 78dcac64..3618df48 100644 --- a/src/main/lib/claude/transform.ts +++ b/src/main/lib/claude/transform.ts @@ -16,23 +16,43 @@ import type { } from "./types" /** - * Message classification (roadmap step 09). Every member of - * `ClaudeStreamMessage` is either translated (`HANDLED`) or deliberately - * internal with a named reason (`INTERNAL`); the compile guards below fail - * typecheck when the SDK grows a member that is neither. + * Message classification (roadmap step 09, rechecked against the 0.3.270 pin by + * step 12). Every member of `ClaudeStreamMessage` is either translated + * (`HANDLED`) or deliberately internal with a named reason (`INTERNAL`); the + * compile guards below fail typecheck when the SDK grows a member that is + * neither, and the union in `types.ts` is built by exclusion from `SDKMessage`, + * so a new member arrives on the bump instead of being missed. * - * Internal members, by `msg.type`: + * The pinned SDK declares 39 message members across 11 top-level types, and 28 + * of the 39 are `system` subtypes. + * + * Internal top-level types: * - `tool_progress`: per-tool streaming progress; the transcript already shows * the tool round-trip, and no renderer surface consumes a second live feed. * - `auth_status`: auth state changes mid-turn; no consumer owns them yet. * - `tool_use_summary`: batch summaries of past tool use; the individual tool * parts are already in the transcript. + * - `rate_limit_event`: a subscription rate-limit gauge. The retry path below + * already says when a turn is waiting, and no surface renders the gauge; a + * usage panel is the consumer that would change this. + * - `conversation_reset`: the app persists its own transcript and never replays + * history on resume, so a reset notice has nothing to invalidate. * - * Internal `system` subtypes: `hook_started`, `hook_progress`, - * `hook_response`, `task_notification`, `task_started`, `files_persisted`. - * Each is harness bookkeeping with no chat-stream content and no named - * consumer; the full table lives in - * `.dump/app/research/2026-09-13-event-mapping.md`. + * Internal `system` subtypes are harness bookkeeping with no chat-stream + * content and no named consumer: the hook trio (`hook_started`, + * `hook_progress`, `hook_response`), the background-task family + * (`task_notification`, `task_started`, `task_updated`, `task_progress`, + * `background_tasks_changed`), session plumbing (`control_request_progress`, + * `session_state_changed`, `worker_shutting_down`, `commands_changed`, + * `files_persisted`, `memory_recall`, `plugin_install`), model fallbacks the + * result message also reports (`model_refusal_fallback`, + * `model_refusal_no_fallback`), and single-field notices (`local_command_output`, + * `thinking_tokens`, `notification`, `elicitation_complete`, + * `permission_denied`, `mirror_error`, `informational`). The permission denial + * is the one worth revisiting: the gate already surfaces a denial in chat, so + * this copy would be a second, differently-shaped report of the same event. + * The full table lives in + * `.dump/app/research/2026-09-13-sdk-0-3-bump.md`. */ export const HANDLED_STREAM_MESSAGE_TYPES = [ "stream_event", @@ -40,18 +60,22 @@ export const HANDLED_STREAM_MESSAGE_TYPES = [ "user", "system", "result", + "prompt_suggestion", ] as const satisfies readonly ClaudeStreamMessage["type"][] export const INTERNAL_STREAM_MESSAGE_TYPES = [ "tool_progress", "auth_status", "tool_use_summary", + "rate_limit_event", + "conversation_reset", ] as const satisfies readonly ClaudeStreamMessage["type"][] export const HANDLED_SYSTEM_SUBTYPES = [ "init", "status", "compact_boundary", + "api_retry", ] as const satisfies readonly Extract["subtype"][] export const INTERNAL_SYSTEM_SUBTYPES = [ @@ -60,7 +84,25 @@ export const INTERNAL_SYSTEM_SUBTYPES = [ "hook_response", "task_notification", "task_started", + "task_updated", + "task_progress", + "background_tasks_changed", "files_persisted", + "control_request_progress", + "session_state_changed", + "worker_shutting_down", + "commands_changed", + "memory_recall", + "plugin_install", + "model_refusal_fallback", + "model_refusal_no_fallback", + "local_command_output", + "thinking_tokens", + "notification", + "elicitation_complete", + "permission_denied", + "mirror_error", + "informational", ] as const satisfies readonly Extract["subtype"][] type AssertNever = [T] extends [never] ? true : never @@ -114,10 +156,27 @@ export function toClaudeStreamMessage(raw: unknown): ClaudeStreamMessage | null return raw as ClaudeStreamMessage } -/** A failed tool result's text: structured content renders its text parts. */ +/** A failed tool result's text: structured content renders its text parts and + * names the parts that carry no text, which at the pinned SDK is more than the + * image the old dialect could send. */ function toolResultErrorText(content: ClaudeToolResultBlock["content"]): string { if (typeof content === "string") return content - return content.map((part) => (part.type === "text" ? part.text : "[image]")).join("\n") + if (!content) return "" + return content.map((part) => (part.type === "text" ? part.text : `[${part.type}]`)).join("\n") +} + +/** + * One line for the retry toast: what failed, which attempt, and how long the + * CLI is going to wait. The error arrives as a slug, so it is spaced rather + * than shown to a person as `oauth_org_not_allowed`. + */ +function apiRetryMessage( + msg: Extract, +): string { + const reason = msg.error.replaceAll("_", " ") + const status = msg.error_status == null ? "" : ` (HTTP ${msg.error_status})` + const waitSeconds = Math.max(1, Math.round(msg.retry_delay_ms / 1000)) + return `Claude API retry: ${reason}${status}, attempt ${msg.attempt} of ${msg.max_retries}, waiting ${waitSeconds}s` } /** Resolve a tool result's output payload, preferring the CLI's own result. */ @@ -675,6 +734,13 @@ export function createTransformer(options?: { isUsingOllama?: boolean }) { } } + // A retry the CLI is already performing. Without it the stream looks + // stalled for the length of the backoff, and both chat transports toast + // this chunk. + if (msg.subtype === "api_retry") { + yield { type: "retry-notification", message: apiRetryMessage(msg) } + } + // Compact boundary - mark the compacting tool as complete if (msg.subtype === "compact_boundary") { let compactId = lastCompactId @@ -697,6 +763,19 @@ export function createTransformer(options?: { isUsingOllama?: boolean }) { } } + /** + * A suggested next prompt, asked for with `Options.promptSuggestions` and sent + * after the result message. Bounded here because the string is + * provider-authored and crosses IPC into the composer. + */ + function* handlePromptSuggestion( + msg: Extract, + ): Generator { + const suggestion = msg.suggestion.trim().slice(0, 2000) + if (!suggestion) return + yield { type: "prompt-suggestion", suggestion, sessionId: msg.session_id } + } + function* handleResultMessage(msg: SDKResultMessage): Generator { // ===== RESULT (final) ===== currentParentToolUseId = null @@ -770,6 +849,9 @@ export function createTransformer(options?: { isUsingOllama?: boolean }) { case "result": yield* handleResultMessage(msg) break + case "prompt_suggestion": + yield* handlePromptSuggestion(msg) + break default: break } diff --git a/src/main/lib/claude/types.ts b/src/main/lib/claude/types.ts index d76d46fd..6ed5057f 100644 --- a/src/main/lib/claude/types.ts +++ b/src/main/lib/claude/types.ts @@ -1,16 +1,8 @@ import type { - SDKAuthStatusMessage, - SDKCompactBoundaryMessage, - SDKFilesPersistedEvent, - SDKHookProgressMessage, - SDKHookResponseMessage, - SDKHookStartedMessage, - SDKResultMessage, - SDKStatusMessage, - SDKTaskNotificationMessage, - SDKTaskStartedMessage, - SDKToolProgressMessage, - SDKToolUseSummaryMessage, + SDKAssistantMessage, + SDKMessage, + SDKPartialAssistantMessage, + SDKUserMessage, } from "@anthropic-ai/claude-agent-sdk" // AI SDK UIMessageChunk format @@ -57,6 +49,7 @@ export type UIMessageChunk = | { type: "error"; errorText: string } | { type: "auth-error"; errorText: string } | { type: "retry-notification"; message: string } + | { type: "prompt-suggestion"; suggestion: string; sessionId: string } | { type: "ask-user-question"; toolUseId: string; questions: ToolApprovalQuestion[] } | { type: "ask-user-question-timeout"; toolUseId: string } | { type: "ask-user-question-result"; toolUseId: string; result: unknown } @@ -114,46 +107,97 @@ export type MessageMetadata = { /** * The stream messages and content blocks the translator consumes, declared - * locally (roadmap step 09). + * locally (roadmap step 09) and reconciled with the pinned SDK (step 12). * - * Provenance: the pinned `@anthropic-ai/claude-agent-sdk` 0.2.45 cannot type - * this boundary itself. Its `sdk.d.ts` builds `SDKMessage` from 18 members but - * never declares `SDKRateLimitEvent`, and the assistant/user/stream-event - * payload types import `BetaMessage`, `BetaRawMessageStreamEvent` and - * `MessageParam` from `@anthropic-ai/sdk`, which is not in the dependency - * tree. Under `skipLibCheck` the top-level union therefore collapses to an - * `any`-like type and those payload fields resolve to `any`. The local shapes - * below mirror what the CLI actually emits; 13 member types that ARE sound in - * the SDK are imported directly, and this file is the one place to revisit on - * an SDK pin bump (roadmap step 12 owns that step). + * Provenance: at 0.2.45 the SDK could not type this boundary itself. Its + * `sdk.d.ts` built `SDKMessage` from 18 members but never declared + * `SDKRateLimitEvent`, and the assistant, user and stream-event payload types + * imported `BetaMessage`, `BetaRawMessageStreamEvent` and `MessageParam` from + * `@anthropic-ai/sdk`, which was not in the dependency tree, so under + * `skipLibCheck` those fields resolved to `any` and the local shapes below were + * the only real contract. + * + * At 0.3.270 `@anthropic-ai/sdk` is a dependency, all 39 members are declared, + * and those payload types resolve. The local shapes stay, because they are what + * a fixture has to spell and what the translator is allowed to read, but they + * are no longer guesses: each one is derived from the SDK type it mirrors and + * kept a supertype of it, so a payload the SDK types still satisfies the read + * contract and a field the translator reads but the SDK stopped declaring is a + * compile error rather than an `undefined` in the transcript. + * `SDKSystemMessage` is still mirrored field by field, because its + * `mcp_servers` entry stops at `{name, status}` while the CLI also sends + * `serverInfo` and `error`, and the translator renders both. + */ + +/** + * The block names the translator reads. Every other block the pinned SDK can + * send stays in the unions below unread, which is what keeps a local shape + * narrow enough for a fixture and wide enough for the wire. */ +type ReadBlockType = "text" | "thinking" | "tool_use" | "tool_result" + +/** Anthropic's own block unions, reached through the SDK types that carry them + * so this file does not import a package the app does not depend on directly. */ +type SdkAssistantBlock = SDKAssistantMessage["message"]["content"][number] +type SdkUserBlock = Extract[number] +type SdkStreamEvent = SDKPartialAssistantMessage["event"] +type SdkStreamBlock = Extract["content_block"] +type SdkStreamDelta = Extract["delta"] /** Anthropic content blocks as the CLI streams or persists them. */ export type ClaudeTextBlock = { type: "text"; text: string } export type ClaudeThinkingBlock = { type: "thinking"; thinking: string; signature?: string } export type ClaudeToolUseBlock = { type: "tool_use"; id: string; name: string; input: unknown } -export type ClaudeToolResultContentBlock = { type: "text"; text: string } | { type: "image" } +/** + * What a tool result's content array can hold. The translator renders the text + * parts and names the rest, so the two read shapes stay explicit and the blocks + * it does not read (a document, a search result, a browser state) stay in the + * union through the SDK's own type. + */ +type SdkToolResultBlock = Extract +type SdkToolResultContentBlock = Extract< + NonNullable, + readonly unknown[] +>[number] +export type ClaudeToolResultContentBlock = + | { type: "text"; text: string } + | { type: "image" } + | Exclude export type ClaudeToolResultBlock = { type: "tool_result" tool_use_id: string - content: string | ClaudeToolResultContentBlock[] + /** Optional, because the SDK's own tool result block allows it to be absent. */ + content?: string | ClaudeToolResultContentBlock[] is_error?: boolean } -export type ClaudeContentBlock = ClaudeTextBlock | ClaudeThinkingBlock | ClaudeToolUseBlock -export type ClaudeUserContentBlock = ClaudeContentBlock | ClaudeToolResultBlock +export type ClaudeContentBlock = + | ClaudeTextBlock + | ClaudeThinkingBlock + | ClaudeToolUseBlock + | Exclude +export type ClaudeUserContentBlock = + | ClaudeContentBlock + | ClaudeToolResultBlock + | Exclude -/** Usage fields the translator reads for per-turn context metrics. */ +/** + * Usage fields the translator reads for per-turn context metrics. Nullable + * because the pinned SDK reports `null` for a cache tier that did not apply, + * and a null that lands in an arithmetic expression is a NaN in the context + * indicator rather than a missing number. + */ export type ClaudeUsage = { - input_tokens?: number - cache_read_input_tokens?: number - cache_creation_input_tokens?: number - output_tokens?: number + input_tokens?: number | null + cache_read_input_tokens?: number | null + cache_creation_input_tokens?: number | null + output_tokens?: number | null } /** * Raw Anthropic streaming events the translator reads, as forwarded inside * `stream_event` messages. Members the translator never reads (for example - * `message_delta`) are deliberately omitted. + * `message_delta`) stay in the union through the SDK's own event type, so the + * discriminated switch in the translator keeps exactly one member per name. */ export type ClaudeStreamApiEvent = | { type: "message_start" } @@ -164,8 +208,15 @@ export type ClaudeStreamApiEvent = | { type: "text_delta"; text: string } | { type: "input_json_delta"; partial_json: string } | { type: "thinking_delta"; thinking: string } + | Exclude } | { type: "content_block_stop" } + | Exclude< + SdkStreamEvent, + { + type: "message_start" | "content_block_start" | "content_block_delta" | "content_block_stop" + } + > export type ClaudeAssistantStreamMessage = { type: "assistant" @@ -207,8 +258,9 @@ export type ClaudeSystemInitMessage = { } /** - * The translator's message union: local shapes where the SDK resolves to - * `any`, direct SDK imports where the SDK types are sound. Widened with the + * The translator's message union: local read contracts for the four shapes the + * translator pulls fields out of, and the rest of the SDK's members by + * exclusion. Widened with the * nested-tool marker (`parent_tool_use_id`) on every member, because the * translator reads it before narrowing. Typing the translator against this * makes a CLI/SDK message-shape change a compile error at the read site @@ -219,16 +271,14 @@ export type ClaudeStreamMessage = ( | ClaudeUserStreamMessage | ClaudeStreamEventMessage | ClaudeSystemInitMessage - | SDKResultMessage - | SDKStatusMessage - | SDKCompactBoundaryMessage - | SDKHookStartedMessage - | SDKHookProgressMessage - | SDKHookResponseMessage - | SDKToolProgressMessage - | SDKAuthStatusMessage - | SDKTaskNotificationMessage - | SDKTaskStartedMessage - | SDKFilesPersistedEvent - | SDKToolUseSummaryMessage + // Every other member the pinned SDK declares, by exclusion rather than by + // hand: a release that adds one joins this union on the bump and fails the + // classification guards in `transform.ts` until someone decides what it means. + | Exclude< + SDKMessage, + | { type: "assistant" } + | { type: "user" } + | { type: "stream_event" } + | { type: "system"; subtype: "init" } + > ) & { readonly parent_tool_use_id?: string | null } From 20f92b919b1101cea2a47a1a6a553f7b974eb40d Mon Sep 17 00:00:00 2001 From: Owie6789 <151057755+Owie6789@users.noreply.github.com> Date: Wed, 23 Sep 2026 12:49:47 +0000 Subject: [PATCH 03/60] Register the renamed sub-agent and background task tools The pinned Claude CLI emits Agent where it used to emit Task, and TaskOutput and TaskStop where it used to emit BashOutput and KillShell. Grepping the 2.1.270 platform binary for its emitted tool table returns TaskCreate, TaskGet, TaskList, TaskUpdate, TaskStop, TaskOutput, Agent and TodoWrite, with a normalization table mapping KillShell and KillBash to TaskStop and BashOutput, BashOutputTool, AgentOutput and AgentOutputTool to TaskOutput. The registry knew only the old names, so a current session rendered those steps as unstyled generic tool calls, and assistant-message-item suppressed TaskOutput rows the CLI now emits under a name it did not suppress, which is why background output had no home in the transcript. One module names the sub-agent tool types so grouping, dispatch and the nested tool lookup read one list instead of three string comparisons. tool-Task and tool-Agent share one meta object, and TaskOutput shares its meta with BashOutput, so the rename cannot drift into two entries with different wording. TaskStop joins KillShell the same way, which is one entry beyond the two the roadmap names and comes from the same normalization table. A single subtitle reader accepts pid, task_id and the persisted taskId spelling. MultiEdit is deliberately not registered. It appears in that binary only in permission, deny-rule, display-label and legacy-alias tables and never in the emitted tool set, so a row for it would be unreachable; the SDK's tool types agree, declaring AgentInput and AgentOutput with no TaskInput. The evidence is in .dump/app/research/2026-09-13-sdk-0-3-bump.md and on the issue. The dead variant field and its ToolVariant type go with this: nothing in the repo read them, and the structural registry type in isolated-message-group only asks for icon and title, so giving TaskOutput a collapsible value would have described behaviour this codebase does not have. Co-authored-by: arena-agent <297053741+arena-agent@users.noreply.github.com> --- .../agents/lib/subagent-tool-types.ts | 26 ++++ .../agents/main/assistant-message-item.tsx | 7 +- .../agents/ui/agent-tool-registry.test.ts | 111 +++++++++++++++ .../agents/ui/agent-tool-registry.tsx | 134 ++++++++++-------- 4 files changed, 212 insertions(+), 66 deletions(-) create mode 100644 src/renderer/features/agents/lib/subagent-tool-types.ts create mode 100644 src/renderer/features/agents/ui/agent-tool-registry.test.ts diff --git a/src/renderer/features/agents/lib/subagent-tool-types.ts b/src/renderer/features/agents/lib/subagent-tool-types.ts new file mode 100644 index 00000000..b7895d45 --- /dev/null +++ b/src/renderer/features/agents/lib/subagent-tool-types.ts @@ -0,0 +1,26 @@ +/** + * The tool types that mean "a sub-agent is running", under every name the + * pinned Claude CLI has used for it. + * + * The 2.1.270 binary emits `Agent`. Its own normalization table maps the older + * spellings to the current ones, and the SDK changelog at 0.2.69 records why + * both exist: the wire name was reverted to `Task` with the note that it "will + * migrate to `Agent` in the next minor release", which the 0.3 line did. So a + * transcript persisted before the bump carries `Task` and one recorded after it + * carries `Agent`, and grouping, dispatch and suppression all have to treat the + * two as one tool or a resumed session renders differently from a new one. + * + * The background-task family is NOT in this list. `TaskCreate`, `TaskUpdate`, + * `TaskGet`, `TaskList` and `TaskStop` share the `Task` prefix and nothing else: + * they manage background shells and tasks, not sub-agents, and + * `assistant-message-item.tsx` keeps its own `TASK_TOOLS` set for them. + */ +export const SUBAGENT_TOOL_TYPES = ["tool-Task", "tool-Agent"] as const + +export type SubagentToolType = (typeof SUBAGENT_TOOL_TYPES)[number] + +const SUBAGENT_TOOL_TYPE_SET: ReadonlySet = new Set(SUBAGENT_TOOL_TYPES) + +export function isSubagentToolType(type: string): type is SubagentToolType { + return SUBAGENT_TOOL_TYPE_SET.has(type) +} diff --git a/src/renderer/features/agents/main/assistant-message-item.tsx b/src/renderer/features/agents/main/assistant-message-item.tsx index 55ba0486..d33c4dfd 100644 --- a/src/renderer/features/agents/main/assistant-message-item.tsx +++ b/src/renderer/features/agents/main/assistant-message-item.tsx @@ -19,6 +19,7 @@ import { cn } from "../../../lib/utils" import { selectedProjectAtom, showMessageJsonAtom } from "../atoms" import { isAssistantMessageQuestion } from "../lib/is-question" import { playQuestionSound } from "../lib/play-question-sound" +import { isSubagentToolType } from "../lib/subagent-tool-types" import { useFileOpen } from "../mentions" import type { Message } from "../stores/message-store" import { @@ -638,7 +639,7 @@ export const AssistantMessageItem = memo(function AssistantMessageItem({ messageParts .filter( (p): p is NormalizedPart & { toolCallId: string } => - p.type === "tool-Task" && !!p.toolCallId, + isSubagentToolType(p.type) && !!p.toolCallId, ) .map((p) => p.toolCallId), ) @@ -746,7 +747,6 @@ export const AssistantMessageItem = memo(function AssistantMessageItem({ shouldCollapse && collapseBeforeIndex !== -1 ? messageParts.slice(0, collapseBeforeIndex) : [] const visibleStepsCount = stepParts.filter((p) => { if (p.type === "step-start") return false - if (p.type === "tool-TaskOutput") return false if (p.type === "tool-ExitPlanMode") return false if (p.toolCallId && nestedToolIds.has(p.toolCallId)) return false if ( @@ -803,7 +803,6 @@ export const AssistantMessageItem = memo(function AssistantMessageItem({ (part: NormalizedPart, idx: number, isFinal = false) => { const toolInput = part.input as { file_path?: string } | null | undefined if (part.type === "step-start") return null - if (part.type === "tool-TaskOutput") return null if (part.toolCallId && orphanToolCallIds.has(part.toolCallId)) { if (!orphanFirstToolCallIds.has(part.toolCallId)) return null @@ -845,7 +844,7 @@ export const AssistantMessageItem = memo(function AssistantMessageItem({ ) } - if (part.type === "tool-Task") { + if (isSubagentToolType(part.type)) { const nestedTools = nestedToolsMap.get(part.toolCallId ?? "") || [] return } diff --git a/src/renderer/features/agents/ui/agent-tool-registry.test.ts b/src/renderer/features/agents/ui/agent-tool-registry.test.ts new file mode 100644 index 00000000..2683e3a2 --- /dev/null +++ b/src/renderer/features/agents/ui/agent-tool-registry.test.ts @@ -0,0 +1,111 @@ +/** + * The registry entries the 0.3.270 pin renamed, and the sharing that keeps the + * two spellings of one tool from drifting apart. The names come from the pinned + * CLI itself: grepping the 2.1.270 platform binary for its emitted tool table + * returns `Agent`, `TaskOutput` and `TaskStop` with no `Task`, no `BashOutput` + * and no `KillShell`, and its normalization table maps `KillShell` and + * `KillBash` to `TaskStop` and `BashOutput`, `BashOutputTool`, `AgentOutput`, + * `AgentOutputTool` to `TaskOutput`. The legacy keys stay registered because + * transcripts persisted before the bump carry them. + */ +import { describe, expect, it } from "vitest" +import { isSubagentToolType, SUBAGENT_TOOL_TYPES } from "../lib/subagent-tool-types" +import { AgentToolRegistry, type ToolDisplayPart } from "./agent-tool-registry" + +const pendingSubagent: ToolDisplayPart = { + state: "input-available", + input: { subagent_type: "Explore", description: "Audit the permission gate" }, +} + +const streamingSubagent: ToolDisplayPart = { + state: "input-streaming", + input: { subagent_type: "Explore", description: "Audit the permission gate" }, +} + +const finishedSubagent: ToolDisplayPart = { + state: "output-available", + input: { + subagent_type: "Explore", + description: "Read the transform and list every chunk kind it emits", + }, + output: { task: { subject: "Run the gate" } }, +} + +const taskOutputById: ToolDisplayPart = { + state: "output-available", + input: { task_id: "bg_7" }, + output: { task: { subject: "Run the gate" } }, +} + +const shellOutputByPid: ToolDisplayPart = { + state: "input-available", + input: { pid: 4242 }, +} + +describe("agent tool registry: renamed sub-agent and background task tools", () => { + it("registers the sub-agent under both names as one meta", () => { + expect(AgentToolRegistry["tool-Agent"]).toBeDefined() + // Same object, not an equal copy: two entries for one tool drift apart the + // first time someone edits the wording of one of them. + expect(AgentToolRegistry["tool-Agent"]).toBe(AgentToolRegistry["tool-Task"]) + }) + + it("registers TaskOutput as the meta BashOutput already had", () => { + expect(AgentToolRegistry["tool-TaskOutput"]).toBe(AgentToolRegistry["tool-BashOutput"]) + expect(AgentToolRegistry["tool-TaskStop"]).toBeDefined() + expect(AgentToolRegistry["tool-KillShell"]).toBeDefined() + }) + + it("titles a sub-agent by state", () => { + const meta = AgentToolRegistry["tool-Agent"] + expect(meta.title(streamingSubagent)).toBe("Preparing agent") + expect(meta.title(pendingSubagent)).toBe("Running Explore") + expect(meta.title(finishedSubagent)).toBe("Explore completed") + }) + + it("names the sub-agent even when the payload carries no type", () => { + const meta = AgentToolRegistry["tool-Agent"] + expect(meta.title({ state: "input-available", input: {} })).toBe("Running Agent") + expect(meta.title({ state: "output-available", input: {} })).toBe("Agent completed") + }) + + it("truncates a long sub-agent description and hides it while streaming", () => { + const meta = AgentToolRegistry["tool-Agent"] + expect(meta.subtitle?.(streamingSubagent)).toBe("") + expect(meta.subtitle?.(pendingSubagent)).toBe("Audit the permission gate") + // The registry's rule is 47 characters plus an ellipsis, so a 53 character + // description loses its last word rather than growing the row. + expect(meta.subtitle?.(finishedSubagent)).toBe( + "Read the transform and list every chunk kind it...", + ) + }) + + it("reads a task id or a shell pid into one subtitle", () => { + const meta = AgentToolRegistry["tool-TaskOutput"] + expect(meta.title(shellOutputByPid)).toBe("Getting output") + expect(meta.subtitle?.(shellOutputByPid)).toBe("PID: 4242") + expect(meta.title(taskOutputById)).toBe("Got output") + expect(meta.subtitle?.(taskOutputById)).toBe("Task: bg_7") + expect(meta.subtitle?.({ state: "output-available", input: {} })).toBe("") + }) + + it("says task for TaskStop and shell for the name it replaced", () => { + expect(AgentToolRegistry["tool-TaskStop"].title(shellOutputByPid)).toBe("Stopping task") + expect(AgentToolRegistry["tool-TaskStop"].title(taskOutputById)).toBe("Stopped task") + expect(AgentToolRegistry["tool-KillShell"].title(shellOutputByPid)).toBe("Stopping shell") + expect(AgentToolRegistry["tool-KillShell"].title(taskOutputById)).toBe("Stopped shell") + // Both read the same subtitle, so a task id renders under either name. + expect(AgentToolRegistry["tool-TaskStop"].subtitle?.(taskOutputById)).toBe("Task: bg_7") + }) + + it("counts the sub-agent family and leaves the background task family out", () => { + expect([...SUBAGENT_TOOL_TYPES]).toEqual(["tool-Task", "tool-Agent"]) + expect(isSubagentToolType("tool-Task")).toBe(true) + expect(isSubagentToolType("tool-Agent")).toBe(true) + // Same prefix, different tool: these manage background work, not sub-agents. + expect(isSubagentToolType("tool-TaskCreate")).toBe(false) + expect(isSubagentToolType("tool-TaskStop")).toBe(false) + expect(isSubagentToolType("tool-TaskOutput")).toBe(false) + expect(isSubagentToolType("tool-Bash")).toBe(false) + }) +}) diff --git a/src/renderer/features/agents/ui/agent-tool-registry.tsx b/src/renderer/features/agents/ui/agent-tool-registry.tsx index e649c9c7..4057fa80 100644 --- a/src/renderer/features/agents/ui/agent-tool-registry.tsx +++ b/src/renderer/features/agents/ui/agent-tool-registry.tsx @@ -33,8 +33,6 @@ import { getToolLifecycleState } from "./agent-tool-state" export { getToolStatus } from "./agent-tool-state" -export type ToolVariant = "simple" | "collapsible" - /** Tool input/output fields read by the registry display callbacks. */ export type ToolDisplayPart = { state?: string @@ -56,6 +54,7 @@ export type ToolDisplayPart = { subject?: string status?: string taskId?: string | number + task_id?: string | number pid?: string | number text?: string plan?: { status?: string; title?: string; steps?: { status?: string }[] } @@ -74,7 +73,6 @@ export interface ToolMeta { title: (part: ToolDisplayPart) => string subtitle?: (part: ToolDisplayPart) => string tooltipContent?: (part: ToolDisplayPart, projectPath?: string) => string - variant: ToolVariant } function isInputStreaming(part: { state?: unknown; output?: unknown; result?: unknown }) { @@ -152,23 +150,73 @@ function calculateDiffStats(oldString: string, newString: string) { return { addedLines, removedLines } } -export const AgentToolRegistry: Record = { - "tool-Task": { - icon: SparklesIcon, - title: (part) => { - if (isInputStreaming(part)) return "Preparing agent" - const subagentType = part.input?.subagent_type || "Agent" - return isPendingState(part) ? `Running ${subagentType}` : `${subagentType} completed` - }, - subtitle: (part) => { - // Don't show subtitle while input is still streaming - if (isInputStreaming(part)) return "" - const description = part.input?.description || "" - return description.length > 50 ? `${description.slice(0, 47)}...` : description - }, - variant: "simple", +/** + * The sub-agent tool under both names it has had. The pinned CLI emits `Agent`, + * and its own history explains why two names exist: the SDK changelog at 0.2.69 + * reverted the wire name with the note that it "will migrate to `Agent` in the + * next minor release", so transcripts this app already persisted carry `Task`. + * One object behind two keys, because two entries for one tool drift apart the + * first time someone edits the wording of one of them. + */ +const subagentTool: ToolMeta = { + icon: SparklesIcon, + title: (part) => { + if (isInputStreaming(part)) return "Preparing agent" + const subagentType = part.input?.subagent_type || "Agent" + return isPendingState(part) ? `Running ${subagentType}` : `${subagentType} completed` + }, + subtitle: (part) => { + // Don't show subtitle while input is still streaming + if (isInputStreaming(part)) return "" + const description = part.input?.description || "" + return description.length > 50 ? `${description.slice(0, 47)}...` : description }, +} + +/** + * A shell reports `pid`, a background task reports `task_id`, and the persisted + * app shape spells the same field `taskId`. One reader for all three so the + * renamed tools do not each grow their own subtitle rule. + */ +function backgroundTaskSubtitle(part: ToolDisplayPart): string { + const pid = part.input?.pid + if (pid) return `PID: ${pid}` + const taskId = part.input?.task_id ?? part.input?.taskId + return taskId ? `Task: ${taskId}` : "" +} +/** + * Background output under the name the CLI emits now and the name it emitted + * before: the 2.1.270 binary normalizes `BashOutput`, `BashOutputTool`, + * `AgentOutput` and `AgentOutputTool` to `TaskOutput`, so one meta serves the + * current wire name and the legacy rows already in the transcript store. + */ +const backgroundOutputTool: ToolMeta = { + icon: Terminal, + title: (part) => (isPendingState(part) ? "Getting output" : "Got output"), + subtitle: backgroundTaskSubtitle, +} + +const stopShellTool: ToolMeta = { + icon: XCircle, + title: (part) => (isPendingState(part) ? "Stopping shell" : "Stopped shell"), + subtitle: backgroundTaskSubtitle, +} + +/** + * `TaskStop` is what the same binary normalizes `KillShell` and `KillBash` to, + * so its wording says task rather than shell: the thing being stopped at that + * name is a background task, which may be a sub-agent rather than a shell. + */ +const stopTaskTool: ToolMeta = { + icon: XCircle, + title: (part) => (isPendingState(part) ? "Stopping task" : "Stopped task"), + subtitle: backgroundTaskSubtitle, +} + +export const AgentToolRegistry: Record = { + "tool-Task": subagentTool, + "tool-Agent": subagentTool, "tool-Grep": { icon: SearchIcon, title: (part) => { @@ -204,7 +252,6 @@ export const AgentToolRegistry: Record = { return pattern.length > 40 ? `${pattern.slice(0, 37)}...` : pattern }, - variant: "simple", }, "tool-Glob": { @@ -231,7 +278,6 @@ export const AgentToolRegistry: Record = { return pattern.length > 40 ? `${pattern.slice(0, 37)}...` : pattern }, - variant: "simple", }, "tool-Read": { @@ -252,7 +298,6 @@ export const AgentToolRegistry: Record = { const filePath = part.input?.file_path || "" return getDisplayPath(filePath, projectPath) }, - variant: "simple", }, "tool-Edit": { @@ -283,14 +328,12 @@ export const AgentToolRegistry: Record = { return "" }, - variant: "simple", }, // Cloning indicator - shown while sandbox is being created "tool-cloning": { icon: GitBranch, title: () => "Cloning repo", - variant: "simple", }, // Planning indicator - shown when streaming starts but no content yet @@ -312,7 +355,6 @@ export const AgentToolRegistry: Record = { ] return messages[Math.floor(Math.random() * messages.length)] }, - variant: "simple", }, "tool-Write": { @@ -328,7 +370,6 @@ export const AgentToolRegistry: Record = { if (!filePath) return "" // Don't show "file" placeholder during streaming return filePath.split("/").pop() || "" }, - variant: "simple", }, "tool-Bash": { @@ -350,7 +391,6 @@ export const AgentToolRegistry: Record = { }) return normalized.length > 50 ? `${normalized.slice(0, 47)}...` : normalized }, - variant: "simple", }, "tool-WebFetch": { @@ -369,7 +409,6 @@ export const AgentToolRegistry: Record = { return url.slice(0, 30) } }, - variant: "simple", }, "tool-WebSearch": { @@ -384,7 +423,6 @@ export const AgentToolRegistry: Record = { const query = part.input?.query || "" return query.length > 40 ? `${query.slice(0, 37)}...` : query }, - variant: "collapsible", }, // Planning tools @@ -402,7 +440,6 @@ export const AgentToolRegistry: Record = { if (todos.length === 0) return "" return `${todos.length} ${todos.length === 1 ? "item" : "items"}` }, - variant: "simple", }, // Task management tools @@ -415,7 +452,6 @@ export const AgentToolRegistry: Record = { const subject = part.input?.subject || "" return subject.length > 40 ? `${subject.slice(0, 37)}...` : subject }, - variant: "simple", }, "tool-TaskUpdate": { @@ -442,7 +478,6 @@ export const AgentToolRegistry: Record = { } return taskId ? `#${taskId}` : "" }, - variant: "simple", }, "tool-TaskGet": { @@ -458,7 +493,6 @@ export const AgentToolRegistry: Record = { } return taskId ? `#${taskId}` : "" }, - variant: "simple", }, "tool-TaskList": { @@ -469,7 +503,6 @@ export const AgentToolRegistry: Record = { return count !== undefined ? `Listed ${count} tasks` : "Listed tasks" }, subtitle: () => "", - variant: "simple", }, "tool-PlanWrite": { @@ -498,7 +531,6 @@ export const AgentToolRegistry: Record = { } return steps.length > 0 ? `${completed}/${steps.length} steps` : "" }, - variant: "simple", }, "tool-ExitPlanMode": { @@ -507,7 +539,6 @@ export const AgentToolRegistry: Record = { return isPendingState(part) ? "Finishing plan" : "Plan complete" }, subtitle: () => "", - variant: "simple", }, // Notebook tools @@ -521,33 +552,14 @@ export const AgentToolRegistry: Record = { if (!filePath) return "" return filePath.split("/").pop() || "" }, - variant: "simple", }, - // Shell management tools - "tool-BashOutput": { - icon: Terminal, - title: (part) => { - return isPendingState(part) ? "Getting output" : "Got output" - }, - subtitle: (part) => { - const pid = part.input?.pid - return pid ? `PID: ${pid}` : "" - }, - variant: "simple", - }, - - "tool-KillShell": { - icon: XCircle, - title: (part) => { - return isPendingState(part) ? "Stopping shell" : "Stopped shell" - }, - subtitle: (part) => { - const pid = part.input?.pid - return pid ? `PID: ${pid}` : "" - }, - variant: "simple", - }, + // Shell and background task management. The first name in each pair is the one + // the pinned CLI emits, the second the one older transcripts carry. + "tool-TaskOutput": backgroundOutputTool, + "tool-BashOutput": backgroundOutputTool, + "tool-TaskStop": stopTaskTool, + "tool-KillShell": stopShellTool, // Note: ListMcpResources, ReadMcpResource and their "Tool"-suffixed variants // are handled by AgentMcpToolCall via parseMcpToolType() for richer output display @@ -558,7 +570,6 @@ export const AgentToolRegistry: Record = { title: (part) => { return isPendingState(part) ? "Compacting..." : "Compacted" }, - variant: "simple", }, // Extended Thinking @@ -572,7 +583,6 @@ export const AgentToolRegistry: Record = { // Show first 50 chars as preview return text.length > 50 ? `${text.slice(0, 47)}...` : text }, - variant: "collapsible", }, } From 76a14122bd543e3c42052e5b1b6dcd3053ae4bee Mon Sep 17 00:00:00 2001 From: Owie6789 <151057755+Owie6789@users.noreply.github.com> Date: Wed, 23 Sep 2026 12:54:27 +0000 Subject: [PATCH 04/60] Make the effort, adaptive thinking and prompt suggestion surfaces reachable The pinned SDK adds three options this app could not send. `Options.effort` is low, medium, high, xhigh or max. `Options.thinking` is adaptive, enabled with a budget, or disabled, and supersedes the deprecated `maxThinkingTokens`, which on a model that supports adaptive thinking only ever meant "adaptive" when non-zero and meant the model chose a budget anyway when absent, so the Extended thinking switch could not turn thinking off. `Options.promptSuggestions` asks for one suggested next prompt after the result, which transform.ts already classified but nothing could request and nothing rendered. One vocabulary in src/shared/effort.ts now backs both pickers. Codex's four levels carry a satisfies proof against it and its own label function is gone in favour of formatEffortLabel, so the two backends cannot drift to two names for one concept. The sub-menu Codex already had is generalized to EffortSubMenu and gains an optional row that clears the pick: Claude needs a Default row because a chat that never opened the picker must keep the model's own default rather than a level this app guessed, while Codex always sends a concrete level and passes no clearer, so its rows and its behaviour are unchanged. Whether the row appears comes from the backend's capability profile rather than a provider name in the renderer. features.effort, features.adaptiveThinking and features.promptSuggestions are declared for all ten backends and true only where a turn can actually carry the value end to end, which today is effort on Claude and Codex and both other flags on Claude. The transport sends thinking as adaptive when the switch is on and disabled when it is off, which is the first time off has meant off. A prompt suggestion is asked for only when the new preference is on, which is off by default, and the chunk it produces goes to a per-sub-chat atom instead of the message stream, so split panes do not share a suggestion and the AI SDK never sees a chunk type it does not know. The composer renders it as one row: clicking fills the draft and focuses the editor, Dismiss clears it. The atom is not persisted, because a suggestion belongs to the turn that produced it. Co-authored-by: arena-agent <297053741+arena-agent@users.noreply.github.com> --- src/main/lib/providers/claude.ts | 6 ++ src/main/lib/providers/cline.ts | 6 ++ src/main/lib/providers/codex.ts | 6 ++ src/main/lib/providers/cursor.ts | 6 ++ src/main/lib/providers/grok.ts | 6 ++ src/main/lib/providers/hermes.ts | 6 ++ src/main/lib/providers/openclaw.ts | 6 ++ src/main/lib/providers/opencode.ts | 6 ++ src/main/lib/providers/qwen.ts | 6 ++ src/main/lib/providers/roo.ts | 6 ++ src/main/lib/trpc/routers/claude.ts | 28 ++++++++-- .../settings-tabs/agents-preferences-tab.tsx | 17 ++++++ src/renderer/features/agents/atoms/index.ts | 13 +++++ .../components/agent-model-selector.tsx | 55 ++++++++++++++++--- .../features/agents/lib/ipc-chat-transport.ts | 31 +++++++++-- src/renderer/features/agents/lib/models.ts | 1 - .../features/agents/main/chat-input-area.tsx | 43 ++++++++++++++- .../features/agents/main/new-chat-form.tsx | 11 ++++ src/renderer/lib/atoms/index.ts | 22 ++++++++ src/shared/codex-model-id.ts | 25 ++++++--- src/shared/effort.test.ts | 34 ++++++++++++ src/shared/effort.ts | 35 ++++++++++++ src/shared/provider-capabilities.ts | 6 ++ 23 files changed, 354 insertions(+), 27 deletions(-) create mode 100644 src/shared/effort.test.ts create mode 100644 src/shared/effort.ts diff --git a/src/main/lib/providers/claude.ts b/src/main/lib/providers/claude.ts index f9effaaa..aeaf9fa7 100644 --- a/src/main/lib/providers/claude.ts +++ b/src/main/lib/providers/claude.ts @@ -101,6 +101,12 @@ export function getClaudeCapability(): ProviderCapability { skills: true, structuredOutput: false, fileCheckpointing: false, + // True only where a turn can actually carry the value end to end; the + // evidence per backend is in + // `.dump/app/research/2026-09-13-sdk-0-3-bump.md`. + effort: true, + adaptiveThinking: true, + promptSuggestions: true, }, notes: [ "Per-action approvals via canUseTool + in-chat approval prompts.", diff --git a/src/main/lib/providers/cline.ts b/src/main/lib/providers/cline.ts index 1cf10495..4d86f468 100644 --- a/src/main/lib/providers/cline.ts +++ b/src/main/lib/providers/cline.ts @@ -70,6 +70,12 @@ export function getClineCapability(): ProviderCapability { skills: true, structuredOutput: false, fileCheckpointing: false, + // True only where a turn can actually carry the value end to end; the + // evidence per backend is in + // `.dump/app/research/2026-09-13-sdk-0-3-bump.md`. + effort: false, + adaptiveThinking: false, + promptSuggestions: false, }, notes: [ "Images travel as prompt path references the agent reads via tools.", diff --git a/src/main/lib/providers/codex.ts b/src/main/lib/providers/codex.ts index 94cac7b2..fbe77d08 100644 --- a/src/main/lib/providers/codex.ts +++ b/src/main/lib/providers/codex.ts @@ -72,6 +72,12 @@ export function getCodexCapability(): ProviderCapability { skills: false, structuredOutput: false, fileCheckpointing: false, + // True only where a turn can actually carry the value end to end; the + // evidence per backend is in + // `.dump/app/research/2026-09-13-sdk-0-3-bump.md`. + effort: true, + adaptiveThinking: false, + promptSuggestions: false, }, notes: [ "Approvals auto-grant session-wide (parity with the former ACP path).", diff --git a/src/main/lib/providers/cursor.ts b/src/main/lib/providers/cursor.ts index 4899b872..78fd1d47 100644 --- a/src/main/lib/providers/cursor.ts +++ b/src/main/lib/providers/cursor.ts @@ -62,6 +62,12 @@ export function getCursorCapability(): ProviderCapability { skills: false, structuredOutput: false, fileCheckpointing: false, + // True only where a turn can actually carry the value end to end; the + // evidence per backend is in + // `.dump/app/research/2026-09-13-sdk-0-3-bump.md`. + effort: false, + adaptiveThinking: false, + promptSuggestions: false, }, notes: [ "Images travel as prompt path references the agent reads via tools.", diff --git a/src/main/lib/providers/grok.ts b/src/main/lib/providers/grok.ts index 0d1f17df..3e9ea8a2 100644 --- a/src/main/lib/providers/grok.ts +++ b/src/main/lib/providers/grok.ts @@ -71,6 +71,12 @@ export function getGrokCapability(): ProviderCapability { skills: false, structuredOutput: false, fileCheckpointing: false, + // True only where a turn can actually carry the value end to end; the + // evidence per backend is in + // `.dump/app/research/2026-09-13-sdk-0-3-bump.md`. + effort: false, + adaptiveThinking: false, + promptSuggestions: false, }, notes: [ "Images travel as prompt path references the agent reads via tools.", diff --git a/src/main/lib/providers/hermes.ts b/src/main/lib/providers/hermes.ts index d1e44efa..cbd388b7 100644 --- a/src/main/lib/providers/hermes.ts +++ b/src/main/lib/providers/hermes.ts @@ -58,6 +58,12 @@ export function getHermesCapability(): ProviderCapability { skills: true, structuredOutput: false, fileCheckpointing: false, + // True only where a turn can actually carry the value end to end; the + // evidence per backend is in + // `.dump/app/research/2026-09-13-sdk-0-3-bump.md`. + effort: false, + adaptiveThinking: false, + promptSuggestions: false, }, notes: [ "ACP sessions live in the running server process; resume across restarts is best-effort.", diff --git a/src/main/lib/providers/openclaw.ts b/src/main/lib/providers/openclaw.ts index f703582b..22893a83 100644 --- a/src/main/lib/providers/openclaw.ts +++ b/src/main/lib/providers/openclaw.ts @@ -72,6 +72,12 @@ export function getOpenclawCapability(): ProviderCapability { skills: false, structuredOutput: false, fileCheckpointing: false, + // True only where a turn can actually carry the value end to end; the + // evidence per backend is in + // `.dump/app/research/2026-09-13-sdk-0-3-bump.md`. + effort: false, + adaptiveThinking: false, + promptSuggestions: false, }, notes: [ "One JSON envelope per turn — no streaming; progress appears only when the turn settles.", diff --git a/src/main/lib/providers/opencode.ts b/src/main/lib/providers/opencode.ts index ab5732e5..83cbdd12 100644 --- a/src/main/lib/providers/opencode.ts +++ b/src/main/lib/providers/opencode.ts @@ -58,6 +58,12 @@ export function getOpencodeCapability(): ProviderCapability { skills: false, structuredOutput: false, fileCheckpointing: false, + // True only where a turn can actually carry the value end to end; the + // evidence per backend is in + // `.dump/app/research/2026-09-13-sdk-0-3-bump.md`. + effort: false, + adaptiveThinking: false, + promptSuggestions: false, }, notes: [ "Permissions auto-reply session-wide; opencode.json can tighten per-tool policy.", diff --git a/src/main/lib/providers/qwen.ts b/src/main/lib/providers/qwen.ts index b2ba6905..0daa3e6c 100644 --- a/src/main/lib/providers/qwen.ts +++ b/src/main/lib/providers/qwen.ts @@ -69,6 +69,12 @@ export function getQwenCapability(): ProviderCapability { skills: true, structuredOutput: false, fileCheckpointing: false, + // True only where a turn can actually carry the value end to end; the + // evidence per backend is in + // `.dump/app/research/2026-09-13-sdk-0-3-bump.md`. + effort: false, + adaptiveThinking: false, + promptSuggestions: false, }, notes: [ "Images travel as prompt path references the agent reads via tools.", diff --git a/src/main/lib/providers/roo.ts b/src/main/lib/providers/roo.ts index 84e521f7..2a5ccf68 100644 --- a/src/main/lib/providers/roo.ts +++ b/src/main/lib/providers/roo.ts @@ -69,6 +69,12 @@ export function getRooCapability(): ProviderCapability { skills: false, structuredOutput: false, fileCheckpointing: false, + // True only where a turn can actually carry the value end to end; the + // evidence per backend is in + // `.dump/app/research/2026-09-13-sdk-0-3-bump.md`. + effort: false, + adaptiveThinking: false, + promptSuggestions: false, }, notes: [ "Streams NDJSON events per turn (text deltas, thinking, tool calls, command output, cost).", diff --git a/src/main/lib/trpc/routers/claude.ts b/src/main/lib/trpc/routers/claude.ts index 5e5ae92b..7db8156f 100644 --- a/src/main/lib/trpc/routers/claude.ts +++ b/src/main/lib/trpc/routers/claude.ts @@ -14,6 +14,7 @@ import { and, eq } from "drizzle-orm" import { app, BrowserWindow } from "electron" import { z } from "zod" import { agentModeSchema, DEFAULT_AGENT_MODE } from "../../../../shared/agent-mode" +import { EFFORT_LEVELS } from "../../../../shared/effort" import { describePermissionDecision } from "../../../../shared/permissions/decision" import { setConnectionMethod } from "../../analytics" import { @@ -851,7 +852,26 @@ export const claudeRouter = router({ baseUrl: z.string().min(1), }) .optional(), - maxThinkingTokens: z.number().optional(), // Enable extended thinking + // The pinned SDK deprecates `maxThinkingTokens`: on a model that + // supports adaptive thinking a non-zero budget only ever meant + // "adaptive", and leaving it out meant the model chose a budget anyway, + // so the old field could not turn thinking off. Take the config the SDK + // asks for instead. + thinking: z + .discriminatedUnion("type", [ + z.object({ type: z.literal("adaptive") }), + z.object({ type: z.literal("disabled") }), + z.object({ + type: z.literal("enabled"), + budgetTokens: z.number().int().positive().optional(), + }), + ]) + .optional(), + // One vocabulary for every backend that accepts an effort; the CLI + // clamps a level above what the chosen model reports. + effort: z.enum(EFFORT_LEVELS).optional(), + // Asks the SDK for one suggested next prompt after the result message. + promptSuggestions: z.boolean().optional(), images: z.array(imageAttachmentSchema).optional(), // Image attachments historyEnabled: z.boolean().optional(), offlineModeEnabled: z.boolean().optional(), // Whether offline mode (Ollama) is enabled in settings @@ -1947,9 +1967,9 @@ ${prompt} ...(!resumeSessionId && { continue: true }), ...(resolvedModel && { model: resolvedModel }), // fallbackModel: "claude-opus-4-5-20251101", - ...(input.maxThinkingTokens && { - maxThinkingTokens: input.maxThinkingTokens, - }), + ...(input.thinking && { thinking: input.thinking }), + ...(input.effort && { effort: input.effort }), + ...(input.promptSuggestions && { promptSuggestions: true }), }, } diff --git a/src/renderer/components/dialogs/settings-tabs/agents-preferences-tab.tsx b/src/renderer/components/dialogs/settings-tabs/agents-preferences-tab.tsx index c743d256..94c2055b 100644 --- a/src/renderer/components/dialogs/settings-tabs/agents-preferences-tab.tsx +++ b/src/renderer/components/dialogs/settings-tabs/agents-preferences-tab.tsx @@ -38,6 +38,7 @@ import { localOnlyModeAtom, notifyWhenFocusedAtom, preferredEditorAtom, + promptSuggestionsEnabledAtom, soundNotificationsEnabledAtom, } from "../../../lib/atoms" @@ -140,6 +141,9 @@ function useIsNarrowScreen(): boolean { export function AgentsPreferencesTab() { const [thinkingEnabled, setThinkingEnabled] = useAtom(extendedThinkingEnabledAtom) + const [promptSuggestionsEnabled, setPromptSuggestionsEnabled] = useAtom( + promptSuggestionsEnabledAtom, + ) const [soundEnabled, setSoundEnabled] = useAtom(soundNotificationsEnabledAtom) const [desktopNotificationsEnabled, setDesktopNotificationsEnabled] = useAtom( desktopNotificationsEnabledAtom, @@ -209,6 +213,19 @@ export function AgentsPreferencesTab() { +
+
+ Prompt Suggestions + + Ask Claude for one suggested next prompt when a turn finishes, and show it above the + composer. Off by default. + +
+ +
Default Mode diff --git a/src/renderer/features/agents/atoms/index.ts b/src/renderer/features/agents/atoms/index.ts index bcd36d5c..c7b546bd 100644 --- a/src/renderer/features/agents/atoms/index.ts +++ b/src/renderer/features/agents/atoms/index.ts @@ -637,6 +637,19 @@ export const subChatRooModelIdAtomFamily = atomFamily((subChatId: string) => ), ) +/** + * The one prompt suggestion the last finished turn produced for this sub-chat. + * Not persisted: a suggestion belongs to the turn that produced it, so a + * reloaded chat starts with none instead of offering a stale next step. The + * transport writes it when the SDK sends `prompt_suggestion`, and the composer + * renders it and clears it on use or dismiss. The key is the sub-chat the + * suggestion belongs to; jotai does the keying, so the created atom itself has + * nothing to read from it. + */ +export const subChatPromptSuggestionAtomFamily = atomFamily((_subChatId: string) => + atom(null), +) + export const subChatCodexThinkingAtomFamily = atomFamily((subChatId: string) => atom( (get) => { diff --git a/src/renderer/features/agents/components/agent-model-selector.tsx b/src/renderer/features/agents/components/agent-model-selector.tsx index 4a9643ff..ef515eba 100644 --- a/src/renderer/features/agents/components/agent-model-selector.tsx +++ b/src/renderer/features/agents/components/agent-model-selector.tsx @@ -4,6 +4,7 @@ import { Brain, ChevronRight, Zap } from "lucide-react" import { AnimatePresence, motion } from "motion/react" import { useCallback, useEffect, useMemo, useRef, useState } from "react" import { createPortal } from "react-dom" +import { type EffortLevel, formatEffortLabel } from "../../../../shared/effort" import { Button } from "../../../components/ui/button" import { Checkbox } from "../../../components/ui/checkbox" import { @@ -31,7 +32,6 @@ import { Popover, PopoverContent, PopoverTrigger } from "../../../components/ui/ import { Switch } from "../../../components/ui/switch" import { cn } from "../../../lib/utils" import type { CodexThinkingLevel } from "../lib/models" -import { formatCodexThinkingLabel } from "../lib/models" const CROSS_PROVIDER_DIALOG_DISMISSED_KEY = "agent-model-selector:skip-cross-provider-dialog" @@ -158,6 +158,10 @@ interface AgentModelSelectorProps { isConnected: boolean thinkingEnabled: boolean onThinkingChange: (enabled: boolean) => void + /** Empty while the backend's capability profile reports no effort control. */ + efforts: readonly EffortLevel[] + selectedEffort: EffortLevel | null + onSelectEffort: (effort: EffortLevel | null) => void } codex: { models: CodexModelOption[] @@ -231,14 +235,28 @@ type FlatModelItem = | { type: "ollama"; modelName: string; isRecommended: boolean } | { type: "custom" } -function CodexThinkingSubMenu({ +/** + * The effort sub-menu both backends that take an effort share. Codex has always + * had one; Claude gained one with the SDK pin that added `Options.effort`. The + * levels come from `src/shared/effort.ts` so two pickers cannot drift to two + * vocabularies, and the trigger row is the one Codex already shipped. + */ +function EffortSubMenu({ thinkings, selectedThinking, onSelectThinking, + onClear, }: { - thinkings: CodexThinkingLevel[] - selectedThinking: CodexThinkingLevel - onSelectThinking: (thinking: CodexThinkingLevel) => void + thinkings: readonly TLevel[] + selectedThinking: TLevel | null + onSelectThinking: (thinking: TLevel) => void + /** + * Renders a row that hands the choice back to the runtime. Claude passes it: + * a chat that never opened the picker must keep the model's own default + * instead of a level this app guessed. Codex always sends a concrete level, + * so it passes nothing and the row does not appear. + */ + onClear?: () => void }) { const triggerRef = useRef(null) const subMenuRef = useRef(null) @@ -324,7 +342,9 @@ function CodexThinkingSubMenu({ Thinking
- {formatCodexThinkingLabel(selectedThinking)} + + {selectedThinking === null ? "Default" : formatEffortLabel(selectedThinking)} +
@@ -339,6 +359,17 @@ function CodexThinkingSubMenu({ className="fixed z-50 min-w-[180px] overflow-auto rounded-[10px] border border-border bg-popover text-sm text-popover-foreground shadow-lg py-1 animate-in fade-in-0 zoom-in-95 slide-in-from-left-2" style={{ top: subPos.top, left: subPos.left }} > + {onClear && ( + + )} {thinkings.map((thinking) => { const isSelected = selectedThinking === thinking return ( @@ -349,7 +380,7 @@ function CodexThinkingSubMenu({ onFocus={cancelClose} className="flex items-center justify-between gap-4 min-h-[32px] py-[5px] px-1.5 mx-1 w-[calc(100%-8px)] rounded-md text-sm cursor-default select-none outline-none dark:hover:bg-neutral-800 hover:text-foreground transition-colors" > - {formatCodexThinkingLabel(thinking)} + {formatEffortLabel(thinking)} {isSelected && } ) @@ -940,6 +971,14 @@ export function AgentModelSelector({ className="scale-75" /> + {claude.efforts.length > 0 && ( + claude.onSelectEffort(null)} + /> + )} )} @@ -952,7 +991,7 @@ export function AgentModelSelector({ if (!selectedCodexModel) return null return ( <> - { // Read extended thinking setting dynamically (so toggle applies to existing chats) const thinkingEnabled = appStore.get(extendedThinkingEnabledAtom) - // Max thinking tokens for extended thinking mode - // SDK adds +1 internally, so 64000 becomes 64001 which exceeds Opus 4.5 limit - // Using 32000 to stay safely under the 64000 max output tokens limit - const maxThinkingTokens = thinkingEnabled ? 32_000 : undefined + // Adaptive lets the model pick its own budget; disabled is what the toggle + // off has always meant but could not say while the field was a token count. + const thinking = thinkingEnabled + ? ({ type: "adaptive" } as const) + : ({ type: "disabled" } as const) + // null is "let the CLI choose", so a chat that never opened the picker keeps + // the model's own default instead of a level this app guessed. + const effort = appStore.get(claudeEffortAtom) + const promptSuggestions = appStore.get(promptSuggestionsEnabledAtom) const historyEnabled = appStore.get(historyEnabledAtom) const enableTasks = appStore.get(enableTasksAtom) @@ -194,7 +202,9 @@ export class IPCChatTransport implements ChatTransport { projectPath: this.config.projectPath, // Original project path for MCP config lookup mode: currentMode, sessionId, - ...(maxThinkingTokens && { maxThinkingTokens }), + thinking, + ...(effort && { effort }), + ...(promptSuggestions && { promptSuggestions: true }), ...(modelString && { model: modelString }), ...(customConfig && { customConfig }), ...(selectedOllamaModel && { selectedOllamaModel }), @@ -340,6 +350,17 @@ export class IPCChatTransport implements ChatTransport { } // Handle retry notification - show friendly toast instead of scary error + // A suggestion is not part of the assistant message, so it goes + // to the composer atom for this sub-chat and is never enqueued as + // a stream chunk the AI SDK would not recognize. + if (chunk.type === "prompt-suggestion") { + appStore.set( + subChatPromptSuggestionAtomFamily(this.config.subChatId), + chunk.suggestion, + ) + return + } + if (chunk.type === "retry-notification") { toast.info("Retrying request", { description: chunk.message || "Request was unsuccessful, trying again...", diff --git a/src/renderer/features/agents/lib/models.ts b/src/renderer/features/agents/lib/models.ts index 0262cfbe..7777572c 100644 --- a/src/renderer/features/agents/lib/models.ts +++ b/src/renderer/features/agents/lib/models.ts @@ -8,7 +8,6 @@ export { CODEX_MODELS, CODEX_SUBSCRIPTION_ONLY_MODEL_IDS, type CodexThinkingLevel, - formatCodexThinkingLabel, } from "../../../../shared/codex-model-id" export const CLAUDE_MODELS = [ diff --git a/src/renderer/features/agents/main/chat-input-area.tsx b/src/renderer/features/agents/main/chat-input-area.tsx index 0930aee2..753766bf 100644 --- a/src/renderer/features/agents/main/chat-input-area.tsx +++ b/src/renderer/features/agents/main/chat-input-area.tsx @@ -9,10 +9,11 @@ */ import { useAtom, useAtomValue, useSetAtom } from "jotai" -import { ChevronDown, Zap } from "lucide-react" +import { ChevronDown, Sparkles, Zap } from "lucide-react" import { memo, useCallback, useEffect, useMemo, useRef, useState } from "react" import { createPortal } from "react-dom" import { toast } from "sonner" +import { EFFORT_LEVELS } from "../../../../shared/effort" import { nativeModeRefusal } from "../../../../shared/permissions/native-mode-floor" import { permissionFloorFor } from "../../../../shared/provider-capabilities" import { Button } from "../../../components/ui/button" @@ -33,6 +34,7 @@ import { agentsSettingsDialogOpenAtom, anthropicOnboardingCompletedAtom, apiKeyOnboardingCompletedAtom, + claudeEffortAtom, codexApiKeyAtom, codexOnboardingCompletedAtom, customClaudeConfigAtom, @@ -83,6 +85,7 @@ import { subChatModelIdAtomFamily, subChatOpenclawModelIdAtomFamily, subChatOpenRouterModelIdAtomFamily, + subChatPromptSuggestionAtomFamily, subChatQwenModelIdAtomFamily, subChatRooModelIdAtomFamily, } from "../atoms" @@ -469,6 +472,12 @@ export const ChatInputArea = memo(function ChatInputArea({ // Model dropdown state const [isModelDropdownOpen, setIsModelDropdownOpen] = useState(false) + const promptSuggestionAtom = useMemo( + () => subChatPromptSuggestionAtomFamily(subChatId), + [subChatId], + ) + const [promptSuggestion, setPromptSuggestion] = useAtom(promptSuggestionAtom) + const subChatModelIdAtom = useMemo(() => subChatModelIdAtomFamily(subChatId), [subChatId]) const [selectedSubChatModelId, setSelectedSubChatModelId] = useAtom(subChatModelIdAtom) const subChatCodexModelIdAtom = useMemo( @@ -800,6 +809,12 @@ export const ChatInputArea = memo(function ChatInputArea({ // Extended thinking (reasoning) toggle const [thinkingEnabled, setThinkingEnabled] = useAtom(extendedThinkingEnabledAtom) + // The effort rows come from the backend's own capability profile, so a + // provider that reports no effort control shows no sub-menu. + const { data: claudeCapability } = trpc.providers.get.useQuery({ id: "claude" }) + const claudeEfforts = claudeCapability?.features.effort ? EFFORT_LEVELS : [] + const [selectedClaudeEffort, setSelectedClaudeEffort] = useAtom(claudeEffortAtom) + const selectedModelLabel = useMemo(() => { if (provider === "codex") { return selectedCodexModel.name @@ -1869,6 +1884,29 @@ export const ChatInputArea = memo(function ChatInputArea({ ) : null } > + {promptSuggestion && ( +
+ + + +
+ )}
( { getOnInit: true }, ) +// Preferences - Claude reasoning effort +// The levels come from `src/shared/effort.ts`, the one vocabulary the main +// process validates a request against. `null` means the chat never picked one +// and the CLI decides, which is the behaviour before this setting existed. +export const claudeEffortAtom = atomWithStorage( + "preferences:claude-effort", + null, + undefined, + { getOnInit: true }, +) + +// Preferences - Prompt suggestions +// Off by default: turning it on asks the backend for a suggested next prompt at +// the end of a turn and puts one clickable row above the composer. +export const promptSuggestionsEnabledAtom = atomWithStorage( + "preferences:prompt-suggestions-enabled", + false, + undefined, + { getOnInit: true }, +) + // Preferences - History (Rollback) // When enabled, allow rollback to previous assistant messages export const historyEnabledAtom = atomWithStorage( diff --git a/src/shared/codex-model-id.ts b/src/shared/codex-model-id.ts index 7daa3c07..4f73ca87 100644 --- a/src/shared/codex-model-id.ts +++ b/src/shared/codex-model-id.ts @@ -21,7 +21,19 @@ * `.dump/global/decisions.md`; the two literals the release note carried, * `gpt-5.5` and `gpt-5.4`, were both refused. */ -export const CODEX_REASONING_EFFORTS = ["low", "medium", "high", "xhigh"] as const +import type { EffortLevel } from "./effort" + +/** + * The four levels Codex sends. `satisfies` keeps the tuple type the picker + * already depends on while proving every member is in the one effort vocabulary + * in `src/shared/effort.ts`. + */ +export const CODEX_REASONING_EFFORTS = [ + "low", + "medium", + "high", + "xhigh", +] as const satisfies readonly EffortLevel[] export type CodexReasoningEffort = (typeof CODEX_REASONING_EFFORTS)[number] @@ -41,8 +53,10 @@ export type CodexReasoningEffort = (typeof CODEX_REASONING_EFFORTS)[number] * Whoever retires it should rename all 25 at once, drop this alias and the * `CodexThinkingLevel` line from the re-export in * `src/renderer/features/agents/lib/models.ts`, and confirm - * `npm run typecheck` still reports 0 errors. The effort label itself is - * `formatCodexThinkingLabel` below, which stays. + * `npm run typecheck` still reports 0 errors. The effort label is + * `formatEffortLabel` in `src/shared/effort.ts`; the picker calls that directly + * now that both backends share one effort vocabulary, and the Codex-named + * wrapper it used to call is gone rather than left as a second name for it. */ export type CodexThinkingLevel = CodexReasoningEffort @@ -73,11 +87,6 @@ export const CODEX_MODELS = [ }, ] as const -export function formatCodexThinkingLabel(thinking: CodexThinkingLevel): string { - if (thinking === "xhigh") return "Extra High" - return thinking.charAt(0).toUpperCase() + thinking.slice(1) -} - export const DEFAULT_CODEX_UI_MODEL = "gpt-5.5" /** Effort sent when a chat has no stored effort and the CLI answers nothing. */ diff --git a/src/shared/effort.test.ts b/src/shared/effort.test.ts new file mode 100644 index 00000000..1b870400 --- /dev/null +++ b/src/shared/effort.test.ts @@ -0,0 +1,34 @@ +import { describe, expect, it } from "vitest" +import { CODEX_REASONING_EFFORTS } from "./codex-model-id" +import { EFFORT_LEVELS, formatEffortLabel, isEffortLevel } from "./effort" + +describe("the shared effort vocabulary", () => { + it("keeps the levels the pinned Claude Agent SDK declares", () => { + // `Options.effort` at 0.3.270 is low | medium | high | xhigh | max. This + // fails on purpose if the list is edited without re-reading the SDK type. + expect([...EFFORT_LEVELS]).toEqual(["low", "medium", "high", "xhigh", "max"]) + }) + + it("keeps every Codex effort inside the shared vocabulary", () => { + for (const effort of CODEX_REASONING_EFFORTS) { + expect(isEffortLevel(effort)).toBe(true) + } + }) + + it("accepts the advertised levels and nothing else", () => { + expect(EFFORT_LEVELS.every((level) => isEffortLevel(level))).toBe(true) + expect(isEffortLevel("default")).toBe(false) + expect(isEffortLevel("")).toBe(false) + expect(isEffortLevel("MAX")).toBe(false) + }) + + it("labels the levels the picker already labelled them", () => { + expect(formatEffortLabel("low")).toBe("Low") + expect(formatEffortLabel("medium")).toBe("Medium") + expect(formatEffortLabel("high")).toBe("High") + // The one level whose capitalized form is not the label the Codex picker + // shipped, and the reason this is a function rather than a capitalize call. + expect(formatEffortLabel("xhigh")).toBe("Extra High") + expect(formatEffortLabel("max")).toBe("Max") + }) +}) diff --git a/src/shared/effort.ts b/src/shared/effort.ts new file mode 100644 index 00000000..89ef82b3 --- /dev/null +++ b/src/shared/effort.ts @@ -0,0 +1,35 @@ +/** + * One reasoning-effort vocabulary for every backend that accepts one. + * + * The Claude Agent SDK at the pin in `package.json` (0.3.270) types its + * `Options.effort` as `"low" | "medium" | "high" | "xhigh" | "max"`, and its + * `ModelInfo` reports `supportedEffortLevels` from the same set. Codex has + * always shipped four of those five, `max` being the one it does not send. Two + * lists for one concept is how two backends drift, the same way the Codex model + * id and its effort once drifted apart as one slash-joined string, so the wider + * set lives here and each backend proves it is a subset: + * `CODEX_REASONING_EFFORTS` in `codex-model-id.ts` carries + * `satisfies readonly EffortLevel[]`. + * + * Which levels a given model accepts is the CLI's answer, not this module's. + * The SDK documents that an effort above a model's `maxEffortLevel` is clamped + * to it, so the picker offers the vocabulary and the runtime settles the model. + * Reading `ModelInfo.supportedEffortLevels` per model is the follow-up that + * retires a static list, mirroring what `providers/codex-models.ts` already does + * for the Codex catalog. + */ +export const EFFORT_LEVELS = ["low", "medium", "high", "xhigh", "max"] as const + +export type EffortLevel = (typeof EFFORT_LEVELS)[number] + +const EFFORT_VALUES: ReadonlySet = new Set(EFFORT_LEVELS) + +export function isEffortLevel(value: string): value is EffortLevel { + return EFFORT_VALUES.has(value) +} + +/** The row label the picker shows for an effort level. */ +export function formatEffortLabel(level: EffortLevel): string { + if (level === "xhigh") return "Extra High" + return level.charAt(0).toUpperCase() + level.slice(1) +} diff --git a/src/shared/provider-capabilities.ts b/src/shared/provider-capabilities.ts index 5bcfb4d0..efd2a44a 100644 --- a/src/shared/provider-capabilities.ts +++ b/src/shared/provider-capabilities.ts @@ -72,6 +72,12 @@ export const featureFlagsSchema = z.object({ skills: z.boolean(), structuredOutput: z.boolean(), fileCheckpointing: z.boolean(), + /** The backend accepts a reasoning-effort level from `src/shared/effort.ts` on a turn. */ + effort: z.boolean(), + /** The backend can pick its own thinking budget per turn instead of being handed one. */ + adaptiveThinking: z.boolean(), + /** The backend can suggest a next prompt after it finishes a turn. */ + promptSuggestions: z.boolean(), }) export const providerCapabilitySchema = z.object({ From f6601752ec4b9915af85b8f8949d494fc17ab9de Mon Sep 17 00:00:00 2001 From: Owie6789 <151057755+Owie6789@users.noreply.github.com> Date: Wed, 23 Sep 2026 12:54:50 +0000 Subject: [PATCH 05/60] Add the three drag and drop packages the roadmap ratified Steps 17, 18 and 37 all need the same drag interaction: panes, sub-chat rows and the queue. The human approved @dnd-kit as an exception to the no-new-dependency rule so those three do not grow three hand-rolled implementations, and this is the only step allowed to touch package.json and bun.lock, so the packages land here rather than in the step that first imports one. Ratified 2026-09-13 in .dump/global/decisions.md and recorded in .dump/app/plans/release-parity-v0.0.75-0.0.84-plan.md section 5.2, which withdrew the native HTML5 recommendation this file argued for. Exact pins, no carets: core 6.3.1, sortable 10.0.0, utilities 3.2.2. The lockfile adds those three plus the transitive @dnd-kit/accessibility 3.1.1 and tslib. Nothing imports them yet, which is the point of the exception and the reason the bundle delta belongs to step 30 rather than to a feature commit. Co-authored-by: arena-agent <297053741+arena-agent@users.noreply.github.com> --- bun.lock | 11 +++++++++++ package.json | 3 +++ 2 files changed, 14 insertions(+) diff --git a/bun.lock b/bun.lock index 293b2f4c..9f09c621 100644 --- a/bun.lock +++ b/bun.lock @@ -8,6 +8,9 @@ "@ai-sdk/google": "^3.0.65", "@ai-sdk/react": "^3.0.14", "@anthropic-ai/claude-agent-sdk": "0.3.270", + "@dnd-kit/core": "6.3.1", + "@dnd-kit/sortable": "10.0.0", + "@dnd-kit/utilities": "3.2.2", "@effect/platform-node": "4.0.0-rc.112", "@effect/platform-node-shared": "4.0.0-rc.112", "@git-diff-view/react": "^0.0.35", @@ -269,6 +272,14 @@ "@develar/schema-utils": ["@develar/schema-utils@2.6.5", "", { "dependencies": { "ajv": "^6.12.0", "ajv-keywords": "^3.4.1" } }, "sha512-0cp4PsWQ/9avqTVMCtZ+GirikIA36ikvjtHweU4/j8yLtgObI0+JUPhYFScgwlteveGB1rt3Cm8UhN04XayDig=="], + "@dnd-kit/accessibility": ["@dnd-kit/accessibility@3.1.1", "", { "dependencies": { "tslib": "^2.0.0" }, "peerDependencies": { "react": ">=16.8.0" } }, "sha512-2P+YgaXF+gRsIihwwY1gCsQSYnu9Zyj2py8kY5fFvUM1qm2WA2u639R6YNVfU4GWr+ZM5mqEsfHZZLoRONbemw=="], + + "@dnd-kit/core": ["@dnd-kit/core@6.3.1", "", { "dependencies": { "@dnd-kit/accessibility": "^3.1.1", "@dnd-kit/utilities": "^3.2.2", "tslib": "^2.0.0" }, "peerDependencies": { "react": ">=16.8.0", "react-dom": ">=16.8.0" } }, "sha512-xkGBRQQab4RLwgXxoqETICr6S5JlogafbhNsidmrkVv2YRs5MLwpjoF2qpiGjQt8S9AoxtIV603s0GIUpY5eYQ=="], + + "@dnd-kit/sortable": ["@dnd-kit/sortable@10.0.0", "", { "dependencies": { "@dnd-kit/utilities": "^3.2.2", "tslib": "^2.0.0" }, "peerDependencies": { "@dnd-kit/core": "^6.3.0", "react": ">=16.8.0" } }, "sha512-+xqhmIIzvAYMGfBYYnbKuNicfSsk4RksY2XdmJhT+HAC01nix6fHCztU68jooFiMUB01Ky3F0FyOvhG/BZrWkg=="], + + "@dnd-kit/utilities": ["@dnd-kit/utilities@3.2.2", "", { "dependencies": { "tslib": "^2.0.0" }, "peerDependencies": { "react": ">=16.8.0" } }, "sha512-+MKAJEOfaBe5SmV6t34p80MMKhjvUz0vRrvVJbPT0WElzaOJ/1xs+D+KDv+tD/NE5ujfrChEcshd4fLn0wpiqg=="], + "@drizzle-team/brocli": ["@drizzle-team/brocli@0.10.2", "", {}, "sha512-z33Il7l5dKjUgGULTqBsQBQwckHh5AbIuxhdsIxDDiZAzBOrZO6q9ogcWC65kU382AfynTfgNumVcNIjuIua6w=="], "@effect/platform-node": ["@effect/platform-node@4.0.0-rc.112", "", { "dependencies": { "@effect/platform-node-shared": "^4.0.0-rc.112", "mime": "^4.1.0", "undici": "^8.10.0" }, "peerDependencies": { "effect": "^4.0.0-rc.112", "redis": ">=5.0.0 <7.0.0" } }, "sha512-/BMAcdNGQQskLmI0Zoa95KfTZkr9HV9N4NSxaSrusG6GeW6Ulp9KvZ+Rlaiw8lnOt43CXjFLdfll5/k5rxL4hQ=="], diff --git a/package.json b/package.json index 7cb8454a..df379a53 100644 --- a/package.json +++ b/package.json @@ -48,6 +48,9 @@ "@ai-sdk/google": "^3.0.65", "@ai-sdk/react": "^3.0.14", "@anthropic-ai/claude-agent-sdk": "0.3.270", + "@dnd-kit/core": "6.3.1", + "@dnd-kit/sortable": "10.0.0", + "@dnd-kit/utilities": "3.2.2", "@effect/platform-node": "4.0.0-rc.112", "@effect/platform-node-shared": "4.0.0-rc.112", "@git-diff-view/react": "^0.0.35", From f5c104406c7837ca1f251e1576625d21953884c8 Mon Sep 17 00:00:00 2001 From: Owie6789 <151057755+Owie6789@users.noreply.github.com> Date: Wed, 23 Sep 2026 12:55:05 +0000 Subject: [PATCH 06/60] Declare the sharp the icon script already imports scripts/generate-icon.mjs imports sharp at line 21 and package.json declared nothing, so the script only ran when a transitive copy happened to be hoisted. There is no such copy any more: the old Claude Agent SDK pulled the @img/sharp-* platform binaries in as optional dependencies, the pin bump dropped all fifteen from the lockfile, and a fresh install has no sharp at all. The icon script is part of the release path, so this declares it instead of leaving it to hoisting. Exact pin as a devDependency, 0.35.4, because it is a build-time tool and never ships in the app. The lockfile gains sharp and its @img platform binaries for every target. Verified by importing it in this checkout, which resolves and reports its libvips version. This is the second and last approved dependency exception after the drag and drop packages. Co-authored-by: arena-agent <297053741+arena-agent@users.noreply.github.com> --- bun.lock | 61 ++++++++++++++++++++++++++++++++++++++++++++++++++++ package.json | 1 + 2 files changed, 62 insertions(+) diff --git a/bun.lock b/bun.lock index 9f09c621..0954319f 100644 --- a/bun.lock +++ b/bun.lock @@ -117,6 +117,7 @@ "electron-builder": "^25.1.8", "electron-vite": "^3.0.0", "postcss": "^8.5.1", + "sharp": "0.35.4", "tailwindcss": "^3.4.17", "typescript": "^5.4.5", "vite": "^6.3.4", @@ -304,6 +305,8 @@ "@electron/universal": ["@electron/universal@2.0.1", "", { "dependencies": { "@electron/asar": "^3.2.7", "@malept/cross-spawn-promise": "^2.0.0", "debug": "^4.3.1", "dir-compare": "^4.2.0", "fs-extra": "^11.1.1", "minimatch": "^9.0.3", "plist": "^3.1.0" } }, "sha512-fKpv9kg4SPmt+hY7SVBnIYULE9QJl8L3sCfcBsnqbJwwBwAeTLokJ9TRt9y7bK0JAzIW2y78TVVjvnQEms/yyA=="], + "@emnapi/runtime": ["@emnapi/runtime@1.11.3", "", { "dependencies": { "tslib": "^2.4.0" } }, "sha512-Xz4Tpyki7XyrpbUK1jR1AhdAdaXyhhY4lZ3neLodmhpuWfy2PAQN5B46sAiU4liOXGLkHypn/qU+jvfWSCYYLA=="], + "@esbuild-kit/core-utils": ["@esbuild-kit/core-utils@3.3.2", "", { "dependencies": { "esbuild": "~0.18.20", "source-map-support": "^0.5.21" } }, "sha512-sPRAnw9CdSsRmEtnsl2WXWdyquogVpB3yZ3dgwJfe8zrOzTsV7cJvmwrKVa+0ma5BoiGJ+BoqkMvawbayKUsqQ=="], "@esbuild-kit/esm-loader": ["@esbuild-kit/esm-loader@2.6.5", "", { "dependencies": { "@esbuild-kit/core-utils": "^3.3.2", "get-tsconfig": "^4.7.0" } }, "sha512-FxEMIkJKnodyA1OaCUoEvbYRkoZlLZ4d/eXFu9Fh8CbBBgP5EmZxrfTRyN0qpXZ4vOvqnE5YdRdcrmUUXuU+dA=="], @@ -384,6 +387,60 @@ "@iconify/utils": ["@iconify/utils@3.1.0", "", { "dependencies": { "@antfu/install-pkg": "^1.1.0", "@iconify/types": "^2.0.0", "mlly": "^1.8.0" } }, "sha512-Zlzem1ZXhI1iHeeERabLNzBHdOa4VhQbqAcOQaMKuTuyZCpwKbC2R4Dd0Zo3g9EAc+Y4fiarO8HIHRAth7+skw=="], + "@img/colour": ["@img/colour@1.1.0", "", {}, "sha512-Td76q7j57o/tLVdgS746cYARfSyxk8iEfRxewL9h4OMzYhbW4TAcppl0mT4eyqXddh6L/jwoM75mo7ixa/pCeQ=="], + + "@img/sharp-darwin-arm64": ["@img/sharp-darwin-arm64@0.35.4", "", { "optionalDependencies": { "@img/sharp-libvips-darwin-arm64": "1.3.3" }, "os": "darwin", "cpu": "arm64" }, "sha512-Uhfl4V4lhP2nbUVF9+hyH1+luj86f1gUFeo8ALYxFoULoU+G87D43BfeMP8XHsk9boxAnCY/bf2EHwhA7MuGsA=="], + + "@img/sharp-darwin-x64": ["@img/sharp-darwin-x64@0.35.4", "", { "optionalDependencies": { "@img/sharp-libvips-darwin-x64": "1.3.3" }, "os": "darwin", "cpu": "x64" }, "sha512-hWniXY3bG5qKpkKrAwPe4y+VTPmf086YQAnkxWh7uA1YrlRouWGa0M0Mxj3ZjnXFkv7/TD1bTy9lGUK26vRvWw=="], + + "@img/sharp-freebsd-wasm32": ["@img/sharp-freebsd-wasm32@0.35.4", "", { "dependencies": { "@img/sharp-wasm32": "0.35.4" }, "os": "freebsd" }, "sha512-lIsKw/BU+kjB4eZjxrYrZmwOJYi3Ajrv66iAlBmUPyKc3HpnloevB1g3wxGD9P/5BbQ1brBGl65VRRrCvQDEqA=="], + + "@img/sharp-libvips-darwin-arm64": ["@img/sharp-libvips-darwin-arm64@1.3.3", "", { "os": "darwin", "cpu": "arm64" }, "sha512-suTBPTDGrI9WodccaDdwZItTSaBYASlBk1NSfElSHrUfzu3szG6lvIF58+WiFvnfzuK8ZBFS5zE00PxqxnRiPg=="], + + "@img/sharp-libvips-darwin-x64": ["@img/sharp-libvips-darwin-x64@1.3.3", "", { "os": "darwin", "cpu": "x64" }, "sha512-FVJZ5mITMobmXIz/hPDTw0EintTW5H3WfrxwLqEqjiIihlu+hVRyGrFQ60xl0Lxn7Bt3zdpevPaQi0HEzqz9fw=="], + + "@img/sharp-libvips-linux-arm": ["@img/sharp-libvips-linux-arm@1.3.3", "", { "os": "linux", "cpu": "arm" }, "sha512-3rbU4vqXXc3hY/OiXdl52xZvT0F1yEngWfvqudtPJg/KkyiaQw2DRsFrNzpmLvfavbwOq3qXn36GP8obHRULQA=="], + + "@img/sharp-libvips-linux-arm64": ["@img/sharp-libvips-linux-arm64@1.3.3", "", { "os": "linux", "cpu": "arm64" }, "sha512-0DaL0A6Xu6sQSQFwe4iVCrKWU2cCTItnRsYsCdxAMm9NF6twAA9BKnoqy4hqz4+azQ0JHuA26qiUKsf1XJ/v5A=="], + + "@img/sharp-libvips-linux-ppc64": ["@img/sharp-libvips-linux-ppc64@1.3.3", "", { "os": "linux", "cpu": "ppc64" }, "sha512-cdn1OvUBwsXhbC0zSzJnNzf5MZ/mTrobawDvNXBTxe8VtqKAm0sRuEY2Evzovb/w9JMk4TvRxqt1mekSuJz64w=="], + + "@img/sharp-libvips-linux-riscv64": ["@img/sharp-libvips-linux-riscv64@1.3.3", "", { "os": "linux", "cpu": "none" }, "sha512-HjPVx7yKz+0lqdhDlTw1tt90wamBoxhiXpvl1XZpJLiHH4RCJ5yDTqH+VlYPv2fwFs89JFw4c1IexYOcQUi4IQ=="], + + "@img/sharp-libvips-linux-s390x": ["@img/sharp-libvips-linux-s390x@1.3.3", "", { "os": "linux", "cpu": "s390x" }, "sha512-neWLh+3yCNThxnfy3c4BbVBeGgt9aftno+XbT56iK28RgeDs3UOFWviLWlUu0bArYVYJaFDK+RRohbicUNCm8Q=="], + + "@img/sharp-libvips-linux-x64": ["@img/sharp-libvips-linux-x64@1.3.3", "", { "os": "linux", "cpu": "x64" }, "sha512-4vKmvAst9nrowcqquKFAyZJUDolUaIp8uRiN0mWFguJ1IplC9/pitXtlnnlU4aa/eJw3J7i67V+pwUL+wZGdsA=="], + + "@img/sharp-libvips-linuxmusl-arm64": ["@img/sharp-libvips-linuxmusl-arm64@1.3.3", "", { "os": "linux", "cpu": "arm64" }, "sha512-Y9kQaLMuNoB0bPYOOdcZMaseNrFpPodIWWMrx+CZyydf2xn68j9WYc6sWWRrDwNkzCQjKYfc68L7jKjGlHMibw=="], + + "@img/sharp-libvips-linuxmusl-x64": ["@img/sharp-libvips-linuxmusl-x64@1.3.3", "", { "os": "linux", "cpu": "x64" }, "sha512-fj8Mv0HHfD1Rr+4I68+3agJynxDWtBFgicTbSOb9Bke6pIwzGcJ+RX/yHjmiEGFMCavY/dxvem7MyNaJF+wDiw=="], + + "@img/sharp-linux-arm": ["@img/sharp-linux-arm@0.35.4", "", { "optionalDependencies": { "@img/sharp-libvips-linux-arm": "1.3.3" }, "os": "linux", "cpu": "arm" }, "sha512-7OAS8gI0EReKGVN2HssHlM6umJgxF5VI3xN0p9FA91p/YO+ou5hiNghLdZ5BEHztwaaK5+bLKRf8x/o2L2nk9A=="], + + "@img/sharp-linux-arm64": ["@img/sharp-linux-arm64@0.35.4", "", { "optionalDependencies": { "@img/sharp-libvips-linux-arm64": "1.3.3" }, "os": "linux", "cpu": "arm64" }, "sha512-De4jpEnAU8Hd5oT0j1G3uL4ZvTuipVMn7YC6vPaJhy6/7EwEae0SVAoBrUMYQbkLGDm85taVWwuPc1a44LTzCQ=="], + + "@img/sharp-linux-ppc64": ["@img/sharp-linux-ppc64@0.35.4", "", { "optionalDependencies": { "@img/sharp-libvips-linux-ppc64": "1.3.3" }, "os": "linux", "cpu": "ppc64" }, "sha512-2oYZJeIl4kCcMGk4ouZVjnkCtFrpQFlNEtJ6GbxzhHQchwH0NH/qEb9ykmOl29dqwMq+JhFdZn+1ak2FKhI9fQ=="], + + "@img/sharp-linux-riscv64": ["@img/sharp-linux-riscv64@0.35.4", "", { "optionalDependencies": { "@img/sharp-libvips-linux-riscv64": "1.3.3" }, "os": "linux", "cpu": "none" }, "sha512-cPbNChoRURAWdebDIHSenxRpgEdy7JkPydSnUxRm9VvKD7m0/xVaR/8Fzlu81pk5nHEvHH87UZUA7cTtwnbJSA=="], + + "@img/sharp-linux-s390x": ["@img/sharp-linux-s390x@0.35.4", "", { "optionalDependencies": { "@img/sharp-libvips-linux-s390x": "1.3.3" }, "os": "linux", "cpu": "s390x" }, "sha512-RY0JFY8Fd6RonCBtHz+DvadaPkXDSI1AUn6yWL9TipqkZ1vY8w8evqdgyDFnkm4/K1ve1TvZiaePP5oSd4+WVQ=="], + + "@img/sharp-linux-x64": ["@img/sharp-linux-x64@0.35.4", "", { "optionalDependencies": { "@img/sharp-libvips-linux-x64": "1.3.3" }, "os": "linux", "cpu": "x64" }, "sha512-9qvvEAuk8k89TfWUoX2htWjbAMX8p+NxCppjpcg5k6xMsjhBQPTsoIh36h9Qde4WRuGpJeYnOjdosDn/cnv+OA=="], + + "@img/sharp-linuxmusl-arm64": ["@img/sharp-linuxmusl-arm64@0.35.4", "", { "optionalDependencies": { "@img/sharp-libvips-linuxmusl-arm64": "1.3.3" }, "os": "linux", "cpu": "arm64" }, "sha512-KB5jxpfWQTr0nc3xdHtWChdbifHrBGsd2SM62Eyxrl8afikm+f5qGBU75SJIZBT/S1MC8XyacdlXBMSWq6OURA=="], + + "@img/sharp-linuxmusl-x64": ["@img/sharp-linuxmusl-x64@0.35.4", "", { "optionalDependencies": { "@img/sharp-libvips-linuxmusl-x64": "1.3.3" }, "os": "linux", "cpu": "x64" }, "sha512-f+eZJZIQNEEd26RPSW+76chwOf1XtA2Y/O+5ocVyLliHkeih3e+jhLVBdNTd2rS3IbNXK8+ug93Vf5ZXtF5Lxg=="], + + "@img/sharp-wasm32": ["@img/sharp-wasm32@0.35.4", "", { "dependencies": { "@emnapi/runtime": "^1.11.3" } }, "sha512-zQnl4Kwp7Q6NHsENtU2T/00Zi+w3AQNwz3+UaTyVBy2FpXrzXzGjndpK61onhZjRtRpQXxCTeqw19bVyXOh7jA=="], + + "@img/sharp-webcontainers-wasm32": ["@img/sharp-webcontainers-wasm32@0.35.4", "", { "dependencies": { "@img/sharp-wasm32": "0.35.4" }, "cpu": "none" }, "sha512-ESfNkywmCfPNyaZjxooddJQiQ+l/nTpGEOGthxiLnIHXC/CmcBixnfwUleX9mCz9ovrUUvKMap/pm8RYbzfwaA=="], + + "@img/sharp-win32-arm64": ["@img/sharp-win32-arm64@0.35.4", "", { "os": "win32", "cpu": "arm64" }, "sha512-iNdlBX9gLVvqe2I3uIJSIKTq6wckP/DYxZtcqxm09x5Gi24DnFBmPAWZmr60ZyYMG0xlzo6goG3670ar+RXvRw=="], + + "@img/sharp-win32-ia32": ["@img/sharp-win32-ia32@0.35.4", "", { "os": "win32", "cpu": "ia32" }, "sha512-kqRsbaa5CS6KHlpxnN7WhE6vAAugXyZButpRdvDWetlv6Qv4N9WTcrWzF7tXfB9T7MsoadqdI8hmwLq6UlLvtw=="], + + "@img/sharp-win32-x64": ["@img/sharp-win32-x64@0.35.4", "", { "os": "win32", "cpu": "x64" }, "sha512-XtmnYhBcrORsJ4XJngyzr/EWP0hRZLAZRFaApdKuviyqF78+ylxh2y06ZmtULAMOnObJ3ucpN0AcwSWnMowTRg=="], + "@isaacs/balanced-match": ["@isaacs/balanced-match@4.0.1", "", {}, "sha512-yzMTt9lEb8Gv7zRioUilSglI0c0smZ9k5D65677DLWLtWJaXIS3CqcGyUFByYKlnUj6TkjLVs54fBl6+TiGQDQ=="], "@isaacs/brace-expansion": ["@isaacs/brace-expansion@5.0.0", "", { "dependencies": { "@isaacs/balanced-match": "^4.0.1" } }, "sha512-ZT55BDLV0yv0RBm2czMiZ+SqCGO7AvmOM3G/w2xhVPH+te0aKgFjmBvGlL1dH+ql2tgGO3MVrbb3jCKyvpgnxA=="], @@ -2172,6 +2229,8 @@ "setprototypeof": ["setprototypeof@1.2.0", "", {}, "sha512-E5LDX7Wrp85Kil5bhZv46j8jOeboKq5JMmYM3gVGdGH8xFpPWXUMsNrlODCrkoxMEeNi/XZIwuRvY4XNwYMJpw=="], + "sharp": ["sharp@0.35.4", "", { "dependencies": { "@img/colour": "^1.1.0", "detect-libc": "^2.1.2", "semver": "^7.8.5" }, "optionalDependencies": { "@img/sharp-darwin-arm64": "0.35.4", "@img/sharp-darwin-x64": "0.35.4", "@img/sharp-freebsd-wasm32": "0.35.4", "@img/sharp-libvips-darwin-arm64": "1.3.3", "@img/sharp-libvips-darwin-x64": "1.3.3", "@img/sharp-libvips-linux-arm": "1.3.3", "@img/sharp-libvips-linux-arm64": "1.3.3", "@img/sharp-libvips-linux-ppc64": "1.3.3", "@img/sharp-libvips-linux-riscv64": "1.3.3", "@img/sharp-libvips-linux-s390x": "1.3.3", "@img/sharp-libvips-linux-x64": "1.3.3", "@img/sharp-libvips-linuxmusl-arm64": "1.3.3", "@img/sharp-libvips-linuxmusl-x64": "1.3.3", "@img/sharp-linux-arm": "0.35.4", "@img/sharp-linux-arm64": "0.35.4", "@img/sharp-linux-ppc64": "0.35.4", "@img/sharp-linux-riscv64": "0.35.4", "@img/sharp-linux-s390x": "0.35.4", "@img/sharp-linux-x64": "0.35.4", "@img/sharp-linuxmusl-arm64": "0.35.4", "@img/sharp-linuxmusl-x64": "0.35.4", "@img/sharp-webcontainers-wasm32": "0.35.4", "@img/sharp-win32-arm64": "0.35.4", "@img/sharp-win32-ia32": "0.35.4", "@img/sharp-win32-x64": "0.35.4" }, "peerDependencies": { "@types/node": "*" }, "optionalPeers": ["@types/node"] }, "sha512-n++8XWcj+jCOr2IOl7h8LbKnGBDY4aPbmprMONBNFdn0ImXqpGVv5zliDs0V9HbmbCQLpbuo2ej9rAoOQTvMDA=="], + "shebang-command": ["shebang-command@2.0.0", "", { "dependencies": { "shebang-regex": "^3.0.0" } }, "sha512-kHxr2zZpYtdmrN1qDjrrX/Z1rR1kG8Dx+gkpK1G4eXmvXswmcE1hTWBWYUzlraYw1/yZp6YuDY77YtvbN0dmDA=="], "shebang-regex": ["shebang-regex@3.0.0", "", {}, "sha512-7++dFhtcx3353uBaq8DDR4NuxBetBzC7ZQOhmTQInHEd6bSrXdiEyzCvG07Z44UYdLShWUyXt5M/yhz8ekcb1A=="], @@ -2666,6 +2725,8 @@ "roarr/sprintf-js": ["sprintf-js@1.1.3", "", {}, "sha512-Oo+0REFV59/rz3gfJNKQiBlwfHaSESl1pcGyABQsnnIfWOFt6JNj5gCog2U6MLZ//IGYD+nA8nI+mTShREReaA=="], + "sharp/semver": ["semver@7.8.5", "", { "bin": { "semver": "bin/semver.js" } }, "sha512-Y7/KDsb8LjooZpwaqGyulO6DQlksgCncchHGk+sZIY4SBvUocMBEFH5Ur1fI4dV+Jvl0w6cjvucaIi40puRioA=="], + "shiki/@shikijs/core": ["@shikijs/core@1.29.2", "", { "dependencies": { "@shikijs/engine-javascript": "1.29.2", "@shikijs/engine-oniguruma": "1.29.2", "@shikijs/types": "1.29.2", "@shikijs/vscode-textmate": "^10.0.1", "@types/hast": "^3.0.4", "hast-util-to-html": "^9.0.4" } }, "sha512-vju0lY9r27jJfOY4Z7+Rt/nIOjzJpZ3y+nYpqtUZInVoXQ/TJZcfGnNOGnKjFdVZb8qexiCuSlZRKcGfhhTTZQ=="], "shiki/@shikijs/engine-javascript": ["@shikijs/engine-javascript@1.29.2", "", { "dependencies": { "@shikijs/types": "1.29.2", "@shikijs/vscode-textmate": "^10.0.1", "oniguruma-to-es": "^2.2.0" } }, "sha512-iNEZv4IrLYPv64Q6k7EPpOCE/nuvGiKl7zxdq0WFuRPF5PAE9PRo2JGq/d8crLusM59BRemJ4eOqrFrC4wiQ+A=="], diff --git a/package.json b/package.json index df379a53..2f7e1457 100644 --- a/package.json +++ b/package.json @@ -155,6 +155,7 @@ "electron-builder": "^25.1.8", "electron-vite": "^3.0.0", "postcss": "^8.5.1", + "sharp": "0.35.4", "tailwindcss": "^3.4.17", "typescript": "^5.4.5", "vite": "^6.3.4", From b66b13cd0b2bd6db88f552f613463425e81402f1 Mon Sep 17 00:00:00 2001 From: Owie6789 <151057755+Owie6789@users.noreply.github.com> Date: Wed, 23 Sep 2026 12:55:19 +0000 Subject: [PATCH 07/60] Keep the SDK's per-platform native binary out of the packaged app The pin bump made the Claude Agent SDK ship its own compiled CLI as eight optional per-platform packages. The manifest lists them between 197.9 MB and 216.5 MB, and a Linux dev install holds two of them, 214 MB and 208 MB. electron-builder copies production node_modules into the app even with an explicit files list, so without this every packaged target would have grown by its own platform binary. The app never asks the SDK for that binary: every query passes pathToClaudeCodeExecutable, resolved from resources/bin through extraResources at the single call site in the claude router, and the SDK's own failure text for a missing override asks for a valid path there rather than falling back to its native package, whose resolver is a lazy existsSync lookup over candidate paths. The glob excludes the platform packages and keeps the 5 MB JS SDK, whose directory it does not match. package:linux cannot run in a 3 GB sandbox, so CI is the verifier for the artifact size. Co-authored-by: arena-agent <297053741+arena-agent@users.noreply.github.com> --- package.json | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/package.json b/package.json index 2f7e1457..2c46e699 100644 --- a/package.json +++ b/package.json @@ -180,7 +180,8 @@ "output": "release" }, "files": [ - "out/**/*" + "out/**/*", + "!node_modules/@anthropic-ai/claude-agent-sdk-*" ], "extraResources": [ { From b4e5c7c7360ddb3f91645b680a51bf9806b912d4 Mon Sep 17 00:00:00 2001 From: Owie6789 <151057755+Owie6789@users.noreply.github.com> Date: Wed, 23 Sep 2026 13:02:02 +0000 Subject: [PATCH 08/60] Record what the SDK bump changed and what it costs The step file's handoff notes ask for the upstream behaviour list at this exact research path and for a benchmark record, and both are the difference between a version move someone can audit and one someone has to re-derive. The research file carries the checksum proof that the SDK pin and the CLI pin are one artifact, the 39-member dialect the classification guards now cover, the tool table read out of the 2.1.270 binary that makes the MultiEdit acceptance criterion stale, the changelog entries across the whole range with the versions that matter here, the environment-semantics flip that cancels out inside the range, the options the bump adds and the deprecated field they replace, the packaging weight the native binary brings, and the three rejected approaches to the stream types so nobody retries them. The benchmark file separates what was measured in a 2 CPU, 3 GB sandbox from what needs a build, and leaves the five measurements that need CI named rather than implied. Co-authored-by: arena-agent <297053741+arena-agent@users.noreply.github.com> --- .../app/benchmarks/2026-09-23-sdk-0-3-pins.md | 90 ++++++ .dump/app/research/2026-09-13-sdk-0-3-bump.md | 305 ++++++++++++++++++ 2 files changed, 395 insertions(+) create mode 100644 .dump/app/benchmarks/2026-09-23-sdk-0-3-pins.md create mode 100644 .dump/app/research/2026-09-13-sdk-0-3-bump.md diff --git a/.dump/app/benchmarks/2026-09-23-sdk-0-3-pins.md b/.dump/app/benchmarks/2026-09-23-sdk-0-3-pins.md new file mode 100644 index 00000000..c0feab9f --- /dev/null +++ b/.dump/app/benchmarks/2026-09-23-sdk-0-3-pins.md @@ -0,0 +1,90 @@ +# Step 12 benchmark record: SDK 0.3.270, CLI 2.1.270, Codex 0.154.0 + +Date: 2026-09-23. Track: app (`arena/01a0cbec-mauscode`, issue #14). Status: +install-time weight measured in this sandbox; bundle size per target and turn +latency NOT measured, so the benchmark gate stays CLOSED for those two and CI +owns them. + +## Environment + +Sandbox: 2 CPUs, 3 GB RAM, no display, egress intercepted for the hosts that +serve the Claude and Codex binaries. `.dump/app/roadmap/12-sdk-and-pins.md` §11 +asks for `NODE_OPTIONS=--max-old-space-size=4096 bun run build` and +`bun run package:linux`. Neither can run here: the heap flag alone exceeds +available memory. Both download scripts were attempted and both failed on egress +rather than on the pins, `claude:download` losing the TLS socket to the object +store and `codex:download` failing leaf-certificate verification. Everything +below is either a measurement taken here with the command that produced it, or an +explicit NOT RUN. + +## NOT RUN + +| Measurement | Why | Who runs it | +| --- | --- | --- | +| Bundle size per target, with and without the bump | needs `bun run build` at a 4 GB heap | CI | +| Bundle size per target, with and without `@dnd-kit` | same, and step 30 owns the renderer budget | CI, then step 30 | +| Packaged artifact size, with and without the `build.files` exclusion | needs `bun run package:linux` | CI | +| Turn latency for one scripted prompt, before and after | needs a packaged app and provider credentials | a dev machine | +| `bun run claude:download` and `codex:download` integrity runs | sandbox egress is intercepted | CI | + +## Measured here + +Install weight, `du -sh` on this checkout after all four dependency commits: + +| Path | Size | +| --- | --- | +| `node_modules` | 1.9 GB | +| `node_modules/@anthropic-ai` | 440 MB | +| `node_modules/@anthropic-ai/claude-agent-sdk-linux-x64` | 214 MB | +| `node_modules/@anthropic-ai/claude-agent-sdk-linux-x64-musl` | 208 MB | +| `node_modules/@anthropic-ai/sdk` | 14 MB | +| `node_modules/@anthropic-ai/claude-agent-sdk` | 5.0 MB | +| `node_modules/@img` (sharp's platform binaries) | 37 MB | +| `node_modules/@dnd-kit` (three packages) | 2.1 MB | +| `node_modules/sharp` | 1.3 MB | + +The two SDK platform packages are the whole story: 422 MB of a 1.9 GB install, +and they did not exist at `0.2.45`, which spawned bundled JavaScript instead of a +native binary (SDK changelog `0.2.113`). + +`manifest.json` at `0.3.270`, the size each packaged target would have carried +without the `build.files` exclusion: + +| Platform | Bytes | MB | +| --- | --- | --- | +| darwin-arm64 | 207500480 | 197.9 | +| darwin-x64 | 216316928 | 206.3 | +| linux-arm64 | 223862184 | 213.5 | +| linux-x64 | 223981040 | 213.6 | +| linux-arm64-musl | 216545032 | 206.5 | +| linux-x64-musl | 217894976 | 207.8 | +| win32-x64 | 227051168 | 216.5 | +| win32-arm64 | 218256032 | 208.1 | + +Lockfile rows: 2888 at `33475d8`, 2960 after all four dependency commits, with 16 +package rows added and 16 removed by the pin bump itself. + +Integrity, the one check this sandbox can make: `sha256sum` of +`node_modules/@anthropic-ai/claude-agent-sdk-linux-x64/claude` is +`3a624a5a7cd79bbad4d32bd7db36f1197ecf458bc5bf1e2aed81834a01ad3ef0`, which equals +the `linux-x64` checksum in the SDK's `manifest.json`, size `223981040`. The SDK +pin and the CLI pin are therefore the same artifact, not two independent guesses. + +Gate wall clocks on the same 2 CPU sandbox, as a regression proxy rather than a +product number: `npm run test` over 103 files and 1885 tests took 50.5 s; +`npm run test:node` 59 tests took 39.0 s; `npm run test:contracts` 382 tests took +10.5 s; `npx biome check .` over 977 files took 3 s; `tsc --noEmit` over the whole +repo took 42 s to 65 s per run and reports 0 errors, which is the ratchet's +baseline rather than new debt. + +## What the numbers mean for the next step + +Step 30 records the renderer budget, and `@dnd-kit` is the dependency a +performance claim will be checked against: 2.1 MB installed, three packages, no +importer yet, so its bundle cost is still zero and becomes measurable the moment +step 17 mounts the shared `DndContext`. The `build.files` exclusion is worth +213.6 MB on a linux-x64 artifact and 207.8 MB on linux-x64-musl, which is the +difference between a packaged app that grew by about a fifth and one that did not. +The `sharp` declaration costs 37 MB of platform binaries in a dev install and +nothing in a packaged app, because it is a devDependency and the icon script runs +before packaging. diff --git a/.dump/app/research/2026-09-13-sdk-0-3-bump.md b/.dump/app/research/2026-09-13-sdk-0-3-bump.md new file mode 100644 index 00000000..e374b010 --- /dev/null +++ b/.dump/app/research/2026-09-13-sdk-0-3-bump.md @@ -0,0 +1,305 @@ +# SDK 0.3 bump: what actually changed between 0.2.45 and 0.3.270 + +Date: written 2026-09-23 on `arena/01a0cbec-mauscode`, the step 12 lane for issue +#14. The file name is the one `.dump/app/roadmap/12-sdk-and-pins.md` prescribes in +its handoff notes, which carries the ratification date rather than the working +date. + +Track: app. Status: research complete; the pins, the stream reconciliation, the +tool registry and the three new option surfaces are implemented on this branch. +The measurements that need a build are still open, and are named as open in +`.dump/app/benchmarks/2026-09-23-sdk-0-3-pins.md` rather than estimated here. + +## 1. The versions, and the proof they belong together + +| Pin | Before | After | Where it is written | +| --- | --- | --- | --- | +| `@anthropic-ai/claude-agent-sdk` | `0.2.45` | `0.3.270` | `package.json` dependencies | +| Claude CLI download | `2.1.45` | `2.1.270` | `package.json` `claude:download`, `claude:download:all`, and the offline fallback at `scripts/download-claude-binary.mjs:152` | +| Codex download | `0.137.0` | `0.154.0` | `package.json` `codex:download`, `codex:download:all` | + +Ratified 2026-09-13 in `.dump/global/decisions.md` and argued in +`.dump/app/plans/release-parity-v0.0.75-0.0.84-plan.md` §5.1, which refused the +`0.2.63` a changelog note suggested. Codex `0.154.0` is the version +`.dump/app/roadmap/35-codex-app-server-parity.md` assumes step 12 lands: it was +published as `rust-v0.154.0` on 2026-09-09, and `rust-v0.156.0` is the current +stable, so this pin sits one release behind on purpose, because step 35 owns +Codex parity and this step only moves the pin the roadmap named. + +The SDK and the CLI are one decision, not two, and the pairing is checkable. +`node_modules/@anthropic-ai/claude-agent-sdk/manifest.json` at `0.3.270` lists +`linux-x64` with checksum +`3a624a5a7cd79bbad4d32bd7db36f1197ecf458bc5bf1e2aed81834a01ad3ef0` and size +`223981040`, and `sha256sum` of the binary that install places at +`node_modules/@anthropic-ai/claude-agent-sdk-linux-x64/claude` is exactly that +value. So `0.3.270` bundles CLI `2.1.270`, which is why the download scripts and +the SDK pin move in the same commit and why the script's offline fallback carries +the same version as the script argument: a retry after a failed manifest lookup +cannot silently fetch `2.1.45`. + +Both download scripts were run in this sandbox and both failed on egress rather +than on the pins: `claude:download` lost the TLS socket to the object store that +serves the binaries, and `codex:download` failed certificate verification with +`UNABLE_TO_VERIFY_LEAF_SIGNATURE`. Neither is evidence about the versions. The +integrity path is verified by the checksum match above, and the two script runs +stay open for CI. + +## 2. The stream dialect at the pin + +At `0.2.45` the SDK could not type its own stream boundary: `SDKMessage` was +built from 18 members but `SDKRateLimitEvent` was never declared, and the +assistant, user and stream-event payloads imported `BetaMessage`, +`BetaRawMessageStreamEvent` and `MessageParam` from `@anthropic-ai/sdk`, which +was not in the dependency tree, so under `skipLibCheck` those fields resolved to +`any`. The local shapes in `src/main/lib/claude/types.ts` were the only real +contract, and they were guesses. + +At `0.3.270` that package is a dependency and every member is declared. The +dialect is 39 message members across 11 top-level types, and 28 of the 39 are +`system` subtypes: + +- Top-level: `assistant`, `user` (two members, `SDKUserMessage` and + `SDKUserMessageReplay`), `result`, `system`, `stream_event`, `tool_progress`, + `auth_status`, `tool_use_summary`, `rate_limit_event`, `prompt_suggestion`, + `conversation_reset`. +- `system` subtypes: `init`, `status`, `compact_boundary`, `api_retry`, + `control_request_progress`, `model_refusal_fallback`, + `model_refusal_no_fallback`, `local_command_output`, `hook_started`, + `hook_progress`, `hook_response`, `plugin_install`, `task_notification`, + `task_started`, `task_updated`, `task_progress`, `background_tasks_changed`, + `thinking_tokens`, `session_state_changed`, `worker_shutting_down`, + `commands_changed`, `notification`, `files_persisted`, `memory_recall`, + `elicitation_complete`, `permission_denied`, `mirror_error`, `informational`. + +`transform.ts` classifies all of them, and the classification is enforced rather +than documented: `HANDLED_*` and `INTERNAL_*` arrays carry `satisfies` against +the union's own `type` and `subtype` literals, and two `AssertNever>` +guards fail typecheck when a member is in neither list. `ClaudeStreamMessage` now +covers everything except the four locally mirrored shapes by exclusion from +`SDKMessage`, so a member a future release adds joins the union on the bump and +breaks the build until someone decides what it means. + +The local read contracts stay, because they are what a fixture has to spell and +what the translator is allowed to read, but they are derived from the SDK types +instead of written beside them: each block union keeps the shapes the translator +reads and adds everything else the SDK can send through +`Exclude`. Two fields had to widen to stay +supertypes of what the wire actually carries: + +- A tool result's `content` is optional, and its array can hold a document, a + search result, a tool reference or a browser state, not only text and image. + The reader that renders a failed result now names a part it cannot render + (`[document]`) instead of calling every non-text part `[image]`. +- `ClaudeUsage` cache counters accept `null`, because the pinned SDK reports null + for a cache tier that did not apply, and a null in an arithmetic expression is + a NaN in the context indicator rather than a missing number. + +Three approaches were tried and rejected, recorded so nobody retries them: +widening the local mirrors to all 16 `ContentBlockParam` members by hand +(duplicates Anthropic's own types and then fails assignability on the members +that carry required fields the app does not read); switching the union members to +`SDKAssistantMessage` and `SDKPartialAssistantMessage` directly (their +`BetaTextBlock` requires `citations` and `BetaMessage` has 11 required fields, +which makes every test fixture a lie about what the translator reads); and using +`SDKSystemMessage` as a union member shortcut (it is init-only, and its +`mcp_servers` entry still stops at `{name, status}` while the CLI sends +`serverInfo` and `error`, which the translator renders, so that one shape is +still mirrored field by field). + +## 3. Tool names at the pin, and the stale acceptance criterion + +Grepping the installed `2.1.270` platform binary for its emitted tool table +returns `TaskCreate`, `TaskGet`, `TaskList`, `TaskUpdate`, `TaskStop`, +`TaskOutput`, `Agent`, `TodoWrite`. There is no `Task` and no `MultiEdit` in that +table. The binary's own normalization map turns `KillShell` and `KillBash` into +`TaskStop`, and `BashOutput`, `BashOutputTool`, `AgentOutput` and +`AgentOutputTool` into `TaskOutput`. The SDK's tool types agree on the sub-agent: +`sdk-tools.d.ts` declares `AgentInput` and `AgentOutput` and no `TaskInput`, and +the SDK changelog at `0.2.69` records the wire-name revert with the note that it +"will migrate to `Agent` in the next minor release", which the `0.3` line did. + +So issue #14's acceptance criterion naming three registry entries, `MultiEdit`, +`tool-Agent` and `TaskOutput`, is stale in one of the three. `MultiEdit` appears +in that binary only in permission, deny-rule, display-label and legacy-alias +tables, for example `Edit:"Editing",MultiEdit:"Editing"` and +`toolName==="MultiEdit"?"Edit"`, which is exactly the shape this app's own +classifier mirrors, and never as a tool the runtime emits. Registering it would +have added a row no transcript can reach. Dropping it was the human's decision, +taken during this step, and the evidence is on the issue as a comment. + +What shipped instead: + +- `tool-Task` and `tool-Agent` share one meta object, so the rename cannot drift + into two entries with different wording, and a session resumed from a + pre-bump transcript renders the same as a new one. +- `tool-TaskOutput` shares its meta with `tool-BashOutput`. +- `tool-TaskStop` joins `tool-KillShell`, one entry beyond the two the criterion + names, from the same normalization table, with wording that says task rather + than shell because at that name the thing being stopped may be a sub-agent. +- `src/renderer/features/agents/lib/subagent-tool-types.ts` holds + `SUBAGENT_TOOL_TYPES` and `isSubagentToolType`, the single predicate grouping, + dispatch and the nested-tool lookup all read. +- `assistant-message-item.tsx` no longer suppresses `TaskOutput` rows in either + of the two places that did, which is why background output had no home in the + transcript before this change. + +Two notes for whoever reads this next. The `variant` field on the registry's +`ToolMeta` was dead: nothing in the repo read it, and the structural registry +type in `isolated-message-group.tsx` only asks for icon and title, so it was +removed rather than given a `collapsible` value for `TaskOutput` that would have +described behaviour this codebase does not have. And `BUILTIN_MCP_TOOLS` at +`agent-tool-registry.tsx:614` still lists `Tool`-suffixed duplicates of the same +pairs; that is the MCP truth lane, step 28, and is left alone here. + +The background-task family (`TaskCreate`, `TaskUpdate`, `TaskGet`, `TaskList`, +`TaskStop`) is not the sub-agent family, even though both start with `Task`. +`assistant-message-item.tsx` keeps its own `TASK_TOOLS` set for the former and +reads `SUBAGENT_TOOL_TYPES` for the latter. + +## 4. Options the pin adds, and what they replaced + +- `Options.effort` is `low | medium | high | xhigh | max`, where `max` arrived at + `0.3.214`. `ModelInfo` reports `supportsEffort`, `supportedEffortLevels`, + `supportsAdaptiveThinking`, `supportsFastMode` and `supportsAutoMode`, and the + SDK documents that an effort above a model's `maxEffortLevel` is clamped to it. + The app now has one vocabulary for the concept in `src/shared/effort.ts`, + Codex's four levels carry a `satisfies` proof against it, and the picker's + Codex sub-menu is generalized to `EffortSubMenu` with an optional row that + clears the pick. Reading `supportedEffortLevels` per model is the follow-up + that retires the static list, mirroring what `providers/codex-models.ts` + already does for the Codex catalog. +- `Options.thinking` is `{ adaptive } | { enabled, budgetTokens? } | { disabled }` + and supersedes the deprecated `maxThinkingTokens`. This is a behaviour fix and + not only a rename: the app sent `32000` when the Extended thinking switch was + on, which on a model that supports adaptive thinking meant "adaptive", and sent + nothing when it was off, which meant the model chose a budget anyway. The + switch therefore could not turn thinking off. The transport now sends + `adaptive` when it is on and `disabled` when it is off. +- `Options.promptSuggestions` asks for a `SDKPromptSuggestionMessage` after the + result. It arrives before the router's deferred finish chunk, so ordering is + safe. It is off by default behind a preference, carried per sub-chat, bounded + to 2000 characters in the translator because the string is provider-authored + and crosses IPC, and rendered as one clickable row above the composer. +The upstream changelog, read across the whole range, with the versions that +matter to this app: + +| Version | Change | What it means here | +| --- | --- | --- | +| `0.2.47` | `promptSuggestion()` method on `Query` | the request half of the suggestion surface | +| `0.2.49` | `ModelInfo` gains `supportsEffort`, `supportedEffortLevels`, `supportsAdaptiveThinking` | the per-model discovery that would retire the static effort list | +| `0.2.69` | `system:init` and `result` emit `Task` again, reverted from `Agent`, "will migrate to `Agent` in the next minor release" | why two spellings of one tool exist, and why the registry keeps both keys | +| `0.2.84` | exported `EffortLevel`, then `low | medium | high | max` | the vocabulary `src/shared/effort.ts` mirrors, since widened by `xhigh` | +| `0.2.113` | spawns a native Claude Code binary through a per-platform optional dependency instead of bundled JavaScript | the packaging weight in section 5 | +| `0.2.113` | **Breaking**: `options.env` replaces `process.env` again, after `0.2.111` made it overlay | see the environment note below | +| `0.2.136`, `0.3.142` | `TodoWrite` deprecated, then headless and SDK sessions use the `Task` tools | the background-task family the registry and `TASK_TOOLS` already name | +| `0.3.144` | `extract` export for `bun build --compile` consumers | not used; the app ships `resources/bin` instead | +| `0.3.149` | fixed `options.env` dropping `CLAUDE_AGENT_SDK_VERSION` when a custom environment is supplied | the app always supplies one, so its `User-Agent` and telemetry were incomplete before this | +| `0.3.178` | spawn failures on an existing native binary explain a libc mismatch and point at `pathToClaudeCodeExecutable` | the error text quoted in section 5 | +| `0.3.179` | optional `tool_use_meta` sidecar with display-friendly names for tool calls | a follow-up for the registry, which hand-writes the titles the CLI can now supply | +| `0.3.193` | `promptSuggestions` option | the option the preference now sends | +| `0.3.214` | `effortLevel` accepts `max` in the TypeScript type | the fifth level in the shared vocabulary | +| `0.3.234` | `SDKSystemMessage` gains an optional `effort` field | the applied effort, readable on init if a surface ever wants to show it | +| `0.3.277` | `updateSettings()` gains a `userSettings` source accepting only `effortLevel` | a per-session effort write path, not used | + +The environment note. `options.env` flipped twice inside this range: `0.2.111` +made it overlay the inherited `process.env`, and `0.2.113` made it replace it +again. The old pin `0.2.45` predates both, so the semantics the app runs under +are the same before and after the bump, and the two flips cancel out. That +matters because `claude.ts` deliberately deletes ambient `ANTHROPIC_API_KEY`, +`ANTHROPIC_AUTH_TOKEN` and `ANTHROPIC_BASE_URL` when a Claude Code OAuth token +exists, and `buildClaudeEnv` strips its own sensitive-key list: under overlay +semantics both deletions would have been undone by the SDK merging +`process.env` back in. Replace semantics is also what the app already assumes, +since `buildClaudeEnv` hands the SDK a complete environment, merging the shell +environment and `process.env` and ensuring `HOME`, `USER`, `TERM`, `SHELL` and +`PATH` are present. + +## 5. Weight, and the packaging decision the pin forces + +The SDK now ships its compiled CLI as eight optional per-platform packages. +`manifest.json` sizes: darwin-arm64 197.9 MB, darwin-x64 206.3 MB, linux-arm64 +213.5 MB, linux-x64 213.6 MB, linux-arm64-musl 206.5 MB, linux-x64-musl +207.8 MB, win32-x64 216.5 MB, win32-arm64 208.1 MB. A Linux dev install holds two +of them, 214 MB and 208 MB, inside a 440 MB `@anthropic-ai` directory and a +1.9 GB `node_modules`. + +electron-builder copies production `node_modules` into the app even with an +explicit `build.files` list, so without a change every packaged target would have +grown by its own platform binary. `build.files` now carries +`"!node_modules/@anthropic-ai/claude-agent-sdk-*"`, which excludes the platform +packages and keeps the 5.0 MB JS SDK, whose directory the glob does not match. + +The exclusion is safe because the app never asks the SDK for that binary: every +query passes `pathToClaudeCodeExecutable`, resolved from `resources/bin` through +`extraResources`, at the single call site in +`src/main/lib/trpc/routers/claude.ts`. The SDK's own failure text for a missing +override is "specify a valid path with `options.pathToClaudeCodeExecutable`", and +its resolver for the native packages is a lazy `existsSync` lookup over candidate +paths, so an absent package is a miss rather than a crash. `package:linux` cannot +run in this sandbox, so CI is the verifier for the artifact size. + +## 6. Dependencies the bump moved + +The lockfile went from 2888 rows to 2960, with 16 package rows added and 16 +removed by the pin bump itself. In: the eight SDK platform packages, +`@anthropic-ai/sdk@0.128.0` with `@babel/runtime`, `@stablelib/base64`, +`fast-sha256`, `json-schema-to-ts`, `standardwebhooks` and `ts-algebra`. Out: the +fifteen `@img/sharp-*` packages that only the old SDK pulled in as optional +dependencies. + +`@anthropic-ai/sdk` is a new transitive dependency, which §8 of the step file says +to ask about first. The human's answer was to take liberties and record them. The +record: it is 14 MB on disk, nothing in `src/` imports it, and `types.ts` reaches +its block types through the SDK's own message types by indexed access rather than +importing it, so the impact on this app is type surface rather than shipped code. +It is not pinned because the app does not depend on it directly, and §8 forbids an +`npm overrides` block to force one. + +The `@img/sharp-*` loss is the evidence for step 5c. `scripts/generate-icon.mjs` +imports `sharp` at line 21 and `package.json` declared nothing, so after the bump +a fresh install had no `sharp` at all and the icon script could not run. It is now +an exact `devDependency`, `0.35.4`, verified by importing it in this checkout, +which resolves and reports libvips 8.18.6. It is a build-time tool and never +ships. + +`@dnd-kit/core@6.3.1`, `@dnd-kit/sortable@10.0.0` and `@dnd-kit/utilities@3.2.2` +arrive in their own commit with exact pins and no importer yet, which is the point +of the ratified exception: steps 17, 18 and 37 share one `DndContext` at the +agents layout root, and this is the only step allowed to touch `package.json` and +`bun.lock`. They cost 2.1 MB installed plus the transitive +`@dnd-kit/accessibility@3.1.1` and `tslib`. + +## 7. Capability profile + +`featureFlagsSchema` gained `effort`, `adaptiveThinking` and `promptSuggestions`, +declared for all ten backends and true only where a turn can actually carry the +value end to end: effort on Claude and Codex, adaptive thinking and prompt +suggestions on Claude. The picker reads the profile through +`trpc.providers.get` rather than a provider name hardcoded in the renderer, so a +backend that reports no effort control shows no sub-menu, and the Settings +backends tab renders the three new facts without a change of its own. + +## 8. Known gaps and follow-ups + +- Bundle bytes per target with and without the bump, with and without `@dnd-kit`, + and turn latency for one scripted prompt: all need `bun run build` at a 4 GB + heap and a packaged app, and this sandbox has 2 CPUs and 3 GB RAM. Not run, and + recorded as not run in the benchmark file. +- A captured transcript showing `tool-Agent` and `tool-TaskOutput` rendering, + which the acceptance criterion asks for and which needs a live session against + a provider. The registry entries have unit tests over fixtures instead. +- Per-model effort discovery from `ModelInfo.supportedEffortLevels`, which retires + the static five-level list. +- `rate_limit_event` is classified internal. A usage or rate-limit panel is the + consumer that would change that, and until one exists the retry toast is the + only rate-limit signal in the UI. +- `permission_denied` is classified internal because the gate already reports a + denial in chat; a second, differently-shaped report of the same event is the + thing to design, not to assume. +- `tool_use_meta` (`0.3.179`) carries display-friendly names for tool calls, + which is the CLI's own answer to part of what `agent-tool-registry.tsx` + hand-writes. Worth reading before the registry grows another entry. +- `BUILTIN_MCP_TOOLS` duplicated pairs at `agent-tool-registry.tsx:614`, step 28. +- Codex schema drift and parity verification this bump makes necessary, step 35. +- The two download scripts, `bun run build` and `package:linux` as the integrity, + type and size verifiers, on CI. From eb4a2bf1cad81b578ecde87d5c97c783091adff4 Mon Sep 17 00:00:00 2001 From: Owie6789 <151057755+Owie6789@users.noreply.github.com> Date: Wed, 23 Sep 2026 13:48:41 +0000 Subject: [PATCH 09/60] Keep a prompt suggestion inside the turn that produced it Two review findings on the suggestion row were both real. A suggestion stayed in the sub-chat atom after the next turn started, so the composer kept offering a previous request's next step and clicking it inserted that into the new prompt. And the transport stored every arriving suggestion by sub-chat id alone, so a late one from an aborted or older run in the same sub-chat could overwrite the current turn's. Starting a turn now clears the atom, and a suggestion is dropped unless its session id is the one this stream reported in its own metadata. The metadata arrives on the result message and the suggestion after it, so the session is always known by the time one lands; a stream that never reports one stores the suggestion rather than silently losing it. The subscription types that metadata as unknown, so it is read through a predicate instead of a cast at the use site. Co-authored-by: arena-agent <297053741+arena-agent@users.noreply.github.com> --- .../features/agents/lib/ipc-chat-transport.ts | 29 ++++++++++++++++++- 1 file changed, 28 insertions(+), 1 deletion(-) diff --git a/src/renderer/features/agents/lib/ipc-chat-transport.ts b/src/renderer/features/agents/lib/ipc-chat-transport.ts index a4a55c94..a3fc2bb5 100644 --- a/src/renderer/features/agents/lib/ipc-chat-transport.ts +++ b/src/renderer/features/agents/lib/ipc-chat-transport.ts @@ -134,6 +134,16 @@ type ImageAttachment = { filename?: string } +/** The session id off a `message-metadata` chunk, whose payload the subscription + * types as unknown. */ +function hasSessionId(value: unknown): value is { sessionId: string } { + return ( + typeof value === "object" && + value !== null && + typeof (value as { sessionId?: unknown }).sessionId === "string" + ) +} + export class IPCChatTransport implements ChatTransport { constructor(private config: IPCChatTransportConfig) {} @@ -186,10 +196,19 @@ export class IPCChatTransport implements ChatTransport { .allSubChats.find((subChat) => subChat.id === this.config.subChatId)?.mode || this.config.mode + // A suggestion belongs to the turn that produced it, so starting a turn + // clears the last one: the composer must not offer a previous request's next + // step, and clicking it must not insert that into this prompt. + appStore.set(subChatPromptSuggestionAtomFamily(this.config.subChatId), null) + // Stream tracking const subId = this.config.subChatId.slice(-8) let _chunkCount = 0 let _lastChunkType = "" + // The session this stream belongs to, learned from its own metadata, so a + // suggestion from an aborted or older run in the same sub-chat is dropped + // instead of overwriting the current turn's. + let streamSessionId: string | null = null return new ReadableStream({ start: (controller) => { @@ -349,11 +368,19 @@ export class IPCChatTransport implements ChatTransport { return } - // Handle retry notification - show friendly toast instead of scary error + // Learn the session before the suggestion that follows it, then + // fall through: this chunk still belongs to the AI SDK. The + // subscription's chunk type carries the metadata as unknown, so it + // is read through a predicate rather than a cast at the use site. + if (chunk.type === "message-metadata" && hasSessionId(chunk.messageMetadata)) { + streamSessionId = chunk.messageMetadata.sessionId + } + // A suggestion is not part of the assistant message, so it goes // to the composer atom for this sub-chat and is never enqueued as // a stream chunk the AI SDK would not recognize. if (chunk.type === "prompt-suggestion") { + if (streamSessionId && chunk.sessionId !== streamSessionId) return appStore.set( subChatPromptSuggestionAtomFamily(this.config.subChatId), chunk.suggestion, From df4b974f6f2e3c8551e91cd1027d14c77500aebc Mon Sep 17 00:00:00 2001 From: Owie6789 <151057755+Owie6789@users.noreply.github.com> Date: Wed, 23 Sep 2026 14:57:46 +0000 Subject: [PATCH 10/60] Register every name the pinned CLI folds into TaskOutput and TaskStop The registry carried two of the six legacy spellings while its own comment named all six, and CodeAnt caught the gap on review. A transcript persisted before the bump can hold KillBash, BashOutputTool, AgentOutput or AgentOutputTool, and with no entry each one fell through to the generic row that prints the raw tool name instead of "Got output" or "Stopped shell". The names are not guessed. The pinned SDK bundle carries the CLI's own normalization table verbatim: Task to Agent, KillShell and KillBash to TaskStop, and BashOutput, BashOutputTool, AgentOutput and AgentOutputTool to TaskOutput. Each alias here therefore maps to the meta of the name it normalizes to. KillBash keeps the shell wording, because what that name killed was a background bash, while TaskStop can stop a sub-agent. The test lists the aliases itself rather than deriving them from the registry, so dropping a key fails instead of quietly shrinking the assertion. Co-authored-by: arena-agent <297053741+arena-agent@users.noreply.github.com> --- .../agents/ui/agent-tool-registry.test.ts | 17 +++++++++++++++++ .../features/agents/ui/agent-tool-registry.tsx | 10 ++++++++-- 2 files changed, 25 insertions(+), 2 deletions(-) diff --git a/src/renderer/features/agents/ui/agent-tool-registry.test.ts b/src/renderer/features/agents/ui/agent-tool-registry.test.ts index 2683e3a2..41ce7c4c 100644 --- a/src/renderer/features/agents/ui/agent-tool-registry.test.ts +++ b/src/renderer/features/agents/ui/agent-tool-registry.test.ts @@ -56,6 +56,23 @@ describe("agent tool registry: renamed sub-agent and background task tools", () expect(AgentToolRegistry["tool-KillShell"]).toBeDefined() }) + // The pinned CLI's normalization table folds these six names into the two it + // emits, so each one has to reach the same meta or a persisted call renders as + // an unnamed generic row. Listed here rather than derived from the registry: + // dropping a key from the registry must fail this test, not shrink it. + it.each(["tool-BashOutput", "tool-BashOutputTool", "tool-AgentOutput", "tool-AgentOutputTool"])( + "routes %s to the TaskOutput meta", + (alias) => { + expect(AgentToolRegistry[alias]).toBe(AgentToolRegistry["tool-TaskOutput"]) + }, + ) + + it.each(["tool-KillShell", "tool-KillBash"])("routes %s to the shell-stopping meta", (alias) => { + expect(AgentToolRegistry[alias]).toBe(AgentToolRegistry["tool-KillShell"]) + // The current name says task, because what it stops may be a sub-agent. + expect(AgentToolRegistry[alias]).not.toBe(AgentToolRegistry["tool-TaskStop"]) + }) + it("titles a sub-agent by state", () => { const meta = AgentToolRegistry["tool-Agent"] expect(meta.title(streamingSubagent)).toBe("Preparing agent") diff --git a/src/renderer/features/agents/ui/agent-tool-registry.tsx b/src/renderer/features/agents/ui/agent-tool-registry.tsx index 4057fa80..f224c859 100644 --- a/src/renderer/features/agents/ui/agent-tool-registry.tsx +++ b/src/renderer/features/agents/ui/agent-tool-registry.tsx @@ -554,12 +554,18 @@ export const AgentToolRegistry: Record = { }, }, - // Shell and background task management. The first name in each pair is the one - // the pinned CLI emits, the second the one older transcripts carry. + // Shell and background task management. The pinned CLI emits `TaskOutput` and + // `TaskStop`; every other key here is a name its own normalization table folds + // into one of those two, kept because a transcript persisted before the bump + // still carries the old spelling and would otherwise render as a generic row. "tool-TaskOutput": backgroundOutputTool, "tool-BashOutput": backgroundOutputTool, + "tool-BashOutputTool": backgroundOutputTool, + "tool-AgentOutput": backgroundOutputTool, + "tool-AgentOutputTool": backgroundOutputTool, "tool-TaskStop": stopTaskTool, "tool-KillShell": stopShellTool, + "tool-KillBash": stopShellTool, // Note: ListMcpResources, ReadMcpResource and their "Tool"-suffixed variants // are handled by AgentMcpToolCall via parseMcpToolType() for richer output display From 57e8cd3316fd5c0b86dc7c7875d284c88a82ce7a Mon Sep 17 00:00:00 2001 From: Owie6789 <151057755+Owie6789@users.noreply.github.com> Date: Wed, 23 Sep 2026 14:58:06 +0000 Subject: [PATCH 11/60] Hoist the prompt-suggestion generator out of the transformer closure SonarCloud flagged the nested generator on new code, and it is right on the merits: handlePromptSuggestion reads only the message it is handed, so holding a slot in createTransformer's closure rebuilt it for every stream and set it apart from the module-scope helpers that already build the retry line and the tool-result text. It now sits with them. Behaviour is unchanged and the transform's 18 tests, which cover the suggestion path, still pass. Co-authored-by: arena-agent <297053741+arena-agent@users.noreply.github.com> --- src/main/lib/claude/transform.ts | 28 +++++++++++++++------------- 1 file changed, 15 insertions(+), 13 deletions(-) diff --git a/src/main/lib/claude/transform.ts b/src/main/lib/claude/transform.ts index 3618df48..5a184a83 100644 --- a/src/main/lib/claude/transform.ts +++ b/src/main/lib/claude/transform.ts @@ -179,6 +179,21 @@ function apiRetryMessage( return `Claude API retry: ${reason}${status}, attempt ${msg.attempt} of ${msg.max_retries}, waiting ${waitSeconds}s` } +/** + * A suggested next prompt, asked for with `Options.promptSuggestions` and sent + * after the result message. Bounded here because the string is + * provider-authored and crosses IPC into the composer. Module scope: it reads + * only its own message, so holding a slot in the transformer closure would just + * re-create it per stream. + */ +function* handlePromptSuggestion( + msg: Extract, +): Generator { + const suggestion = msg.suggestion.trim().slice(0, 2000) + if (!suggestion) return + yield { type: "prompt-suggestion", suggestion, sessionId: msg.session_id } +} + /** Resolve a tool result's output payload, preferring the CLI's own result. */ function resolveToolResultOutput( block: ClaudeToolResultBlock, @@ -763,19 +778,6 @@ export function createTransformer(options?: { isUsingOllama?: boolean }) { } } - /** - * A suggested next prompt, asked for with `Options.promptSuggestions` and sent - * after the result message. Bounded here because the string is - * provider-authored and crosses IPC into the composer. - */ - function* handlePromptSuggestion( - msg: Extract, - ): Generator { - const suggestion = msg.suggestion.trim().slice(0, 2000) - if (!suggestion) return - yield { type: "prompt-suggestion", suggestion, sessionId: msg.session_id } - } - function* handleResultMessage(msg: SDKResultMessage): Generator { // ===== RESULT (final) ===== currentParentToolUseId = null From 3306846a880c4bde93f9ca5aee3e82b05014e0fd Mon Sep 17 00:00:00 2001 From: Owie6789 <151057755+Owie6789@users.noreply.github.com> Date: Wed, 23 Sep 2026 14:58:06 +0000 Subject: [PATCH 12/60] Mark the effort sub-menu props read-only SonarCloud's read-only props rule fired on the one component this step added. The sub-menu never writes to its props: it renders the levels it is handed and calls back with the one the user picked, so the type now says that instead of leaving a reader to check the body for it. Co-authored-by: arena-agent <297053741+arena-agent@users.noreply.github.com> --- .../features/agents/components/agent-model-selector.tsx | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/src/renderer/features/agents/components/agent-model-selector.tsx b/src/renderer/features/agents/components/agent-model-selector.tsx index ef515eba..c0ad5ca1 100644 --- a/src/renderer/features/agents/components/agent-model-selector.tsx +++ b/src/renderer/features/agents/components/agent-model-selector.tsx @@ -246,7 +246,7 @@ function EffortSubMenu({ selectedThinking, onSelectThinking, onClear, -}: { +}: Readonly<{ thinkings: readonly TLevel[] selectedThinking: TLevel | null onSelectThinking: (thinking: TLevel) => void @@ -257,7 +257,7 @@ function EffortSubMenu({ * so it passes nothing and the row does not appear. */ onClear?: () => void -}) { +}>) { const triggerRef = useRef(null) const subMenuRef = useRef(null) const [showSub, setShowSub] = useState(false) From 685aebdd030cfd184581ea1645c278240c597568 Mon Sep 17 00:00:00 2001 From: Owie6789 <151057755+Owie6789@users.noreply.github.com> Date: Wed, 23 Sep 2026 14:58:32 +0000 Subject: [PATCH 13/60] State the new turn-control default once instead of per backend The pin added three turn-shaping features to the capability manifest, and ten backends grew the same six lines for it: a three-line comment pointing at the research record, then effort, adaptiveThinking and promptSuggestions set to false. Eight of the ten were byte-identical, and that copy-paste is what SonarCloud's duplication measure on new code was reading when it failed this PR's quality gate at 4.2% against a 3% limit. TURN_CONTROLS_OFF now holds the "when in doubt, false" answer in the module that owns the rule, and a manifest spreads it and names only what its own backend carries end to end: Codex overrides effort, Claude turns all three on and so inherits nothing. A fourth flag added to the schema now fails typecheck in one place instead of being silently missing from whichever manifest someone forgot, and the evidence comment lives where the doctrine does. Co-authored-by: arena-agent <297053741+arena-agent@users.noreply.github.com> --- src/main/lib/providers/claude.ts | 5 ++--- src/main/lib/providers/cline.ts | 9 ++------- src/main/lib/providers/codex.ts | 10 ++++------ src/main/lib/providers/cursor.ts | 9 ++------- src/main/lib/providers/grok.ts | 9 ++------- src/main/lib/providers/hermes.ts | 9 ++------- src/main/lib/providers/openclaw.ts | 9 ++------- src/main/lib/providers/opencode.ts | 9 ++------- src/main/lib/providers/qwen.ts | 9 ++------- src/main/lib/providers/roo.ts | 9 ++------- src/shared/provider-capabilities.ts | 21 +++++++++++++++++++++ 11 files changed, 43 insertions(+), 65 deletions(-) diff --git a/src/main/lib/providers/claude.ts b/src/main/lib/providers/claude.ts index aeaf9fa7..b4c1a5cf 100644 --- a/src/main/lib/providers/claude.ts +++ b/src/main/lib/providers/claude.ts @@ -101,9 +101,8 @@ export function getClaudeCapability(): ProviderCapability { skills: true, structuredOutput: false, fileCheckpointing: false, - // True only where a turn can actually carry the value end to end; the - // evidence per backend is in - // `.dump/app/research/2026-09-13-sdk-0-3-bump.md`. + // All three on: the 0.3.270 pin carries `Options.effort`, adaptive + // thinking and `Options.promptSuggestions` through a turn end to end. effort: true, adaptiveThinking: true, promptSuggestions: true, diff --git a/src/main/lib/providers/cline.ts b/src/main/lib/providers/cline.ts index 4d86f468..3e7aa6c9 100644 --- a/src/main/lib/providers/cline.ts +++ b/src/main/lib/providers/cline.ts @@ -1,5 +1,5 @@ import { execFile } from "node:child_process" -import type { ProviderCapability } from "../../../shared/provider-capabilities" +import { type ProviderCapability, TURN_CONTROLS_OFF } from "../../../shared/provider-capabilities" import { resolveClineCliLaunch } from "../cline-binary" import { probeClineStoredAuth } from "../cline-print/auth-config" import type { BackendProbe } from "./types" @@ -70,12 +70,7 @@ export function getClineCapability(): ProviderCapability { skills: true, structuredOutput: false, fileCheckpointing: false, - // True only where a turn can actually carry the value end to end; the - // evidence per backend is in - // `.dump/app/research/2026-09-13-sdk-0-3-bump.md`. - effort: false, - adaptiveThinking: false, - promptSuggestions: false, + ...TURN_CONTROLS_OFF, }, notes: [ "Images travel as prompt path references the agent reads via tools.", diff --git a/src/main/lib/providers/codex.ts b/src/main/lib/providers/codex.ts index fbe77d08..16a768ac 100644 --- a/src/main/lib/providers/codex.ts +++ b/src/main/lib/providers/codex.ts @@ -1,7 +1,7 @@ import { execFile } from "node:child_process" import { join } from "node:path" import { app } from "electron" -import type { ProviderCapability } from "../../../shared/provider-capabilities" +import { type ProviderCapability, TURN_CONTROLS_OFF } from "../../../shared/provider-capabilities" import type { BackendProbe } from "./types" function resolveCodexBinary(): string { @@ -72,12 +72,10 @@ export function getCodexCapability(): ProviderCapability { skills: false, structuredOutput: false, fileCheckpointing: false, - // True only where a turn can actually carry the value end to end; the - // evidence per backend is in - // `.dump/app/research/2026-09-13-sdk-0-3-bump.md`. + ...TURN_CONTROLS_OFF, + // The app-server takes a reasoning effort on a turn; Codex chooses its + // own thinking budget and sends no prompt suggestion. effort: true, - adaptiveThinking: false, - promptSuggestions: false, }, notes: [ "Approvals auto-grant session-wide (parity with the former ACP path).", diff --git a/src/main/lib/providers/cursor.ts b/src/main/lib/providers/cursor.ts index 78fd1d47..731fbd96 100644 --- a/src/main/lib/providers/cursor.ts +++ b/src/main/lib/providers/cursor.ts @@ -1,5 +1,5 @@ import { execFile } from "node:child_process" -import type { ProviderCapability } from "../../../shared/provider-capabilities" +import { type ProviderCapability, TURN_CONTROLS_OFF } from "../../../shared/provider-capabilities" import { resolveCursorAgentCliLaunch } from "../cursor-agent-binary" import type { BackendProbe } from "./types" @@ -62,12 +62,7 @@ export function getCursorCapability(): ProviderCapability { skills: false, structuredOutput: false, fileCheckpointing: false, - // True only where a turn can actually carry the value end to end; the - // evidence per backend is in - // `.dump/app/research/2026-09-13-sdk-0-3-bump.md`. - effort: false, - adaptiveThinking: false, - promptSuggestions: false, + ...TURN_CONTROLS_OFF, }, notes: [ "Images travel as prompt path references the agent reads via tools.", diff --git a/src/main/lib/providers/grok.ts b/src/main/lib/providers/grok.ts index 3e9ea8a2..da8f99e7 100644 --- a/src/main/lib/providers/grok.ts +++ b/src/main/lib/providers/grok.ts @@ -1,7 +1,7 @@ import { execFile } from "node:child_process" import { existsSync, readFileSync } from "node:fs" import { join } from "node:path" -import type { ProviderCapability } from "../../../shared/provider-capabilities" +import { type ProviderCapability, TURN_CONTROLS_OFF } from "../../../shared/provider-capabilities" import { resolveGrokCliLaunch, resolveGrokHome } from "../grok-binary" import type { BackendProbe } from "./types" @@ -71,12 +71,7 @@ export function getGrokCapability(): ProviderCapability { skills: false, structuredOutput: false, fileCheckpointing: false, - // True only where a turn can actually carry the value end to end; the - // evidence per backend is in - // `.dump/app/research/2026-09-13-sdk-0-3-bump.md`. - effort: false, - adaptiveThinking: false, - promptSuggestions: false, + ...TURN_CONTROLS_OFF, }, notes: [ "Images travel as prompt path references the agent reads via tools.", diff --git a/src/main/lib/providers/hermes.ts b/src/main/lib/providers/hermes.ts index cbd388b7..859e7097 100644 --- a/src/main/lib/providers/hermes.ts +++ b/src/main/lib/providers/hermes.ts @@ -1,5 +1,5 @@ import { execFile } from "node:child_process" -import type { ProviderCapability } from "../../../shared/provider-capabilities" +import { type ProviderCapability, TURN_CONTROLS_OFF } from "../../../shared/provider-capabilities" import type { BackendProbe } from "./types" function runBinary( @@ -58,12 +58,7 @@ export function getHermesCapability(): ProviderCapability { skills: true, structuredOutput: false, fileCheckpointing: false, - // True only where a turn can actually carry the value end to end; the - // evidence per backend is in - // `.dump/app/research/2026-09-13-sdk-0-3-bump.md`. - effort: false, - adaptiveThinking: false, - promptSuggestions: false, + ...TURN_CONTROLS_OFF, }, notes: [ "ACP sessions live in the running server process; resume across restarts is best-effort.", diff --git a/src/main/lib/providers/openclaw.ts b/src/main/lib/providers/openclaw.ts index 22893a83..8ceb6379 100644 --- a/src/main/lib/providers/openclaw.ts +++ b/src/main/lib/providers/openclaw.ts @@ -1,5 +1,5 @@ import { execFile } from "node:child_process" -import type { ProviderCapability } from "../../../shared/provider-capabilities" +import { type ProviderCapability, TURN_CONTROLS_OFF } from "../../../shared/provider-capabilities" import { resolveOpenclawCliLaunch } from "../openclaw-binary" import { readOpenclawModelsStatus, summarizeModelsStatusAuth } from "../openclaw-print/auth-config" import type { BackendProbe } from "./types" @@ -72,12 +72,7 @@ export function getOpenclawCapability(): ProviderCapability { skills: false, structuredOutput: false, fileCheckpointing: false, - // True only where a turn can actually carry the value end to end; the - // evidence per backend is in - // `.dump/app/research/2026-09-13-sdk-0-3-bump.md`. - effort: false, - adaptiveThinking: false, - promptSuggestions: false, + ...TURN_CONTROLS_OFF, }, notes: [ "One JSON envelope per turn — no streaming; progress appears only when the turn settles.", diff --git a/src/main/lib/providers/opencode.ts b/src/main/lib/providers/opencode.ts index 83cbdd12..bdf4cca5 100644 --- a/src/main/lib/providers/opencode.ts +++ b/src/main/lib/providers/opencode.ts @@ -1,5 +1,5 @@ import { execFile } from "node:child_process" -import type { ProviderCapability } from "../../../shared/provider-capabilities" +import { type ProviderCapability, TURN_CONTROLS_OFF } from "../../../shared/provider-capabilities" import type { BackendProbe } from "./types" function runBinary( @@ -58,12 +58,7 @@ export function getOpencodeCapability(): ProviderCapability { skills: false, structuredOutput: false, fileCheckpointing: false, - // True only where a turn can actually carry the value end to end; the - // evidence per backend is in - // `.dump/app/research/2026-09-13-sdk-0-3-bump.md`. - effort: false, - adaptiveThinking: false, - promptSuggestions: false, + ...TURN_CONTROLS_OFF, }, notes: [ "Permissions auto-reply session-wide; opencode.json can tighten per-tool policy.", diff --git a/src/main/lib/providers/qwen.ts b/src/main/lib/providers/qwen.ts index 0daa3e6c..fe0ebe71 100644 --- a/src/main/lib/providers/qwen.ts +++ b/src/main/lib/providers/qwen.ts @@ -1,5 +1,5 @@ import { execFile } from "node:child_process" -import type { ProviderCapability } from "../../../shared/provider-capabilities" +import { type ProviderCapability, TURN_CONTROLS_OFF } from "../../../shared/provider-capabilities" import { resolveQwenCliLaunch } from "../qwen-binary" import { probeQwenStoredAuth } from "../qwen-print/auth-config" import type { BackendProbe } from "./types" @@ -69,12 +69,7 @@ export function getQwenCapability(): ProviderCapability { skills: true, structuredOutput: false, fileCheckpointing: false, - // True only where a turn can actually carry the value end to end; the - // evidence per backend is in - // `.dump/app/research/2026-09-13-sdk-0-3-bump.md`. - effort: false, - adaptiveThinking: false, - promptSuggestions: false, + ...TURN_CONTROLS_OFF, }, notes: [ "Images travel as prompt path references the agent reads via tools.", diff --git a/src/main/lib/providers/roo.ts b/src/main/lib/providers/roo.ts index 2a5ccf68..e3e330e7 100644 --- a/src/main/lib/providers/roo.ts +++ b/src/main/lib/providers/roo.ts @@ -1,5 +1,5 @@ import { execFile } from "node:child_process" -import type { ProviderCapability } from "../../../shared/provider-capabilities" +import { type ProviderCapability, TURN_CONTROLS_OFF } from "../../../shared/provider-capabilities" import { getClaudeShellEnvironment } from "../claude/env" import { resolveRooCliLaunch } from "../roo-binary" import { resolveRooAmbientAuth } from "../roo-print/auth-config" @@ -69,12 +69,7 @@ export function getRooCapability(): ProviderCapability { skills: false, structuredOutput: false, fileCheckpointing: false, - // True only where a turn can actually carry the value end to end; the - // evidence per backend is in - // `.dump/app/research/2026-09-13-sdk-0-3-bump.md`. - effort: false, - adaptiveThinking: false, - promptSuggestions: false, + ...TURN_CONTROLS_OFF, }, notes: [ "Streams NDJSON events per turn (text deltas, thinking, tool calls, command output, cost).", diff --git a/src/shared/provider-capabilities.ts b/src/shared/provider-capabilities.ts index efd2a44a..d0ff0a52 100644 --- a/src/shared/provider-capabilities.ts +++ b/src/shared/provider-capabilities.ts @@ -99,6 +99,27 @@ export type ProviderCapability = z.infer /** The two answers to "who enforces the permission floor of roadmap step 10". */ export type PermissionFloor = ProviderCapability["security"]["permissionFloor"] +/** The three turn-shaping features the 0.3.270 SDK pin made expressible. */ +export type TurnControlFeatures = Pick< + ProviderCapability["features"], + "effort" | "adaptiveThinking" | "promptSuggestions" +> + +/** + * The "when in doubt, false" rule above, stated once for those three features. + * A manifest spreads this and then names only what its own backend carries end + * to end, so eight backends that support none of them cannot drift to eight + * hand-written copies of the same default, and a fourth flag added to the + * schema fails typecheck here rather than being silently missing from one + * manifest. The per-backend evidence is in + * `.dump/app/research/2026-09-13-sdk-0-3-bump.md`. + */ +export const TURN_CONTROLS_OFF: TurnControlFeatures = { + effort: false, + adaptiveThinking: false, + promptSuggestions: false, +} + /** * The floor behind a sub-chat provider id, which is the vocabulary the chat UI * holds. The manifests are keyed by backend id and live in main, so the renderer From dc182be5ad348e745434d6c252794269b8dd2663 Mon Sep 17 00:00:00 2001 From: Owie6789 <151057755+Owie6789@users.noreply.github.com> Date: Wed, 23 Sep 2026 14:58:57 +0000 Subject: [PATCH 14/60] Give the transport one handler per chunk side effect onData had grown into a 256-line flat chain at a cognitive complexity of 43, and this step added two more links to it for the session and suggestion handling. SonarCloud reports the function as a new-code failure because the PR touched it, but the shape predates the pin, and leaving it longer than the step found it was the worse option. Each side effect is now its own module-scope function over one ChunkContext, and routeChunk calls them in the order the stream needs them: questions first, so the stale-question clear sees the one just asked, then compaction and session info, then the chunks that end the turn or belong to this app rather than to the AI SDK. A handler returns an outcome, so onData keeps a single enqueue path, and the four copies of "close unless it is already closed" become one closeQuietly shared with onError, onComplete and the abort listener. The error chunk keeps its own two pieces: the log and Sentry report, and the toast copy with the text its copy action hands over, which now receives the category and debug payload the caller already read instead of deciding the UNKNOWN fallback twice. The sequence, the early returns and the set of chunks that reach the SDK are unchanged. Typecheck is clean and the full suite passes at 103 files and 1891 tests. Co-authored-by: arena-agent <297053741+arena-agent@users.noreply.github.com> --- .../features/agents/lib/ipc-chat-transport.ts | 602 ++++++++++-------- 1 file changed, 336 insertions(+), 266 deletions(-) diff --git a/src/renderer/features/agents/lib/ipc-chat-transport.ts b/src/renderer/features/agents/lib/ipc-chat-transport.ts index a3fc2bb5..e28c6bb8 100644 --- a/src/renderer/features/agents/lib/ipc-chat-transport.ts +++ b/src/renderer/features/agents/lib/ipc-chat-transport.ts @@ -144,6 +144,325 @@ function hasSessionId(value: unknown): value is { sessionId: string } { ) } +/** + * What the chunk handlers decided: `enqueue` hands the chunk to the AI SDK, + * `consumed` ends handling for a chunk the SDK has no type for, and `failed` + * means the stream was already errored. + */ +type ChunkOutcome = "enqueue" | "consumed" | "failed" + +/** + * What a handler needs besides the chunk: the ids and the turn this transport + * was built with, plus the session id the stream reports about itself. + */ +type ChunkContext = { + chatId: string + subChatId: string + /** The last 8 characters of the sub-chat id, which is what the stream logs tag. */ + subId: string + cwd: string + mode: AgentMode + prompt: string + images: ImageAttachment[] + /** Written by this stream's own metadata chunk, read by the suggestion after it. */ + sessionId: string | null +} + +type ChunkController = ReadableStreamDefaultController + +/** A question the agent is asking now: pending, and no longer expired. */ +function recordPendingQuestion(chunk: SubscriptionChunk, ctx: ChunkContext): void { + if (chunk.type !== "ask-user-question") return + const newMap = new Map(appStore.get(pendingUserQuestionsAtom)) + newMap.set(ctx.subChatId, { + subChatId: ctx.subChatId, + parentChatId: ctx.chatId, + toolUseId: chunk.toolUseId, + questions: chunk.questions, + }) + appStore.set(pendingUserQuestionsAtom, newMap) + + // Clear any expired question (new question replaces it) + const currentExpired = appStore.get(expiredUserQuestionsAtom) + if (!currentExpired.has(ctx.subChatId)) return + const newExpiredMap = new Map(currentExpired) + newExpiredMap.delete(ctx.subChatId) + appStore.set(expiredUserQuestionsAtom, newExpiredMap) +} + +/** A question that timed out: out of pending, kept on screen as expired. */ +function expirePendingQuestion(chunk: SubscriptionChunk, ctx: ChunkContext): void { + if (chunk.type !== "ask-user-question-timeout") return + const currentMap = appStore.get(pendingUserQuestionsAtom) + const pending = currentMap.get(ctx.subChatId) + if (!pending || pending.toolUseId !== chunk.toolUseId) return + + const newPendingMap = new Map(currentMap) + newPendingMap.delete(ctx.subChatId) + appStore.set(pendingUserQuestionsAtom, newPendingMap) + + // Move to expired (so the UI keeps showing the question) + const newExpiredMap = new Map(appStore.get(expiredUserQuestionsAtom)) + newExpiredMap.set(ctx.subChatId, pending) + appStore.set(expiredUserQuestionsAtom, newExpiredMap) +} + +/** An answer, stored for the real-time updates the question UI reads. */ +function storeQuestionResult(chunk: SubscriptionChunk): void { + if (chunk.type !== "ask-user-question-result") return + const newResults = new Map(appStore.get(askUserQuestionResultsAtom)) + newResults.set(chunk.toolUseId, chunk.result) + appStore.set(askUserQuestionResultsAtom, newResults) +} + +function isCompactingStart(chunk: SubscriptionChunk): boolean { + return ( + (chunk.type === "tool-input-start" || chunk.type === "tool-input-available") && + chunk.toolName === "Compact" + ) +} + +function isCompactingEnd(chunk: SubscriptionChunk): boolean { + return ( + (chunk.type === "tool-output-available" || chunk.type === "tool-output-error") && + Boolean(chunk.toolCallId?.startsWith("compact-")) + ) +} + +/** The compaction state the chat header reads while the CLI rewrites history. */ +function updateCompactingState(chunk: SubscriptionChunk, ctx: ChunkContext): void { + if (!isCompactingStart(chunk) && !isCompactingEnd(chunk)) return + const next = new Set(appStore.get(compactingSubChatsAtom)) + if (isCompactingStart(chunk)) next.add(ctx.subChatId) + else next.delete(ctx.subChatId) + appStore.set(compactingSubChatsAtom, next) +} + +/** What this session opened with: tools, MCP servers, plugins, skills. */ +function recordSessionInfo(chunk: SubscriptionChunk): void { + if (chunk.type !== "session-init") return + appStore.set(sessionInfoAtom, { + tools: chunk.tools, + mcpServers: chunk.mcpServers, + plugins: chunk.plugins, + skills: chunk.skills, + }) +} + +/** + * A pending question goes stale once the agent has moved on, but not while it is + * still building the tool input that asks it, so the question chunks, tool input + * and stream start all keep it. + */ +function clearStaleQuestion(chunk: SubscriptionChunk, ctx: ChunkContext): void { + const shouldClearOnChunk = + chunk.type !== "ask-user-question" && + chunk.type !== "ask-user-question-timeout" && + chunk.type !== "ask-user-question-result" && + !chunk.type.startsWith("tool-input") && // Don't clear while input is being built + chunk.type !== "start" && + chunk.type !== "start-step" + if (!shouldClearOnChunk) return + + // NOTE: Do NOT clear expired questions here. After a timeout, the agent + // continues and emits new chunks — that's expected. Expired questions should + // persist until the user answers, dismisses, or sends a new message. + const currentMap = appStore.get(pendingUserQuestionsAtom) + if (!currentMap.has(ctx.subChatId)) return + const newMap = new Map(currentMap) + newMap.delete(ctx.subChatId) + appStore.set(pendingUserQuestionsAtom, newMap) +} + +/** + * An auth failure keeps this tree's modal-and-retry flow rather than a toast: + * park the turn so the modal can resend it after OAuth, then error the stream + * instead of closing it so the chat leaves "streaming" and the user can retry. + */ +function failTurnForAuth(ctx: ChunkContext, controller: ChunkController): ChunkOutcome { + // Store the failed message for retry after successful auth. + // readyToRetry=false prevents immediate retry; the modal sets it to true. + appStore.set(pendingAuthRetryMessageAtom, { + subChatId: ctx.subChatId, + provider: "claude-code", + prompt: ctx.prompt, + ...(ctx.images.length > 0 && { images: ctx.images }), + readyToRetry: false, + }) + appStore.set(claudeLoginModalConfigAtom, { + hideCustomModelSettingsLink: false, + autoStartAuth: false, + }) + // Show the Claude Code login modal + appStore.set(agentsLoginModalOpenAtom, true) + console.log(`[SD] R:AUTH_ERR sub=${ctx.subId}`) + // controller.error() rather than controller.close(), so the SDK Chat resets + // status from "streaming" to "ready". + controller.error(new Error("Authentication required")) + return "failed" +} + +/** + * A prompt suggestion belongs to the turn that produced it. The session id + * arrives on the metadata chunk the transform emits before the suggestion, so a + * late suggestion from an aborted or older run in the same sub-chat is dropped + * instead of overwriting this turn's. Neither chunk is one the AI SDK knows: + * the suggestion is consumed, the metadata is passed on. + */ +function routePromptSuggestion(chunk: SubscriptionChunk, ctx: ChunkContext): ChunkOutcome { + // Learn the session before the suggestion that follows it. The subscription's + // chunk type carries the metadata as unknown, so it is read through a + // predicate rather than a cast at the use site. + if (chunk.type === "message-metadata" && hasSessionId(chunk.messageMetadata)) { + ctx.sessionId = chunk.messageMetadata.sessionId + } + if (chunk.type !== "prompt-suggestion") return "enqueue" + if (ctx.sessionId && chunk.sessionId !== ctx.sessionId) return "consumed" + appStore.set(subChatPromptSuggestionAtomFamily(ctx.subChatId), chunk.suggestion) + return "consumed" +} + +/** A retry the CLI is already performing: said once, and not a stream chunk. */ +function announceRetry(chunk: SubscriptionChunk): ChunkOutcome { + if (chunk.type !== "retry-notification") return "enqueue" + toast.info("Retrying request", { + description: chunk.message || "Request was unsuccessful, trying again...", + duration: 4000, + }) + return "consumed" +} + +/** + * The copy an error toast shows, and the full text its copy action hands over. + * The category and debug payload come from the caller, which already read them + * for the log and Sentry, so the fallback is not decided twice. + */ +function errorToastCopy( + chunk: Extract, + ctx: ChunkContext, + category: string, + debugInfo: unknown, +): { title: string; description: string; details: string } { + // Available for every error, not only the categories this app recognizes. + const details = [ + `Error: ${chunk.errorText || "Unknown error"}`, + `Category: ${category}`, + `Chat ID: ${ctx.chatId}`, + `SubChat ID: ${ctx.subChatId}`, + `CWD: ${ctx.cwd}`, + `Mode: ${ctx.mode}`, + `Timestamp: ${new Date().toISOString()}`, + debugInfo ? `Debug Info: ${JSON.stringify(debugInfo, null, 2)}` : null, + ] + .filter(Boolean) + .join("\n") + + const config = ERROR_TOAST_CONFIG[category] + // For auth and API key failures the backend's own wording wins: it names the + // credential that failed, which this app's copy cannot. + const prefersBackendError = + category === "AUTH_FAILURE" || + category === "INVALID_API_KEY_SDK" || + category === "INVALID_API_KEY" + const rawDescription = prefersBackendError + ? chunk.errorText || config?.description || "An unexpected error occurred" + : config?.description || chunk.errorText || "An unexpected error occurred" + return { + title: config?.title || "Claude error", + // Truncate long descriptions for the toast (keep the first 300 chars). + description: + rawDescription.length > 300 ? `${rawDescription.slice(0, 300)}...` : rawDescription, + details, + } +} + +/** + * An error chunk is logged, sent to Sentry and toasted, and then still handed to + * the AI SDK: its message part is what the transcript shows afterwards. + */ +function reportErrorChunk(chunk: SubscriptionChunk, ctx: ChunkContext): void { + if (chunk.type !== "error") return + const debugInfo = "debugInfo" in chunk ? chunk.debugInfo : undefined + const category = debugInfo?.category || "UNKNOWN" + + // Detailed SDK error logging for debugging + console.error(`[SDK ERROR] ========================================`) + console.error(`[SDK ERROR] Category: ${category}`) + console.error(`[SDK ERROR] Error text: ${chunk.errorText}`) + console.error(`[SDK ERROR] Chat ID: ${ctx.chatId}`) + console.error(`[SDK ERROR] SubChat ID: ${ctx.subChatId}`) + console.error(`[SDK ERROR] CWD: ${ctx.cwd}`) + console.error(`[SDK ERROR] Mode: ${ctx.mode}`) + if (debugInfo) { + console.error(`[SDK ERROR] Debug info:`, JSON.stringify(debugInfo, null, 2)) + } + console.error(`[SDK ERROR] Full chunk:`, JSON.stringify(chunk, null, 2)) + console.error(`[SDK ERROR] ========================================`) + + Sentry.captureException(new Error(chunk.errorText || "Claude transport error"), { + tags: { errorCategory: category, mode: ctx.mode }, + extra: { debugInfo, cwd: ctx.cwd, chatId: ctx.chatId, subChatId: ctx.subChatId }, + }) + + const { title, description, details } = errorToastCopy(chunk, ctx, category, debugInfo) + toast.error(title, { + description, + duration: 12000, + action: { + label: "Copy Error", + onClick: () => { + navigator.clipboard.writeText(details) + toast.success("Error details copied to clipboard") + }, + }, + }) +} + +/** Enqueue without crashing on a stream that is already closed. */ +function enqueueChunk(controller: ChunkController, chunk: SubscriptionChunk): void { + try { + controller.enqueue(chunk as SDKUIMessageChunk) + } catch { + // Stream already closed, ignore enqueue failure + } +} + +/** Close without crashing on a stream that is already closed. */ +function closeQuietly(controller: ChunkController): void { + try { + controller.close() + } catch { + // Already closed + } +} + +/** + * The side effects a chunk has, in the order the stream needs them: questions + * first, so the stale-question clear sees the one just asked, then compaction + * and session info, then the chunks that end the turn or belong to this app + * rather than to the AI SDK. + */ +function routeChunk( + chunk: SubscriptionChunk, + ctx: ChunkContext, + controller: ChunkController, +): ChunkOutcome { + recordPendingQuestion(chunk, ctx) + expirePendingQuestion(chunk, ctx) + storeQuestionResult(chunk) + updateCompactingState(chunk, ctx) + recordSessionInfo(chunk) + clearStaleQuestion(chunk, ctx) + + if (chunk.type === "auth-error") return failTurnForAuth(ctx, controller) + const suggestion = routePromptSuggestion(chunk, ctx) + if (suggestion !== "enqueue") return suggestion + const retry = announceRetry(chunk) + if (retry !== "enqueue") return retry + reportErrorChunk(chunk, ctx) + return "enqueue" +} + export class IPCChatTransport implements ChatTransport { constructor(private config: IPCChatTransportConfig) {} @@ -202,13 +521,20 @@ export class IPCChatTransport implements ChatTransport { appStore.set(subChatPromptSuggestionAtomFamily(this.config.subChatId), null) // Stream tracking - const subId = this.config.subChatId.slice(-8) let _chunkCount = 0 let _lastChunkType = "" - // The session this stream belongs to, learned from its own metadata, so a - // suggestion from an aborted or older run in the same sub-chat is dropped - // instead of overwriting the current turn's. - let streamSessionId: string | null = null + // One context for the chunk handlers, so the session id this stream reports + // about itself is visible to the suggestion that follows it. + const ctx: ChunkContext = { + chatId: this.config.chatId, + subChatId: this.config.subChatId, + subId: this.config.subChatId.slice(-8), + cwd: this.config.cwd, + mode: currentMode, + prompt, + images, + sessionId: null, + } return new ReadableStream({ start: (controller) => { @@ -237,257 +563,9 @@ export class IPCChatTransport implements ChatTransport { _chunkCount++ _lastChunkType = chunk.type - // Handle AskUserQuestion - show question UI - if (chunk.type === "ask-user-question") { - const currentMap = appStore.get(pendingUserQuestionsAtom) - const newMap = new Map(currentMap) - newMap.set(this.config.subChatId, { - subChatId: this.config.subChatId, - parentChatId: this.config.chatId, - toolUseId: chunk.toolUseId, - questions: chunk.questions, - }) - appStore.set(pendingUserQuestionsAtom, newMap) - - // Clear any expired question (new question replaces it) - const currentExpired = appStore.get(expiredUserQuestionsAtom) - if (currentExpired.has(this.config.subChatId)) { - const newExpiredMap = new Map(currentExpired) - newExpiredMap.delete(this.config.subChatId) - appStore.set(expiredUserQuestionsAtom, newExpiredMap) - } - } - - // Handle AskUserQuestion timeout - move to expired (keep UI visible) - if (chunk.type === "ask-user-question-timeout") { - const currentMap = appStore.get(pendingUserQuestionsAtom) - const pending = currentMap.get(this.config.subChatId) - if (pending && pending.toolUseId === chunk.toolUseId) { - // Remove from pending - const newPendingMap = new Map(currentMap) - newPendingMap.delete(this.config.subChatId) - appStore.set(pendingUserQuestionsAtom, newPendingMap) - - // Move to expired (so UI keeps showing the question) - const currentExpired = appStore.get(expiredUserQuestionsAtom) - const newExpiredMap = new Map(currentExpired) - newExpiredMap.set(this.config.subChatId, pending) - appStore.set(expiredUserQuestionsAtom, newExpiredMap) - } - } - - // Handle AskUserQuestion result - store for real-time updates - if (chunk.type === "ask-user-question-result") { - const currentResults = appStore.get(askUserQuestionResultsAtom) - const newResults = new Map(currentResults) - newResults.set(chunk.toolUseId, chunk.result) - appStore.set(askUserQuestionResultsAtom, newResults) - } - - // Handle compacting status - track in atom for UI display - if ( - (chunk.type === "tool-input-start" && chunk.toolName === "Compact") || - (chunk.type === "tool-input-available" && chunk.toolName === "Compact") - ) { - const compacting = appStore.get(compactingSubChatsAtom) - const newCompacting = new Set(compacting) - // Compacting started - newCompacting.add(this.config.subChatId) - appStore.set(compactingSubChatsAtom, newCompacting) - } - if ( - (chunk.type === "tool-output-available" && - chunk.toolCallId?.startsWith("compact-")) || - (chunk.type === "tool-output-error" && chunk.toolCallId?.startsWith("compact-")) - ) { - const compacting = appStore.get(compactingSubChatsAtom) - const newCompacting = new Set(compacting) - // Compacting finished - newCompacting.delete(this.config.subChatId) - appStore.set(compactingSubChatsAtom, newCompacting) - } - - // Handle session init - store MCP servers, plugins, tools info - if (chunk.type === "session-init") { - appStore.set(sessionInfoAtom, { - tools: chunk.tools, - mcpServers: chunk.mcpServers, - plugins: chunk.plugins, - skills: chunk.skills, - }) - } - - // Clear pending questions ONLY when agent has moved on - // Don't clear on tool-input-* chunks (still building the question input) - // Clear when we get tool-output-* (answer received) or text-delta (agent moved on) - const shouldClearOnChunk = - chunk.type !== "ask-user-question" && - chunk.type !== "ask-user-question-timeout" && - chunk.type !== "ask-user-question-result" && - !chunk.type.startsWith("tool-input") && // Don't clear while input is being built - chunk.type !== "start" && - chunk.type !== "start-step" - - if (shouldClearOnChunk) { - const currentMap = appStore.get(pendingUserQuestionsAtom) - if (currentMap.has(this.config.subChatId)) { - const newMap = new Map(currentMap) - newMap.delete(this.config.subChatId) - appStore.set(pendingUserQuestionsAtom, newMap) - } - // NOTE: Do NOT clear expired questions here. After a timeout, - // the agent continues and emits new chunks — that's expected. - // Expired questions should persist until the user answers, - // dismisses, or sends a new message. - } - - // Handle authentication errors - show Claude login modal - // NOTE (mausCode): kept our modal+retry flow; their toast-only - // replacement was NOT transplanted. - if (chunk.type === "auth-error") { - // Store the failed message for retry after successful auth - // readyToRetry=false prevents immediate retry - modal sets it to true on OAuth success - appStore.set(pendingAuthRetryMessageAtom, { - subChatId: this.config.subChatId, - provider: "claude-code", - prompt, - ...(images.length > 0 && { images }), - readyToRetry: false, - }) - appStore.set(claudeLoginModalConfigAtom, { - hideCustomModelSettingsLink: false, - autoStartAuth: false, - }) - // Show the Claude Code login modal - appStore.set(agentsLoginModalOpenAtom, true) - // Use controller.error() instead of controller.close() so that - // the SDK Chat properly resets status from "streaming" to "ready" - // This allows user to retry sending messages after failed auth - console.log(`[SD] R:AUTH_ERR sub=${subId}`) - controller.error(new Error("Authentication required")) - return - } - - // Learn the session before the suggestion that follows it, then - // fall through: this chunk still belongs to the AI SDK. The - // subscription's chunk type carries the metadata as unknown, so it - // is read through a predicate rather than a cast at the use site. - if (chunk.type === "message-metadata" && hasSessionId(chunk.messageMetadata)) { - streamSessionId = chunk.messageMetadata.sessionId - } - - // A suggestion is not part of the assistant message, so it goes - // to the composer atom for this sub-chat and is never enqueued as - // a stream chunk the AI SDK would not recognize. - if (chunk.type === "prompt-suggestion") { - if (streamSessionId && chunk.sessionId !== streamSessionId) return - appStore.set( - subChatPromptSuggestionAtomFamily(this.config.subChatId), - chunk.suggestion, - ) - return - } - - if (chunk.type === "retry-notification") { - toast.info("Retrying request", { - description: chunk.message || "Request was unsuccessful, trying again...", - duration: 4000, - }) - return // don't enqueue retry-notification as a stream chunk - } - - // Handle errors - show toast to user FIRST before anything else - if (chunk.type === "error") { - const debugInfo = "debugInfo" in chunk ? chunk.debugInfo : undefined - const category = debugInfo?.category || "UNKNOWN" - - // Detailed SDK error logging for debugging - console.error(`[SDK ERROR] ========================================`) - console.error(`[SDK ERROR] Category: ${category}`) - console.error(`[SDK ERROR] Error text: ${chunk.errorText}`) - console.error(`[SDK ERROR] Chat ID: ${this.config.chatId}`) - console.error(`[SDK ERROR] SubChat ID: ${this.config.subChatId}`) - console.error(`[SDK ERROR] CWD: ${this.config.cwd}`) - console.error(`[SDK ERROR] Mode: ${currentMode}`) - if (debugInfo) { - console.error(`[SDK ERROR] Debug info:`, JSON.stringify(debugInfo, null, 2)) - } - console.error(`[SDK ERROR] Full chunk:`, JSON.stringify(chunk, null, 2)) - console.error(`[SDK ERROR] ========================================`) - - // Track error in Sentry - Sentry.captureException(new Error(chunk.errorText || "Claude transport error"), { - tags: { - errorCategory: category, - mode: currentMode, - }, - extra: { - debugInfo: debugInfo, - cwd: this.config.cwd, - chatId: this.config.chatId, - subChatId: this.config.subChatId, - }, - }) - - // Build detailed error string for copying (available for ALL errors) - const errorDetails = [ - `Error: ${chunk.errorText || "Unknown error"}`, - `Category: ${category}`, - `Chat ID: ${this.config.chatId}`, - `SubChat ID: ${this.config.subChatId}`, - `CWD: ${this.config.cwd}`, - `Mode: ${currentMode}`, - `Timestamp: ${new Date().toISOString()}`, - debugInfo ? `Debug Info: ${JSON.stringify(debugInfo, null, 2)}` : null, - ] - .filter(Boolean) - .join("\n") - - // Show toast based on error category - const config = ERROR_TOAST_CONFIG[category] - const title = config?.title || "Claude error" - // For auth/API key failures, prefer original backend error to aid debugging - const preferOriginalError = - category === "AUTH_FAILURE" || - category === "INVALID_API_KEY_SDK" || - category === "INVALID_API_KEY" - // Use config description if set, otherwise fall back to errorText - const rawDescription = preferOriginalError - ? chunk.errorText || config?.description || "An unexpected error occurred" - : config?.description || chunk.errorText || "An unexpected error occurred" - // Truncate long descriptions for toast (keep first 300 chars) - const description = - rawDescription.length > 300 - ? `${rawDescription.slice(0, 300)}...` - : rawDescription - - toast.error(title, { - description, - duration: 12000, - action: { - label: "Copy Error", - onClick: () => { - navigator.clipboard.writeText(errorDetails) - toast.success("Error details copied to clipboard") - }, - }, - }) - } - - // Try to enqueue, but don't crash if stream is already closed - try { - controller.enqueue(chunk as SDKUIMessageChunk) - } catch (_e) { - // Stream already closed, ignore enqueue failure - } - - if (chunk.type === "finish") { - try { - controller.close() - } catch { - // Already closed - } - } + if (routeChunk(chunk, ctx, controller) !== "enqueue") return + enqueueChunk(controller, chunk) + if (chunk.type === "finish") closeQuietly(controller) }, onError: (err: Error) => { // Track transport errors in Sentry @@ -509,11 +587,7 @@ export class IPCChatTransport implements ChatTransport { // Note: Don't clear pending questions here - let active-chat.tsx handle it // via the stream stop detection effect. Clearing here causes race conditions // where sync effect immediately restores from messages. - try { - controller.close() - } catch { - // Already closed - } + closeQuietly(controller) }, }, ) @@ -522,11 +596,7 @@ export class IPCChatTransport implements ChatTransport { options.abortSignal?.addEventListener("abort", () => { sub.unsubscribe() // trpcClient.claude.cancel.mutate({ subChatId: this.config.subChatId }) - try { - controller.close() - } catch { - // Already closed - } + closeQuietly(controller) }) }, }) From fe09b5adaf4d487045bb17b01131d05688aebd48 Mon Sep 17 00:00:00 2001 From: Owie6789 <151057755+Owie6789@users.noreply.github.com> Date: Wed, 23 Sep 2026 15:05:46 +0000 Subject: [PATCH 15/60] Take the chunk helpers that were extracted from this transport chat-chunk-atoms.ts says it was extracted from the Claude IPC transport so the native runtime transport could reuse identical question, compacting and prompt-extraction behaviour, and then this transport kept its own copies: applyQuestionChunks, applyCompactingChunks, clearStalePendingQuestion, extractPromptText and extractPromptImages were called only from native-chat-transport.ts, while this file carried the same logic inline plus two private methods byte-identical to the shared extractors. So the handlers the previous commit lifted out of onData are deleted rather than kept, and routeChunk calls the shared module in the order the native transport already uses. session-init stays here, because the shared module leaves it per transport on purpose: the native runtime reads a cached snapshot and fills the gaps, while the CLI reports the full set on init. ChunkContext now widens the shared ChatChunkContext instead of restating its two ids, the local ImageAttachment type gives way to the exported one, and the transplant note points at where the code lives now. 172 lines go, and the two transports cannot drift on the question lifecycle again. Typecheck is clean and the suite still passes at 103 files and 1891 tests. Co-authored-by: arena-agent <297053741+arena-agent@users.noreply.github.com> --- .../features/agents/lib/ipc-chat-transport.ts | 202 +++--------------- 1 file changed, 30 insertions(+), 172 deletions(-) diff --git a/src/renderer/features/agents/lib/ipc-chat-transport.ts b/src/renderer/features/agents/lib/ipc-chat-transport.ts index e28c6bb8..4b76a1fc 100644 --- a/src/renderer/features/agents/lib/ipc-chat-transport.ts +++ b/src/renderer/features/agents/lib/ipc-chat-transport.ts @@ -1,8 +1,10 @@ /** - * NOTE (transplant): inlined question/compact chunk handling, stale-question - * clearing fix, extractText/extractImages, and log removals were transplanted - * from erenbertr/1code (Apache-2.0). Their auth-error toast replacement was - * NOT taken — this tree keeps the login-modal retry flow. + * NOTE (transplant): the question/compact chunk handling, the stale-question + * clearing fix, the prompt and image extraction, and the log removals were + * transplanted from erenbertr/1code (Apache-2.0). The first three now live in + * `./chat-chunk-atoms`, which both transports share, and the provenance record + * is NOTICE and UPSTREAM.md. Their auth-error toast replacement was NOT taken + * — this tree keeps the login-modal retry flow. */ import * as Sentry from "@sentry/electron/renderer" @@ -28,18 +30,23 @@ import { import { appStore } from "../../../lib/jotai-store" import { trpcClient } from "../../../lib/trpc" import { - askUserQuestionResultsAtom, - compactingSubChatsAtom, - expiredUserQuestionsAtom, MODEL_ID_MAP, pendingAuthRetryMessageAtom, - pendingUserQuestionsAtom, subChatModelIdAtomFamily, subChatPromptSuggestionAtomFamily, } from "../atoms" import { useAgentSubChatStore } from "../stores/sub-chat-store" import type { AgentMessageMetadata } from "../ui/agent-message-usage" -import type { LooseUIPart, SubscriptionChunk } from "./chat-chunk-atoms" +import { + applyCompactingChunks, + applyQuestionChunks, + type ChatChunkContext, + clearStalePendingQuestion, + extractPromptImages, + extractPromptText, + type ImageAttachment, + type SubscriptionChunk, +} from "./chat-chunk-atoms" // Error categories and their user-friendly messages const ERROR_TOAST_CONFIG: Record< @@ -127,13 +134,6 @@ type IPCChatTransportConfig = { model?: string } -// Image attachment type matching the tRPC schema -type ImageAttachment = { - base64Data: string - mediaType: string - filename?: string -} - /** The session id off a `message-metadata` chunk, whose payload the subscription * types as unknown. */ function hasSessionId(value: unknown): value is { sessionId: string } { @@ -152,12 +152,11 @@ function hasSessionId(value: unknown): value is { sessionId: string } { type ChunkOutcome = "enqueue" | "consumed" | "failed" /** - * What a handler needs besides the chunk: the ids and the turn this transport - * was built with, plus the session id the stream reports about itself. + * What a handler needs besides the chunk: the shared question context both + * transports pass, widened with the turn this transport was built with and the + * session id the stream reports about itself. */ -type ChunkContext = { - chatId: string - subChatId: string +type ChunkContext = ChatChunkContext & { /** The last 8 characters of the sub-chat id, which is what the stream logs tag. */ subId: string cwd: string @@ -170,75 +169,11 @@ type ChunkContext = { type ChunkController = ReadableStreamDefaultController -/** A question the agent is asking now: pending, and no longer expired. */ -function recordPendingQuestion(chunk: SubscriptionChunk, ctx: ChunkContext): void { - if (chunk.type !== "ask-user-question") return - const newMap = new Map(appStore.get(pendingUserQuestionsAtom)) - newMap.set(ctx.subChatId, { - subChatId: ctx.subChatId, - parentChatId: ctx.chatId, - toolUseId: chunk.toolUseId, - questions: chunk.questions, - }) - appStore.set(pendingUserQuestionsAtom, newMap) - - // Clear any expired question (new question replaces it) - const currentExpired = appStore.get(expiredUserQuestionsAtom) - if (!currentExpired.has(ctx.subChatId)) return - const newExpiredMap = new Map(currentExpired) - newExpiredMap.delete(ctx.subChatId) - appStore.set(expiredUserQuestionsAtom, newExpiredMap) -} - -/** A question that timed out: out of pending, kept on screen as expired. */ -function expirePendingQuestion(chunk: SubscriptionChunk, ctx: ChunkContext): void { - if (chunk.type !== "ask-user-question-timeout") return - const currentMap = appStore.get(pendingUserQuestionsAtom) - const pending = currentMap.get(ctx.subChatId) - if (!pending || pending.toolUseId !== chunk.toolUseId) return - - const newPendingMap = new Map(currentMap) - newPendingMap.delete(ctx.subChatId) - appStore.set(pendingUserQuestionsAtom, newPendingMap) - - // Move to expired (so the UI keeps showing the question) - const newExpiredMap = new Map(appStore.get(expiredUserQuestionsAtom)) - newExpiredMap.set(ctx.subChatId, pending) - appStore.set(expiredUserQuestionsAtom, newExpiredMap) -} - -/** An answer, stored for the real-time updates the question UI reads. */ -function storeQuestionResult(chunk: SubscriptionChunk): void { - if (chunk.type !== "ask-user-question-result") return - const newResults = new Map(appStore.get(askUserQuestionResultsAtom)) - newResults.set(chunk.toolUseId, chunk.result) - appStore.set(askUserQuestionResultsAtom, newResults) -} - -function isCompactingStart(chunk: SubscriptionChunk): boolean { - return ( - (chunk.type === "tool-input-start" || chunk.type === "tool-input-available") && - chunk.toolName === "Compact" - ) -} - -function isCompactingEnd(chunk: SubscriptionChunk): boolean { - return ( - (chunk.type === "tool-output-available" || chunk.type === "tool-output-error") && - Boolean(chunk.toolCallId?.startsWith("compact-")) - ) -} - -/** The compaction state the chat header reads while the CLI rewrites history. */ -function updateCompactingState(chunk: SubscriptionChunk, ctx: ChunkContext): void { - if (!isCompactingStart(chunk) && !isCompactingEnd(chunk)) return - const next = new Set(appStore.get(compactingSubChatsAtom)) - if (isCompactingStart(chunk)) next.add(ctx.subChatId) - else next.delete(ctx.subChatId) - appStore.set(compactingSubChatsAtom, next) -} - -/** What this session opened with: tools, MCP servers, plugins, skills. */ +/** + * What this session opened with. `chat-chunk-atoms.ts` leaves this one in each + * transport on purpose: the native runtime reads a cached snapshot and fills + * the gaps, while the Claude CLI reports the full set on init. + */ function recordSessionInfo(chunk: SubscriptionChunk): void { if (chunk.type !== "session-init") return appStore.set(sessionInfoAtom, { @@ -248,32 +183,6 @@ function recordSessionInfo(chunk: SubscriptionChunk): void { skills: chunk.skills, }) } - -/** - * A pending question goes stale once the agent has moved on, but not while it is - * still building the tool input that asks it, so the question chunks, tool input - * and stream start all keep it. - */ -function clearStaleQuestion(chunk: SubscriptionChunk, ctx: ChunkContext): void { - const shouldClearOnChunk = - chunk.type !== "ask-user-question" && - chunk.type !== "ask-user-question-timeout" && - chunk.type !== "ask-user-question-result" && - !chunk.type.startsWith("tool-input") && // Don't clear while input is being built - chunk.type !== "start" && - chunk.type !== "start-step" - if (!shouldClearOnChunk) return - - // NOTE: Do NOT clear expired questions here. After a timeout, the agent - // continues and emits new chunks — that's expected. Expired questions should - // persist until the user answers, dismisses, or sends a new message. - const currentMap = appStore.get(pendingUserQuestionsAtom) - if (!currentMap.has(ctx.subChatId)) return - const newMap = new Map(currentMap) - newMap.delete(ctx.subChatId) - appStore.set(pendingUserQuestionsAtom, newMap) -} - /** * An auth failure keeps this tree's modal-and-retry flow rather than a toast: * park the turn so the modal can resend it after OAuth, then error the stream @@ -447,12 +356,10 @@ function routeChunk( ctx: ChunkContext, controller: ChunkController, ): ChunkOutcome { - recordPendingQuestion(chunk, ctx) - expirePendingQuestion(chunk, ctx) - storeQuestionResult(chunk) - updateCompactingState(chunk, ctx) + applyQuestionChunks(chunk, ctx) + applyCompactingChunks(chunk, ctx.subChatId) recordSessionInfo(chunk) - clearStaleQuestion(chunk, ctx) + clearStalePendingQuestion(chunk, ctx.subChatId) if (chunk.type === "auth-error") return failTurnForAuth(ctx, controller) const suggestion = routePromptSuggestion(chunk, ctx) @@ -472,8 +379,8 @@ export class IPCChatTransport implements ChatTransport { }): Promise> { // Extract prompt and images from last user message const lastUser = [...options.messages].reverse().find((m) => m.role === "user") - const prompt = this.extractText(lastUser) - const images = this.extractImages(lastUser) + const prompt = extractPromptText(lastUser) + const images = extractPromptImages(lastUser) // Get sessionId for resume (server preserves sessionId on abort so // the next message can resume with full conversation context) @@ -605,53 +512,4 @@ export class IPCChatTransport implements ChatTransport { async reconnectToStream(): Promise | null> { return null // Not needed for local app } - - private extractText(msg: UIMessage | undefined): string { - if (!msg) return "" - if (msg.parts) { - const textParts: string[] = [] - const fileContents: string[] = [] - - for (const p of msg.parts) { - const part = p as LooseUIPart - if (part.type === "text" && part.text) { - textParts.push(part.text) - } else if (part.type === "file-content") { - // Hidden file content - add to prompt but not displayed in UI - const fileName = part.filePath?.split("/").pop() || part.filePath || "file" - fileContents.push(`\n--- ${fileName} ---\n${part.content}`) - } - } - - // Combine text and file contents - return textParts.join("\n") + fileContents.join("") - } - return "" - } - - /** - * Extract images from message parts - * Looks for parts with type "data-image" that have base64Data - */ - private extractImages(msg: UIMessage | undefined): ImageAttachment[] { - if (!msg?.parts) return [] - - const images: ImageAttachment[] = [] - - for (const part of msg.parts) { - // Check for data-image parts with base64 data - const data = (part as LooseUIPart).data - if (part.type === "data-image" && data) { - if (data.base64Data && data.mediaType) { - images.push({ - base64Data: data.base64Data, - mediaType: data.mediaType, - filename: data.filename, - }) - } - } - } - - return images - } } From 2c8be15ec86dfc3053c84dd6d0da48792708a020 Mon Sep 17 00:00:00 2001 From: Owie6789 <151057755+Owie6789@users.noreply.github.com> Date: Wed, 23 Sep 2026 15:11:30 +0000 Subject: [PATCH 16/60] Record what the review round changed and what it left standing Six findings landed as code and two did not, and the difference is the part worth writing down. The duplication that failed SonarCloud's quality gate was this step's own copy-paste across ten capability manifests. The transport's private copies of helpers that had been extracted from it were pre-existing, and only became visible once onData was taken apart. The two cognitive-complexity reports that remain are pre-existing conditions in legacy renderer components this PR touched with three and eleven lines, and the record says what clearing each one would take instead of leaving a reviewer to guess why they were skipped. It also keeps the classifier drift the pin causes and this step must not fix: READ_ONLY_TOOLS still lists BashOutput, which the 2.1.270 binary no longer emits, and the fallback is fail-safe but not free. The permissions lane owns that table. Co-authored-by: arena-agent <297053741+arena-agent@users.noreply.github.com> --- ...2026-09-23-review-and-sonar-remediation.md | 120 ++++++++++++++++++ 1 file changed, 120 insertions(+) create mode 100644 .dump/app/decisions/2026-09-23-review-and-sonar-remediation.md diff --git a/.dump/app/decisions/2026-09-23-review-and-sonar-remediation.md b/.dump/app/decisions/2026-09-23-review-and-sonar-remediation.md new file mode 100644 index 00000000..07e5dae7 --- /dev/null +++ b/.dump/app/decisions/2026-09-23-review-and-sonar-remediation.md @@ -0,0 +1,120 @@ +# What the review round on PR #69 was worth, and what it changed + +Roadmap step 12, issue #14, PR #69 on `arena/01a0cbec-mauscode`, review round of +2026-09-23 against head `eb4a2bf`. This is the record of every finding the round +produced, what each one was worth, and what was done about it, so the next reader +does not have to re-judge a bot's opinion or re-derive why two findings were left +standing. + +AGENTS.md still says no review bot is configured. That is now false in three +directions: Sourcery, CodeAnt and Buoy all review here, and SonarQube Cloud runs a +quality gate on the pull request. Bot output was treated as evidence to verify +against the pinned artifacts and the tree, not as a verdict. + +## The round, finding by finding + +| Source | Finding | Verdict | Where it landed | +| --- | --- | --- | --- | +| Sourcery | A prompt suggestion stayed in the sub-chat atom after the next turn began | Real bug | `eb4a2bf` | +| Sourcery | Suggestions were stored by sub-chat id alone, so a late one from an older session could overwrite the current turn's | Real bug | `eb4a2bf` | +| CodeAnt | `KillBash`, `BashOutputTool`, `AgentOutput` and `AgentOutputTool` had no registry entry, so persisted calls rendered as generic rows | Real gap | `df4b974` | +| Buoy | `min-h-[32px]` should be `min-h-8`, twice, in the effort sub-menu | Declined | PR comment | +| SonarCloud | Quality gate failed: 4.2% duplication on new code against a 3% limit | Real, and this step's own making | `685aebd`, `fe09b5a` | +| SonarCloud | `handlePromptSuggestion` nested in the transformer closure | Real, cheap | `57e8cd3` | +| SonarCloud | Effort sub-menu props not read-only | Real, cheap | `3306846` | +| SonarCloud | `onData` cognitive complexity 43 against 15 | Real shape problem, mostly pre-existing | `dc182be`, `fe09b5a` | +| SonarCloud | `NewChatForm` cognitive complexity 23 against 15 | Pre-existing, deferred | below | +| SonarCloud | `renderPart` in `assistant-message-item.tsx` cognitive complexity 70 against 15 | Pre-existing, deferred | below | + +Duplication was the only condition the quality gate actually failed on. The three +cognitive-complexity reports are annotations against new code, and two of the three +are reported only because this PR touched lines inside functions that were already +over the limit. + +## The duplication was this step's own making + +The pin added three turn-shaping features to the capability manifest. Ten backends +each grew the same six lines: a three-line comment pointing at the research record, +then `effort`, `adaptiveThinking` and `promptSuggestions` set to false. Eight of the +ten were byte-identical, and a scan for repeated eight-line windows over the tree +returned them as the single largest duplicated block touching new code. + +`TURN_CONTROLS_OFF` in `src/shared/provider-capabilities.ts` now holds the "when in +doubt, false" answer in the module that owns the rule. A manifest spreads it and +names only what its own backend carries: Codex overrides `effort`, Claude turns all +three on and so inherits nothing. The forcing function is worth more than the line +count — a fourth flag added to the schema now fails typecheck in one place instead +of being silently missing from whichever manifest someone forgot. + +The second duplication source was not written by this step but was exposed by it. +`chat-chunk-atoms.ts` states that it was extracted from the Claude IPC transport so +the native runtime transport could reuse identical question, compacting and +prompt-extraction behaviour, and then the Claude transport kept its own copies: +`applyQuestionChunks`, `applyCompactingChunks`, `clearStalePendingQuestion`, +`extractPromptText` and `extractPromptImages` were called only from +`native-chat-transport.ts`, while `ipc-chat-transport.ts` carried the same logic +inline and two private methods byte-identical to the shared extractors. Refactoring +`onData` into handlers made that copy-paste visible; `fe09b5a` deletes it, 172 +lines, and the two transports can no longer drift on the question lifecycle. +`session-init` deliberately stays per transport, because the native runtime reads a +cached snapshot and fills the gaps while the CLI reports the full set on init. + +## The two complexity findings this step does not take + +| Function | Reported | What this PR changed inside it | What clearing it needs | +| --- | --- | --- | --- | +| `renderPart`, `assistant-message-item.tsx:803` | 70 against 15 | +3 / −4 lines: one `if` removed, one equality swapped for `isSubagentToolType()` | The 250-line render dispatcher split into components. It closes over roughly a dozen locals — orphan and nested tool-call sets, the nested-tools map, collapse state, the file-open callback — so each extraction is a props contract of its own, and nothing tests the file | +| `NewChatForm`, `new-chat-form.tsx:216` | 23 against 15 | +11 lines, one of them a ternary, so one point of the 23 | Section extraction across a 2300-line component whose remaining complexity is spread through the JSX body as `&&` and ternary expressions rather than sitting in one block | + +Both are pre-existing conditions reported on new code because the PR touched them. +Neither blocks the gate. Both are left standing on purpose: a dependency-pin pull +request that also rewrites two legacy renderer components cannot be reviewed as a +dependency-pin pull request, and there is no test coverage to make the rewrite +safe. They are named here so they are handed off rather than quietly dropped, and +the transport refactor is the model for how to do them — one handler per side +effect, an explicit context, and a single exit path. + +## What was declined, and why + +Buoy asked for `min-h-8` in place of `min-h-[32px]` on two rows of the effort +sub-menu. `min-h-[32px]` appears 14 times across `src/**/*.tsx` and `min-h-8` +appears zero times, so the token form has no precedent in this renderer, one of the +two flagged lines is pre-existing Codex code this PR only generalized, and adopting +the token here would make this component the single exception while leaving the +other 14 arbitrary values in place. A repo-wide move to spacing tokens is a +formatting contract of its own. + +## One finding this step must not fix + +The pin renames tools, and the permission classifier keeps its own name table. +`READ_ONLY_TOOLS` in `src/shared/permissions/classifier.ts` lists `BashOutput`, +which the 2.1.270 binary no longer emits: it emits `TaskOutput`. `classifyToolName` +falls through to `approval` / `unclassified-tool`, "no classification, so it takes +the middle tier", for any name it does not know, so the direction of the drift is +fail-safe — reading background output now costs an approval instead of being +read-only, rather than the other way round. The same is true of `TaskStop` for +`KillShell` and `KillBash`, and of `Agent` for `Task`. + +It is recorded rather than changed here because the classifier belongs to the +permission-floor work of steps 10 and 11, which another lane owns, and a +dependency-pin step editing that table is exactly the cross-lane collision the lane +rules exist to prevent. Whoever owns the classifier should decide whether the +renamed tools take their predecessors' classes. + +## Verification in this sandbox + +2 CPU, 3.9 GB, no Electron binary (the postinstall download is intercepted) and no +local build or package. + +| Gate | Result | +| --- | --- | +| `biome check .` | 977 files, 0 findings | +| `tsc --noEmit` | 0 errors, after building `packages/runtime-client` for its `dist` types | +| `ratchet:typecheck` | passed, 0 errors against a 0 baseline | +| `vitest run` | 103 files, 1891 passed, 1 skipped | +| `npm run test:node` | 59 passed, 0 failed | +| `npm run test:contracts` | 23 files, 382 passed | +| `npm run lint` | 922 files checked, no findings | +| `ratchet:audit` | passed, 3 critical baseline, no new critical advisories | +| `skills:verify` | 50 of 50 locked skills verified, 2 unrecorded project-owned (pre-existing) | +| `build`, `package:linux` | not runnable here; CI runs them and the results are recorded on the PR | From 43ff64da8bec69ebd25f3ad873987fde19ca72e3 Mon Sep 17 00:00:00 2001 From: Owie6789 <151057755+Owie6789@users.noreply.github.com> Date: Wed, 23 Sep 2026 16:06:24 +0000 Subject: [PATCH 17/60] Collapse the provider feature flags onto one all-off default TURN_CONTROLS_OFF named four of the thirteen flags, so every manifest restated the other nine and the shared default covered a third of the object. The flags are not four: they are thirteen booleans that all default off, because a capability nobody proved is one this app must not advertise. - FeatureFlags is now z.infer, so the type cannot drift from the schema; ALL_FEATURES_OFF spells all thirteen false under `satisfies FeatureFlags`, which makes adding a schema flag a compile error here until its default is chosen deliberately. - Each manifest reads features: { ...ALL_FEATURES_OFF, }. A flag that is off because nobody proved it is no longer written; a flag that is off for a reason keeps its reason inline (grok's -r composition, cline's broken --id, openclaw's missing session ids, roo's rejected prompt, the two unverified skills claims). Fail-safe direction preserved: ALL_FEATURES_OFF is a superset of TURN_CONTROLS_OFF, so no provider gained a control it did not have. Evidence: verified the default is 13 keys, all false, covering every schema flag with no extras; compared the effective flag matrix parsed from HEAD against the working tree - 130 values across 10 providers, 0 differences. Biome clean; tsc --noEmit clean; vitest 1891 passed. Co-authored-by: arena-agent <297053741+arena-agent@users.noreply.github.com> --- src/main/lib/providers/claude.ts | 6 ++--- src/main/lib/providers/cline.ts | 8 ++---- src/main/lib/providers/codex.ts | 10 ++----- src/main/lib/providers/cursor.ts | 9 ++----- src/main/lib/providers/grok.ts | 8 ++---- src/main/lib/providers/hermes.ts | 8 ++---- src/main/lib/providers/openclaw.ts | 9 ++----- src/main/lib/providers/opencode.ts | 9 ++----- src/main/lib/providers/qwen.ts | 7 ++--- src/main/lib/providers/roo.ts | 9 ++----- src/shared/provider-capabilities.ts | 41 +++++++++++++++++++---------- 11 files changed, 47 insertions(+), 77 deletions(-) diff --git a/src/main/lib/providers/claude.ts b/src/main/lib/providers/claude.ts index b4c1a5cf..337636e6 100644 --- a/src/main/lib/providers/claude.ts +++ b/src/main/lib/providers/claude.ts @@ -1,7 +1,7 @@ import { execFile } from "node:child_process" import { existsSync } from "node:fs" import { eq } from "drizzle-orm" -import type { ProviderCapability } from "../../../shared/provider-capabilities" +import { ALL_FEATURES_OFF, type ProviderCapability } from "../../../shared/provider-capabilities" import { getBundledClaudeBinaryPath } from "../claude/env" import { getExistingClaudeCredentials } from "../claude-token" import { anthropicAccounts, anthropicSettings, claudeCodeCredentials, getDatabase } from "../db" @@ -91,16 +91,14 @@ export function getClaudeCapability(): ProviderCapability { usageSurface: "native", }, features: { + ...ALL_FEATURES_OFF, chat: true, images: true, resume: true, fork: true, mcp: true, subagents: true, - cron: false, skills: true, - structuredOutput: false, - fileCheckpointing: false, // All three on: the 0.3.270 pin carries `Options.effort`, adaptive // thinking and `Options.promptSuggestions` through a turn end to end. effort: true, diff --git a/src/main/lib/providers/cline.ts b/src/main/lib/providers/cline.ts index 3e7aa6c9..47ce59bd 100644 --- a/src/main/lib/providers/cline.ts +++ b/src/main/lib/providers/cline.ts @@ -1,5 +1,5 @@ import { execFile } from "node:child_process" -import { type ProviderCapability, TURN_CONTROLS_OFF } from "../../../shared/provider-capabilities" +import { ALL_FEATURES_OFF, type ProviderCapability } from "../../../shared/provider-capabilities" import { resolveClineCliLaunch } from "../cline-binary" import { probeClineStoredAuth } from "../cline-print/auth-config" import type { BackendProbe } from "./types" @@ -53,6 +53,7 @@ export function getClineCapability(): ProviderCapability { usageSurface: "native", }, features: { + ...ALL_FEATURES_OFF, chat: true, // `@./path.png` image mentions exist upstream, but headless // image support is unverified — attachments travel as prompt @@ -61,16 +62,11 @@ export function getClineCapability(): ProviderCapability { // --id resume is broken in all headless paths (v3.0.61): // continuity comes from transcript-in-prompt instead. resume: false, - fork: false, mcp: true, // spawn_agent / team_* tools exist upstream and flow through // the tool projector; multi-agent orchestration is CLI-managed. subagents: true, - cron: false, skills: true, - structuredOutput: false, - fileCheckpointing: false, - ...TURN_CONTROLS_OFF, }, notes: [ "Images travel as prompt path references the agent reads via tools.", diff --git a/src/main/lib/providers/codex.ts b/src/main/lib/providers/codex.ts index 16a768ac..4fea3deb 100644 --- a/src/main/lib/providers/codex.ts +++ b/src/main/lib/providers/codex.ts @@ -1,7 +1,7 @@ import { execFile } from "node:child_process" import { join } from "node:path" import { app } from "electron" -import { type ProviderCapability, TURN_CONTROLS_OFF } from "../../../shared/provider-capabilities" +import { ALL_FEATURES_OFF, type ProviderCapability } from "../../../shared/provider-capabilities" import type { BackendProbe } from "./types" function resolveCodexBinary(): string { @@ -62,17 +62,11 @@ export function getCodexCapability(): ProviderCapability { usageSurface: "session-files", }, features: { + ...ALL_FEATURES_OFF, chat: true, images: true, resume: true, - fork: false, mcp: true, - subagents: false, - cron: false, - skills: false, - structuredOutput: false, - fileCheckpointing: false, - ...TURN_CONTROLS_OFF, // The app-server takes a reasoning effort on a turn; Codex chooses its // own thinking budget and sends no prompt suggestion. effort: true, diff --git a/src/main/lib/providers/cursor.ts b/src/main/lib/providers/cursor.ts index 731fbd96..deb9e1cd 100644 --- a/src/main/lib/providers/cursor.ts +++ b/src/main/lib/providers/cursor.ts @@ -1,5 +1,5 @@ import { execFile } from "node:child_process" -import { type ProviderCapability, TURN_CONTROLS_OFF } from "../../../shared/provider-capabilities" +import { ALL_FEATURES_OFF, type ProviderCapability } from "../../../shared/provider-capabilities" import { resolveCursorAgentCliLaunch } from "../cursor-agent-binary" import type { BackendProbe } from "./types" @@ -51,18 +51,13 @@ export function getCursorCapability(): ProviderCapability { usageSurface: "none", }, features: { + ...ALL_FEATURES_OFF, chat: true, images: true, resume: true, - fork: false, mcp: true, // Task/subagent delegation drains before print runs exit. subagents: true, - cron: false, - skills: false, - structuredOutput: false, - fileCheckpointing: false, - ...TURN_CONTROLS_OFF, }, notes: [ "Images travel as prompt path references the agent reads via tools.", diff --git a/src/main/lib/providers/grok.ts b/src/main/lib/providers/grok.ts index da8f99e7..4351dc76 100644 --- a/src/main/lib/providers/grok.ts +++ b/src/main/lib/providers/grok.ts @@ -1,7 +1,7 @@ import { execFile } from "node:child_process" import { existsSync, readFileSync } from "node:fs" import { join } from "node:path" -import { type ProviderCapability, TURN_CONTROLS_OFF } from "../../../shared/provider-capabilities" +import { ALL_FEATURES_OFF, type ProviderCapability } from "../../../shared/provider-capabilities" import { resolveGrokCliLaunch, resolveGrokHome } from "../grok-binary" import type { BackendProbe } from "./types" @@ -57,6 +57,7 @@ export function getGrokCapability(): ProviderCapability { usageSurface: "native", }, features: { + ...ALL_FEATURES_OFF, chat: true, images: true, resume: true, @@ -67,11 +68,6 @@ export function getGrokCapability(): ProviderCapability { // Task-tool delegation flows through the same tool projector as any // other tool call (same posture as the cursor backend). subagents: true, - cron: false, - skills: false, - structuredOutput: false, - fileCheckpointing: false, - ...TURN_CONTROLS_OFF, }, notes: [ "Images travel as prompt path references the agent reads via tools.", diff --git a/src/main/lib/providers/hermes.ts b/src/main/lib/providers/hermes.ts index 859e7097..c514c874 100644 --- a/src/main/lib/providers/hermes.ts +++ b/src/main/lib/providers/hermes.ts @@ -1,5 +1,5 @@ import { execFile } from "node:child_process" -import { type ProviderCapability, TURN_CONTROLS_OFF } from "../../../shared/provider-capabilities" +import { ALL_FEATURES_OFF, type ProviderCapability } from "../../../shared/provider-capabilities" import type { BackendProbe } from "./types" function runBinary( @@ -48,17 +48,13 @@ export function getHermesCapability(): ProviderCapability { usageSurface: "none", }, features: { + ...ALL_FEATURES_OFF, chat: true, images: true, resume: true, - fork: false, mcp: true, subagents: true, - cron: false, skills: true, - structuredOutput: false, - fileCheckpointing: false, - ...TURN_CONTROLS_OFF, }, notes: [ "ACP sessions live in the running server process; resume across restarts is best-effort.", diff --git a/src/main/lib/providers/openclaw.ts b/src/main/lib/providers/openclaw.ts index 8ceb6379..bdc8e0d8 100644 --- a/src/main/lib/providers/openclaw.ts +++ b/src/main/lib/providers/openclaw.ts @@ -1,5 +1,5 @@ import { execFile } from "node:child_process" -import { type ProviderCapability, TURN_CONTROLS_OFF } from "../../../shared/provider-capabilities" +import { ALL_FEATURES_OFF, type ProviderCapability } from "../../../shared/provider-capabilities" import { resolveOpenclawCliLaunch } from "../openclaw-binary" import { readOpenclawModelsStatus, summarizeModelsStatusAuth } from "../openclaw-print/auth-config" import type { BackendProbe } from "./types" @@ -56,6 +56,7 @@ export function getOpenclawCapability(): ProviderCapability { usageSurface: "native", }, features: { + ...ALL_FEATURES_OFF, chat: true, // No image input surface on exec — attachments travel as prompt // path references the agent reads via tools (cline posture). @@ -63,16 +64,10 @@ export function getOpenclawCapability(): ProviderCapability { // Exec accepts no session id: each turn is a fresh session with // bounded transcript context. resume: false, - fork: false, mcp: true, - subagents: false, - cron: false, // Upstream skills exist but exec-run skill loading is // unverified from here. skills: false, - structuredOutput: false, - fileCheckpointing: false, - ...TURN_CONTROLS_OFF, }, notes: [ "One JSON envelope per turn — no streaming; progress appears only when the turn settles.", diff --git a/src/main/lib/providers/opencode.ts b/src/main/lib/providers/opencode.ts index bdf4cca5..4acebadb 100644 --- a/src/main/lib/providers/opencode.ts +++ b/src/main/lib/providers/opencode.ts @@ -1,5 +1,5 @@ import { execFile } from "node:child_process" -import { type ProviderCapability, TURN_CONTROLS_OFF } from "../../../shared/provider-capabilities" +import { ALL_FEATURES_OFF, type ProviderCapability } from "../../../shared/provider-capabilities" import type { BackendProbe } from "./types" function runBinary( @@ -48,17 +48,12 @@ export function getOpencodeCapability(): ProviderCapability { usageSurface: "native", }, features: { + ...ALL_FEATURES_OFF, chat: true, images: true, resume: true, - fork: false, mcp: true, subagents: true, - cron: false, - skills: false, - structuredOutput: false, - fileCheckpointing: false, - ...TURN_CONTROLS_OFF, }, notes: [ "Permissions auto-reply session-wide; opencode.json can tighten per-tool policy.", diff --git a/src/main/lib/providers/qwen.ts b/src/main/lib/providers/qwen.ts index fe0ebe71..ef4b0ccf 100644 --- a/src/main/lib/providers/qwen.ts +++ b/src/main/lib/providers/qwen.ts @@ -1,5 +1,5 @@ import { execFile } from "node:child_process" -import { type ProviderCapability, TURN_CONTROLS_OFF } from "../../../shared/provider-capabilities" +import { ALL_FEATURES_OFF, type ProviderCapability } from "../../../shared/provider-capabilities" import { resolveQwenCliLaunch } from "../qwen-binary" import { probeQwenStoredAuth } from "../qwen-print/auth-config" import type { BackendProbe } from "./types" @@ -53,6 +53,7 @@ export function getQwenCapability(): ProviderCapability { usageSurface: "native", }, features: { + ...ALL_FEATURES_OFF, chat: true, // read_file reads images/PDFs by path; attachments are staged to // temp files and referenced from the prompt (cursor posture). @@ -65,11 +66,7 @@ export function getQwenCapability(): ProviderCapability { // agent/task delegation flows through the same tool projector as // any other tool call (same posture as the claude backend). subagents: true, - cron: false, skills: true, - structuredOutput: false, - fileCheckpointing: false, - ...TURN_CONTROLS_OFF, }, notes: [ "Images travel as prompt path references the agent reads via tools.", diff --git a/src/main/lib/providers/roo.ts b/src/main/lib/providers/roo.ts index e3e330e7..ceac74c1 100644 --- a/src/main/lib/providers/roo.ts +++ b/src/main/lib/providers/roo.ts @@ -1,5 +1,5 @@ import { execFile } from "node:child_process" -import { type ProviderCapability, TURN_CONTROLS_OFF } from "../../../shared/provider-capabilities" +import { ALL_FEATURES_OFF, type ProviderCapability } from "../../../shared/provider-capabilities" import { getClaudeShellEnvironment } from "../claude/env" import { resolveRooCliLaunch } from "../roo-binary" import { resolveRooAmbientAuth } from "../roo-print/auth-config" @@ -53,6 +53,7 @@ export function getRooCapability(): ProviderCapability { usageSurface: "native", }, features: { + ...ALL_FEATURES_OFF, chat: true, // No image input surface on print — attachments travel as prompt // path references the agent reads via tools (cline posture). @@ -60,16 +61,10 @@ export function getRooCapability(): ProviderCapability { // Resume rejects prompt upstream: each turn is a fresh session // with bounded transcript context. resume: false, - fork: false, mcp: true, - subagents: false, - cron: false, // Upstream custom tools exist but print-run skill loading is // unverified from here. skills: false, - structuredOutput: false, - fileCheckpointing: false, - ...TURN_CONTROLS_OFF, }, notes: [ "Streams NDJSON events per turn (text deltas, thinking, tool calls, command output, cost).", diff --git a/src/shared/provider-capabilities.ts b/src/shared/provider-capabilities.ts index d0ff0a52..a8d9ce96 100644 --- a/src/shared/provider-capabilities.ts +++ b/src/shared/provider-capabilities.ts @@ -99,26 +99,39 @@ export type ProviderCapability = z.infer /** The two answers to "who enforces the permission floor of roadmap step 10". */ export type PermissionFloor = ProviderCapability["security"]["permissionFloor"] -/** The three turn-shaping features the 0.3.270 SDK pin made expressible. */ -export type TurnControlFeatures = Pick< - ProviderCapability["features"], - "effort" | "adaptiveThinking" | "promptSuggestions" -> +/** The feature flags themselves, so a manifest can be typed against them. */ +export type FeatureFlags = z.infer /** - * The "when in doubt, false" rule above, stated once for those three features. - * A manifest spreads this and then names only what its own backend carries end - * to end, so eight backends that support none of them cannot drift to eight - * hand-written copies of the same default, and a fourth flag added to the - * schema fails typecheck here rather than being silently missing from one - * manifest. The per-backend evidence is in - * `.dump/app/research/2026-09-13-sdk-0-3-bump.md`. + * Every feature at the rule above's own answer: false. + * + * A manifest spreads this and then claims what its backend carries end to end, + * so the flags nothing in mausCode wires for any backend yet are stated once + * here instead of once per manifest, and a claim reads as a claim. Forgetting + * one fails safe, which is the direction the rule already asks for: the UI must + * not offer what a manifest did not say. A flag added to the schema fails + * typecheck here rather than going silently missing from one manifest. + * + * A negative claim that was investigated keeps its own line and its reason in + * the manifest — `grok` forks, `openclaw` does not resume — because that is + * evidence about a backend, not a default. The evidence behind the three + * turn-shaping flags is in `.dump/app/research/2026-09-13-sdk-0-3-bump.md`. */ -export const TURN_CONTROLS_OFF: TurnControlFeatures = { +export const ALL_FEATURES_OFF = { + chat: false, + images: false, + resume: false, + fork: false, + mcp: false, + subagents: false, + cron: false, + skills: false, + structuredOutput: false, + fileCheckpointing: false, effort: false, adaptiveThinking: false, promptSuggestions: false, -} +} satisfies FeatureFlags /** * The floor behind a sub-chat provider id, which is the vocabulary the chat UI From e57175ba0090f46a4a8fe98b4450b88713e4db06 Mon Sep 17 00:00:00 2001 From: Owie6789 <151057755+Owie6789@users.noreply.github.com> Date: Wed, 23 Sep 2026 16:07:06 +0000 Subject: [PATCH 18/60] Share one Claude model picker hook between the two surfaces chat-input-area.tsx and new-chat-form.tsx each carried a byte-identical 37-line useAvailableModels, an identical connection test, and an identical 26-line claude={{...}} block - about 110 duplicated lines that had already drifted once, when the effort rows were added to both by hand in the round that shipped adaptive thinking. hooks/use-claude-model-picker.ts now owns the model list and the offline-Ollama overlay, the custom-config test, the connection test, the resolved Ollama model, extended thinking, and the capability-driven effort rows. It returns the list plus a ready props object typed against AgentModelSelectorProps["claude"] (now exported), so the block cannot drift from the component it feeds. selectedModelId and onSelectModel stay with each surface: they are the only two lines that differ - the composer also stamps the sub-chat model id - and folding them in would mean handing callbacks back out. Evidence: the 6-line duplication scan reports no shared window containing added lines left in either file (repo-wide, groups touching new code went 21 -> 0 at that window); -177/+22 lines across the two surfaces; Biome clean; tsc --noEmit clean; vitest 1891 passed, 1 skipped. Co-authored-by: arena-agent <297053741+arena-agent@users.noreply.github.com> --- .../components/agent-model-selector.tsx | 2 +- .../agents/hooks/use-claude-model-picker.ts | 130 ++++++++++++++++++ .../features/agents/main/chat-input-area.tsx | 100 ++------------ .../features/agents/main/new-chat-form.tsx | 99 ++----------- 4 files changed, 153 insertions(+), 178 deletions(-) create mode 100644 src/renderer/features/agents/hooks/use-claude-model-picker.ts diff --git a/src/renderer/features/agents/components/agent-model-selector.tsx b/src/renderer/features/agents/components/agent-model-selector.tsx index c0ad5ca1..0c2e9335 100644 --- a/src/renderer/features/agents/components/agent-model-selector.tsx +++ b/src/renderer/features/agents/components/agent-model-selector.tsx @@ -134,7 +134,7 @@ type RooModelOption = { name: string } -interface AgentModelSelectorProps { +export interface AgentModelSelectorProps { open: boolean onOpenChange: (open: boolean) => void selectedAgentId: AgentProviderId diff --git a/src/renderer/features/agents/hooks/use-claude-model-picker.ts b/src/renderer/features/agents/hooks/use-claude-model-picker.ts new file mode 100644 index 00000000..694eb74d --- /dev/null +++ b/src/renderer/features/agents/hooks/use-claude-model-picker.ts @@ -0,0 +1,130 @@ +/** + * The Claude half of the model picker, which two surfaces render: the composer + * of an open chat and the new-chat form. Each held its own copy of the model + * list, the offline read, the connection test, the capability question and the + * thinking and effort atoms, and then built thirteen identical props out of + * them, so a change to one surface missed the other until somebody noticed. The + * hook owns that state and returns the props the two share; each surface keeps + * the two that are genuinely its own — which model is selected, and what + * selecting one means there. + */ +import { useAtom, useAtomValue } from "jotai" +import { EFFORT_LEVELS } from "../../../../shared/effort" +import { + anthropicOnboardingCompletedAtom, + apiKeyOnboardingCompletedAtom, + claudeEffortAtom, + customClaudeConfigAtom, + extendedThinkingEnabledAtom, + normalizeCustomClaudeConfig, + selectedOllamaModelAtom, + showOfflineModeFeaturesAtom, +} from "../../../lib/atoms" +import { trpc } from "../../../lib/trpc" +import type { AgentModelSelectorProps } from "../components/agent-model-selector" +import { CLAUDE_MODELS } from "../lib/models" + +/** + * The picker props both surfaces pass identically. Typed from the component's + * own contract, so a prop added to `AgentModelSelectorProps` that both surfaces + * owe fails typecheck here until it is wired once, rather than being added to + * one call site and forgotten in the other. + */ +export type SharedClaudePickerProps = Omit< + AgentModelSelectorProps["claude"], + "selectedModelId" | "onSelectModel" +> + +// Available Claude models, plus the Ollama ones an offline turn can use. +function useAvailableModels() { + const showOfflineFeatures = useAtomValue(showOfflineModeFeaturesAtom) + const { data: ollamaStatus } = trpc.ollama.getStatus.useQuery(undefined, { + refetchInterval: showOfflineFeatures ? 30000 : false, + enabled: showOfflineFeatures, // Only query Ollama when offline mode is enabled + }) + + const baseModels = CLAUDE_MODELS + + const isOffline = ollamaStatus ? !ollamaStatus.internet.online : false + const hasOllama = ollamaStatus?.ollama.available && (ollamaStatus.ollama.models?.length ?? 0) > 0 + const ollamaModels = ollamaStatus?.ollama.models || [] + const recommendedModel = ollamaStatus?.ollama.recommendedModel + + // Only show offline models if: + // 1. Debug flag is enabled (showOfflineFeatures) + // 2. Ollama is available with models + // 3. User is actually offline + if (showOfflineFeatures && hasOllama && isOffline) { + return { + models: baseModels, + ollamaModels, + recommendedModel, + isOffline, + hasOllama: true, + } + } + + return { + models: baseModels, + ollamaModels: [] as string[], + recommendedModel: undefined as string | undefined, + isOffline, + hasOllama: false, + } +} + +export function useClaudeModelPicker(hiddenModels: readonly string[]) { + const availableModels = useAvailableModels() + + // A custom config counts as connected: the turn goes to the endpoint it names + // rather than to an account this app signed into. + const customClaudeConfig = useAtomValue(customClaudeConfigAtom) + const hasCustomClaudeConfig = Boolean(normalizeCustomClaudeConfig(customClaudeConfig)) + const anthropicOnboardingCompleted = useAtomValue(anthropicOnboardingCompletedAtom) + const apiKeyOnboardingCompleted = useAtomValue(apiKeyOnboardingCompletedAtom) + const { data: claudeCodeIntegration } = trpc.claudeCode.getIntegration.useQuery() + const isClaudeConnected = + Boolean(claudeCodeIntegration?.isConnected) || + anthropicOnboardingCompleted || + apiKeyOnboardingCompleted || + hasCustomClaudeConfig + + const [selectedOllamaModel, setSelectedOllamaModel] = useAtom(selectedOllamaModelAtom) + // The model an offline turn actually runs: the picked one, else the one + // Ollama recommends, else the first one available. + const currentOllamaModel = + selectedOllamaModel || availableModels.recommendedModel || availableModels.ollamaModels[0] + + const [thinkingEnabled, setThinkingEnabled] = useAtom(extendedThinkingEnabledAtom) + + // The effort rows come from the backend's own capability profile, so a + // provider that reports no effort control shows no sub-menu. + const { data: claudeCapability } = trpc.providers.get.useQuery({ id: "claude" }) + const claudeEfforts = claudeCapability?.features.effort ? EFFORT_LEVELS : [] + const [selectedClaudeEffort, setSelectedClaudeEffort] = useAtom(claudeEffortAtom) + + const props: SharedClaudePickerProps = { + models: availableModels.models.filter((model) => !hiddenModels.includes(model.id)), + hasCustomModelConfig: hasCustomClaudeConfig, + isOffline: availableModels.isOffline && availableModels.hasOllama, + ollamaModels: availableModels.ollamaModels, + selectedOllamaModel: currentOllamaModel, + recommendedOllamaModel: availableModels.recommendedModel, + onSelectOllamaModel: setSelectedOllamaModel, + isConnected: isClaudeConnected, + thinkingEnabled, + onThinkingChange: setThinkingEnabled, + efforts: claudeEfforts, + selectedEffort: selectedClaudeEffort, + onSelectEffort: setSelectedClaudeEffort, + } + + // The connection test travels inside `props`; neither surface reads it apart + // from the picker, so it is not part of the return. + return { + availableModels, + hasCustomClaudeConfig, + currentOllamaModel, + props, + } +} diff --git a/src/renderer/features/agents/main/chat-input-area.tsx b/src/renderer/features/agents/main/chat-input-area.tsx index 753766bf..cba6b044 100644 --- a/src/renderer/features/agents/main/chat-input-area.tsx +++ b/src/renderer/features/agents/main/chat-input-area.tsx @@ -13,7 +13,6 @@ import { ChevronDown, Sparkles, Zap } from "lucide-react" import { memo, useCallback, useEffect, useMemo, useRef, useState } from "react" import { createPortal } from "react-dom" import { toast } from "sonner" -import { EFFORT_LEVELS } from "../../../../shared/effort" import { nativeModeRefusal } from "../../../../shared/permissions/native-mode-floor" import { permissionFloorFor } from "../../../../shared/provider-capabilities" import { Button } from "../../../components/ui/button" @@ -32,21 +31,13 @@ import { import { agentsSettingsDialogActiveTabAtom, agentsSettingsDialogOpenAtom, - anthropicOnboardingCompletedAtom, - apiKeyOnboardingCompletedAtom, - claudeEffortAtom, codexApiKeyAtom, codexOnboardingCompletedAtom, - customClaudeConfigAtom, customHotkeysAtom, - extendedThinkingEnabledAtom, hiddenModelsAtom, normalizeCodexApiKey, - normalizeCustomClaudeConfig, pinnedOpenRouterModelsAtom, - selectedOllamaModelAtom, sessionInfoAtom, - showOfflineModeFeaturesAtom, } from "../../../lib/atoms" import { blobToBase64, @@ -93,11 +84,11 @@ import { AgentsSlashCommand, type SlashCommandOption } from "../commands" import { AgentModelSelector, type AgentProviderId } from "../components/agent-model-selector" import { AgentSendButton } from "../components/agent-send-button" import type { UploadedFile, UploadedImage } from "../hooks/use-agents-file-upload" +import { useClaudeModelPicker } from "../hooks/use-claude-model-picker" import type { PastedTextFile } from "../hooks/use-pasted-text-files" import { clearSubChatDraft, saveSubChatDraftWithAttachments } from "../lib/drafts" import { getModeIcon, getModeLabel, getModeTooltip } from "../lib/mode-display" import { - CLAUDE_MODELS, CLINE_MODELS, CODEX_MODELS, CODEX_SUBSCRIPTION_ONLY_MODEL_IDS, @@ -127,44 +118,6 @@ import { AgentTextContextItem } from "../ui/agent-text-context-item" import { VoiceWaveIndicator } from "../ui/voice-wave-indicator" import { handlePasteEvent } from "../utils/paste-text" -// Hook to get available models (including offline models if Ollama is available and debug enabled) -function useAvailableModels() { - const showOfflineFeatures = useAtomValue(showOfflineModeFeaturesAtom) - const { data: ollamaStatus } = trpc.ollama.getStatus.useQuery(undefined, { - refetchInterval: showOfflineFeatures ? 30000 : false, - enabled: showOfflineFeatures, // Only query Ollama when offline mode is enabled - }) - - const baseModels = CLAUDE_MODELS - - const isOffline = ollamaStatus ? !ollamaStatus.internet.online : false - const hasOllama = ollamaStatus?.ollama.available && (ollamaStatus.ollama.models?.length ?? 0) > 0 - const ollamaModels = ollamaStatus?.ollama.models || [] - const recommendedModel = ollamaStatus?.ollama.recommendedModel - - // Only show offline models if: - // 1. Debug flag is enabled (showOfflineFeatures) - // 2. Ollama is available with models - // 3. User is actually offline - if (showOfflineFeatures && hasOllama && isOffline) { - return { - models: baseModels, - ollamaModels, - recommendedModel, - isOffline, - hasOllama: true, - } - } - - return { - models: baseModels, - ollamaModels: [] as string[], - recommendedModel: undefined as string | undefined, - isOffline, - hasOllama: false, - } -} - export interface ChatInputAreaProps { // Editor ref - passed from parent for external access editorRef: React.RefObject @@ -559,8 +512,15 @@ export const ChatInputArea = memo(function ChatInputArea({ refetchOnWindowFocus: false, retry: 1, }) - const [selectedOllamaModel, setSelectedOllamaModel] = useAtom(selectedOllamaModelAtom) - const availableModels = useAvailableModels() + const hiddenModels = useAtomValue(hiddenModelsAtom) + // The Claude half of the picker is shared with the new-chat form; which model + // is selected, and what selecting one does, belong to this surface alone. + const { + availableModels, + hasCustomClaudeConfig, + currentOllamaModel, + props: claudePickerProps, + } = useClaudeModelPicker(hiddenModels) const [selectedModel, setSelectedModel] = useState( () => availableModels.models.find((m) => m.id === selectedSubChatModelId) || @@ -583,13 +543,8 @@ export const ChatInputArea = memo(function ChatInputArea({ setSelectedSubChatModelId(selectedModel.id) }, [provider, selectedModel?.id, setSelectedSubChatModelId]) - const hiddenModels = useAtomValue(hiddenModelsAtom) - // Connection status for providers - const anthropicOnboardingCompleted = useAtomValue(anthropicOnboardingCompletedAtom) - const apiKeyOnboardingCompleted = useAtomValue(apiKeyOnboardingCompletedAtom) const codexOnboardingCompleted = useAtomValue(codexOnboardingCompletedAtom) - const { data: claudeCodeIntegration } = trpc.claudeCode.getIntegration.useQuery() const { data: cursorIntegration } = trpc.cursor.getIntegration.useQuery() const { data: grokIntegration } = trpc.grok.getIntegration.useQuery() const { data: qwenIntegration } = trpc.qwen.getIntegration.useQuery() @@ -787,14 +742,6 @@ export const ChatInputArea = memo(function ChatInputArea({ setSelectedSubChatRooModelId(selectedRooModel.id) }, [provider, selectedRooModel?.id, setSelectedSubChatRooModelId]) - const customClaudeConfig = useAtomValue(customClaudeConfigAtom) - const normalizedCustomClaudeConfig = normalizeCustomClaudeConfig(customClaudeConfig) - const hasCustomClaudeConfig = Boolean(normalizedCustomClaudeConfig) - const isClaudeConnected = - Boolean(claudeCodeIntegration?.isConnected) || - anthropicOnboardingCompleted || - apiKeyOnboardingCompleted || - hasCustomClaudeConfig const isCursorConnected = Boolean(cursorIntegration?.isConnected) const isGrokConnected = Boolean(grokIntegration?.isConnected) const isQwenConnected = Boolean(qwenIntegration?.isConnected) @@ -802,19 +749,6 @@ export const ChatInputArea = memo(function ChatInputArea({ const isOpenclawConnected = Boolean(openclawIntegration?.isConnected) const isRooConnected = Boolean(rooIntegration?.isConnected) - // Determine current Ollama model (selected or recommended) - const currentOllamaModel = - selectedOllamaModel || availableModels.recommendedModel || availableModels.ollamaModels[0] - - // Extended thinking (reasoning) toggle - const [thinkingEnabled, setThinkingEnabled] = useAtom(extendedThinkingEnabledAtom) - - // The effort rows come from the backend's own capability profile, so a - // provider that reports no effort control shows no sub-menu. - const { data: claudeCapability } = trpc.providers.get.useQuery({ id: "claude" }) - const claudeEfforts = claudeCapability?.features.effort ? EFFORT_LEVELS : [] - const [selectedClaudeEffort, setSelectedClaudeEffort] = useAtom(claudeEffortAtom) - const selectedModelLabel = useMemo(() => { if (provider === "codex") { return selectedCodexModel.name @@ -2074,7 +2008,7 @@ export const ChatInputArea = memo(function ChatInputArea({ setSettingsOpen(true) }} claude={{ - models: availableModels.models.filter((m) => !hiddenModels.includes(m.id)), + ...claudePickerProps, selectedModelId: selectedModel?.id, onSelectModel: (modelId) => { const model = @@ -2085,18 +2019,6 @@ export const ChatInputArea = memo(function ChatInputArea({ setSelectedSubChatModelId(model.id) setLastSelectedModelId(model.id) }, - hasCustomModelConfig: hasCustomClaudeConfig, - isOffline: availableModels.isOffline && availableModels.hasOllama, - ollamaModels: availableModels.ollamaModels, - selectedOllamaModel: currentOllamaModel, - recommendedOllamaModel: availableModels.recommendedModel, - onSelectOllamaModel: setSelectedOllamaModel, - isConnected: isClaudeConnected, - thinkingEnabled, - onThinkingChange: setThinkingEnabled, - efforts: claudeEfforts, - selectedEffort: selectedClaudeEffort, - onSelectEffort: setSelectedClaudeEffort, }} codex={{ models: codexUiModels, diff --git a/src/renderer/features/agents/main/new-chat-form.tsx b/src/renderer/features/agents/main/new-chat-form.tsx index 545a73c6..6c3dd7b1 100644 --- a/src/renderer/features/agents/main/new-chat-form.tsx +++ b/src/renderer/features/agents/main/new-chat-form.tsx @@ -5,7 +5,6 @@ import { atom, useAtom, useAtomValue, useSetAtom } from "jotai" import { AlignJustify, Plus } from "lucide-react" import { useCallback, useEffect, useMemo, useRef, useState } from "react" import { createPortal } from "react-dom" -import { EFFORT_LEVELS } from "../../../../shared/effort" import { permissionFloorFor } from "../../../../shared/provider-capabilities" import { Button } from "../../../components/ui/button" import { @@ -79,21 +78,13 @@ import { import { agentsSettingsDialogActiveTabAtom, agentsSettingsDialogOpenAtom, - anthropicOnboardingCompletedAtom, - apiKeyOnboardingCompletedAtom, chatSourceModeAtom, - claudeEffortAtom, codexApiKeyAtom, codexOnboardingCompletedAtom, - customClaudeConfigAtom, customHotkeysAtom, - extendedThinkingEnabledAtom, hiddenModelsAtom, normalizeCodexApiKey, - normalizeCustomClaudeConfig, pinnedOpenRouterModelsAtom, - selectedOllamaModelAtom, - showOfflineModeFeaturesAtom, } from "../../../lib/atoms" import { blobToBase64, @@ -113,6 +104,7 @@ import { AgentModelSelector, type AgentProviderId } from "../components/agent-mo import { AgentSendButton } from "../components/agent-send-button" import { CreateBranchDialog } from "../components/create-branch-dialog" import { useAgentsFileUpload } from "../hooks/use-agents-file-upload" +import { useClaudeModelPicker } from "../hooks/use-claude-model-picker" import { useFocusInputOnEnter } from "../hooks/use-focus-input-on-enter" import { usePastedTextFiles } from "../hooks/use-pasted-text-files" import { useToggleFocusOnCmdEsc } from "../hooks/use-toggle-focus-on-cmd-esc" @@ -124,7 +116,6 @@ import { saveGlobalDrafts, } from "../lib/drafts" import { - CLAUDE_MODELS, CLINE_MODELS, CODEX_MODELS, CODEX_SUBSCRIPTION_ONLY_MODEL_IDS, @@ -151,44 +142,6 @@ import { VoiceWaveIndicator } from "../ui/voice-wave-indicator" import { formatTimeAgo } from "../utils/format-time-ago" import { handlePasteEvent } from "../utils/paste-text" -// Hook to get available models (including offline models if Ollama is available and debug enabled) -function useAvailableModels() { - const showOfflineFeatures = useAtomValue(showOfflineModeFeaturesAtom) - const { data: ollamaStatus } = trpc.ollama.getStatus.useQuery(undefined, { - refetchInterval: showOfflineFeatures ? 30000 : false, - enabled: showOfflineFeatures, // Only query Ollama when offline mode is enabled - }) - - const baseModels = CLAUDE_MODELS - - const isOffline = ollamaStatus ? !ollamaStatus.internet.online : false - const hasOllama = ollamaStatus?.ollama.available && (ollamaStatus.ollama.models?.length ?? 0) > 0 - const ollamaModels = ollamaStatus?.ollama.models || [] - const recommendedModel = ollamaStatus?.ollama.recommendedModel - - // Only show offline models if: - // 1. Debug flag is enabled (showOfflineFeatures) - // 2. Ollama is available with models - // 3. User is actually offline - if (showOfflineFeatures && hasOllama && isOffline) { - return { - models: baseModels, - ollamaModels, - recommendedModel, - isOffline, - hasOllama: true, - } - } - - return { - models: baseModels, - ollamaModels: [] as string[], - recommendedModel: undefined as string | undefined, - isOffline, - hasOllama: false, - } -} - // Agent providers const agents: { id: string @@ -265,25 +218,14 @@ export function NewChatForm({ isMobileFullscreen = false, onBackToChats }: NewCh }, []) const [workMode, setWorkMode] = useAtom(lastSelectedWorkModeAtom) const debugMode = useAtomValue(agentsDebugModeAtom) - const customClaudeConfig = useAtomValue(customClaudeConfigAtom) - const normalizedCustomClaudeConfig = normalizeCustomClaudeConfig(customClaudeConfig) - const hasCustomClaudeConfig = Boolean(normalizedCustomClaudeConfig) // Connection status for providers - const anthropicOnboardingCompleted = useAtomValue(anthropicOnboardingCompletedAtom) - const apiKeyOnboardingCompleted = useAtomValue(apiKeyOnboardingCompletedAtom) const codexOnboardingCompleted = useAtomValue(codexOnboardingCompletedAtom) - const { data: claudeCodeIntegration } = trpc.claudeCode.getIntegration.useQuery() const { data: cursorIntegration } = trpc.cursor.getIntegration.useQuery() const { data: grokIntegration } = trpc.grok.getIntegration.useQuery() const { data: qwenIntegration } = trpc.qwen.getIntegration.useQuery() const { data: clineIntegration } = trpc.cline.getIntegration.useQuery() const { data: openclawIntegration } = trpc.openclaw.getIntegration.useQuery() const { data: rooIntegration } = trpc.roo.getIntegration.useQuery() - const isClaudeConnected = - Boolean(claudeCodeIntegration?.isConnected) || - anthropicOnboardingCompleted || - apiKeyOnboardingCompleted || - hasCustomClaudeConfig const setSettingsDialogOpen = useSetAtom(agentsSettingsDialogOpenAtom) const setSettingsActiveTab = useSetAtom(agentsSettingsDialogActiveTabAtom) const setJustCreatedIds = useSetAtom(justCreatedIdsAtom) @@ -347,9 +289,15 @@ export function NewChatForm({ isMobileFullscreen = false, onBackToChats }: NewCh } }, [enabledAgents, fallbackAgent, lastSelectedAgentId, selectedAgent.id]) - // Get available models (with offline support) - const availableModels = useAvailableModels() - const [selectedOllamaModel, setSelectedOllamaModel] = useAtom(selectedOllamaModelAtom) + const hiddenModels = useAtomValue(hiddenModelsAtom) + // The Claude half of the picker is shared with the chat composer; which model + // is selected, and what selecting one does, belong to this surface alone. + const { + availableModels, + hasCustomClaudeConfig, + currentOllamaModel, + props: claudePickerProps, + } = useClaudeModelPicker(hiddenModels) const [lastSelectedCodexModelId, setLastSelectedCodexModelId] = useAtom( lastSelectedCodexModelIdAtom, ) @@ -371,13 +319,6 @@ export function NewChatForm({ isMobileFullscreen = false, onBackToChats }: NewCh const [lastSelectedGeminiModelId, setLastSelectedGeminiModelId] = useAtom( lastSelectedGeminiModelIdAtom, ) - const [thinkingEnabled, setThinkingEnabled] = useAtom(extendedThinkingEnabledAtom) - - // The effort rows come from the backend's own capability profile, so a - // provider that reports no effort control shows no sub-menu. - const { data: claudeCapability } = trpc.providers.get.useQuery({ id: "claude" }) - const claudeEfforts = claudeCapability?.features.effort ? EFFORT_LEVELS : [] - const [selectedClaudeEffort, setSelectedClaudeEffort] = useAtom(claudeEffortAtom) const { data: geminiAuth } = trpc.gemini.getAuthStatus.useQuery() const { data: geminiCliStatus } = trpc.gemini.getCliStatus.useQuery() const isGeminiConnected = @@ -410,7 +351,6 @@ export function NewChatForm({ isMobileFullscreen = false, onBackToChats }: NewCh } }, [lastSelectedModelId, selectedModel.id, availableModels.models.find]) - const hiddenModels = useAtomValue(hiddenModelsAtom) const storedCodexApiKey = useAtomValue(codexApiKeyAtom) const hasAppCodexApiKey = Boolean(normalizeCodexApiKey(storedCodexApiKey)) const codexUiModels = useMemo(() => { @@ -596,9 +536,6 @@ export function NewChatForm({ isMobileFullscreen = false, onBackToChats }: NewCh selectedModel?.id, ]) - // Determine current Ollama model (selected or recommended) - const currentOllamaModel = - selectedOllamaModel || availableModels.recommendedModel || availableModels.ollamaModels[0] const claudeAgent = enabledAgents.find((agent) => agent.id === "claude-code") || fallbackAgent const selectedModelLabel = useMemo(() => { if (selectedAgent.id === "codex") { @@ -2197,9 +2134,7 @@ export function NewChatForm({ isMobileFullscreen = false, onBackToChats }: NewCh setSettingsDialogOpen(true) }} claude={{ - models: availableModels.models.filter( - (m) => !hiddenModels.includes(m.id), - ), + ...claudePickerProps, selectedModelId: selectedModel?.id, onSelectModel: (modelId) => { const model = @@ -2209,18 +2144,6 @@ export function NewChatForm({ isMobileFullscreen = false, onBackToChats }: NewCh setSelectedModel(model) setLastSelectedModelId(model.id) }, - hasCustomModelConfig: hasCustomClaudeConfig, - isOffline: availableModels.isOffline && availableModels.hasOllama, - ollamaModels: availableModels.ollamaModels, - selectedOllamaModel: currentOllamaModel, - recommendedOllamaModel: availableModels.recommendedModel, - onSelectOllamaModel: setSelectedOllamaModel, - isConnected: isClaudeConnected, - thinkingEnabled, - onThinkingChange: setThinkingEnabled, - efforts: claudeEfforts, - selectedEffort: selectedClaudeEffort, - onSelectEffort: setSelectedClaudeEffort, }} codex={{ models: codexUiModels, From dbb9457a68d4a4b57eef7ef8c156749a924f5144 Mon Sep 17 00:00:00 2001 From: Owie6789 <151057755+Owie6789@users.noreply.github.com> Date: Wed, 23 Sep 2026 16:12:37 +0000 Subject: [PATCH 19/60] Take the SDK's own sendMessages options type in both transports Both chat transports restated sendMessages(options: { messages: UIMessage[]; abortSignal?: AbortSignal }) and then repeated the same three lines reading the last user message. Both classes implements ChatTransport, so the shape was the library's all along - and restating it narrowed it: the real options carry trigger, chatId, messageId and ChatRequestOptions, which neither transport could see. chat-chunk-atoms.ts, which already holds the shared transport helpers, now also holds SendMessagesOptions = Parameters ["sendMessages"]>[0] and lastUserPrompt(messages). Each transport declares the derived type and reads the turn once, so the two engines can no longer drift on either the contract or the reading. Evidence: no direct sendMessages caller exists in the repo (the AI SDK's Chat calls it), so widening the accepted options changes no call site; tsc --noEmit clean; vitest 1891 passed, 1 skipped; Biome clean. Co-authored-by: arena-agent <297053741+arena-agent@users.noreply.github.com> --- .../features/agents/lib/chat-chunk-atoms.ts | 27 ++++++++++++++++++- .../features/agents/lib/ipc-chat-transport.ts | 14 +++------- .../agents/lib/native-chat-transport.ts | 13 +++------ 3 files changed, 34 insertions(+), 20 deletions(-) diff --git a/src/renderer/features/agents/lib/chat-chunk-atoms.ts b/src/renderer/features/agents/lib/chat-chunk-atoms.ts index 83f64e38..c60dea51 100644 --- a/src/renderer/features/agents/lib/chat-chunk-atoms.ts +++ b/src/renderer/features/agents/lib/chat-chunk-atoms.ts @@ -6,7 +6,7 @@ * handling that is provider-specific (session-init, auth modals, error * toasts) stays in each transport. */ -import type { UIMessageChunk as SDKUIMessageChunk, UIMessage } from "ai" +import type { ChatTransport, UIMessageChunk as SDKUIMessageChunk, UIMessage } from "ai" import type { UIMessageChunk as WireUIMessageChunk } from "../../../../main/lib/claude/types" import type { SessionInfo } from "../../../lib/atoms" import { appStore } from "../../../lib/jotai-store" @@ -232,3 +232,28 @@ export function extractPromptImages(msg: UIMessage | undefined): ImageAttachment return images } + +/** + * What the AI SDK hands a transport on send. Derived from the library's own + * `ChatTransport` contract rather than restated per transport, so both engines + * accept the same options and neither narrows them by hand: when the SDK adds a + * field, both see it without an edit here. + */ +export type SendMessagesOptions = Parameters["sendMessages"]>[0] + +/** + * The newest user turn's prompt text and image attachments. Both engines read + * the last user message the same way, so the reading lives here; a missing + * message yields an empty prompt and no images, which each engine then rejects + * on its own terms. + */ +export function lastUserPrompt(messages: readonly UIMessage[]): { + prompt: string + images: ImageAttachment[] +} { + const lastUser = [...messages].reverse().find((message) => message.role === "user") + return { + prompt: extractPromptText(lastUser), + images: extractPromptImages(lastUser), + } +} diff --git a/src/renderer/features/agents/lib/ipc-chat-transport.ts b/src/renderer/features/agents/lib/ipc-chat-transport.ts index 4b76a1fc..4a044715 100644 --- a/src/renderer/features/agents/lib/ipc-chat-transport.ts +++ b/src/renderer/features/agents/lib/ipc-chat-transport.ts @@ -42,9 +42,9 @@ import { applyQuestionChunks, type ChatChunkContext, clearStalePendingQuestion, - extractPromptImages, - extractPromptText, type ImageAttachment, + lastUserPrompt, + type SendMessagesOptions, type SubscriptionChunk, } from "./chat-chunk-atoms" @@ -373,14 +373,8 @@ function routeChunk( export class IPCChatTransport implements ChatTransport { constructor(private config: IPCChatTransportConfig) {} - async sendMessages(options: { - messages: UIMessage[] - abortSignal?: AbortSignal - }): Promise> { - // Extract prompt and images from last user message - const lastUser = [...options.messages].reverse().find((m) => m.role === "user") - const prompt = extractPromptText(lastUser) - const images = extractPromptImages(lastUser) + async sendMessages(options: SendMessagesOptions): Promise> { + const { prompt, images } = lastUserPrompt(options.messages) // Get sessionId for resume (server preserves sessionId on abort so // the next message can resume with full conversation context) diff --git a/src/renderer/features/agents/lib/native-chat-transport.ts b/src/renderer/features/agents/lib/native-chat-transport.ts index 1510570d..246b37dd 100644 --- a/src/renderer/features/agents/lib/native-chat-transport.ts +++ b/src/renderer/features/agents/lib/native-chat-transport.ts @@ -28,8 +28,8 @@ import { applyCompactingChunks, applyQuestionChunks, clearStalePendingQuestion, - extractPromptImages, - extractPromptText, + lastUserPrompt, + type SendMessagesOptions, type SubscriptionChunk, } from "./chat-chunk-atoms" @@ -74,13 +74,8 @@ const NATIVE_ERROR_TOAST_CONFIG: Record { constructor(private config: NativeChatTransportConfig) {} - async sendMessages(options: { - messages: UIMessage[] - abortSignal?: AbortSignal - }): Promise> { - const lastUser = [...options.messages].reverse().find((m) => m.role === "user") - const prompt = extractPromptText(lastUser) - const images = extractPromptImages(lastUser) + async sendMessages(options: SendMessagesOptions): Promise> { + const { prompt, images } = lastUserPrompt(options.messages) // Read model selection dynamically per sub-chat (so split panes stay independent) const selectedModelId = appStore.get(subChatModelIdAtomFamily(this.config.subChatId)) From b75064c8ec315ab65d9d2d3d04e3c7d862be50ab Mon Sep 17 00:00:00 2001 From: Owie6789 <151057755+Owie6789@users.noreply.github.com> Date: Wed, 23 Sep 2026 16:15:57 +0000 Subject: [PATCH 20/60] Record the duplication sweep in the step-12 decision file Round 2 of the duplication work: what was removed (the all-off feature-flag default, the shared Claude model picker hook, the SDK's own sendMessages options type), what was verified rather than assumed (130 flag values across 10 providers compared key by key against 2c8be15, 0 differences), and what was left standing with reasons - import references, declarative manifest data that has no fail-safe default, the two call sites of the new abstractions, and coincidental windows with pre-existing code. The verification table is re-run in full against head dbb9457: biome 978 files 0 findings, lint 923 files no findings, tsc and the typecheck ratchet 0 errors, vitest 1891 passed 1 skipped, test:node 59 passed, test:contracts 382 passed, audit ratchet passed, 50 of 50 locked skills verified. Co-authored-by: arena-agent <297053741+arena-agent@users.noreply.github.com> --- ...2026-09-23-review-and-sonar-remediation.md | 112 +++++++++++++++++- 1 file changed, 109 insertions(+), 3 deletions(-) diff --git a/.dump/app/decisions/2026-09-23-review-and-sonar-remediation.md b/.dump/app/decisions/2026-09-23-review-and-sonar-remediation.md index 07e5dae7..76e26515 100644 --- a/.dump/app/decisions/2026-09-23-review-and-sonar-remediation.md +++ b/.dump/app/decisions/2026-09-23-review-and-sonar-remediation.md @@ -4,7 +4,8 @@ Roadmap step 12, issue #14, PR #69 on `arena/01a0cbec-mauscode`, review round of 2026-09-23 against head `eb4a2bf`. This is the record of every finding the round produced, what each one was worth, and what was done about it, so the next reader does not have to re-judge a bot's opinion or re-derive why two findings were left -standing. +standing. A second round follows it: the duplication sweep instructed after head +`2c8be15`, recorded under its own heading below against head `dbb9457`. AGENTS.md still says no review bot is configured. That is now false in three directions: Sourcery, CodeAnt and Buoy all review here, and SonarQube Cloud runs a @@ -59,6 +60,110 @@ lines, and the two transports can no longer drift on the question lifecycle. `session-init` deliberately stays per transport, because the native runtime reads a cached snapshot and fills the gaps while the CLI reports the full set on init. +## Round 2: taking the duplication to zero, and where zero stops being the goal + +The instruction after head `2c8be15` was "eliminate all code duplication +correctly". Round 1 removed the two blocks a reviewer would have pointed at; this +round removed what was left that could be removed honestly, and writes down what +stayed and why. The measure is a windowed scan over the whole tree at 4- and +6-line windows, keeping only blocks whose copies contain lines this PR added — +stricter than the gate it feeds, since Sonar's CPD threshold for TypeScript is +about 100 tokens and a six-line window of short object entries is well under it. + +### `ALL_FEATURES_OFF` replaces `TURN_CONTROLS_OFF` + +`TURN_CONTROLS_OFF` named four of the thirteen flags, so every manifest still +restated the other nine and the shared default covered a third of the object. The +flags are not four. They are thirteen booleans that default off, because a +capability nobody proved is a capability this app must not advertise — round 1's +rule, applied to the whole object instead of to the slice that had just grown: + +- `FeatureFlags` is now `z.infer`, exported in place of a + hand-named partial, so the type cannot drift from the schema. +- `ALL_FEATURES_OFF` spells all thirteen `false` under `satisfies FeatureFlags`. + Adding a flag to the schema is a compile error there until its default is chosen + deliberately, and off is the only default that needs no edit. +- Each manifest reads `features: { ...ALL_FEATURES_OFF, }`. + A flag that is off because nobody proved it is no longer written at all; a flag + that is off *for a reason* keeps its reason inline — grok's undocumented `-r` + composition, cline's broken `--id`, openclaw's missing session ids, roo's + rejected prompt, the two unverified skills claims. + +The direction of the default did not move, so neither did the fail-safe: +`ALL_FEATURES_OFF` is a superset of `TURN_CONTROLS_OFF`, and no provider gained a +control it did not have. That was verified rather than assumed. The effective flag +matrix parsed out of every manifest at `2c8be15` was compared key by key against +the rewritten tree: 130 values across 10 providers, 0 differences; the default +itself checked for 13 keys, all `false`, covering every schema flag with none +outside it. Net: −77/+47 lines across eleven files, most of it deleted negatives. + +### One hook for the Claude half of the model picker + +`chat-input-area.tsx` and `new-chat-form.tsx` each carried a byte-identical +37-line `useAvailableModels`, an identical connection test and an identical +26-line `claude={{ ... }}` block — about 110 duplicated lines that had already +drifted once, when the effort rows were added to both surfaces by hand in the +round that shipped adaptive thinking. This duplication predates step 12; step 12 +added to it, which is what made it worth ending. + +`hooks/use-claude-model-picker.ts` now owns the model list and the offline-Ollama +overlay, the custom-config test, the connection test, the resolved Ollama model, +extended thinking, and the capability-driven effort rows. It returns the list plus +a ready props object typed against `AgentModelSelectorProps["claude"]`, exported +from the selector for the purpose, so the block cannot drift from the component it +feeds: add a prop to the selector and the shared object fails typecheck until it +supplies one. The connection test travels inside that object rather than in the +return, because neither surface reads it apart from the picker. + +Two things stay with each surface on purpose: `selectedModelId` and +`onSelectModel`. They are the only lines that differ — the composer also stamps the +sub-chat model id — and pulling them into the hook would mean passing callbacks +back out, which is the same coupling wearing a different hat. Net across the two +surfaces: −177/+22, and the drift class is gone rather than merely smaller. + +### The transports take the SDK's own options type + +Both chat transports restated +`sendMessages(options: { messages: UIMessage[]; abortSignal?: AbortSignal })` and +then repeated the same three lines reading the last user message. Both classes +`implements ChatTransport`, so the shape was the library's all along, +and restating it did not merely duplicate it — it narrowed it: the real options +carry `trigger`, `chatId`, `messageId` and `ChatRequestOptions`, none of which +either transport could see. + +`chat-chunk-atoms.ts`, which already holds the shared transport helpers, now also +holds `SendMessagesOptions = Parameters["sendMessages"]>[0]` +and `lastUserPrompt(messages)`. Each transport declares the derived type and reads +the turn once. This is the one duplication the compiler was already holding equal — +an implementation that stops matching its interface does not typecheck — and it was +still worth removing, because what the copies shared was a loss of contract. + +### Left standing, with reasons + +1. **Import statements.** `import { ALL_FEATURES_OFF, type ProviderCapability } from + "../../../shared/provider-capabilities"` is identical in ten manifests, and the + two surfaces share atom import lines. These are references, not behaviour: + nothing inside them can drift out of step with anything, and the only way to + "share" an import is a barrel module whose entire job is being imported. +2. **Declarative manifest data.** Seven manifests share + `contextWindow: null, latencyClass: "cloud", usageSurface: "native"`, and the two + that share `usageSurface: "none"` share it for the same reason. Unlike the flags, + these have no fail-safe default: a spread that quietly handed a new provider + cloud latency or a native usage surface would be a wrong value inherited + silently, where an off flag is a safe one. Restating a fact per provider is what + a capability table is for. +3. **Call sites of the new abstractions.** `const { availableModels, ... } = + useClaudeModelPicker(hiddenModels)` and `claude={{ ...claudePickerProps, ... }}` + appear in both surfaces because both surfaces use the shared thing. Two call + sites of one hook are the point, not the residue. +4. **Coincidental windows with pre-existing code.** A `break / default: / break` + switch tail in `transform.ts` matches three unrelated files; two providers' + comment prose matches. Neither is a copy of anything this step wrote. + +After this round the scan reports no duplicated block of six or more lines in which +both copies contain lines this PR added, other than the manifest data and the call +sites itemised above. The same scan before round 2 reported 21 such groups. + ## The two complexity findings this step does not take | Function | Reported | What this PR changed inside it | What clearing it needs | @@ -103,18 +208,19 @@ renamed tools take their predecessors' classes. ## Verification in this sandbox +Re-run in full against the round-2 head `dbb9457`; every row below is that run. 2 CPU, 3.9 GB, no Electron binary (the postinstall download is intercepted) and no local build or package. | Gate | Result | | --- | --- | -| `biome check .` | 977 files, 0 findings | +| `biome check .` | 978 files, 0 findings | | `tsc --noEmit` | 0 errors, after building `packages/runtime-client` for its `dist` types | | `ratchet:typecheck` | passed, 0 errors against a 0 baseline | | `vitest run` | 103 files, 1891 passed, 1 skipped | | `npm run test:node` | 59 passed, 0 failed | | `npm run test:contracts` | 23 files, 382 passed | -| `npm run lint` | 922 files checked, no findings | +| `npm run lint` | 923 files checked, no findings | | `ratchet:audit` | passed, 3 critical baseline, no new critical advisories | | `skills:verify` | 50 of 50 locked skills verified, 2 unrecorded project-owned (pre-existing) | | `build`, `package:linux` | not runnable here; CI runs them and the results are recorded on the PR | From 19ffbe18cfdd86274c907e02d0eb601afe1da732 Mon Sep 17 00:00:00 2001 From: Owie6789 <151057755+Owie6789@users.noreply.github.com> Date: Wed, 23 Sep 2026 16:26:10 +0000 Subject: [PATCH 21/60] Record the round-2 gate result and the complexity finding it cleared SonarQube Cloud on head b75064c: quality gate passed, duplication on new code 0.5% (7 duplicated lines of 1284 new ones) against the 3% limit, the 4.2% that failed it and the 1.0% round 1 reached; 0 security hotspots; 1 new issue where round 1 left 2. CodeAnt's five gates pass and it approved. All fifteen substantive GitHub Actions checks pass; Buoy, Sourcery and DeepSource skip as before, and Buoy re-posted the two min-h-[32px] suggestions that are declined on the evidence already recorded. Sonar's per-file measure puts all 7 remaining duplicated lines in provider manifests, one each in cline, codex, cursor, grok, hermes, opencode and qwen; every file an abstraction touched reports zero. The two manifests at zero are the two whose feature blocks carry reasoned false flags with comments, which breaks the window. The cleared issue is NewChatForm's cognitive complexity: e57175b moved the ternary that fed the effort rows and the || chains that resolved the custom config, the connection state and the Ollama model out of the component. typescript:S3776 on assistant-message-item.tsx:803 stands, untouched. Co-authored-by: arena-agent <297053741+arena-agent@users.noreply.github.com> --- ...2026-09-23-review-and-sonar-remediation.md | 41 ++++++++++++++++++- 1 file changed, 40 insertions(+), 1 deletion(-) diff --git a/.dump/app/decisions/2026-09-23-review-and-sonar-remediation.md b/.dump/app/decisions/2026-09-23-review-and-sonar-remediation.md index 76e26515..4244667a 100644 --- a/.dump/app/decisions/2026-09-23-review-and-sonar-remediation.md +++ b/.dump/app/decisions/2026-09-23-review-and-sonar-remediation.md @@ -164,7 +164,35 @@ After this round the scan reports no duplicated block of six or more lines in wh both copies contain lines this PR added, other than the manifest data and the call sites itemised above. The same scan before round 2 reported 21 such groups. -## The two complexity findings this step does not take +### The gate after the sweep + +SonarQube Cloud on head `b75064c`: quality gate passed, duplication on new code +0.5% — 7 duplicated lines out of 1284 new ones — against the 3% limit, the 4.2% +that failed it, and the 1.0% round 1 reached. Security hotspots 0. New issues 1, +down from 2, for the reason recorded in the complexity section below. CodeAnt's +five gates pass and it approved this head. All fifteen substantive GitHub Actions +checks pass: Build on ubuntu-24.04, macos-14 and windows-2022, Package unsigned on +ubuntu-24.04 and macos-14, the lint/test/typecheck quality gates, the security +gates, both Socket reports and CodeRabbit; Buoy, Sourcery and DeepSource skip as +before. Buoy re-posted the same two `min-h-[32px]` → `min-h-8` suggestions and they +are declined again on the evidence already recorded here: the arbitrary form appears +14 times across `src/**/*.tsx`, the token form zero times. + +Sonar's per-file measure says where the 7 lines are, and it is one line in each of +seven provider manifests — cline, codex, cursor, grok, hermes, opencode, qwen. Every +file an abstraction touched reports zero duplicated new lines: both surfaces, the new +hook, the selector, both transports, the shared chunk helpers, +`provider-capabilities.ts`, `transform.ts`, `types.ts`, every test file and +`package.json`. The two manifests that report zero are openclaw and roo, which are +the two whose feature blocks carry reasoned false flags with comments — the comment +lines break the window. One duplicated line per manifest is consistent with the two +things the local scan also flags there, the data tail +(`latencyClass: "cloud", usageSurface: "native"`) and the spread line +(`features: { ...ALL_FEATURES_OFF, chat: true`), and with nothing else those files +contain. Neither is a decision stated twice, which is what item 2 above argues, and +0.5% is what is left when the decisions are each stated once. + +## The complexity findings this step does not take | Function | Reported | What this PR changed inside it | What clearing it needs | | --- | --- | --- | --- | @@ -179,6 +207,17 @@ safe. They are named here so they are handed off rather than quietly dropped, an the transport refactor is the model for how to do them — one handler per side effect, an explicit context, and a single exit path. +**Update from round 2.** One of the two no longer stands, and nothing attacked it: +SonarCloud's issue list for this PR now returns a single issue, `typescript:S3776` +on `assistant-message-item.tsx:803`. Moving the picker wiring out of `NewChatForm` +(`e57175b`) took the ternary that fed the effort rows and the `||` chains that +resolved the custom config, the connection state and the Ollama model with it, and +the component came out under the threshold. A component that stops owning a decision +stops paying for its branches — the same argument as the transport refactor, arrived +at from the other end, and a reason to expect the `renderPart` split to be worth +doing on its own terms rather than as gate relief. `renderPart` is untouched by this +round and stands at 70 against 15. + ## What was declined, and why Buoy asked for `min-h-8` in place of `min-h-[32px]` on two rows of the effort From 41688f4aa68c36bc1fbb585b528b930265f540d1 Mon Sep 17 00:00:00 2001 From: Owie6789 <151057755+Owie6789@users.noreply.github.com> Date: Wed, 23 Sep 2026 18:43:18 +0000 Subject: [PATCH 22/60] Select the Claude model from the list the picker actually shows CodeAnt flagged the same gap twice, once per surface, and it is one gap: the picker received the hidden-model-filtered list while the selection, the label and the id written back to the sub-chat all read the unfiltered one, so hiding a model left it selected, displayed and sent. The other nine providers do not have this shape. Each derives its selection from the filtered list - codexUiModels.find(...) || codexUiModels[0], rooUiModels.find(...) || rooUiModels[0] - so a hidden model cannot be selected at all. Claude instead mirrored that list into useState and synced it with an effect, and the effect could only ever move towards a model it could find, so a model hidden mid-session survived until the next mount. useClaudeModelPicker now returns the visible list as availableModels.models, the same list it hands the picker, and both surfaces derive selectedModel from it with useMemo. The existing materialize effect then persists the visible id to the sub-chat, exactly as the other providers do. When every Claude model is hidden the list is empty and each surface falls back on its own guard: the composer offers nothing, the new-chat form sends its existing "opus" default. Evidence: tsc --noEmit clean; vitest 1891 passed, 1 skipped; Biome clean; no setSelectedModel remains in either surface. Co-authored-by: arena-agent <297053741+arena-agent@users.noreply.github.com> --- .../agents/hooks/use-claude-model-picker.ts | 19 +++++++++++++++++-- .../features/agents/main/chat-input-area.tsx | 19 ++++++++----------- .../features/agents/main/new-chat-form.tsx | 17 ++++++----------- 3 files changed, 31 insertions(+), 24 deletions(-) diff --git a/src/renderer/features/agents/hooks/use-claude-model-picker.ts b/src/renderer/features/agents/hooks/use-claude-model-picker.ts index 694eb74d..1848f098 100644 --- a/src/renderer/features/agents/hooks/use-claude-model-picker.ts +++ b/src/renderer/features/agents/hooks/use-claude-model-picker.ts @@ -9,6 +9,7 @@ * selecting one means there. */ import { useAtom, useAtomValue } from "jotai" +import { useMemo } from "react" import { EFFORT_LEVELS } from "../../../../shared/effort" import { anthropicOnboardingCompletedAtom, @@ -74,7 +75,21 @@ function useAvailableModels() { } export function useClaudeModelPicker(hiddenModels: readonly string[]) { - const availableModels = useAvailableModels() + const modelSets = useAvailableModels() + // Every other provider reads its selection from the list the hidden-model + // setting has already been applied to (`codexUiModels`, `rooUiModels` and the + // rest), and Claude was the exception: the picker received a filtered list + // while the selection, the label and the id written back to the sub-chat all + // read the unfiltered one, so hiding a model left it selected, displayed and + // sent. One list, the visible one, so the two cannot disagree. When every + // Claude model is hidden the list is empty and both surfaces fall back on + // their own — the composer offers nothing, the new-chat form sends its + // existing `?? "opus"` default — which is what hiding them all means. + const models = useMemo( + () => modelSets.models.filter((model) => !hiddenModels.includes(model.id)), + [modelSets.models, hiddenModels], + ) + const availableModels = { ...modelSets, models } // A custom config counts as connected: the turn goes to the endpoint it names // rather than to an account this app signed into. @@ -104,7 +119,7 @@ export function useClaudeModelPicker(hiddenModels: readonly string[]) { const [selectedClaudeEffort, setSelectedClaudeEffort] = useAtom(claudeEffortAtom) const props: SharedClaudePickerProps = { - models: availableModels.models.filter((model) => !hiddenModels.includes(model.id)), + models, hasCustomModelConfig: hasCustomClaudeConfig, isOffline: availableModels.isOffline && availableModels.hasOllama, ollamaModels: availableModels.ollamaModels, diff --git a/src/renderer/features/agents/main/chat-input-area.tsx b/src/renderer/features/agents/main/chat-input-area.tsx index cba6b044..7ffc656f 100644 --- a/src/renderer/features/agents/main/chat-input-area.tsx +++ b/src/renderer/features/agents/main/chat-input-area.tsx @@ -521,20 +521,18 @@ export const ChatInputArea = memo(function ChatInputArea({ currentOllamaModel, props: claudePickerProps, } = useClaudeModelPicker(hiddenModels) - const [selectedModel, setSelectedModel] = useState( + // Derived from the visible list, the way every other provider derives its + // selection (`codexUiModels.find(...) || codexUiModels[0]` and the rest). Local + // state plus a sync effect kept a model that the picker no longer offers: hide + // one mid-session and it stayed selected, labelled and sent until the next + // mount, because the effect only ever moved towards a model it could find. + const selectedModel = useMemo( () => - availableModels.models.find((m) => m.id === selectedSubChatModelId) || + availableModels.models.find((model) => model.id === selectedSubChatModelId) || availableModels.models[0], + [availableModels.models, selectedSubChatModelId], ) - // Sync selectedModel when per-subChat atom value changes (e.g., after localStorage hydration) - useEffect(() => { - const model = availableModels.models.find((m) => m.id === selectedSubChatModelId) - if (model && model.id !== selectedModel.id) { - setSelectedModel(model) - } - }, [availableModels.models, selectedModel.id, selectedSubChatModelId]) - // Materialize the resolved Claude model into per-subChat storage once mounted. // This prevents later global default changes from affecting existing sub-chats. useEffect(() => { @@ -2015,7 +2013,6 @@ export const ChatInputArea = memo(function ChatInputArea({ availableModels.models.find((item) => item.id === modelId) || availableModels.models[0] if (!model) return - setSelectedModel(model) setSelectedSubChatModelId(model.id) setLastSelectedModelId(model.id) }, diff --git a/src/renderer/features/agents/main/new-chat-form.tsx b/src/renderer/features/agents/main/new-chat-form.tsx index 6c3dd7b1..cd8546fa 100644 --- a/src/renderer/features/agents/main/new-chat-form.tsx +++ b/src/renderer/features/agents/main/new-chat-form.tsx @@ -338,19 +338,15 @@ export function NewChatForm({ isMobileFullscreen = false, onBackToChats }: NewCh retry: 1, }) - const [selectedModel, setSelectedModel] = useState( + // Derived from the visible list, as the other nine providers derive theirs, + // so a model hidden in settings cannot stay selected for the next chat. + const selectedModel = useMemo( () => - availableModels.models.find((m) => m.id === lastSelectedModelId) || availableModels.models[0], + availableModels.models.find((model) => model.id === lastSelectedModelId) || + availableModels.models[0], + [availableModels.models, lastSelectedModelId], ) - // Sync selectedModel when atom value changes (e.g., after localStorage hydration) - useEffect(() => { - const model = availableModels.models.find((m) => m.id === lastSelectedModelId) - if (model && model.id !== selectedModel.id) { - setSelectedModel(model) - } - }, [lastSelectedModelId, selectedModel.id, availableModels.models.find]) - const storedCodexApiKey = useAtomValue(codexApiKeyAtom) const hasAppCodexApiKey = Boolean(normalizeCodexApiKey(storedCodexApiKey)) const codexUiModels = useMemo(() => { @@ -2141,7 +2137,6 @@ export function NewChatForm({ isMobileFullscreen = false, onBackToChats }: NewCh availableModels.models.find((m) => m.id === modelId) || availableModels.models[0] if (!model) return - setSelectedModel(model) setLastSelectedModelId(model.id) }, }} From e5b602ae8dc521bbd031ba010b35931e1cbc6b0a Mon Sep 17 00:00:00 2001 From: Owie6789 <151057755+Owie6789@users.noreply.github.com> Date: Wed, 23 Sep 2026 18:43:26 +0000 Subject: [PATCH 23/60] Give the ten backend manifests one probe runner SonarCloud's duplication blocks on this PR's manifests are not the feature flags. They are a sixteen-line execFile wrapper that all ten manifests carried byte-identically, eight of them as runBinary and two as runLaunch, differing only in the bound: fifteen seconds in eight, thirty in openclaw and roo. Ten copies of one rule about what a capability probe may do - run a CLI once, bounded, and resolve rather than reject, because the ordinary failure is a binary that is not installed, and read error.code as a string errno for a spawn failure versus a number for a non-zero exit. providers/probe-command.ts holds it once, with the bound as a named export. The two manifests that have always used thirty seconds pass LAUNCH_PROBE_TIMEOUT_MS explicitly rather than inheriting a default they did not choose. Behaviour at the fifteen call sites is unchanged: same arguments, same result shape, same never-throws contract. codex-models.ts keeps its own codexVersion, which takes an env, uses its own VERSION_TIMEOUT_MS and resolves to a trimmed string or "unknown" - a different contract, not an eleventh copy. Evidence: -195/+46 across the ten manifests, plus 43 lines for the new module; tsc --noEmit clean; vitest 1891 passed, 1 skipped; Biome clean. Co-authored-by: arena-agent <297053741+arena-agent@users.noreply.github.com> --- src/main/lib/providers/claude.ts | 21 ++---------- src/main/lib/providers/cline.ts | 21 ++---------- src/main/lib/providers/codex.ts | 23 ++----------- src/main/lib/providers/cursor.ts | 23 ++----------- src/main/lib/providers/grok.ts | 21 ++---------- src/main/lib/providers/hermes.ts | 25 +++----------- src/main/lib/providers/openclaw.ts | 22 +++--------- src/main/lib/providers/opencode.ts | 23 ++----------- src/main/lib/providers/probe-command.ts | 45 +++++++++++++++++++++++++ src/main/lib/providers/qwen.ts | 21 ++---------- src/main/lib/providers/roo.ts | 22 +++--------- 11 files changed, 74 insertions(+), 193 deletions(-) create mode 100644 src/main/lib/providers/probe-command.ts diff --git a/src/main/lib/providers/claude.ts b/src/main/lib/providers/claude.ts index 337636e6..2ffb413b 100644 --- a/src/main/lib/providers/claude.ts +++ b/src/main/lib/providers/claude.ts @@ -1,29 +1,12 @@ -import { execFile } from "node:child_process" import { existsSync } from "node:fs" import { eq } from "drizzle-orm" import { ALL_FEATURES_OFF, type ProviderCapability } from "../../../shared/provider-capabilities" import { getBundledClaudeBinaryPath } from "../claude/env" import { getExistingClaudeCredentials } from "../claude-token" import { anthropicAccounts, anthropicSettings, claudeCodeCredentials, getDatabase } from "../db" +import { runProbeCommand } from "./probe-command" import type { BackendProbe } from "./types" -function runBinary( - binary: string, - args: string[], -): Promise<{ stdout: string; stderr: string; exitCode: number | null }> { - return new Promise((resolve) => { - execFile(binary, args, { timeout: 15000 }, (error, stdout, stderr) => { - // Spawn failures (ENOENT) carry a string errno, not a numeric code. - const exitCode = error ? (typeof error.code === "number" ? error.code : null) : 0 - resolve({ - stdout: String(stdout ?? ""), - stderr: String(stderr ?? ""), - exitCode, - }) - }) - }) -} - /** * Local-only credential presence (no decrypt, no network): env API key, * app DB tokens (multi-account + legacy), or CLI-managed local credentials. @@ -118,7 +101,7 @@ export async function probeClaude(): Promise { (binary, index) => index > 0 || existsSync(binary), ) for (const binary of candidates) { - const version = await runBinary(binary, ["--version"]) + const version = await runProbeCommand(binary, ["--version"]) if (version.exitCode === 0) { return { available: true, diff --git a/src/main/lib/providers/cline.ts b/src/main/lib/providers/cline.ts index 47ce59bd..5fe04024 100644 --- a/src/main/lib/providers/cline.ts +++ b/src/main/lib/providers/cline.ts @@ -1,26 +1,9 @@ -import { execFile } from "node:child_process" import { ALL_FEATURES_OFF, type ProviderCapability } from "../../../shared/provider-capabilities" import { resolveClineCliLaunch } from "../cline-binary" import { probeClineStoredAuth } from "../cline-print/auth-config" +import { runProbeCommand } from "./probe-command" import type { BackendProbe } from "./types" -function runLaunch( - command: string, - args: string[], -): Promise<{ stdout: string; stderr: string; exitCode: number | null }> { - return new Promise((resolve) => { - execFile(command, args, { timeout: 15000 }, (error, stdout, stderr) => { - // Spawn failures (ENOENT) carry a string errno, not a numeric code. - const exitCode = error ? (typeof error.code === "number" ? error.code : null) : 0 - resolve({ - stdout: String(stdout ?? ""), - stderr: String(stderr ?? ""), - exitCode, - }) - }) - }) -} - export function getClineCapability(): ProviderCapability { return { id: "cline", @@ -102,7 +85,7 @@ export async function probeCline(): Promise { } catch { return { available: false, detail: "cline CLI binary not found" } } - const version = await runLaunch(launch.command, launch.args) + const version = await runProbeCommand(launch.command, launch.args) if (version.exitCode === null) { // Spawn failure (missing/not executable, e.g. a broken $CLINE_BINARY). return { diff --git a/src/main/lib/providers/codex.ts b/src/main/lib/providers/codex.ts index 4fea3deb..31bf016e 100644 --- a/src/main/lib/providers/codex.ts +++ b/src/main/lib/providers/codex.ts @@ -1,7 +1,7 @@ -import { execFile } from "node:child_process" import { join } from "node:path" import { app } from "electron" import { ALL_FEATURES_OFF, type ProviderCapability } from "../../../shared/provider-capabilities" +import { runProbeCommand } from "./probe-command" import type { BackendProbe } from "./types" function resolveCodexBinary(): string { @@ -18,23 +18,6 @@ function resolveCodexBinary(): string { ) } -function runBinary( - binary: string, - args: string[], -): Promise<{ stdout: string; stderr: string; exitCode: number | null }> { - return new Promise((resolve) => { - execFile(binary, args, { timeout: 15000 }, (error, stdout, stderr) => { - // Spawn failures (ENOENT) carry a string errno, not a numeric code. - const exitCode = error ? (typeof error.code === "number" ? error.code : null) : 0 - resolve({ - stdout: String(stdout ?? ""), - stderr: String(stderr ?? ""), - exitCode, - }) - }) - }) -} - export function getCodexCapability(): ProviderCapability { return { id: "codex", @@ -83,9 +66,9 @@ export async function probeCodex(): Promise { const candidates = [resolveCodexBinary(), "codex"] for (const binary of candidates) { try { - const version = await runBinary(binary, ["--version"]) + const version = await runProbeCommand(binary, ["--version"]) if (version.exitCode === 0) { - const login = await runBinary(binary, ["login", "status"]) + const login = await runProbeCommand(binary, ["login", "status"]) const combined = `${login.stdout}\n${login.stderr}`.toLowerCase() return { available: true, diff --git a/src/main/lib/providers/cursor.ts b/src/main/lib/providers/cursor.ts index deb9e1cd..dbd2e098 100644 --- a/src/main/lib/providers/cursor.ts +++ b/src/main/lib/providers/cursor.ts @@ -1,25 +1,8 @@ -import { execFile } from "node:child_process" import { ALL_FEATURES_OFF, type ProviderCapability } from "../../../shared/provider-capabilities" import { resolveCursorAgentCliLaunch } from "../cursor-agent-binary" +import { runProbeCommand } from "./probe-command" import type { BackendProbe } from "./types" -function runLaunch( - command: string, - args: string[], -): Promise<{ stdout: string; stderr: string; exitCode: number | null }> { - return new Promise((resolve) => { - execFile(command, args, { timeout: 15000 }, (error, stdout, stderr) => { - // Spawn failures (ENOENT) carry a string errno, not a numeric code. - const exitCode = error ? (typeof error.code === "number" ? error.code : null) : 0 - resolve({ - stdout: String(stdout ?? ""), - stderr: String(stderr ?? ""), - exitCode, - }) - }) - }) -} - export function getCursorCapability(): ProviderCapability { return { id: "cursor", @@ -75,14 +58,14 @@ export async function probeCursor(): Promise { } catch { return { available: false, detail: "cursor agent binary not found" } } - const version = await runLaunch(launch.command, launch.args) + const version = await runProbeCommand(launch.command, launch.args) if (version.exitCode !== 0) { return { available: false, detail: "cursor agent binary not found" } } // `status` is the documented auth check (displays whether the CLI is // authenticated plus account/endpoint info). const statusLaunch = resolveCursorAgentCliLaunch(["status"]) - const status = await runLaunch(statusLaunch.command, statusLaunch.args) + const status = await runProbeCommand(statusLaunch.command, statusLaunch.args) const combined = `${status.stdout}\n${status.stderr}`.toLowerCase() const loggedOut = combined.includes("not logged in") || diff --git a/src/main/lib/providers/grok.ts b/src/main/lib/providers/grok.ts index 4351dc76..8c2bdccd 100644 --- a/src/main/lib/providers/grok.ts +++ b/src/main/lib/providers/grok.ts @@ -1,27 +1,10 @@ -import { execFile } from "node:child_process" import { existsSync, readFileSync } from "node:fs" import { join } from "node:path" import { ALL_FEATURES_OFF, type ProviderCapability } from "../../../shared/provider-capabilities" import { resolveGrokCliLaunch, resolveGrokHome } from "../grok-binary" +import { runProbeCommand } from "./probe-command" import type { BackendProbe } from "./types" -function runLaunch( - command: string, - args: string[], -): Promise<{ stdout: string; stderr: string; exitCode: number | null }> { - return new Promise((resolve) => { - execFile(command, args, { timeout: 15000 }, (error, stdout, stderr) => { - // Spawn failures (ENOENT) carry a string errno, not a numeric code. - const exitCode = error ? (typeof error.code === "number" ? error.code : null) : 0 - resolve({ - stdout: String(stdout ?? ""), - stderr: String(stderr ?? ""), - exitCode, - }) - }) - }) -} - export function getGrokCapability(): ProviderCapability { return { id: "grok", @@ -140,7 +123,7 @@ export async function probeGrok(): Promise { } catch { return { available: false, detail: "grok CLI binary not found" } } - const version = await runLaunch(launch.command, launch.args) + const version = await runProbeCommand(launch.command, launch.args) if (version.exitCode === null) { // Spawn failure (missing/not executable, e.g. a broken $GROK_BINARY). return { diff --git a/src/main/lib/providers/hermes.ts b/src/main/lib/providers/hermes.ts index c514c874..d169fe21 100644 --- a/src/main/lib/providers/hermes.ts +++ b/src/main/lib/providers/hermes.ts @@ -1,24 +1,7 @@ -import { execFile } from "node:child_process" import { ALL_FEATURES_OFF, type ProviderCapability } from "../../../shared/provider-capabilities" +import { runProbeCommand } from "./probe-command" import type { BackendProbe } from "./types" -function runBinary( - binary: string, - args: string[], -): Promise<{ stdout: string; stderr: string; exitCode: number | null }> { - return new Promise((resolve) => { - execFile(binary, args, { timeout: 15000 }, (error, stdout, stderr) => { - // Spawn failures (ENOENT) carry a string errno, not a numeric code. - const exitCode = error ? (typeof error.code === "number" ? error.code : null) : 0 - resolve({ - stdout: String(stdout ?? ""), - stderr: String(stderr ?? ""), - exitCode, - }) - }) - }) -} - export function getHermesCapability(): ProviderCapability { return { id: "hermes", @@ -67,12 +50,12 @@ export function getHermesCapability(): ProviderCapability { } export async function probeHermes(): Promise { - const version = await runBinary("hermes", ["--version"]) + const version = await runProbeCommand("hermes", ["--version"]) if (version.exitCode !== 0) { return { available: false, detail: "hermes binary not found" } } - const check = await runBinary("hermes", ["acp", "--check"]) - const auth = await runBinary("hermes", ["auth", "status"]) + const check = await runProbeCommand("hermes", ["acp", "--check"]) + const auth = await runProbeCommand("hermes", ["auth", "status"]) return { available: true, version: `${version.stdout} ${version.stderr}`.trim(), diff --git a/src/main/lib/providers/openclaw.ts b/src/main/lib/providers/openclaw.ts index bdc8e0d8..f851135f 100644 --- a/src/main/lib/providers/openclaw.ts +++ b/src/main/lib/providers/openclaw.ts @@ -1,25 +1,11 @@ -import { execFile } from "node:child_process" import { ALL_FEATURES_OFF, type ProviderCapability } from "../../../shared/provider-capabilities" import { resolveOpenclawCliLaunch } from "../openclaw-binary" import { readOpenclawModelsStatus, summarizeModelsStatusAuth } from "../openclaw-print/auth-config" +import { runProbeCommand } from "./probe-command" import type { BackendProbe } from "./types" -function runLaunch( - command: string, - args: string[], -): Promise<{ stdout: string; stderr: string; exitCode: number | null }> { - return new Promise((resolve) => { - execFile(command, args, { timeout: 30000 }, (error, stdout, stderr) => { - // Spawn failures (ENOENT) carry a string errno, not a numeric code. - const exitCode = error ? (typeof error.code === "number" ? error.code : null) : 0 - resolve({ - stdout: String(stdout ?? ""), - stderr: String(stderr ?? ""), - exitCode, - }) - }) - }) -} +/** This manifest's launch probe has always been given twice the shared bound. */ +const LAUNCH_PROBE_TIMEOUT_MS = 30_000 export function getOpenclawCapability(): ProviderCapability { return { @@ -130,7 +116,7 @@ export async function probeOpenclawBinary(): Promise<{ } catch { return { available: false, detail: "openclaw CLI binary not found" } } - const version = await runLaunch(launch.command, launch.args) + const version = await runProbeCommand(launch.command, launch.args, LAUNCH_PROBE_TIMEOUT_MS) if (version.exitCode === null) { // Spawn failure (missing/not executable, e.g. a broken $OPENCLAW_BINARY). return { diff --git a/src/main/lib/providers/opencode.ts b/src/main/lib/providers/opencode.ts index 4acebadb..24968fee 100644 --- a/src/main/lib/providers/opencode.ts +++ b/src/main/lib/providers/opencode.ts @@ -1,24 +1,7 @@ -import { execFile } from "node:child_process" import { ALL_FEATURES_OFF, type ProviderCapability } from "../../../shared/provider-capabilities" +import { runProbeCommand } from "./probe-command" import type { BackendProbe } from "./types" -function runBinary( - binary: string, - args: string[], -): Promise<{ stdout: string; stderr: string; exitCode: number | null }> { - return new Promise((resolve) => { - execFile(binary, args, { timeout: 15000 }, (error, stdout, stderr) => { - // Spawn failures (ENOENT) carry a string errno, not a numeric code. - const exitCode = error ? (typeof error.code === "number" ? error.code : null) : 0 - resolve({ - stdout: String(stdout ?? ""), - stderr: String(stderr ?? ""), - exitCode, - }) - }) - }) -} - export function getOpencodeCapability(): ProviderCapability { return { id: "opencode", @@ -66,11 +49,11 @@ export function getOpencodeCapability(): ProviderCapability { export async function probeOpencode(): Promise { try { - const version = await runBinary("opencode", ["--version"]) + const version = await runProbeCommand("opencode", ["--version"]) if (version.exitCode !== 0) { return { available: false, detail: "opencode binary not found" } } - const auth = await runBinary("opencode", ["auth", "list"]) + const auth = await runProbeCommand("opencode", ["auth", "list"]) const combined = `${auth.stdout}\n${auth.stderr}`.trim() return { available: true, diff --git a/src/main/lib/providers/probe-command.ts b/src/main/lib/providers/probe-command.ts new file mode 100644 index 00000000..4a885283 --- /dev/null +++ b/src/main/lib/providers/probe-command.ts @@ -0,0 +1,45 @@ +import { execFile } from "node:child_process" + +/** What one probe invocation returns: the CLI's own words and how it ended. */ +export type ProbeCommandResult = { + stdout: string + stderr: string + exitCode: number | null +} + +/** + * Bound on a probe invocation. A probe asks a CLI for its version or its auth + * status and nothing else, so one that has not answered within the bound is + * reported as unavailable instead of being waited on. + */ +export const PROBE_TIMEOUT_MS = 15_000 + +/** + * Run a CLI once on behalf of a capability probe, and resolve rather than + * reject, because the failure that matters here is the ordinary one: the binary + * is not installed. `execFile` reports a spawn failure as a string errno + * (`ENOENT`) on `error.code` and a non-zero exit as a number, so `exitCode` is + * `number | null` and callers read `null` as "the process never ran". + * + * All ten backend manifests carried their own byte-identical copy of this — + * eight as `runBinary`, two as `runLaunch` — which is ten places for one rule + * about what a probe may and may not do. The two manifests that have always + * given their launch probe a longer bound pass it explicitly. + */ +export function runProbeCommand( + command: string, + args: string[], + timeoutMs: number = PROBE_TIMEOUT_MS, +): Promise { + return new Promise((resolve) => { + execFile(command, args, { timeout: timeoutMs }, (error, stdout, stderr) => { + // Spawn failures (ENOENT) carry a string errno, not a numeric code. + const exitCode = error ? (typeof error.code === "number" ? error.code : null) : 0 + resolve({ + stdout: String(stdout ?? ""), + stderr: String(stderr ?? ""), + exitCode, + }) + }) + }) +} diff --git a/src/main/lib/providers/qwen.ts b/src/main/lib/providers/qwen.ts index ef4b0ccf..1473eaec 100644 --- a/src/main/lib/providers/qwen.ts +++ b/src/main/lib/providers/qwen.ts @@ -1,26 +1,9 @@ -import { execFile } from "node:child_process" import { ALL_FEATURES_OFF, type ProviderCapability } from "../../../shared/provider-capabilities" import { resolveQwenCliLaunch } from "../qwen-binary" import { probeQwenStoredAuth } from "../qwen-print/auth-config" +import { runProbeCommand } from "./probe-command" import type { BackendProbe } from "./types" -function runLaunch( - command: string, - args: string[], -): Promise<{ stdout: string; stderr: string; exitCode: number | null }> { - return new Promise((resolve) => { - execFile(command, args, { timeout: 15000 }, (error, stdout, stderr) => { - // Spawn failures (ENOENT) carry a string errno, not a numeric code. - const exitCode = error ? (typeof error.code === "number" ? error.code : null) : 0 - resolve({ - stdout: String(stdout ?? ""), - stderr: String(stderr ?? ""), - exitCode, - }) - }) - }) -} - export function getQwenCapability(): ProviderCapability { return { id: "qwen", @@ -101,7 +84,7 @@ export async function probeQwen(): Promise { } catch { return { available: false, detail: "qwen CLI binary not found" } } - const version = await runLaunch(launch.command, launch.args) + const version = await runProbeCommand(launch.command, launch.args) if (version.exitCode === null) { // Spawn failure (missing/not executable, e.g. a broken $QWEN_BINARY). return { diff --git a/src/main/lib/providers/roo.ts b/src/main/lib/providers/roo.ts index ceac74c1..7cab246d 100644 --- a/src/main/lib/providers/roo.ts +++ b/src/main/lib/providers/roo.ts @@ -1,26 +1,12 @@ -import { execFile } from "node:child_process" import { ALL_FEATURES_OFF, type ProviderCapability } from "../../../shared/provider-capabilities" import { getClaudeShellEnvironment } from "../claude/env" import { resolveRooCliLaunch } from "../roo-binary" import { resolveRooAmbientAuth } from "../roo-print/auth-config" +import { runProbeCommand } from "./probe-command" import type { BackendProbe } from "./types" -function runLaunch( - command: string, - args: string[], -): Promise<{ stdout: string; stderr: string; exitCode: number | null }> { - return new Promise((resolve) => { - execFile(command, args, { timeout: 30000 }, (error, stdout, stderr) => { - // Spawn failures (ENOENT) carry a string errno, not a numeric code. - const exitCode = error ? (typeof error.code === "number" ? error.code : null) : 0 - resolve({ - stdout: String(stdout ?? ""), - stderr: String(stderr ?? ""), - exitCode, - }) - }) - }) -} +/** This manifest's launch probe has always been given twice the shared bound. */ +const LAUNCH_PROBE_TIMEOUT_MS = 30_000 export function getRooCapability(): ProviderCapability { return { @@ -127,7 +113,7 @@ export async function probeRooBinary(): Promise<{ } catch { return { available: false, detail: "roo CLI binary not found" } } - const version = await runLaunch(launch.command, launch.args) + const version = await runProbeCommand(launch.command, launch.args, LAUNCH_PROBE_TIMEOUT_MS) if (version.exitCode === null) { // Spawn failure (missing/not executable, e.g. a broken $ROO_BINARY). return { From 5a98fb453f9dd1d9841e8cad1e35ff2715d93a54 Mon Sep 17 00:00:00 2001 From: Owie6789 <151057755+Owie6789@users.noreply.github.com> Date: Wed, 23 Sep 2026 18:59:32 +0000 Subject: [PATCH 24/60] Pin the transcript renderer's output before restructuring it SonarQube reports renderPart in assistant-message-item.tsx at a cognitive complexity of 70 against a threshold of 15, and nothing in the suite rendered any of it: the vitest environment is node, so a ~250-line dispatcher that decides what every kind of message part looks like was verified by reading it. Splitting it on that basis means trusting that the output did not change, which is not verification. So the output is recorded first: 28 snapshots, one message per branch of the dispatcher - text, whitespace-only text, step-start, a part that is neither text nor tool, Bash success and Bash failure, reasoning, a completed thinking tool, Edit with its patch, Write, a plan file as a card, a second plan operation as a mini indicator, web search, web fetch, PlanWrite, ExitPlanMode, a todo list, a question awaiting an answer, a registry tool, the renamed TaskOutput, a subagent task with nested tools, an orphaned nested group, an MCP call, an unregistered tool, the collapsed-steps path, the streaming path, the exploring group and the usage badges. jsdom 30.1.1, @testing-library/react 16.3.3 and its @testing-library/dom 10.4.2 peer join as exact devDependencies, and the vitest include list gains *.test.tsx. The environment stays node globally; this file declares jsdom for itself. The message id is reset per test because it lands in the rendered DOM, so a snapshot cannot depend on how many tests ran before it. No child component reaches for trpc, the router, Sentry or electron, so the only context a render needs is the tooltip provider the agents layout already supplies; the question chime is mocked because jsdom has no Audio and a sound is not part of the output. Evidence: 28 snapshots written, then a second run with 0 written and 28 passed, so the baseline is stable and order-independent; full suite 104 files, 1919 passed, 1 skipped; bun install --frozen-lockfile accepts the regenerated lock (no existing dependency changed version); Biome clean. Co-authored-by: arena-agent <297053741+arena-agent@users.noreply.github.com> --- bun.lock | 105 ++++- package.json | 3 + .../assistant-message-item.test.tsx.snap | 60 +++ .../main/assistant-message-item.test.tsx | 442 ++++++++++++++++++ vitest.config.ts | 5 +- 5 files changed, 607 insertions(+), 8 deletions(-) create mode 100644 src/renderer/features/agents/main/__snapshots__/assistant-message-item.test.tsx.snap create mode 100644 src/renderer/features/agents/main/assistant-message-item.test.tsx diff --git a/bun.lock b/bun.lock index 0954319f..c4998aee 100644 --- a/bun.lock +++ b/bun.lock @@ -103,6 +103,8 @@ "@electron-toolkit/utils": "^4.0.0", "@electron/rebuild": "^4.0.3", "@tailwindcss/container-queries": "^0.1.1", + "@testing-library/dom": "10.4.2", + "@testing-library/react": "16.3.3", "@types/better-sqlite3": "^7.6.13", "@types/diff": "^8.0.0", "@types/node": "^20.17.50", @@ -116,6 +118,7 @@ "electron": "~39.4.0", "electron-builder": "^25.1.8", "electron-vite": "^3.0.0", + "jsdom": "30.1.1", "postcss": "^8.5.1", "sharp": "0.35.4", "tailwindcss": "^3.4.17", @@ -199,6 +202,10 @@ "@apm-js-collab/tracing-hooks": ["@apm-js-collab/tracing-hooks@0.3.1", "", { "dependencies": { "@apm-js-collab/code-transformer": "^0.8.0", "debug": "^4.4.1", "module-details-from-path": "^1.0.4" } }, "sha512-Vu1CbmPURlN5fTboVuKMoJjbO5qcq9fA5YXpskx3dXe/zTBvjODFoerw+69rVBlRLrJpwPqSDqEuJDEKIrTldw=="], + "@asamuzakjp/css-color": ["@asamuzakjp/css-color@7.0.1", "", { "dependencies": { "@csstools/css-calc": "^3.4.0", "@csstools/css-color-parser": "^4.2.3", "@csstools/css-parser-algorithms": "^4.0.0", "@csstools/css-tokenizer": "^4.0.1", "lru-cache": "^11.5.3" } }, "sha512-C9duntabagkBZ1LebM7FKmphR4Q1pBclLxVbZETQV0akkFjV0ooFxo8FlvAQyKj9F6l8rEnmidgcTlpyxFYizg=="], + + "@asamuzakjp/dom-selector": ["@asamuzakjp/dom-selector@9.2.1", "", { "dependencies": { "bidi-js": "^1.1.0", "css-tree": "^3.2.1", "is-potential-custom-element-name": "^1.0.1", "lru-cache": "^11.5.3" } }, "sha512-NT4s3yZLjovPpliRpTvdsdzyPjqRqiCZj9MxnarBihbaY5MbAG7DWAJcrLlhmC7xKTNCXsTELcqB8YsqpLnUFQ=="], + "@babel/code-frame": ["@babel/code-frame@7.28.6", "", { "dependencies": { "@babel/helper-validator-identifier": "^7.28.5", "js-tokens": "^4.0.0", "picocolors": "^1.1.1" } }, "sha512-JYgintcMjRiCvS8mMECzaEn+m3PfoQiyqukOMCCVQtoJGYJw8j/8LBJEiqkHLkfwCcs74E3pbAUFNg7d9VNJ+Q=="], "@babel/compat-data": ["@babel/compat-data@7.28.6", "", {}, "sha512-2lfu57JtzctfIrcGMz992hyLlByuzgIk58+hhGCxjKZ3rWI82NnVLjXcaTqkI2NvlcvOskZaiZ5kjUALo3Lpxg=="], @@ -261,6 +268,8 @@ "@braintree/sanitize-url": ["@braintree/sanitize-url@7.1.1", "", {}, "sha512-i1L7noDNxtFyL5DmZafWy1wRVhGehQmzZaz1HiN5e7iylJMSZR7ekOV7NsIqa5qBldlLrsKv4HbgFUVlQrz8Mw=="], + "@bramus/specificity": ["@bramus/specificity@2.4.2", "", { "dependencies": { "css-tree": "^3.0.0" }, "bin": { "specificity": "bin/cli.js" } }, "sha512-ctxtJ/eA+t+6q2++vj5j7FYX3nRu311q1wfYH3xjlLOsczhlhxAg2FWNUXhpGvAw3BWo1xBcvOV6/YLc2r5FJw=="], + "@chevrotain/cst-dts-gen": ["@chevrotain/cst-dts-gen@11.0.3", "", { "dependencies": { "@chevrotain/gast": "11.0.3", "@chevrotain/types": "11.0.3", "lodash-es": "4.17.21" } }, "sha512-BvIKpRLeS/8UbfxXxgC33xOumsacaeCKAjAeLyOn7Pcp95HiRbrpl14S+9vaZLolnbssPIUuiUd8IvgkRyt6NQ=="], "@chevrotain/gast": ["@chevrotain/gast@11.0.3", "", { "dependencies": { "@chevrotain/types": "11.0.3", "lodash-es": "4.17.21" } }, "sha512-+qNfcoNk70PyS/uxmj3li5NiECO+2YKZZQMbmjTqRI3Qchu8Hig/Q9vgkHpI3alNjr7M+a2St5pw5w5F6NL5/Q=="], @@ -271,6 +280,18 @@ "@chevrotain/utils": ["@chevrotain/utils@11.0.3", "", {}, "sha512-YslZMgtJUyuMbZ+aKvfF3x1f5liK4mWNxghFRv7jqRR9C3R3fAOGTTKvxXDa2Y1s9zSbcpuO0cAxDYsc9SrXoQ=="], + "@csstools/color-helpers": ["@csstools/color-helpers@6.1.1", "", {}, "sha512-gLNsunvwf3mCi5u5o46/Z/JcJMnhbHSaZ69rkgPzNM3J4s8hWwpPUQB6/tt0EDFyCiWzxANlx+2LJwpYj4zS1w=="], + + "@csstools/css-calc": ["@csstools/css-calc@3.4.0", "", { "peerDependencies": { "@csstools/css-parser-algorithms": "^4.0.0", "@csstools/css-tokenizer": "^4.0.0" } }, "sha512-XQKj5B7QiZcHiegCOCAzcAOJdhGgWOHbbu62h5e5mkHnn8lWcfiJhllkqWmxu5zWR9jucPHuo1iTB56P033hcg=="], + + "@csstools/css-color-parser": ["@csstools/css-color-parser@4.2.3", "", { "dependencies": { "@csstools/color-helpers": "^6.1.1", "@csstools/css-calc": "^3.4.0" }, "peerDependencies": { "@csstools/css-parser-algorithms": "^4.0.0", "@csstools/css-tokenizer": "^4.0.0" } }, "sha512-y4LpL+lmpuyKDiEFq2PnZUVFdAjsoB/qQJod79yLNokXyW7jewi+/WJ69EfItj8A2unWtxXnGjw6LYXgXu5ZjA=="], + + "@csstools/css-parser-algorithms": ["@csstools/css-parser-algorithms@4.0.0", "", { "peerDependencies": { "@csstools/css-tokenizer": "^4.0.0" } }, "sha512-+B87qS7fIG3L5h3qwJ/IFbjoVoOe/bpOdh9hAjXbvx0o8ImEmUsGXN0inFOnk2ChCFgqkkGFQ+TpM5rbhkKe4w=="], + + "@csstools/css-syntax-patches-for-csstree": ["@csstools/css-syntax-patches-for-csstree@1.1.14", "", { "peerDependencies": { "css-tree": "^3.2.1" }, "optionalPeers": ["css-tree"] }, "sha512-HpbVXyrofRXpHpgkNIjU/3EWR4WJvOkO3emNK/L6X/mTJU7bGUI3AkkpoTNXznQLp0KRjLHELTGeKI5dIkI9JQ=="], + + "@csstools/css-tokenizer": ["@csstools/css-tokenizer@4.0.1", "", {}, "sha512-bPlN9S9O1A0euCpEWE4qnvB5YDuyYVsUTrxSgmAM1Is0j4tICHoVyOVAXfWMP/kS9ZrjvyIXWV2PmomiAXXqOw=="], + "@develar/schema-utils": ["@develar/schema-utils@2.6.5", "", { "dependencies": { "ajv": "^6.12.0", "ajv-keywords": "^3.4.1" } }, "sha512-0cp4PsWQ/9avqTVMCtZ+GirikIA36ikvjtHweU4/j8yLtgObI0+JUPhYFScgwlteveGB1rt3Cm8UhN04XayDig=="], "@dnd-kit/accessibility": ["@dnd-kit/accessibility@3.1.1", "", { "dependencies": { "tslib": "^2.0.0" }, "peerDependencies": { "react": ">=16.8.0" } }, "sha512-2P+YgaXF+gRsIihwwY1gCsQSYnu9Zyj2py8kY5fFvUM1qm2WA2u639R6YNVfU4GWr+ZM5mqEsfHZZLoRONbemw=="], @@ -363,6 +384,8 @@ "@esbuild/win32-x64": ["@esbuild/win32-x64@0.25.12", "", { "os": "win32", "cpu": "x64" }, "sha512-alJC0uCZpTFrSL0CCDjcgleBXPnCrEAhTBILpeAp7M/OFgoqtAetfBzX0xM00MUsVVPpVjlPuMbREqnZCXaTnA=="], + "@exodus/bytes": ["@exodus/bytes@1.16.0", "", { "peerDependencies": { "@noble/hashes": "^1.8.0 || ^2.0.0" }, "optionalPeers": ["@noble/hashes"] }, "sha512-IcpW84uEn3N7ETtNZMlxKhfl6Pec8rUNGOTBtWbK1FKhJxIFAptZyVrvVRVBimAJxJCgc3PxepxkdWWG4DVzfA=="], + "@floating-ui/core": ["@floating-ui/core@1.7.3", "", { "dependencies": { "@floating-ui/utils": "^0.2.10" } }, "sha512-sGnvb5dmrJaKEZ+LDIpguvdX3bDlEllmv4/ClQ9awcmCZrlx5jQyyMWFM5kBI+EyNOCDDiKk8il0zeuX3Zlg/w=="], "@floating-ui/dom": ["@floating-ui/dom@1.7.4", "", { "dependencies": { "@floating-ui/core": "^1.7.3", "@floating-ui/utils": "^0.2.10" } }, "sha512-OOchDgh4F2CchOX94cRVqhvy7b3AFb+/rQXyswmzmGakRfkMgoWVjfnLWkRirfLEfuD4ysVW16eXzwt3jHIzKA=="], @@ -817,6 +840,10 @@ "@tanstack/virtual-core": ["@tanstack/virtual-core@3.13.18", "", {}, "sha512-Mx86Hqu1k39icq2Zusq+Ey2J6dDWTjDvEv43PJtRCoEYTLyfaPnxIQ6iy7YAOK0NV/qOEmZQ/uCufrppZxTgcg=="], + "@testing-library/dom": ["@testing-library/dom@10.4.2", "", { "dependencies": { "@babel/code-frame": "^7.10.4", "@babel/runtime": "^7.12.5", "@types/aria-query": "^5.0.1", "aria-query": "5.3.0", "dom-accessibility-api": "^0.5.9", "lz-string": "^1.5.0", "picocolors": "1.1.1", "pretty-format": "^27.0.2" } }, "sha512-yzr2S9HyAIdhz2/6qHgbs665Q7PKVcDF05vsOlHPxG1mo36gKVesdYVeDLnXgfjJ03CrKRk08knc6+E/9m8v2Q=="], + + "@testing-library/react": ["@testing-library/react@16.3.3", "", { "dependencies": { "@babel/runtime": "^7.12.5" }, "peerDependencies": { "@testing-library/dom": "^10.0.0", "@types/react": "^18.0.0 || ^19.0.0", "@types/react-dom": "^18.0.0 || ^19.0.0", "react": "^18.0.0 || ^19.0.0", "react-dom": "^18.0.0 || ^19.0.0" }, "optionalPeers": ["@types/react", "@types/react-dom"] }, "sha512-Uo193NgQbPMz6lrrhtRQQFcMC6Re/ELLFbbuVL30WDlZxlpZf9/lMHTAVxPRLw1q1iu9OJmR1c2BLiENRstdBg=="], + "@tootallnate/once": ["@tootallnate/once@2.0.0", "", {}, "sha512-XCuKFP5PS55gnMVu3dty8KPatLqUoy/ZYzDzAGCQ8JNFCkLXzmI7vNHCR+XpbZaMWQK/vQubr7PkYq8g470J/A=="], "@trpc/client": ["@trpc/client@11.8.1", "", { "peerDependencies": { "@trpc/server": "11.8.1", "typescript": ">=5.7.2" } }, "sha512-L/SJFGanr9xGABmuDoeXR4xAdHJmsXsiF9OuH+apecJ+8sUITzVT1EPeqp0ebqA6lBhEl5pPfg3rngVhi/h60Q=="], @@ -825,6 +852,8 @@ "@trpc/server": ["@trpc/server@11.8.1", "", { "peerDependencies": { "typescript": ">=5.7.2" } }, "sha512-P4rzZRpEL7zDFgjxK65IdyH0e41FMFfTkQkuq0BA5tKcr7E6v9/v38DEklCpoDN6sPiB1Sigy/PUEzHENhswDA=="], + "@types/aria-query": ["@types/aria-query@5.0.4", "", {}, "sha512-rfT93uj5s0PRL7EzccGMs3brplhcrghnDoV26NqKhCAS1hVo+WdNsPvE/yb6ilfr5hi2MEk6d5EWJTKdxg8jVw=="], + "@types/babel__core": ["@types/babel__core@7.20.5", "", { "dependencies": { "@babel/parser": "^7.20.7", "@babel/types": "^7.20.7", "@types/babel__generator": "*", "@types/babel__template": "*", "@types/babel__traverse": "*" } }, "sha512-qoQprZvz5wQFJwMDqeseRXWv3rqMvhgpbXFfVyWhbx9X47POIA6i/+dXefEmZKoAgOaTdaIgNSMqMIU61yRyzA=="], "@types/babel__generator": ["@types/babel__generator@7.27.0", "", { "dependencies": { "@babel/types": "^7.0.0" } }, "sha512-ufFd2Xi92OAVPYsy+P4n7/U7e68fex0+Ee8gSG9KX7eo084CWiQ4sdxktvdl0bOPupXtVJPY19zk6EwWqUQ8lg=="], @@ -1037,7 +1066,7 @@ "ansi-regex": ["ansi-regex@5.0.1", "", {}, "sha512-quJQXlTSUGL2LH9SUXo8VwsY4soanhgo6LNSm84E1LBcE8s3O0wpdiRzyR9z/ZZJMlMWv37qOOb9pdJlMUEKFQ=="], - "ansi-styles": ["ansi-styles@4.3.0", "", { "dependencies": { "color-convert": "^2.0.1" } }, "sha512-zbB9rCJAT1rbjiVDb2hqKFHNYLxgtk8NURxZ3IZwD3F6NtxbXZQCnnSi1Lkx+IDohdPlFp222wVALIheZJQSEg=="], + "ansi-styles": ["ansi-styles@5.2.0", "", {}, "sha512-Cxwpt2SfTzTtXcfOlzGEee8O+c+MmUgGrNiBcXnuWxuFJHe6a5Hz7qwhwe5OgaSYI0IJvkLqWX1ASG+cJOkEiA=="], "any-promise": ["any-promise@1.3.0", "", {}, "sha512-7UvmKalWRt1wgjL1RrGxoSJW/0QZFIegpeGvZG9kjp8vrRu55XTHbwnqq2GpXm9uLbcuhxm3IqX9OB4MZR1b2A=="], @@ -1061,6 +1090,8 @@ "aria-hidden": ["aria-hidden@1.2.6", "", { "dependencies": { "tslib": "^2.0.0" } }, "sha512-ik3ZgC9dY/lYVVM++OISsaYDeg1tb0VtP5uL3ouh1koGOaUMDPpbFIei4JkFimWUFPn90sbMNMXQAIVOlnYKJA=="], + "aria-query": ["aria-query@5.3.0", "", { "dependencies": { "dequal": "^2.0.3" } }, "sha512-b0P0sZPKtyu8HkeRAfCq0IfURZK+SuwMjY1UXGBU27wpAiTwQAIlq56IbIO+ytk/JjS1fMR14ee5WBBfKi5J6A=="], + "assert-plus": ["assert-plus@1.0.0", "", {}, "sha512-NfJ4UzBCcQGLDlQq7nHxH+tv3kyZ0hHQqF5BO6J7tNJeP5do1llPr8dZ8zHonfhAu0PHAdMkSo+8o0wxg9lZWw=="], "assertion-error": ["assertion-error@2.0.1", "", {}, "sha512-Izi8RQcffqCeNVgFigKli1ssklIbpHnCYc6AknXGYoB6grJqyeby7jv12JUQgmTAnIDnbck1uxksT4dzN3PWBA=="], @@ -1089,6 +1120,8 @@ "better-sqlite3": ["better-sqlite3@12.6.2", "", { "dependencies": { "bindings": "^1.5.0", "prebuild-install": "^7.1.1" } }, "sha512-8VYKM3MjCa9WcaSAI3hzwhmyHVlH8tiGFwf0RlTsZPWJ1I5MkzjiudCo4KC4DxOaL/53A5B1sI/IbldNFDbsKA=="], + "bidi-js": ["bidi-js@1.1.0", "", { "dependencies": { "require-from-string": "^2.0.2" } }, "sha512-fX1Onk0tdVPC7obPWB5EbJ1z7NVhLq4m2xZLq2YXBkxzMXIGRpNMU88n0EPgWseKl12J7zXs7qrDxPK4sRs2fg=="], + "binary-extensions": ["binary-extensions@2.3.0", "", {}, "sha512-Ceh+7ox5qe7LJuLHoY0feh3pHuUDHAcRUeyL2VYghZwfpkNIy/+8Ocg0a3UuSoYzavmylwuLWQOf3hl0jjMMIw=="], "bindings": ["bindings@1.5.0", "", { "dependencies": { "file-uri-to-path": "1.0.0" } }, "sha512-p2q/t/mhvuOj/UeLlV6566GD/guowlr0hHxClI0W9m7MWYkL1F0hLo+0Aexs9HSPCtR1SXQ0TD3MMKrXZajbiQ=="], @@ -1237,6 +1270,8 @@ "cross-spawn": ["cross-spawn@7.0.6", "", { "dependencies": { "path-key": "^3.1.0", "shebang-command": "^2.0.0", "which": "^2.0.1" } }, "sha512-uV2QOWP2nWzsy2aMp8aRibhi9dlzF5Hgh5SHaB9OiTGEyDTiJJyx0uy51QXdyWbtAHNua4XJzUKca3OzKUd3vA=="], + "css-tree": ["css-tree@3.2.1", "", { "dependencies": { "mdn-data": "2.27.1", "source-map-js": "^1.2.1" } }, "sha512-X7sjQzceUhu1u7Y/ylrRZFU2FS6LRiFVp6rKLPg23y3x3c3DOKAwuXGDp+PAGjh6CSnCjYeAul8pcT8bAl+lSA=="], + "cssesc": ["cssesc@3.0.0", "", { "bin": { "cssesc": "bin/cssesc" } }, "sha512-/Tb/JcjK111nNScGob5MNtsntNM1aCNUDipB/TkwZFhyDrrE47SOx/18wF2bbjgc3ZzCSKW1T5nt5EbFoAz/Vg=="], "csstype": ["csstype@3.2.3", "", {}, "sha512-z1HGKcYy2xA8AGQfwrn0PAy+PB7X/GSj3UVJW9qKyn43xWa+gl5nXmU4qqLMRzWVLFC8KusUX8T/0kCiOYpAIQ=="], @@ -1313,12 +1348,16 @@ "dagre-d3-es": ["dagre-d3-es@7.0.13", "", { "dependencies": { "d3": "^7.9.0", "lodash-es": "^4.17.21" } }, "sha512-efEhnxpSuwpYOKRm/L5KbqoZmNNukHa/Flty4Wp62JRvgH2ojwVgPgdYyr4twpieZnyRDdIH7PY2mopX26+j2Q=="], + "data-urls": ["data-urls@7.0.0", "", { "dependencies": { "whatwg-mimetype": "^5.0.0", "whatwg-url": "^16.0.0" } }, "sha512-23XHcCF+coGYevirZceTVD7NdJOqVn+49IHyxgszm+JIiHLoB2TkmPtsYkNWT1pvRSGkc35L6NHs0yHkN2SumA=="], + "date-fns": ["date-fns@3.6.0", "", {}, "sha512-fRHTG8g/Gif+kSh50gaGEdToemgfj74aRX3swtiouboip5JDLAyDE9F11nHMIcvOaXeOC6D7SpNhi7uFyB7Uww=="], "dayjs": ["dayjs@1.11.19", "", {}, "sha512-t5EcLVS6QPBNqM2z8fakk/NKel+Xzshgt8FFKAn+qwlD1pzZWxh0nVCrvFK7ZDb6XucZeF9z8C7CBWTRIVApAw=="], "debug": ["debug@4.4.3", "", { "dependencies": { "ms": "^2.1.3" } }, "sha512-RGwwWnwQvkVfavKVt22FGLw+xYSdzARwm0ru6DhTVA3umU5hZc28V3kO4stgYryrTlLpuvgI9GiijltAjNbcqA=="], + "decimal.js": ["decimal.js@10.6.0", "", {}, "sha512-YpgQiITW3JXGntzdUmyUR1V812Hn8T1YVXhCu+wO3OpS4eU9l4YdD3qjyiKdV6mvV29zapkMeD390UVEf2lkUg=="], + "decode-named-character-reference": ["decode-named-character-reference@1.3.0", "", { "dependencies": { "character-entities": "^2.0.0" } }, "sha512-GtpQYB283KrPp6nRw50q3U9/VfOutZOe103qlN7BPP6Ad27xYnOIWv4lPzo8HCAL+mMZofJ9KEy30fq6MfaK6Q=="], "decompress-response": ["decompress-response@6.0.0", "", { "dependencies": { "mimic-response": "^3.1.0" } }, "sha512-aW35yZM6Bb/4oJlZncMH2LCoZtJXTRxES17vE3hoRiowU2kWHaJKFkSBDnDR+cm9J+9QhXmREyIfv0pji9ejCQ=="], @@ -1363,6 +1402,8 @@ "dmg-license": ["dmg-license@1.0.11", "", { "dependencies": { "@types/plist": "^3.0.1", "@types/verror": "^1.10.3", "ajv": "^6.10.0", "crc": "^3.8.0", "iconv-corefoundation": "^1.1.7", "plist": "^3.0.4", "smart-buffer": "^4.0.2", "verror": "^1.10.0" }, "os": "darwin", "bin": { "dmg-license": "bin/dmg-license.js" } }, "sha512-ZdzmqwKmECOWJpqefloC5OJy1+WZBBse5+MR88z9g9Zn4VY+WYUkAyojmhzJckH5YbbZGcYIuGAkY5/Ys5OM2Q=="], + "dom-accessibility-api": ["dom-accessibility-api@0.5.16", "", {}, "sha512-X7BJ2yElsnOJ30pZF4uIIDfBEVgF4XEBxL9Bxhy6dnrm5hkzqmsWHGTiHqRiITNhMyFLyAiWndIJP7Z1NTteDg=="], + "dompurify": ["dompurify@3.3.1", "", { "optionalDependencies": { "@types/trusted-types": "^2.0.7" } }, "sha512-qkdCKzLNtrgPFP1Vo+98FRzJnBRGe4ffyCea9IwHB1fyxPOeNTHpLKYGd4Uk9xvNoH0ZoOjwZxNptyMwqrId1Q=="], "dotenv": ["dotenv@16.6.1", "", {}, "sha512-uBq4egWHTcTt33a72vpSG0z3HnPuIl6NqYcTrKEg2azoEyl2hpW0zqlxysq2pK9HlDIHyHyakeYaYnSAwd8bow=="], @@ -1409,7 +1450,7 @@ "end-of-stream": ["end-of-stream@1.4.5", "", { "dependencies": { "once": "^1.4.0" } }, "sha512-ooEGc6HP26xXq/N+GCGOT0JKCLDGrq2bQUZrQ7gyrJiZANJ/8YDTxTpQBXGMn+WbIQXNVpyWymm7KYVICQnyOg=="], - "entities": ["entities@6.0.1", "", {}, "sha512-aN97NXWF6AWBTahfVOIrB/NShkzi5H7F9r1s9mD3cDj4Ko5f2qhhVoYMibXF7GlLveb/D2ioWay8lxI97Ven3g=="], + "entities": ["entities@8.1.0", "", {}, "sha512-kxL7msIffSuh9aaFAMD7rxAIuTRMAHMeBtgHW2yUdWw732ZNh4MehkF2gdjvtdmikkaIP9bFDDJOPlsvm7avrA=="], "env-paths": ["env-paths@2.2.1", "", {}, "sha512-+h1lkLKhZMTYjog1VEpJNG7NZJWcuc2DDk/qsqSTRRCOXiLjeQ1d1/udrUGhqMxUgAlwKNZ0cf2uqan5GLuS2A=="], @@ -1595,6 +1636,8 @@ "hosted-git-info": ["hosted-git-info@4.1.0", "", { "dependencies": { "lru-cache": "^6.0.0" } }, "sha512-kyCuEOWjJqZuDbRHzL8V93NzQhwIB71oFWSyzVo+KPZI+pnQPPxucdkrOZvkLRnrf5URsQM+IJ09Dw29cRALIA=="], + "html-encoding-sniffer": ["html-encoding-sniffer@7.0.0", "", { "dependencies": { "@exodus/bytes": "^1.15.1" } }, "sha512-UikN5yr7xsCDAq87Or5or0PAlD3HJJOKVzM05az588WnpDJ4Ux7a2A53Qi6gofGg2/EtvF/H4hCi/TXfCW4Y6w=="], + "html-url-attributes": ["html-url-attributes@3.0.1", "", {}, "sha512-ol6UPyBWqsrO6EJySPz2O7ZSr856WDrEzM5zMqp+FJJLGMW35cLYmmZnl0vztAZxRUoNZJFTCohfjuIJ8I4QBQ=="], "html-void-elements": ["html-void-elements@3.0.0", "", {}, "sha512-bEqo66MRXsUGxWHV5IP0PUiAWwoEjba4VCzg0LjFJBpchPaTfyfCKTG6bc5F8ucKec3q5y6qOdGyYTSBEvhCrg=="], @@ -1669,6 +1712,8 @@ "is-plain-obj": ["is-plain-obj@4.1.0", "", {}, "sha512-+Pgi+vMuUNkJyExiMBt5IlFoMyKnr5zhJ4Uspz58WOhBF5QoIZkFyNHIbBAtHwzVAgk5RtndVNsDRN61/mmDqg=="], + "is-potential-custom-element-name": ["is-potential-custom-element-name@1.0.1", "", {}, "sha512-bCYeRA2rVibKZd+s2625gGnGF/t7DSqDs4dP7CrLA1m7jKWz6pps0LpYLJN8Q64HtmPKJ1hrN3nzPNKFEKOUiQ=="], + "is-promise": ["is-promise@4.0.0", "", {}, "sha512-hvpoI6korhJMnej285dSg6nu1+e6uxs7zG3BYAm5byqDsgJNWwxzM6z6iZiAgQR4TJ30JmBTOwqZUw3WlyH3AQ=="], "is-unicode-supported": ["is-unicode-supported@0.1.0", "", {}, "sha512-knxG2q4UC3u8stRGyAVJCOdxFmv5DZiRcdlIaAQXAbSfJya+OhopNotLQrstBhququ4ZpuKbDc/8S6mgXgPFPw=="], @@ -1695,6 +1740,8 @@ "js-yaml": ["js-yaml@4.1.1", "", { "dependencies": { "argparse": "^2.0.1" }, "bin": { "js-yaml": "bin/js-yaml.js" } }, "sha512-qQKT4zQxXl8lLwBtHMWwaTcGfFOZviOJet3Oy/xmGk2gZH677CJM9EvtfdSkgWcATZhj/55JZ0rmy3myCT5lsA=="], + "jsdom": ["jsdom@30.1.1", "", { "dependencies": { "@asamuzakjp/css-color": "^7.0.0", "@asamuzakjp/dom-selector": "^9.2.1", "@bramus/specificity": "^2.4.2", "@csstools/css-syntax-patches-for-csstree": "^1.1.13", "@exodus/bytes": "^1.15.1", "css-tree": "^3.2.1", "data-urls": "^7.0.0", "decimal.js": "^10.6.0", "html-encoding-sniffer": "^7.0.0", "is-potential-custom-element-name": "^1.0.1", "lru-cache": "^11.5.2", "parse5": "^8.0.1", "saxes": "^6.0.0", "tough-cookie": "^6.0.2", "undici": "^8.10.2", "w3c-xmlserializer": "^6.0.0", "webidl-conversions": "^8.0.1", "whatwg-mimetype": "^5.0.0", "whatwg-url": "^17.1.1", "xml-name-validator": "^5.0.0" }, "peerDependencies": { "canvas": "^3.2.3" }, "optionalPeers": ["canvas"] }, "sha512-FahmoPK5vbPc+jxV1iErMHmAZypCZ942NHF4+qqaWAuvaKKTBZxawnmAtrbGWLU7MtlxfqIP0qw6aSI+aWGtLg=="], + "jsesc": ["jsesc@3.1.0", "", { "bin": { "jsesc": "bin/jsesc" } }, "sha512-/sM3dO2FOzXjKQhJuo0Q173wf2KOo8t4I8vHy6lF9poUp7bKT0/NHE8fPX23PwfhnykfqnC2xRxOnVw5XuGIaA=="], "json-buffer": ["json-buffer@3.0.1", "", {}, "sha512-4bV5BfR2mqfQTJm+V5tPPdf+ZpuhiIvTuAB5g8kcrXOZpTT/QwwVRWBywX1ozr6lEuPdbHxwaJlm9G6mI2sfSQ=="], @@ -1763,12 +1810,14 @@ "lowlight": ["lowlight@3.3.0", "", { "dependencies": { "@types/hast": "^3.0.0", "devlop": "^1.0.0", "highlight.js": "~11.11.0" } }, "sha512-0JNhgFoPvP6U6lE/UdVsSq99tn6DhjjpAj5MxG49ewd2mOBVtwWYIT8ClyABhq198aXXODMU6Ox8DrGy/CpTZQ=="], - "lru-cache": ["lru-cache@5.1.1", "", { "dependencies": { "yallist": "^3.0.2" } }, "sha512-KpNARQA3Iwv+jTA0utUVVbrh+Jlrr1Fv0e56GGzAFOXN7dk/FviaDW8LHmK52DlcH4WP2n6gI8vN1aesBFgo9w=="], + "lru-cache": ["lru-cache@11.5.3", "", {}, "sha512-U4N8FgzmWxc8k1VH8Kr6lQg18U7Fjvby6wXHVRX/ZZ7IwWbRMgrRbP0Wrb5q5NVinryp4SQampHKdvtecItxUg=="], "lru_map": ["lru_map@0.4.1", "", {}, "sha512-I+lBvqMMFfqaV8CJCISjI3wbjmwVu/VyOoU7+qtu9d7ioW5klMgsTTiUOUp+DJvfTTzKXoPbyC6YfgkNcyPSOg=="], "lucide-react": ["lucide-react@0.468.0", "", { "peerDependencies": { "react": "^16.5.1 || ^17.0.0 || ^18.0.0 || ^19.0.0-rc" } }, "sha512-6koYRhnM2N0GGZIdXzSeiNwguv1gt/FAjZOiPl76roBi3xKEXa4WmfpxgQwTTL4KipXjefrnf3oV4IsYhi4JFA=="], + "lz-string": ["lz-string@1.5.0", "", { "bin": { "lz-string": "bin/bin.js" } }, "sha512-h5bgJWpxJNswbU7qCrV0tIKQCaS3blPDrqKWx+QxzuzL1zGUzij9XCWLrSLsJPu5t+eWA/ycetzYAO5IOMcWAQ=="], + "magic-string": ["magic-string@0.30.21", "", { "dependencies": { "@jridgewell/sourcemap-codec": "^1.5.5" } }, "sha512-vd2F4YUyEXKGcLHoq+TEyCjxueSeHnFxyyjNp80yg0XV4vUhnDer/lvvlqM/arB5bXQN5K2/3oinyCRyx8T2CQ=="], "make-fetch-happen": ["make-fetch-happen@14.0.3", "", { "dependencies": { "@npmcli/agent": "^3.0.0", "cacache": "^19.0.1", "http-cache-semantics": "^4.1.1", "minipass": "^7.0.2", "minipass-fetch": "^4.0.0", "minipass-flush": "^1.0.5", "minipass-pipeline": "^1.2.4", "negotiator": "^1.0.0", "proc-log": "^5.0.0", "promise-retry": "^2.0.1", "ssri": "^12.0.0" } }, "sha512-QMjGbFTP0blj97EeidG5hk/QhKQ3T4ICckQGLgz38QF7Vgbk6e6FTARN8KhKxyBbWn8R0HU+bnw8aSoFPD4qtQ=="], @@ -1813,6 +1862,8 @@ "mdast-util-to-string": ["mdast-util-to-string@4.0.0", "", { "dependencies": { "@types/mdast": "^4.0.0" } }, "sha512-0H44vDimn51F0YwvxSJSm0eCDOJTRlmN0R1yBh4HLj9wiV1Dn0QoXGbvFAWj2hSItVTlCmBF1hqKlIyUBVFLPg=="], + "mdn-data": ["mdn-data@2.27.1", "", {}, "sha512-9Yubnt3e8A0OKwxYSXyhLymGW4sCufcLG6VdiDdUGVkPhpqLxlvP5vl1983gQjJl3tqbrM731mjaZaP68AgosQ=="], + "media-typer": ["media-typer@1.1.0", "", {}, "sha512-aisnrDP4GNe06UcKFnV5bfMNPBUw4jsLGaWwWfnH3v02GnBuXX2MCVn5RbrWo0j3pczUilYblq7fQ7Nw2t5XKw=="], "merge-descriptors": ["merge-descriptors@2.0.0", "", {}, "sha512-Snk314V5ayFLhp3fkUREub6WtjBfPdCPY1Ln8/8munuLuiYhsABgBVWsozAG+MWMbVEvcdcpbi9R7ww22l9Q3g=="], @@ -1995,7 +2046,7 @@ "parse-entities": ["parse-entities@4.0.2", "", { "dependencies": { "@types/unist": "^2.0.0", "character-entities-legacy": "^3.0.0", "character-reference-invalid": "^2.0.0", "decode-named-character-reference": "^1.0.0", "is-alphanumerical": "^2.0.0", "is-decimal": "^2.0.0", "is-hexadecimal": "^2.0.0" } }, "sha512-GG2AQYWoLgL877gQIKeRPGO1xF9+eG1ujIb5soS5gPvLQ1y2o8FL90w2QWNdf9I361Mpp7726c+lj3U0qK1uGw=="], - "parse5": ["parse5@7.3.0", "", { "dependencies": { "entities": "^6.0.0" } }, "sha512-IInvU7fabl34qmi9gY8XOVxhYyMyuH2xUNpb2q8/Y+7552KlejkRvqvD19nMoUW/uQGGbqNpA6Tufu5FL5BZgw=="], + "parse5": ["parse5@8.0.1", "", { "dependencies": { "entities": "^8.0.0" } }, "sha512-z1e/HMG90obSGeidlli3hj7cbocou0/wa5HacvI3ASx34PecNjNQeaHNo5WIZpWofN9kgkqV1q5YvXe3F0FoPw=="], "parseurl": ["parseurl@1.3.3", "", {}, "sha512-CiyeOxFT/JZyN5m0z9PfXw4SCBJ6Sygz1Dpl0wqjlhDEGGBP1GnsUVEL0p63hoG1fcj3fHynXi9NYO4nWOL+qQ=="], @@ -2073,6 +2124,8 @@ "prebuild-install": ["prebuild-install@7.1.3", "", { "dependencies": { "detect-libc": "^2.0.0", "expand-template": "^2.0.3", "github-from-package": "0.0.0", "minimist": "^1.2.3", "mkdirp-classic": "^0.5.3", "napi-build-utils": "^2.0.0", "node-abi": "^3.3.0", "pump": "^3.0.0", "rc": "^1.2.7", "simple-get": "^4.0.0", "tar-fs": "^2.0.0", "tunnel-agent": "^0.6.0" }, "bin": { "prebuild-install": "bin.js" } }, "sha512-8Mf2cbV7x1cXPUILADGI3wuhfqWvtiLA1iclTDbFRZkgRQS0NqsPZphna9V+HyTEadheuPmjaJMsbzKQFOzLug=="], + "pretty-format": ["pretty-format@27.5.1", "", { "dependencies": { "ansi-regex": "^5.0.1", "ansi-styles": "^5.0.0", "react-is": "^17.0.1" } }, "sha512-Qb1gy5OrP5+zDf2Bvnzdl3jsTf1qXVMazbvCoKhtKqVs4/YK4ozX4gKQJJVyNe+cajNPn0KoC0MC3FUmaHWEmQ=="], + "proc-log": ["proc-log@5.0.0", "", {}, "sha512-Azwzvl90HaF0aCz1JrDdXQykFakSSNPaPoiZ9fm5qJIMHioDZEi7OAdRwSm6rSoPtY3Qutnm3L7ogmg3dc+wbQ=="], "process-nextick-args": ["process-nextick-args@2.0.1", "", {}, "sha512-3ouUOpQhtgrbOa17J7+uxOTpITYWaGP7/AhoR3+A+/1e9skrzelGi/dXzEYyvbxubEF6Wn2ypscTKiKJFFn1ag=="], @@ -2117,6 +2170,8 @@ "react-icons": ["react-icons@5.5.0", "", { "peerDependencies": { "react": "*" } }, "sha512-MEFcXdkP3dLo8uumGI5xN3lDFNsRtrjbOEKDLD7yv76v4wpnEq2Lt2qeHaQOr34I/wPN3s3+N08WkQ+CW37Xiw=="], + "react-is": ["react-is@17.0.2", "", {}, "sha512-w2GsyukL62IJnlaff/nRegPQR94C/XXamvMWmSHRJ4y7Ts/4ocGRmTHvOs8PSE6pB3dWOrD/nueuU5sduBsQ4w=="], + "react-refresh": ["react-refresh@0.17.0", "", {}, "sha512-z6F7K9bV85EfseRCp2bzrpyQ0Gkw1uLoCel9XBVWPg/TjRj94SkJzUTGfOa4bs7iJvBWtQG0Wq7wnI0syw3EBQ=="], "react-remove-scroll": ["react-remove-scroll@2.7.2", "", { "dependencies": { "react-remove-scroll-bar": "^2.3.7", "react-style-singleton": "^2.2.3", "tslib": "^2.1.0", "use-callback-ref": "^1.3.3", "use-sidecar": "^1.1.3" }, "peerDependencies": { "@types/react": "*", "react": "^16.8.0 || ^17.0.0 || ^18.0.0 || ^19.0.0 || ^19.0.0-rc" }, "optionalPeers": ["@types/react"] }, "sha512-Iqb9NjCCTt6Hf+vOdNIZGdTiH1QSqr27H/Ek9sv/a97gfueI/5h1s3yRi1nngzMUaOOToin5dI1dXKdXiF+u0Q=="], @@ -2211,6 +2266,8 @@ "sax": ["sax@1.4.4", "", {}, "sha512-1n3r/tGXO6b6VXMdFT54SHzT9ytu9yr7TaELowdYpMqY/Ao7EnlQGmAQ1+RatX7Tkkdm6hONI2owqNx2aZj5Sw=="], + "saxes": ["saxes@6.0.0", "", { "dependencies": { "xmlchars": "^2.2.0" } }, "sha512-xAg7SOnEhrm5zI3puOOKyy1OMcMlIJZYNJY7xLBwSze0UjhPLnWfj2GF2EpT0jmzaJKIWKHLsaSSajf35bcYnA=="], + "scheduler": ["scheduler@0.27.0", "", {}, "sha512-eNv+WrVbKu1f3vbYJT/xtiF5syA5HPIMtf9IgY/nKg0sWqzAUEvqY/xm7OcZc/qafLx/iO9FgOmeSAp4v5ti/Q=="], "section-matter": ["section-matter@1.0.0", "", { "dependencies": { "extend-shallow": "^2.0.1", "kind-of": "^6.0.0" } }, "sha512-vfD3pmTzGpufjScBh50YHKzEu2lxBWhVEHsNGoEXmCmn2hKGfeNLYMzCJpe8cD7gqX7TJluOVpBkAequ6dgMmA=="], @@ -2357,6 +2414,10 @@ "tinyrainbow": ["tinyrainbow@3.1.1", "", {}, "sha512-yau8yJdTt989Mm0Bd/236QnzEiPf2xLLTqUZRUJOo/3CB078LSwzei343DgtJVmfJKJE3TMINY1u42SQsP6mXw=="], + "tldts": ["tldts@7.4.14", "", { "dependencies": { "tldts-core": "^7.4.14" }, "bin": { "tldts": "bin/cli.js" } }, "sha512-EahQoi+Q5oqmG3yxRHx+bwn36OaLbtw6jnTFcGjc8fcmktzTgk45xRnzAx8Ch87PFEiqRhoc3FWBmY+LBrvEGQ=="], + + "tldts-core": ["tldts-core@7.4.14", "", {}, "sha512-KYkjfJHnIC5t+Gy7hPrMHgJmWc/N9FfmS+fggRPSlpFckidpB9HD6OzPsSTkfDH+MpZgcu2tJPGgLfHl979N1Q=="], + "tmp": ["tmp@0.2.5", "", {}, "sha512-voyz6MApa1rQGUxT3E+BK7/ROe8itEx7vD8/HEvt4xwXucvQ5G5oeEiHkmHZJuBO21RpOf+YYm9MOivj709jow=="], "tmp-promise": ["tmp-promise@3.0.3", "", { "dependencies": { "tmp": "^0.2.0" } }, "sha512-RwM7MoPojPxsOBYnyd2hy0bxtIlVrihNs9pj5SUvY8Zz1sQcQG2tG1hSr8PDxfgEB8RNKDhqbIlroIarSNDNsQ=="], @@ -2365,6 +2426,10 @@ "toidentifier": ["toidentifier@1.0.1", "", {}, "sha512-o5sSPKEkg/DIQNmH43V0/uerLrpzVedkUh8tGNvaeXpfpuwjKenlSox/2O/BTlZUtEe+JG7s5YhEz608PlAHRA=="], + "tough-cookie": ["tough-cookie@6.0.2", "", { "dependencies": { "tldts": "^7.0.5" } }, "sha512-exgYmnmL/sJpR3upZfXG5PoatXQii55xAiXGXzY+sROLZ/Y+SLcp9PgJNI9Vz37HpQ74WvDcLT8eqm+kV3FzrA=="], + + "tr46": ["tr46@6.0.0", "", { "dependencies": { "punycode": "^2.3.1" } }, "sha512-bLVMLPtstlZ4iMQHpFHTR7GAGj2jxi8Dg0s2h2MafAE4uSWF98FC/3MomU51iQAMf8/qDUbKWf5GxuvvVcXEhw=="], + "trim-lines": ["trim-lines@3.0.1", "", {}, "sha512-kRj8B+YHZCc9kQYdWfJB2/oUl9rA99qbowYYBtr4ui4mZyAQ2JpvVBd/6U2YloATfqBhBTSMhTpgBHtU0Mf3Rg=="], "trough": ["trough@2.2.0", "", {}, "sha512-tmMpK00BjZiUyVyvrBK7knerNgmgvcV/KLVyuma/SC+TQN167GrMRciANTz09+k3zW8L8t60jWO1GpfkZdjTaw=="], @@ -2459,12 +2524,20 @@ "vscode-uri": ["vscode-uri@3.0.8", "", {}, "sha512-AyFQ0EVmsOZOlAnxoFOGOq1SQDWAB7C6aqMGS23svWAllfOaxbuFvcT8D1i8z3Gyn8fraVeZNNmN6e9bxxXkKw=="], + "w3c-xmlserializer": ["w3c-xmlserializer@6.0.0", "", { "dependencies": { "xml-name-validator": "^5.0.0" } }, "sha512-4Nsy8K5Tr6SPDH9jhKJOHf7ChDrc1zufZTVSF7x72hwuEXBqxqk9G6cK+K2NRUtB3iELRJqjXb4JPDMBjMTl2Q=="], + "wcwidth": ["wcwidth@1.0.1", "", { "dependencies": { "defaults": "^1.0.3" } }, "sha512-XHPEwS0q6TaxcvG85+8EYkbiCux2XtWG2mkc47Ng2A77BQu9+DqIOJldST4HgPkuea7dvKSj5VgX3P1d4rW8Tg=="], "web-namespaces": ["web-namespaces@2.0.1", "", {}, "sha512-bKr1DkiNa2krS7qxNtdrtHAmzuYGFQLiQ13TsorsdT6ULTkPLKuu5+GsFpDlg6JFjUTwX2DyhMPG2be8uPrqsQ=="], "web-vitals": ["web-vitals@4.2.4", "", {}, "sha512-r4DIlprAGwJ7YM11VZp4R884m0Vmgr6EAKe3P+kO0PPj3Unqyvv59rczf6UiGcb9Z8QxZVcqKNwv/g0WNdWwsw=="], + "webidl-conversions": ["webidl-conversions@8.0.1", "", {}, "sha512-BMhLD/Sw+GbJC21C/UgyaZX41nPt8bUTg+jWyDeg7e7YN4xOM05YPSIXceACnXVtqyEw/LMClUQMtMZ+PGGpqQ=="], + + "whatwg-mimetype": ["whatwg-mimetype@5.0.0", "", {}, "sha512-sXcNcHOC51uPGF0P/D4NVtrkjSU2fNsm9iog4ZvZJsL3rjoDAzXZhkm2MWt1y+PUdggKAYVoMAIYcs78wJ51Cw=="], + + "whatwg-url": ["whatwg-url@17.1.2", "", { "dependencies": { "@exodus/bytes": "^1.15.1", "tr46": "^6.0.0", "webidl-conversions": "^8.0.1" } }, "sha512-TEZA+Zqxin7Jjsm2cjRohCmen5awh+hT6Zi3VZdqZlNRk7zvOI/9WpBFg/DWlA56bWnzwm6DuB8NS0EsxQH9uQ=="], + "which": ["which@5.0.0", "", { "dependencies": { "isexe": "^3.1.1" }, "bin": { "node-which": "bin/which.js" } }, "sha512-JEdGzHwwkrbWoGOlIHqQ5gtprKGOenpDHpxE9zVR1bWbOtYRyPPHMe9FaP6x61CmNaTThSkb0DAJte5jD+DmzQ=="], "why-is-node-running": ["why-is-node-running@2.3.0", "", { "dependencies": { "siginfo": "^2.0.0", "stackback": "0.0.2" }, "bin": { "why-is-node-running": "cli.js" } }, "sha512-hUrmaWBdVDcxvYqnyh09zunKzROWjbZTiNy8dBEjkS7ehEDQibXJ7XvlmtbwuTclUiIyN+CyXQD4Vmko8fNm8w=="], @@ -2479,8 +2552,12 @@ "ws": ["ws@8.21.3", "", { "peerDependencies": { "bufferutil": "^4.0.1", "utf-8-validate": ">=5.0.2" }, "optionalPeers": ["bufferutil", "utf-8-validate"] }, "sha512-201TZ/kPWxoPr/OKWjquZR1SWKXcvxdH+e1xrx89b3YbmzLMFCLfnaG1HFIgWzJOEWZ7MvpK++odZufgYR50Rw=="], + "xml-name-validator": ["xml-name-validator@5.0.0", "", {}, "sha512-EvGK8EJ3DhaHfbRlETOWAS5pO9MZITeauHKJyb8wyajUfQUenkIg2MvLDTZ4T/TgIcm3HU0TFBgWWboAZ30UHg=="], + "xmlbuilder": ["xmlbuilder@15.1.1", "", {}, "sha512-yMqGBqtXyeN1e3TGYvgNgDVZ3j84W4cwkOXQswghol6APgZWaff9lnbvN7MHYJOiXsvGPXtjTYJEiC9J2wv9Eg=="], + "xmlchars": ["xmlchars@2.2.0", "", {}, "sha512-JZnDKK8B0RCDw84FNdDAIpZK+JuJw+s7Lz8nksI7SIuU3UXJJslUthsi+uWBUYOwPFwW7W7PRLRfUKpxjtjFCw=="], + "xtend": ["xtend@4.0.2", "", {}, "sha512-LKYU1iAXJXUgAXn9URjiu+MWhyUXHsvfp7mcuYm9dSUKK0/CjtrUwFAxD82/mCWbtLsGjFIad0wIsod4zrTAEQ=="], "xterm": ["xterm@5.3.0", "", {}, "sha512-8QqjlekLUFTrU6x7xck1MsPzPA571K5zNqWm0M0oroYEWVOptZ0+ubQSkQ3uxIEhcIHRujJy6emDWX4A7qyFzg=="], @@ -2521,6 +2598,8 @@ "@babel/core/semver": ["semver@6.3.1", "", { "bin": { "semver": "bin/semver.js" } }, "sha512-BR7VvDCVHO+q2xBEWskxS6DJE1qRnb7DxzUrogb71CWoSficBxYsiAGd+Kl0mmq/MprG9yArRkyrQxTO6XjMzA=="], + "@babel/helper-compilation-targets/lru-cache": ["lru-cache@5.1.1", "", { "dependencies": { "yallist": "^3.0.2" } }, "sha512-KpNARQA3Iwv+jTA0utUVVbrh+Jlrr1Fv0e56GGzAFOXN7dk/FviaDW8LHmK52DlcH4WP2n6gI8vN1aesBFgo9w=="], + "@babel/helper-compilation-targets/semver": ["semver@6.3.1", "", { "bin": { "semver": "bin/semver.js" } }, "sha512-BR7VvDCVHO+q2xBEWskxS6DJE1qRnb7DxzUrogb71CWoSficBxYsiAGd+Kl0mmq/MprG9yArRkyrQxTO6XjMzA=="], "@chevrotain/cst-dts-gen/lodash-es": ["lodash-es@4.17.21", "", {}, "sha512-mKnC+QJ9pWVzv+C4/U3rRsHapFfHvQFoFB92e52xeyGMcX6/OlIl78je1u8vePzYZSkkogMPJ2yjxxsb89cxyw=="], @@ -2649,6 +2728,8 @@ "cacache/lru-cache": ["lru-cache@10.4.3", "", {}, "sha512-JNAzZcXrCt42VGLuYz0zfAzDfAvJWW6AfYlDBQyDV5DClI2m5sAmK+OIO7s59XfsRsWHp02jAJrRadPRGTt6SQ=="], + "chalk/ansi-styles": ["ansi-styles@4.3.0", "", { "dependencies": { "color-convert": "^2.0.1" } }, "sha512-zbB9rCJAT1rbjiVDb2hqKFHNYLxgtk8NURxZ3IZwD3F6NtxbXZQCnnSi1Lkx+IDohdPlFp222wVALIheZJQSEg=="], + "chevrotain/lodash-es": ["lodash-es@4.17.21", "", {}, "sha512-mKnC+QJ9pWVzv+C4/U3rRsHapFfHvQFoFB92e52xeyGMcX6/OlIl78je1u8vePzYZSkkogMPJ2yjxxsb89cxyw=="], "clone-response/mimic-response": ["mimic-response@1.0.1", "", {}, "sha512-j5EctnkH7amfV/q5Hgmoal1g2QHFJRraOtmx0JpIqkxhBhI/lJSl1nMpQ45hVarwNETOoWEimndZ4QK0RHxuxQ=="], @@ -2663,6 +2744,8 @@ "d3-sankey/d3-shape": ["d3-shape@1.3.7", "", { "dependencies": { "d3-path": "1" } }, "sha512-EUkvKjqPFUAZyOlhY5gzCxCeI0Aep04LwIRpsZ/mLFelJiUfnK56jo5JMDSE7yyP2kLSb6LtF+S5chMk7uqPqw=="], + "data-urls/whatwg-url": ["whatwg-url@16.0.1", "", { "dependencies": { "@exodus/bytes": "^1.11.0", "tr46": "^6.0.0", "webidl-conversions": "^8.0.1" } }, "sha512-1to4zXBxmXHV3IiSSEInrreIlu02vUOvrhxJJH5vcxYTBDAx51cqZiKdyTxlecdKNSjj8EcxGBxNf6Vg+945gw=="], + "dir-compare/minimatch": ["minimatch@3.1.2", "", { "dependencies": { "brace-expansion": "^1.1.7" } }, "sha512-J7p63hRiAjw1NDEww1W7i37+ByIrOWO5XQQAzZ3VOcL0PNybwpfmV/N05zFAzwQ9USyEcX6t3UO+K5aqBQOIHw=="], "dmg-license/ajv": ["ajv@6.12.6", "", { "dependencies": { "fast-deep-equal": "^3.1.1", "fast-json-stable-stringify": "^2.0.0", "json-schema-traverse": "^0.4.1", "uri-js": "^4.2.2" } }, "sha512-j3fVLgvTo527anyYyJOGTYJbG+vnnQYvE0m5mmkc1TK+nxAppkCLMIL0aZ4dblVCNoGShhm+kzE4ZUykBoMg4g=="], @@ -2687,6 +2770,8 @@ "gray-matter/js-yaml": ["js-yaml@3.14.2", "", { "dependencies": { "argparse": "^1.0.7", "esprima": "^4.0.0" }, "bin": { "js-yaml": "bin/js-yaml.js" } }, "sha512-PMSmkqxr106Xa156c2M265Z+FTrPl+oxd/rgOQy2tijQeK5TxQ43psO1ZCwhVOSdnn+RzkzlRz/eY4BgJBYVpg=="], + "hast-util-raw/parse5": ["parse5@7.3.0", "", { "dependencies": { "entities": "^6.0.0" } }, "sha512-IInvU7fabl34qmi9gY8XOVxhYyMyuH2xUNpb2q8/Y+7552KlejkRvqvD19nMoUW/uQGGbqNpA6Tufu5FL5BZgw=="], + "hosted-git-info/lru-cache": ["lru-cache@6.0.0", "", { "dependencies": { "yallist": "^4.0.0" } }, "sha512-Jo6dJ04CmSjuznwJSS3pUeWmd/H0ffTlkXXgwZi+eq1UCmqQwCh+eLsYOYCwY991i2Fah4h1BEMCx4qThGbsiA=="], "iconv-corefoundation/node-addon-api": ["node-addon-api@1.7.2", "", {}, "sha512-ibPK3iA+vaY1eEjESkQkM0BbCqFOaZMiXRTtdB0u7b4djtY6JnsjvPdUHVMg6xQt3B8fpTTWHI9A+ADjM9frzg=="], @@ -2695,8 +2780,6 @@ "lazystream/readable-stream": ["readable-stream@2.3.8", "", { "dependencies": { "core-util-is": "~1.0.0", "inherits": "~2.0.3", "isarray": "~1.0.0", "process-nextick-args": "~2.0.0", "safe-buffer": "~5.1.1", "string_decoder": "~1.1.1", "util-deprecate": "~1.0.1" } }, "sha512-8p0AUk4XODgIewSi0l8Epjs+EVnWiK7NoDIEGU0HhE7+ZyY8D1IMY7odu5lRrFXGg71L15KG8QrPmum45RTtdA=="], - "lru-cache/yallist": ["yallist@3.1.1", "", {}, "sha512-a4UGQaWPH59mOXUYnAG2ewncQS4i4F43Tv3JoAM+s2VDAmS9NsK8GpDMLrCHPksFT7h3K6TOoUNn2pb7RoXx4g=="], - "matcher/escape-string-regexp": ["escape-string-regexp@4.0.0", "", {}, "sha512-TtpcNJ3XAzx3Gq8sWRzJaVajRs0uVxA2YAkdb1jm2YkPz4G6egUFAyA3n5vtEIZefPk5Wa4UXbKuS5fKkJWdgA=="], "micromatch/picomatch": ["picomatch@2.3.1", "", {}, "sha512-JU3teHTNjmE2VCGFzuY8EXzCDVwEqB2a8fsIvwaStHhAWJEeVd1o1QD80CU6+ZdEXXSLbSsuLwJjkCBWqRQUVA=="], @@ -2731,6 +2814,8 @@ "shiki/@shikijs/engine-javascript": ["@shikijs/engine-javascript@1.29.2", "", { "dependencies": { "@shikijs/types": "1.29.2", "@shikijs/vscode-textmate": "^10.0.1", "oniguruma-to-es": "^2.2.0" } }, "sha512-iNEZv4IrLYPv64Q6k7EPpOCE/nuvGiKl7zxdq0WFuRPF5PAE9PRo2JGq/d8crLusM59BRemJ4eOqrFrC4wiQ+A=="], + "slice-ansi/ansi-styles": ["ansi-styles@4.3.0", "", { "dependencies": { "color-convert": "^2.0.1" } }, "sha512-zbB9rCJAT1rbjiVDb2hqKFHNYLxgtk8NURxZ3IZwD3F6NtxbXZQCnnSi1Lkx+IDohdPlFp222wVALIheZJQSEg=="], + "streamdown/marked": ["marked@17.0.1", "", { "bin": { "marked": "bin/marked.js" } }, "sha512-boeBdiS0ghpWcSwoNm/jJBwdpFaMnZWRzjA6SkUMYb40SVaN1x7mmfGKp0jvexGcx+7y2La5zRZsYFZI6Qpypg=="], "streamdown/tailwind-merge": ["tailwind-merge@3.4.0", "", {}, "sha512-uSaO4gnW+b3Y2aWoWfFpX62vn2sR3skfhbjsEnaBI81WD1wBLlHZe5sWf0AqjksNdYTbGBEd0UasQMT3SNV15g=="], @@ -2743,10 +2828,16 @@ "vitest/tinyglobby": ["tinyglobby@0.2.17", "", { "dependencies": { "fdir": "^6.5.0", "picomatch": "^4.0.4" } }, "sha512-wXR/dYpcqKmfWpEdZjiKJOwCNFndD0DMnrW/cYjVGttEkBfVgcLFHoNrlj47mjOVic9yyNu65alsgF4NQyTa2g=="], + "wrap-ansi/ansi-styles": ["ansi-styles@4.3.0", "", { "dependencies": { "color-convert": "^2.0.1" } }, "sha512-zbB9rCJAT1rbjiVDb2hqKFHNYLxgtk8NURxZ3IZwD3F6NtxbXZQCnnSi1Lkx+IDohdPlFp222wVALIheZJQSEg=="], + + "wrap-ansi-cjs/ansi-styles": ["ansi-styles@4.3.0", "", { "dependencies": { "color-convert": "^2.0.1" } }, "sha512-zbB9rCJAT1rbjiVDb2hqKFHNYLxgtk8NURxZ3IZwD3F6NtxbXZQCnnSi1Lkx+IDohdPlFp222wVALIheZJQSEg=="], + "zip-stream/archiver-utils": ["archiver-utils@3.0.4", "", { "dependencies": { "glob": "^7.2.3", "graceful-fs": "^4.2.0", "lazystream": "^1.0.0", "lodash.defaults": "^4.2.0", "lodash.difference": "^4.5.0", "lodash.flatten": "^4.4.0", "lodash.isplainobject": "^4.0.6", "lodash.union": "^4.6.0", "normalize-path": "^3.0.0", "readable-stream": "^3.6.0" } }, "sha512-KVgf4XQVrTjhyWmx6cte4RxonPLR9onExufI1jhvw/MQ4BB6IsZD5gT8Lq+u/+pRkWna/6JoHpiQioaqFP5Rzw=="], "@ai-sdk/react/@ai-sdk/provider-utils/@ai-sdk/provider": ["@ai-sdk/provider@3.0.4", "", { "dependencies": { "json-schema": "^0.4.0" } }, "sha512-5KXyBOSEX+l67elrEa+wqo/LSsSTtrPj9Uoh3zMbe/ceQX4ucHI3b9nUEfNkGF3Ry1svv90widAt+aiKdIJasQ=="], + "@babel/helper-compilation-targets/lru-cache/yallist": ["yallist@3.1.1", "", {}, "sha512-a4UGQaWPH59mOXUYnAG2ewncQS4i4F43Tv3JoAM+s2VDAmS9NsK8GpDMLrCHPksFT7h3K6TOoUNn2pb7RoXx4g=="], + "@develar/schema-utils/ajv/json-schema-traverse": ["json-schema-traverse@0.4.1", "", {}, "sha512-xbbCH5dCYU5T8LcEhhuh7HJ88HXuW3qsI3Y0zOZFKfZEHcpWiHU/Jxzk629Brsab/mMiHQti9wMP+845RPe3Vg=="], "@electron/asar/minimatch/brace-expansion": ["brace-expansion@1.1.12", "", { "dependencies": { "balanced-match": "^1.0.0", "concat-map": "0.0.1" } }, "sha512-9T9UjW3r0UW5c1Q7GTwllptXwhvYmEzFhzMfZ9H7FQWt+uZePjZPjBP/W1ZEyZ1twGWom5/56TF4lPcqjnDHcg=="], @@ -2857,6 +2948,8 @@ "gray-matter/js-yaml/argparse": ["argparse@1.0.10", "", { "dependencies": { "sprintf-js": "~1.0.2" } }, "sha512-o5Roy6tNG4SL/FOkCAN6RzjiakZS25RLYFrcMttJqbdd8BWrnA+fGz57iN5Pb06pvBGvl5gQ0B48dJlslXvoTg=="], + "hast-util-raw/parse5/entities": ["entities@6.0.1", "", {}, "sha512-aN97NXWF6AWBTahfVOIrB/NShkzi5H7F9r1s9mD3cDj4Ko5f2qhhVoYMibXF7GlLveb/D2ioWay8lxI97Ven3g=="], + "hosted-git-info/lru-cache/yallist": ["yallist@4.0.0", "", {}, "sha512-3wdGidZyq5PB084XLES5TpOSRA3wjXAlIWMhum2kRcv/41Sn2emQ0dycQW4uZXLejwKvg6EsvbdlVL+FYEct7A=="], "lazystream/readable-stream/safe-buffer": ["safe-buffer@5.1.2", "", {}, "sha512-Gd2UZBJDkXlY7GbJxfsE8/nvKkUEU1G38c1siN6QP6a9PT9MmHB8GnpscSmMJSoF8LOIrt8ud/wPtojys4G6+g=="], diff --git a/package.json b/package.json index 2c46e699..93bab397 100644 --- a/package.json +++ b/package.json @@ -142,6 +142,8 @@ "@electron-toolkit/utils": "^4.0.0", "@electron/rebuild": "^4.0.3", "@tailwindcss/container-queries": "^0.1.1", + "@testing-library/dom": "10.4.2", + "@testing-library/react": "16.3.3", "@types/better-sqlite3": "^7.6.13", "@types/diff": "^8.0.0", "@types/node": "^20.17.50", @@ -154,6 +156,7 @@ "electron": "~39.4.0", "electron-builder": "^25.1.8", "electron-vite": "^3.0.0", + "jsdom": "30.1.1", "postcss": "^8.5.1", "sharp": "0.35.4", "tailwindcss": "^3.4.17", diff --git a/src/renderer/features/agents/main/__snapshots__/assistant-message-item.test.tsx.snap b/src/renderer/features/agents/main/__snapshots__/assistant-message-item.test.tsx.snap new file mode 100644 index 00000000..69f0e3db --- /dev/null +++ b/src/renderer/features/agents/main/__snapshots__/assistant-message-item.test.tsx.snap @@ -0,0 +1,60 @@ +// Vitest Snapshot v1, https://vitest.dev/guide/snapshot.html + +exports[`AssistantMessageItem, one message per branch of the part dispatcher > collapses the steps under a final text part 1`] = `"
Response

The lockfile is clean.

"`; + +exports[`AssistantMessageItem, one message per branch of the part dispatcher > groups three consecutive exploring tools 1`] = `"
Response

Read all three.

"`; + +exports[`AssistantMessageItem, one message per branch of the part dispatcher > keeps every part visible while the last message streams 1`] = `"

Lint is clean so far.

"`; + +exports[`AssistantMessageItem, one message per branch of the part dispatcher > renders a Bash call with its command, output and exit code 1`] = ` +"
" +`; + +exports[`AssistantMessageItem, one message per branch of the part dispatcher > renders a PlanWrite 1`] = `"
"`; + +exports[`AssistantMessageItem, one message per branch of the part dispatcher > renders a Write with its content 1`] = `"
"`; + +exports[`AssistantMessageItem, one message per branch of the part dispatcher > renders a completed thinking tool 1`] = `"
"`; + +exports[`AssistantMessageItem, one message per branch of the part dispatcher > renders a failing Bash call 1`] = `"
"`; + +exports[`AssistantMessageItem, one message per branch of the part dispatcher > renders a plan file as a plan card 1`] = `"
"`; + +exports[`AssistantMessageItem, one message per branch of the part dispatcher > renders a question awaiting an answer 1`] = `"
SDK pin•Interrupted
"`; + +exports[`AssistantMessageItem, one message per branch of the part dispatcher > renders a reasoning part 1`] = `"
"`; + +exports[`AssistantMessageItem, one message per branch of the part dispatcher > renders a registry tool as a single row 1`] = `"
Readeffort.ts
"`; + +exports[`AssistantMessageItem, one message per branch of the part dispatcher > renders a second plan operation as a mini indicator, not a card 1`] = `"
Created plan
"`; + +exports[`AssistantMessageItem, one message per branch of the part dispatcher > renders a subagent task with its nested tools 1`] = `"
"`; + +exports[`AssistantMessageItem, one message per branch of the part dispatcher > renders a text part 1`] = `"

Both pins moved together.

"`; + +exports[`AssistantMessageItem, one message per branch of the part dispatcher > renders a todo list 1`] = `"
"`; + +exports[`AssistantMessageItem, one message per branch of the part dispatcher > renders a tool nobody registered as its bare name 1`] = `"
SomethingTheNextCliAdds
"`; + +exports[`AssistantMessageItem, one message per branch of the part dispatcher > renders a web fetch 1`] = `"
"`; + +exports[`AssistantMessageItem, one message per branch of the part dispatcher > renders a web search with its results 1`] = `"
"`; + +exports[`AssistantMessageItem, one message per branch of the part dispatcher > renders an Edit with its patch 1`] = `"
"`; + +exports[`AssistantMessageItem, one message per branch of the part dispatcher > renders an MCP tool call 1`] = `"
"`; + +exports[`AssistantMessageItem, one message per branch of the part dispatcher > renders an orphaned nested group as an incomplete task 1`] = `"
Subagent interruptedIncomplete task
"`; + +exports[`AssistantMessageItem, one message per branch of the part dispatcher > renders nothing for ExitPlanMode 1`] = `"
"`; + +exports[`AssistantMessageItem, one message per branch of the part dispatcher > renders nothing for a part that is neither text nor a tool 1`] = `"
"`; + +exports[`AssistantMessageItem, one message per branch of the part dispatcher > renders nothing for a step-start marker 1`] = `"
"`; + +exports[`AssistantMessageItem, one message per branch of the part dispatcher > renders nothing for a text part that is only whitespace 1`] = `"
"`; + +exports[`AssistantMessageItem, one message per branch of the part dispatcher > renders the renamed background-output tool 1`] = `"
Got outputTask: task_1
"`; + +exports[`AssistantMessageItem, one message per branch of the part dispatcher > renders usage and git badges from message metadata 1`] = `"

Done.

"`; diff --git a/src/renderer/features/agents/main/assistant-message-item.test.tsx b/src/renderer/features/agents/main/assistant-message-item.test.tsx new file mode 100644 index 00000000..29a93005 --- /dev/null +++ b/src/renderer/features/agents/main/assistant-message-item.test.tsx @@ -0,0 +1,442 @@ +// @vitest-environment jsdom +/** + * What the transcript renderer produces, pinned before it is restructured. + * + * `renderPart` in `assistant-message-item.tsx` decides what every kind of + * message part looks like. It is a ~250-line dispatcher at a cognitive + * complexity SonarQube reports as 70 against a threshold of 15, and until this + * file existed nothing in the suite rendered any of it: the vitest environment + * is `node`, so the dispatcher's behaviour was verified by reading it. + * + * These snapshots are the golden output for one message per branch of that + * dispatcher, written against the code as it stands and committed before the + * split, so the restructuring can be checked against what the renderer actually + * produced rather than against a reviewer's memory of it. Nothing here asserts + * intent or good taste. It records output, and the only acceptable result after + * a refactor is that it does not change. + */ +import { render } from "@testing-library/react" +import { beforeEach, describe, expect, it, vi } from "vitest" +import { TooltipProvider } from "../../../components/ui/tooltip" +import type { Message, MessagePart } from "../stores/message-store" +import { AssistantMessageItem } from "./assistant-message-item" + +// The chime plays when a message ends on a question awaiting an answer. jsdom +// has no Audio, and a sound is not part of the output these snapshots pin. +vi.mock("../lib/play-question-sound", () => ({ playQuestionSound: vi.fn() })) + +// jsdom implements neither, and Radix's collapsible reaches for both while the +// step group renders. +vi.stubGlobal( + "ResizeObserver", + class { + observe() {} + unobserve() {} + disconnect() {} + }, +) +vi.stubGlobal( + "IntersectionObserver", + class { + observe() {} + unobserve() {} + disconnect() {} + takeRecords() { + return [] + } + }, +) + +/** + * The transcript renders inside the agents layout, which wraps its tree in a + * tooltip provider; the tool rows need that context to render at all. + */ +function renderMessage(message: Message, streaming: boolean): string { + const { container } = render( + + + , + ) + return container.innerHTML +} + +/** + * Each message gets its own id: the module keeps a per-message state cache to + * survive the AI SDK mutating parts in place, and a shared id would let one + * test's cache decide the next test's memo comparison. + */ +let messageSequence = 0 + +// The id lands in the rendered DOM, so it is reset per test: a snapshot must not +// depend on how many tests ran before it. +beforeEach(() => { + messageSequence = 0 +}) + +function renderParts(parts: MessagePart[], streaming = false): string { + const message: Message = { + id: `msg-${++messageSequence}`, + role: "assistant", + parts, + } + return renderMessage(message, streaming) +} + +/** A completed tool call, which is the state every fixture below starts from. */ +function tool(type: string, id: string, input: unknown, output?: unknown): MessagePart { + return { type, toolCallId: id, state: "output-available", input, output } +} + +describe("AssistantMessageItem, one message per branch of the part dispatcher", () => { + it("renders a text part", () => { + expect(renderParts([{ type: "text", text: "Both pins moved together." }])).toMatchSnapshot() + }) + + it("renders nothing for a text part that is only whitespace", () => { + expect(renderParts([{ type: "text", text: " \n " }])).toMatchSnapshot() + }) + + it("renders nothing for a step-start marker", () => { + expect(renderParts([{ type: "step-start" }])).toMatchSnapshot() + }) + + it("renders nothing for a part that is neither text nor a tool", () => { + expect( + renderParts([{ type: "file-content", filePath: "notes.txt", content: "hidden" }]), + ).toMatchSnapshot() + }) + + it("renders a Bash call with its command, output and exit code", () => { + expect( + renderParts([ + tool( + "tool-Bash", + "toolu_bash_1", + { command: "rg -n 'ALL_FEATURES_OFF' src" }, + { + stdout: "src/shared/provider-capabilities.ts:41:export const ALL_FEATURES_OFF\n", + exitCode: 0, + }, + ), + ]), + ).toMatchSnapshot() + }) + + it("renders a failing Bash call", () => { + expect( + renderParts([ + tool( + "tool-Bash", + "toolu_bash_2", + { command: "bun run build" }, + { stdout: "", stderr: "electron-builder exited with 1", exitCode: 1 }, + ), + ]), + ).toMatchSnapshot() + }) + + it("renders a reasoning part", () => { + expect( + renderParts([ + { type: "reasoning", text: "The registry names six legacy spellings.", state: "done" }, + ]), + ).toMatchSnapshot() + }) + + it("renders a completed thinking tool", () => { + expect( + renderParts([ + { + type: "tool-Thinking", + toolCallId: "toolu_think_1", + toolName: "Thinking", + state: "output-available", + input: { text: "Check the changelog for the wire name." }, + output: { completed: true }, + }, + ]), + ).toMatchSnapshot() + }) + + it("renders an Edit with its patch", () => { + expect( + renderParts([ + tool( + "tool-Edit", + "toolu_edit_1", + { + file_path: "/repo/src/shared/provider-capabilities.ts", + old_string: "export const TURN_CONTROLS_OFF", + new_string: "export const ALL_FEATURES_OFF", + }, + { + structuredPatch: [ + { lines: ["- export const TURN_CONTROLS_OFF", "+ export const ALL_FEATURES_OFF"] }, + ], + }, + ), + ]), + ).toMatchSnapshot() + }) + + it("renders a Write with its content", () => { + expect( + renderParts([ + tool("tool-Write", "toolu_write_1", { + file_path: "/repo/src/shared/effort.ts", + content: "export const EFFORT_LEVELS = ['minimal', 'low', 'medium', 'high'] as const\n", + }), + ]), + ).toMatchSnapshot() + }) + + it("renders a plan file as a plan card", () => { + expect( + renderParts([ + tool("tool-Write", "toolu_plan_1", { + file_path: "/repo/claude-sessions/plans/step-12-plan.md", + content: "# Step 12\n\nMove the pins together.\n", + }), + ]), + ).toMatchSnapshot() + }) + + it("renders a second plan operation as a mini indicator, not a card", () => { + expect( + renderParts([ + tool("tool-Write", "toolu_plan_2", { + file_path: "/repo/plans-a-plan.md", + content: "# First\n", + }), + tool("tool-Edit", "toolu_plan_3", { + file_path: "/repo/plans-a-plan.md", + old_string: "# First", + new_string: "# Second", + }), + ]), + ).toMatchSnapshot() + }) + + it("renders a web search with its results", () => { + expect( + renderParts([ + tool( + "tool-WebSearch", + "toolu_search_1", + { query: "claude agent sdk 0.3.270 changelog" }, + { + results: [ + { + content: [ + { title: "SDK changelog", url: "https://example.com/changelog" }, + { title: "Release notes", url: "https://example.com/releases" }, + ], + }, + ], + }, + ), + ]), + ).toMatchSnapshot() + }) + + it("renders a web fetch", () => { + expect( + renderParts([ + tool( + "tool-WebFetch", + "toolu_fetch_1", + { url: "https://example.com/docs/pins" }, + { result: "Fetched the pin table.", bytes: 12048, code: 200 }, + ), + ]), + ).toMatchSnapshot() + }) + + it("renders a PlanWrite", () => { + expect( + renderParts([ + tool("tool-PlanWrite", "toolu_planwrite_1", { + plan: { + status: "completed", + steps: [ + { title: "Pin the SDK and the CLI together", status: "completed" }, + { title: "Register the renamed tools", status: "completed" }, + ], + }, + }), + ]), + ).toMatchSnapshot() + }) + + it("renders nothing for ExitPlanMode", () => { + expect(renderParts([tool("tool-ExitPlanMode", "toolu_exit_1", {})])).toMatchSnapshot() + }) + + it("renders a todo list", () => { + expect( + renderParts([ + tool( + "tool-TodoWrite", + "toolu_todo_1", + { + todos: [ + { content: "Pin the SDK", status: "completed", activeForm: "Pinning the SDK" }, + { + content: "Register the tools", + status: "in_progress", + activeForm: "Registering the tools", + }, + ], + }, + { oldTodos: [], newTodos: [{ content: "Pin the SDK", status: "completed" }] }, + ), + ]), + ).toMatchSnapshot() + }) + + it("renders a question awaiting an answer", () => { + expect( + renderParts([ + { + type: "tool-AskUserQuestion", + toolCallId: "toolu_ask_1", + state: "call", + input: { + questions: [ + { + question: "Which SDK version should the pin take?", + header: "SDK pin", + multiSelect: false, + options: [ + { label: "0.3.270", description: "Matches the CLI pin." }, + { label: "0.3.280", description: "One release ahead." }, + ], + }, + ], + }, + }, + ]), + ).toMatchSnapshot() + }) + + it("renders a registry tool as a single row", () => { + expect( + renderParts([tool("tool-Read", "toolu_read_1", { file_path: "/repo/src/shared/effort.ts" })]), + ).toMatchSnapshot() + }) + + it("renders the renamed background-output tool", () => { + expect( + renderParts([ + tool("tool-TaskOutput", "toolu_taskoutput_1", { task_id: "task_1" }, { output: "done" }), + ]), + ).toMatchSnapshot() + }) + + it("renders a subagent task with its nested tools", () => { + expect( + renderParts([ + tool("tool-Agent", "toolu_agent_1", { + subagent_type: "general-purpose", + description: "Find every pin the step has to move", + }), + tool("tool-Read", "toolu_agent_1:0", { file_path: "/repo/package.json" }), + tool( + "tool-Bash", + "toolu_agent_1:1", + { command: "git diff --stat" }, + { stdout: "", exitCode: 0 }, + ), + ]), + ).toMatchSnapshot() + }) + + it("renders an orphaned nested group as an incomplete task", () => { + expect( + renderParts([ + tool("tool-Bash", "toolu_ghost_1:0", { command: "ls" }, { stdout: "", exitCode: 0 }), + tool("tool-Read", "toolu_ghost_1:1", { file_path: "/repo/AGENTS.md" }), + ]), + ).toMatchSnapshot() + }) + + it("renders an MCP tool call", () => { + expect( + renderParts([ + tool("tool-mcp__filesystem__read_file", "toolu_mcp_1", { path: "/tmp/pins.json" }), + ]), + ).toMatchSnapshot() + }) + + it("renders a tool nobody registered as its bare name", () => { + expect( + renderParts([tool("tool-SomethingTheNextCliAdds", "toolu_future_1", {})]), + ).toMatchSnapshot() + }) + + it("collapses the steps under a final text part", () => { + expect( + renderParts([ + tool( + "tool-Bash", + "toolu_collapse_1", + { command: "bun install --frozen-lockfile" }, + { stdout: "", exitCode: 0 }, + ), + tool("tool-Read", "toolu_collapse_2", { file_path: "/repo/bun.lock" }), + { type: "text", text: "The lockfile is clean." }, + ]), + ).toMatchSnapshot() + }) + + it("keeps every part visible while the last message streams", () => { + expect( + renderParts( + [ + tool( + "tool-Bash", + "toolu_stream_1", + { command: "bun run lint" }, + { stdout: "", exitCode: 0 }, + ), + { type: "text", text: "Lint is clean so far." }, + ], + true, + ), + ).toMatchSnapshot() + }) + + it("groups three consecutive exploring tools", () => { + expect( + renderParts([ + tool("tool-Read", "toolu_explore_1", { file_path: "/repo/src/a.ts" }), + tool("tool-Read", "toolu_explore_2", { file_path: "/repo/src/b.ts" }), + tool("tool-Read", "toolu_explore_3", { file_path: "/repo/src/c.ts" }), + { type: "text", text: "Read all three." }, + ]), + ).toMatchSnapshot() + }) + + it("renders usage and git badges from message metadata", () => { + const message: Message = { + id: `msg-${++messageSequence}`, + role: "assistant", + parts: [{ type: "text", text: "Done." }], + metadata: { + inputTokens: 1200, + outputTokens: 340, + cacheReadInputTokens: 800, + cacheCreationInputTokens: 0, + }, + } + expect(renderMessage(message, false)).toMatchSnapshot() + }) +}) diff --git a/vitest.config.ts b/vitest.config.ts index 14d6a002..d279d7a1 100644 --- a/vitest.config.ts +++ b/vitest.config.ts @@ -2,9 +2,10 @@ import { defineConfig } from "vitest/config" export default defineConfig({ test: { - // Node environment: tests cover main-process logic, not the renderer. + // Node environment by default: tests cover main-process logic. A renderer + // test declares `// @vitest-environment jsdom` at the top of its own file. environment: "node", - include: ["src/**/*.test.ts"], + include: ["src/**/*.test.ts", "src/**/*.test.tsx"], // node:test suites (not vitest) — run via `npm run test:node`. exclude: ["src/main/lib/runtime/*.test.ts", "node_modules"], // Main-process modules pull in electron/native deps; keep tests pure. From 1900d14db2327da35d116e517505b537ae3b6da1 Mon Sep 17 00:00:00 2001 From: Owie6789 <151057755+Owie6789@users.noreply.github.com> Date: Wed, 23 Sep 2026 19:08:22 +0000 Subject: [PATCH 25/60] Take the transcript dispatcher out of its closure SonarQube measured renderPart at a cognitive complexity of 70 against a threshold of 15: a useCallback inside AssistantMessageItem holding eighteen branch decisions, eighteen closure reads and the JSX for each one. The previous commit pinned what it renders in 28 snapshots; this changes how the decision is made and proves the output did not move. The branch bodies are unchanged. They now live at module scope as one function per shape - text, sub-agent task, Bash, thinking, plan operation, file edit, web search, web fetch, plan write, todo list, question, registry row, unregistered tool - each taking the part, its index, and one PartRenderContext carrying what only the component knows: message id, status, stream state, sub-chat and project, the file-open handler, the nesting maps, the orphan groups, the plan summary and the collapse window. renderMessagePart dispatches in the order it always did: suppressions, text, sub-agent, plan file, the type table, the registry, then the shapes nobody registered. Order is the part that carries meaning - a sub-agent Task and a Write to a plan file both claim a type the table also names, and they have to win. Two things fall out. Write and Edit rendered the identical AgentEditTool in two branches, so one function now serves both table entries. And the closure's eighteen-entry deps list becomes a useMemo around the context, with renderPart depending on that single object. Evidence: the 28 snapshots pass byte-identically - git reports no change to the .snap file and vitest wrote 0, obsoleted 0; full suite 104 files, 1919 passed, 1 skipped; typecheck ratchet 0 errors against a 0 baseline; Biome 980 files and 0 findings; test:node 59/59; test:contracts 382; audit ratchet passed; skills:verify 50/50. The decision now sits in renderMessagePart at eleven increments instead of one function at seventy, and SonarCloud re-measures on this push. Co-authored-by: arena-agent <297053741+arena-agent@users.noreply.github.com> --- .../agents/main/assistant-message-item.tsx | 630 +++++++++++------- 1 file changed, 396 insertions(+), 234 deletions(-) diff --git a/src/renderer/features/agents/main/assistant-message-item.tsx b/src/renderer/features/agents/main/assistant-message-item.tsx index d33c4dfd..d74c708e 100644 --- a/src/renderer/features/agents/main/assistant-message-item.tsx +++ b/src/renderer/features/agents/main/assistant-message-item.tsx @@ -2,7 +2,16 @@ import { useAtomValue } from "jotai" import { ListTree, MoreHorizontal } from "lucide-react" -import { memo, useCallback, useContext, useEffect, useMemo, useRef, useState } from "react" +import { + memo, + type ReactNode, + useCallback, + useContext, + useEffect, + useMemo, + useRef, + useState, +} from "react" import { normalizeCodexToolPart } from "../../../../shared/codex-tool-normalizer" import { DropdownMenu, @@ -595,6 +604,359 @@ function areMessagePropsEqual( return true } +/** One plan-file operation this message carries, in the order it arrived. */ +type PlanOperation = { type: "write" | "edit"; part: NormalizedPart; index: number } + +/** What the plan-file pass found across the whole message. */ +type PlanOpsSummary = { + operations: PlanOperation[] + hasAnyPlanOperation: boolean + isStreaming: boolean + lastOperationType: "write" | "edit" | null +} + +/** + * Everything a part renderer reads that is not the part itself: this message's + * identity and stream state, the grouping its parts implied, and how it + * collapses. One object instead of eighteen closure reads, which is what lets + * the dispatch and the renderers live at module scope, be read one branch at a + * time, and be tested without the component that owns the values. + */ +type PartRenderContext = { + messageId: string + status: string + isStreaming: boolean + isLastMessage: boolean + subChatId: string + projectPath: string | undefined + onOpenFile: ReturnType + nestedToolsMap: Map + nestedToolIds: Set + /** Nested calls whose parent task part never arrived. */ + orphans: { + toolCallIds: Set + firstToolCallIds: Set + taskGroups: Map + } + planOps: PlanOpsSummary + collapse: { + shouldCollapse: boolean + collapseBeforeIndex: number + visibleStepsCount: number + lastCollapsedPlanOp: PlanOperation | null + } +} + +type PartRenderer = (part: NormalizedPart, idx: number, ctx: PartRenderContext) => ReactNode + +/** + * A nested call under a parent that never arrived, which is not the first of its + * group: the first one stands in for the missing parent and renders the rest + * inside itself, so the others are suppressed where they sit. + */ +function isSuppressedOrphan(part: NormalizedPart, ctx: PartRenderContext): boolean { + const { toolCallIds, firstToolCallIds } = ctx.orphans + if (!part.toolCallId || !toolCallIds.has(part.toolCallId)) return false + return !firstToolCallIds.has(part.toolCallId) +} + +/** The incomplete task the first orphaned nested call of a group stands in for. */ +function renderOrphanTaskGroup( + part: NormalizedPart, + idx: number, + ctx: PartRenderContext, +): ReactNode { + const { toolCallIds, firstToolCallIds, taskGroups } = ctx.orphans + if (!part.toolCallId || !toolCallIds.has(part.toolCallId)) return null + if (!firstToolCallIds.has(part.toolCallId)) return null + const parentId = part.toolCallId.split(":")[0] + const group = taskGroups.get(parentId) + if (!group) return null + return ( + + ) +} + +function renderTextPart( + part: NormalizedPart, + idx: number, + isFinal: boolean, + ctx: PartRenderContext, +): ReactNode { + const { messageId, isLastMessage, isStreaming, collapse } = ctx + const { collapseBeforeIndex, visibleStepsCount } = collapse + if (!part.text?.trim()) return null + const isFinalText = isFinal && idx === collapseBeforeIndex + const isTextStreaming = isLastMessage && isStreaming + return ( + + ) +} + +function renderSubagentTask(part: NormalizedPart, idx: number, ctx: PartRenderContext): ReactNode { + const nestedTools = ctx.nestedToolsMap.get(part.toolCallId ?? "") || [] + return +} + +function renderBashTool(part: NormalizedPart, idx: number, ctx: PartRenderContext): ReactNode { + return ( + + ) +} + +function renderThinkingTool(part: NormalizedPart, idx: number, ctx: PartRenderContext): ReactNode { + return ( + + ) +} + +/** A Write or Edit whose target is a plan file, which the transcript shows as plan steps. */ +function isPlanOperationPart(part: NormalizedPart): boolean { + if (part.type !== "tool-Write" && part.type !== "tool-Edit") return false + const toolInput = part.input as { file_path?: string } | null | undefined + return isPlanFile(toolInput?.file_path || "") +} + +/** + * Plan files: unified handling + * - In collapsed steps: all show mini indicator, last collapsed op's card shown separately after finalParts + * - In final parts: all but last show mini indicator, last shows full card + */ +function renderPlanOperation(part: NormalizedPart, idx: number, ctx: PartRenderContext): ReactNode { + const { planOps, status, subChatId, isStreaming, isLastMessage, collapse } = ctx + const { shouldCollapse, collapseBeforeIndex, lastCollapsedPlanOp } = collapse + + // Use part.toolCallId to find operation since idx may be adjusted for collapsed parts + const opIndex = planOps.operations.findIndex((op) => op.part.toolCallId === part.toolCallId) + if (opIndex === -1) return null + + const originalIndex = planOps.operations[opIndex]?.index ?? -1 + const isInCollapsedSteps = + shouldCollapse && collapseBeforeIndex !== -1 && originalIndex < collapseBeforeIndex + const isLastCollapsedOp = lastCollapsedPlanOp?.part.toolCallId === part.toolCallId + const isLastOperation = opIndex === planOps.operations.length - 1 + + // If this is the last collapsed plan op, hide it here (card shown after CollapsibleSteps) + if (isInCollapsedSteps && isLastCollapsedOp) { + return null + } + + // Show mini indicator for: + // - All operations in collapsed steps (except last collapsed, handled above) + // - All operations except last in final parts + const showMiniIndicator = isInCollapsedSteps || !isLastOperation + + if (showMiniIndicator) { + const isWrite = part.type === "tool-Write" + const { isPending } = getToolStatus(part, status) + const isOpStreaming = + isPending || (part.state === "input-streaming" && isStreaming && isLastMessage) + + return ( +
+ + {isOpStreaming ? ( + + {isWrite ? "Creating plan..." : "Updating plan..."} + + ) : isWrite ? ( + "Created plan" + ) : ( + "Updated plan" + )} + +
+ ) + } + + // Last operation in final parts: show full card + return ( + + ) +} + +/** A file edit that is not a plan step: Write and Edit render the same card. */ +function renderFileEditTool(part: NormalizedPart, idx: number, ctx: PartRenderContext): ReactNode { + return ( + + ) +} + +function renderWebSearch(part: NormalizedPart, idx: number, ctx: PartRenderContext): ReactNode { + return +} + +function renderWebFetch(part: NormalizedPart, idx: number, ctx: PartRenderContext): ReactNode { + return +} + +function renderPlanWrite(part: NormalizedPart, idx: number, ctx: PartRenderContext): ReactNode { + return ( + + ) +} + +function renderTodoList(part: NormalizedPart, idx: number, ctx: PartRenderContext): ReactNode { + return ( + + ) +} + +function renderQuestionTool(part: NormalizedPart, idx: number, ctx: PartRenderContext): ReactNode { + const { isPending, isError } = getToolStatus(part, ctx.status) + return ( + + ) +} + +/** A tool the registry knows: one row, clickable when it was a file read. */ +function renderRegistryTool(part: NormalizedPart, idx: number, ctx: PartRenderContext): ReactNode { + const { onOpenFile, projectPath, status } = ctx + const meta = AgentToolRegistry[part.type] + const { isPending, isError } = getToolStatus(part, status) + // Make Read tool clickable to open file in viewer + // Capture the path at render: part objects can be mutated in place during streaming. + const toolInput = part.input as { file_path?: string } | null | undefined + const readFilePath = + part.type === "tool-Read" && onOpenFile ? (toolInput?.file_path ?? null) : null + const handleClick = readFilePath && onOpenFile ? () => onOpenFile(readFilePath) : undefined + return ( + + ) +} + +/** A tool nobody registered: an MCP call by its server and tool, else its bare name. */ +function renderUnlistedTool(part: NormalizedPart, idx: number, ctx: PartRenderContext): ReactNode { + // MCP tool calls (pattern: tool-mcp____) + const mcpInfo = parseMcpToolType(part.type) + if (mcpInfo) { + return + } + + if (part.type?.startsWith("tool-")) { + return ( +
+ {part.type.replace("tool-", "")} +
+ ) + } + + return null +} + +/** + * The part types with a renderer of their own. Order does not matter here + * because the keys are distinct; what does matter is that the dispatcher consults + * this table only after the shapes that claim a type before its own renderer — + * a sub-agent task, and a Write or Edit aimed at a plan file. + */ +const PART_RENDERERS: Record = { + "tool-Bash": renderBashTool, + reasoning: renderThinkingTool, + "tool-Thinking": renderThinkingTool, + "tool-Write": renderFileEditTool, + "tool-Edit": renderFileEditTool, + "tool-WebSearch": renderWebSearch, + "tool-WebFetch": renderWebFetch, + "tool-PlanWrite": renderPlanWrite, + // ExitPlanMode tool is hidden - plan is shown in sidebar instead + "tool-ExitPlanMode": () => null, + "tool-TodoWrite": renderTodoList, + "tool-AskUserQuestion": renderQuestionTool, +} + +/** + * What one part of a message looks like, decided in the order the transcript has + * always decided it: the suppressions first, then text, then the two shapes that + * claim a type before its own renderer does, then the table, then the registry, + * then the shapes nobody registered. + */ +function renderMessagePart( + part: NormalizedPart, + idx: number, + isFinal: boolean, + ctx: PartRenderContext, +): ReactNode { + if (part.type === "step-start") return null + if (isSuppressedOrphan(part, ctx)) return null + + const orphanTask = renderOrphanTaskGroup(part, idx, ctx) + if (orphanTask) return orphanTask + + if (part.toolCallId && ctx.nestedToolIds.has(part.toolCallId)) return null + if (part.type === "exploring-group") return null + if (part.type === "text") return renderTextPart(part, idx, isFinal, ctx) + if (isSubagentToolType(part.type)) return renderSubagentTask(part, idx, ctx) + if (isPlanOperationPart(part)) return renderPlanOperation(part, idx, ctx) + + const renderer = PART_RENDERERS[part.type] + if (renderer) return renderer(part, idx, ctx) + if (part.type in AgentToolRegistry) return renderRegistryTool(part, idx, ctx) + return renderUnlistedTool(part, idx, ctx) +} + export const AssistantMessageItem = memo(function AssistantMessageItem({ message, isLastMessage, @@ -799,239 +1161,33 @@ export const AssistantMessageItem = memo(function AssistantMessageItem({ const msgMetadata = message?.metadata as AgentMessageMetadata - const renderPart = useCallback( - (part: NormalizedPart, idx: number, isFinal = false) => { - const toolInput = part.input as { file_path?: string } | null | undefined - if (part.type === "step-start") return null - - if (part.toolCallId && orphanToolCallIds.has(part.toolCallId)) { - if (!orphanFirstToolCallIds.has(part.toolCallId)) return null - const parentId = part.toolCallId.split(":")[0] - const group = orphanTaskGroups.get(parentId) - if (group) { - return ( - - ) - } - } - - if (part.toolCallId && nestedToolIds.has(part.toolCallId)) return null - if (part.type === "exploring-group") return null - - if (part.type === "text") { - if (!part.text?.trim()) return null - const isFinalText = isFinal && idx === collapseBeforeIndex - const isTextStreaming = isLastMessage && isStreaming - return ( - - ) - } - - if (isSubagentToolType(part.type)) { - const nestedTools = nestedToolsMap.get(part.toolCallId ?? "") || [] - return - } - - if (part.type === "tool-Bash") - return ( - - ) - if (part.type === "reasoning" || part.type === "tool-Thinking") { - return ( - - ) - } - - // Plan files: unified handling - // - In collapsed steps: all show mini indicator, last collapsed op's card shown separately after finalParts - // - In final parts: all but last show mini indicator, last shows full card - if (part.type === "tool-Write" || part.type === "tool-Edit") { - const filePath = toolInput?.file_path || "" - if (isPlanFile(filePath)) { - // Use part.toolCallId to find operation since idx may be adjusted for collapsed parts - const opIndex = planOpsSummary.operations.findIndex( - (op) => op.part.toolCallId === part.toolCallId, - ) - if (opIndex === -1) return null - - const originalIndex = planOpsSummary.operations[opIndex]?.index ?? -1 - const isInCollapsedSteps = - shouldCollapse && collapseBeforeIndex !== -1 && originalIndex < collapseBeforeIndex - const isLastCollapsedOp = lastCollapsedPlanOp?.part.toolCallId === part.toolCallId - const isLastOperation = opIndex === planOpsSummary.operations.length - 1 - - // If this is the last collapsed plan op, hide it here (card shown after CollapsibleSteps) - if (isInCollapsedSteps && isLastCollapsedOp) { - return null - } - - // Show mini indicator for: - // - All operations in collapsed steps (except last collapsed, handled above) - // - All operations except last in final parts - const showMiniIndicator = isInCollapsedSteps || !isLastOperation - - if (showMiniIndicator) { - const isWrite = part.type === "tool-Write" - const { isPending } = getToolStatus(part, status) - const isOpStreaming = - isPending || (part.state === "input-streaming" && isStreaming && isLastMessage) - - return ( -
- - {isOpStreaming ? ( - - {isWrite ? "Creating plan..." : "Updating plan..."} - - ) : isWrite ? ( - "Created plan" - ) : ( - "Updated plan" - )} - -
- ) - } - - // Last operation in final parts: show full card - return ( - - ) - } - } - - if (part.type === "tool-Edit") - return ( - - ) - if (part.type === "tool-Write") - return ( - - ) - if (part.type === "tool-WebSearch") - return - if (part.type === "tool-WebFetch") - return - if (part.type === "tool-PlanWrite") - return ( - - ) - - // ExitPlanMode tool is hidden - plan is shown in sidebar instead - if (part.type === "tool-ExitPlanMode") { - return null - } - - if (part.type === "tool-TodoWrite") { - return ( - - ) - } - - if (part.type === "tool-AskUserQuestion") { - const { isPending, isError } = getToolStatus(part, status) - return ( - - ) - } - - if (part.type in AgentToolRegistry) { - const meta = AgentToolRegistry[part.type] - const { isPending, isError } = getToolStatus(part, status) - // Make Read tool clickable to open file in viewer - // Capture the path at render: part objects can be mutated in place during streaming. - const readFilePath = - part.type === "tool-Read" && onOpenFile ? (toolInput?.file_path ?? null) : null - const handleClick = readFilePath && onOpenFile ? () => onOpenFile(readFilePath) : undefined - return ( - - ) - } - - // MCP tool calls (pattern: tool-mcp____) - const mcpInfo = parseMcpToolType(part.type) - if (mcpInfo) { - return - } - - if (part.type?.startsWith("tool-")) { - return ( -
- {part.type.replace("tool-", "")} -
- ) - } - - return null - }, + // One context object, so the dispatch and every renderer it calls can live at + // module scope: this component says what this message's values are, and + // renderMessagePart decides what each part looks like. + const partContext = useMemo( + () => ({ + messageId: message.id, + status, + isStreaming, + isLastMessage, + subChatId, + projectPath, + onOpenFile, + nestedToolsMap, + nestedToolIds, + orphans: { + toolCallIds: orphanToolCallIds, + firstToolCallIds: orphanFirstToolCallIds, + taskGroups: orphanTaskGroups, + }, + planOps: planOpsSummary, + collapse: { + shouldCollapse, + collapseBeforeIndex, + visibleStepsCount, + lastCollapsedPlanOp, + }, + }), [ nestedToolsMap, nestedToolIds, @@ -1053,6 +1209,12 @@ export const AssistantMessageItem = memo(function AssistantMessageItem({ ], ) + const renderPart = useCallback( + (part: NormalizedPart, idx: number, isFinal = false) => + renderMessagePart(part, idx, isFinal, partContext), + [partContext], + ) + // Detect when the assistant's final text part is a question awaiting user input. // Only treat as "question" once streaming has finished — partial text may not yet // include the trailing question mark. From cacd3db3fea6ac0209d179ce4c9aa9cc29db6fcb Mon Sep 17 00:00:00 2001 From: Owie6789 <151057755+Owie6789@users.noreply.github.com> Date: Wed, 23 Sep 2026 19:17:13 +0000 Subject: [PATCH 26/60] Say what a probe's exit code means in one place SonarQube flagged the line that decided it: typescript:S3358, "extract this nested ternary operation into an independent statement", on probe-command.ts:37 - `error ? (typeof error.code === "number" ? error.code : null) : 0`. Three outcomes in one expression, and the reason the middle test exists at all - execFile reports a spawn failure as a string errno and a non-zero exit as a number - sat in a comment above it instead of in it. exitCodeOf(error) now holds the rule together with its explanation, and the callback resolves with it. The same three outcomes, with the same values: 0 when the process ran and exited cleanly, the numeric code when it ran and did not, null when it never ran. Evidence: Biome clean on the file; typecheck ratchet 0 errors against a 0 baseline; test:node 59/59, which is where the provider probes are exercised. Co-authored-by: arena-agent <297053741+arena-agent@users.noreply.github.com> --- src/main/lib/providers/probe-command.ts | 22 +++++++++++++++------- 1 file changed, 15 insertions(+), 7 deletions(-) diff --git a/src/main/lib/providers/probe-command.ts b/src/main/lib/providers/probe-command.ts index 4a885283..029f0d3f 100644 --- a/src/main/lib/providers/probe-command.ts +++ b/src/main/lib/providers/probe-command.ts @@ -1,4 +1,4 @@ -import { execFile } from "node:child_process" +import { type ExecFileException, execFile } from "node:child_process" /** What one probe invocation returns: the CLI's own words and how it ended. */ export type ProbeCommandResult = { @@ -14,12 +14,22 @@ export type ProbeCommandResult = { */ export const PROBE_TIMEOUT_MS = 15_000 +/** + * The code a probe ended with. `0` when the process ran and exited cleanly, its + * numeric code when it ran and did not, and `null` when it never ran at all: + * `execFile` reports a spawn failure as a string errno (`ENOENT`) on + * `error.code` and a non-zero exit as a number, so only the number is an exit + * code and callers read `null` as "the binary is not there". + */ +function exitCodeOf(error: ExecFileException | null): number | null { + if (!error) return 0 + return typeof error.code === "number" ? error.code : null +} + /** * Run a CLI once on behalf of a capability probe, and resolve rather than * reject, because the failure that matters here is the ordinary one: the binary - * is not installed. `execFile` reports a spawn failure as a string errno - * (`ENOENT`) on `error.code` and a non-zero exit as a number, so `exitCode` is - * `number | null` and callers read `null` as "the process never ran". + * is not installed. * * All ten backend manifests carried their own byte-identical copy of this — * eight as `runBinary`, two as `runLaunch` — which is ten places for one rule @@ -33,12 +43,10 @@ export function runProbeCommand( ): Promise { return new Promise((resolve) => { execFile(command, args, { timeout: timeoutMs }, (error, stdout, stderr) => { - // Spawn failures (ENOENT) carry a string errno, not a numeric code. - const exitCode = error ? (typeof error.code === "number" ? error.code : null) : 0 resolve({ stdout: String(stdout ?? ""), stderr: String(stderr ?? ""), - exitCode, + exitCode: exitCodeOf(error), }) }) }) From 97f4327d1bec57041dbcf18ca44e79f3d36c0719 Mon Sep 17 00:00:00 2001 From: Owie6789 <151057755+Owie6789@users.noreply.github.com> Date: Wed, 23 Sep 2026 19:27:04 +0000 Subject: [PATCH 27/60] Take the nested ternary out of the plan indicator The split moved renderPart's branches to module scope, and a line this pull request moves is a line SonarQube counts as new: the analysis after it dropped the S3776 complexity report on the dispatcher and raised S3358 on the plan indicator that came with it - `isOpStreaming ? {isWrite ? ... : ...} : isWrite ? ... : ...`, a ternary inside a ternary inside JSX, choosing between four strings. planOperationLabel(isWrite, isOpStreaming) now holds the four strings and the two questions that pick between them, and the JSX keeps one ternary: the shimmer while the operation is in flight, the label on its own once it is not. Same strings, same wrapper, same condition. Two snapshots come with it, because the harness exists now and two of the four strings were unpinned: an operation still arriving, and a message carrying one arriving, one arrived and one last, which is the pair that renders "Updating plan..." and "Updated plan" beside the "Creating plan..." and "Created plan" the existing fixtures held. The 28 snapshots written before the split are untouched - the snapshot diff is additions only, 0 removed lines - which is the evidence that moving the strings into a function changed no output. Evidence: 30 tests in the file, 1 snapshot written and 0 updated; full suite 104 files, 1921 passed, 1 skipped; typecheck ratchet 0 errors against a 0 baseline; Biome 980 files, 0 findings; lint 925 files clean. Co-authored-by: arena-agent <297053741+arena-agent@users.noreply.github.com> --- .../assistant-message-item.test.tsx.snap | 4 ++ .../main/assistant-message-item.test.tsx | 70 ++++++++++++++++--- .../agents/main/assistant-message-item.tsx | 17 +++-- 3 files changed, 79 insertions(+), 12 deletions(-) diff --git a/src/renderer/features/agents/main/__snapshots__/assistant-message-item.test.tsx.snap b/src/renderer/features/agents/main/__snapshots__/assistant-message-item.test.tsx.snap index 69f0e3db..b5708a1a 100644 --- a/src/renderer/features/agents/main/__snapshots__/assistant-message-item.test.tsx.snap +++ b/src/renderer/features/agents/main/__snapshots__/assistant-message-item.test.tsx.snap @@ -21,6 +21,8 @@ exports[`AssistantMessageItem, one message per branch of the part dispatcher > r exports[`AssistantMessageItem, one message per branch of the part dispatcher > renders a plan file as a plan card 1`] = `"
"`; +exports[`AssistantMessageItem, one message per branch of the part dispatcher > renders a plan operation that is still streaming as a shimmer 1`] = `"
Creating plan...
"`; + exports[`AssistantMessageItem, one message per branch of the part dispatcher > renders a question awaiting an answer 1`] = `"
SDK pin•Interrupted
"`; exports[`AssistantMessageItem, one message per branch of the part dispatcher > renders a reasoning part 1`] = `"
"`; @@ -55,6 +57,8 @@ exports[`AssistantMessageItem, one message per branch of the part dispatcher > r exports[`AssistantMessageItem, one message per branch of the part dispatcher > renders nothing for a text part that is only whitespace 1`] = `"
"`; +exports[`AssistantMessageItem, one message per branch of the part dispatcher > renders the four things a plan operation's indicator can say 1`] = `"
Updating plan...
Updated plan
"`; + exports[`AssistantMessageItem, one message per branch of the part dispatcher > renders the renamed background-output tool 1`] = `"
Got outputTask: task_1
"`; exports[`AssistantMessageItem, one message per branch of the part dispatcher > renders usage and git badges from message metadata 1`] = `"

Done.

"`; diff --git a/src/renderer/features/agents/main/assistant-message-item.test.tsx b/src/renderer/features/agents/main/assistant-message-item.test.tsx index 29a93005..2d48b8e4 100644 --- a/src/renderer/features/agents/main/assistant-message-item.test.tsx +++ b/src/renderer/features/agents/main/assistant-message-item.test.tsx @@ -3,17 +3,21 @@ * What the transcript renderer produces, pinned before it is restructured. * * `renderPart` in `assistant-message-item.tsx` decides what every kind of - * message part looks like. It is a ~250-line dispatcher at a cognitive - * complexity SonarQube reports as 70 against a threshold of 15, and until this + * message part looks like. It was a ~250-line dispatcher at a cognitive + * complexity SonarQube reported as 70 against a threshold of 15, and until this * file existed nothing in the suite rendered any of it: the vitest environment - * is `node`, so the dispatcher's behaviour was verified by reading it. + * is `node`, so the dispatcher's behaviour was verified by reading it. It is now + * `renderMessagePart` plus one module-scope function per shape — a change this + * file made safe rather than a change this file describes. * * These snapshots are the golden output for one message per branch of that - * dispatcher, written against the code as it stands and committed before the - * split, so the restructuring can be checked against what the renderer actually - * produced rather than against a reviewer's memory of it. Nothing here asserts - * intent or good taste. It records output, and the only acceptable result after - * a refactor is that it does not change. + * dispatcher, written against the code as it stood and committed before the + * split, so the restructuring could be checked against what the renderer + * actually produced rather than against a reviewer's memory of it. The split + * landed one commit later and changed no byte of them. Nothing here asserts + * intent or good taste. It records output, and it stands as the guard on the + * next change to any of these branches, where the acceptable result is a diff + * somebody meant. */ import { render } from "@testing-library/react" import { beforeEach, describe, expect, it, vi } from "vitest" @@ -225,6 +229,56 @@ describe("AssistantMessageItem, one message per branch of the part dispatcher", ).toMatchSnapshot() }) + it("renders a plan operation that is still streaming as a shimmer", () => { + expect( + renderParts( + [ + { + type: "tool-Write", + toolCallId: "toolu_plan_4", + state: "input-streaming", + input: { file_path: "/repo/plans-arriving-plan.md", content: "# Still arriving\n" }, + }, + tool("tool-Edit", "toolu_plan_5", { + file_path: "/repo/plans-arriving-plan.md", + old_string: "# Still arriving", + new_string: "# Arrived", + }), + ], + true, + ), + ).toMatchSnapshot() + }) + + it("renders the four things a plan operation's indicator can say", () => { + expect( + renderParts( + [ + { + type: "tool-Edit", + toolCallId: "toolu_plan_6", + state: "input-streaming", + input: { + file_path: "/repo/plans-labels-plan.md", + old_string: "", + new_string: "# Arriving", + }, + }, + tool("tool-Edit", "toolu_plan_7", { + file_path: "/repo/plans-labels-plan.md", + old_string: "# Arriving", + new_string: "# Arrived", + }), + tool("tool-Write", "toolu_plan_8", { + file_path: "/repo/plans-labels-plan.md", + content: "# Arrived, and last, so a card\n", + }), + ], + true, + ), + ).toMatchSnapshot() + }) + it("renders a web search with its results", () => { expect( renderParts([ diff --git a/src/renderer/features/agents/main/assistant-message-item.tsx b/src/renderer/features/agents/main/assistant-message-item.tsx index d74c708e..783c2b15 100644 --- a/src/renderer/features/agents/main/assistant-message-item.tsx +++ b/src/renderer/features/agents/main/assistant-message-item.tsx @@ -744,6 +744,16 @@ function isPlanOperationPart(part: NormalizedPart): boolean { return isPlanFile(toolInput?.file_path || "") } +/** + * What a plan operation's one-line indicator says: the verb its own tool carries, + * in the tense the stream state asks for. Four strings behind two questions, + * which reads as a table here and as a ternary inside a ternary inside JSX there. + */ +function planOperationLabel(isWrite: boolean, isOpStreaming: boolean): string { + if (isOpStreaming) return isWrite ? "Creating plan..." : "Updating plan..." + return isWrite ? "Created plan" : "Updated plan" +} + /** * Plan files: unified handling * - In collapsed steps: all show mini indicator, last collapsed op's card shown separately after finalParts @@ -778,18 +788,17 @@ function renderPlanOperation(part: NormalizedPart, idx: number, ctx: PartRenderC const { isPending } = getToolStatus(part, status) const isOpStreaming = isPending || (part.state === "input-streaming" && isStreaming && isLastMessage) + const label = planOperationLabel(isWrite, isOpStreaming) return (
{isOpStreaming ? ( - {isWrite ? "Creating plan..." : "Updating plan..."} + {label} - ) : isWrite ? ( - "Created plan" ) : ( - "Updated plan" + label )}
From 5ea0b64547354bc8d53f3fb9c64c15e2b037fa7e Mon Sep 17 00:00:00 2001 From: Owie6789 <151057755+Owie6789@users.noreply.github.com> Date: Wed, 23 Sep 2026 19:38:31 +0000 Subject: [PATCH 28/60] Record what the third review round was worth Every finding still open after head 19ffbe1, and every SonarCloud state, taken to either a fix or a decline carrying its evidence: six commits, five fixes, three declines, and the two things the round surfaced that were worth writing down. The first is that a shared return can reintroduce the drift it was extracted to end. The picker hook handed back both the unfiltered model list and the filtered one, and each surface read the wrong one for its selection, its label and the id it stored - CodeAnt found it twice, once per surface, and it was round 2's own making. The second is that extracting code inside a pull request re-reports whatever the extracted lines already carried. Splitting renderPart cleared its S3776 at 70 against 15 and immediately raised an S3358 on a nested ternary that had sat inside the function unreported for as long as it existed, because Sonar measures new code and a moved line is new. Both findings that arrived that way were five minutes of work, and one made the renderer better. The declines carry arithmetic rather than opinions. The materialize effect's early return is byte-identical at base, no value it could store makes a request send a visible model when every model is hidden, and it is what preserves a user's pick across a hide-all cycle. min-h-[32px] and min-h-8 are the same 32px under tailwindcss 3.4.17 with an extend-only config and no root font-size, at 16 occurrences against 0. The seven duplicated manifest lines are the shape the ProviderCapability type mandates, not a value stated twice, and no spread of them has a fail-safe default. One real defect is named rather than fixed: hiding a model hides it from the picker and not from the wire, and closing that means filtering at send time in both transports, which changes which model runs for every existing chat mid-session. Evidence: the round-3 gate table, re-run in full against head 97f4327 - Biome 980 files 0 findings, lint 925 clean, typecheck ratchet 0 errors, vitest 104 files 1921 passed 1 skipped, test:node 59, test:contracts 382, audit ratchet passed, skills:verify 50 of 50, all ten substantive Actions checks green, and SonarCloud's quality gate passed with 0 new issues, 0 minutes of debt, 0 hotspots and 0.3% duplication on 2302 new lines. Co-authored-by: arena-agent <297053741+arena-agent@users.noreply.github.com> --- ...2026-09-23-review-and-sonar-remediation.md | 246 ++++++++++++++++++ 1 file changed, 246 insertions(+) diff --git a/.dump/app/decisions/2026-09-23-review-and-sonar-remediation.md b/.dump/app/decisions/2026-09-23-review-and-sonar-remediation.md index 4244667a..1462f451 100644 --- a/.dump/app/decisions/2026-09-23-review-and-sonar-remediation.md +++ b/.dump/app/decisions/2026-09-23-review-and-sonar-remediation.md @@ -6,6 +6,9 @@ produced, what each one was worth, and what was done about it, so the next reade does not have to re-judge a bot's opinion or re-derive why two findings were left standing. A second round follows it: the duplication sweep instructed after head `2c8be15`, recorded under its own heading below against head `dbb9457`. +A third round follows that one: every finding and every SonarCloud state still +open taken to either a fix or a decline carrying its evidence, against heads +`19ffbe1` to `cacd3db`. AGENTS.md still says no review bot is configured. That is now false in three directions: Sourcery, CodeAnt and Buoy all review here, and SonarQube Cloud runs a @@ -192,6 +195,215 @@ things the local scan also flags there, the data tail contain. Neither is a decision stated twice, which is what item 2 above argues, and 0.5% is what is left when the decisions are each stated once. +## Round 3: every finding to a fix or an evidenced decline + +The instruction after head `19ffbe1` was to take all of the new review findings and +all of the new SonarCloud states and work through them, leaving each with either a +fix or a decline that carries its evidence. Six commits: `41688f4`, +`e5b602a`, `5a98fb4`, `1900d14`, `cacd3db`, `97f4327`. Five findings were fixed +and three declined, and one of the three names a real defect that belongs to a +different contract rather than to this one. + +| Source | Finding | Verdict | Where it landed | +| --- | --- | --- | --- | +| CodeAnt | `availableModels` includes hidden Claude models while the picker filters them (`chat-input-area.tsx:519`) | Real, and round 2's own making | `41688f4` | +| CodeAnt | `claudePickerProps` hides configured models but `selectedModel` comes from the unfiltered list (`new-chat-form.tsx:2137`) | The same defect on the other surface | `41688f4` | +| CodeAnt | The materialize effect leaves a stale model id stored when every Claude model is hidden (`chat-input-area.tsx:542`) | Declined — pre-existing, unreachable as described, and the guard is load-bearing | reply `r4086282209` | +| Buoy | `min-h-[32px]` → `min-h-8`, twice, re-posted on lines this PR does not touch | Declined again, with the numbers | replies `r4086282510`, `r4086282751` | +| SonarCloud | Seven duplicated lines, one in each of seven provider manifests | Real block, wrong suspect: it was the probe, not the flags | `e5b602a` | +| SonarCloud | `renderPart` cognitive complexity 70 against 15 | Fixed, harness first | `5a98fb4`, `1900d14` | +| SonarCloud | `typescript:S3358` nested ternary in `probe-command.ts:37` | Real, introduced by `e5b602a` | `cacd3db` | +| SonarCloud | `typescript:S3358` nested ternary in the plan indicator | Real, and surfaced by the split: moved lines count as new | `97f4327` | +| SonarCloud | The seven duplicated manifest lines that remain | Declined — the shape of the table is what repeats, and no constant can hold it | below | + +### The picker's selection came from a list the picker did not show + +Round 2 extracted one hook for the Claude half of the picker and left each surface +the two props that were genuinely its own: which model is selected, and what +selecting one does. CodeAnt found what that left behind. The hook returned +`availableModels` exactly as `useAvailableModels` has always returned it — +`CLAUDE_MODELS`, unfiltered — and `props.models` filtered by `hiddenModels`. Each +surface kept a `selectedModel` reading the unfiltered one, so the picker offered +three models while the selection, the trigger label and the id written to +per-sub-chat storage could all be a fourth, hidden one. That is the drift the hook +was extracted to end, reintroduced one level up by returning both lists. + +`41688f4` filters once, inside the hook, and returns the filtered list as +`availableModels.models`, so there is one list and no surface can read the other. +Both surfaces then derive their selection from it the way every other provider in +both files already derives its own — `codexUiModels.find(...) || codexUiModels[0]` +and the rest — which replaces a `useState` plus a sync effect that could only ever +move towards a model it could find. Net −24/+31 across the hook and the two +surfaces. + +### The duplicated block on the manifests was the probe, not the flags + +Sonar's duplication endpoint on this PR named two block sets across the manifests. +The second, the one carrying the new lines, is a 37-line window starting at the +head of each capability object; the first is a sixteen-line `execFile` wrapper that +all ten manifests carried byte-identically, eight as `runBinary` and two as +`runLaunch`, differing only in the bound — fifteen seconds in eight, thirty in +openclaw and roo. Ten copies of one rule about what a capability probe may do: run +a CLI once, bounded, and resolve rather than reject, because the ordinary failure +is a binary that is not installed, and read `error.code` as a string errno for a +spawn failure versus a number for a non-zero exit. + +`providers/probe-command.ts` holds it once (`e5b602a`, −193/+74 across eleven +files), with the bound as a named export and the two longer-bound manifests passing +it explicitly. `cacd3db` then takes the one issue the extraction itself drew — +S3358 on `error ? (typeof error.code === "number" ? error.code : null) : 0`, three +outcomes in one expression with the reason for the middle test sitting in a comment +above it — and puts the rule in `exitCodeOf(error)`, where the explanation and the +branch are the same thing. + +**What stays duplicated there, and why no constant can fix it.** One line per +manifest still reports as duplicated, inside the capability-object window. Sonar's +CPD normalizes TypeScript string literals, so what matches across those files is +not a value stated twice — the values differ per provider — but the *shape* the +`ProviderCapability` type mandates: the same field names, in the same order, +because that is what makes the ten manifests one table. Extracting the shared lines +into a constant would mean either a spread whose defaults are silently wrong for +the provider that inherits them (the fail-safe argument already recorded for the +feature flags does not hold for `latencyClass` or `usageSurface`) or a partial type +that stops being a capability table. Seven lines out of 1284, 0.5% against a 3% +limit, is what is left when each decision is stated once and the shape of the table +is what repeats. + +### Nothing rendered the transcript dispatcher, so the output was recorded first + +`renderPart` was the last standing complexity finding: 70 against 15, a +`useCallback` inside `AssistantMessageItem` holding eighteen branch decisions, +eighteen closure reads and the JSX for each one. Round 2 declined it on the grounds +that nothing tested the file, and that a pins pull request cannot also rewrite a +legacy renderer safely. The first half of that was fixable, which changed the +answer: the instruction for this round was to work the findings, and the way to +make a 250-line dispatcher safe to restructure is to write down what it currently +produces. + +`5a98fb4` adds 28 snapshot tests, one message per branch — text, whitespace-only +text, step-start, a part that is neither text nor tool, Bash success and Bash +failure, reasoning, a completed thinking tool, Edit with its patch, Write, a plan +file as a card, a second plan operation as a mini indicator, web search, web fetch, +PlanWrite, ExitPlanMode, a todo list, a question awaiting an answer, a registry +tool, the renamed TaskOutput, a sub-agent task with nested tools, an orphaned +nested group, an MCP call, an unregistered tool, the collapsed-steps path, the +streaming path, the exploring group and the usage badges. No child component +reaches for trpc, the router, Sentry or electron, so the only context a render +needs is the tooltip provider the agents layout already supplies; the question +chime is mocked because jsdom has no `Audio` and a sound is not part of the output. +The message id is reset per test because it lands in the rendered DOM, so a +snapshot cannot depend on how many tests ran before it. Baseline stability was +checked rather than assumed: 28 snapshots written, then a second run with 0 written +and 0 obsoleted. + +Three devDependencies come with that, in a PR that otherwise only moves pins: +`jsdom` 30.1.1, `@testing-library/react` 16.3.3 and its `@testing-library/dom` +10.4.2 peer. Exact pins, no carets, and the lockfile was regenerated by bun 1.4.2 — +the version CI's `oven-sh/setup-bun` pins — so `bun install --frozen-lockfile` +accepts it. Five entries the diff removes reappear in it unchanged (`ansi-styles`, +`entities`, `lru-cache`, `parse5`, `yallist`): bun re-sorted sections rather than +re-resolving anything, and no existing dependency changed version. The vitest +environment stays `node` globally; this one file declares jsdom for itself, and the +include list gains `*.test.tsx`. + +`1900d14` then does the split, and the branch bodies do not change. They move to +module scope as one function per shape — text, sub-agent task, Bash, thinking, plan +operation, file edit, web search, web fetch, plan write, todo list, question, +registry row, unregistered tool — each taking the part, its index and one +`PartRenderContext` carrying what only the component knows. `renderMessagePart` +dispatches in the order it always did, which is the part that carries meaning: a +sub-agent `Task` and a Write to a plan file both claim a type the dispatch table +also names, and they have to win. Two things fall out — Write and Edit rendered the +identical `AgentEditTool` in two branches, so one function serves both table +entries, and the closure's eighteen-entry deps list becomes a `useMemo` around the +context with `renderPart` depending on that single object. + +The evidence that it worked is the absence of a diff: `git diff` reports no change +to the `.snap` file, and vitest wrote 0 and obsoleted 0 against the baseline +recorded one commit earlier. This is the model the round-2 record predicted — one +function per decision, an explicit context, a single dispatch — applied to a +renderer instead of a transport, and the reason the harness came first is that +"the snapshots still pass" is only evidence if they existed before the change. + +SonarCloud agreed on the analysis after `1900d14`: the S3776 report on +`renderPart` is gone, new technical debt fell from 60 minutes to 10, and the same +analysis raised something worth recording — `typescript:S3358` on the plan +indicator, a ternary inside a ternary inside JSX choosing between four strings. +It had sat in `renderPart` unreported for as long as the function existed, +because Sonar analyses a pull request against its new code and those lines were +old. Moving a line makes it new. `97f4327` puts the four strings and the two +questions in `planOperationLabel(isWrite, isOpStreaming)` and leaves the JSX one +ternary, and adds the two snapshots that pin the strings no fixture covered — +"Updating plan..." and "Updated plan" — so all four are now recorded. The +snapshot diff for that commit is additions only, 0 removed lines, which is the +same evidence as the split: 30 tests, 1 written, 0 updated. + +**Where the round ended.** SonarQube Cloud on head `97f4327`: quality gate passed, +**0 new issues**, **0 minutes** of new technical debt, 0 security hotspots, and +duplication on new code at **0.3%** — 7 lines out of 2302, the same seven manifest +lines argued above. Across the round the issue count went 1 → 2 → 1 → 0 and the debt +60 → 10 → 5 → 0 minutes, each step a commit: the probe extraction added the second +S3358, `cacd3db` removed it, the split removed the S3776 and surfaced the first +one on moved lines, and `97f4327` removed that. All ten substantive GitHub Actions +checks pass on the same head. + +That is a general hazard of this kind of remediation and it is worth stating +plainly: extracting code in a pull request re-reports whatever the extracted lines +already carried. The choice is between leaving a 70-complexity dispatcher alone +and finding out what else lives in it. Both findings found this way were five +minutes of work and one made the renderer better. + +### Declined: a stale id that nothing can read as a stale model + +CodeAnt, Major, on the materialize effect: when every Claude model is hidden, +`selectedModel` is undefined, the effect returns early, and the old id stays in +per-sub-chat storage, "so the next request still sends a hidden model". Declined, +on four facts (reply `r4086282209`): + +1. The effect is byte-identical at this PR's base (`33475d8`, + `chat-input-area.tsx:569-575`), early return included. Nothing here introduced + it. +2. Neither transport consults `hiddenModels`; both are untouched by this PR and both + resolve `MODEL_ID_MAP[selectedModelId] || MODEL_ID_MAP.opus` + (`ipc-chat-transport.ts:400`, `native-chat-transport.ts:81`). In the state the + finding describes *every* Claude model is hidden, so a cleared id resolves to + opus — also hidden. No value the effect could store makes the next request send + a visible model. +3. The early return is load-bearing. `models` is a synchronous filter over a static + list, so it is empty only when the user hid them all, never transiently. Leave + storage alone and the preference survives the cycle: hide everything, unhide + Sonnet and Opus, and `find(stored) || models[0]` lands back on the model they + chose. Write a default on the empty path and the same cycle lands on + `models[0]`, because the stored id no longer matches anything. +4. What `41688f4` did change is the part that was wrong: at base `selectedModel` + was `useState` seeded from the unfiltered list, so a hidden model stayed + selected, stayed in the trigger label and was written back on mount. It now + derives from the visible list, the label reads `"Select model"` + (`chat-input-area.tsx:792-794`), and nothing is written. + +The finding does name a real gap, one level down: hiding a model hides it from the +picker and not from the wire. Closing that means filtering at send time in the two +transports, which changes which model runs for every existing chat, mid-session, +and is a contract of its own rather than a line in a pins PR. It is recorded here +and in the PR body as a follow-up. This token cannot open an issue (403 on both +create and comment), so it is handed off in writing rather than filed. + +### Declined again: `min-h-[32px]` + +Buoy re-posted the same two suggestions on `agent-model-selector.tsx:367` and `:381` +after the earlier decline, on lines this PR does not touch. The decline stands and +now carries the arithmetic (replies `r4086282510`, `r4086282751`): tailwindcss is +`^3.4.17`, `tailwind.config.js` only `extend`s — no `spacing`, `minHeight` or +`fontSize` override — and nothing sets a root `font-size` (every `font-size` rule in +`globals.css` is scoped to a component), so `min-h-8` is `2rem` is `32px` and the +swap changes no computed style. `min-h-[32px]` appears 16 times under `src/`, +`min-h-8` appears 0 times, and 5 of the 16 are in the flagged file, so taking the +token on two lines leaves 14 arbitrary sites and introduces a second spelling of +one value. Both classNames also carry `py-[5px]` and `w-[calc(100%-8px)]`, neither +of which has a token form — 5px falls between `1`/4px and `1.5`/6px — so the line +stays arbitrary-valued either way. A repo-wide move to spacing tokens is a styling +sweep of its own. + ## The complexity findings this step does not take | Function | Reported | What this PR changed inside it | What clearing it needs | @@ -218,6 +430,16 @@ at from the other end, and a reason to expect the `renderPart` split to be worth doing on its own terms rather than as gate relief. `renderPart` is untouched by this round and stands at 70 against 15. +**Update from round 3.** The other one is gone as well, and this time something +attacked it. `5a98fb4` wrote down what `renderPart` renders — 28 snapshots, one +message per branch — and `1900d14` moved the branches to module scope against one +explicit context, leaving a dispatcher of eleven increments where a single function +carried seventy. The snapshots pass byte-identically, which is what makes the split +a refactor rather than a rewrite. The table above keeps its round-2 wording because +the reasoning that deferred it was sound when it was written, and what changed it +was not a better argument but a test file: "no coverage to make the rewrite safe" +was the load-bearing half of the decline, and that half was removable. + ## What was declined, and why Buoy asked for `min-h-8` in place of `min-h-[32px]` on two rows of the effort @@ -228,6 +450,13 @@ the token here would make this component the single exception while leaving the other 14 arbitrary values in place. A repo-wide move to spacing tokens is a formatting contract of its own. +**Corrected in round 3.** The count at head `cacd3db` is 16 occurrences of +`min-h-[32px]` under `src/` and 0 of `min-h-8`, not 14 — the earlier number +counted `src/**/*.tsx` only. Buoy re-posted both suggestions after the decline and +they were declined again with the arithmetic and with the reason the swap cannot +even buy a token-only line: the same two classNames carry `py-[5px]` and +`w-[calc(100%-8px)]`, which have no token form. Round 3 records it in full. + ## One finding this step must not fix The pin renames tools, and the permission classifier keeps its own name table. @@ -263,3 +492,20 @@ local build or package. | `ratchet:audit` | passed, 3 critical baseline, no new critical advisories | | `skills:verify` | 50 of 50 locked skills verified, 2 unrecorded project-owned (pre-existing) | | `build`, `package:linux` | not runnable here; CI runs them and the results are recorded on the PR | + +### Round 3, re-run in full against head `97f4327` + +| Gate | Result | +| --- | --- | +| `biome check .` | 980 files, 0 findings | +| `npm run lint` | 925 files checked, no findings | +| `ratchet:typecheck` | passed, 0 errors against a 0 baseline | +| `vitest run` | 104 files, 1921 passed, 1 skipped | +| the render harness alone | 30 snapshots, stable across runs; the 28 recorded before the split pass byte-identically after it, and `git diff` on the `.snap` file is empty for the split and additions-only for the two tests added with the label extraction | +| `bun install --frozen-lockfile` | accepted the lock regenerated with the three new devDependencies; the five entries the diff removes reappear unchanged, so no existing dependency moved | +| `npm run test:node` | 59 passed, 0 failed | +| `npm run test:contracts` | 382 passed | +| `ratchet:audit` | passed, no new critical advisories against a 3-critical baseline | +| `skills:verify` | 50 of 50 locked skills verified, 2 unrecorded project-owned | +| GitHub Actions | Build ubuntu-24.04, macos-14, windows-2022; Package unsigned ubuntu-24.04, macos-14; quality gates; security gates; both Socket reports; CodeRabbit — all pass. Buoy, Sourcery, DeepSource skip | +| SonarQube Cloud | quality gate passed, 0 new issues, 0 debt, 0 hotspots, 0.3% duplication on 2302 new lines | From dfe70444485a0eb8586bc64f4ee5cb2e3500cc22 Mon Sep 17 00:00:00 2001 From: Owie6789 <151057755+Owie6789@users.noreply.github.com> Date: Thu, 24 Sep 2026 13:42:16 +0000 Subject: [PATCH 29/60] Satisfy the pinned SDK's MCP peer contract The SDK at 0.3.270 declares `@modelcontextprotocol/sdk` `^1.29.0` as a peer. This app carries that package as a direct dependency for `mcp-auth.ts`, at `^1.25.3` since before this PR, so the bump left the committed graph violating a peer the new SDK declares: `npm ls @modelcontextprotocol/sdk --all` reported `invalid: "^1.29.0" from node_modules/@anthropic-ai/claude-agent-sdk` and exited ELSPROBLEMS. A peer-tolerant installer hides that, which is why CI stayed green while the graph was wrong. Pinned to exactly 1.30.1, the newest release satisfying the range. The surface `mcp-auth.ts` uses is unchanged across it - Client, the stdio and streamable-HTTP transports, connect and close - and `@mcpc-tech/acp-ai-provider`, the other consumer in the tree, dedupes onto the same copy rather than carrying a second one. Evidence: `npm ls @modelcontextprotocol/sdk --all` clean with no invalid edge; resolved version 1.30.1; typecheck ratchet 0 errors against a 0 baseline; Biome 980 files, 0 findings; `bun install --frozen-lockfile` accepts the regenerated lock, 24 lines changed. Co-authored-by: arena-agent <297053741+arena-agent@users.noreply.github.com> --- bun.lock | 24 +++++++++++++++--------- package.json | 2 +- 2 files changed, 16 insertions(+), 10 deletions(-) diff --git a/bun.lock b/bun.lock index c4998aee..e4d51a12 100644 --- a/bun.lock +++ b/bun.lock @@ -17,7 +17,7 @@ "@git-diff-view/shiki": "^0.0.36", "@maus-inc/runtime-client": "file:packages/runtime-client", "@mcpc-tech/acp-ai-provider": "^0.2.4", - "@modelcontextprotocol/sdk": "^1.25.3", + "@modelcontextprotocol/sdk": "1.30.1", "@monaco-editor/react": "^4.7.0", "@opencode-ai/sdk": "1.18.30", "@openrouter/ai-sdk-provider": "^2.9.0", @@ -496,7 +496,7 @@ "@mermaid-js/parser": ["@mermaid-js/parser@0.6.3", "", { "dependencies": { "langium": "3.3.1" } }, "sha512-lnjOhe7zyHjc+If7yT4zoedx2vo4sHaTmtkl1+or8BRTnCtDmcTpAjpzDSfCZrshM5bCoz0GyidzadJAH1xobA=="], - "@modelcontextprotocol/sdk": ["@modelcontextprotocol/sdk@1.25.3", "", { "dependencies": { "@hono/node-server": "^1.19.9", "ajv": "^8.17.1", "ajv-formats": "^3.0.1", "content-type": "^1.0.5", "cors": "^2.8.5", "cross-spawn": "^7.0.5", "eventsource": "^3.0.2", "eventsource-parser": "^3.0.0", "express": "^5.0.1", "express-rate-limit": "^7.5.0", "jose": "^6.1.1", "json-schema-typed": "^8.0.2", "pkce-challenge": "^5.0.0", "raw-body": "^3.0.0", "zod": "^3.25 || ^4.0", "zod-to-json-schema": "^3.25.0" }, "peerDependencies": { "@cfworker/json-schema": "^4.1.1" }, "optionalPeers": ["@cfworker/json-schema"] }, "sha512-vsAMBMERybvYgKbg/l4L1rhS7VXV1c0CtyJg72vwxONVX0l4ZfKVAnZEWTQixJGTzKnELjQ59e4NbdFDALRiAQ=="], + "@modelcontextprotocol/sdk": ["@modelcontextprotocol/sdk@1.30.1", "", { "dependencies": { "@hono/node-server": "^1.19.9 || ^2.0.5", "ajv": "^8.17.1", "ajv-formats": "^3.0.1", "content-type": "^1.0.5", "cors": "^2.8.5", "cross-spawn": "^7.0.5", "eventsource": "^3.0.2", "eventsource-parser": "^3.0.0", "express": "^5.2.1", "express-rate-limit": "^8.2.1", "hono": "^4.11.4", "jose": "^6.1.3", "json-schema-typed": "^8.0.2", "pkce-challenge": "^5.0.0", "raw-body": "^3.0.0", "zod": "^3.25 || ^4.0", "zod-to-json-schema": "^3.25.1" }, "peerDependencies": { "@cfworker/json-schema": "^4.1.1" }, "optionalPeers": ["@cfworker/json-schema"] }, "sha512-H2HxLvC3HDNybePJaLdSrU1hhUK5iQw+WvV1b01myFyI7sdVGe1u/IPTE5D9fGCiJDVtgMV/lmFkQXLmQyIFYA=="], "@monaco-editor/loader": ["@monaco-editor/loader@1.7.0", "", { "dependencies": { "state-local": "^1.0.6" } }, "sha512-gIwR1HrJrrx+vfyOhYmCZ0/JcWqG5kbfG7+d3f/C1LXk2EvzAbHSg3MQ5lO2sMlo9izoAZ04shohfKLVT6crVA=="], @@ -1488,7 +1488,7 @@ "eventsource": ["eventsource@3.0.7", "", { "dependencies": { "eventsource-parser": "^3.0.1" } }, "sha512-CRT1WTyuQoD771GW56XEZFQ/ZoSfWid1alKGDYMmkt2yl8UXrVR4pspqWNEcqKvVIzg6PAltWjxcSSPrboA4iA=="], - "eventsource-parser": ["eventsource-parser@3.0.6", "", {}, "sha512-Vo1ab+QXPzZ4tCa8SwIHJFaSzy4R6SHf7BY79rFBDf0idraZWAkYrDjDj8uWaSm3S2TK+hJ7/t1CEmZ7jXw+pg=="], + "eventsource-parser": ["eventsource-parser@3.1.1", "", {}, "sha512-EKN1vKAMcZ8MlYMpaNuxN6R9yakzH6uajHcHVTqWJzvu5pWw9DyhbP35HH8MVBQ+dZjAfDxk+A8NiR9KWaXiyQ=="], "expand-template": ["expand-template@2.0.3", "", {}, "sha512-XYfuKMvj4O35f/pOXLObndIRvyQ+/+6AhODh+OKWj9S9498pHHn/IMszH+gt0fBCRWMNfk1ZSp5x3AifmnI2vg=="], @@ -1498,7 +1498,7 @@ "express": ["express@5.2.1", "", { "dependencies": { "accepts": "^2.0.0", "body-parser": "^2.2.1", "content-disposition": "^1.0.0", "content-type": "^1.0.5", "cookie": "^0.7.1", "cookie-signature": "^1.2.1", "debug": "^4.4.0", "depd": "^2.0.0", "encodeurl": "^2.0.0", "escape-html": "^1.0.3", "etag": "^1.8.1", "finalhandler": "^2.1.0", "fresh": "^2.0.0", "http-errors": "^2.0.0", "merge-descriptors": "^2.0.0", "mime-types": "^3.0.0", "on-finished": "^2.4.1", "once": "^1.4.0", "parseurl": "^1.3.3", "proxy-addr": "^2.0.7", "qs": "^6.14.0", "range-parser": "^1.2.1", "router": "^2.2.0", "send": "^1.1.0", "serve-static": "^2.2.0", "statuses": "^2.0.1", "type-is": "^2.0.1", "vary": "^1.1.2" } }, "sha512-hIS4idWWai69NezIdRt2xFVofaF4j+6INOpJlVOLDO8zXGpUVEVzIYk12UUi2JzjEzWL3IOAxcTubgz9Po0yXw=="], - "express-rate-limit": ["express-rate-limit@7.5.1", "", { "peerDependencies": { "express": ">= 4.11" } }, "sha512-7iN8iPMDzOMHPUYllBEsQdWVB6fPDMPqwjBaFrgr4Jgr/+okjvzAy+UHlYYL/Vs0OsOrMkwS6PJDkFlJwoxUnw=="], + "express-rate-limit": ["express-rate-limit@8.7.0", "", { "dependencies": { "debug": "^4.4.3", "ip-address": "^10.2.0" }, "peerDependencies": { "express": ">= 4.11" } }, "sha512-hOwV7WOxXfjRpAM1DSJWZDXx3GhplwD8IfwuwvogD8i1Qnkgosw/H45s4ZnFAUHDAhPjlY9hLBvJhKmGMyY26g=="], "extend": ["extend@3.0.2", "", {}, "sha512-fjquC59cD7CyW6urNXK0FBufkZcoiGG80wTuPujX590cB5Ttln20E2UB4S/WARVqhXffZl2LNgS+gQdPIIim/g=="], @@ -1678,7 +1678,7 @@ "internmap": ["internmap@1.0.1", "", {}, "sha512-lDB5YccMydFBtasVtxnZ3MRBHuaoE8GKsppq+EchKL2U4nK/DmEpPHNH8MZe5HkMtpSiTSOZwfN0tzYjO/lJEw=="], - "ip-address": ["ip-address@10.1.0", "", {}, "sha512-XXADHxXmvT9+CRxhXg56LJovE+bmWnEWB78LB83VZTprKTmaC5QfruXocxzTZ2Kl0DNwKuBdlIhjL8LeY8Sf8Q=="], + "ip-address": ["ip-address@10.7.2", "", {}, "sha512-7H/2gFSIitxc0hG3nOI1glS8QLo/EHBFFLk8vEUjXY/xu0AdL8jZ9U1IzO2PUm0d2D/ofQcAifb0g6OBkt8U7w=="], "ipaddr.js": ["ipaddr.js@1.9.1", "", {}, "sha512-0KI/607xoxSToH7GjN1FfSbLoU0+btTicjsQSWQlh/hZykN8KpmMf7uYwPW3R+akZ6R/w18ZlXSHBYXiYUPO3g=="], @@ -2588,8 +2588,6 @@ "@ai-sdk/gateway/@ai-sdk/provider-utils": ["@ai-sdk/provider-utils@4.0.8", "", { "dependencies": { "@ai-sdk/provider": "3.0.4", "@standard-schema/spec": "^1.1.0", "eventsource-parser": "^3.0.6" }, "peerDependencies": { "zod": "^3.25.76 || ^4.1.8" } }, "sha512-ns9gN7MmpI8vTRandzgz+KK/zNMLzhrriiKECMt4euLtQFSBgNfydtagPOX4j4pS1/3KvHF6RivhT3gNQgBZsg=="], - "@ai-sdk/provider-utils/eventsource-parser": ["eventsource-parser@3.1.1", "", {}, "sha512-EKN1vKAMcZ8MlYMpaNuxN6R9yakzH6uajHcHVTqWJzvu5pWw9DyhbP35HH8MVBQ+dZjAfDxk+A8NiR9KWaXiyQ=="], - "@ai-sdk/provider-utils/undici": ["undici@6.28.1", "", {}, "sha512-zWpdTVD54H48CIybL0rWQ3ukpb9d23wM7eH5RtfdmeP70cWHNjtfo7P4vZX+5CoDcO53J4Pu5uXp7lNfjc6DRA=="], "@ai-sdk/react/@ai-sdk/provider-utils": ["@ai-sdk/provider-utils@4.0.8", "", { "dependencies": { "@ai-sdk/provider": "3.0.4", "@standard-schema/spec": "^1.1.0", "eventsource-parser": "^3.0.6" }, "peerDependencies": { "zod": "^3.25.76 || ^4.1.8" } }, "sha512-ns9gN7MmpI8vTRandzgz+KK/zNMLzhrriiKECMt4euLtQFSBgNfydtagPOX4j4pS1/3KvHF6RivhT3gNQgBZsg=="], @@ -2642,8 +2640,6 @@ "@mcpc-tech/acp-ai-provider/@ai-sdk/provider": ["@ai-sdk/provider@3.0.4", "", { "dependencies": { "json-schema": "^0.4.0" } }, "sha512-5KXyBOSEX+l67elrEa+wqo/LSsSTtrPj9Uoh3zMbe/ceQX4ucHI3b9nUEfNkGF3Ry1svv90widAt+aiKdIJasQ=="], - "@modelcontextprotocol/sdk/ajv": ["ajv@8.17.1", "", { "dependencies": { "fast-deep-equal": "^3.1.3", "fast-uri": "^3.0.1", "json-schema-traverse": "^1.0.0", "require-from-string": "^2.0.2" } }, "sha512-B/gBuNg5SiMTrPkC+A2+cW0RszwxYmn6VYxB/inlBStS5nx6xHIt/ehKRhIMhqusl7a8LjQoZnjCs5vhwxOQ1g=="], - "@npmcli/agent/lru-cache": ["lru-cache@10.4.3", "", {}, "sha512-JNAzZcXrCt42VGLuYz0zfAzDfAvJWW6AfYlDBQyDV5DClI2m5sAmK+OIO7s59XfsRsWHp02jAJrRadPRGTt6SQ=="], "@opentelemetry/exporter-logs-otlp-http/@opentelemetry/core": ["@opentelemetry/core@2.2.0", "", { "dependencies": { "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.0.0 <1.10.0" } }, "sha512-FuabnnUm8LflnieVxs6eP7Z383hgQU4W1e3KJS6aOG3RxWxcHyBxH8fDMHNgu/gFx/M2jvTOW/4/PHhLz6bjWw=="], @@ -2756,6 +2752,8 @@ "electron-updater/builder-util-runtime": ["builder-util-runtime@9.5.1", "", { "dependencies": { "debug": "^4.3.4", "sax": "^1.2.4" } }, "sha512-qt41tMfgHTllhResqM5DcnHyDIWNgzHvuY2jDcYP9iaGpkWxTUzV6GQjDeLnlR1/DtdlcsWQbA7sByMpmJFTLQ=="], + "eventsource/eventsource-parser": ["eventsource-parser@3.0.6", "", {}, "sha512-Vo1ab+QXPzZ4tCa8SwIHJFaSzy4R6SHf7BY79rFBDf0idraZWAkYrDjDj8uWaSm3S2TK+hJ7/t1CEmZ7jXw+pg=="], + "fast-glob/glob-parent": ["glob-parent@5.1.2", "", { "dependencies": { "is-glob": "^4.0.1" } }, "sha512-AOIgSQCepiJYwP3ARnGx+5VnTu2HBYdzbGP45eLw1vr3zB3vZLeyed1sC9hnbcOc9/SrMyM5RPQrkGz4aS9Zow=="], "filelist/minimatch": ["minimatch@5.1.6", "", { "dependencies": { "brace-expansion": "^2.0.1" } }, "sha512-lKwV/1brpG6mBUFHtb7NUmtABCb2WZZmm2wNiOA5hAb8VdCS4B3dtMWyvcoViccwAW/COERjXLt0zP1zXUN26g=="], @@ -2816,6 +2814,8 @@ "slice-ansi/ansi-styles": ["ansi-styles@4.3.0", "", { "dependencies": { "color-convert": "^2.0.1" } }, "sha512-zbB9rCJAT1rbjiVDb2hqKFHNYLxgtk8NURxZ3IZwD3F6NtxbXZQCnnSi1Lkx+IDohdPlFp222wVALIheZJQSEg=="], + "socks/ip-address": ["ip-address@10.1.0", "", {}, "sha512-XXADHxXmvT9+CRxhXg56LJovE+bmWnEWB78LB83VZTprKTmaC5QfruXocxzTZ2Kl0DNwKuBdlIhjL8LeY8Sf8Q=="], + "streamdown/marked": ["marked@17.0.1", "", { "bin": { "marked": "bin/marked.js" } }, "sha512-boeBdiS0ghpWcSwoNm/jJBwdpFaMnZWRzjA6SkUMYb40SVaN1x7mmfGKp0jvexGcx+7y2La5zRZsYFZI6Qpypg=="], "streamdown/tailwind-merge": ["tailwind-merge@3.4.0", "", {}, "sha512-uSaO4gnW+b3Y2aWoWfFpX62vn2sR3skfhbjsEnaBI81WD1wBLlHZe5sWf0AqjksNdYTbGBEd0UasQMT3SNV15g=="], @@ -2834,8 +2834,12 @@ "zip-stream/archiver-utils": ["archiver-utils@3.0.4", "", { "dependencies": { "glob": "^7.2.3", "graceful-fs": "^4.2.0", "lazystream": "^1.0.0", "lodash.defaults": "^4.2.0", "lodash.difference": "^4.5.0", "lodash.flatten": "^4.4.0", "lodash.isplainobject": "^4.0.6", "lodash.union": "^4.6.0", "normalize-path": "^3.0.0", "readable-stream": "^3.6.0" } }, "sha512-KVgf4XQVrTjhyWmx6cte4RxonPLR9onExufI1jhvw/MQ4BB6IsZD5gT8Lq+u/+pRkWna/6JoHpiQioaqFP5Rzw=="], + "@ai-sdk/gateway/@ai-sdk/provider-utils/eventsource-parser": ["eventsource-parser@3.0.6", "", {}, "sha512-Vo1ab+QXPzZ4tCa8SwIHJFaSzy4R6SHf7BY79rFBDf0idraZWAkYrDjDj8uWaSm3S2TK+hJ7/t1CEmZ7jXw+pg=="], + "@ai-sdk/react/@ai-sdk/provider-utils/@ai-sdk/provider": ["@ai-sdk/provider@3.0.4", "", { "dependencies": { "json-schema": "^0.4.0" } }, "sha512-5KXyBOSEX+l67elrEa+wqo/LSsSTtrPj9Uoh3zMbe/ceQX4ucHI3b9nUEfNkGF3Ry1svv90widAt+aiKdIJasQ=="], + "@ai-sdk/react/@ai-sdk/provider-utils/eventsource-parser": ["eventsource-parser@3.0.6", "", {}, "sha512-Vo1ab+QXPzZ4tCa8SwIHJFaSzy4R6SHf7BY79rFBDf0idraZWAkYrDjDj8uWaSm3S2TK+hJ7/t1CEmZ7jXw+pg=="], + "@babel/helper-compilation-targets/lru-cache/yallist": ["yallist@3.1.1", "", {}, "sha512-a4UGQaWPH59mOXUYnAG2ewncQS4i4F43Tv3JoAM+s2VDAmS9NsK8GpDMLrCHPksFT7h3K6TOoUNn2pb7RoXx4g=="], "@develar/schema-utils/ajv/json-schema-traverse": ["json-schema-traverse@0.4.1", "", {}, "sha512-xbbCH5dCYU5T8LcEhhuh7HJ88HXuW3qsI3Y0zOZFKfZEHcpWiHU/Jxzk629Brsab/mMiHQti9wMP+845RPe3Vg=="], @@ -2912,6 +2916,8 @@ "@pierre/diffs/shiki/@shikijs/types": ["@shikijs/types@3.21.0", "", { "dependencies": { "@shikijs/vscode-textmate": "^10.0.2", "@types/hast": "^3.0.4" } }, "sha512-zGrWOxZ0/+0ovPY7PvBU2gIS9tmhSUUt30jAcNV0Bq0gb2S98gwfjIs1vxlmH5zM7/4YxLamT6ChlqqAJmPPjA=="], + "ai/@ai-sdk/provider-utils/eventsource-parser": ["eventsource-parser@3.0.6", "", {}, "sha512-Vo1ab+QXPzZ4tCa8SwIHJFaSzy4R6SHf7BY79rFBDf0idraZWAkYrDjDj8uWaSm3S2TK+hJ7/t1CEmZ7jXw+pg=="], + "ajv-keywords/ajv/json-schema-traverse": ["json-schema-traverse@0.4.1", "", {}, "sha512-xbbCH5dCYU5T8LcEhhuh7HJ88HXuW3qsI3Y0zOZFKfZEHcpWiHU/Jxzk629Brsab/mMiHQti9wMP+845RPe3Vg=="], "app-builder-lib/@electron/rebuild/node-abi": ["node-abi@3.87.0", "", { "dependencies": { "semver": "^7.3.5" } }, "sha512-+CGM1L1CgmtheLcBuleyYOn7NWPVu0s0EJH2C4puxgEZb9h8QpR9G2dBfZJOAUhi7VQxuBPMd0hiISWcTyiYyQ=="], diff --git a/package.json b/package.json index 93bab397..5c8fca89 100644 --- a/package.json +++ b/package.json @@ -57,7 +57,7 @@ "@git-diff-view/shiki": "^0.0.36", "@maus-inc/runtime-client": "file:packages/runtime-client", "@mcpc-tech/acp-ai-provider": "^0.2.4", - "@modelcontextprotocol/sdk": "^1.25.3", + "@modelcontextprotocol/sdk": "1.30.1", "@monaco-editor/react": "^4.7.0", "@opencode-ai/sdk": "1.18.30", "@openrouter/ai-sdk-provider": "^2.9.0", From ead68c1ea2df72017b9c64b08e5a1b551df23a52 Mon Sep 17 00:00:00 2001 From: Owie6789 <151057755+Owie6789@users.noreply.github.com> Date: Thu, 24 Sep 2026 13:55:30 +0000 Subject: [PATCH 30/60] Keep a malformed provider line from taking the process down `toClaudeStreamMessage` proves only that a record carries a string `type`; everything after that was trusted. The two handlers this PR added trusted the most: `apiRetryMessage` called `replaceAll` on `msg.error` and `handlePromptSuggestion` called `trim` on `msg.suggestion`, so a Qwen stdout line or a Claude-dialect record with the right type and the wrong fields threw inside the child's `data` callback. In the Claude router that throw is caught and ends the turn; in `qwen-print`'s `feedLine` nothing caught it, and the throw escaped into the main process. Both handlers now read their fields only when they are the shapes the SDK documents: a retry without them still reports one readable line ("request failed", no attempt phrase, a one-second wait), and a suggestion without a string body, or without a session id to attribute it to, is dropped rather than rendered against no turn. The record's shape, not its absence, decides. `feedLine` itself now wraps everything after `JSON.parse` - premap, result handling, transform - in a boundary that settles the turn with `Malformed provider message: ` and emits the same error chunk the router already holds and re-raises. That catch covers the lines no one has guarded yet too: the assistant handler's `for (const block of content)` iterates a truthy non-iterable the same way it did at this PR's base, and a line hitting it now ends one turn instead of the process. `settle` is idempotent, so a result line arriving after the failure cannot resurrect the turn. Tests: four new transform fixtures built from bare `asProviderLine` records - non-string suggestion, suggestion with no session, a retry with none of the documented fields, and a retry whose attempt numbers are strings - plus a `malformed` mock mode that plays the non-iterable-content assistant line through the real spawn path and asserts the turn settles with the error chunk and no finish. Evidence: biome 980 files 0 findings; typecheck ratchet 0 errors against a 0 baseline; lint 925 files clean; vitest 104 files, 1926 passed, 1 skipped (five new); test:node and test:contracts pass; ratchet:audit passes at the 3-critical baseline; skills:verify 50 of 50. Co-authored-by: arena-agent <297053741+arena-agent@users.noreply.github.com> --- src/main/lib/claude/transform.test.ts | 61 +++++++++++++++++++ src/main/lib/claude/transform.ts | 25 ++++++-- src/main/lib/qwen-print/session.test.ts | 14 +++++ src/main/lib/qwen-print/session.ts | 30 ++++++--- .../test/fixtures/qwen-print-mock.mjs | 11 ++++ 5 files changed, 128 insertions(+), 13 deletions(-) diff --git a/src/main/lib/claude/transform.test.ts b/src/main/lib/claude/transform.test.ts index 481e126b..633a42fc 100644 --- a/src/main/lib/claude/transform.test.ts +++ b/src/main/lib/claude/transform.test.ts @@ -32,6 +32,16 @@ function translateChunks(...messages: ClaudeStreamMessage[]): UIMessageChunk[] { return out } +/** + * A record exactly as `toClaudeStreamMessage` lets it through: any value + * carrying a string `type`, whatever else it holds. The malformed-fixture + * tests below are this and nothing more, which is the point — the handlers + * must survive what the boundary accepts. + */ +function asProviderLine(line: Record): ClaudeStreamMessage { + return line as unknown as ClaudeStreamMessage +} + afterEach(() => { vi.restoreAllMocks() }) @@ -372,4 +382,55 @@ describe("claude transform", () => { translate({ type: "prompt_suggestion", suggestion: " ", uuid: U4, session_id: "sess-9" }), ).toEqual(["start", "start-step"]) }) + + it("drops a prompt suggestion whose text is not a string", () => { + expect( + translate( + asProviderLine({ + type: "prompt_suggestion", + suggestion: 42, + uuid: U4, + session_id: "sess-9", + }), + ), + ).toEqual(["start", "start-step"]) + }) + + it("drops a prompt suggestion that names no session to belong to", () => { + expect( + translate( + asProviderLine({ type: "prompt_suggestion", suggestion: "Next: ship it", uuid: U4 }), + ), + ).toEqual(["start", "start-step"]) + }) + + it("reports a retry record that carries none of the documented fields", () => { + const chunks = translateChunks( + asProviderLine({ type: "system", subtype: "api_retry", uuid: U1, session_id: "sess-1" }), + ) + const retry = chunks.find((chunk) => chunk.type === "retry-notification") + expect(retry?.type === "retry-notification" && retry.message).toBe( + "Claude API retry: request failed, waiting 1s", + ) + }) + + it("keeps the attempt numbers out of a retry that has them wrong-typed", () => { + const chunks = translateChunks( + asProviderLine({ + type: "system", + subtype: "api_retry", + error: "overloaded_error", + error_status: 529, + attempt: "2", + max_retries: "5", + retry_delay_ms: 400, + uuid: U2, + session_id: "sess-1", + }), + ) + const retry = chunks.find((chunk) => chunk.type === "retry-notification") + expect(retry?.type === "retry-notification" && retry.message).toBe( + "Claude API retry: overloaded error (HTTP 529), waiting 1s", + ) + }) }) diff --git a/src/main/lib/claude/transform.ts b/src/main/lib/claude/transform.ts index 5a184a83..c40fca7d 100644 --- a/src/main/lib/claude/transform.ts +++ b/src/main/lib/claude/transform.ts @@ -173,10 +173,22 @@ function toolResultErrorText(content: ClaudeToolResultBlock["content"]): string function apiRetryMessage( msg: Extract, ): string { - const reason = msg.error.replaceAll("_", " ") - const status = msg.error_status == null ? "" : ` (HTTP ${msg.error_status})` - const waitSeconds = Math.max(1, Math.round(msg.retry_delay_ms / 1000)) - return `Claude API retry: ${reason}${status}, attempt ${msg.attempt} of ${msg.max_retries}, waiting ${waitSeconds}s` + // The line reaches this through `toClaudeStreamMessage`, which proves only + // that the record carries a string `type`, so every field read here is + // checked against the shape the SDK documents. A record written in another + // dialect still yields one readable line instead of throwing inside the + // child's stdout callback. + const reason = typeof msg.error === "string" ? msg.error.replaceAll("_", " ") : "request failed" + const status = typeof msg.error_status === "number" ? ` (HTTP ${msg.error_status})` : "" + const attempt = + typeof msg.attempt === "number" + ? `, attempt ${msg.attempt}${ + typeof msg.max_retries === "number" ? ` of ${msg.max_retries}` : "" + }` + : "" + const waitSeconds = + typeof msg.retry_delay_ms === "number" ? Math.max(1, Math.round(msg.retry_delay_ms / 1000)) : 1 + return `Claude API retry: ${reason}${status}${attempt}, waiting ${waitSeconds}s` } /** @@ -189,6 +201,11 @@ function apiRetryMessage( function* handlePromptSuggestion( msg: Extract, ): Generator { + // Same untrusted boundary as the retry line: unless both fields are the + // strings the SDK documents, the record is dropped rather than thrown on — + // and a suggestion with no session id has no turn to be attributed to, so + // there is nothing to render even if the text were well-formed. + if (typeof msg.suggestion !== "string" || typeof msg.session_id !== "string") return const suggestion = msg.suggestion.trim().slice(0, 2000) if (!suggestion) return yield { type: "prompt-suggestion", suggestion, sessionId: msg.session_id } diff --git a/src/main/lib/qwen-print/session.test.ts b/src/main/lib/qwen-print/session.test.ts index b8e3166c..075c28cd 100644 --- a/src/main/lib/qwen-print/session.test.ts +++ b/src/main/lib/qwen-print/session.test.ts @@ -153,6 +153,20 @@ it("turns error envelopes into held error chunks (no finish)", async () => { assert.ok(!chunks.map((c) => c.type).includes("finish")) }) +it("settles the turn when a provider line the parser rejects would otherwise escape", async () => { + const { chunks, done } = runTurn("malformed") + const result = await done + assert.strictEqual(result.status, "error") + assert.ok(result.errorMessage?.includes("Malformed provider message")) + const errorChunk = requireChunk( + chunks.find((c) => c.type === "error"), + "error", + ) + assert.ok(errorChunk.errorText?.includes("Malformed provider message")) + // The turn ends on the spot: no finish follows a line that broke the stream. + assert.ok(!chunks.map((c) => c.type).includes("finish")) +}) + it("reports stderr diagnostics when the CLI exits nonzero", async () => { const { done } = runTurn("error-exit") const result = await done diff --git a/src/main/lib/qwen-print/session.ts b/src/main/lib/qwen-print/session.ts index 55f0a82c..6bffa4c9 100644 --- a/src/main/lib/qwen-print/session.ts +++ b/src/main/lib/qwen-print/session.ts @@ -333,15 +333,27 @@ export function runQwenPrintTurn(opts: RunQwenPrintTurnOptions): QwenPrintTurn { sessionId = parsed.session_id opts.onSessionId?.(parsed.session_id) } - premapQwenLine(parsed) - if (isRecord(parsed) && parsed.type === "result") { - handleResultLine(parsed) - return - } - const message = toClaudeStreamMessage(parsed) - if (!message) return - for (const chunk of transform(message)) { - emit(chunk) + try { + premapQwenLine(parsed) + if (isRecord(parsed) && parsed.type === "result") { + handleResultLine(parsed) + return + } + const message = toClaudeStreamMessage(parsed) + if (!message) return + for (const chunk of transform(message)) { + emit(chunk) + } + } catch (error) { + // The child's stdout is an untrusted boundary. A line the transform + // cannot read must settle the turn with that line's failure rather + // than throw out of the `data` callback and take the main process + // with it; `settle` is idempotent, so a later result line cannot + // resurrect a turn that already ended. + const detail = error instanceof Error ? error.message : String(error) + const errorMessage = `Malformed provider message: ${detail}` + emit({ type: "error", errorText: errorMessage }) + settle({ status: "error", errorMessage, sessionId }) } } diff --git a/src/main/lib/qwen-print/test/fixtures/qwen-print-mock.mjs b/src/main/lib/qwen-print/test/fixtures/qwen-print-mock.mjs index 15116eac..8d75e4c3 100644 --- a/src/main/lib/qwen-print/test/fixtures/qwen-print-mock.mjs +++ b/src/main/lib/qwen-print/test/fixtures/qwen-print-mock.mjs @@ -64,6 +64,17 @@ if (mode === "never") { permission_denials: [], error: { message: "Missing API key for OpenAI-compatible auth." }, }) +} else if (mode === "malformed") { + init() + // `content` is truthy but not iterable: the assistant handler throws on + // `for (const block of content)`, so the boundary catch has to settle the + // turn instead of letting the throw escape the stdout callback. + line({ + type: "assistant", + uuid: "uuid-malformed", + session_id: SID, + message: { id: "m1", type: "message", role: "assistant", model: "test-model", content: 42 }, + }) } else if (mode === "denials") { init({ permission_mode: "default" }) line({ From 6804c04bad72e93e6512267e1c4317635b70cd83 Mon Sep 17 00:00:00 2001 From: Owie6789 <151057755+Owie6789@users.noreply.github.com> Date: Thu, 24 Sep 2026 14:15:49 +0000 Subject: [PATCH 31/60] Say what a subagent actually did, and keep its descendants with it MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two rows lied about the current `Agent` wire, both newly reachable because this PR routes `tool-Agent` through the task renderer and the grouping predicate. The title asked only whether the tool was still streaming, so a result the SDK types as `async_launched` or `remote_launched` — the run was handed off to the background or to a remote session and is still going there — rendered as `Completed Subagent`. A reader would take a subagent that has not finished for one that has. The title now reads the output's `status`: `Launched Subagent` for the two hand-offs, `Completed Subagent` for `completed`, and the registry's flat-row phrasing says `${type} launched` for the same two. `isLaunchedAgentOutput` in `agent-tool-utils` is the one question both surfaces ask. The grouping looked a nested call's parent up by the first segment of its composite id among the TOP-LEVEL task ids. The wire composes ids as `parentOriginal:childOriginal` — the SDK's `parent_tool_use_id` names the immediate parent by its original tool id — so a grandchild `B:C` looked for a task named `B` while the actual parent row is `A:B`: not found, so the descendants stood up a fake `Incomplete task` at the top level, split from the tree they belong to. The lookup now goes through a map of every task's original id to its full composite, keys the children by the parent's FULL id, and skips a part that names itself as its own parent — the cycle guard. Rendering follows the grouping: a subagent inside a subagent now renders as its own expandable `AgentTaskTool` with its own children, reached through a `nestedChildren` lookup from the message's nesting map, instead of collapsing into one flat registry line. The recursion is capped at three levels — real composites stop at two — and a level past the cap falls back to the flat row. The memo comparator watches the lookup's identity, because a grandchild is not in this task's own `nestedTools` and the deep compare cannot see it otherwise. Tests: three statuses pinned in one snapshot (`Launched` ×2, `Completed` ×1 — additions only, the 30 recorded before unchanged); a three-level fixture asserting no `Incomplete task` appears; the same fixture clicked open through both headers to prove the Read's basename renders under the inner agent; and a registry case for both launch statuses plus a real completion. Evidence: biome 980 files 0 findings; typecheck ratchet 0 errors against a 0 baseline; lint 925 files clean; vitest 104 files, 1930 passed, 1 skipped (four new); test:node 59 pass; test:contracts 382 passed; ratchet:audit passes; skills:verify 50 of 50. Co-authored-by: arena-agent <297053741+arena-agent@users.noreply.github.com> --- .../assistant-message-item.test.tsx.snap | 4 + .../main/assistant-message-item.test.tsx | 103 +++++++++++++++++- .../agents/main/assistant-message-item.tsx | 57 +++++++--- .../features/agents/ui/agent-task-tool.tsx | 51 ++++++++- .../agents/ui/agent-tool-registry.test.ts | 12 ++ .../agents/ui/agent-tool-registry.tsx | 10 +- .../features/agents/ui/agent-tool-utils.ts | 42 ++++++- 7 files changed, 257 insertions(+), 22 deletions(-) diff --git a/src/renderer/features/agents/main/__snapshots__/assistant-message-item.test.tsx.snap b/src/renderer/features/agents/main/__snapshots__/assistant-message-item.test.tsx.snap index b5708a1a..7527c50c 100644 --- a/src/renderer/features/agents/main/__snapshots__/assistant-message-item.test.tsx.snap +++ b/src/renderer/features/agents/main/__snapshots__/assistant-message-item.test.tsx.snap @@ -4,6 +4,8 @@ exports[`AssistantMessageItem, one message per branch of the part dispatcher > c exports[`AssistantMessageItem, one message per branch of the part dispatcher > groups three consecutive exploring tools 1`] = `"
Response

Read all three.

"`; +exports[`AssistantMessageItem, one message per branch of the part dispatcher > keeps a nested subagent's descendants under it instead of orphaning them 1`] = `"
"`; + exports[`AssistantMessageItem, one message per branch of the part dispatcher > keeps every part visible while the last message streams 1`] = `"

Lint is clean so far.

"`; exports[`AssistantMessageItem, one message per branch of the part dispatcher > renders a Bash call with its command, output and exit code 1`] = ` @@ -62,3 +64,5 @@ exports[`AssistantMessageItem, one message per branch of the part dispatcher > r exports[`AssistantMessageItem, one message per branch of the part dispatcher > renders the renamed background-output tool 1`] = `"
Got outputTask: task_1
"`; exports[`AssistantMessageItem, one message per branch of the part dispatcher > renders usage and git badges from message metadata 1`] = `"

Done.

"`; + +exports[`AssistantMessageItem, one message per branch of the part dispatcher > tells a launch apart from a finished subagent 1`] = `"
"`; diff --git a/src/renderer/features/agents/main/assistant-message-item.test.tsx b/src/renderer/features/agents/main/assistant-message-item.test.tsx index 2d48b8e4..86eacfa4 100644 --- a/src/renderer/features/agents/main/assistant-message-item.test.tsx +++ b/src/renderer/features/agents/main/assistant-message-item.test.tsx @@ -19,7 +19,7 @@ * next change to any of these branches, where the acceptable result is a diff * somebody meant. */ -import { render } from "@testing-library/react" +import { fireEvent, render } from "@testing-library/react" import { beforeEach, describe, expect, it, vi } from "vitest" import { TooltipProvider } from "../../../components/ui/tooltip" import type { Message, MessagePart } from "../stores/message-store" @@ -94,6 +94,24 @@ function renderParts(parts: MessagePart[], streaming = false): string { return renderMessage(message, streaming) } +/** The same message as `renderParts`, but with the DOM still attached to click on. */ +function renderPartsDom(parts: MessagePart[]): HTMLElement { + const { container } = render( + + + , + ) + return container +} + /** A completed tool call, which is the state every fixture below starts from. */ function tool(type: string, id: string, input: unknown, output?: unknown): MessagePart { return { type, toolCallId: id, state: "output-available", input, output } @@ -422,6 +440,89 @@ describe("AssistantMessageItem, one message per branch of the part dispatcher", ).toMatchSnapshot() }) + it("tells a launch apart from a finished subagent", () => { + const html = renderParts([ + tool( + "tool-Agent", + "toolu_async_1", + { subagent_type: "general-purpose", description: "Watch the queue" }, + { + status: "async_launched", + agentId: "agent_1", + description: "Watch the queue", + prompt: "watch", + outputFile: "/tmp/agent.log", + }, + ), + tool( + "tool-Agent", + "toolu_remote_1", + { subagent_type: "general-purpose", description: "Run elsewhere" }, + { + status: "remote_launched", + taskId: "task_9", + description: "Run elsewhere", + prompt: "run", + }, + ), + tool( + "tool-Agent", + "toolu_done_1", + { subagent_type: "general-purpose", description: "Finished run" }, + { status: "completed", prompt: "done" }, + ), + ]) + // Two launches say launched; only the `completed` status says completed. + expect(html.match(/Launched Subagent/g)).toHaveLength(2) + expect(html).toContain("Completed Subagent") + expect(html).toMatchSnapshot() + }) + + it("keeps a nested subagent's descendants under it instead of orphaning them", () => { + const html = renderParts([ + tool("tool-Agent", "toolu_agent_1", { + subagent_type: "general-purpose", + description: "Outer agent", + }), + tool("tool-Agent", "toolu_agent_1:toolu_agent_2", { + subagent_type: "general-purpose", + description: "Inner agent", + }), + tool("tool-Read", "toolu_agent_2:toolu_read_9", { file_path: "/repo/nested.txt" }), + ]) + // The parent is resolved through the full id map: no top-level task is + // named `toolu_agent_2`, and before that lookup existed the Read stood up + // a fake "Incomplete task" instead of living under the inner agent. + expect(html).not.toContain("Incomplete task") + expect(html).toMatchSnapshot() + }) + + it("renders a three-level subagent chain once each level is expanded", () => { + const container = renderPartsDom([ + tool("tool-Agent", "toolu_agent_1", { + subagent_type: "general-purpose", + description: "Outer agent", + }), + tool("tool-Agent", "toolu_agent_1:toolu_agent_2", { + subagent_type: "general-purpose", + description: "Inner agent", + }), + tool("tool-Read", "toolu_agent_2:toolu_read_9", { file_path: "/repo/nested.txt" }), + ]) + const clickHeader = (needle: string) => { + const header = Array.from(container.querySelectorAll('[role="button"]')).find( + (el) => el.textContent?.includes(needle), + ) + expect(header, `a header containing ${needle}`).toBeTruthy() + fireEvent.click(header as HTMLElement) + } + clickHeader("Outer agent") + clickHeader("Inner agent") + // The Read row's subtitle is the file's basename — proof the descendant + // renders under the inner agent rather than at the top level or nowhere. + expect(container.textContent).toContain("nested.txt") + }) + it("renders an MCP tool call", () => { expect( renderParts([ diff --git a/src/renderer/features/agents/main/assistant-message-item.tsx b/src/renderer/features/agents/main/assistant-message-item.tsx index 783c2b15..edc78a29 100644 --- a/src/renderer/features/agents/main/assistant-message-item.tsx +++ b/src/renderer/features/agents/main/assistant-message-item.tsx @@ -631,6 +631,8 @@ type PartRenderContext = { projectPath: string | undefined onOpenFile: ReturnType nestedToolsMap: Map + /** Children of any subagent, by the subagent's full composite id. */ + nestedChildren: (toolCallId: string) => NormalizedPart[] nestedToolIds: Set /** Nested calls whose parent task part never arrived. */ orphans: { @@ -681,6 +683,7 @@ function renderOrphanTaskGroup( input: { subagent_type: "unknown-agent", description: "Incomplete task" }, }} nestedTools={group.parts} + nestedChildren={ctx.nestedChildren} chatStatus={ctx.status} /> ) @@ -712,7 +715,15 @@ function renderTextPart( function renderSubagentTask(part: NormalizedPart, idx: number, ctx: PartRenderContext): ReactNode { const nestedTools = ctx.nestedToolsMap.get(part.toolCallId ?? "") || [] - return + return ( + + ) } function renderBashTool(part: NormalizedPart, idx: number, ctx: PartRenderContext): ReactNode { @@ -1006,32 +1017,47 @@ export const AssistantMessageItem = memo(function AssistantMessageItem({ } = useMemo(() => { const nestedToolsMap = new Map() const nestedToolIds = new Set() - const taskPartIds = new Set( - messageParts - .filter( - (p): p is NormalizedPart & { toolCallId: string } => - isSubagentToolType(p.type) && !!p.toolCallId, - ) - .map((p) => p.toolCallId), + const taskParts = messageParts.filter( + (p): p is NormalizedPart & { toolCallId: string } => + isSubagentToolType(p.type) && !!p.toolCallId, ) + // A composite id is `parentOriginal:childOriginal` — the SDK names a + // child's parent by that parent's ORIGINAL tool id, never by the + // parent's own composite — so a nested task's original id is its last + // segment, and it is what the task's own children carry before the + // colon. Looking the first segment up in the top-level ids alone (what + // this did before) finds `A` under `A:B`, but orphans everything under + // `A:B`, because no top-level task is ever named just `B`. + const taskFullIdByOriginalId = new Map() + for (const task of taskParts) { + const segments = task.toolCallId.split(":") + taskFullIdByOriginalId.set(segments[segments.length - 1] ?? task.toolCallId, task.toolCallId) + } const orphanTaskGroups = new Map() const orphanToolCallIds = new Set() const orphanFirstToolCallIds = new Set() for (const part of messageParts) { if (part.toolCallId?.includes(":")) { - const parentId = part.toolCallId.split(":")[0] - if (taskPartIds.has(parentId)) { - if (!nestedToolsMap.has(parentId)) { - nestedToolsMap.set(parentId, []) + const parentOriginalId = part.toolCallId.split(":")[0] + const parentFullId = + parentOriginalId === undefined ? undefined : taskFullIdByOriginalId.get(parentOriginalId) + // The self check is the cycle guard: a part that names itself as its + // own parent would otherwise sit in its own children forever. + if (parentFullId !== undefined && parentFullId !== part.toolCallId) { + // Keyed by the parent's FULL id: that is the id `renderSubagentTask` + // looks children up by, whether the parent sits at the top level + // (`A`) or inside another task (`A:B`). + if (!nestedToolsMap.has(parentFullId)) { + nestedToolsMap.set(parentFullId, []) } - nestedToolsMap.get(parentId)?.push(part) + nestedToolsMap.get(parentFullId)?.push(part) nestedToolIds.add(part.toolCallId) } else { - let group = orphanTaskGroups.get(parentId) + let group = orphanTaskGroups.get(parentOriginalId ?? "") if (!group) { group = { parts: [], firstToolCallId: part.toolCallId } - orphanTaskGroups.set(parentId, group) + orphanTaskGroups.set(parentOriginalId ?? "", group) orphanFirstToolCallIds.add(part.toolCallId) } group.parts.push(part) @@ -1183,6 +1209,7 @@ export const AssistantMessageItem = memo(function AssistantMessageItem({ projectPath, onOpenFile, nestedToolsMap, + nestedChildren: (toolCallId: string) => nestedToolsMap.get(toolCallId) ?? [], nestedToolIds, orphans: { toolCallIds: orphanToolCallIds, diff --git a/src/renderer/features/agents/ui/agent-task-tool.tsx b/src/renderer/features/agents/ui/agent-task-tool.tsx index a51173de..f3d2a7d5 100644 --- a/src/renderer/features/agents/ui/agent-task-tool.tsx +++ b/src/renderer/features/agents/ui/agent-task-tool.tsx @@ -7,22 +7,42 @@ import { TextShimmer } from "../../../components/ui/text-shimmer" import { keyItems } from "../../../lib/react-keys" import { cn } from "../../../lib/utils" import { selectedProjectAtom } from "../atoms" +import { isSubagentToolType } from "../lib/subagent-tool-types" import { useFileOpen } from "../mentions" import { AgentToolCall } from "./agent-tool-call" import { AgentToolInterrupted } from "./agent-tool-interrupted" import { AgentToolRegistry, getToolStatus, type ToolDisplayPart } from "./agent-tool-registry" import type { ToolPartLike } from "./agent-tool-state" -import { areTaskToolPropsEqual } from "./agent-tool-utils" +import { + areTaskToolPropsEqual, + isLaunchedAgentOutput, + type NestedToolsLookup, +} from "./agent-tool-utils" interface AgentTaskToolProps { part: ToolPartLike nestedTools: ToolPartLike[] + /** + * Children of any nested subagent, so a subagent inside a subagent renders + * as its own expandable task rather than a flat line. Absent means flat + * rows only (the depth cap, and callers that have no map to offer). + */ + nestedChildren?: NestedToolsLookup + /** How many subagent levels deep this call already is; the cap counts them. */ + depth?: number chatStatus?: string } // Constants for rendering const MAX_VISIBLE_TOOLS = 5 const TOOL_HEIGHT_PX = 24 +/** + * Subagent levels this component will nest into itself. The wire composes ids + * as `parentOriginal:childOriginal`, so real transcripts stop at depth two; + * the cap exists so an id scheme nobody has seen yet cannot recurse without + * bound — a level at or past it falls back to the flat registry row. + */ +const MAX_SUBAGENT_RENDER_DEPTH = 3 // Format elapsed time in a human-readable format function formatElapsedTime(ms: number): string { @@ -38,6 +58,8 @@ function formatElapsedTime(ms: number): string { export const AgentTaskTool = memo(function AgentTaskTool({ part, nestedTools, + nestedChildren, + depth = 0, chatStatus, }: AgentTaskToolProps) { const selectedProject = useAtomValue(selectedProjectAtom) @@ -124,9 +146,11 @@ export const AgentTaskTool = memo(function AgentTaskTool({ const subtitle = getSubtitle() - // Get title text based on status + // Get title text based on status: a launch is not a finish, and the row + // says which one it was. const getTitle = () => { - return isPending ? "Running Subagent" : "Completed Subagent" + if (isPending) return "Running Subagent" + return isLaunchedAgentOutput(part.output) ? "Launched Subagent" : "Completed Subagent" } // Show interrupted state if task was interrupted without completing @@ -217,6 +241,27 @@ export const AgentTaskTool = memo(function AgentTaskTool({ (nestedPart as { toolCallId?: unknown }).toolCallId ?? nestedPart.type ?? "part", ), ).map(({ key, item: nestedPart }) => { + // A subagent inside a subagent is its own task row — same + // header, same expansion, its own children — so its descendants + // render under it instead of being flattened into this level. + if ( + nestedPart.type && + isSubagentToolType(nestedPart.type) && + nestedChildren && + depth < MAX_SUBAGENT_RENDER_DEPTH + ) { + const childId = String((nestedPart as { toolCallId?: unknown }).toolCallId ?? "") + return ( + + ) + } const nestedMeta = nestedPart.type ? AgentToolRegistry[nestedPart.type] : undefined if (!nestedMeta) { return ( diff --git a/src/renderer/features/agents/ui/agent-tool-registry.test.ts b/src/renderer/features/agents/ui/agent-tool-registry.test.ts index 41ce7c4c..f6c8ad22 100644 --- a/src/renderer/features/agents/ui/agent-tool-registry.test.ts +++ b/src/renderer/features/agents/ui/agent-tool-registry.test.ts @@ -86,6 +86,18 @@ describe("agent tool registry: renamed sub-agent and background task tools", () expect(meta.title({ state: "output-available", input: {} })).toBe("Agent completed") }) + it("calls a launched sub-agent a launch, not a completion", () => { + const meta = AgentToolRegistry["tool-Agent"] + const base: ToolDisplayPart = { + state: "output-available", + input: { subagent_type: "Explore", description: "Hand the run off" }, + } + expect(meta.title({ ...base, output: { status: "async_launched" } })).toBe("Explore launched") + expect(meta.title({ ...base, output: { status: "remote_launched" } })).toBe("Explore launched") + // A real completion still says so. + expect(meta.title({ ...base, output: { status: "completed" } })).toBe("Explore completed") + }) + it("truncates a long sub-agent description and hides it while streaming", () => { const meta = AgentToolRegistry["tool-Agent"] expect(meta.subtitle?.(streamingSubagent)).toBe("") diff --git a/src/renderer/features/agents/ui/agent-tool-registry.tsx b/src/renderer/features/agents/ui/agent-tool-registry.tsx index f224c859..114347ea 100644 --- a/src/renderer/features/agents/ui/agent-tool-registry.tsx +++ b/src/renderer/features/agents/ui/agent-tool-registry.tsx @@ -30,6 +30,7 @@ import { WriteFileIcon, } from "../../../components/ui/icons" import { getToolLifecycleState } from "./agent-tool-state" +import { isLaunchedAgentOutput } from "./agent-tool-utils" export { getToolStatus } from "./agent-tool-state" @@ -65,6 +66,8 @@ export type ToolDisplayPart = { numLines?: number task?: { subject?: string } tasks?: unknown[] + /** `AgentOutput.status`: `completed`, or one of the two launch hand-offs. */ + status?: string } } @@ -163,7 +166,12 @@ const subagentTool: ToolMeta = { title: (part) => { if (isInputStreaming(part)) return "Preparing agent" const subagentType = part.input?.subagent_type || "Agent" - return isPendingState(part) ? `Running ${subagentType}` : `${subagentType} completed` + if (isPendingState(part)) return `Running ${subagentType}` + // A launched run is still running somewhere else; only a `completed` + // status has actually finished. + return isLaunchedAgentOutput(part.output) + ? `${subagentType} launched` + : `${subagentType} completed` }, subtitle: (part) => { // Don't show subtitle while input is still streaming diff --git a/src/renderer/features/agents/ui/agent-tool-utils.ts b/src/renderer/features/agents/ui/agent-tool-utils.ts index 2d046071..fb70145d 100644 --- a/src/renderer/features/agents/ui/agent-tool-utils.ts +++ b/src/renderer/features/agents/ui/agent-tool-utils.ts @@ -129,10 +129,48 @@ export function areToolPropsEqual( /** * Compare function for AgentTaskTool which has additional nestedTools prop. */ +/** + * A subagent's own children by its id. The message-level map stands behind it + * so the memo can compare identity: a grandchild is not in this task's own + * `nestedTools`, and only the lookup's identity says it changed. + */ +export type NestedToolsLookup = (toolCallId: string) => ToolPartLike[] + +/** + * A result that launched work instead of finishing it. The pinned SDK types + * `AgentOutput.status` as `completed` | `async_launched` | `remote_launched`: + * the latter two mean the run was handed off — to the background, or to a + * remote session — and is still going there, so a row that calls them a + * completion reads as subagent work that ended when it has not. + */ +export function isLaunchedAgentOutput(output: unknown): boolean { + const status = (output as { status?: unknown } | null | undefined)?.status + return status === "async_launched" || status === "remote_launched" +} + export function areTaskToolPropsEqual( - prevProps: { part: ToolPartLike; nestedTools: ToolPartLike[]; chatStatus?: string }, - nextProps: { part: ToolPartLike; nestedTools: ToolPartLike[]; chatStatus?: string }, + prevProps: { + part: ToolPartLike + nestedTools: ToolPartLike[] + nestedChildren?: NestedToolsLookup + depth?: number + chatStatus?: string + }, + nextProps: { + part: ToolPartLike + nestedTools: ToolPartLike[] + nestedChildren?: NestedToolsLookup + depth?: number + chatStatus?: string + }, ): boolean { + // The lookup's identity changes when the message's nesting map is rebuilt, + // which is the only signal that reaches here about a descendant deeper than + // this task's own `nestedTools`. Checked first so the completed short + // circuit below cannot hide it. + if (prevProps.nestedChildren !== nextProps.nestedChildren) return false + if (prevProps.depth !== nextProps.depth) return false + // Compare main part first if (!arePartsEqual(prevProps.part, nextProps.part)) return false From b4c9ebb64a074dcd3ce59fff74ebdeb691fd1a73 Mon Sep 17 00:00:00 2001 From: Owie6789 <151057755+Owie6789@users.noreply.github.com> Date: Thu, 24 Sep 2026 14:23:38 +0000 Subject: [PATCH 32/60] Only an actionable subtitle wears button semantics MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit This PR routes `tool-TaskOutput` through the generic registry row, where every string subtitle got `role="button"`, a tab stop and Enter/Space handling, while the handler was attached only when the row had one. A completed TaskOutput with `{ task_id }` is the common case and has no action: the recorded snapshot itself showed `Task: task_1` doing nothing — focusable, announced as interactive, inert. The base suppressed `tool-TaskOutput` entirely, so the base could not reach this; the row is new here, the span's rule was not. `subtitleSpan` now takes the handler as the question it answers: with one, the span is the button it says it is and Enter and Space press it; without one, it is text — no role, no tab stop, nothing in the tab order. `subtitleWithTooltip` keeps the tooltip wrapper for both, so a row that only describes itself still shows its detail on hover. Two snapshots changed, by exactly these deletions and nothing else: `role="button"` and `tabindex="0"` leave the registry row and the TaskOutput row. `TaskStop`'s second half: the shared subtitle reader took `task_id` and `taskId` but not `shell_id`, which `TaskStopInput` still accepts as the deprecated spelling of `task_id` — a valid `{ shell_id }` call stopped something with no name for it. The reader takes it now, under the same `Task:` label, because it is the same identifier. Tests: `agent-tool-call.test.tsx` pins both halves of the affordance against the component directly (no action: no role, no tab stop; action: role button, tabindex 0); the dispatcher test pins TaskOutput's subtitle as plain text in a real transcript; the registry test pins the `shell_id` subtitle. The positive Read-row control lives in the component test rather than the transcript test because the transcript renders without the file-open provider, where no row has an action — the old snapshot's Read button was inert there for the same reason. Evidence: biome 981 files 0 findings; typecheck ratchet 0 errors against a 0 baseline; lint 926 files clean; vitest 105 files, 1933 passed, 1 skipped (three new); test:node 59 pass; test:contracts 382 passed; ratchet:audit passes; skills:verify 50 of 50. Co-authored-by: arena-agent <297053741+arena-agent@users.noreply.github.com> --- .../assistant-message-item.test.tsx.snap | 4 +- .../main/assistant-message-item.test.tsx | 17 +++ .../agents/ui/agent-tool-call.test.tsx | 48 ++++++++ .../features/agents/ui/agent-tool-call.tsx | 106 ++++++++++-------- .../agents/ui/agent-tool-registry.test.ts | 8 ++ .../agents/ui/agent-tool-registry.tsx | 7 +- 6 files changed, 139 insertions(+), 51 deletions(-) create mode 100644 src/renderer/features/agents/ui/agent-tool-call.test.tsx diff --git a/src/renderer/features/agents/main/__snapshots__/assistant-message-item.test.tsx.snap b/src/renderer/features/agents/main/__snapshots__/assistant-message-item.test.tsx.snap index 7527c50c..0ea838d2 100644 --- a/src/renderer/features/agents/main/__snapshots__/assistant-message-item.test.tsx.snap +++ b/src/renderer/features/agents/main/__snapshots__/assistant-message-item.test.tsx.snap @@ -29,7 +29,7 @@ exports[`AssistantMessageItem, one message per branch of the part dispatcher > r exports[`AssistantMessageItem, one message per branch of the part dispatcher > renders a reasoning part 1`] = `"
"`; -exports[`AssistantMessageItem, one message per branch of the part dispatcher > renders a registry tool as a single row 1`] = `"
Readeffort.ts
"`; +exports[`AssistantMessageItem, one message per branch of the part dispatcher > renders a registry tool as a single row 1`] = `"
Readeffort.ts
"`; exports[`AssistantMessageItem, one message per branch of the part dispatcher > renders a second plan operation as a mini indicator, not a card 1`] = `"
Created plan
"`; @@ -61,7 +61,7 @@ exports[`AssistantMessageItem, one message per branch of the part dispatcher > r exports[`AssistantMessageItem, one message per branch of the part dispatcher > renders the four things a plan operation's indicator can say 1`] = `"
Updating plan...
Updated plan
"`; -exports[`AssistantMessageItem, one message per branch of the part dispatcher > renders the renamed background-output tool 1`] = `"
Got outputTask: task_1
"`; +exports[`AssistantMessageItem, one message per branch of the part dispatcher > renders the renamed background-output tool 1`] = `"
Got outputTask: task_1
"`; exports[`AssistantMessageItem, one message per branch of the part dispatcher > renders usage and git badges from message metadata 1`] = `"

Done.

"`; diff --git a/src/renderer/features/agents/main/assistant-message-item.test.tsx b/src/renderer/features/agents/main/assistant-message-item.test.tsx index 86eacfa4..b1bdf875 100644 --- a/src/renderer/features/agents/main/assistant-message-item.test.tsx +++ b/src/renderer/features/agents/main/assistant-message-item.test.tsx @@ -497,6 +497,23 @@ describe("AssistantMessageItem, one message per branch of the part dispatcher", expect(html).toMatchSnapshot() }) + it("keeps a non-actionable subtitle out of the tab order", () => { + const container = renderPartsDom([ + tool("tool-TaskOutput", "toolu_taskoutput_1", { task_id: "task_1" }, { output: "done" }), + ]) + // TaskOutput's subtitle identifies the output and does nothing else, so + // it must be plain text: no role, no tab stop. (The positive case — a + // subtitle with an action — is pinned in agent-tool-call.test.tsx; here + // the transcript renders without the file-open provider, so no row would + // have an action to assert.) + const inert = Array.from(container.querySelectorAll("span")).find( + (s) => s.textContent === "Task: task_1", + ) + expect(inert).toBeTruthy() + expect(inert?.getAttribute("role")).toBeNull() + expect(inert?.getAttribute("tabindex")).toBeNull() + }) + it("renders a three-level subagent chain once each level is expanded", () => { const container = renderPartsDom([ tool("tool-Agent", "toolu_agent_1", { diff --git a/src/renderer/features/agents/ui/agent-tool-call.test.tsx b/src/renderer/features/agents/ui/agent-tool-call.test.tsx new file mode 100644 index 00000000..c15c1f3c --- /dev/null +++ b/src/renderer/features/agents/ui/agent-tool-call.test.tsx @@ -0,0 +1,48 @@ +// @vitest-environment jsdom +/** + * When a registry row's subtitle is a button and when it is only text. + * + * `AgentToolCall` used to give every subtitle `role="button"`, a tab stop and + * Enter/Space handling, and attached the handler only when the row had an + * action — so every `TaskOutput` and `TaskStop` row (and any row rendered + * without its file-open provider) was a focusable, screen-reader-announced + * control that did nothing. The two halves of that are pinned here: no + * action, no button; action, button that works. + */ +import { render } from "@testing-library/react" +import { describe, expect, it } from "vitest" +import { EyeIcon } from "../../../components/ui/icons" +import { AgentToolCall } from "./agent-tool-call" + +describe("AgentToolCall subtitle affordance", () => { + it("is plain text when the row has no action", () => { + const { getByText } = render( + , + ) + const subtitle = getByText("Task: task_1") + expect(subtitle.getAttribute("role")).toBeNull() + expect(subtitle.getAttribute("tabindex")).toBeNull() + }) + + it("is a button when the row has an action to press", () => { + const { getByText } = render( + {}} + />, + ) + const subtitle = getByText("effort.ts") + expect(subtitle.getAttribute("role")).toBe("button") + expect(subtitle.getAttribute("tabindex")).toBe("0") + }) +}) diff --git a/src/renderer/features/agents/ui/agent-tool-call.tsx b/src/renderer/features/agents/ui/agent-tool-call.tsx index 2f8ccd9b..bca65ebf 100644 --- a/src/renderer/features/agents/ui/agent-tool-call.tsx +++ b/src/renderer/features/agents/ui/agent-tool-call.tsx @@ -4,6 +4,59 @@ import { memo } from "react" import { TextShimmer } from "../../../components/ui/text-shimmer" import { Tooltip, TooltipContent, TooltipTrigger } from "../../../components/ui/tooltip" +/** + * The subtitle span, wearing button semantics only when there is an action to + * press. A `role="button"` whose Enter and Space do nothing is a control that + * lies: keyboard and screen-reader users can focus it, it announces itself as + * interactive, and nothing happens — which is every `TaskOutput` and + * `TaskStop` row, since neither offers an action beyond being read. + */ +function subtitleSpan( + content: React.ReactNode, + className: string, + onClick?: () => void, +): React.ReactElement { + if (!onClick) return {content} + return ( + /* biome-ignore lint/a11y/useSemanticElements: compact inline action; a native button would require style resets. */ + { + if (e.key === "Enter" || e.key === " ") { + e.preventDefault() + onClick() + } + }} + > + {content} + + ) +} + +/** The subtitle, wrapped in its tooltip when the meta describes one. */ +function subtitleWithTooltip( + span: React.ReactElement, + tooltipContent?: string, +): React.ReactElement { + if (!tooltipContent) return span + return ( + + {span} + + + {tooltipContent} + + + + ) +} + interface AgentToolCallProps { icon: React.ComponentType<{ className?: string }> title: string @@ -31,58 +84,15 @@ export const AgentToolCall = memo( const titleStr = String(title) const subtitleContent = subtitle ? subtitle : undefined - // Render subtitle with optional tooltip + // Render subtitle with optional tooltip; only an actionable one is a button. const clickableClass = onClick ? " cursor-pointer hover:text-muted-foreground transition-colors" : "" + const subtitleClass = `text-muted-foreground/60 font-normal truncate min-w-0${clickableClass}` - const subtitleElement = subtitleContent ? ( - tooltipContent ? ( - - - {/* biome-ignore lint/a11y/useSemanticElements: compact inline action; a native button would require style resets. */} - { - if (e.key === "Enter" || e.key === " ") { - e.preventDefault() - onClick?.() - } - }} - > - {subtitleContent} - - - - - {tooltipContent} - - - - ) : ( - /* biome-ignore lint/a11y/useSemanticElements: compact inline action; a native button would require style resets. */ - { - if (e.key === "Enter" || e.key === " ") { - e.preventDefault() - onClick?.() - } - }} - > - {subtitleContent} - - ) - ) : null + const subtitleElement = subtitleContent + ? subtitleWithTooltip(subtitleSpan(subtitleContent, subtitleClass, onClick), tooltipContent) + : null return (
diff --git a/src/renderer/features/agents/ui/agent-tool-registry.test.ts b/src/renderer/features/agents/ui/agent-tool-registry.test.ts index f6c8ad22..a54ab5c1 100644 --- a/src/renderer/features/agents/ui/agent-tool-registry.test.ts +++ b/src/renderer/features/agents/ui/agent-tool-registry.test.ts @@ -125,6 +125,14 @@ describe("agent tool registry: renamed sub-agent and background task tools", () expect(AgentToolRegistry["tool-KillShell"].title(taskOutputById)).toBe("Stopped shell") // Both read the same subtitle, so a task id renders under either name. expect(AgentToolRegistry["tool-TaskStop"].subtitle?.(taskOutputById)).toBe("Task: bg_7") + // `shell_id` is TaskStopInput's deprecated spelling of `task_id`: a call + // that carries only it still has to say which task it stopped. + expect( + AgentToolRegistry["tool-TaskStop"].subtitle?.({ + state: "output-available", + input: { shell_id: "shell_1" }, + }), + ).toBe("Task: shell_1") }) it("counts the sub-agent family and leaves the background task family out", () => { diff --git a/src/renderer/features/agents/ui/agent-tool-registry.tsx b/src/renderer/features/agents/ui/agent-tool-registry.tsx index 114347ea..8eb56810 100644 --- a/src/renderer/features/agents/ui/agent-tool-registry.tsx +++ b/src/renderer/features/agents/ui/agent-tool-registry.tsx @@ -56,6 +56,8 @@ export type ToolDisplayPart = { status?: string taskId?: string | number task_id?: string | number + /** `TaskStopInput`'s deprecated spelling of `task_id`; same value. */ + shell_id?: string | number pid?: string | number text?: string plan?: { status?: string; title?: string; steps?: { status?: string }[] } @@ -189,7 +191,10 @@ const subagentTool: ToolMeta = { function backgroundTaskSubtitle(part: ToolDisplayPart): string { const pid = part.input?.pid if (pid) return `PID: ${pid}` - const taskId = part.input?.task_id ?? part.input?.taskId + // `shell_id` is `TaskStopInput`'s deprecated spelling of `task_id` — the + // same identifier under the name older transcripts carry — so a stop that + // sends only it still says which task it stopped. + const taskId = part.input?.task_id ?? part.input?.taskId ?? part.input?.shell_id return taskId ? `Task: ${taskId}` : "" } From 34ea81bb224e3aad74e28a03af5b107d6bc1d411 Mon Sep 17 00:00:00 2001 From: Owie6789 <151057755+Owie6789@users.noreply.github.com> Date: Thu, 24 Sep 2026 14:33:43 +0000 Subject: [PATCH 33/60] Give each prompt suggestion a turn, an engine and a preference MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Three ways a suggestion could outlive its welcome, all answered wrongly. Session equality stood in for turn ownership, but a session id is reused across turns, so a late chunk from turn N landing during turn N+1 passed its check and the composer offered turn N's next step for turn N+1's prompt. The native engine never cleared the atom at all, so switching a sub-chat mid-conversation left the legacy stream's suggestion on screen with nothing that would ever replace it. And the transport stored whatever arrived: with the app's switch off, an inherited `CLAUDE_CODE_ENABLE_PROMPT_SUGGESTION` in the shell still made the CLI emit, the store still accepted, and the composer — which rendered the bare atom — still offered. The switch said off and the row appeared. A suggestion is now `{ text, turn, engine }` and `suggestion-ownership.ts` is the one place that decides what any of it means. Every send — legacy and native alike — bumps a per-sub-chat turn generation and clears the atom before streaming starts; the legacy store refuses a chunk whose captured generation is no longer current, and refuses any chunk while the preference is off; the composer renders only an entry whose engine and turn are still its own. Abort and error withdraw this turn's suggestion under the same generation guard, so a superseding send's fresh suggestion is not wiped by the stream it replaced. The preference is made authoritative the way the SDK documents winning it: the env var beats settings, so the router overwrites the inherited `CLAUDE_CODE_ENABLE_PROMPT_SUGGESTION` with the toggle's value in both directions and sends `promptSuggestions: false` as explicitly as true. Absent used to mean undecided, and undecided is where the shell got to decide. The shared rules are pure and carry the tests: seven cases in `suggestion-ownership.test.ts` pin the store's three refusals and the composer's three gates — switch off, superseded turn, engine change — rather than leaving them to be rediscovered in each transport. Evidence: biome 983 files 0 findings; typecheck ratchet 0 errors against a 0 baseline; lint 928 files clean; vitest 106 files, 1940 passed, 1 skipped (seven new); test:node 59 pass; test:contracts 382 passed; ratchet:audit passes; skills:verify 50 of 50. Co-authored-by: arena-agent <297053741+arena-agent@users.noreply.github.com> --- src/main/lib/trpc/routers/claude.ts | 12 ++++- src/renderer/features/agents/atoms/index.ts | 12 ++++- .../features/agents/lib/ipc-chat-transport.ts | 51 +++++++++++++++++-- .../agents/lib/native-chat-transport.ts | 16 +++++- .../agents/lib/suggestion-ownership.test.ts | 50 ++++++++++++++++++ .../agents/lib/suggestion-ownership.ts | 47 +++++++++++++++++ .../features/agents/main/chat-input-area.tsx | 17 +++++-- 7 files changed, 195 insertions(+), 10 deletions(-) create mode 100644 src/renderer/features/agents/lib/suggestion-ownership.test.ts create mode 100644 src/renderer/features/agents/lib/suggestion-ownership.ts diff --git a/src/main/lib/trpc/routers/claude.ts b/src/main/lib/trpc/routers/claude.ts index 7db8156f..add8388e 100644 --- a/src/main/lib/trpc/routers/claude.ts +++ b/src/main/lib/trpc/routers/claude.ts @@ -1201,6 +1201,13 @@ export const claudeRouter = router({ }), enableTasks: input.enableTasks ?? true, }) + // The app's switch is the authority, and the env var is how the SDK + // documents winning: `CLAUDE_CODE_ENABLE_PROMPT_SUGGESTION` beats + // settings, so an inherited shell value is replaced with the + // preference rather than left to fight it. Both directions are + // set — a turn with the switch off must not inherit an on. + claudeEnv.CLAUDE_CODE_ENABLE_PROMPT_SUGGESTION = + (input.promptSuggestions ?? false) ? "true" : "false" // Debug logging in dev if (process.env.NODE_ENV !== "production") { @@ -1969,7 +1976,10 @@ ${prompt} // fallbackModel: "claude-opus-4-5-20251101", ...(input.thinking && { thinking: input.thinking }), ...(input.effort && { effort: input.effort }), - ...(input.promptSuggestions && { promptSuggestions: true }), + // Both directions, matching the env override above: absent + // means undecided, and undecided is where an inherited shell + // variable slips in and overrides the app's switch. + promptSuggestions: input.promptSuggestions ?? false, }, } diff --git a/src/renderer/features/agents/atoms/index.ts b/src/renderer/features/agents/atoms/index.ts index c7b546bd..2d3f2ff2 100644 --- a/src/renderer/features/agents/atoms/index.ts +++ b/src/renderer/features/agents/atoms/index.ts @@ -6,6 +6,7 @@ import { DEFAULT_CODEX_UI_MODEL, } from "../../../../shared/codex-model-id" import { atomWithWindowStorage } from "../../../lib/window-storage" +import type { PromptSuggestionEntry } from "../lib/suggestion-ownership" import type { FileMentionOption } from "../mentions/agents-mentions-editor" export type { AgentMode } from "../../../../shared/agent-mode" @@ -647,9 +648,18 @@ export const subChatRooModelIdAtomFamily = atomFamily((subChatId: string) => * nothing to read from it. */ export const subChatPromptSuggestionAtomFamily = atomFamily((_subChatId: string) => - atom(null), + atom(null), ) +/** + * The send counter behind `PromptSuggestionEntry.turn`. Every transport bumps + * it at the start of a turn; a suggestion carries the generation that + * produced it, and both the store and the composer refuse one whose + * generation is no longer current. In-memory on purpose: a restart has no + * late chunks to reject, and a persisted counter would only desynchronize. + */ +export const subChatTurnGenerationAtomFamily = atomFamily((_subChatId: string) => atom(0)) + export const subChatCodexThinkingAtomFamily = atomFamily((subChatId: string) => atom( (get) => { diff --git a/src/renderer/features/agents/lib/ipc-chat-transport.ts b/src/renderer/features/agents/lib/ipc-chat-transport.ts index 4a044715..2c4f4bed 100644 --- a/src/renderer/features/agents/lib/ipc-chat-transport.ts +++ b/src/renderer/features/agents/lib/ipc-chat-transport.ts @@ -34,6 +34,7 @@ import { pendingAuthRetryMessageAtom, subChatModelIdAtomFamily, subChatPromptSuggestionAtomFamily, + subChatTurnGenerationAtomFamily, } from "../atoms" import { useAgentSubChatStore } from "../stores/sub-chat-store" import type { AgentMessageMetadata } from "../ui/agent-message-usage" @@ -47,6 +48,7 @@ import { type SendMessagesOptions, type SubscriptionChunk, } from "./chat-chunk-atoms" +import { mayStoreSuggestion } from "./suggestion-ownership" // Error categories and their user-friendly messages const ERROR_TOAST_CONFIG: Record< @@ -165,6 +167,12 @@ type ChunkContext = ChatChunkContext & { images: ImageAttachment[] /** Written by this stream's own metadata chunk, read by the suggestion after it. */ sessionId: string | null + /** + * The turn generation this stream bumped when it started. A suggestion is + * stored under it, so a late chunk arriving after a newer send is refused + * instead of overwriting the newer turn's own. + */ + turnGeneration: number } type ChunkController = ReadableStreamDefaultController @@ -227,7 +235,21 @@ function routePromptSuggestion(chunk: SubscriptionChunk, ctx: ChunkContext): Chu } if (chunk.type !== "prompt-suggestion") return "enqueue" if (ctx.sessionId && chunk.sessionId !== ctx.sessionId) return "consumed" - appStore.set(subChatPromptSuggestionAtomFamily(ctx.subChatId), chunk.suggestion) + // The preference and the turn generation decide together: an inherited + // environment variable can make the CLI emit this while the app's switch is + // off, and a session id is reused across turns, so neither the switch nor + // the session alone answers whether the composer may still offer it. + const mayStore = mayStoreSuggestion({ + preferenceOn: appStore.get(promptSuggestionsEnabledAtom), + capturedTurn: ctx.turnGeneration, + currentTurn: appStore.get(subChatTurnGenerationAtomFamily(ctx.subChatId)), + }) + if (!mayStore) return "consumed" + appStore.set(subChatPromptSuggestionAtomFamily(ctx.subChatId), { + text: chunk.suggestion, + turn: ctx.turnGeneration, + engine: "legacy", + }) return "consumed" } @@ -417,9 +439,22 @@ export class IPCChatTransport implements ChatTransport { this.config.mode // A suggestion belongs to the turn that produced it, so starting a turn - // clears the last one: the composer must not offer a previous request's next - // step, and clicking it must not insert that into this prompt. + // bumps the generation — the store refuses a late chunk from the stream + // this send supersedes — and clears the last one: the composer must not + // offer a previous request's next step, and clicking it must not insert + // that into this prompt. + const turnGeneration = appStore.get(subChatTurnGenerationAtomFamily(this.config.subChatId)) + 1 + appStore.set(subChatTurnGenerationAtomFamily(this.config.subChatId), turnGeneration) appStore.set(subChatPromptSuggestionAtomFamily(this.config.subChatId), null) + // Aborting or failing this turn takes its suggestion with it, but only if + // it is still this turn's: a superseding send may already have begun, and + // its own suggestion must not be wiped by the stream it replaced. + const clearOwnSuggestion = () => { + const entry = appStore.get(subChatPromptSuggestionAtomFamily(this.config.subChatId)) + if (entry?.turn === turnGeneration) { + appStore.set(subChatPromptSuggestionAtomFamily(this.config.subChatId), null) + } + } // Stream tracking let _chunkCount = 0 @@ -435,6 +470,7 @@ export class IPCChatTransport implements ChatTransport { prompt, images, sessionId: null, + turnGeneration, } return new ReadableStream({ @@ -450,7 +486,10 @@ export class IPCChatTransport implements ChatTransport { sessionId, thinking, ...(effort && { effort }), - ...(promptSuggestions && { promptSuggestions: true }), + // Sent in both directions: with the option absent, an inherited + // `CLAUDE_CODE_ENABLE_PROMPT_SUGGESTION` in the shell decides, + // and the app's switch stops being the switch. + promptSuggestions, ...(modelString && { model: modelString }), ...(customConfig && { customConfig }), ...(selectedOllamaModel && { selectedOllamaModel }), @@ -482,6 +521,7 @@ export class IPCChatTransport implements ChatTransport { }, }) + clearOwnSuggestion() controller.error(err) }, onComplete: () => { @@ -495,6 +535,9 @@ export class IPCChatTransport implements ChatTransport { // Handle abort options.abortSignal?.addEventListener("abort", () => { + // A stopped turn leaves no next step behind: whatever arrived from + // it is withdrawn with the same generation guard as the error path. + clearOwnSuggestion() sub.unsubscribe() // trpcClient.claude.cancel.mutate({ subChatId: this.config.subChatId }) closeQuietly(controller) diff --git a/src/renderer/features/agents/lib/native-chat-transport.ts b/src/renderer/features/agents/lib/native-chat-transport.ts index 246b37dd..598e8bfd 100644 --- a/src/renderer/features/agents/lib/native-chat-transport.ts +++ b/src/renderer/features/agents/lib/native-chat-transport.ts @@ -22,7 +22,13 @@ import { } from "../../../lib/atoms" import { appStore } from "../../../lib/jotai-store" import { trpcClient } from "../../../lib/trpc" -import { MODEL_ID_MAP, pendingAuthRetryMessageAtom, subChatModelIdAtomFamily } from "../atoms" +import { + MODEL_ID_MAP, + pendingAuthRetryMessageAtom, + subChatModelIdAtomFamily, + subChatPromptSuggestionAtomFamily, + subChatTurnGenerationAtomFamily, +} from "../atoms" import { useAgentSubChatStore } from "../stores/sub-chat-store" import { applyCompactingChunks, @@ -109,6 +115,14 @@ export class NativeChatTransport implements ChatTransport { .allSubChats.find((subChat) => subChat.id === this.config.subChatId)?.mode || this.config.mode + // Turn ownership is shared with the legacy transport: bump the generation + // so a late suggestion from a still-open legacy stream is refused at the + // store, and clear whatever the previous turn left — this engine emits no + // suggestions of its own, and it inherits no stale ones. + const turnGeneration = appStore.get(subChatTurnGenerationAtomFamily(this.config.subChatId)) + 1 + appStore.set(subChatTurnGenerationAtomFamily(this.config.subChatId), turnGeneration) + appStore.set(subChatPromptSuggestionAtomFamily(this.config.subChatId), null) + const subId = this.config.subChatId.slice(-8) let chunkCount = 0 let lastChunkType = "" diff --git a/src/renderer/features/agents/lib/suggestion-ownership.test.ts b/src/renderer/features/agents/lib/suggestion-ownership.test.ts new file mode 100644 index 00000000..0df0b422 --- /dev/null +++ b/src/renderer/features/agents/lib/suggestion-ownership.test.ts @@ -0,0 +1,50 @@ +/** + * The ownership rules a prompt suggestion passes through on its way from a + * provider stream into the composer, and the four ways it is refused. + */ +import { describe, expect, it } from "vitest" +import { + mayStoreSuggestion, + type PromptSuggestionEntry, + suggestionIsCurrent, +} from "./suggestion-ownership" + +describe("mayStoreSuggestion", () => { + it("stores a live turn with the preference on", () => { + expect(mayStoreSuggestion({ preferenceOn: true, capturedTurn: 3, currentTurn: 3 })).toBe(true) + }) + + it("refuses when the app's switch is off, whatever the stream emitted", () => { + // An inherited CLAUDE_CODE_ENABLE_PROMPT_SUGGESTION can make the CLI send + // this; the preference still decides whether the composer ever sees it. + expect(mayStoreSuggestion({ preferenceOn: false, capturedTurn: 3, currentTurn: 3 })).toBe(false) + }) + + it("refuses a late arrival from a superseded turn", () => { + expect(mayStoreSuggestion({ preferenceOn: true, capturedTurn: 3, currentTurn: 4 })).toBe(false) + }) +}) + +describe("suggestionIsCurrent", () => { + const entry: PromptSuggestionEntry = { + text: "Now run the gate", + turn: 7, + engine: "legacy", + } + + it("keeps a suggestion while its engine and turn are still the composer's", () => { + expect(suggestionIsCurrent(entry, { engineNow: "legacy", turnNow: 7 })).toBe(true) + }) + + it("hides a suggestion after the sub-chat switched engines", () => { + expect(suggestionIsCurrent(entry, { engineNow: "native", turnNow: 7 })).toBe(false) + }) + + it("hides a suggestion once another turn has started", () => { + expect(suggestionIsCurrent(entry, { engineNow: "legacy", turnNow: 8 })).toBe(false) + }) + + it("treats an empty atom as nothing to show", () => { + expect(suggestionIsCurrent(null, { engineNow: "legacy", turnNow: 7 })).toBe(false) + }) +}) diff --git a/src/renderer/features/agents/lib/suggestion-ownership.ts b/src/renderer/features/agents/lib/suggestion-ownership.ts new file mode 100644 index 00000000..571b577f --- /dev/null +++ b/src/renderer/features/agents/lib/suggestion-ownership.ts @@ -0,0 +1,47 @@ +/** + * Which prompt suggestions may be stored, and which stored ones may still be + * shown. Session equality alone answered neither: a session id is reused + * across turns, outlives an engine switch, and says nothing about whether the + * app's own switch is on. The rules live here so the transport that stores, + * the transport that clears, and the composer that renders all ask the same + * three questions — of the turn generation, the engine, and the preference — + * instead of each inventing its own. + */ +import type { SubChatEngine } from "../atoms" + +/** What the composer is allowed to offer, and the provenance that gates it. */ +export type PromptSuggestionEntry = { + text: string + /** The turn generation that produced it; every send bumps the family. */ + turn: number + /** The engine whose stream carried it: the legacy SDK, or the native runtime. */ + engine: SubChatEngine +} + +/** + * Whether a suggestion that just arrived may be written to the sub-chat's + * atom. The switch is authoritative — an inherited environment variable can + * make the CLI emit suggestions while the app's preference is off, and a + * consumed chunk with nowhere to go is the store's job to refuse — and a + * captured turn that is no longer the current one is a late arrival from a + * stream this app has already superseded. + */ +export function mayStoreSuggestion(args: { + preferenceOn: boolean + capturedTurn: number + currentTurn: number +}): boolean { + return args.preferenceOn && args.capturedTurn === args.currentTurn +} + +/** + * Whether a stored suggestion still describes the composer it would insert + * into: same engine (a switch mid-flight changed who the next prompt would be + * addressed to) and same turn (something newer has started since). + */ +export function suggestionIsCurrent( + entry: PromptSuggestionEntry | null, + args: { engineNow: SubChatEngine; turnNow: number }, +): entry is PromptSuggestionEntry { + return entry !== null && entry.engine === args.engineNow && entry.turn === args.turnNow +} diff --git a/src/renderer/features/agents/main/chat-input-area.tsx b/src/renderer/features/agents/main/chat-input-area.tsx index 7ffc656f..5c860f62 100644 --- a/src/renderer/features/agents/main/chat-input-area.tsx +++ b/src/renderer/features/agents/main/chat-input-area.tsx @@ -79,6 +79,7 @@ import { subChatPromptSuggestionAtomFamily, subChatQwenModelIdAtomFamily, subChatRooModelIdAtomFamily, + subChatTurnGenerationAtomFamily, } from "../atoms" import { AgentsSlashCommand, type SlashCommandOption } from "../commands" import { AgentModelSelector, type AgentProviderId } from "../components/agent-model-selector" @@ -101,6 +102,7 @@ import { ROO_MODELS, } from "../lib/models" import type { DiffTextContext, SelectedTextContext } from "../lib/queue-utils" +import { suggestionIsCurrent } from "../lib/suggestion-ownership" import { AgentsFileMention, AgentsMentionsEditor, @@ -430,6 +432,12 @@ export const ChatInputArea = memo(function ChatInputArea({ [subChatId], ) const [promptSuggestion, setPromptSuggestion] = useAtom(promptSuggestionAtom) + // The generation the composer is on: a stored suggestion names the turn + // that produced it, and one from an older turn — or the other engine — is + // not this composer's next step, whatever a late stream wrote. + const turnGeneration = useAtomValue( + useMemo(() => subChatTurnGenerationAtomFamily(subChatId), [subChatId]), + ) const subChatModelIdAtom = useMemo(() => subChatModelIdAtomFamily(subChatId), [subChatId]) const [selectedSubChatModelId, setSelectedSubChatModelId] = useAtom(subChatModelIdAtom) @@ -1816,19 +1824,22 @@ export const ChatInputArea = memo(function ChatInputArea({ ) : null } > - {promptSuggestion && ( + {suggestionIsCurrent(promptSuggestion, { + engineNow: engine, + turnNow: turnGeneration, + }) && (
From 29a11b1184c5e3c31b451a747b317116e558c8e3 Mon Sep 17 00:00:00 2001 From: Owie6789 <151057755+Owie6789@users.noreply.github.com> Date: Thu, 24 Sep 2026 16:27:08 +0000 Subject: [PATCH 36/60] Give each sub-chat its own Claude effort, with a shared default One global claudeEffortAtom let a split pane change what its neighbour sent, while the model beside it was already per-sub-chat. Mirror the Codex thinking family: subChatClaudeEffortAtomFamily reads and writes this pane's slot, and lastSelected keeps the old storage key so a chat that never picked a level inherits the most recent one (and the new-chat form still has somewhere to write before a chat exists). Use `in`, not `??`, so an explicit "no effort" stays this chat's answer. The composer passes its subChatId into the picker; both transports read the family. Co-authored-by: arena-agent <297053741+arena-agent@users.noreply.github.com> --- .../agents/hooks/use-claude-model-picker.ts | 13 ++-- .../features/agents/lib/ipc-chat-transport.ts | 7 +- .../features/agents/main/chat-input-area.tsx | 2 +- src/renderer/lib/atoms/claude-effort.test.ts | 70 +++++++++++++++++++ src/renderer/lib/atoms/index.ts | 38 +++++++++- 5 files changed, 120 insertions(+), 10 deletions(-) create mode 100644 src/renderer/lib/atoms/claude-effort.test.ts diff --git a/src/renderer/features/agents/hooks/use-claude-model-picker.ts b/src/renderer/features/agents/hooks/use-claude-model-picker.ts index 1848f098..012b79a8 100644 --- a/src/renderer/features/agents/hooks/use-claude-model-picker.ts +++ b/src/renderer/features/agents/hooks/use-claude-model-picker.ts @@ -14,12 +14,12 @@ import { EFFORT_LEVELS } from "../../../../shared/effort" import { anthropicOnboardingCompletedAtom, apiKeyOnboardingCompletedAtom, - claudeEffortAtom, customClaudeConfigAtom, extendedThinkingEnabledAtom, normalizeCustomClaudeConfig, selectedOllamaModelAtom, showOfflineModeFeaturesAtom, + subChatClaudeEffortAtomFamily, } from "../../../lib/atoms" import { trpc } from "../../../lib/trpc" import type { AgentModelSelectorProps } from "../components/agent-model-selector" @@ -74,7 +74,7 @@ function useAvailableModels() { } } -export function useClaudeModelPicker(hiddenModels: readonly string[]) { +export function useClaudeModelPicker(hiddenModels: readonly string[], subChatId = "") { const modelSets = useAvailableModels() // Every other provider reads its selection from the list the hidden-model // setting has already been applied to (`codexUiModels`, `rooUiModels` and the @@ -113,10 +113,15 @@ export function useClaudeModelPicker(hiddenModels: readonly string[]) { const [thinkingEnabled, setThinkingEnabled] = useAtom(extendedThinkingEnabledAtom) // The effort rows come from the backend's own capability profile, so a - // provider that reports no effort control shows no sub-menu. + // provider that reports no effort control shows no sub-menu. The VALUE is + // owned by the sub-chat beside the model it will be sent with: the composer + // passes its id, and the new-chat form passes none, which reads and writes + // the last-selected value a fresh chat inherits. const { data: claudeCapability } = trpc.providers.get.useQuery({ id: "claude" }) const claudeEfforts = claudeCapability?.features.effort ? EFFORT_LEVELS : [] - const [selectedClaudeEffort, setSelectedClaudeEffort] = useAtom(claudeEffortAtom) + const [selectedClaudeEffort, setSelectedClaudeEffort] = useAtom( + subChatClaudeEffortAtomFamily(subChatId), + ) const props: SharedClaudePickerProps = { models, diff --git a/src/renderer/features/agents/lib/ipc-chat-transport.ts b/src/renderer/features/agents/lib/ipc-chat-transport.ts index 2c4f4bed..87a21a33 100644 --- a/src/renderer/features/agents/lib/ipc-chat-transport.ts +++ b/src/renderer/features/agents/lib/ipc-chat-transport.ts @@ -15,7 +15,6 @@ import { agentsLoginModalOpenAtom, autoOfflineModeAtom, type CustomClaudeConfig, - claudeEffortAtom, claudeLoginModalConfigAtom, customClaudeConfigAtom, enableTasksAtom, @@ -26,6 +25,7 @@ import { selectedOllamaModelAtom, sessionInfoAtom, showOfflineModeFeaturesAtom, + subChatClaudeEffortAtomFamily, } from "../../../lib/atoms" import { appStore } from "../../../lib/jotai-store" import { trpcClient } from "../../../lib/trpc" @@ -412,8 +412,9 @@ export class IPCChatTransport implements ChatTransport { ? ({ type: "adaptive" } as const) : ({ type: "disabled" } as const) // null is "let the CLI choose", so a chat that never opened the picker keeps - // the model's own default instead of a level this app guessed. - const effort = appStore.get(claudeEffortAtom) + // the model's own default instead of a level this app guessed. Read from + // THIS sub-chat's slot: two split panes carry two answers. + const effort = appStore.get(subChatClaudeEffortAtomFamily(this.config.subChatId)) const promptSuggestions = appStore.get(promptSuggestionsEnabledAtom) const historyEnabled = appStore.get(historyEnabledAtom) const enableTasks = appStore.get(enableTasksAtom) diff --git a/src/renderer/features/agents/main/chat-input-area.tsx b/src/renderer/features/agents/main/chat-input-area.tsx index 967ed248..1adfb17a 100644 --- a/src/renderer/features/agents/main/chat-input-area.tsx +++ b/src/renderer/features/agents/main/chat-input-area.tsx @@ -529,7 +529,7 @@ export const ChatInputArea = memo(function ChatInputArea({ hasCustomClaudeConfig, currentOllamaModel, props: claudePickerProps, - } = useClaudeModelPicker(hiddenModels) + } = useClaudeModelPicker(hiddenModels, subChatId) // Derived from the visible list, the way every other provider derives its // selection (`codexUiModels.find(...) || codexUiModels[0]` and the rest). Local // state plus a sync effect kept a model that the picker no longer offers: hide diff --git a/src/renderer/lib/atoms/claude-effort.test.ts b/src/renderer/lib/atoms/claude-effort.test.ts new file mode 100644 index 00000000..ef37e57b --- /dev/null +++ b/src/renderer/lib/atoms/claude-effort.test.ts @@ -0,0 +1,70 @@ +// @vitest-environment jsdom +/** + * Claude effort is owned by the sub-chat that chose it: one split pane + * setting `max` must not change what the other pane sends. The pre-existing + * global value survives as `lastSelected`, which is what a chat that has + * never picked reads and what the new-chat form (no sub-chat id yet) reads + * and writes. + */ +import { createStore } from "jotai" +import { afterEach, describe, expect, it, vi } from "vitest" +import { lastSelectedClaudeEffortAtom, subChatClaudeEffortAtomFamily } from "." + +afterEach(() => { + localStorage.clear() +}) + +describe("sub-chat Claude effort", () => { + it("keeps two sub-chats' picks independent", () => { + const store = createStore() + store.set(subChatClaudeEffortAtomFamily("chat-a"), "max") + store.set(subChatClaudeEffortAtomFamily("chat-b"), "low") + + expect(store.get(subChatClaudeEffortAtomFamily("chat-a"))).toBe("max") + expect(store.get(subChatClaudeEffortAtomFamily("chat-b"))).toBe("low") + // The global a third chat falls back on has not moved. + expect(store.get(lastSelectedClaudeEffortAtom)).toBe(null) + }) + + it("falls back to the last-selected pick for a chat that has never chosen", () => { + const store = createStore() + store.set(lastSelectedClaudeEffortAtom, "high") + expect(store.get(subChatClaudeEffortAtomFamily("chat-untouched"))).toBe("high") + }) + + it("treats a chat's explicit null as its own answer, not a miss", () => { + const store = createStore() + store.set(lastSelectedClaudeEffortAtom, "high") + store.set(subChatClaudeEffortAtomFamily("chat-a"), null) + // `high` is what a fresh chat gets; chat-a asked for the CLI default and + // must keep reading null rather than someone else's level. + expect(store.get(subChatClaudeEffortAtomFamily("chat-a"))).toBe(null) + expect(store.get(subChatClaudeEffortAtomFamily("chat-b"))).toBe("high") + }) + + it("reads and writes the last-selected pick when there is no chat yet", () => { + const store = createStore() + store.set(subChatClaudeEffortAtomFamily(""), "xhigh") + expect(store.get(lastSelectedClaudeEffortAtom)).toBe("xhigh") + // ...and a chat created afterwards inherits it. + store.set( + subChatClaudeEffortAtomFamily("chat-new"), + store.get(subChatClaudeEffortAtomFamily("")), + ) + expect(store.get(subChatClaudeEffortAtomFamily("chat-new"))).toBe("xhigh") + }) + + it("carries the pre-existing global storage key over as the last-selected pick", async () => { + // `getOnInit` reads storage when the atom module loads, so the seed has + // to be in place before the atoms are imported — which is exactly the + // production shape: an upgraded app starts with last version's value + // already on disk. Resetting the module registry replays that start. + vi.resetModules() + localStorage.setItem("preferences:claude-effort", JSON.stringify("low")) + const { lastSelectedClaudeEffortAtom: lastSelected, subChatClaudeEffortAtomFamily: family } = + await import(".") + const store = createStore() + expect(store.get(lastSelected)).toBe("low") + expect(store.get(family("chat-untouched"))).toBe("low") + }) +}) diff --git a/src/renderer/lib/atoms/index.ts b/src/renderer/lib/atoms/index.ts index 9af6a4a4..7fcfbdaf 100644 --- a/src/renderer/lib/atoms/index.ts +++ b/src/renderer/lib/atoms/index.ts @@ -1,5 +1,5 @@ import { atom } from "jotai" -import { atomWithStorage } from "jotai/utils" +import { atomFamily, atomWithStorage } from "jotai/utils" import type { EffortLevel } from "../../../shared/effort" import { desktopViewAtom as _desktopViewAtom } from "../../features/agents/atoms" import { createRendererSecretStorage } from "../renderer-secrets" @@ -355,12 +355,46 @@ export const extendedThinkingEnabledAtom = atomWithStorage( // The levels come from `src/shared/effort.ts`, the one vocabulary the main // process validates a request against. `null` means the chat never picked one // and the CLI decides, which is the behaviour before this setting existed. -export const claudeEffortAtom = atomWithStorage( +// +// Effort belongs to the sub-chat that chose it: the model beside it in the +// picker has always been per-sub-chat, and one split pane setting `max` must +// not change what the other pane sends. The pre-existing global value is kept +// as `lastSelected` — what a chat that has never picked one reads, and what +// the new-chat form writes before there is a chat to write into. The storage +// key for it is the old one, so every existing pick carries over untouched. +export const lastSelectedClaudeEffortAtom = atomWithStorage( "preferences:claude-effort", null, undefined, { getOnInit: true }, ) +const subChatClaudeEffortStorageAtom = atomWithStorage>( + "preferences:claude-effort-by-subchat", + {}, + undefined, + { getOnInit: true }, +) +export const subChatClaudeEffortAtomFamily = atomFamily((subChatId: string) => + atom( + (get) => { + // `in`, not `??`: a chat that explicitly chose "no effort" stores null, + // and a null that means "this chat decided" must not read as a miss and + // fall back to someone else's pick. + const stored = get(subChatClaudeEffortStorageAtom) + if (subChatId && subChatId in stored) return stored[subChatId] + return get(lastSelectedClaudeEffortAtom) + }, + (get, set, effort: EffortLevel | null) => { + if (!subChatId) { + set(lastSelectedClaudeEffortAtom, effort) + return + } + const current = get(subChatClaudeEffortStorageAtom) + if (current[subChatId] === effort) return + set(subChatClaudeEffortStorageAtom, { ...current, [subChatId]: effort }) + }, + ), +) // Preferences - Prompt suggestions // Off by default: turning it on asks the backend for a suggested next prompt at From 79294c9950250760da2a4a34de425e62d3e75b01 Mon Sep 17 00:00:00 2001 From: Owie6789 <151057755+Owie6789@users.noreply.github.com> Date: Thu, 24 Sep 2026 16:27:08 +0000 Subject: [PATCH 37/60] Send the chosen effort down the Native transport too The picker shows effort on the Native engine, but the transport omitted the field, so the daemon always ran its default. Validate effort with the same EFFORT_LEVELS enum the legacy chat router uses, and apply it best-effort via setReasoningEffort after set_model: a refusal (unsupported level, older daemon) warns and leaves the turn running rather than failing a prompt over a reasoning preference. Co-authored-by: arena-agent <297053741+arena-agent@users.noreply.github.com> --- src/main/lib/trpc/routers/runtime.ts | 22 +++++++++++++++++++ .../agents/lib/native-chat-transport.ts | 7 ++++++ 2 files changed, 29 insertions(+) diff --git a/src/main/lib/trpc/routers/runtime.ts b/src/main/lib/trpc/routers/runtime.ts index 777f726a..a721ffdd 100644 --- a/src/main/lib/trpc/routers/runtime.ts +++ b/src/main/lib/trpc/routers/runtime.ts @@ -12,6 +12,7 @@ import { observable } from "@trpc/server/observable" import { and, eq } from "drizzle-orm" import { z } from "zod" import { type AgentMode, agentModeSchema, DEFAULT_AGENT_MODE } from "../../../../shared/agent-mode" +import { EFFORT_LEVELS } from "../../../../shared/effort" import { nativeModeRefusal } from "../../../../shared/permissions/native-mode-floor" import { approvalWasDenied } from "../../claude/tool-approval" import type { UIMessageChunk } from "../../claude/types" @@ -284,6 +285,10 @@ export const runtimeRouter = router({ projectPath: z.string().optional(), mode: agentModeSchema.default(DEFAULT_AGENT_MODE), model: z.string().optional(), + // Same vocabulary and same validation as the legacy chat router: one + // effort scale across both engines, so a pane's visible level means + // the same thing whichever transport runs it. + effort: z.enum(EFFORT_LEVELS).optional(), customToken: z.string().optional(), customBaseUrl: z.string().optional(), images: z.array(imageAttachmentSchema).optional(), @@ -405,6 +410,23 @@ export const runtimeRouter = router({ await setNativeModelWithRetry(client, sessionId, input.model, safeEmit) + // The daemon takes effort as its own request, after the model so + // the provider context is settled. Applied best-effort: a refusal + // (a level this provider will not accept, an older daemon) leaves + // the turn running at the daemon's default rather than failing a + // prompt over a reasoning preference. + if (input.effort) { + try { + await client.setReasoningEffort(sessionId, input.effort) + } catch (error) { + console.warn( + `[Native] set_reasoning_effort refused (${input.effort}): ${ + error instanceof Error ? error.message : String(error) + }`, + ) + } + } + // The turn may have been cancelled while the daemon was starting; // never send a doomed turn (it would run uncancelled server-side). if (!isActive || turn.cancelled) { diff --git a/src/renderer/features/agents/lib/native-chat-transport.ts b/src/renderer/features/agents/lib/native-chat-transport.ts index 598e8bfd..0b27e488 100644 --- a/src/renderer/features/agents/lib/native-chat-transport.ts +++ b/src/renderer/features/agents/lib/native-chat-transport.ts @@ -19,6 +19,7 @@ import { normalizeCustomClaudeConfig, sessionInfoAtom, showOfflineModeFeaturesAtom, + subChatClaudeEffortAtomFamily, } from "../../../lib/atoms" import { appStore } from "../../../lib/jotai-store" import { trpcClient } from "../../../lib/trpc" @@ -86,6 +87,9 @@ export class NativeChatTransport implements ChatTransport { // Read model selection dynamically per sub-chat (so split panes stay independent) const selectedModelId = appStore.get(subChatModelIdAtomFamily(this.config.subChatId)) const modelString = MODEL_ID_MAP[selectedModelId] || MODEL_ID_MAP.opus + // ...and the effort beside it, from the same family the legacy transport + // reads: both engines send what their pane's picker last chose. + const effort = appStore.get(subChatClaudeEffortAtomFamily(this.config.subChatId)) // Offline/Ollama routing is a legacy-path feature; refuse loudly rather // than silently running the turn against cloud credentials. @@ -139,6 +143,9 @@ export class NativeChatTransport implements ChatTransport { projectPath: this.config.projectPath, mode: currentMode, ...(modelString && { model: modelString }), + // The same per-sub-chat effort the legacy transport sends: the + // daemon's `set_reasoning_effort` carries it the rest of the way. + ...(effort && { effort }), ...(customConfig?.token && { customToken: customConfig.token }), ...(customConfig?.baseUrl && { customBaseUrl: customConfig.baseUrl }), ...(images.length > 0 && { images }), From 5bbd966458605fc48d511d3d8a6e4cd4eb96d1e6 Mon Sep 17 00:00:00 2001 From: Owie6789 <151057755+Owie6789@users.noreply.github.com> Date: Thu, 24 Sep 2026 16:27:08 +0000 Subject: [PATCH 38/60] Correct the docs the review proved wrong, and ratify the pins - research record: drop the false "null coerces to NaN" claim; Node 22 coerces null to 0 and the transform normalizes with ?? 0. - benchmarks: label the 103-file / 1885-test row as the pre-harness measurement it is and name CI as the live count. - CLAUDE.md, openspec/project.md, permission-hook: move 0.2.45 / 2.1.45 / 0.137.0 to 0.3.270 / 2.1.270 / 0.154.0 after re-reading the deny/ask claim in the 0.3.270 bundle. - probe-command: narrow the null exit-code comment to "no usable exit code", naming ENOENT as the one case the helper can prove. - roadmap: add section 16 ratifying every lockfile addition this step made. Co-authored-by: arena-agent <297053741+arena-agent@users.noreply.github.com> --- .../app/benchmarks/2026-09-23-sdk-0-3-pins.md | 14 +++++++++----- .dump/app/research/2026-09-13-sdk-0-3-bump.md | 9 +++++++-- .dump/app/roadmap/12-sdk-and-pins.md | 18 ++++++++++++++++++ CLAUDE.md | 6 +++--- openspec/project.md | 2 +- src/main/lib/claude/permission-hook.ts | 4 +++- src/main/lib/providers/probe-command.ts | 12 ++++++++---- 7 files changed, 49 insertions(+), 16 deletions(-) diff --git a/.dump/app/benchmarks/2026-09-23-sdk-0-3-pins.md b/.dump/app/benchmarks/2026-09-23-sdk-0-3-pins.md index c0feab9f..a7cd93c3 100644 --- a/.dump/app/benchmarks/2026-09-23-sdk-0-3-pins.md +++ b/.dump/app/benchmarks/2026-09-23-sdk-0-3-pins.md @@ -71,11 +71,15 @@ the `linux-x64` checksum in the SDK's `manifest.json`, size `223981040`. The SDK pin and the CLI pin are therefore the same artifact, not two independent guesses. Gate wall clocks on the same 2 CPU sandbox, as a regression proxy rather than a -product number: `npm run test` over 103 files and 1885 tests took 50.5 s; -`npm run test:node` 59 tests took 39.0 s; `npm run test:contracts` 382 tests took -10.5 s; `npx biome check .` over 977 files took 3 s; `tsc --noEmit` over the whole -repo took 42 s to 65 s per run and reports 0 errors, which is the ratchet's -baseline rather than new debt. +product number. Measured at the head that added this file, before the render +test harness landed — the counts are that run's, kept because the timings +beside them are its own: `npm run test` over 103 files and 1885 tests took +50.5 s; `npm run test:node` 59 tests took 39.0 s; `npm run test:contracts` 382 +tests took 10.5 s; `npx biome check .` over 977 files took 3 s; `tsc +--noEmit` over the whole repo took 42 s to 65 s per run and reports 0 errors, +which is the ratchet's baseline rather than new debt. The suite has grown since +(harness, review fixes): the PR's CI quality job on the final head is the live +count, and it — not this snapshot — is what a later comparison reads. ## What the numbers mean for the next step diff --git a/.dump/app/research/2026-09-13-sdk-0-3-bump.md b/.dump/app/research/2026-09-13-sdk-0-3-bump.md index e374b010..190a8c2f 100644 --- a/.dump/app/research/2026-09-13-sdk-0-3-bump.md +++ b/.dump/app/research/2026-09-13-sdk-0-3-bump.md @@ -91,8 +91,13 @@ supertypes of what the wire actually carries: The reader that renders a failed result now names a part it cannot render (`[document]`) instead of calling every non-text part `[image]`. - `ClaudeUsage` cache counters accept `null`, because the pinned SDK reports null - for a cache tier that did not apply, and a null in an arithmetic expression is - a NaN in the context indicator rather than a missing number. + for a cache tier that did not apply. JavaScript does not turn that null into a + NaN — `null + 1` is `1`, because additive arithmetic coerces null to zero — + so the requirement is not "guard against NaN" but "normalize before anything + needs a number": consumer arithmetic, JSON that must carry a real integer, and + APIs whose signatures are not nullable. The transform does that with `?? 0` + at the two sites it copies usage out of a message, which is why the context + indicator has always shown a number rather than a failed one. Three approaches were tried and rejected, recorded so nobody retries them: widening the local mirrors to all 16 `ContentBlockParam` members by hand diff --git a/.dump/app/roadmap/12-sdk-and-pins.md b/.dump/app/roadmap/12-sdk-and-pins.md index 6a5569d5..8022a547 100644 --- a/.dump/app/roadmap/12-sdk-and-pins.md +++ b/.dump/app/roadmap/12-sdk-and-pins.md @@ -81,3 +81,21 @@ The Codex schema drift and parity verification that this bump makes necessary, w ## 15. Handoff notes Write the upstream behaviour list to `.dump/app/research/2026-09-13-sdk-0-3-bump.md` and name it in {{S19}} and {{S20}}, which assume those three tools exist. + +## 16. As shipped: what the lockfile moved beyond the three pins + +The step is the only one allowed to touch `package.json` and `bun.lock`, so +everything this head added there is ratified here rather than left for the next +reader to reconcile against §1's original allowance: + +| Added | Why | Ratified by | +| --- | --- | --- | +| `@dnd-kit/core` 6.3.1, `@dnd-kit/sortable` 10.0.0, `@dnd-kit/utilities` 3.2.2 (exact) | the drag-and-drop decision §1 already names | this step's original allowance | +| `jsdom` 30.1.1, `@testing-library/react` 16.3.3, `@testing-library/dom` 10.4.2 (exact, dev) | the `renderPart` split was conditioned on pinning the current renderer output first — a snapshot harness over the dispatcher's 30 branches before any line moved | the split-with-tests decision taken on this PR, accepted lockfile change and all | +| `sharp` 0.35.4 (exact, dev) | `scripts/generate-icon.mjs` needs it as a direct dev dependency once the lock was regenerated; previously satisfied by hoisting | carried in the PR body's dependency row | +| `@modelcontextprotocol/sdk` `^1.25.3` → exact `1.30.1` | SDK 0.3.270 declares that package as a peer at `^1.29.0`; the old range left `npm ls` reporting `invalid: "^1.29.0"` and exiting `ELSPROBLEMS` | the SDK's own peer contract — an exact pin inside the declared range, no override | +| `@anthropic-ai/sdk` (lock-only, peer of the pinned SDK) | bun's auto-install of the peer the SDK declares at `>=0.93.0` | the transitive-peer decision recorded on this PR | + +No override, no caret on a CI-resolved dependency, and no runtime dependency +the three pins did not earn: everything above is dev tooling, the dnd +allowance, or a contract the new SDK imposes. diff --git a/CLAUDE.md b/CLAUDE.md index 39a94475..87eb7f34 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -80,8 +80,8 @@ npm run db:generate # generate a migration from the schema npm run db:push # push the schema directly, dev only # Bundled agent binaries, version pinned in the script -npm run claude:download # 2.1.45 -npm run codex:download # 0.137.0 +npm run claude:download # 2.1.270 +npm run codex:download # 0.154.0 ``` Three scripts need a note. @@ -204,7 +204,7 @@ Versions are the ones `bun.lock` resolves. | Components | Radix UI, Lucide icons, Motion, Sonner | | State | Jotai, Zustand, React Query | | Backend | tRPC over Electron IPC, Drizzle ORM, better-sqlite3 | -| AI | `@anthropic-ai/claude-agent-sdk` 0.2.45, plus the Codex app-server adapter | +| AI | `@anthropic-ai/claude-agent-sdk` 0.3.270, plus the Codex app-server adapter | | Schemas | Effect 4.0.0-rc.112, an exact pin, load bearing for `src/shared/contracts` | | Lint and format | Biome 2.5.13, every rule at error | | Tests | Vitest 4.1.11 and `node --test` | diff --git a/openspec/project.md b/openspec/project.md index d523a27a..7e1fbc3b 100644 --- a/openspec/project.md +++ b/openspec/project.md @@ -15,7 +15,7 @@ Roadmap step 01 measured every path, name and count on this page against the tre | Components | Radix UI, Lucide icons, Motion, Sonner | | State | Jotai, Zustand, React Query | | Backend | tRPC over Electron IPC, Drizzle ORM, better-sqlite3 | -| AI | `@anthropic-ai/claude-agent-sdk` 0.2.45, plus the Codex app-server adapter | +| AI | `@anthropic-ai/claude-agent-sdk` 0.3.270, plus the Codex app-server adapter | | Schemas | `effect` 4.0.0-rc.112, an exact pin, load bearing for `src/shared/contracts` | | Lint and format | Biome 2.5.13, every rule at error | | Tests | Vitest 4.1.11 and `node --test` | diff --git a/src/main/lib/claude/permission-hook.ts b/src/main/lib/claude/permission-hook.ts index 84838b51..9e322e1f 100644 --- a/src/main/lib/claude/permission-hook.ts +++ b/src/main/lib/claude/permission-hook.ts @@ -80,7 +80,9 @@ export async function permissionFloorDecision( if (decision.decision === "allow") return {} // `ask` goes out as `ask` rather than being hardened into `deny`, and that rests // on the pinned engine honouring the value. Checked against the bundled CLI in - // `@anthropic-ai/claude-agent-sdk` 0.2.45, which switches on it and sets + // `@anthropic-ai/claude-agent-sdk` 0.3.270 (the same `["deny", "ask"]` + // validation it had at 0.2.45, re-read in the 0.3.270 bundle), which switches + // on it and sets // `permissionBehavior` to `ask`, and throws on a value it does not know, so an // unsupported spelling cannot slip through as an approval. Hardening it here // would break the critical-path breaker's own contract, which is that Turbo asks diff --git a/src/main/lib/providers/probe-command.ts b/src/main/lib/providers/probe-command.ts index 029f0d3f..5744a6a6 100644 --- a/src/main/lib/providers/probe-command.ts +++ b/src/main/lib/providers/probe-command.ts @@ -16,10 +16,14 @@ export const PROBE_TIMEOUT_MS = 15_000 /** * The code a probe ended with. `0` when the process ran and exited cleanly, its - * numeric code when it ran and did not, and `null` when it never ran at all: - * `execFile` reports a spawn failure as a string errno (`ENOENT`) on - * `error.code` and a non-zero exit as a number, so only the number is an exit - * code and callers read `null` as "the binary is not there". + * numeric code when it ran and did not, and `null` when `execFile` reported no + * numeric code at all — a spawn failure arrives as a string errno (`ENOENT`, + * but also `EACCES`), and a timeout, a signal and a max-buffer cut arrive with + * a non-numeric `error.code` too. `null` therefore means "no usable exit + * code", which callers may read as "the binary is not there" only knowing the + * ENOENT case is the one this helper can name; the cause lives on the error + * itself, and collapsing it here is the classification the follow-up replaces + * with a structured result. */ function exitCodeOf(error: ExecFileException | null): number | null { if (!error) return 0 From 3018d7cf6023af3911550b590d561aadb86c46f4 Mon Sep 17 00:00:00 2001 From: Owie6789 <151057755+Owie6789@users.noreply.github.com> Date: Thu, 24 Sep 2026 16:27:08 +0000 Subject: [PATCH 39/60] Record round 4: twenty-one threads, sixteen fixes, four declines MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Add the kilo-code-bot disposition table to the remediation record: every claim, the evidence that settled it, and the commit that carried the fix — plus the four evidenced declines (conversation reset, per-model matrix, TaskOutput classifier handoff, packaged size) and the gate results at the round's end. Co-authored-by: arena-agent <297053741+arena-agent@users.noreply.github.com> --- ...2026-09-23-review-and-sonar-remediation.md | 73 +++++++++++++++++++ 1 file changed, 73 insertions(+) diff --git a/.dump/app/decisions/2026-09-23-review-and-sonar-remediation.md b/.dump/app/decisions/2026-09-23-review-and-sonar-remediation.md index 1462f451..1d80ed7c 100644 --- a/.dump/app/decisions/2026-09-23-review-and-sonar-remediation.md +++ b/.dump/app/decisions/2026-09-23-review-and-sonar-remediation.md @@ -509,3 +509,76 @@ local build or package. | `skills:verify` | 50 of 50 locked skills verified, 2 unrecorded project-owned | | GitHub Actions | Build ubuntu-24.04, macos-14, windows-2022; Package unsigned ubuntu-24.04, macos-14; quality gates; security gates; both Socket reports; CodeRabbit — all pass. Buoy, Sourcery, DeepSource skip | | SonarQube Cloud | quality gate passed, 0 new issues, 0 debt, 0 hotspots, 0.3% duplication on 2302 new lines | + +## Round 4: the independent review, twenty-one threads + +A fourth reviewer — `kilo-code-bot`, six multipass reviews plus targeted +confirmations — opened twenty-one inline threads against head `5ea0b64` with a +`CHANGES_REQUESTED` review: five Major, six Moderate (later ten), five Low +across three addenda, plus a mediation roadmap in the PR thread. Each claim was +checked against the pinned SDK's own `.d.ts`, the installed bundle, this +repo's code and its recorded snapshots before anything was touched; the +disposition table below is the round's result. Nothing was accepted on the +reviewer's word alone, and nothing was declined without evidence in the reply. + +### Fixed — sixteen of twenty-one + +| Thread | What it claimed | What settled it | Commit | +| --- | --- | --- | --- | +| MCP peer contract (Moderate) | SDK 0.3.270 declares `@modelcontextprotocol/sdk` `^1.29.0`; the graph resolves `1.25.3` | Reproduced: `npm ls` → `invalid: "^1.29.0"`, `ELSPROBLEMS`. Exact `1.30.1` pinned, newest in range | `dfe7044` | +| Transform throws on malformed lines (Major) | `apiRetryMessage` and `handlePromptSuggestion` trust fields `toClaudeStreamMessage` proved only have a string `type`; qwen `feedLine` has no catch, so the throw escapes into the main process | Both handlers read only documented shapes; `feedLine` wraps premap/result/transform in a boundary that settles the turn. Claude router already caught; qwen was the process-killer | `ead68c1` | +| Launch rendered as completion (Major) | `AgentOutput.status` is `completed \| async_launched \| remote_launched`; the row asked only "streaming?" | `sdk-tools.d.ts` confirms the union; `isLaunchedAgentOutput` branches the title and the registry phrase. Three statuses pinned in a snapshot | `6804c04` | +| Nested ancestry orphaned (Major) | Grouping resolved a child's parent by first id segment among top-level tasks only, so `B:C` never found `A:B` | The transform composes `parentOriginal:childOriginal` from the SDK's immediate `parent_tool_use_id`; lookup now goes through every task's original id, keys children by the parent's full id, self-parent skipped as the cycle guard, recursive rows capped at three levels | `6804c04` | +| Inert focusable subtitles (Moderate) | Every registry subtitle wore `role="button"` and a tab stop; TaskOutput rows have no action | Our own snapshot contained `Task: task_1`; button semantics now require a handler. Snapshot deltas verified as exactly those two attribute deletions | `b4c9ebb` | +| `shell_id` dropped (Low) | `TaskStopInput` accepts the deprecated `shell_id`; the shared reader took only `task_id`/`taskId` | `sdk-tools.d.ts:920` confirms; reader takes `shell_id` under the same `Task:` label; registry test pins it | `b4c9ebb` | +| Effort is global (Major) | One `claudeEffortAtom` for every pane while the model beside it is per-sub-chat | Mirrored the Codex thinking family: `subChatClaudeEffortAtomFamily` + `lastSelected` keeping the old storage key; `in` not `??` so an explicit null stays that chat's answer. Five tests including migration | `07bf80a` | +| Native sends no effort (Moderate) | The picker shows effort on the Native engine; the transport omits the field | Contract verified before building: the pinned runtime-client exposes `setReasoningEffort` (`set_reasoning_effort`). Router validates `z.enum(EFFORT_LEVELS)` and applies it best-effort after `set_model` | `7f7a8f9` | +| Stale suggestion survives turn/engine (Moderate ×2) | Session equality stood in for turn ownership; native never cleared; abort/error left the row | `{text, turn, engine}` entry, per-sub-chat generation bumped by both transports at send, store-time and render-time gates, guarded clears on abort/error. Seven tests on the pure rules | `34ea81b` | +| Preference not authoritative (Moderate) | Option sent only when true, so an inherited `CLAUDE_CODE_ENABLE_PROMPT_SUGGESTION` decided | Env var overridden both directions from the toggle (the SDK documents env beating settings), `promptSuggestions: false` sent as explicitly as true, store refuses chunks while off | `34ea81b` | +| Click erases draft (Moderate) | `setValue(suggestion)` clears and rebuilds the editor | `mergeDraftWithSuggestion`: replace only an empty/whitespace draft, otherwise append after one space, draft byte-stable. Five tests; voice path uses the same join | `5e24490` | +| Switch has no name (Moderate) | Sibling ``, no `aria-label`; 31 switches, 0 labelled | `aria-label="Prompt Suggestions"` on the new control; test queries `getByRole("switch", { name: ... })` | `7e2c988` | +| NaN claim (Low) | JS coerces `null` to 0; the transform normalizes with `?? 0` | Verified on Node 22; research record rewritten to the real contract | `2456829` | +| Benchmark counts (Low) | Row says 103 files / 1885 tests; head CI reports more | Row labelled as the pre-harness measurement it is, with CI named as the live count | `2456829` | +| Stale version strings (Low) | CLAUDE.md, openspec/project.md, permission-hook comment still name 0.2.45 / 2.1.45 / 0.137.0 | All three moved to 0.3.270 / 2.1.270 / 0.154.0; the hook's `["deny", "ask"]` claim re-read in the 0.3.270 bundle before the number moved; roadmap §16 added ratifying every lockfile addition | `2456829` | +| Probe `null` doc (Low) | `null` also follows timeout, signal, EACCES, max-buffer — not only "never ran" | Comment narrowed to "no usable exit code", naming ENOENT as the one case the helper can prove; behavior deliberately unchanged (pre-existing, deferred with the structured-result follow-up) | `2456829` | + +### Declined — four threads, evidence in each reply + +- **`conversation_reset` (Major).** `/clear` is a client-side builtin in this + app — it "creates new sub-chat" (`builtin-commands.ts`) — so the CLI's reset + event never reaches the transformer from any path the app owns. The + router lines the thread cites are unreachable for it. +- **Per-model capability matrix (Major).** `src/shared/effort.ts` already + records the SDK's documented clamp ("an effort above a model's + `maxEffortLevel` is clamped to it") and files per-model + `supportedEffortLevels` as the named follow-up; no model-info source exists + in-repo to build the matrix from. The engine half of the same thread was + fixed instead (native now sends effort), and the "hidden control still + sent" premise was checked: nothing hides the effort rows for custom or + offline — `efforts` is passed unconditionally — so there is no hidden + control to disagree with. +- **`TaskOutput` in `READ_ONLY_TOOLS` (Moderate).** The classifier belongs to + the sibling permissions lane; checked `arena/01a0bb80-mauscode` at push time + — still `BashOutput` only — so the handoff stands as a handoff, and touching + it here would collide with the lane that owns it. +- **Packaged artifact size (Moderate).** The benchmark already lists packaged + and bundle size as *needs `bun run package:linux`* rows owned by CI and step + 30; this sandbox cannot produce the artifact, and the record says so rather + than claiming a measurement it does not have. + +Two sub-asks rode along declined with their threads: capping the raw qwen line +before `JSON.parse` (identical feedLine at base, outside the diff cause, and +JSON.parse was already guarded) and restructuring the probe's null into a +structured result (behavior inherited from the base wrappers; the comment now +says what it proves, the follow-up keeps the redesign). + +### Where the round stood + +Ten commits from `5ea0b64` to `2456829`, every gate re-run in full at each of +the three code commits that needed it and at the docs commit: biome 987 files +0 findings, typecheck ratchet 0 errors against a 0 baseline, lint 932 files +clean, vitest 109 files / 1951 passed / 1 skipped (29 new tests across the +round), test:node 59, test:contracts 382, audit ratchet at the 3-critical +baseline, skills:verify 50 of 50. Pushes after `5e24490` are queued locally — +the session's GitHub token expired mid-round (401 on REST and GraphQL) — and +the replies to the twenty-one threads post when the connection is restored. From c84a571876a276a9da0b7e4294c5ecdba2315d9a Mon Sep 17 00:00:00 2001 From: Owie6789 <151057755+Owie6789@users.noreply.github.com> Date: Thu, 24 Sep 2026 16:46:04 +0000 Subject: [PATCH 40/60] Reap the Qwen child when a malformed line ends the turn MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit feedLine's catch settled the turn and resolved `done`, but left the child running: the router cleared `activeTurn`, stdout kept flowing, and every subsequent line still hit a turn nobody owned. Settled turns now refuse further lines, and the catch escalates the same SIGINT→SIGTERM→SIGKILL path a user interrupt uses, so nothing outlives the error the renderer was already told about. `settle` is idempotent, so the close handler does not resurrect the turn. Co-authored-by: arena-agent <297053741+arena-agent@users.noreply.github.com> --- src/main/lib/qwen-print/session.ts | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/src/main/lib/qwen-print/session.ts b/src/main/lib/qwen-print/session.ts index 6bffa4c9..01accb26 100644 --- a/src/main/lib/qwen-print/session.ts +++ b/src/main/lib/qwen-print/session.ts @@ -318,6 +318,10 @@ export function runQwenPrintTurn(opts: RunQwenPrintTurnOptions): QwenPrintTurn { } const feedLine = (rawLine: string) => { + // A settled turn (error, result, interrupt) must not keep consuming + // stdout: the router has already cleared `activeTurn`, and a late line + // would emit chunks against nothing. + if (settled) return const text = rawLine.trim() if (text.length === 0) return let parsed: unknown @@ -354,6 +358,11 @@ export function runQwenPrintTurn(opts: RunQwenPrintTurnOptions): QwenPrintTurn { const errorMessage = `Malformed provider message: ${detail}` emit({ type: "error", errorText: errorMessage }) settle({ status: "error", errorMessage, sessionId }) + // The child is still running the turn the renderer has already been + // told ended. Reap it (same SIGINT→SIGTERM→SIGKILL escalation as a + // user interrupt) so it cannot keep executing or emitting; `settle` + // is idempotent, so the close handler will not resurrect the turn. + interrupt() } } From e429440e9f7fafc2d1ac7a4ee13395be655ed969 Mon Sep 17 00:00:00 2001 From: Owie6789 <151057755+Owie6789@users.noreply.github.com> Date: Thu, 24 Sep 2026 16:46:04 +0000 Subject: [PATCH 41/60] Withdraw the suggestion when a provider error chunk arrives Abort and transport-level onError already cleared this turn's suggestion, but a provider failure is an ordinary data chunk: it is toasted by routeChunk and never reaches the subscription's onError, so a suggestion stored before the failure stayed in the atom and the composer kept offering a next step from a turn that failed. error and auth-error chunks now clear under the same generation guard. Co-authored-by: arena-agent <297053741+arena-agent@users.noreply.github.com> --- src/renderer/features/agents/lib/ipc-chat-transport.ts | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/src/renderer/features/agents/lib/ipc-chat-transport.ts b/src/renderer/features/agents/lib/ipc-chat-transport.ts index 87a21a33..78455b47 100644 --- a/src/renderer/features/agents/lib/ipc-chat-transport.ts +++ b/src/renderer/features/agents/lib/ipc-chat-transport.ts @@ -504,6 +504,16 @@ export class IPCChatTransport implements ChatTransport { _chunkCount++ _lastChunkType = chunk.type + // A provider failure arrives as an ordinary data chunk and is + // toasted by `routeChunk` — it never hits this subscription's + // `onError`. A failed turn leaves no next step behind: withdraw + // this turn's suggestion under the same generation guard as + // abort and transport failure, so a row stored before the + // failure cannot still be offered afterwards. + if (chunk.type === "error" || chunk.type === "auth-error") { + clearOwnSuggestion() + } + if (routeChunk(chunk, ctx, controller) !== "enqueue") return enqueueChunk(controller, chunk) if (chunk.type === "finish") closeQuietly(controller) From 3c3500e92c7432f7d649839a271a80a4df421bc1 Mon Sep 17 00:00:00 2001 From: Owie6789 <151057755+Owie6789@users.noreply.github.com> Date: Thu, 24 Sep 2026 16:46:04 +0000 Subject: [PATCH 42/60] Hide a stored suggestion once the preference is turned off MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit mayStoreSuggestion checked the switch when a chunk arrived; suggestionIsCurrent asked only engine and turn — so turning Prompt Suggestions off left the stored row visible, and turning it back on resurrected it. The preference is now the third question the render gate asks, matching what the module's own contract already promised. Co-authored-by: arena-agent <297053741+arena-agent@users.noreply.github.com> --- .../agents/lib/suggestion-ownership.test.ts | 26 +++++++++++++++---- .../agents/lib/suggestion-ownership.ts | 15 ++++++++--- .../features/agents/main/chat-input-area.tsx | 5 ++++ 3 files changed, 37 insertions(+), 9 deletions(-) diff --git a/src/renderer/features/agents/lib/suggestion-ownership.test.ts b/src/renderer/features/agents/lib/suggestion-ownership.test.ts index 0df0b422..c22ae36b 100644 --- a/src/renderer/features/agents/lib/suggestion-ownership.test.ts +++ b/src/renderer/features/agents/lib/suggestion-ownership.test.ts @@ -32,19 +32,35 @@ describe("suggestionIsCurrent", () => { engine: "legacy", } - it("keeps a suggestion while its engine and turn are still the composer's", () => { - expect(suggestionIsCurrent(entry, { engineNow: "legacy", turnNow: 7 })).toBe(true) + it("keeps a suggestion while its engine, turn and preference still allow it", () => { + expect( + suggestionIsCurrent(entry, { engineNow: "legacy", turnNow: 7, preferenceOn: true }), + ).toBe(true) }) it("hides a suggestion after the sub-chat switched engines", () => { - expect(suggestionIsCurrent(entry, { engineNow: "native", turnNow: 7 })).toBe(false) + expect( + suggestionIsCurrent(entry, { engineNow: "native", turnNow: 7, preferenceOn: true }), + ).toBe(false) }) it("hides a suggestion once another turn has started", () => { - expect(suggestionIsCurrent(entry, { engineNow: "legacy", turnNow: 8 })).toBe(false) + expect( + suggestionIsCurrent(entry, { engineNow: "legacy", turnNow: 8, preferenceOn: true }), + ).toBe(false) + }) + + it("hides a stored suggestion once the preference is turned off", () => { + // Turning Prompt Suggestions off withdraws what is already in the atom; + // turning it back on must not resurrect a row the user dismissed. + expect( + suggestionIsCurrent(entry, { engineNow: "legacy", turnNow: 7, preferenceOn: false }), + ).toBe(false) }) it("treats an empty atom as nothing to show", () => { - expect(suggestionIsCurrent(null, { engineNow: "legacy", turnNow: 7 })).toBe(false) + expect(suggestionIsCurrent(null, { engineNow: "legacy", turnNow: 7, preferenceOn: true })).toBe( + false, + ) }) }) diff --git a/src/renderer/features/agents/lib/suggestion-ownership.ts b/src/renderer/features/agents/lib/suggestion-ownership.ts index 571b577f..695424ab 100644 --- a/src/renderer/features/agents/lib/suggestion-ownership.ts +++ b/src/renderer/features/agents/lib/suggestion-ownership.ts @@ -36,12 +36,19 @@ export function mayStoreSuggestion(args: { /** * Whether a stored suggestion still describes the composer it would insert - * into: same engine (a switch mid-flight changed who the next prompt would be - * addressed to) and same turn (something newer has started since). + * into: the preference is still on (turning Prompt Suggestions off withdraws + * whatever is already stored — and turning it back on must not resurrect it), + * same engine (a switch mid-flight changed who the next prompt would be + * addressed to), and same turn (something newer has started since). */ export function suggestionIsCurrent( entry: PromptSuggestionEntry | null, - args: { engineNow: SubChatEngine; turnNow: number }, + args: { engineNow: SubChatEngine; turnNow: number; preferenceOn: boolean }, ): entry is PromptSuggestionEntry { - return entry !== null && entry.engine === args.engineNow && entry.turn === args.turnNow + return ( + entry !== null && + args.preferenceOn && + entry.engine === args.engineNow && + entry.turn === args.turnNow + ) } diff --git a/src/renderer/features/agents/main/chat-input-area.tsx b/src/renderer/features/agents/main/chat-input-area.tsx index 1adfb17a..494cc360 100644 --- a/src/renderer/features/agents/main/chat-input-area.tsx +++ b/src/renderer/features/agents/main/chat-input-area.tsx @@ -37,6 +37,7 @@ import { hiddenModelsAtom, normalizeCodexApiKey, pinnedOpenRouterModelsAtom, + promptSuggestionsEnabledAtom, sessionInfoAtom, } from "../../../lib/atoms" import { @@ -439,6 +440,9 @@ export const ChatInputArea = memo(function ChatInputArea({ const turnGeneration = useAtomValue( useMemo(() => subChatTurnGenerationAtomFamily(subChatId), [subChatId]), ) + // Turning the switch off withdraws whatever suggestion is already stored; + // the render gate asks for the preference so a re-enable cannot resurrect it. + const promptSuggestionsOn = useAtomValue(promptSuggestionsEnabledAtom) const subChatModelIdAtom = useMemo(() => subChatModelIdAtomFamily(subChatId), [subChatId]) const [selectedSubChatModelId, setSelectedSubChatModelId] = useAtom(subChatModelIdAtom) @@ -1826,6 +1830,7 @@ export const ChatInputArea = memo(function ChatInputArea({ {suggestionIsCurrent(promptSuggestion, { engineNow: engine, turnNow: turnGeneration, + preferenceOn: promptSuggestionsOn, }) && (
From faae3950dca705aa83f408dfc9bbd9811f81e5ed Mon Sep 17 00:00:00 2001 From: Owie6789 <151057755+Owie6789@users.noreply.github.com> Date: Thu, 24 Sep 2026 16:46:04 +0000 Subject: [PATCH 43/60] Compare the nesting map by content, so task-row memo survives MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit areTaskToolPropsEqual rejected whenever nestedChildren was a new function — and messageParts is rebuilt every render (the AI SDK mutates parts in place), so the map and any callback over it were fresh every time. Every AgentTaskTool re-rendered on every stream tick, undoing the memo the comparator exists to protect. The message-level map now rides along as a prop and is compared by content (same arePartsEqual as every other part), so a grandchild change still reaches the row and an unchanged tree does not. Co-authored-by: arena-agent <297053741+arena-agent@users.noreply.github.com> --- .../agents/main/assistant-message-item.tsx | 2 + .../features/agents/ui/agent-task-tool.tsx | 9 ++++ .../features/agents/ui/agent-tool-utils.ts | 52 ++++++++++++++++--- 3 files changed, 55 insertions(+), 8 deletions(-) diff --git a/src/renderer/features/agents/main/assistant-message-item.tsx b/src/renderer/features/agents/main/assistant-message-item.tsx index edc78a29..e718f7b4 100644 --- a/src/renderer/features/agents/main/assistant-message-item.tsx +++ b/src/renderer/features/agents/main/assistant-message-item.tsx @@ -684,6 +684,7 @@ function renderOrphanTaskGroup( }} nestedTools={group.parts} nestedChildren={ctx.nestedChildren} + nestedToolsMap={ctx.nestedToolsMap} chatStatus={ctx.status} /> ) @@ -721,6 +722,7 @@ function renderSubagentTask(part: NormalizedPart, idx: number, ctx: PartRenderCo part={part} nestedTools={nestedTools} nestedChildren={ctx.nestedChildren} + nestedToolsMap={ctx.nestedToolsMap} chatStatus={ctx.status} /> ) diff --git a/src/renderer/features/agents/ui/agent-task-tool.tsx b/src/renderer/features/agents/ui/agent-task-tool.tsx index f3d2a7d5..eb9bbb2b 100644 --- a/src/renderer/features/agents/ui/agent-task-tool.tsx +++ b/src/renderer/features/agents/ui/agent-task-tool.tsx @@ -17,6 +17,7 @@ import { areTaskToolPropsEqual, isLaunchedAgentOutput, type NestedToolsLookup, + type NestedToolsMapLike, } from "./agent-tool-utils" interface AgentTaskToolProps { @@ -28,6 +29,12 @@ interface AgentTaskToolProps { * rows only (the depth cap, and callers that have no map to offer). */ nestedChildren?: NestedToolsLookup + /** + * The message-level map behind `nestedToolsMap`, carried for the memo only: + * a grandchild lives under another key, and content comparison is what lets + * a re-render through without defeating the memo with a fresh callback. + */ + nestedToolsMap?: NestedToolsMapLike /** How many subagent levels deep this call already is; the cap counts them. */ depth?: number chatStatus?: string @@ -59,6 +66,7 @@ export const AgentTaskTool = memo(function AgentTaskTool({ part, nestedTools, nestedChildren, + nestedToolsMap, depth = 0, chatStatus, }: AgentTaskToolProps) { @@ -257,6 +265,7 @@ export const AgentTaskTool = memo(function AgentTaskTool({ part={nestedPart} nestedTools={childId ? nestedChildren(childId) : []} nestedChildren={nestedChildren} + nestedToolsMap={nestedToolsMap} depth={depth + 1} chatStatus={chatStatus} /> diff --git a/src/renderer/features/agents/ui/agent-tool-utils.ts b/src/renderer/features/agents/ui/agent-tool-utils.ts index fb70145d..831f969f 100644 --- a/src/renderer/features/agents/ui/agent-tool-utils.ts +++ b/src/renderer/features/agents/ui/agent-tool-utils.ts @@ -130,11 +130,36 @@ export function areToolPropsEqual( * Compare function for AgentTaskTool which has additional nestedTools prop. */ /** - * A subagent's own children by its id. The message-level map stands behind it - * so the memo can compare identity: a grandchild is not in this task's own - * `nestedTools`, and only the lookup's identity says it changed. + * A subagent's own children by its id, and the message-level map that stands + * behind it. The map's CONTENT is what the memo compares: a grandchild is not + * in this task's own `nestedTools`, and only a change under some other key + * says it moved. Identity of either is useless here — `messageParts` is + * rebuilt every render (the AI SDK mutates parts in place), so the map and + * any callback over it are fresh objects with unchanged contents. */ export type NestedToolsLookup = (toolCallId: string) => ToolPartLike[] +export type NestedToolsMapLike = ReadonlyMap + +function nestedMapsEqual( + prev: NestedToolsMapLike | undefined, + next: NestedToolsMapLike | undefined, +): boolean { + if (prev === next) return true + const prevSize = prev?.size ?? 0 + const nextSize = next?.size ?? 0 + if (prevSize !== nextSize) return false + if (!prev || !next) return prevSize === 0 + for (const [id, prevParts] of prev) { + const nextParts = next.get(id) + if (!nextParts || nextParts.length !== prevParts.length) return false + for (let i = 0; i < prevParts.length; i++) { + if (!arePartsEqual(prevParts[i] as ToolPartLike, nextParts[i] as ToolPartLike)) { + return false + } + } + } + return true +} /** * A result that launched work instead of finishing it. The pinned SDK types @@ -153,6 +178,7 @@ export function areTaskToolPropsEqual( part: ToolPartLike nestedTools: ToolPartLike[] nestedChildren?: NestedToolsLookup + nestedToolsMap?: NestedToolsMapLike depth?: number chatStatus?: string }, @@ -160,15 +186,25 @@ export function areTaskToolPropsEqual( part: ToolPartLike nestedTools: ToolPartLike[] nestedChildren?: NestedToolsLookup + nestedToolsMap?: NestedToolsMapLike depth?: number chatStatus?: string }, ): boolean { - // The lookup's identity changes when the message's nesting map is rebuilt, - // which is the only signal that reaches here about a descendant deeper than - // this task's own `nestedTools`. Checked first so the completed short - // circuit below cannot hide it. - if (prevProps.nestedChildren !== nextProps.nestedChildren) return false + // Descendants beyond this task's own `nestedTools` are only visible through + // the message-level map. Compare it by CONTENT — the map (and the lookup + // over it) is rebuilt every render, so identity would reject every time and + // undo the memo this comparator exists to protect. Checked first so the + // completed short circuit below cannot hide a grandchild change. + if (!nestedMapsEqual(prevProps.nestedToolsMap, nextProps.nestedToolsMap)) return false + // Fallback for callers that offer a lookup without the map. + if ( + prevProps.nestedToolsMap === undefined && + nextProps.nestedToolsMap === undefined && + prevProps.nestedChildren !== nextProps.nestedChildren + ) { + return false + } if (prevProps.depth !== nextProps.depth) return false // Compare main part first From 8b0ee3d3222da854e4d0d7dc46ebca1697b8b7ba Mon Sep 17 00:00:00 2001 From: Owie6789 <151057755+Owie6789@users.noreply.github.com> Date: Thu, 24 Sep 2026 16:46:04 +0000 Subject: [PATCH 44/60] Give a tooltip-only subtitle a tab stop without calling it a button MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Removing role and tabindex from action-less subtitles fixed the inert control, but TooltipTrigger hangs off focus — a truncated Bash or Read path that only a mouse could reveal. A subtitle with a tooltip and no action is now focusable (tabIndex 0, no role, no Enter/Space handler): keyboard users can open the tooltip, and the span still claims no affordance it does not have. The snapshot delta is exactly one tabindex on the Read row's path; role=button count unchanged. Co-authored-by: arena-agent <297053741+arena-agent@users.noreply.github.com> --- .../assistant-message-item.test.tsx.snap | 2 +- .../agents/ui/agent-tool-call.test.tsx | 29 +++++++++++++++++-- .../features/agents/ui/agent-tool-call.tsx | 22 ++++++++++++-- 3 files changed, 47 insertions(+), 6 deletions(-) diff --git a/src/renderer/features/agents/main/__snapshots__/assistant-message-item.test.tsx.snap b/src/renderer/features/agents/main/__snapshots__/assistant-message-item.test.tsx.snap index 0ea838d2..3e21a71f 100644 --- a/src/renderer/features/agents/main/__snapshots__/assistant-message-item.test.tsx.snap +++ b/src/renderer/features/agents/main/__snapshots__/assistant-message-item.test.tsx.snap @@ -29,7 +29,7 @@ exports[`AssistantMessageItem, one message per branch of the part dispatcher > r exports[`AssistantMessageItem, one message per branch of the part dispatcher > renders a reasoning part 1`] = `"
"`; -exports[`AssistantMessageItem, one message per branch of the part dispatcher > renders a registry tool as a single row 1`] = `"
Readeffort.ts
"`; +exports[`AssistantMessageItem, one message per branch of the part dispatcher > renders a registry tool as a single row 1`] = `"
Readeffort.ts
"`; exports[`AssistantMessageItem, one message per branch of the part dispatcher > renders a second plan operation as a mini indicator, not a card 1`] = `"
Created plan
"`; diff --git a/src/renderer/features/agents/ui/agent-tool-call.test.tsx b/src/renderer/features/agents/ui/agent-tool-call.test.tsx index c15c1f3c..5182b235 100644 --- a/src/renderer/features/agents/ui/agent-tool-call.test.tsx +++ b/src/renderer/features/agents/ui/agent-tool-call.test.tsx @@ -12,11 +12,16 @@ import { render } from "@testing-library/react" import { describe, expect, it } from "vitest" import { EyeIcon } from "../../../components/ui/icons" +import { TooltipProvider } from "../../../components/ui/tooltip" import { AgentToolCall } from "./agent-tool-call" +function renderCall(ui: React.ReactElement) { + return render({ui}) +} + describe("AgentToolCall subtitle affordance", () => { - it("is plain text when the row has no action", () => { - const { getByText } = render( + it("is plain text when the row has no action and no tooltip", () => { + const { getByText } = renderCall( { expect(subtitle.getAttribute("tabindex")).toBeNull() }) + it("is a focusable non-button when only a tooltip needs a keyboard entry", () => { + // TooltipTrigger hangs off focus; a truncated path that only a mouse can + // reveal is not keyboard-accessible. Focusable is not the same as a button. + const { getByText } = renderCall( + , + ) + const subtitle = getByText("src/very/long/path/to/file.ts") + expect(subtitle.getAttribute("role")).toBeNull() + expect(subtitle.getAttribute("tabindex")).toBe("0") + }) + it("is a button when the row has an action to press", () => { - const { getByText } = render( + const { getByText } = renderCall( void, + tooltipFocusable = false, ): React.ReactElement { - if (!onClick) return {content} + if (!onClick) { + if (!tooltipFocusable) return {content} + return ( + /* biome-ignore lint/a11y/noNoninteractiveTabindex: keyboard entry point for the tooltip trigger; focus opens it, and the span deliberately claims no interactive role. */ + + {content} + + ) + } return ( /* biome-ignore lint/a11y/useSemanticElements: compact inline action; a native button would require style resets. */ Date: Thu, 24 Sep 2026 17:01:28 +0000 Subject: [PATCH 45/60] Clear every open Sonar finding on this pull request MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Ten issues sat on PR 69 after the round-4/5 heads (gate still passed; the debt was ours either way): - runtime.ts S3776 (18→): the chat IIFE's session/credential/model/effort setup becomes openNativeTurnSession + applyNativeEffort — the same named-step shape the mode floor and credential prepare already use, so the handler stays flat and the gate stays green. - assistant-message-item S3776 (20→) + S7755: buildNestingIndex owns the composite-id classification at module scope; last-segment lookup uses .at(-1). - transform S3358: the retry attempt phrase extracts its max-retries half before the outer ternary. - agent-tool-utils S6582: length check is one optional chain. - agent-task-tool S6551: a non-string toolCallId is "" rather than "[object Object]". - claude-effort.test ×2 S5906: toBeNull(). - agent-tool-call S6819: the action subtitle is a native ) } diff --git a/src/renderer/features/agents/ui/agent-tool-utils.ts b/src/renderer/features/agents/ui/agent-tool-utils.ts index 831f969f..37e64684 100644 --- a/src/renderer/features/agents/ui/agent-tool-utils.ts +++ b/src/renderer/features/agents/ui/agent-tool-utils.ts @@ -151,7 +151,7 @@ function nestedMapsEqual( if (!prev || !next) return prevSize === 0 for (const [id, prevParts] of prev) { const nextParts = next.get(id) - if (!nextParts || nextParts.length !== prevParts.length) return false + if (nextParts?.length !== prevParts.length) return false for (let i = 0; i < prevParts.length; i++) { if (!arePartsEqual(prevParts[i] as ToolPartLike, nextParts[i] as ToolPartLike)) { return false diff --git a/src/renderer/lib/atoms/claude-effort.test.ts b/src/renderer/lib/atoms/claude-effort.test.ts index ef37e57b..4be23465 100644 --- a/src/renderer/lib/atoms/claude-effort.test.ts +++ b/src/renderer/lib/atoms/claude-effort.test.ts @@ -23,7 +23,7 @@ describe("sub-chat Claude effort", () => { expect(store.get(subChatClaudeEffortAtomFamily("chat-a"))).toBe("max") expect(store.get(subChatClaudeEffortAtomFamily("chat-b"))).toBe("low") // The global a third chat falls back on has not moved. - expect(store.get(lastSelectedClaudeEffortAtom)).toBe(null) + expect(store.get(lastSelectedClaudeEffortAtom)).toBeNull() }) it("falls back to the last-selected pick for a chat that has never chosen", () => { @@ -38,7 +38,7 @@ describe("sub-chat Claude effort", () => { store.set(subChatClaudeEffortAtomFamily("chat-a"), null) // `high` is what a fresh chat gets; chat-a asked for the CLI default and // must keep reading null rather than someone else's level. - expect(store.get(subChatClaudeEffortAtomFamily("chat-a"))).toBe(null) + expect(store.get(subChatClaudeEffortAtomFamily("chat-a"))).toBeNull() expect(store.get(subChatClaudeEffortAtomFamily("chat-b"))).toBe("high") }) From 8071737d5f0d3c257871c28c6f59ca4930d75a2d Mon Sep 17 00:00:00 2001 From: Owie6789 <151057755+Owie6789@users.noreply.github.com> Date: Thu, 24 Sep 2026 17:07:11 +0000 Subject: [PATCH 46/60] Record round 5 and the Sonar leak-period triage MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Five new Kilo threads (all fixed), the ten open PR findings with nine cleared and S6845 left open as a documented false positive, and the state of the queue after the sandbox reset that wiped the round-4 patches — rebuilt, pushed, replied. Co-authored-by: arena-agent <297053741+arena-agent@users.noreply.github.com> --- ...2026-09-23-review-and-sonar-remediation.md | 41 +++++++++++++++++++ 1 file changed, 41 insertions(+) diff --git a/.dump/app/decisions/2026-09-23-review-and-sonar-remediation.md b/.dump/app/decisions/2026-09-23-review-and-sonar-remediation.md index 1d80ed7c..09e74bad 100644 --- a/.dump/app/decisions/2026-09-23-review-and-sonar-remediation.md +++ b/.dump/app/decisions/2026-09-23-review-and-sonar-remediation.md @@ -582,3 +582,44 @@ round), test:node 59, test:contracts 382, audit ratchet at the 3-critical baseline, skills:verify 50 of 50. Pushes after `5e24490` are queued locally — the session's GitHub token expired mid-round (401 on REST and GraphQL) — and the replies to the twenty-one threads post when the connection is restored. + +## Round 5: five new threads, five fixes, and the Sonar leak period + +Kilo's reconciliation at 16:01Z opened five new inline findings against +`5e24490` (the twelve already-discussed rows in that summary were answers to +the round-4 threads). Each was verified in the code before anything was +touched; all five were real; all five were fixed: + +| Thread | Claim | Commit | +| --- | --- | --- | +| `qwen-print/session.ts` | Settling a malformed line left the child running; stdout kept feeding a dead turn | `c84a571` — settled turns refuse further lines; catch escalates SIGINT→SIGTERM→SIGKILL | +| `ipc-chat-transport.ts:506` | Provider `error` chunks never hit subscription `onError`, so a stored suggestion survived the failure | `e429440` — `error`/`auth-error` chunks clear under the generation guard | +| `suggestion-ownership.ts:42` | `suggestionIsCurrent` asked engine and turn but not the preference; off→on resurrected a withdrawn row | `3c3500e` — preference is the third argument the render gate passes | +| `agent-tool-utils.ts:171` | `nestedChildren` identity defeated task-row memo on every stream render | `faae395` — nesting map compared by content, not callback identity | +| `agent-tool-call.tsx:19` | Tooltip-only subtitles lost keyboard access when role/tabindex were removed | `8b0ee3d` — tab stop with no role, only when a tooltip exists | + +Replies posted on each thread; summary comment `5818332768`. + +### Sonar on the same heads + +After the round-4/5 pushes the leak period carried **10 open issues** (gate +still passed: 0 hotspots, 0.7% duplication on new code). Nine were cleared in +`1fe7cf9` — both S3776s by extraction (`openNativeTurnSession` / +`applyNativeEffort`, `buildNestingIndex`), the S3358 by hoisting the +max-retries half, S7755/S6582/S6551/S5906×2 as one-liners, and S6819 by +making the action subtitle a native `
"`; exports[`AssistantMessageItem, one message per branch of the part dispatcher > renders a second plan operation as a mini indicator, not a card 1`] = `"
Created plan
"`; diff --git a/src/renderer/features/agents/ui/agent-tool-call.test.tsx b/src/renderer/features/agents/ui/agent-tool-call.test.tsx index 4f7c15ad..57f7a2d1 100644 --- a/src/renderer/features/agents/ui/agent-tool-call.test.tsx +++ b/src/renderer/features/agents/ui/agent-tool-call.test.tsx @@ -6,8 +6,11 @@ * Enter/Space handling, and attached the handler only when the row had an * action — so every `TaskOutput` and `TaskStop` row (and any row rendered * without its file-open provider) was a focusable, screen-reader-announced - * control that did nothing. The two halves of that are pinned here: no - * action, no button; action, button that works. + * control that did nothing. Pinned here: no affordance, no button; an + * action or a tooltip that needs a keyboard entry, a native button that + * carries the focus without declaring a `tabIndex` (Sonar S6845 reads the + * declaration on a non-interactive element, and Radix's own TooltipTrigger + * is a button). */ import { render } from "@testing-library/react" import { describe, expect, it } from "vitest" @@ -35,10 +38,13 @@ describe("AgentToolCall subtitle affordance", () => { expect(subtitle.getAttribute("tabindex")).toBeNull() }) - it("is a focusable non-button when only a tooltip needs a keyboard entry", () => { + it("is a button when only a tooltip needs a keyboard entry", () => { // TooltipTrigger hangs off focus; a truncated path that only a mouse can - // reveal is not keyboard-accessible. Focusable is not the same as a button. - const { getByText } = renderCall( + // reveal is not keyboard-accessible. A native button is Radix's own + // default trigger: it carries the tab stop natively, so no tabIndex sits + // on a non-interactive element (Sonar S6845), and with no action to + // press it gets no handler to fake one. + const { getByRole } = renderCall( { isError={false} />, ) - const subtitle = getByText("src/very/long/path/to/file.ts") - expect(subtitle.getAttribute("role")).toBeNull() - expect(subtitle.getAttribute("tabindex")).toBe("0") + const subtitle = getByRole("button", { name: "src/very/long/path/to/file.ts" }) + expect(subtitle.tagName).toBe("BUTTON") + expect(subtitle.getAttribute("tabindex")).toBeNull() }) it("is a button when the row has an action to press", () => { diff --git a/src/renderer/features/agents/ui/agent-tool-call.tsx b/src/renderer/features/agents/ui/agent-tool-call.tsx index 8d526887..bfb7efe3 100644 --- a/src/renderer/features/agents/ui/agent-tool-call.tsx +++ b/src/renderer/features/agents/ui/agent-tool-call.tsx @@ -5,17 +5,18 @@ import { TextShimmer } from "../../../components/ui/text-shimmer" import { Tooltip, TooltipContent, TooltipTrigger } from "../../../components/ui/tooltip" /** - * The subtitle span, wearing button semantics only when there is an action to - * press. A `role="button"` whose Enter and Space do nothing is a control that - * lies: keyboard and screen-reader users can focus it, it announces itself as - * interactive, and nothing happens — which is every `TaskOutput` and - * `TaskStop` row, since neither offers an action beyond being read. + * The subtitle span, wearing button semantics when there is an affordance to + * press — an action, or a tooltip that needs a keyboard entry point — and + * nothing at all when there is neither. * - * Without an action but WITH a tooltip, the span still takes a tab stop: - * `TooltipTrigger` hangs off focus, and a truncated path that only a mouse - * can reveal is not keyboard-accessible. A bare tab stop is not a button — - * no role, no Enter/Space handler — so it does not claim an affordance it - * does not have; it only lets focus open the tooltip that is already there. + * The tooltip-only case used to be a focusable span with `tabIndex={0}` and + * no role: Sonar's S6845 reads that as a tab stop on a non-interactive + * element, and the alternative of `role="button"` with an Enter handler that + * does nothing is the control that lies (which is the `TaskOutput` and + * `TaskStop` rows this component was already fixed for). The button here is + * the resolution Radix itself picks: `TooltipTrigger` renders a button by + * default, so focus is the affordance, native focusability carries the tab + * stop without declaring one, and activation has nothing to fake. */ function subtitleSpan( content: React.ReactNode, @@ -28,17 +29,13 @@ function subtitleSpan( // semantics (implicit role, keyboard activation, no hand-rolled keydown). const buttonReset = "appearance-none border-0 bg-transparent p-0 m-0 font-[inherit] text-[inherit]" - if (!onClick) { - if (!tooltipFocusable) return {content} - return ( - /* biome-ignore lint/a11y/noNoninteractiveTabindex: keyboard entry point for the tooltip trigger; focus opens it, and the span deliberately claims no interactive role. */ - - {content} - - ) - } + // No action and nothing to reveal: passive text, out of the tab order. + if (!onClick && !tooltipFocusable) return {content} + // The tooltip-only button is a trigger, not a link: it must not advertise a + // pointer click the UA stylesheet would otherwise promise. + const cursor = onClick ? "" : " cursor-default" return ( - ) From 646b8e843817a000552678932f0da841001b0ca0 Mon Sep 17 00:00:00 2001 From: Owie6789 <151057755+Owie6789@users.noreply.github.com> Date: Thu, 24 Sep 2026 22:09:37 +0000 Subject: [PATCH 55/60] Bound the round-8 serializers to parts that can still change MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Round 9's three perf reviews all landed on the same cost the round-8 blind-spot fix introduced: areMessagePropsEqual stringified every part's input+output on every memo comparison, and nestingFingerprintOf stringified every nested tool's input+output on every render — quadratic in transcript size for a finished session. Both sites now share one settled-part rule. A part whose state string is terminal (output-available, output-error, result, error — the new isTerminalStateString, the state string alone, so an output that arrived before its state caught up stays live) and whose input and output REFERENCES are unchanged has its cached string reused in O(1); the SDK does not reopen a completed part, so the string still describes it. Only live parts — streaming input, growing output — serialize per comparison, bounded by the active tool's payload instead of the transcript's. Reference or state changes still serialize, so the round-7/8 guarantees hold on their real shape. The fingerprint's reuse lives in a private segment cache keyed by toolCallId, cleared alongside the other per-tool caches, written once per render in the message component; row comparators only read the returned string, preserving round 7's single-consumer property. Tests pin both halves: a streaming grandchild deep-mutating through one input object still changes the fingerprint, and a settled terminal part reuses its segment until a reference moves. Evidence: biome full 0 findings; lint-changed clean; typecheck 0 errors; tsgo --singleThreaded 0 errors; vitest 1964 passed 1 skipped (110 files); test:node 59 pass; contracts 382 passed; audit ratchet clean; skills 50 of 50; typecheck ratchet 0 <= 0. Co-authored-by: arena-agent <297053741+arena-agent@users.noreply.github.com> --- .../agents/main/assistant-message-item.tsx | 66 ++++++++++++++----- .../features/agents/ui/agent-tool-state.ts | 19 ++++-- .../agents/ui/agent-tool-utils.test.ts | 37 +++++++++-- .../features/agents/ui/agent-tool-utils.ts | 62 +++++++++++++++-- 4 files changed, 151 insertions(+), 33 deletions(-) diff --git a/src/renderer/features/agents/main/assistant-message-item.tsx b/src/renderer/features/agents/main/assistant-message-item.tsx index 31a5260d..3c497d2b 100644 --- a/src/renderer/features/agents/main/assistant-message-item.tsx +++ b/src/renderer/features/agents/main/assistant-message-item.tsx @@ -54,6 +54,7 @@ import { parseMcpToolType, type ToolDisplayPart, } from "../ui/agent-tool-registry" +import { isTerminalStateString } from "../ui/agent-tool-state" import { isPlanFile, nestingFingerprintOf } from "../ui/agent-tool-utils" import { AgentWebFetchTool } from "../ui/agent-web-fetch-tool" import { AgentWebSearchCollapsible } from "../ui/agent-web-search-collapsible" @@ -566,18 +567,29 @@ export interface AssistantMessageItemProps { // Cache for tracking previous message state per sub-chat/message // (to detect AI SDK in-place mutations without cross-chat collisions) // Stores both text lengths and tool states for complete change detection +interface PartIOSnapshot { + state: string | undefined + input: unknown + output: unknown + json: string | undefined +} + interface MessageStateSnapshot { textLengths: number[] partStates: (string | undefined)[] /** - * Every part's input and output, stringified. A nested tool can mutate - * either in place while its state and every text length around it stay - * unchanged, and nothing downstream of this memo runs when it skips a - * render — the task-row fingerprint included — so the row would never see - * the streamed update. Replaces the last-part-only input tracking: the last - * part is covered by this the same way, and one array is not two rules. + * Every part's input and output, stringified — but only once per state of + * the part. A nested tool can mutate either in place while its state and + * every text length around it stay unchanged, and nothing downstream of + * this memo runs when it skips a render, so the check has to see it. The + * cost stays bounded because a part whose state string is terminal and + * whose input/output references are unchanged is SETTLED: the SDK does not + * reopen a completed part, so its cached string still describes it and the + * comparison reuses it in O(1). Only live parts (streaming input, growing + * output) serialize per comparison, bounded by the active tool's payload + * rather than the whole transcript's — the round-9 reviews' point. */ - partIOJsons: (string | undefined)[] + partIO: PartIOSnapshot[] } const messageStateCache = new Map() @@ -638,22 +650,39 @@ function areMessagePropsEqual( // Get current message state from parts const nextParts = next.message?.parts || [] + // Read the previous snapshot first: the per-part IO check below reuses its + // strings for settled parts instead of serializing them again. + const cachedState = cacheKey ? messageStateCache.get(cacheKey) : undefined + const currentState: MessageStateSnapshot = { textLengths: nextParts.map((p) => getTrackedPartTextLength(p)), // Track ALL part states - critical for detecting Edit plan file streaming! partStates: nextParts.map((p) => p.state), // Track every part's input AND output — tool streaming arrives as in-place // mutation of both, on non-last parts too (parallel calls, nested tools). - partIOJsons: nextParts.map((p) => - p.input === undefined && p.output === undefined - ? undefined - : JSON.stringify([p.input, p.output]), - ), + partIO: nextParts.map((p, i) => { + const prev = cachedState?.partIO?.[i] + if ( + prev !== undefined && + prev.state === p.state && + prev.input === p.input && + prev.output === p.output && + isTerminalStateString(prev.state) + ) { + return prev // settled: same terminal state, same references + } + return { + state: p.state, + input: p.input, + output: p.output, + json: + p.input === undefined && p.output === undefined + ? undefined + : JSON.stringify([p.input, p.output]), + } + }), } - // Get cached state from previous render - const cachedState = cacheKey ? messageStateCache.get(cacheKey) : undefined - // If no cache, this is first comparison - cache and allow render if (!cachedState || !cacheKey) { if (cacheKey) messageStateCache.set(cacheKey, currentState) @@ -675,9 +704,10 @@ function areMessagePropsEqual( } // Compare every part's input/output (detects in-place tool streaming the - // state and text-length checks cannot see) - for (let i = 0; i < currentState.partIOJsons.length; i++) { - if (cachedState.partIOJsons?.[i] !== currentState.partIOJsons[i]) { + // state and text-length checks cannot see). Settled parts carried their + // cached string over above, so this is a reference compare for them. + for (let i = 0; i < currentState.partIO.length; i++) { + if (cachedState.partIO?.[i]?.json !== currentState.partIO[i].json) { messageStateCache.set(cacheKey, currentState) return false // A part's input or output changed } diff --git a/src/renderer/features/agents/ui/agent-tool-state.ts b/src/renderer/features/agents/ui/agent-tool-state.ts index be3e9ae8..bc78b3d1 100644 --- a/src/renderer/features/agents/ui/agent-tool-state.ts +++ b/src/renderer/features/agents/ui/agent-tool-state.ts @@ -33,16 +33,25 @@ export type ToolPartLike = { result?: unknown } +/** + * The state strings that say the SDK has finished with this part — the state + * string alone, deliberately not `getToolLifecycleState().isTerminal`, which + * also counts a present `output`: a part whose output arrived before its + * state string caught up is still live, and the settled-part serializers + * below must keep serializing it every time. + */ +const TERMINAL_STATE_STRINGS = new Set(["output-available", "output-error", "result", "error"]) + +export function isTerminalStateString(state: unknown): boolean { + return typeof state === "string" && TERMINAL_STATE_STRINGS.has(state) +} + export function getToolLifecycleState(part: ToolPartLike): ToolLifecycleState { const state = typeof part?.state === "string" ? part.state : undefined const hasOutput = hasValue(part?.output) const hasResult = hasValue(part?.result) const isInputStreaming = state === "input-streaming" - const isTerminalState = - state === "output-available" || - state === "output-error" || - state === "result" || - state === "error" + const isTerminalState = isTerminalStateString(state) const isError = state === "output-error" || state === "error" || diff --git a/src/renderer/features/agents/ui/agent-tool-utils.test.ts b/src/renderer/features/agents/ui/agent-tool-utils.test.ts index e5729868..721be19b 100644 --- a/src/renderer/features/agents/ui/agent-tool-utils.test.ts +++ b/src/renderer/features/agents/ui/agent-tool-utils.test.ts @@ -21,11 +21,15 @@ function taskPart(id: string): ToolPartLike { } } -function grandchild(id: string, filePath: string): ToolPartLike { +function grandchild( + id: string, + filePath: string, + state: string = "output-available", +): ToolPartLike { return { type: "tool-Read", toolCallId: id, - state: "output-available", + state, input: { file_path: filePath }, output: { content: `contents of ${filePath}` }, } @@ -60,17 +64,40 @@ describe("nestingFingerprintOf", () => { it("changes when a streaming grandchild mutates in place", () => { const map = new Map([ ["A", [taskPart("A")]], - ["A:B", [grandchild("A:B:C", "src/one.ts")]], + ["A:B", [grandchild("A:B:C", "src/one.ts", "input-streaming")]], ]) const before = nestingFingerprintOf(map) - // The AI SDK mutates parts in place: same Map, same array, new output. + // The AI SDK mutates parts in place — same part, same input object, the + // file_path rewritten underneath. A live (non-terminal) part is + // re-serialized every render, so the fingerprint sees it. const child = map.get("A:B")?.[0] expect(child).toBeDefined() - if (child) child.output = { content: "halfway through the file" } + const input = child?.input as { file_path: string } + input.file_path = "src/two.ts" const after = nestingFingerprintOf(map) expect(after).not.toBe(before) }) + it("holds a settled terminal part's segment until a reference moves", () => { + // The round-9 cost bound: a completed part whose input/output references + // have not moved is not re-stringified, so a deep rewrite of a FINISHED + // part does not reach the fingerprint — the SDK does not reopen one. + // What keeps it honest is the other half: a changed reference or state + // re-serializes immediately, which is the case the row memos feed on. + const map = new Map([ + ["A:B", [grandchild("A:B:C", "src/one.ts")]], // terminal by default + ]) + const before = nestingFingerprintOf(map) + const child = map.get("A:B")?.[0] + const input = child?.input as { file_path: string } + input.file_path = "src/deep-rewritten.ts" // same reference, settled part + expect(nestingFingerprintOf(map)).toBe(before) + + // A new input object (the shape a replacement takes) breaks the settle. + if (child) child.input = { file_path: "src/replaced.ts" } + expect(nestingFingerprintOf(map)).not.toBe(before) + }) + it("changes when a nested part's state moves from pending to done", () => { const map = new Map([ [ diff --git a/src/renderer/features/agents/ui/agent-tool-utils.ts b/src/renderer/features/agents/ui/agent-tool-utils.ts index e36ed76e..100bfae3 100644 --- a/src/renderer/features/agents/ui/agent-tool-utils.ts +++ b/src/renderer/features/agents/ui/agent-tool-utils.ts @@ -7,7 +7,7 @@ * and compare cached values, not object references. */ -import { getToolLifecycleState, type ToolPartLike } from "./agent-tool-state" +import { getToolLifecycleState, isTerminalStateString, type ToolPartLike } from "./agent-tool-state" // ============================================================================ // TOOL STATE CACHE @@ -28,6 +28,7 @@ export function clearToolStateCachesByToolCallIds(toolCallIds: string[]) { for (const toolCallId of toolCallIds) { toolStateCache.delete(toolCallId) askUserStateCache.delete(toolCallId) + fingerprintSegmentCache.delete(toolCallId) } } @@ -141,6 +142,20 @@ export function areToolPropsEqual( export type NestedToolsLookup = (toolCallId: string) => ToolPartLike[] export type NestedToolsMapLike = ReadonlyMap +interface FingerprintSegment { + mapKey: string + state: unknown + input: unknown + output: unknown + segment: string +} + +/** + * Settled segments from the previous fingerprint, keyed by toolCallId — + * see `nestingFingerprintOf` for when a segment may be reused. + */ +const fingerprintSegmentCache = new Map() + /** * An immutable snapshot of the message-level nesting map, as one string. * @@ -153,20 +168,57 @@ export type NestedToolsMapLike = ReadonlyMap * same answer, because neither of them writes anything. * * Reads the same fields the tool-state snapshot records (`state`, `input`, - * `output`) plus the identity fields, deliberately without touching the cache. + * `output`) plus the identity fields, and never the tool-state cache. + * + * Cost is bounded like the outer memo's: a part whose state string is + * terminal and whose input/output references are unchanged is settled (the + * SDK does not reopen a completed part), so its segment is reused from a + * private cache in O(1) instead of re-stringified — a finished transcript + * costs reference compares, and only live parts pay for serialization on + * each render. The cache is written here, once per render in the message + * component; the row comparators only ever read the returned string, so the + * single-consumer property that motivated the fingerprint is untouched. */ export function nestingFingerprintOf(map: NestedToolsMapLike | undefined): string { if (!map || map.size === 0) return "" const segments: string[] = [] for (const [id, parts] of map) { for (const part of parts) { + const key = typeof part.toolCallId === "string" ? part.toolCallId : undefined + const prev = key !== undefined ? fingerprintSegmentCache.get(key) : undefined + if ( + prev !== undefined && + prev.mapKey === id && + prev.state === part.state && + prev.input === part.input && + prev.output === part.output && + isTerminalStateString(prev.state) + ) { + segments.push(prev.segment) // settled: same terminal state, same references + continue + } // JSON.stringify the tuple rather than join(): `state` is `unknown`, and // join() would fall back to Object's default stringification for any // non-string it meets, collapsing two different objects into one // "[object Object]" and hiding a change behind it. - segments.push( - JSON.stringify([id, part.type, part.toolCallId, part.state, part.input, part.output]), - ) + const segment = JSON.stringify([ + id, + part.type, + part.toolCallId, + part.state, + part.input, + part.output, + ]) + segments.push(segment) + if (key !== undefined) { + fingerprintSegmentCache.set(key, { + mapKey: id, + state: part.state, + input: part.input, + output: part.output, + segment, + }) + } } } return segments.join("\u0001") From d2ee05c7fa6980a5309ed56c002106ec6f41165d Mon Sep 17 00:00:00 2001 From: Owie6789 <151057755+Owie6789@users.noreply.github.com> Date: Thu, 24 Sep 2026 22:11:35 +0000 Subject: [PATCH 56/60] Record round 9: S6845, the settled serializers, reset five MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The S6845 span/button reversal of round 4, the three round-9 perf reviews verified real against round 8's unbounded serialization, the settled-part rule that bounds both sites, and the fifth sandbox reset that took node_modules, bun, and the runtime-client dist. Evidence: Markdown only — the code research gate is exempt by AGENTS.md; the code commits carry the full battery. Co-authored-by: arena-agent <297053741+arena-agent@users.noreply.github.com> --- ...2026-09-23-review-and-sonar-remediation.md | 47 +++++++++++++++++++ 1 file changed, 47 insertions(+) diff --git a/.dump/app/decisions/2026-09-23-review-and-sonar-remediation.md b/.dump/app/decisions/2026-09-23-review-and-sonar-remediation.md index 9c2fee03..18639fb4 100644 --- a/.dump/app/decisions/2026-09-23-review-and-sonar-remediation.md +++ b/.dump/app/decisions/2026-09-23-review-and-sonar-remediation.md @@ -723,3 +723,50 @@ pull_request-triggered run failed its vitest step after 30s. Its merge ref (`git diff HEAD origin/pr-69-merge` is empty), log storage is unreachable from this sandbox, and `gh run rerun` refuses — so the failure reads as environmental and the re-triggered run is the arbiter. + +## Round 9 — S6845 fixed, and the round-8 fix's own price + +Sonar's leak period closed on one issue: **S6845**, the bare `tabIndex="0"` +on the tool-call header subtitle (`agent-tool-call.tsx:35`) — every subtitle +was a tab stop, plain-text ones included, with nothing for a keyboard user to +reach. The fix reverses the round-4 ruling that kept bare spans: that call +assumed a click handler existed to protect, and it rests on native buttons +never overriding appearance — Radix's `TooltipTrigger` default is a button, +which the snapshot change records. `920a934` renders a span only when the +subtitle has neither an `onClick` nor a tooltip, otherwise a native +`