From bfd7936dbe7c55e4be4d895507d14dbcb7b8a301 Mon Sep 17 00:00:00 2001 From: Epinephrine Date: Sun, 27 Sep 2026 04:51:04 +0000 Subject: [PATCH 1/2] fix(adapters): keep tool call reminting linear --- .../openai-chat/tool-call-id-remint.ts | 6 +++++- structure/providers-and-adapters.md | 2 +- .../openai-chat-tool-call-id-remint.test.ts | 18 ++++++++++++++++++ 3 files changed, 24 insertions(+), 2 deletions(-) diff --git a/src/adapters/openai-chat/tool-call-id-remint.ts b/src/adapters/openai-chat/tool-call-id-remint.ts index bddfb4cb6a2..9b8e1bbd1c9 100644 --- a/src/adapters/openai-chat/tool-call-id-remint.ts +++ b/src/adapters/openai-chat/tool-call-id-remint.ts @@ -17,6 +17,7 @@ import type { OcxMessage } from "../../types"; */ export function createToolCallIdReminter(reservedIds: Iterable): (rawId: string) => string { const occupied = new Set(reservedIds); + const nextSuffixByBase = new Map(); return rawId => { if (!occupied.has(rawId)) { occupied.add(rawId); @@ -25,13 +26,16 @@ export function createToolCallIdReminter(reservedIds: Iterable): (rawId: // A non-conforming source is sanitized, never dropped: the wire still needs an id, and the // occupied check below covers a sanitized form that now equals some other call's id. const base = isConformingToolCallId(rawId) ? rawId : rawId.replace(/[^a-zA-Z0-9_-]/g, "_"); - for (let n = 2; ; n++) { + // Resume after the last suffix considered for this effective base. Restarting at 2 made a + // response containing N copies of one id perform O(N²) occupied-set probes synchronously. + for (let n = nextSuffixByBase.get(base) ?? 2; ; n++) { // Hyphen, not underscore: an id that extends another id as `_` is parsed by // at least one client as a batch sub-call of ``, which pairs the second call's // result to the first call. A `-` suffix is in the same id family without that reading. const suffix = `-${n}`; const candidate = base.slice(0, Math.max(1, MAX_TOOL_CALL_ID_LENGTH - suffix.length)) + suffix; if (!occupied.has(candidate)) { + nextSuffixByBase.set(base, n + 1); occupied.add(candidate); return candidate; } diff --git a/structure/providers-and-adapters.md b/structure/providers-and-adapters.md index ad719ac37ec..e12fff74eab 100644 --- a/structure/providers-and-adapters.md +++ b/structure/providers-and-adapters.md @@ -123,7 +123,7 @@ rewrite rules and the routed-id settlement. | `src/adapters/openai-chat.ts`, `src/adapters/openai-chat/` | OpenAI-compatible Chat Completions bridge, split into leaves (`wire.ts`, `messages.ts`, `response-events.ts`, `passthrough.ts`, `parallel-tool-calls.ts`, `reasoning-wire.ts`, `serialized-tool-call-content.ts`, `tool-call-validation.ts`, `tool-call-id-remint.ts`, `tool-schema.ts`, `errors.ts`). `parallel-tool-calls.ts` owns the `parallel_tool_calls` wire value for both the translated and native builders, so the three provider states — configured opt-out, configured opt-in, and the unset default that forwards only a caller's explicit `false` — cannot drift between them. `reasoning-wire.ts` applies explicit gateway-object and tool-bearing effort-omission declarations to both builders; absent declarations leave native raw forwarding unchanged. Its client delivery shapes in `src/chat/outbound.ts` and `src/server/chat-native-sse.ts` relay the upstream `service_tier` echo on non-stream, folded-stream, and synthesized-SSE bodies, never inventing the key when the upstream omits it. | | `src/adapters/anthropic.ts` | Anthropic Messages bridge. A `refusal` or `content_filter` stop reason yields an explicit `incomplete` event with `retryable: false` rather than `done` with that stopReason (#4312); `max_tokens` remains `done`. It is the wire that defines `tools[*].strict` and `tools[*].allowed_callers`, so a rebuilt declaration carries both: an explicit `strict: true` and any `allowed_callers` the caller declared. An absent `strict` stays absent, because the Messages inbound records it as `false` and a `false` on the wire would read as an opt-out nobody asked for. Anthropic Fast uses the native `anthropic-speed` FastWire: a set decision sends `speed: "fast"` with `fast-mode-2026-02-01` in one case-insensitively merged, deduplicated `anthropic-beta` header that preserves OAuth betas. Stream and buffered `usage.speed` echoes confirm fast or downgrade to standard; no echo leaves the request assumed. `tests/adapters/anthropic/anthropic-fast-speed.test.ts` pins the wire and echoes. Anthropic Fast is opt-in: the registry marks both Anthropic entries `fastOptIn`, and `src/providers/fast-opt-in.ts` (`providerFastSwitchOff`) keeps Fast off until `providers..fastEnabled` is `true`. An off switch is provider capability `false`, applied in the FastPolicy authority (`service-tier.ts`), `resolveModelPolicy`, and router registry enrichment, so no model-level Fast toggle, `--fast` row, or proxy-generated `speed` field is produced. Native Claude Messages passthrough still forwards a `speed` field the caller sends itself, outside the proxy Fast policy. `tests/adapters/anthropic/anthropic-fast-opt-in.test.ts` pins the default, the switch, and the management PATCH/GET. | | `src/adapters/google.ts` | Gemini bridge. The final wire compiler owns [endpoint-scoped tool-schema loss policy](providers/google.md#google-tool-schema-loss-reporting): compatible mode changes no request bytes, strict initial loss creates no physical send, and strict non-direct repair creates no changed repair send. A caller-declared strict tool selects `functionCallingConfig.mode: "VALIDATED"` in place of the absent-choice default; `NONE`, `ANY` and a forced-name choice are stronger constraints the caller asked for and are never overwritten. | -| `src/adapters/unique-tool-call-ids.ts` | Request-scoped tool-call-id uniqueness for every `openai-chat` provider. An upstream that mints an id from the call's position in its response repeats `call-0-0` on every turn; a Messages client has already paired that id, drops the duplicate, and is left with a call that has no result, so the turn reads as empty and the model re-issues it indefinitely. Only a **repeat** is rewritten — the first occurrence stays byte-identical, leaving prompt-cache keys, reasoning-replay lookups and already-unique upstreams untouched. The ids to avoid come from the caller's history, captured in `buildRequest` (the only point that sees it) and applied at emission, never at ingestion: ingestion matches streamed deltas against the id upstream sent, so rewriting there would strip a pending call of its identity mid-stream. A repeat takes a `-` suffix, never `_`, because `_` reads as a batch sub-call of ``. Covered by `tests/adapters/openai/openai-chat-tool-call-id-remint.test.ts`. | +| `src/adapters/unique-tool-call-ids.ts` | Request-scoped tool-call-id uniqueness for every `openai-chat` provider. An upstream that mints an id from the call's position in its response repeats `call-0-0` on every turn; a Messages client has already paired that id, drops the duplicate, and is left with a call that has no result, so the turn reads as empty and the model re-issues it indefinitely. Only a **repeat** is rewritten — the first occurrence stays byte-identical, leaving prompt-cache keys, reasoning-replay lookups and already-unique upstreams untouched. The ids to avoid come from the caller's history, captured in `buildRequest` (the only point that sees it) and applied at emission, never at ingestion: ingestion matches streamed deltas against the id upstream sent, so rewriting there would strip a pending call of its identity mid-stream. A repeat takes a `-` suffix, never `_`, because `_` reads as a batch sub-call of ``; collision search resumes per effective base so a response with repeated ids takes linear rather than quadratic occupied-set probes. Covered by `tests/adapters/openai/openai-chat-tool-call-id-remint.test.ts`. | | `src/adapters/declaration-carrier.ts`, `src/adapters/input-media-guard.ts` | Default-deny allowlists for constraints the normalized request carries but a wire may not be able to express: `tools[*].allowed_callers`, which fences a tool off from callers, and inline document bytes. Both are refused with a 400 at the single guard every registered adapter passes through, rather than left to each adapter, because an adapter that never learned about the carrier rebuilds without it and answers normally. `allowed_callers` reaches the `anthropic` wire; document bytes reach `anthropic`, `openai-chat` and `google`; the `openai-responses` wire is exempt from the whole guard because it forwards the original body. Adding an `AdapterWire` member makes the omission visible in these lists instead of at a customer's upstream. The unrestricted `["direct"]` caller default is not a restriction. | | `src/adapters/azure.ts` | Azure OpenAI bridge. | | `src/adapters/cursor.ts`, `src/adapters/cursor/` | Cursor protobuf transport: discovery, request builder, event decoding, MCP, thread continuity, native-exec policy. | diff --git a/tests/adapters/openai/openai-chat-tool-call-id-remint.test.ts b/tests/adapters/openai/openai-chat-tool-call-id-remint.test.ts index c638fd86127..5963e2ec15d 100644 --- a/tests/adapters/openai/openai-chat-tool-call-id-remint.test.ts +++ b/tests/adapters/openai/openai-chat-tool-call-id-remint.test.ts @@ -49,6 +49,24 @@ describe("createToolCallIdReminter", () => { expect(new Set(ids).size).toBe(3); }); + test("keeps repeated-id collision searches linear", () => { + const remint = createToolCallIdReminter([]); + const originalHas = Set.prototype.has; + let probes = 0; + Set.prototype.has = function (value) { + probes++; + return originalHas.call(this, value); + }; + + try { + for (let index = 0; index < 10_000; index++) remint("call-0-0"); + } finally { + Set.prototype.has = originalHas; + } + + expect(probes).toBeLessThan(20_001); + }); + test("skips a suffix the reserved set already occupies", () => { const remint = createToolCallIdReminter(["call-0-0", "call-0-0-2"]); From ccfb8e63e1bba96642db1477144136369da366b1 Mon Sep 17 00:00:00 2001 From: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Sun, 27 Sep 2026 05:35:46 +0000 Subject: [PATCH 2/2] fix(adapters): key remint cursors by truncated collision domain MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The suffix cursor was keyed by the full sanitized base, but a candidate keeps at most MAX_TOOL_CALL_ID_LENGTH - 2 characters of it — longer suffixes only truncate deeper. Distinct ids agreeing on that prefix therefore produce the same candidate sequence while each restarting the search at -2, so M prefix-sharing ids emitted twice still cost ~M^2/2 occupied-set probes. Key the cursor by the post-truncation prefix — the actual collision domain — so all ids sharing it resume the same search. The occupied-set check still decides acceptance, keeping emitted ids unique and within the 64-char bound. Regression test: 1,000 conforming 64-char ids plus 500 overlength non-conforming ids, all sharing the first 62 characters and each emitted twice, stay under 10k probes (was ~1.13M before). Co-Authored-By: Epinephrine --- .../openai-chat/tool-call-id-remint.ts | 16 +++++--- structure/providers-and-adapters.md | 2 +- .../openai-chat-tool-call-id-remint.test.ts | 38 +++++++++++++++++++ 3 files changed, 49 insertions(+), 7 deletions(-) diff --git a/src/adapters/openai-chat/tool-call-id-remint.ts b/src/adapters/openai-chat/tool-call-id-remint.ts index 9b8e1bbd1c9..ffae1a1d83a 100644 --- a/src/adapters/openai-chat/tool-call-id-remint.ts +++ b/src/adapters/openai-chat/tool-call-id-remint.ts @@ -17,7 +17,7 @@ import type { OcxMessage } from "../../types"; */ export function createToolCallIdReminter(reservedIds: Iterable): (rawId: string) => string { const occupied = new Set(reservedIds); - const nextSuffixByBase = new Map(); + const nextSuffixByPrefix = new Map(); return rawId => { if (!occupied.has(rawId)) { occupied.add(rawId); @@ -26,16 +26,20 @@ export function createToolCallIdReminter(reservedIds: Iterable): (rawId: // A non-conforming source is sanitized, never dropped: the wire still needs an id, and the // occupied check below covers a sanitized form that now equals some other call's id. const base = isConformingToolCallId(rawId) ? rawId : rawId.replace(/[^a-zA-Z0-9_-]/g, "_"); - // Resume after the last suffix considered for this effective base. Restarting at 2 made a - // response containing N copies of one id perform O(N²) occupied-set probes synchronously. - for (let n = nextSuffixByBase.get(base) ?? 2; ; n++) { + // A candidate keeps at most `MAX - 2` characters of the base (longer suffixes truncate + // deeper), so ids agreeing on that prefix compete for the same candidate sequence — the + // prefix is the collision domain. The resume cursor has to live on the domain, not the full + // base: distinct ids sharing it would otherwise each restart at -2 and re-run every probe an + // earlier base already made, which is quadratic in the number of such ids. + const prefix = base.slice(0, MAX_TOOL_CALL_ID_LENGTH - 2); + for (let n = nextSuffixByPrefix.get(prefix) ?? 2; ; n++) { // Hyphen, not underscore: an id that extends another id as `_` is parsed by // at least one client as a batch sub-call of ``, which pairs the second call's // result to the first call. A `-` suffix is in the same id family without that reading. const suffix = `-${n}`; - const candidate = base.slice(0, Math.max(1, MAX_TOOL_CALL_ID_LENGTH - suffix.length)) + suffix; + const candidate = prefix.slice(0, Math.max(1, MAX_TOOL_CALL_ID_LENGTH - suffix.length)) + suffix; if (!occupied.has(candidate)) { - nextSuffixByBase.set(base, n + 1); + nextSuffixByPrefix.set(prefix, n + 1); occupied.add(candidate); return candidate; } diff --git a/structure/providers-and-adapters.md b/structure/providers-and-adapters.md index e12fff74eab..d793152ca01 100644 --- a/structure/providers-and-adapters.md +++ b/structure/providers-and-adapters.md @@ -123,7 +123,7 @@ rewrite rules and the routed-id settlement. | `src/adapters/openai-chat.ts`, `src/adapters/openai-chat/` | OpenAI-compatible Chat Completions bridge, split into leaves (`wire.ts`, `messages.ts`, `response-events.ts`, `passthrough.ts`, `parallel-tool-calls.ts`, `reasoning-wire.ts`, `serialized-tool-call-content.ts`, `tool-call-validation.ts`, `tool-call-id-remint.ts`, `tool-schema.ts`, `errors.ts`). `parallel-tool-calls.ts` owns the `parallel_tool_calls` wire value for both the translated and native builders, so the three provider states — configured opt-out, configured opt-in, and the unset default that forwards only a caller's explicit `false` — cannot drift between them. `reasoning-wire.ts` applies explicit gateway-object and tool-bearing effort-omission declarations to both builders; absent declarations leave native raw forwarding unchanged. Its client delivery shapes in `src/chat/outbound.ts` and `src/server/chat-native-sse.ts` relay the upstream `service_tier` echo on non-stream, folded-stream, and synthesized-SSE bodies, never inventing the key when the upstream omits it. | | `src/adapters/anthropic.ts` | Anthropic Messages bridge. A `refusal` or `content_filter` stop reason yields an explicit `incomplete` event with `retryable: false` rather than `done` with that stopReason (#4312); `max_tokens` remains `done`. It is the wire that defines `tools[*].strict` and `tools[*].allowed_callers`, so a rebuilt declaration carries both: an explicit `strict: true` and any `allowed_callers` the caller declared. An absent `strict` stays absent, because the Messages inbound records it as `false` and a `false` on the wire would read as an opt-out nobody asked for. Anthropic Fast uses the native `anthropic-speed` FastWire: a set decision sends `speed: "fast"` with `fast-mode-2026-02-01` in one case-insensitively merged, deduplicated `anthropic-beta` header that preserves OAuth betas. Stream and buffered `usage.speed` echoes confirm fast or downgrade to standard; no echo leaves the request assumed. `tests/adapters/anthropic/anthropic-fast-speed.test.ts` pins the wire and echoes. Anthropic Fast is opt-in: the registry marks both Anthropic entries `fastOptIn`, and `src/providers/fast-opt-in.ts` (`providerFastSwitchOff`) keeps Fast off until `providers..fastEnabled` is `true`. An off switch is provider capability `false`, applied in the FastPolicy authority (`service-tier.ts`), `resolveModelPolicy`, and router registry enrichment, so no model-level Fast toggle, `--fast` row, or proxy-generated `speed` field is produced. Native Claude Messages passthrough still forwards a `speed` field the caller sends itself, outside the proxy Fast policy. `tests/adapters/anthropic/anthropic-fast-opt-in.test.ts` pins the default, the switch, and the management PATCH/GET. | | `src/adapters/google.ts` | Gemini bridge. The final wire compiler owns [endpoint-scoped tool-schema loss policy](providers/google.md#google-tool-schema-loss-reporting): compatible mode changes no request bytes, strict initial loss creates no physical send, and strict non-direct repair creates no changed repair send. A caller-declared strict tool selects `functionCallingConfig.mode: "VALIDATED"` in place of the absent-choice default; `NONE`, `ANY` and a forced-name choice are stronger constraints the caller asked for and are never overwritten. | -| `src/adapters/unique-tool-call-ids.ts` | Request-scoped tool-call-id uniqueness for every `openai-chat` provider. An upstream that mints an id from the call's position in its response repeats `call-0-0` on every turn; a Messages client has already paired that id, drops the duplicate, and is left with a call that has no result, so the turn reads as empty and the model re-issues it indefinitely. Only a **repeat** is rewritten — the first occurrence stays byte-identical, leaving prompt-cache keys, reasoning-replay lookups and already-unique upstreams untouched. The ids to avoid come from the caller's history, captured in `buildRequest` (the only point that sees it) and applied at emission, never at ingestion: ingestion matches streamed deltas against the id upstream sent, so rewriting there would strip a pending call of its identity mid-stream. A repeat takes a `-` suffix, never `_`, because `_` reads as a batch sub-call of ``; collision search resumes per effective base so a response with repeated ids takes linear rather than quadratic occupied-set probes. Covered by `tests/adapters/openai/openai-chat-tool-call-id-remint.test.ts`. | +| `src/adapters/unique-tool-call-ids.ts` | Request-scoped tool-call-id uniqueness for every `openai-chat` provider. An upstream that mints an id from the call's position in its response repeats `call-0-0` on every turn; a Messages client has already paired that id, drops the duplicate, and is left with a call that has no result, so the turn reads as empty and the model re-issues it indefinitely. Only a **repeat** is rewritten — the first occurrence stays byte-identical, leaving prompt-cache keys, reasoning-replay lookups and already-unique upstreams untouched. The ids to avoid come from the caller's history, captured in `buildRequest` (the only point that sees it) and applied at emission, never at ingestion: ingestion matches streamed deltas against the id upstream sent, so rewriting there would strip a pending call of its identity mid-stream. A repeat takes a `-` suffix, never `_`, because `_` reads as a batch sub-call of ``; collision search resumes per truncated-prefix collision domain so a response with repeated or prefix-sharing ids takes linear rather than quadratic occupied-set probes. Covered by `tests/adapters/openai/openai-chat-tool-call-id-remint.test.ts`. | | `src/adapters/declaration-carrier.ts`, `src/adapters/input-media-guard.ts` | Default-deny allowlists for constraints the normalized request carries but a wire may not be able to express: `tools[*].allowed_callers`, which fences a tool off from callers, and inline document bytes. Both are refused with a 400 at the single guard every registered adapter passes through, rather than left to each adapter, because an adapter that never learned about the carrier rebuilds without it and answers normally. `allowed_callers` reaches the `anthropic` wire; document bytes reach `anthropic`, `openai-chat` and `google`; the `openai-responses` wire is exempt from the whole guard because it forwards the original body. Adding an `AdapterWire` member makes the omission visible in these lists instead of at a customer's upstream. The unrestricted `["direct"]` caller default is not a restriction. | | `src/adapters/azure.ts` | Azure OpenAI bridge. | | `src/adapters/cursor.ts`, `src/adapters/cursor/` | Cursor protobuf transport: discovery, request builder, event decoding, MCP, thread continuity, native-exec policy. | diff --git a/tests/adapters/openai/openai-chat-tool-call-id-remint.test.ts b/tests/adapters/openai/openai-chat-tool-call-id-remint.test.ts index 5963e2ec15d..933caff99bb 100644 --- a/tests/adapters/openai/openai-chat-tool-call-id-remint.test.ts +++ b/tests/adapters/openai/openai-chat-tool-call-id-remint.test.ts @@ -67,6 +67,44 @@ describe("createToolCallIdReminter", () => { expect(probes).toBeLessThan(20_001); }); + test("distinct ids sharing a truncated prefix share one collision domain", () => { + // A candidate keeps at most MAX-2 characters of the base, so ids agreeing on those characters + // truncate to the same candidate sequence. A per-base cursor restarts each of them at -2: + // M such ids emitted twice cost ~M²/2 probes. The cursor has to live on the shared domain. + const remint = createToolCallIdReminter([]); + const originalHas = Set.prototype.has; + let probes = 0; + Set.prototype.has = function (value) { + probes++; + return originalHas.call(this, value); + }; + + // Conforming 64-char ids and overlength non-conforming ones, all agreeing on `prefix`. + const prefix = "p".repeat(MAX_TOOL_CALL_ID_LENGTH - 2); + const rawIds = [ + ...Array.from({ length: 1_000 }, (_, i) => prefix + i.toString(36).padStart(2, "0")), + ...Array.from({ length: 500 }, (_, i) => `${prefix}:${i}`), + ]; + const emitted: string[] = []; + try { + for (let round = 0; round < 2; round++) { + for (const rawId of rawIds) emitted.push(remint(rawId)); + } + } finally { + Set.prototype.has = originalHas; + } + + // First occurrences pass through byte-identical; each repeat draws the domain's next suffix. + expect(emitted.slice(0, rawIds.length)).toEqual(rawIds); + for (const [index, id] of emitted.slice(rawIds.length).entries()) { + const suffix = `-${index + 2}`; + expect(id).toBe(prefix.slice(0, MAX_TOOL_CALL_ID_LENGTH - suffix.length) + suffix); + expect(id.length).toBeLessThanOrEqual(MAX_TOOL_CALL_ID_LENGTH); + } + expect(new Set(emitted).size).toBe(emitted.length); + expect(probes).toBeLessThan(10_000); + }); + test("skips a suffix the reserved set already occupies", () => { const remint = createToolCallIdReminter(["call-0-0", "call-0-0-2"]);