From 214704f5295c8b5578ec77965c8c0b6f9dabe7d4 Mon Sep 17 00:00:00 2001 From: Yujun Liu Date: Wed, 16 Sep 2026 21:55:34 -0700 Subject: [PATCH] fix: stop rewriting Responses history by default MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Every /v1/responses request was trimmed pre-flight against a byte ceiling derived from max_prompt_tokens, rewriting old input items into placeholders before Copilot ever saw the payload. Two problems with that: The ratio was wrong. CHARS_PER_TOKEN_ESTIMATE = 3.5 came from context-manager.ts, where it measures *unescaped content length*; here it measures serialized JSON. Measured against multi-turn payloads built from this repo's own source, the real figure is 4.4 bytes per content token (3.7 per token of the JSON itself), so trimming engaged at roughly 77% of a model's real window — destroying context Copilot would have accepted. The rewrite broke prompt caching. dropOldInputItems replaces from the oldest item forward, which is precisely the stable prefix a cache is keyed on. In a simulated growing session the reusable prefix collapsed from 277 items to 1 on the turn trimming first engaged, and was re-poisoned each time the boundary advanced. Default behavior is now to forward the client's history verbatim and let Copilot enforce its own limits, so a rejection is upstream's real answer instead of a local guess. The ~5 MB Azure Front Door transport cliff is still enforced pre-flight — that one is a verified wire limit, not an estimate. Token-derived trimming remains available behind --responses-context-trim, with the ratio corrected to 4.3. Also repairs the reactive path it now depends on: - Context-overflow errors returned upstream's reported token counts instead of a local bytes/4 estimate. A genuine "300000 > 272000" was being rewritten as "16 + 1000 > 272000", and Claude Code sizes its compaction pass off that gap. The original upstream body is preserved in error.upstream_error. - isContextOverflow treated any 5xx on a >2 MB payload as context overflow, so a transient 503 made clients compact away history to work around what was really a Copilot outage. Now restricted to explicit overflow markers and plain-500 dead-zone hangs, with transient-outage phrasings excluded. Ratio, cache-prefix collapse, and tokenizer cost were measured independently by two models reaching the same conclusions. Fixes #5 Co-Authored-By: Claude Opus 5 (1M context) --- src/lib/state.ts | 8 + src/services/copilot/create-responses.ts | 158 +++++++++++++++-- src/start.ts | 14 ++ tests/create-responses-context.test.ts | 212 +++++++++++++++++++++++ tests/create-responses.test.ts | 25 ++- 5 files changed, 394 insertions(+), 23 deletions(-) create mode 100644 tests/create-responses-context.test.ts diff --git a/src/lib/state.ts b/src/lib/state.ts index 42d1224..d325ad9 100644 --- a/src/lib/state.ts +++ b/src/lib/state.ts @@ -13,6 +13,13 @@ export interface State { showToken: boolean localApiKeys: Array + /** + * Opt in to lossy pre-flight trimming of /responses history. Off by default: + * the proxy forwards the client's conversation verbatim and lets Copilot + * enforce its own limits, rather than rewriting history on a local estimate. + */ + responsesContextTrim: boolean + // Rate limiting configuration rateLimitSeconds?: number lastRequestTimestamp?: number @@ -24,4 +31,5 @@ export const state: State = { rateLimitWait: false, showToken: false, localApiKeys: [], + responsesContextTrim: false, } diff --git a/src/services/copilot/create-responses.ts b/src/services/copilot/create-responses.ts index 8eb1596..7e60d97 100644 --- a/src/services/copilot/create-responses.ts +++ b/src/services/copilot/create-responses.ts @@ -23,8 +23,30 @@ import { stripEncryptedOutputParts, } from "~/services/copilot/encrypted-output-recovery" +/** + * Hard transport ceiling. Azure Front Door rejects bodies above ~5.4 MB before + * Copilot ever sees them, so trimming to stay under this is not a guess about + * token accounting — it is a verified wire limit. Enforced unconditionally. + */ const MAX_RESPONSES_PAYLOAD_BYTES = 5_000_000 -const CHARS_PER_TOKEN_ESTIMATE = 3.5 + +/** + * Bytes per token for Responses payloads, used only when opt-in token-derived + * trimming is enabled. + * + * Measured against multi-turn payloads built from this repository's own source + * (user messages, function_call items, large function_call_output bodies, + * assistant replies): 4.4 JSON bytes per content token, 3.7 per token of the + * serialized JSON itself. The previous value of 3.5 was inherited from + * context-manager.ts, where it measures *unescaped content length* rather than + * serialized JSON — copying the constant silently changed its basis and made + * the derived ceiling ~26% too aggressive, engaging trimming at roughly 77% of + * a model's real window. + * + * 4.3 is the conservative end of what was measured. Neither figure is Copilot's + * own prompt-token count, which is why this no longer gates the default path. + */ +const CHARS_PER_TOKEN_ESTIMATE = 4.3 const TOKEN_RESERVE = 8_000 const IMAGE_STRIPPED_PLACEHOLDER = "[image removed to stay under upstream payload limit]" @@ -92,14 +114,22 @@ export async function createResponses( consola.error(`Request payload size: ${result.body.length} bytes`) if (isContextOverflow(response, errorBody, result.body.length)) { - const estimatedTokens = Math.ceil(result.body.length / 4) const modelCaps = state.models?.data.find((m) => m.id === payload.model) ?.capabilities.limits - const modelLimit = getModelPromptLimit(payload.model, modelCaps) const maxOutputTokens = payload.max_output_tokens ?? 0 + // Prefer the numbers Copilot reported over a local estimate. Claude Code + // uses the gap between them to size its compaction pass, so substituting + // a bytes/4 guess for upstream's real count makes it compact by the wrong + // amount. Fall back to the estimate only when upstream gave us nothing. + const upstream = parseUpstreamTokenCounts(errorBody) + const estimatedTokens = upstream.used ?? Math.ceil(result.body.length / 4) + const modelLimit = + upstream.limit ?? getModelPromptLimit(payload.model, modelCaps) + const source = upstream.used ? "upstream-reported" : "estimated" + consola.warn( - `Responses context overflow -> returning 400 prompt-too-long (~${estimatedTokens} + ${maxOutputTokens} > ${modelLimit})`, + `Responses context overflow -> returning 400 prompt-too-long (${source}: ~${estimatedTokens} + ${maxOutputTokens} > ${modelLimit})`, ) throw new HTTPError( @@ -110,6 +140,9 @@ export async function createResponses( error: { type: "invalid_request_error", message: `prompt is too long: input length and \`max_tokens\` exceed context limit: ${estimatedTokens} + ${maxOutputTokens} > ${modelLimit} tokens`, + // Preserve what upstream actually said, so a client that wants + // to diagnose rather than just compact is not flying blind. + upstream_error: errorBody.slice(0, 2_000), }, }), { @@ -277,7 +310,22 @@ function fitResponsesPayload( return current } +/** + * Byte ceiling for the forwarded payload. + * + * By default this is the transport ceiling alone: the proxy forwards the + * client's history verbatim and lets Copilot enforce its own token limits, so + * a rejection is upstream's real answer rather than a local guess. Rewriting + * history pre-emptively both discards context Copilot would have accepted and + * breaks prompt-cache continuity, because `dropOldInputItems` rewrites from the + * oldest item forward — exactly the stable prefix a cache is keyed on. + * + * With `responsesContextTrim` enabled, the smaller token-derived ceiling is + * applied too, restoring the previous lossy behavior for anyone who wants it. + */ function computeResponsesPayloadCeiling(modelId: string): number { + if (!state.responsesContextTrim) return MAX_RESPONSES_PAYLOAD_BYTES + const limits = state.models?.data.find((m) => m.id === modelId)?.capabilities .limits const maxPromptTokens = getKnownModelPromptLimit(modelId, limits) @@ -350,6 +398,20 @@ function replaceImageWithPlaceholder( return { ...payload, input } } +/** + * Replace the oldest history with placeholders until the payload fits. + * + * Walks oldest → newest: the trailing items carry the active request, so + * anything dropped has to come off the front. This is unavoidably hostile to + * prompt caching — a cache is keyed on a stable prefix, and rewriting item 0 + * invalidates everything after it — but the alternative (dropping recent turns + * to preserve the cacheable prefix) throws away exactly the context the model + * needs most. The real mitigation is not running this at all by default; see + * computeResponsesPayloadCeiling. + * + * The two most recent droppable items are always preserved, as is anything + * system/developer. + */ function dropOldInputItems( payload: ResponsesApiRequest, ceiling: number, @@ -622,23 +684,89 @@ function isRecord(value: unknown): value is Record { return typeof value === "object" && value !== null } +/** + * Error markers that unambiguously mean "the prompt did not fit". These are + * upstream saying so explicitly, so they are safe to act on regardless of + * payload size. + */ +const CONTEXT_OVERFLOW_PATTERNS = [ + /request entity too large/i, + /exceeds the limit of \d+/i, + /context_length_exceeded/i, + /model_max_prompt_tokens_exceeded/i, + /payload too large/i, + /maximum context length/i, + /prompt is too long/i, +] + +/** + * Markers of a transient upstream outage. These can coincide with a large + * payload without the payload being the cause, so they must never be read as + * context overflow — doing so makes a client compact away history to work + * around what is really a Copilot hiccup. + */ +const TRANSIENT_OUTAGE_PATTERNS = [ + /service unavailable/i, + /bad gateway/i, + /temporarily unavailable/i, + /upstream connect error/i, + /overloaded/i, + /try again later/i, +] + function isContextOverflow( response: Response, errorBody: string, bodyLength: number, ): boolean { - return ( - response.status === 413 - || /request entity too large/i.test(errorBody) - || /exceeds the limit of \d+/i.test(errorBody) - || /context_length_exceeded/i.test(errorBody) - || /operation timed out/i.test(errorBody) - || /payload too large/i.test(errorBody) - || /maximum context length/i.test(errorBody) - || (response.status >= 500 - && response.status < 600 - && bodyLength > 2_000_000) + if (response.status === 413) return true + if (CONTEXT_OVERFLOW_PATTERNS.some((pattern) => pattern.test(errorBody))) { + return true + } + if (TRANSIENT_OUTAGE_PATTERNS.some((pattern) => pattern.test(errorBody))) { + return false + } + + // Copilot's backend hangs rather than cleanly rejecting in the ~2.5-5.3 MB + // dead zone, which Bun surfaces as a 500 or an upstream timeout. Restrict + // that inference to plain 500s: 502/503/504 are gateway-level outages that + // say nothing about whether the prompt fit. + const isDeadZoneHang = + (response.status === 500 || /operation timed out/i.test(errorBody)) + && bodyLength > 2_000_000 + + return isDeadZoneHang +} + +interface UpstreamTokenCounts { + limit?: number + used?: number +} + +/** + * Pull the token counts Copilot actually reported out of an error body, so the + * client sees upstream's numbers rather than a local guess. Covers the phrasings + * Copilot has emitted: "maximum context length is N tokens ... resulted in M + * tokens", "exceeds the limit of N", and JSON bodies carrying explicit fields. + */ +function parseUpstreamTokenCounts(errorBody: string): UpstreamTokenCounts { + const counts: UpstreamTokenCounts = {} + + const maxContext = /maximum context length is (\d+) tokens/i.exec(errorBody) + if (maxContext) counts.limit = Number(maxContext[1]) + + const resulted = /(?:resulted in|you requested|requested) (\d+) tokens/i.exec( + errorBody, ) + if (resulted) counts.used = Number(resulted[1]) + + const exceedsLimit = /exceeds the limit of (\d+)/i.exec(errorBody) + if (exceedsLimit) counts.limit ??= Number(exceedsLimit[1]) + + const promptTokens = /(\d+) prompt tokens/i.exec(errorBody) + if (promptTokens) counts.used ??= Number(promptTokens[1]) + + return counts } function sanitizeResponsesPayload( diff --git a/src/start.ts b/src/start.ts index f6e30e7..35de1bc 100644 --- a/src/start.ts +++ b/src/start.ts @@ -27,6 +27,7 @@ interface RunServerOptions { showToken: boolean proxyEnv: boolean apiKey?: string + responsesContextTrim: boolean } export async function runServer(options: RunServerOptions): Promise { @@ -49,6 +50,12 @@ export async function runServer(options: RunServerOptions): Promise { state.rateLimitWait = options.rateLimitWait state.showToken = options.showToken state.localApiKeys = parseLocalApiKeys(options.apiKey) + state.responsesContextTrim = options.responsesContextTrim + if (state.responsesContextTrim) { + consola.warn( + "Responses context trimming enabled — old history will be rewritten to fit an estimated budget, which degrades prompt caching", + ) + } if (state.localApiKeys.length > 0) { consola.info("Local API-key auth enabled") } @@ -203,6 +210,12 @@ export const start = defineCommand({ description: "Require one of these comma-separated local API keys for proxy routes", }, + "responses-context-trim": { + type: "boolean", + default: false, + description: + "Trim old /responses history to fit an estimated token budget. Lossy: rewrites conversation history and breaks prompt caching. Off by default — Copilot enforces its own limits", + }, }, run({ args }) { const rateLimitRaw = args["rate-limit"] @@ -221,6 +234,7 @@ export const start = defineCommand({ showToken: args["show-token"], proxyEnv: args["proxy-env"], apiKey: args["api-key"], + responsesContextTrim: args["responses-context-trim"], }) }, }) diff --git a/tests/create-responses-context.test.ts b/tests/create-responses-context.test.ts new file mode 100644 index 0000000..594c832 --- /dev/null +++ b/tests/create-responses-context.test.ts @@ -0,0 +1,212 @@ +import { afterEach, expect, mock, test } from "bun:test" + +import type { ResponsesApiRequest } from "~/routes/responses/types" + +import { state } from "~/lib/state" +import { createResponses } from "~/services/copilot/create-responses" + +/** + * Context handling for /responses. + * + * The proxy forwards client history verbatim by default and lets Copilot + * enforce its own token limits, so a rejection reflects upstream's real answer + * instead of a local estimate. Only the ~5 MB Azure Front Door transport cliff + * is enforced pre-flight, because that one is a verified wire limit. + */ + +state.copilotToken = "test-token" +state.vsCodeVersion = "1.0.0" +state.accountType = "individual" +state.models = { + object: "list", + data: [ + { + id: "gpt-5.5", + object: "model", + name: "GPT 5.5", + model_picker_enabled: true, + preview: false, + vendor: "openai", + version: "1", + capabilities: { + family: "gpt-5.5", + limits: { + max_context_window_tokens: 400_000, + max_output_tokens: 16_000, + max_prompt_tokens: 272_000, + }, + object: "model_capabilities", + supports: {}, + tokenizer: "o200k_base", + type: "chat", + }, + }, + ], +} + +afterEach(() => { + mock.restore() + state.responsesContextTrim = false +}) + +function bodyToString(body: unknown): string { + if (typeof body !== "string") { + throw new TypeError("expected fetch body to be a string") + } + return body +} + +test("forwards oversized history unchanged by default", async () => { + // Well over the old token-derived ceiling (924_000) but under the 5 MB + // transport cliff: previously this was silently rewritten before Copilot + // ever saw it. The client's conversation must now reach upstream intact. + const payload: ResponsesApiRequest = { + model: "gpt-5.5", + instructions: "Keep the latest task context.", + input: Array.from({ length: 8 }, (_, index) => ({ + role: index % 2 === 0 ? "user" : "assistant", + content: `Turn ${index}\n${"x".repeat(180_000)}`, + })), + } + + const fetchMock = mock( + (_url: string, _opts: RequestInit) => + new Response(JSON.stringify({ id: "resp_untrimmed" }), { + status: 200, + headers: { "content-type": "application/json" }, + }), + ) + globalThis.fetch = fetchMock as unknown as typeof fetch + + const response = await createResponses(payload) + const sentBody = bodyToString(fetchMock.mock.calls[0][1].body) + const forwarded = JSON.parse(sentBody) as ResponsesApiRequest + + expect(response.status).toBe(200) + expect(sentBody.length).toBeGreaterThan(1_135_200) + expect(JSON.stringify(forwarded.input)).not.toContain( + "older response input omitted", + ) + // Every turn survives, including the oldest — the cacheable prefix is intact. + for (let index = 0; index < 8; index++) { + expect(JSON.stringify(forwarded.input)).toContain(`Turn ${index}`) + } +}) + +test("still enforces the hard transport ceiling without the flag", async () => { + const payload: ResponsesApiRequest = { + model: "gpt-5.5", + input: Array.from({ length: 12 }, (_, index) => ({ + role: "user", + content: `Huge ${index}\n${"x".repeat(600_000)}`, + })), + } + + const fetchMock = mock( + (_url: string, _opts: RequestInit) => + new Response(JSON.stringify({ id: "resp_capped" }), { + status: 200, + headers: { "content-type": "application/json" }, + }), + ) + globalThis.fetch = fetchMock as unknown as typeof fetch + + await createResponses(payload) + const sentBody = bodyToString(fetchMock.mock.calls[0][1].body) + + // Azure Front Door rejects above ~5.4 MB regardless of tokens, so this + // ceiling is a verified wire limit rather than an estimate. + expect(sentBody.length).toBeLessThanOrEqual(5_000_000) +}) + +test("surfaces upstream token counts instead of a local estimate", async () => { + const payload: ResponsesApiRequest = { + model: "gpt-5.5", + input: "hello", + max_output_tokens: 1_000, + } + + globalThis.fetch = mock( + () => + new Response( + JSON.stringify({ + error: { + code: "context_length_exceeded", + message: + "This model's maximum context length is 272000 tokens, however your messages resulted in 300000 tokens.", + }, + }), + { status: 400, headers: { "content-type": "application/json" } }, + ), + ) as unknown as typeof fetch + + const error = (await createResponses(payload).catch( + (caught: unknown) => caught, + )) as { response: Response } + + const body = (await error.response.json()) as { + error: { message: string; upstream_error?: string } + } + + expect(error.response.status).toBe(400) + // Upstream said 300000 > 272000; the proxy must not replace those with its + // own bytes/4 guess, which Claude Code would then size compaction against. + expect(body.error.message).toContain("300000") + expect(body.error.message).toContain("272000") + expect(body.error.upstream_error).toContain("context_length_exceeded") +}) + +test("does not treat an unrelated 503 on a large payload as context overflow", async () => { + const payload: ResponsesApiRequest = { + model: "gpt-5.5", + input: [{ role: "user", content: "x".repeat(2_500_000) }], + } + + globalThis.fetch = mock( + () => + new Response("Service Unavailable: upstream connect error", { + status: 503, + statusText: "Service Unavailable", + }), + ) as unknown as typeof fetch + + const error = (await createResponses(payload).catch( + (caught: unknown) => caught, + )) as { response: Response } + + // A transient outage that happens to coincide with a big payload must stay a + // 503. Reporting it as prompt-too-long makes the client compact away history + // to work around what is really a Copilot hiccup. + expect(error.response.status).toBe(503) + const body = await error.response.text() + expect(body).not.toContain("prompt is too long") +}) + +test("responsesContextTrim opt-in restores token-derived trimming", async () => { + state.responsesContextTrim = true + + const payload: ResponsesApiRequest = { + model: "gpt-5.5", + input: Array.from({ length: 8 }, (_, index) => ({ + role: index % 2 === 0 ? "user" : "assistant", + content: `Turn ${index}\n${"x".repeat(180_000)}`, + })), + } + + const fetchMock = mock( + (_url: string, _opts: RequestInit) => + new Response(JSON.stringify({ id: "resp_trimmed" }), { + status: 200, + headers: { "content-type": "application/json" }, + }), + ) + globalThis.fetch = fetchMock as unknown as typeof fetch + + await createResponses(payload) + const sentBody = bodyToString(fetchMock.mock.calls[0][1].body) + + expect(sentBody.length).toBeLessThanOrEqual(1_135_200) + expect(sentBody).toContain("older response input omitted") + // The active request is always preserved, whatever else is dropped. + expect(sentBody).toContain("Turn 7") +}) diff --git a/tests/create-responses.test.ts b/tests/create-responses.test.ts index 91d3856..deeec7b 100644 --- a/tests/create-responses.test.ts +++ b/tests/create-responses.test.ts @@ -37,6 +37,7 @@ state.models = { afterEach(() => { mock.restore() + state.responsesContextTrim = false }) function bodyToString(body: unknown): string { @@ -81,6 +82,8 @@ test("strips old Responses images when payload exceeds upstream byte limit", asy }) test("drops old Responses input history when payload exceeds model token budget", async () => { + state.responsesContextTrim = true + const payload: ResponsesApiRequest = { model: "gpt-5.5", instructions: "Keep the latest task context.", @@ -93,7 +96,7 @@ test("drops old Responses input history when payload exceeds model token budget" const fetchMock = mock((_url: string, opts: RequestInit) => { const body = bodyToString(opts.body) return new Response(JSON.stringify({ id: "resp_456" }), { - status: body.length > 924_000 ? 400 : 200, + status: body.length > 1_135_200 ? 400 : 200, headers: { "content-type": "application/json" }, }) }) @@ -104,7 +107,7 @@ test("drops old Responses input history when payload exceeds model token budget" const forwarded = JSON.parse(sentBody) as ResponsesApiRequest expect(response.status).toBe(200) - expect(sentBody.length).toBeLessThanOrEqual(924_000) + expect(sentBody.length).toBeLessThanOrEqual(1_135_200) expect(JSON.stringify(forwarded.input)).toContain( "older response input omitted to stay under context limit", ) @@ -112,6 +115,8 @@ test("drops old Responses input history when payload exceeds model token budget" }) test("drops unknown large Responses item fields from old history", async () => { + state.responsesContextTrim = true + const payload = { model: "gpt-5.5", input: [ @@ -135,7 +140,7 @@ test("drops unknown large Responses item fields from old history", async () => { const fetchMock = mock((_url: string, opts: RequestInit) => { const body = bodyToString(opts.body) return new Response(JSON.stringify({ id: "resp_789" }), { - status: body.length > 924_000 ? 413 : 200, + status: body.length > 1_135_200 ? 413 : 200, headers: { "content-type": "application/json" }, }) }) @@ -146,12 +151,14 @@ test("drops unknown large Responses item fields from old history", async () => { const forwarded = JSON.parse(sentBody) as ResponsesApiRequest expect(response.status).toBe(200) - expect(sentBody.length).toBeLessThanOrEqual(924_000) + expect(sentBody.length).toBeLessThanOrEqual(1_135_200) expect(JSON.stringify(forwarded.input)).not.toContain("read_image_batch") expect(JSON.stringify(forwarded.input)).toContain("continue") }) test("does not forward orphaned Responses function call outputs after fitting", async () => { + state.responsesContextTrim = true + const payload = { model: "gpt-5.5", input: [ @@ -163,7 +170,7 @@ test("does not forward orphaned Responses function call outputs after fitting", }, ...Array.from({ length: 4 }, (_, index) => ({ role: index % 2 === 0 ? "user" : "assistant", - content: `Old turn ${index}\n${"x".repeat(260_000)}`, + content: `Old turn ${index}\n${"x".repeat(340_000)}`, })), { type: "function_call_output", @@ -217,7 +224,7 @@ test("does not forward orphaned Responses function call outputs after fitting", const forwarded = JSON.parse(sentBody) as ResponsesApiRequest expect(response.status).toBe(200) - expect(sentBody.length).toBeLessThanOrEqual(924_000) + expect(sentBody.length).toBeLessThanOrEqual(1_135_200) expect(JSON.stringify(forwarded.input)).not.toContain( '"type":"function_call_output"', ) @@ -227,6 +234,8 @@ test("does not forward orphaned Responses function call outputs after fitting", }) test("does not forward orphaned Responses custom tool call outputs after fitting", async () => { + state.responsesContextTrim = true + const payload = { model: "gpt-5.5", input: [ @@ -238,7 +247,7 @@ test("does not forward orphaned Responses custom tool call outputs after fitting }, ...Array.from({ length: 4 }, (_, index) => ({ role: index % 2 === 0 ? "user" : "assistant", - content: `Old custom turn ${index}\n${"x".repeat(260_000)}`, + content: `Old custom turn ${index}\n${"x".repeat(340_000)}`, })), { type: "custom_tool_call_output", @@ -292,7 +301,7 @@ test("does not forward orphaned Responses custom tool call outputs after fitting const forwarded = JSON.parse(sentBody) as ResponsesApiRequest expect(response.status).toBe(200) - expect(sentBody.length).toBeLessThanOrEqual(924_000) + expect(sentBody.length).toBeLessThanOrEqual(1_135_200) expect(JSON.stringify(forwarded.input)).not.toContain( '"type":"custom_tool_call_output"', )