From b8f20f120d0eb3797d279b96d0148329309539a7 Mon Sep 17 00:00:00 2001 From: vyrnsynx <153433026+vyrnsynx@users.noreply.github.com> Date: Fri, 11 Sep 2026 18:27:09 +0000 Subject: [PATCH] fix(workers-ai): map binding error 3021 to retryable 429 Workers AI throws "3021: rate limiting: inference request per min rate reached" from the binding. Without a status mapping, normalizeBindingError built a non-retryable APICallError and AI SDK retries stopped. Treat 3021 like 3036/3040 (HTTP 429). Fixes #657 --- .changeset/workers-ai-3021-retryable.md | 6 ++++++ packages/gateway-core/src/workers-ai-errors.ts | 1 + .../test/workersai-error.test.ts | 13 +++++++++++++ 3 files changed, 20 insertions(+) create mode 100644 .changeset/workers-ai-3021-retryable.md diff --git a/.changeset/workers-ai-3021-retryable.md b/.changeset/workers-ai-3021-retryable.md new file mode 100644 index 0000000000..7eb4fb10c1 --- /dev/null +++ b/.changeset/workers-ai-3021-retryable.md @@ -0,0 +1,6 @@ +--- +"@cloudflare/gateway-core": patch +"workers-ai-provider": patch +--- + +Map Workers AI binding error `3021` (inference per-minute rate limit) to HTTP 429 so SDK retries treat it as retryable, matching gateway responses. diff --git a/packages/gateway-core/src/workers-ai-errors.ts b/packages/gateway-core/src/workers-ai-errors.ts index 6902aa12e4..2d4cd06f5b 100644 --- a/packages/gateway-core/src/workers-ai-errors.ts +++ b/packages/gateway-core/src/workers-ai-errors.ts @@ -37,6 +37,7 @@ export const WORKERS_AI_ERROR_CODE_TO_STATUS: Record = { 3008: 408, // Aborted 3036: 429, // Account limited (daily free allocation used up) 3040: 429, // Out of capacity (no data center to forward to) + 3021: 429, // Inference request per-minute rate limit reached }; /** Read a human-readable message from any thrown value (Error, DOMException, plain object, string). */ diff --git a/packages/workers-ai-provider/test/workersai-error.test.ts b/packages/workers-ai-provider/test/workersai-error.test.ts index 33cc308ab4..357cea0237 100644 --- a/packages/workers-ai-provider/test/workersai-error.test.ts +++ b/packages/workers-ai-provider/test/workersai-error.test.ts @@ -60,6 +60,18 @@ describe("normalizeBindingError", () => { expect(api.data).toEqual({ workersAIErrorCode: 3040 }); }); + it("maps an inference RPM limit (3021) binding error to a retryable 429 APICallError", () => { + const err = normalizeBindingError( + new Error("3021: rate limiting: inference request per min rate reached"), + ctx, + ); + expect(APICallError.isInstance(err)).toBe(true); + const api = err as APICallError; + expect(api.statusCode).toBe(429); + expect(api.isRetryable).toBe(true); + expect(api.data).toEqual({ workersAIErrorCode: 3021 }); + }); + it("maps a client error (5007) to a non-retryable 400 APICallError", () => { const err = normalizeBindingError(new Error("5007: No such model"), ctx) as APICallError; expect(APICallError.isInstance(err)).toBe(true); @@ -139,6 +151,7 @@ describe("WORKERS_AI_ERROR_CODE_TO_STATUS", () => { it("maps the documented transient codes to retryable statuses", () => { expect(WORKERS_AI_ERROR_CODE_TO_STATUS[3040]).toBe(429); expect(WORKERS_AI_ERROR_CODE_TO_STATUS[3036]).toBe(429); + expect(WORKERS_AI_ERROR_CODE_TO_STATUS[3021]).toBe(429); expect(WORKERS_AI_ERROR_CODE_TO_STATUS[3007]).toBe(408); }); });