From 745ea995f39c8c50c9ed917ad33b7e30ee8e8b3e Mon Sep 17 00:00:00 2001 From: waldekmastykarz Date: Sat, 3 Oct 2026 20:18:09 +0200 Subject: [PATCH 1/2] Return rate_limit_exceeded instead of insufficient_quota from LanguageModelRateLimitingPlugin. Closes #1911 Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- .../BehaviorPluginsIntegrationTests.cs | 47 +++++++++++++++++++ .../LanguageModelRateLimitingPlugin.cs | 25 ++++++---- skills/dev-proxy/references/test-llm-apps.md | 9 ++-- 3 files changed, 69 insertions(+), 12 deletions(-) diff --git a/DevProxy.Integration.Tests/BehaviorPluginsIntegrationTests.cs b/DevProxy.Integration.Tests/BehaviorPluginsIntegrationTests.cs index be4772437..97d511a16 100644 --- a/DevProxy.Integration.Tests/BehaviorPluginsIntegrationTests.cs +++ b/DevProxy.Integration.Tests/BehaviorPluginsIntegrationTests.cs @@ -5,6 +5,8 @@ using System.Diagnostics; using System.Globalization; using System.Net; +using System.Text; +using System.Text.Json; using DevProxy.Abstractions.Plugins; using DevProxy.Abstractions.Proxy; using DevProxy.Plugins.Behavior; @@ -165,4 +167,49 @@ public async Task RateLimitingPlusRetryAfter_ThrottledResponseCarriesRetryAfter( out _), "Retry-After header should be an integer seconds value."); } + + [Fact] + public async Task LanguageModelRateLimiting_Throttle_ReturnsOpenAIRateLimitError() + { + await using var origin = await FakeOrigin.StartAsync(); + var urls = KestrelProxyHarness.BuildUrlsToWatch(origin.Host); + + var plugin = new LanguageModelRateLimitingPlugin( + SharedHttpClient, + NullLogger.Instance, + urls, + ProxyConfig, + PluginConfig.FromJson(""" + { "promptTokenLimit": 10, "completionTokenLimit": 100, "resetTimeWindowSeconds": 300 } + """)); + + await using var proxy = await KestrelProxyHarness.StartAsync( + origin.Host, [plugin]); + using var client = proxy.CreateHttpClient(); + + // /echo returns the request body, so the usage below is read back as the + // response's token usage. #1 drains the prompt token limit, #2 is throttled. + const string requestBody = """ + { + "model": "gpt-4o", + "messages": [ { "role": "user", "content": "hi" } ], + "usage": { "prompt_tokens": 10, "completion_tokens": 5, "total_tokens": 15 } + } + """; + using var firstContent = new StringContent(requestBody, Encoding.UTF8, "application/json"); + using var first = await client.PostAsync(new Uri($"http://{origin.Host}/echo"), firstContent); + using var secondContent = new StringContent(requestBody, Encoding.UTF8, "application/json"); + using var throttled = await client.PostAsync(new Uri($"http://{origin.Host}/echo"), secondContent); + + Assert.Equal(HttpStatusCode.OK, first.StatusCode); + Assert.Equal(HttpStatusCode.TooManyRequests, throttled.StatusCode); + Assert.True(throttled.Headers.Contains("retry-after")); + + using var json = JsonDocument.Parse(await throttled.Content.ReadAsStringAsync()); + var error = json.RootElement.GetProperty("error"); + Assert.Equal("rate_limit_exceeded", error.GetProperty("code").GetString()); + Assert.Equal("tokens", error.GetProperty("type").GetString()); + var message = error.GetProperty("message").GetString(); + Assert.StartsWith("Rate limit reached for gpt-4o on tokens per min (TPM): Limit 10, Used 10.", message, StringComparison.Ordinal); + } } diff --git a/DevProxy.Plugins/Behavior/LanguageModelRateLimitingPlugin.cs b/DevProxy.Plugins/Behavior/LanguageModelRateLimitingPlugin.cs index a5238b1cd..e984a69d1 100644 --- a/DevProxy.Plugins/Behavior/LanguageModelRateLimitingPlugin.cs +++ b/DevProxy.Plugins/Behavior/LanguageModelRateLimitingPlugin.cs @@ -140,7 +140,7 @@ public override Task BeforeRequestAsync(ProxyRequestArgs e, CancellationToken ca ShouldThrottle, _resetTime )); - ThrottleResponse(e); + ThrottleResponse(e, openAiRequest?.Model); state.HasBeenSet = true; } else @@ -275,26 +275,33 @@ private ThrottlingInfo ShouldThrottle(IHttpRequest request, string throttlingKey Configuration.HeaderRetryAfter); } - private void ThrottleResponse(ProxyRequestArgs e) + private void ThrottleResponse(ProxyRequestArgs e, string? model) { var headers = new List(); - var body = string.Empty; var request = e.ProxySession.Request; + var retryAfterSeconds = (int)(_resetTime - DateTime.Now).TotalSeconds; + + // Report the limit that's been exhausted, matching OpenAI's + // tokens-per-minute rate limit error so that clients back off and retry + var (limit, remaining) = _promptTokensRemaining <= 0 ? + (Configuration.PromptTokenLimit, _promptTokensRemaining) : + (Configuration.CompletionTokenLimit, _completionTokensRemaining); + var used = limit - Math.Max(remaining, 0); + var modelInfo = string.IsNullOrEmpty(model) ? string.Empty : $" for {model}"; - // Build standard OpenAI error response for token limit exceeded var openAiError = new { error = new { - message = "You exceeded your current quota, please check your plan and billing details.", - type = "insufficient_quota", + message = string.Create(CultureInfo.InvariantCulture, $"Rate limit reached{modelInfo} on tokens per min (TPM): Limit {limit}, Used {used}. Please try again in {retryAfterSeconds}s."), + type = "tokens", param = (object?)null, - code = "insufficient_quota" + code = "rate_limit_exceeded" } }; - body = JsonSerializer.Serialize(openAiError, ProxyUtils.JsonSerializerOptions); + var body = JsonSerializer.Serialize(openAiError, ProxyUtils.JsonSerializerOptions); - headers.Add(new(Configuration.HeaderRetryAfter, ((int)(_resetTime - DateTime.Now).TotalSeconds).ToString(CultureInfo.InvariantCulture))); + headers.Add(new(Configuration.HeaderRetryAfter, retryAfterSeconds.ToString(CultureInfo.InvariantCulture))); if (request.Headers.Any(h => h.Name.Equals("Origin", StringComparison.OrdinalIgnoreCase))) { headers.Add(new("Access-Control-Allow-Origin", "*")); diff --git a/skills/dev-proxy/references/test-llm-apps.md b/skills/dev-proxy/references/test-llm-apps.md index 8fcf72961..0fe504d1e 100644 --- a/skills/dev-proxy/references/test-llm-apps.md +++ b/skills/dev-proxy/references/test-llm-apps.md @@ -166,9 +166,10 @@ Use `LanguageModelRateLimitingPlugin` to test token quota handling. ], "body": { "error": { - "message": "Token quota exceeded. Please wait.", - "type": "insufficient_quota", - "code": "token_quota_exceeded" + "message": "Rate limit reached on tokens per min (TPM). Please try again later.", + "type": "tokens", + "param": null, + "code": "rate_limit_exceeded" } } } @@ -176,6 +177,8 @@ Use `LanguageModelRateLimitingPlugin` to test token quota handling. Use `@dynamic` for the retry-after header to auto-calculate seconds until reset. +With `whenLimitExceeded: "Throttle"`, the plugin returns a 429 with OpenAI's `rate_limit_exceeded` error (`type: "tokens"`) so clients back off and retry. To simulate a billing/quota error instead (`insufficient_quota`), use `Custom` with your own response. + ### Scenario Configs **Tight limits (stress testing):** `promptTokenLimit: 500, completionTokenLimit: 250, resetTimeWindowSeconds: 30` From 458b449df7958c37f538781e25666c3e6fd5bc4a Mon Sep 17 00:00:00 2001 From: waldekmastykarz Date: Sat, 3 Oct 2026 20:27:04 +0200 Subject: [PATCH 2/2] Report actual consumed tokens in LanguageModelRateLimitingPlugin throttle message Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- .../BehaviorPluginsIntegrationTests.cs | 6 +++--- .../Behavior/LanguageModelRateLimitingPlugin.cs | 13 +++++++++---- 2 files changed, 12 insertions(+), 7 deletions(-) diff --git a/DevProxy.Integration.Tests/BehaviorPluginsIntegrationTests.cs b/DevProxy.Integration.Tests/BehaviorPluginsIntegrationTests.cs index 97d511a16..b28ead2dd 100644 --- a/DevProxy.Integration.Tests/BehaviorPluginsIntegrationTests.cs +++ b/DevProxy.Integration.Tests/BehaviorPluginsIntegrationTests.cs @@ -188,12 +188,12 @@ public async Task LanguageModelRateLimiting_Throttle_ReturnsOpenAIRateLimitError using var client = proxy.CreateHttpClient(); // /echo returns the request body, so the usage below is read back as the - // response's token usage. #1 drains the prompt token limit, #2 is throttled. + // response's token usage. #1 exceeds the prompt token limit, #2 is throttled. const string requestBody = """ { "model": "gpt-4o", "messages": [ { "role": "user", "content": "hi" } ], - "usage": { "prompt_tokens": 10, "completion_tokens": 5, "total_tokens": 15 } + "usage": { "prompt_tokens": 15, "completion_tokens": 5, "total_tokens": 20 } } """; using var firstContent = new StringContent(requestBody, Encoding.UTF8, "application/json"); @@ -210,6 +210,6 @@ public async Task LanguageModelRateLimiting_Throttle_ReturnsOpenAIRateLimitError Assert.Equal("rate_limit_exceeded", error.GetProperty("code").GetString()); Assert.Equal("tokens", error.GetProperty("type").GetString()); var message = error.GetProperty("message").GetString(); - Assert.StartsWith("Rate limit reached for gpt-4o on tokens per min (TPM): Limit 10, Used 10.", message, StringComparison.Ordinal); + Assert.StartsWith("Rate limit reached for gpt-4o on tokens per min (TPM): Limit 10, Used 15.", message, StringComparison.Ordinal); } } diff --git a/DevProxy.Plugins/Behavior/LanguageModelRateLimitingPlugin.cs b/DevProxy.Plugins/Behavior/LanguageModelRateLimitingPlugin.cs index e984a69d1..21d340c13 100644 --- a/DevProxy.Plugins/Behavior/LanguageModelRateLimitingPlugin.cs +++ b/DevProxy.Plugins/Behavior/LanguageModelRateLimitingPlugin.cs @@ -51,6 +51,8 @@ public sealed class LanguageModelRateLimitingPlugin( // first request and can set the initial values private int _promptTokensRemaining = -1; private int _completionTokensRemaining = -1; + private int _promptTokensUsed; + private int _completionTokensUsed; private DateTime _resetTime = DateTime.MinValue; private LanguageModelRateLimitingCustomResponseLoader? _loader; @@ -118,6 +120,8 @@ public override Task BeforeRequestAsync(ProxyRequestArgs e, CancellationToken ca { _promptTokensRemaining = Configuration.PromptTokenLimit; _completionTokensRemaining = Configuration.CompletionTokenLimit; + _promptTokensUsed = 0; + _completionTokensUsed = 0; _resetTime = DateTime.Now.AddSeconds(Configuration.ResetTimeWindowSeconds); } @@ -243,6 +247,8 @@ public override Task BeforeResponseAsync(ProxyResponseArgs e, CancellationToken _promptTokensRemaining -= promptTokens; _completionTokensRemaining -= completionTokens; + _promptTokensUsed += promptTokens; + _completionTokensUsed += completionTokens; if (_promptTokensRemaining < 0) { @@ -283,10 +289,9 @@ private void ThrottleResponse(ProxyRequestArgs e, string? model) // Report the limit that's been exhausted, matching OpenAI's // tokens-per-minute rate limit error so that clients back off and retry - var (limit, remaining) = _promptTokensRemaining <= 0 ? - (Configuration.PromptTokenLimit, _promptTokensRemaining) : - (Configuration.CompletionTokenLimit, _completionTokensRemaining); - var used = limit - Math.Max(remaining, 0); + var (limit, used) = _promptTokensRemaining <= 0 ? + (Configuration.PromptTokenLimit, _promptTokensUsed) : + (Configuration.CompletionTokenLimit, _completionTokensUsed); var modelInfo = string.IsNullOrEmpty(model) ? string.Empty : $" for {model}"; var openAiError = new