diff --git a/DevProxy.Integration.Tests/BehaviorPluginsIntegrationTests.cs b/DevProxy.Integration.Tests/BehaviorPluginsIntegrationTests.cs index be477243..b28ead2d 100644 --- a/DevProxy.Integration.Tests/BehaviorPluginsIntegrationTests.cs +++ b/DevProxy.Integration.Tests/BehaviorPluginsIntegrationTests.cs @@ -5,6 +5,8 @@ using System.Diagnostics; using System.Globalization; using System.Net; +using System.Text; +using System.Text.Json; using DevProxy.Abstractions.Plugins; using DevProxy.Abstractions.Proxy; using DevProxy.Plugins.Behavior; @@ -165,4 +167,49 @@ public async Task RateLimitingPlusRetryAfter_ThrottledResponseCarriesRetryAfter( out _), "Retry-After header should be an integer seconds value."); } + + [Fact] + public async Task LanguageModelRateLimiting_Throttle_ReturnsOpenAIRateLimitError() + { + await using var origin = await FakeOrigin.StartAsync(); + var urls = KestrelProxyHarness.BuildUrlsToWatch(origin.Host); + + var plugin = new LanguageModelRateLimitingPlugin( + SharedHttpClient, + NullLogger.Instance, + urls, + ProxyConfig, + PluginConfig.FromJson(""" + { "promptTokenLimit": 10, "completionTokenLimit": 100, "resetTimeWindowSeconds": 300 } + """)); + + await using var proxy = await KestrelProxyHarness.StartAsync( + origin.Host, [plugin]); + using var client = proxy.CreateHttpClient(); + + // /echo returns the request body, so the usage below is read back as the + // response's token usage. #1 exceeds the prompt token limit, #2 is throttled. + const string requestBody = """ + { + "model": "gpt-4o", + "messages": [ { "role": "user", "content": "hi" } ], + "usage": { "prompt_tokens": 15, "completion_tokens": 5, "total_tokens": 20 } + } + """; + using var firstContent = new StringContent(requestBody, Encoding.UTF8, "application/json"); + using var first = await client.PostAsync(new Uri($"http://{origin.Host}/echo"), firstContent); + using var secondContent = new StringContent(requestBody, Encoding.UTF8, "application/json"); + using var throttled = await client.PostAsync(new Uri($"http://{origin.Host}/echo"), secondContent); + + Assert.Equal(HttpStatusCode.OK, first.StatusCode); + Assert.Equal(HttpStatusCode.TooManyRequests, throttled.StatusCode); + Assert.True(throttled.Headers.Contains("retry-after")); + + using var json = JsonDocument.Parse(await throttled.Content.ReadAsStringAsync()); + var error = json.RootElement.GetProperty("error"); + Assert.Equal("rate_limit_exceeded", error.GetProperty("code").GetString()); + Assert.Equal("tokens", error.GetProperty("type").GetString()); + var message = error.GetProperty("message").GetString(); + Assert.StartsWith("Rate limit reached for gpt-4o on tokens per min (TPM): Limit 10, Used 15.", message, StringComparison.Ordinal); + } } diff --git a/DevProxy.Plugins/Behavior/LanguageModelRateLimitingPlugin.cs b/DevProxy.Plugins/Behavior/LanguageModelRateLimitingPlugin.cs index a5238b1c..21d340c1 100644 --- a/DevProxy.Plugins/Behavior/LanguageModelRateLimitingPlugin.cs +++ b/DevProxy.Plugins/Behavior/LanguageModelRateLimitingPlugin.cs @@ -51,6 +51,8 @@ public sealed class LanguageModelRateLimitingPlugin( // first request and can set the initial values private int _promptTokensRemaining = -1; private int _completionTokensRemaining = -1; + private int _promptTokensUsed; + private int _completionTokensUsed; private DateTime _resetTime = DateTime.MinValue; private LanguageModelRateLimitingCustomResponseLoader? _loader; @@ -118,6 +120,8 @@ public override Task BeforeRequestAsync(ProxyRequestArgs e, CancellationToken ca { _promptTokensRemaining = Configuration.PromptTokenLimit; _completionTokensRemaining = Configuration.CompletionTokenLimit; + _promptTokensUsed = 0; + _completionTokensUsed = 0; _resetTime = DateTime.Now.AddSeconds(Configuration.ResetTimeWindowSeconds); } @@ -140,7 +144,7 @@ public override Task BeforeRequestAsync(ProxyRequestArgs e, CancellationToken ca ShouldThrottle, _resetTime )); - ThrottleResponse(e); + ThrottleResponse(e, openAiRequest?.Model); state.HasBeenSet = true; } else @@ -243,6 +247,8 @@ public override Task BeforeResponseAsync(ProxyResponseArgs e, CancellationToken _promptTokensRemaining -= promptTokens; _completionTokensRemaining -= completionTokens; + _promptTokensUsed += promptTokens; + _completionTokensUsed += completionTokens; if (_promptTokensRemaining < 0) { @@ -275,26 +281,32 @@ private ThrottlingInfo ShouldThrottle(IHttpRequest request, string throttlingKey Configuration.HeaderRetryAfter); } - private void ThrottleResponse(ProxyRequestArgs e) + private void ThrottleResponse(ProxyRequestArgs e, string? model) { var headers = new List(); - var body = string.Empty; var request = e.ProxySession.Request; + var retryAfterSeconds = (int)(_resetTime - DateTime.Now).TotalSeconds; + + // Report the limit that's been exhausted, matching OpenAI's + // tokens-per-minute rate limit error so that clients back off and retry + var (limit, used) = _promptTokensRemaining <= 0 ? + (Configuration.PromptTokenLimit, _promptTokensUsed) : + (Configuration.CompletionTokenLimit, _completionTokensUsed); + var modelInfo = string.IsNullOrEmpty(model) ? string.Empty : $" for {model}"; - // Build standard OpenAI error response for token limit exceeded var openAiError = new { error = new { - message = "You exceeded your current quota, please check your plan and billing details.", - type = "insufficient_quota", + message = string.Create(CultureInfo.InvariantCulture, $"Rate limit reached{modelInfo} on tokens per min (TPM): Limit {limit}, Used {used}. Please try again in {retryAfterSeconds}s."), + type = "tokens", param = (object?)null, - code = "insufficient_quota" + code = "rate_limit_exceeded" } }; - body = JsonSerializer.Serialize(openAiError, ProxyUtils.JsonSerializerOptions); + var body = JsonSerializer.Serialize(openAiError, ProxyUtils.JsonSerializerOptions); - headers.Add(new(Configuration.HeaderRetryAfter, ((int)(_resetTime - DateTime.Now).TotalSeconds).ToString(CultureInfo.InvariantCulture))); + headers.Add(new(Configuration.HeaderRetryAfter, retryAfterSeconds.ToString(CultureInfo.InvariantCulture))); if (request.Headers.Any(h => h.Name.Equals("Origin", StringComparison.OrdinalIgnoreCase))) { headers.Add(new("Access-Control-Allow-Origin", "*")); diff --git a/skills/dev-proxy/references/test-llm-apps.md b/skills/dev-proxy/references/test-llm-apps.md index 8fcf7296..0fe504d1 100644 --- a/skills/dev-proxy/references/test-llm-apps.md +++ b/skills/dev-proxy/references/test-llm-apps.md @@ -166,9 +166,10 @@ Use `LanguageModelRateLimitingPlugin` to test token quota handling. ], "body": { "error": { - "message": "Token quota exceeded. Please wait.", - "type": "insufficient_quota", - "code": "token_quota_exceeded" + "message": "Rate limit reached on tokens per min (TPM). Please try again later.", + "type": "tokens", + "param": null, + "code": "rate_limit_exceeded" } } } @@ -176,6 +177,8 @@ Use `LanguageModelRateLimitingPlugin` to test token quota handling. Use `@dynamic` for the retry-after header to auto-calculate seconds until reset. +With `whenLimitExceeded: "Throttle"`, the plugin returns a 429 with OpenAI's `rate_limit_exceeded` error (`type: "tokens"`) so clients back off and retry. To simulate a billing/quota error instead (`insufficient_quota`), use `Custom` with your own response. + ### Scenario Configs **Tight limits (stress testing):** `promptTokenLimit: 500, completionTokenLimit: 250, resetTimeWindowSeconds: 30`