Arm the shared cooldown for rate limits tunneled through HTTP 200

Azure is the only zero-data-retention route for the gpt-5.6 family, so
its capacity 429s arrive frequently and OpenRouter forwards them inside
an HTTP 200 envelope. Those bypassed the rate controller entirely: a
paced run kept sending a request every three seconds into a throttled
endpoint, failing row by row. An in-envelope 429 now records the same
escalating cooldown as a transport 429, so later acquisitions fail fast
until the deadline passes.
This commit is contained in:
Lars Nolden
2026-09-13 11:43:49 +02:00
parent 1b09edc692
commit 4d8a187079
3 changed files with 69 additions and 23 deletions
+38 -23
View File
@@ -109,6 +109,43 @@ func (g *Controller) Release() {
<-g.active
}
// recordLimit escalates the consecutive-failure backoff, retains the cooldown
// and learns spacing. Callers hold the Acquire gate, like Do's 429 branch.
func (g *Controller) recordLimit(header string) *RateLimitError {
if g.backoff <= 0 {
g.backoff = g.InitialBackoff
if g.backoff <= 0 {
g.backoff = time.Second
}
} else if g.backoff >= maxBackoff/2 {
g.backoff = max(g.backoff, maxBackoff)
} else {
g.backoff *= 2
}
fallback := max(g.backoff, g.MinimumInterval, g.learnedInterval)
limit := retryLimit(header, time.Now(), fallback)
g.mu.Lock()
g.limit = limit
g.mu.Unlock()
// Keep the most conservative learned cadence for this controller's
// lifetime, capped at 30 seconds. The actual provider deadline is never
// capped; persistent failures separately escalate up to 15 minutes.
learned := maxLearnedInterval
if !limit.unbounded {
learned = min(learned, time.Until(limit.next))
}
g.learnedInterval = max(g.learnedInterval, learned)
return limit
}
// ReportLimit records a rate limit the provider communicated outside the HTTP
// status — typically inside an HTTP 200 error envelope — so later Acquire
// calls fail fast during the cooldown exactly as after a transport HTTP 429.
// It must be called while holding an Acquire, like Do.
func (g *Controller) ReportLimit() *RateLimitError {
return g.recordLimit("")
}
// retryLimit never converts a positive overflowing delay into a short wait.
// Delays beyond time.Duration's range disable retries rather than truncate the
// provider's instruction. HTTP dates retain their absolute timestamp unchanged.
@@ -202,29 +239,7 @@ func (g *Controller) Do(ctx context.Context, attempt func(context.Context) (*htt
}
return resp, nil
}
if g.backoff <= 0 {
g.backoff = g.InitialBackoff
if g.backoff <= 0 {
g.backoff = time.Second
}
} else if g.backoff >= maxBackoff/2 {
g.backoff = max(g.backoff, maxBackoff)
} else {
g.backoff *= 2
}
fallback := max(g.backoff, g.MinimumInterval, g.learnedInterval)
limit := retryLimit(resp.Header.Get("Retry-After"), time.Now(), fallback)
g.mu.Lock()
g.limit = limit
g.mu.Unlock()
// Keep the most conservative learned cadence for this controller's
// lifetime, capped at 30 seconds. The actual provider deadline is never
// capped; persistent failures separately escalate up to 15 minutes.
learned := maxLearnedInterval
if !limit.unbounded {
learned = min(learned, time.Until(limit.next))
}
g.learnedInterval = max(g.learnedInterval, learned)
limit := g.recordLimit(resp.Header.Get("Retry-After"))
// Never read or expose provider errors, and release each response before
// any sleep or retry. Other responses are processed by the caller.
resp.Body.Close()