Arm the shared cooldown for rate limits tunneled through HTTP 200
Azure is the only zero-data-retention route for the gpt-5.6 family, so its capacity 429s arrive frequently and OpenRouter forwards them inside an HTTP 200 envelope. Those bypassed the rate controller entirely: a paced run kept sending a request every three seconds into a throttled endpoint, failing row by row. An in-envelope 429 now records the same escalating cooldown as a transport 429, so later acquisitions fail fast until the deadline passes.
This commit is contained in:
@@ -109,6 +109,43 @@ func (g *Controller) Release() {
|
||||
<-g.active
|
||||
}
|
||||
|
||||
// recordLimit escalates the consecutive-failure backoff, retains the cooldown
|
||||
// and learns spacing. Callers hold the Acquire gate, like Do's 429 branch.
|
||||
func (g *Controller) recordLimit(header string) *RateLimitError {
|
||||
if g.backoff <= 0 {
|
||||
g.backoff = g.InitialBackoff
|
||||
if g.backoff <= 0 {
|
||||
g.backoff = time.Second
|
||||
}
|
||||
} else if g.backoff >= maxBackoff/2 {
|
||||
g.backoff = max(g.backoff, maxBackoff)
|
||||
} else {
|
||||
g.backoff *= 2
|
||||
}
|
||||
fallback := max(g.backoff, g.MinimumInterval, g.learnedInterval)
|
||||
limit := retryLimit(header, time.Now(), fallback)
|
||||
g.mu.Lock()
|
||||
g.limit = limit
|
||||
g.mu.Unlock()
|
||||
// Keep the most conservative learned cadence for this controller's
|
||||
// lifetime, capped at 30 seconds. The actual provider deadline is never
|
||||
// capped; persistent failures separately escalate up to 15 minutes.
|
||||
learned := maxLearnedInterval
|
||||
if !limit.unbounded {
|
||||
learned = min(learned, time.Until(limit.next))
|
||||
}
|
||||
g.learnedInterval = max(g.learnedInterval, learned)
|
||||
return limit
|
||||
}
|
||||
|
||||
// ReportLimit records a rate limit the provider communicated outside the HTTP
|
||||
// status — typically inside an HTTP 200 error envelope — so later Acquire
|
||||
// calls fail fast during the cooldown exactly as after a transport HTTP 429.
|
||||
// It must be called while holding an Acquire, like Do.
|
||||
func (g *Controller) ReportLimit() *RateLimitError {
|
||||
return g.recordLimit("")
|
||||
}
|
||||
|
||||
// retryLimit never converts a positive overflowing delay into a short wait.
|
||||
// Delays beyond time.Duration's range disable retries rather than truncate the
|
||||
// provider's instruction. HTTP dates retain their absolute timestamp unchanged.
|
||||
@@ -202,29 +239,7 @@ func (g *Controller) Do(ctx context.Context, attempt func(context.Context) (*htt
|
||||
}
|
||||
return resp, nil
|
||||
}
|
||||
if g.backoff <= 0 {
|
||||
g.backoff = g.InitialBackoff
|
||||
if g.backoff <= 0 {
|
||||
g.backoff = time.Second
|
||||
}
|
||||
} else if g.backoff >= maxBackoff/2 {
|
||||
g.backoff = max(g.backoff, maxBackoff)
|
||||
} else {
|
||||
g.backoff *= 2
|
||||
}
|
||||
fallback := max(g.backoff, g.MinimumInterval, g.learnedInterval)
|
||||
limit := retryLimit(resp.Header.Get("Retry-After"), time.Now(), fallback)
|
||||
g.mu.Lock()
|
||||
g.limit = limit
|
||||
g.mu.Unlock()
|
||||
// Keep the most conservative learned cadence for this controller's
|
||||
// lifetime, capped at 30 seconds. The actual provider deadline is never
|
||||
// capped; persistent failures separately escalate up to 15 minutes.
|
||||
learned := maxLearnedInterval
|
||||
if !limit.unbounded {
|
||||
learned = min(learned, time.Until(limit.next))
|
||||
}
|
||||
g.learnedInterval = max(g.learnedInterval, learned)
|
||||
limit := g.recordLimit(resp.Header.Get("Retry-After"))
|
||||
// Never read or expose provider errors, and release each response before
|
||||
// any sleep or retry. Other responses are processed by the caller.
|
||||
resp.Body.Close()
|
||||
|
||||
Reference in New Issue
Block a user