From 692b54046cebc5e1a6969a2da81ddb761966e67e Mon Sep 17 00:00:00 2001 From: unolife <38601861+unolife@users.noreply.github.com> Date: Thu, 17 Sep 2026 16:47:14 +0900 Subject: [PATCH] Put the response status and headers on the error the backoff reads MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `calculateDelay` in core/llm/utils/retry.ts looks for `error.headers["retry-after"]`, `["x-ratelimit-reset"]` and friends so a retry waits exactly as long as the provider asked. Nothing ever put them there: `parseError` builds its error from the response and keeps only the text, so that branch could not fire and every rate limit fell through to exponential backoff — a guess, when the provider had already said the answer. `parseError` now annotates whatever it built with the response's `status` and a lower-cased plain-object copy of its headers. Lower-cased because that is how `Headers` yields them and how the reader spells the names it checks first, and a plain object because the reader uses bracket access, which returns undefined on a `Headers` instance. Headers are left unset when the response carried none: an empty object reads as "the provider answered with no limits", and absent says it never answered. No behaviour changes on its own — this only makes the existing header path reachable. Verified: `jest llm/index.test.ts` — 189 passed, including three new cases. --- core/llm/index.test.ts | 57 ++++++++++++++++++++++++++++++++++++++++++ core/llm/index.ts | 32 ++++++++++++++++++++++++ 2 files changed, 89 insertions(+) diff --git a/core/llm/index.test.ts b/core/llm/index.test.ts index 488252b95a4..14398f34a51 100644 --- a/core/llm/index.test.ts +++ b/core/llm/index.test.ts @@ -199,3 +199,60 @@ describe("BaseLLM", () => { }); }); }); + +describe("BaseLLM error annotation", () => { + class ErrorLLM extends BaseLLM { + static providerName = "openai"; + // parseError is private; these tests are about what it leaves on the error + // for llm/utils/retry.ts to read. + parse(resp: unknown): Promise { + return (this as any).parseError(resp); + } + } + + const llm = () => new ErrorLLM({ model: "dummy-model" }); + + const response = (init: { status: number; headers?: Record }) => ({ + status: init.status, + statusText: "Too Many Requests", + url: "https://api.test-api-dummy.com/v1/chat/completions", + headers: new Headers(init.headers ?? {}), + text: async () => "rate limited", + }); + + it("puts the provider's retry-after where the backoff reads it", async () => { + // calculateDelay() in llm/utils/retry.ts looks for error.headers["retry-after"] + // to wait exactly as long as the provider asked. Nothing was setting it, so + // every rate limit fell through to exponential backoff -- a guess, when the + // provider had already said the answer. + const error = (await llm().parse( + response({ status: 429, headers: { "Retry-After": "17" } }), + )) as Error & { status?: number; headers?: Record }; + + expect(error.status).toBe(429); + expect(error.headers?.["retry-after"]).toBe("17"); + }); + + it("lower-cases the names the reader looks for", async () => { + const error = (await llm().parse( + response({ + status: 429, + headers: { "X-RateLimit-Reset": "1789621158", "X-RateLimit-Remaining": "0" }, + }), + )) as Error & { headers?: Record }; + + expect(error.headers?.["x-ratelimit-reset"]).toBe("1789621158"); + expect(error.headers?.["x-ratelimit-remaining"]).toBe("0"); + }); + + it("leaves headers unset when the response carried none", async () => { + // An empty object would read as "the provider answered with no limits"; + // absent says it never answered at all. + const error = (await llm().parse(response({ status: 500 }))) as Error & { + headers?: Record; + }; + + expect(error.headers).toBeUndefined(); + expect(error.status).toBe(500); + }); +}); diff --git a/core/llm/index.ts b/core/llm/index.ts index 1af44b25614..9b775e3be50 100644 --- a/core/llm/index.ts +++ b/core/llm/index.ts @@ -402,7 +402,39 @@ export abstract class BaseLLM implements ILLM { } } + /** + * Copies the response's status and headers onto the error. + * + * `calculateDelay` in llm/utils/retry.ts reads `error.headers["retry-after"]` + * and friends to back off for as long as the provider asked. Nothing was ever + * putting them there, so that path could not fire and every rate limit fell + * through to exponential backoff -- a guess, when the provider had already + * said the answer. + * + * Header names are lower-cased because that is how `Headers` yields them and + * how the reader spells the ones it looks for first. + */ + private annotateError(error: Error, resp: any): Error { + const annotated = error as Error & { + status?: number; + headers?: Record; + }; + annotated.status = resp?.status; + const headers: Record = {}; + resp?.headers?.forEach?.((value: string, name: string) => { + headers[name.toLowerCase()] = value; + }); + if (Object.keys(headers).length > 0) { + annotated.headers = headers; + } + return annotated; + } + private async parseError(resp: any): Promise { + return this.annotateError(await this.buildError(resp), resp); + } + + private async buildError(resp: any): Promise { let text = await resp.text(); if (resp.status === 404 && !resp.url.includes("/v1")) {