fix(model-fallback): add HTTP statusCode check for GLM rate limit fallback

isRetryableModelError() now checks the HTTP status code (429/503/529)
in addition to existing message pattern matching. This ensures rate
limit errors trigger model fallback regardless of error message format
or language (e.g., Chinese GLM errors).

Changes:
- ErrorInfo interface extended with statusCode?: number
- isRetryableModelError() checks statusCode after STOP patterns, before
  message pattern fallback
- extractErrorStatusCode() added to error-classifier.ts (supports
  statusCode, status, code, response.status fields)
- GLM-specific STOP patterns added: daily call limit, in arrears,
  fair use policy, recharge and try — these prevent quota/billing 429s
  from being treated as transient rate limits
- statusCode propagated through tryFallbackRetry and manager.ts

400 intentionally excluded from statusCode check (permanent client error).
This commit is contained in:
cailgarrisk-collab
2026-05-03 11:59:13 +02:00
parent 9ba3b574a7
commit 61d2f1195b
5 changed files with 187 additions and 4 deletions
+134 -1
View File
@@ -3,7 +3,7 @@ const { describe, expect, test, beforeEach, afterEach, mock, spyOn } = require("
import * as connectedProvidersCache from "./connected-providers-cache"
let readConnectedProvidersCacheSpy: ReturnType<typeof spyOn> | undefined
const { shouldRetryError, selectFallbackProvider } = await import("./model-error-classifier")
const { shouldRetryError, selectFallbackProvider, isRetryableModelError } = await import("./model-error-classifier")
describe("model-error-classifier", () => {
beforeEach(() => {
@@ -270,6 +270,139 @@ describe("model-error-classifier", () => {
//#then
expect(result).toBe(false)
})
test("GLM 429 rate limit with statusCode and Chinese message triggers fallback (statusCode check)", () => {
//#given
const error = { statusCode: 429, message: "请求频率过高" }
//#when
const result = isRetryableModelError(error)
//#then
expect(result).toBe(true)
})
test("GLM 429 rate limit with statusCode and no message at all triggers fallback", () => {
//#given
const error = { statusCode: 429 }
//#when
const result = isRetryableModelError(error)
//#then
expect(result).toBe(true)
})
test("GLM 503 service unavailable with statusCode triggers fallback", () => {
//#given
const error = { statusCode: 503, message: "Service Unavailable" }
//#when
const result = isRetryableModelError(error)
//#then
expect(result).toBe(true)
})
test("GLM 529 overloaded with statusCode triggers fallback", () => {
//#given
const error = { statusCode: 529 }
//#when
const result = isRetryableModelError(error)
//#then
expect(result).toBe(true)
})
test("HTTP 400 with statusCode does NOT trigger fallback via statusCode alone (400 excluded)", () => {
//#given — message does NOT match any retryable pattern
const error = { statusCode: 400, message: "Invalid parameter: model_name" }
//#when
const result = isRetryableModelError(error)
//#then
expect(result).toBe(false)
})
test("HTTP 401 with statusCode does NOT trigger fallback (not a rate limit)", () => {
//#given
const error = { statusCode: 401, message: "Unauthorized" }
//#when
const result = isRetryableModelError(error)
//#then
expect(result).toBe(false)
})
test("GLM code 1304 daily quota 429 does NOT trigger fallback (STOP pattern wins)", () => {
//#given
const error = {
statusCode: 429,
message: "Daily call limit for this API key has been reached. Limit will reset at midnight UTC.",
}
//#when
const result = isRetryableModelError(error)
//#then
expect(result).toBe(false)
})
test("GLM account in arrears 429 does NOT trigger fallback (STOP pattern wins)", () => {
//#given
const error = {
statusCode: 429,
message: "Your account is in arrears, please recharge and try again.",
}
//#when
const result = isRetryableModelError(error)
//#then
expect(result).toBe(false)
})
test("GLM fair use policy violation 429 does NOT trigger fallback (STOP pattern wins)", () => {
//#given
const error = {
statusCode: 429,
message: "Request blocked under Fair Use Policy. Your request rate has been restricted.",
}
//#when
const result = isRetryableModelError(error)
//#then
expect(result).toBe(false)
})
test("STOP message pattern takes precedence over 429 statusCode", () => {
//#given
const error = {
statusCode: 429,
message: "quota exceeded for this account, usage limit has been reached",
}
//#when
const result = isRetryableModelError(error)
//#then
expect(result).toBe(false)
})
test("rate limit message without statusCode still works (backward compat)", () => {
//#given
const error = { message: "rate limit reached for requests" }
//#when
const result = isRetryableModelError(error)
//#then
expect(result).toBe(true)
})
})
export {}
+19
View File
@@ -99,6 +99,13 @@ const STOP_MESSAGE_PATTERNS = [
"credit balance",
"usage limit for this month",
"exhausted your capacity",
// GLM/Z.ai business error codes that indicate permanent quota/billing exhaustion
"daily call limit",
"daily limit",
"usage limit reached for",
"in arrears",
"fair use policy",
"recharge and try",
]
const AUTO_RETRY_GATE_PATTERNS = [
@@ -117,6 +124,8 @@ function hasProviderAutoRetrySignal(message: string): boolean {
export interface ErrorInfo {
name?: string
message?: string
/** HTTP status code from the provider response (e.g., 429 for rate limit) */
statusCode?: number
}
/**
@@ -151,6 +160,16 @@ export function isRetryableModelError(error: ErrorInfo): boolean {
if (hasProviderAutoRetrySignal(msg)) {
return true
}
// HTTP status code check: catches rate-limit errors regardless of message format/language.
// Uses the same codes as runtime-fallback config (400 excluded as it is a permanent client error).
if (
error.statusCode != null &&
(error.statusCode === 429 || error.statusCode === 503 || error.statusCode === 529)
) {
return true
}
return RETRYABLE_MESSAGE_PATTERNS.some((pattern) => msg.includes(pattern))
}