fix(model-fallback): add HTTP statusCode check for GLM rate limit fallback
isRetryableModelError() now checks the HTTP status code (429/503/529) in addition to existing message pattern matching. This ensures rate limit errors trigger model fallback regardless of error message format or language (e.g., Chinese GLM errors). Changes: - ErrorInfo interface extended with statusCode?: number - isRetryableModelError() checks statusCode after STOP patterns, before message pattern fallback - extractErrorStatusCode() added to error-classifier.ts (supports statusCode, status, code, response.status fields) - GLM-specific STOP patterns added: daily call limit, in arrears, fair use policy, recharge and try — these prevent quota/billing 429s from being treated as transient rate limits - statusCode propagated through tryFallbackRetry and manager.ts 400 intentionally excluded from statusCode check (permanent client error).
This commit is contained in:
@@ -3,7 +3,7 @@ const { describe, expect, test, beforeEach, afterEach, mock, spyOn } = require("
|
||||
import * as connectedProvidersCache from "./connected-providers-cache"
|
||||
|
||||
let readConnectedProvidersCacheSpy: ReturnType<typeof spyOn> | undefined
|
||||
const { shouldRetryError, selectFallbackProvider } = await import("./model-error-classifier")
|
||||
const { shouldRetryError, selectFallbackProvider, isRetryableModelError } = await import("./model-error-classifier")
|
||||
|
||||
describe("model-error-classifier", () => {
|
||||
beforeEach(() => {
|
||||
@@ -270,6 +270,139 @@ describe("model-error-classifier", () => {
|
||||
//#then
|
||||
expect(result).toBe(false)
|
||||
})
|
||||
|
||||
test("GLM 429 rate limit with statusCode and Chinese message triggers fallback (statusCode check)", () => {
|
||||
//#given
|
||||
const error = { statusCode: 429, message: "请求频率过高" }
|
||||
|
||||
//#when
|
||||
const result = isRetryableModelError(error)
|
||||
|
||||
//#then
|
||||
expect(result).toBe(true)
|
||||
})
|
||||
|
||||
test("GLM 429 rate limit with statusCode and no message at all triggers fallback", () => {
|
||||
//#given
|
||||
const error = { statusCode: 429 }
|
||||
|
||||
//#when
|
||||
const result = isRetryableModelError(error)
|
||||
|
||||
//#then
|
||||
expect(result).toBe(true)
|
||||
})
|
||||
|
||||
test("GLM 503 service unavailable with statusCode triggers fallback", () => {
|
||||
//#given
|
||||
const error = { statusCode: 503, message: "Service Unavailable" }
|
||||
|
||||
//#when
|
||||
const result = isRetryableModelError(error)
|
||||
|
||||
//#then
|
||||
expect(result).toBe(true)
|
||||
})
|
||||
|
||||
test("GLM 529 overloaded with statusCode triggers fallback", () => {
|
||||
//#given
|
||||
const error = { statusCode: 529 }
|
||||
|
||||
//#when
|
||||
const result = isRetryableModelError(error)
|
||||
|
||||
//#then
|
||||
expect(result).toBe(true)
|
||||
})
|
||||
|
||||
test("HTTP 400 with statusCode does NOT trigger fallback via statusCode alone (400 excluded)", () => {
|
||||
//#given — message does NOT match any retryable pattern
|
||||
const error = { statusCode: 400, message: "Invalid parameter: model_name" }
|
||||
|
||||
//#when
|
||||
const result = isRetryableModelError(error)
|
||||
|
||||
//#then
|
||||
expect(result).toBe(false)
|
||||
})
|
||||
|
||||
test("HTTP 401 with statusCode does NOT trigger fallback (not a rate limit)", () => {
|
||||
//#given
|
||||
const error = { statusCode: 401, message: "Unauthorized" }
|
||||
|
||||
//#when
|
||||
const result = isRetryableModelError(error)
|
||||
|
||||
//#then
|
||||
expect(result).toBe(false)
|
||||
})
|
||||
|
||||
test("GLM code 1304 daily quota 429 does NOT trigger fallback (STOP pattern wins)", () => {
|
||||
//#given
|
||||
const error = {
|
||||
statusCode: 429,
|
||||
message: "Daily call limit for this API key has been reached. Limit will reset at midnight UTC.",
|
||||
}
|
||||
|
||||
//#when
|
||||
const result = isRetryableModelError(error)
|
||||
|
||||
//#then
|
||||
expect(result).toBe(false)
|
||||
})
|
||||
|
||||
test("GLM account in arrears 429 does NOT trigger fallback (STOP pattern wins)", () => {
|
||||
//#given
|
||||
const error = {
|
||||
statusCode: 429,
|
||||
message: "Your account is in arrears, please recharge and try again.",
|
||||
}
|
||||
|
||||
//#when
|
||||
const result = isRetryableModelError(error)
|
||||
|
||||
//#then
|
||||
expect(result).toBe(false)
|
||||
})
|
||||
|
||||
test("GLM fair use policy violation 429 does NOT trigger fallback (STOP pattern wins)", () => {
|
||||
//#given
|
||||
const error = {
|
||||
statusCode: 429,
|
||||
message: "Request blocked under Fair Use Policy. Your request rate has been restricted.",
|
||||
}
|
||||
|
||||
//#when
|
||||
const result = isRetryableModelError(error)
|
||||
|
||||
//#then
|
||||
expect(result).toBe(false)
|
||||
})
|
||||
|
||||
test("STOP message pattern takes precedence over 429 statusCode", () => {
|
||||
//#given
|
||||
const error = {
|
||||
statusCode: 429,
|
||||
message: "quota exceeded for this account, usage limit has been reached",
|
||||
}
|
||||
|
||||
//#when
|
||||
const result = isRetryableModelError(error)
|
||||
|
||||
//#then
|
||||
expect(result).toBe(false)
|
||||
})
|
||||
|
||||
test("rate limit message without statusCode still works (backward compat)", () => {
|
||||
//#given
|
||||
const error = { message: "rate limit reached for requests" }
|
||||
|
||||
//#when
|
||||
const result = isRetryableModelError(error)
|
||||
|
||||
//#then
|
||||
expect(result).toBe(true)
|
||||
})
|
||||
})
|
||||
|
||||
export {}
|
||||
|
||||
@@ -99,6 +99,13 @@ const STOP_MESSAGE_PATTERNS = [
|
||||
"credit balance",
|
||||
"usage limit for this month",
|
||||
"exhausted your capacity",
|
||||
// GLM/Z.ai business error codes that indicate permanent quota/billing exhaustion
|
||||
"daily call limit",
|
||||
"daily limit",
|
||||
"usage limit reached for",
|
||||
"in arrears",
|
||||
"fair use policy",
|
||||
"recharge and try",
|
||||
]
|
||||
|
||||
const AUTO_RETRY_GATE_PATTERNS = [
|
||||
@@ -117,6 +124,8 @@ function hasProviderAutoRetrySignal(message: string): boolean {
|
||||
export interface ErrorInfo {
|
||||
name?: string
|
||||
message?: string
|
||||
/** HTTP status code from the provider response (e.g., 429 for rate limit) */
|
||||
statusCode?: number
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -151,6 +160,16 @@ export function isRetryableModelError(error: ErrorInfo): boolean {
|
||||
if (hasProviderAutoRetrySignal(msg)) {
|
||||
return true
|
||||
}
|
||||
|
||||
// HTTP status code check: catches rate-limit errors regardless of message format/language.
|
||||
// Uses the same codes as runtime-fallback config (400 excluded as it is a permanent client error).
|
||||
if (
|
||||
error.statusCode != null &&
|
||||
(error.statusCode === 429 || error.statusCode === 503 || error.statusCode === 529)
|
||||
) {
|
||||
return true
|
||||
}
|
||||
|
||||
return RETRYABLE_MESSAGE_PATTERNS.some((pattern) => msg.includes(pattern))
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user