Merge pull request #3773 from cailgarrisk-collab/fix/glm-rate-limit-fallback-statuscode

fix(model-fallback): add HTTP statusCode check for GLM rate limit fallback
This commit is contained in:
YeonGyu-Kim
2026-05-15 20:04:22 +09:00
committed by GitHub
5 changed files with 187 additions and 4 deletions
@@ -65,6 +65,33 @@ export function extractErrorMessage(error: unknown): string | undefined {
}
}
export function extractErrorStatusCode(error: unknown): number | undefined {
if (!isRecord(error)) return undefined
for (const key of ["statusCode", "status", "code"]) {
const val = (error as Record<string, unknown>)[key]
if (typeof val === "number" && val >= 100 && val < 600) return val
}
const statusVal = (error as Record<string, unknown>)["status"]
if (typeof statusVal === "string") {
const parsed = parseInt(statusVal, 10)
if (parsed >= 100 && parsed < 600) return parsed
}
const responseRaw = (error as Record<string, unknown>)["response"]
if (isRecord(responseRaw)) {
const respStatus = responseRaw["status"]
if (typeof respStatus === "number" && respStatus >= 100 && respStatus < 600) return respStatus
if (typeof respStatus === "string") {
const parsed = parseInt(respStatus, 10)
if (parsed >= 100 && parsed < 600) return parsed
}
}
return undefined
}
interface EventPropertiesLike {
[key: string]: unknown
}
@@ -41,7 +41,7 @@ const defaultFallbackRetryHandlerDeps: FallbackRetryHandlerDeps = {
export async function tryFallbackRetry(args: {
task: BackgroundTask
errorInfo: { name?: string; message?: string }
errorInfo: { name?: string; message?: string; statusCode?: number }
source: string
concurrencyManager: ConcurrencyManager
client: OpencodeClient
+6 -2
View File
@@ -48,6 +48,7 @@ import {
isAbortedSessionError,
extractErrorName,
extractErrorMessage,
extractErrorStatusCode,
getSessionErrorMessage,
isRecord,
} from "./error-classifier"
@@ -944,6 +945,7 @@ The fallback retry session is now created and can be inspected directly.
const errorInfo = {
name: extractErrorName(error),
message: extractErrorMessage(error),
statusCode: extractErrorStatusCode(error),
}
if (await this.tryFallbackRetry(existingTask, errorInfo, "promptAsync.launch")) {
return
@@ -1311,6 +1313,7 @@ The fallback retry session is now created and can be inspected directly.
const errorInfo = {
name: extractErrorName(error),
message: extractErrorMessage(error),
statusCode: extractErrorStatusCode(error),
}
if (await this.tryFallbackRetry(existingTask, errorInfo, "promptAsync.resume")) {
return
@@ -1644,6 +1647,7 @@ The fallback retry session is now created and can be inspected directly.
const errorInfo = {
name: extractErrorName(assistantError),
message: extractErrorMessage(assistantError),
statusCode: extractErrorStatusCode(assistantError),
}
void this.tryFallbackRetry(task, errorInfo, "message.updated").catch((error) => {
log("[background-agent] Error handling message.updated fallback retry:", {
@@ -1969,7 +1973,7 @@ The fallback retry session is now created and can be inspected directly.
private async handleSessionErrorEvent(args: {
task: BackgroundTask
errorInfo: { name?: string; message?: string }
errorInfo: { name?: string; message?: string; statusCode?: number }
errorName: string | undefined
errorMessage: string | undefined
}): Promise<void> {
@@ -2078,7 +2082,7 @@ The fallback retry session is now created and can be inspected directly.
private async tryFallbackRetry(
task: BackgroundTask,
errorInfo: { name?: string; message?: string },
errorInfo: { name?: string; message?: string; statusCode?: number },
source: string,
): Promise<boolean> {
const previousSessionID = task.sessionId
+134 -1
View File
@@ -3,7 +3,7 @@ const { describe, expect, test, beforeEach, afterEach, mock, spyOn } = require("
import * as connectedProvidersCache from "./connected-providers-cache"
let readConnectedProvidersCacheSpy: ReturnType<typeof spyOn> | undefined
const { shouldRetryError, selectFallbackProvider } = await import("./model-error-classifier")
const { shouldRetryError, selectFallbackProvider, isRetryableModelError } = await import("./model-error-classifier")
describe("model-error-classifier", () => {
beforeEach(() => {
@@ -270,6 +270,139 @@ describe("model-error-classifier", () => {
//#then
expect(result).toBe(false)
})
test("GLM 429 rate limit with statusCode and Chinese message triggers fallback (statusCode check)", () => {
//#given
const error = { statusCode: 429, message: "请求频率过高" }
//#when
const result = isRetryableModelError(error)
//#then
expect(result).toBe(true)
})
test("GLM 429 rate limit with statusCode and no message at all triggers fallback", () => {
//#given
const error = { statusCode: 429 }
//#when
const result = isRetryableModelError(error)
//#then
expect(result).toBe(true)
})
test("GLM 503 service unavailable with statusCode triggers fallback", () => {
//#given
const error = { statusCode: 503, message: "Service Unavailable" }
//#when
const result = isRetryableModelError(error)
//#then
expect(result).toBe(true)
})
test("GLM 529 overloaded with statusCode triggers fallback", () => {
//#given
const error = { statusCode: 529 }
//#when
const result = isRetryableModelError(error)
//#then
expect(result).toBe(true)
})
test("HTTP 400 with statusCode does NOT trigger fallback via statusCode alone (400 excluded)", () => {
//#given — message does NOT match any retryable pattern
const error = { statusCode: 400, message: "Invalid parameter: model_name" }
//#when
const result = isRetryableModelError(error)
//#then
expect(result).toBe(false)
})
test("HTTP 401 with statusCode does NOT trigger fallback (not a rate limit)", () => {
//#given
const error = { statusCode: 401, message: "Unauthorized" }
//#when
const result = isRetryableModelError(error)
//#then
expect(result).toBe(false)
})
test("GLM code 1304 daily quota 429 does NOT trigger fallback (STOP pattern wins)", () => {
//#given
const error = {
statusCode: 429,
message: "Daily call limit for this API key has been reached. Limit will reset at midnight UTC.",
}
//#when
const result = isRetryableModelError(error)
//#then
expect(result).toBe(false)
})
test("GLM account in arrears 429 does NOT trigger fallback (STOP pattern wins)", () => {
//#given
const error = {
statusCode: 429,
message: "Your account is in arrears, please recharge and try again.",
}
//#when
const result = isRetryableModelError(error)
//#then
expect(result).toBe(false)
})
test("GLM fair use policy violation 429 does NOT trigger fallback (STOP pattern wins)", () => {
//#given
const error = {
statusCode: 429,
message: "Request blocked under Fair Use Policy. Your request rate has been restricted.",
}
//#when
const result = isRetryableModelError(error)
//#then
expect(result).toBe(false)
})
test("STOP message pattern takes precedence over 429 statusCode", () => {
//#given
const error = {
statusCode: 429,
message: "quota exceeded for this account, usage limit has been reached",
}
//#when
const result = isRetryableModelError(error)
//#then
expect(result).toBe(false)
})
test("rate limit message without statusCode still works (backward compat)", () => {
//#given
const error = { message: "rate limit reached for requests" }
//#when
const result = isRetryableModelError(error)
//#then
expect(result).toBe(true)
})
})
export {}
+19
View File
@@ -99,6 +99,13 @@ const STOP_MESSAGE_PATTERNS = [
"credit balance",
"usage limit for this month",
"exhausted your capacity",
// GLM/Z.ai business error codes that indicate permanent quota/billing exhaustion
"daily call limit",
"daily limit",
"usage limit reached for",
"in arrears",
"fair use policy",
"recharge and try",
]
const AUTO_RETRY_GATE_PATTERNS = [
@@ -117,6 +124,8 @@ function hasProviderAutoRetrySignal(message: string): boolean {
export interface ErrorInfo {
name?: string
message?: string
/** HTTP status code from the provider response (e.g., 429 for rate limit) */
statusCode?: number
}
/**
@@ -151,6 +160,16 @@ export function isRetryableModelError(error: ErrorInfo): boolean {
if (hasProviderAutoRetrySignal(msg)) {
return true
}
// HTTP status code check: catches rate-limit errors regardless of message format/language.
// Uses the same codes as runtime-fallback config (400 excluded as it is a permanent client error).
if (
error.statusCode != null &&
(error.statusCode === 429 || error.statusCode === 503 || error.statusCode === 529)
) {
return true
}
return RETRYABLE_MESSAGE_PATTERNS.some((pattern) => msg.includes(pattern))
}