Merge pull request #3185 from code-yeongyu/fix/issue-2462
fix: detect token-limit errors to prevent todo-continuation infinite loop (#2462)
This commit is contained in:
@@ -26,6 +26,7 @@ import {
|
|||||||
} from "./constants"
|
} from "./constants"
|
||||||
import { isCompactionGuardActive } from "./compaction-guard"
|
import { isCompactionGuardActive } from "./compaction-guard"
|
||||||
import { getMessageDir } from "./message-directory"
|
import { getMessageDir } from "./message-directory"
|
||||||
|
import { isTokenLimitError } from "./token-limit-detection"
|
||||||
import { getIncompleteCount } from "./todo"
|
import { getIncompleteCount } from "./todo"
|
||||||
import type { ResolvedMessageInfo, Todo } from "./types"
|
import type { ResolvedMessageInfo, Todo } from "./types"
|
||||||
import type { SessionStateStore } from "./session-state"
|
import type { SessionStateStore } from "./session-state"
|
||||||
@@ -204,6 +205,14 @@ ${todoList}`
|
|||||||
injectionState.inFlight = false
|
injectionState.inFlight = false
|
||||||
injectionState.lastInjectedAt = Date.now()
|
injectionState.lastInjectedAt = Date.now()
|
||||||
injectionState.consecutiveFailures = (injectionState.consecutiveFailures ?? 0) + 1
|
injectionState.consecutiveFailures = (injectionState.consecutiveFailures ?? 0) + 1
|
||||||
|
|
||||||
|
const errorObj = error instanceof Error
|
||||||
|
? { name: error.name, message: error.message }
|
||||||
|
: { message: String(error) }
|
||||||
|
if (isTokenLimitError(errorObj)) {
|
||||||
|
injectionState.tokenLimitDetected = true
|
||||||
|
log(`[${HOOK_NAME}] Token limit error detected during injection, stopping continuation`, { sessionID })
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -11,6 +11,7 @@ import { armCompactionGuard } from "./compaction-guard"
|
|||||||
import type { SessionStateStore } from "./session-state"
|
import type { SessionStateStore } from "./session-state"
|
||||||
import { handleSessionIdle } from "./idle-event"
|
import { handleSessionIdle } from "./idle-event"
|
||||||
import { handleNonIdleEvent } from "./non-idle-events"
|
import { handleNonIdleEvent } from "./non-idle-events"
|
||||||
|
import { isTokenLimitError } from "./token-limit-detection"
|
||||||
|
|
||||||
export function createTodoContinuationHandler(args: {
|
export function createTodoContinuationHandler(args: {
|
||||||
ctx: PluginInput
|
ctx: PluginInput
|
||||||
@@ -34,7 +35,7 @@ export function createTodoContinuationHandler(args: {
|
|||||||
const sessionID = props?.sessionID as string | undefined
|
const sessionID = props?.sessionID as string | undefined
|
||||||
if (!sessionID) return
|
if (!sessionID) return
|
||||||
|
|
||||||
const error = props?.error as { name?: string } | undefined
|
const error = props?.error as { name?: string; message?: string } | undefined
|
||||||
if (error?.name === "MessageAbortedError" || error?.name === "AbortError") {
|
if (error?.name === "MessageAbortedError" || error?.name === "AbortError") {
|
||||||
const state = sessionStateStore.getState(sessionID)
|
const state = sessionStateStore.getState(sessionID)
|
||||||
state.wasCancelled = true
|
state.wasCancelled = true
|
||||||
@@ -45,6 +46,10 @@ export function createTodoContinuationHandler(args: {
|
|||||||
state.stagnationCount = 0
|
state.stagnationCount = 0
|
||||||
state.consecutiveFailures = 0
|
state.consecutiveFailures = 0
|
||||||
log(`[${HOOK_NAME}] Abort detected via session.error`, { sessionID, errorName: error.name })
|
log(`[${HOOK_NAME}] Abort detected via session.error`, { sessionID, errorName: error.name })
|
||||||
|
} else if (isTokenLimitError(error)) {
|
||||||
|
const state = sessionStateStore.getState(sessionID)
|
||||||
|
state.tokenLimitDetected = true
|
||||||
|
log(`[${HOOK_NAME}] Token limit error detected via session.error`, { sessionID, errorName: error?.name, errorMessage: error?.message })
|
||||||
}
|
}
|
||||||
|
|
||||||
sessionStateStore.cancelCountdown(sessionID)
|
sessionStateStore.cancelCountdown(sessionID)
|
||||||
|
|||||||
@@ -55,6 +55,11 @@ export async function handleSessionIdle(args: {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
|
if (state.tokenLimitDetected) {
|
||||||
|
log(`[${HOOK_NAME}] Skipped: token limit error detected, retry would worsen context overflow`, { sessionID })
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
if (state.abortDetectedAt) {
|
if (state.abortDetectedAt) {
|
||||||
const timeSinceAbort = Date.now() - state.abortDetectedAt
|
const timeSinceAbort = Date.now() - state.abortDetectedAt
|
||||||
if (timeSinceAbort < ABORT_WINDOW_MS) {
|
if (timeSinceAbort < ABORT_WINDOW_MS) {
|
||||||
|
|||||||
@@ -28,6 +28,7 @@ export function handleNonIdleEvent(args: {
|
|||||||
if (state) {
|
if (state) {
|
||||||
state.abortDetectedAt = undefined
|
state.abortDetectedAt = undefined
|
||||||
state.wasCancelled = false
|
state.wasCancelled = false
|
||||||
|
state.tokenLimitDetected = false
|
||||||
sessionStateStore.recordActivity(sessionID)
|
sessionStateStore.recordActivity(sessionID)
|
||||||
}
|
}
|
||||||
sessionStateStore.cancelCountdown(sessionID)
|
sessionStateStore.cancelCountdown(sessionID)
|
||||||
|
|||||||
@@ -1962,5 +1962,188 @@ describe("todo-continuation-enforcer", () => {
|
|||||||
expect(promptCalls).toHaveLength(1)
|
expect(promptCalls).toHaveLength(1)
|
||||||
}, { timeout: 20000 })
|
}, { timeout: 20000 })
|
||||||
|
|
||||||
|
// ============================================================
|
||||||
|
// TOKEN-LIMIT ERROR DETECTION TESTS (#2462)
|
||||||
|
// These tests verify that the enforcer does NOT retry continuation
|
||||||
|
// when the model returns a token-limit / context-length error.
|
||||||
|
// ============================================================
|
||||||
|
|
||||||
|
test("should stop continuation when session.error carries a ContextLengthError", async () => {
|
||||||
|
// given - session with incomplete todos
|
||||||
|
const sessionID = "main-token-limit-event"
|
||||||
|
setMainSession(sessionID)
|
||||||
|
mockMessages = [
|
||||||
|
{ info: { id: "msg-1", role: "user" } },
|
||||||
|
{ info: { id: "msg-2", role: "assistant" } },
|
||||||
|
]
|
||||||
|
|
||||||
|
const hook = createTodoContinuationEnforcer(createMockPluginInput(), {})
|
||||||
|
|
||||||
|
// when - token limit error event fires
|
||||||
|
await hook.handler({
|
||||||
|
event: {
|
||||||
|
type: "session.error",
|
||||||
|
properties: {
|
||||||
|
sessionID,
|
||||||
|
error: { name: "ContextLengthError", message: "prompt is too long: 250000 tokens > 200000 maximum" },
|
||||||
|
},
|
||||||
|
},
|
||||||
|
})
|
||||||
|
|
||||||
|
// when - session goes idle
|
||||||
|
await hook.handler({
|
||||||
|
event: { type: "session.idle", properties: { sessionID } },
|
||||||
|
})
|
||||||
|
|
||||||
|
await fakeTimers.advanceBy(3000)
|
||||||
|
|
||||||
|
// then - no continuation injected (token limit error blocks retry)
|
||||||
|
expect(promptCalls).toHaveLength(0)
|
||||||
|
})
|
||||||
|
|
||||||
|
test("should stop continuation when session.error message contains token limit keywords", async () => {
|
||||||
|
// given - session with incomplete todos
|
||||||
|
const sessionID = "main-token-limit-message"
|
||||||
|
setMainSession(sessionID)
|
||||||
|
mockMessages = [
|
||||||
|
{ info: { id: "msg-1", role: "user" } },
|
||||||
|
{ info: { id: "msg-2", role: "assistant" } },
|
||||||
|
]
|
||||||
|
|
||||||
|
const hook = createTodoContinuationEnforcer(createMockPluginInput(), {})
|
||||||
|
|
||||||
|
// when - error with token limit message fires (no specific error name)
|
||||||
|
await hook.handler({
|
||||||
|
event: {
|
||||||
|
type: "session.error",
|
||||||
|
properties: {
|
||||||
|
sessionID,
|
||||||
|
error: { name: "APIError", message: "context_length_exceeded: the prompt is too long" },
|
||||||
|
},
|
||||||
|
},
|
||||||
|
})
|
||||||
|
|
||||||
|
// when - session goes idle
|
||||||
|
await hook.handler({
|
||||||
|
event: { type: "session.idle", properties: { sessionID } },
|
||||||
|
})
|
||||||
|
|
||||||
|
await fakeTimers.advanceBy(3000)
|
||||||
|
|
||||||
|
// then - no continuation injected
|
||||||
|
expect(promptCalls).toHaveLength(0)
|
||||||
|
})
|
||||||
|
|
||||||
|
test("should stop continuation when promptAsync throws a token-limit error", async () => {
|
||||||
|
// given - session where promptAsync will throw a token limit error
|
||||||
|
const sessionID = "main-token-limit-injection"
|
||||||
|
setMainSession(sessionID)
|
||||||
|
const mockInput = createMockPluginInput()
|
||||||
|
mockInput.client.session.promptAsync = async () => {
|
||||||
|
const error = new Error("prompt is too long: 150000 tokens > 100000 maximum")
|
||||||
|
;(error as any).name = "ContextLengthError"
|
||||||
|
throw error
|
||||||
|
}
|
||||||
|
|
||||||
|
const hook = createTodoContinuationEnforcer(mockInput, {})
|
||||||
|
|
||||||
|
// when - first idle triggers injection that fails with token limit
|
||||||
|
await hook.handler({
|
||||||
|
event: { type: "session.idle", properties: { sessionID } },
|
||||||
|
})
|
||||||
|
await fakeTimers.advanceBy(2500, true)
|
||||||
|
|
||||||
|
// when - wait past any cooldown, try again
|
||||||
|
await fakeTimers.advanceClockBy(CONTINUATION_COOLDOWN_MS * 100)
|
||||||
|
await hook.handler({
|
||||||
|
event: { type: "session.idle", properties: { sessionID } },
|
||||||
|
})
|
||||||
|
await fakeTimers.advanceBy(3000, true)
|
||||||
|
|
||||||
|
// then - no second injection attempt (token limit permanently stops continuation)
|
||||||
|
expect(promptCalls).toHaveLength(0)
|
||||||
|
})
|
||||||
|
|
||||||
|
test("should still allow retries for non-token-limit errors (existing behavior)", async () => {
|
||||||
|
// given - session where promptAsync throws a generic error
|
||||||
|
const sessionID = "main-generic-error-retry"
|
||||||
|
setMainSession(sessionID)
|
||||||
|
let callCount = 0
|
||||||
|
const mockInput = createMockPluginInput()
|
||||||
|
mockInput.client.session.promptAsync = async (opts: any) => {
|
||||||
|
callCount++
|
||||||
|
if (callCount === 1) {
|
||||||
|
throw new Error("simulated network error")
|
||||||
|
}
|
||||||
|
promptCalls.push({
|
||||||
|
sessionID: opts.path.id,
|
||||||
|
agent: opts.body.agent,
|
||||||
|
model: opts.body.model,
|
||||||
|
text: opts.body.parts[0].text,
|
||||||
|
})
|
||||||
|
return {}
|
||||||
|
}
|
||||||
|
|
||||||
|
const hook = createTodoContinuationEnforcer(mockInput, {})
|
||||||
|
|
||||||
|
// when - first idle triggers injection that fails with generic error
|
||||||
|
await hook.handler({
|
||||||
|
event: { type: "session.idle", properties: { sessionID } },
|
||||||
|
})
|
||||||
|
await fakeTimers.advanceBy(2500, true)
|
||||||
|
|
||||||
|
// when - wait past cooldown, try again
|
||||||
|
await fakeTimers.advanceClockBy(CONTINUATION_COOLDOWN_MS * 2)
|
||||||
|
await hook.handler({
|
||||||
|
event: { type: "session.idle", properties: { sessionID } },
|
||||||
|
})
|
||||||
|
await fakeTimers.advanceBy(2500, true)
|
||||||
|
|
||||||
|
// then - second attempt succeeds (generic errors still allow retry)
|
||||||
|
expect(callCount).toBe(2)
|
||||||
|
expect(promptCalls).toHaveLength(1)
|
||||||
|
}, { timeout: 30000 })
|
||||||
|
|
||||||
|
test("should clear token limit flag when user sends new message after recovery", async () => {
|
||||||
|
fakeTimers.restore()
|
||||||
|
// given - session that hit token limit
|
||||||
|
const sessionID = "main-token-limit-recovery"
|
||||||
|
setMainSession(sessionID)
|
||||||
|
mockMessages = [
|
||||||
|
{ info: { id: "msg-1", role: "user" } },
|
||||||
|
{ info: { id: "msg-2", role: "assistant" } },
|
||||||
|
]
|
||||||
|
|
||||||
|
const hook = createTodoContinuationEnforcer(createMockPluginInput(), {})
|
||||||
|
|
||||||
|
// when - token limit error fires
|
||||||
|
await hook.handler({
|
||||||
|
event: {
|
||||||
|
type: "session.error",
|
||||||
|
properties: {
|
||||||
|
sessionID,
|
||||||
|
error: { name: "ContextLengthError", message: "prompt is too long" },
|
||||||
|
},
|
||||||
|
},
|
||||||
|
})
|
||||||
|
|
||||||
|
// when - user sends new message (clears token limit flag via activity)
|
||||||
|
await hook.handler({
|
||||||
|
event: {
|
||||||
|
type: "message.updated",
|
||||||
|
properties: { info: { sessionID, role: "user" } },
|
||||||
|
},
|
||||||
|
})
|
||||||
|
|
||||||
|
// when - session goes idle
|
||||||
|
await hook.handler({
|
||||||
|
event: { type: "session.idle", properties: { sessionID } },
|
||||||
|
})
|
||||||
|
|
||||||
|
await wait(2500)
|
||||||
|
|
||||||
|
// then - continuation injected (token limit flag cleared by user activity)
|
||||||
|
expect(promptCalls.length).toBe(1)
|
||||||
|
}, { timeout: 15000 })
|
||||||
|
|
||||||
})
|
})
|
||||||
|
|||||||
@@ -0,0 +1,27 @@
|
|||||||
|
const TOKEN_LIMIT_ERROR_NAMES = new Set([
|
||||||
|
"contextlengtherror",
|
||||||
|
])
|
||||||
|
|
||||||
|
const TOKEN_LIMIT_KEYWORDS = [
|
||||||
|
"prompt is too long",
|
||||||
|
"is too long",
|
||||||
|
"context_length_exceeded",
|
||||||
|
"token limit",
|
||||||
|
"context length",
|
||||||
|
"too many tokens",
|
||||||
|
]
|
||||||
|
|
||||||
|
export function isTokenLimitError(error: { name?: string; message?: string } | undefined): boolean {
|
||||||
|
if (!error) return false
|
||||||
|
|
||||||
|
if (error.name && TOKEN_LIMIT_ERROR_NAMES.has(error.name.toLowerCase())) {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
|
||||||
|
if (error.message) {
|
||||||
|
const lower = error.message.toLowerCase()
|
||||||
|
return TOKEN_LIMIT_KEYWORDS.some((keyword) => lower.includes(keyword))
|
||||||
|
}
|
||||||
|
|
||||||
|
return false
|
||||||
|
}
|
||||||
@@ -27,6 +27,7 @@ export interface SessionState {
|
|||||||
countdownInterval?: ReturnType<typeof setInterval>
|
countdownInterval?: ReturnType<typeof setInterval>
|
||||||
isRecovering?: boolean
|
isRecovering?: boolean
|
||||||
wasCancelled?: boolean
|
wasCancelled?: boolean
|
||||||
|
tokenLimitDetected?: boolean
|
||||||
countdownStartedAt?: number
|
countdownStartedAt?: number
|
||||||
abortDetectedAt?: number
|
abortDetectedAt?: number
|
||||||
lastIncompleteCount?: number
|
lastIncompleteCount?: number
|
||||||
|
|||||||
Reference in New Issue
Block a user