import type { PluginInput } from "@opencode-ai/plugin" import { resolveActualContextLimit, type ContextLimitModelCacheState, } from "../shared/context-limit-resolver" import { isCompactionAgent } from "../shared/compaction-marker" import { resolveMessageEventSessionID, resolveSessionEventID } from "../shared/event-session-id" import { createSystemDirective, SystemDirectiveTypes } from "../shared/system-directive" const CONTEXT_WARNING_THRESHOLD = 0.70 function createContextReminder(actualLimit: number): string { const limitTokens = actualLimit.toLocaleString() return `${createSystemDirective(SystemDirectiveTypes.CONTEXT_WINDOW_MONITOR)} You are using a ${limitTokens}-token context window. You still have context remaining - do NOT rush or skip tasks. Complete your work thoroughly and methodically.` } interface TokenInfo { input: number output: number reasoning: number cache: { read: number; write: number } } interface CachedTokenState { providerID: string modelID: string tokens: TokenInfo } export function createContextWindowMonitorHook( _ctx: PluginInput, modelCacheState?: ContextLimitModelCacheState, ) { const remindedSessions = new Set() const tokenCache = new Map() const toolExecuteAfter = async ( input: { tool: string; sessionID: string; callID: string }, output: { title: string; output: string; metadata: unknown } ) => { const { sessionID } = input if (remindedSessions.has(sessionID)) return const cached = tokenCache.get(sessionID) if (!cached) return const actualLimit = resolveActualContextLimit( cached.providerID, cached.modelID, modelCacheState, ) if (!actualLimit) return const lastTokens = cached.tokens const totalInputTokens = (lastTokens?.input ?? 0) + (lastTokens?.cache?.read ?? 0) const actualUsagePercentage = totalInputTokens / actualLimit if (actualUsagePercentage < CONTEXT_WARNING_THRESHOLD) return remindedSessions.add(sessionID) // Clamp the displayed percentages so the block stays trustworthy when the // resolved actualLimit underestimates the model's real context window // (e.g. a 1M-context Anthropic model that falls back to the 200K default). // Without clamping, the block would advertise >100% used and a negative // "remaining" - safety-tuned models flag exactly that pattern as a prompt // injection and refuse to follow the directive (issue #3655). const clampedPercentage = Math.min(Math.max(actualUsagePercentage, 0), 1) const usedPct = (clampedPercentage * 100).toFixed(1) const remainingPct = ((1 - clampedPercentage) * 100).toFixed(1) const usedTokens = totalInputTokens.toLocaleString() const limitTokens = actualLimit.toLocaleString() output.output += `\n\n${createContextReminder(actualLimit)} [Context Status: ${usedPct}% used (${usedTokens}/${limitTokens} tokens), ${remainingPct}% remaining]` } const eventHandler = async ({ event }: { event: { type: string; properties?: unknown } }) => { const props = event.properties as Record | undefined if (event.type === "session.deleted") { const sessionID = resolveSessionEventID(props) if (sessionID) { remindedSessions.delete(sessionID) tokenCache.delete(sessionID) } } if (event.type === "message.updated") { const info = props?.info as { agent?: unknown role?: string sessionID?: string providerID?: string modelID?: string finish?: boolean tokens?: TokenInfo } | undefined if (!info || info.role !== "assistant" || !info.finish) return if (isCompactionAgent(info.agent)) return const sessionID = resolveMessageEventSessionID(props) if (!sessionID || !info.providerID || !info.tokens) return tokenCache.set(sessionID, { providerID: info.providerID, modelID: info.modelID ?? "", tokens: info.tokens, }) } } return { "tool.execute.after": toolExecuteAfter, event: eventHandler, } }