fix(agents): deny apply_patch for GPT models to prevent verification hangs (#2935)

GPT models (5.3-codex, 5.4, etc.) frequently hang when using apply_patch
due to verification loops. This adds:

1. Tool restriction: apply_patch is denied for GPT variants of
   Hephaestus, Sisyphus-Junior, and Sisyphus agents
2. Prompt guidance: GPT-specific prompts now explicitly instruct using
   edit/write tools instead of apply_patch
3. Removed the 'Always use apply_patch' instruction from
   sisyphus-junior/gpt-5-4.ts that contradicted the fix

The deny is model-conditional — Claude variants retain apply_patch
access since it works reliably there.
This commit is contained in:
YeonGyu-Kim
2026-04-07 11:28:48 +09:00
parent 12106ffccc
commit 1140080927
14 changed files with 100 additions and 6 deletions
+4 -2
View File
@@ -30,6 +30,7 @@ const MODE: AgentMode = "subagent"
// Core tools that Sisyphus-Junior must NEVER have access to
// Note: call_omo_agent is ALLOWED so subagents can spawn explore/librarian
const BLOCKED_TOOLS = ["task"]
const GPT_BLOCKED_TOOLS = ["task", "apply_patch"]
export const SISYPHUS_JUNIOR_DEFAULTS = {
model: "anthropic/claude-sonnet-4-6",
@@ -91,13 +92,14 @@ export function createSisyphusJuniorAgentWithOverrides(
const promptAppend = override?.prompt_append
const prompt = buildSisyphusJuniorPrompt(model, useTaskSystem, promptAppend)
const blockedTools = isGptModel(model) ? GPT_BLOCKED_TOOLS : BLOCKED_TOOLS
const baseRestrictions = createAgentToolRestrictions(BLOCKED_TOOLS)
const baseRestrictions = createAgentToolRestrictions(blockedTools)
const userPermission = (override?.permission ?? {}) as Record<string, PermissionValue>
const basePermission = baseRestrictions.permission
const merged: Record<string, PermissionValue> = { ...userPermission }
for (const tool of BLOCKED_TOOLS) {
for (const tool of blockedTools) {
merged[tool] = "deny"
}
merged.call_omo_agent = "allow"
@@ -92,6 +92,7 @@ Style:
1. SEARCH existing codebase for similar patterns/styles
2. Match naming, indentation, import styles, error handling conventions
3. Default to ASCII. Add comments only for non-obvious blocks
4. Use the \`edit\` and \`write\` tools for file changes. Do not use \`apply_patch\` on GPT models - it is unreliable here and can hang during verification.
### After Implementation (MANDATORY - DO NOT SKIP)
+1 -1
View File
@@ -96,7 +96,7 @@ Style:
1. SEARCH existing codebase for similar patterns/styles
2. Match naming, indentation, import styles, error handling conventions
3. Default to ASCII. Add comments only for non-obvious blocks
4. Always use apply_patch for manual code edits. Do not use cat or echo for file creation/editing. Formatting commands or bulk edits don't need apply_patch
4. Use the \`edit\` and \`write\` tools for file changes. Do not use \`apply_patch\` on GPT models - it is unreliable here and can hang during verification.
5. Do not chain bash commands with separators - each command should be a separate tool call
### After Implementation (MANDATORY - DO NOT SKIP)
+1
View File
@@ -93,6 +93,7 @@ Style:
1. SEARCH existing codebase for similar patterns/styles
2. Match naming, indentation, import styles, error handling conventions
3. Default to ASCII. Add comments only for non-obvious blocks
4. Use the \`edit\` and \`write\` tools for file changes. Do not use \`apply_patch\` on GPT models - it is unreliable here and can hang during verification.
### After Implementation (MANDATORY - DO NOT SKIP)
+30
View File
@@ -350,6 +350,8 @@ describe("createSisyphusJuniorAgentWithOverrides", () => {
expect(result.prompt).toContain("Scope Discipline")
expect(result.prompt).toContain("<tool_usage_rules>")
expect(result.prompt).toContain("Progress Updates")
expect(result.prompt).toContain("Do not use `apply_patch`")
expect(result.prompt).toContain("`edit` and `write`")
})
test("GPT 5.4 model uses GPT-5.4 specific prompt", () => {
@@ -362,6 +364,9 @@ describe("createSisyphusJuniorAgentWithOverrides", () => {
// then
expect(result.prompt).toContain("expert coding agent")
expect(result.prompt).toContain("<tool_usage_rules>")
expect(result.prompt).toContain("Do not use `apply_patch`")
expect(result.prompt).toContain("`edit` and `write`")
expect(result.prompt).not.toContain("Always use apply_patch")
})
test("GPT 5.3 Codex model uses GPT-5.3-codex specific prompt", () => {
@@ -374,6 +379,28 @@ describe("createSisyphusJuniorAgentWithOverrides", () => {
// then
expect(result.prompt).toContain("Senior Engineer")
expect(result.prompt).toContain("<tool_usage_rules>")
expect(result.prompt).toContain("Do not use `apply_patch`")
expect(result.prompt).toContain("`edit` and `write`")
})
test("GPT variants deny apply_patch while Claude variants do not", () => {
// given
const gpt54Override = { model: "openai/gpt-5.4" }
const gpt53Override = { model: "openai/gpt-5.3-codex" }
const gptGenericOverride = { model: "openai/gpt-4o" }
const claudeOverride = { model: "anthropic/claude-sonnet-4-6" }
// when
const gpt54Result = createSisyphusJuniorAgentWithOverrides(gpt54Override)
const gpt53Result = createSisyphusJuniorAgentWithOverrides(gpt53Override)
const gptGenericResult = createSisyphusJuniorAgentWithOverrides(gptGenericOverride)
const claudeResult = createSisyphusJuniorAgentWithOverrides(claudeOverride)
// then
expect(gpt54Result.permission ?? {}).toHaveProperty("apply_patch", "deny")
expect(gpt53Result.permission ?? {}).toHaveProperty("apply_patch", "deny")
expect(gptGenericResult.permission ?? {}).toHaveProperty("apply_patch", "deny")
expect(claudeResult.permission ?? {}).not.toHaveProperty("apply_patch")
})
test("prompt_append is added after base prompt", () => {
@@ -494,6 +521,7 @@ describe("buildSisyphusJuniorPrompt", () => {
expect(prompt).toContain("expert coding agent")
expect(prompt).toContain("Scope Discipline")
expect(prompt).toContain("<tool_usage_rules>")
expect(prompt).toContain("Do not use `apply_patch`")
})
test("GPT 5.3 Codex model uses GPT-5.3-codex prompt", () => {
@@ -507,6 +535,7 @@ describe("buildSisyphusJuniorPrompt", () => {
expect(prompt).toContain("Senior Engineer")
expect(prompt).toContain("Scope Discipline")
expect(prompt).toContain("<tool_usage_rules>")
expect(prompt).toContain("Do not use `apply_patch`")
})
test("generic GPT model uses generic GPT prompt", () => {
@@ -521,6 +550,7 @@ describe("buildSisyphusJuniorPrompt", () => {
expect(prompt).toContain("Scope Discipline")
expect(prompt).toContain("<tool_usage_rules>")
expect(prompt).toContain("Progress Updates")
expect(prompt).toContain("Do not use `apply_patch`")
})
test("Claude model prompt contains Claude-specific sections", () => {