fix(background-task): clarify task id contracts
This commit is contained in:
@@ -51,7 +51,7 @@ describe("buildAntiDuplicationSection", () => {
|
||||
expect(result).toContain("Wait for Results Properly")
|
||||
expect(result).toContain("End your response")
|
||||
expect(result).toContain("Wait for the completion notification")
|
||||
expect(result).toContain("background_output")
|
||||
expect(result).toContain('background_output(task_id="bg_...")')
|
||||
})
|
||||
|
||||
it("#given no arguments #when building #then explains why this matters", () => {
|
||||
|
||||
@@ -126,6 +126,13 @@ describe("Atlas prompts use task_id (not session_id) for retries", () => {
|
||||
expect(prompt, `${name}: missing task_id retry reference`).toMatch(/task_id/)
|
||||
}
|
||||
})
|
||||
|
||||
test("all variants should separate background ids from continuation task ids", () => {
|
||||
for (const [name, prompt] of ALL_VARIANTS) {
|
||||
expect(prompt, `${name}: missing bg result collection contract`).toContain('background_output(task_id="bg_...")')
|
||||
expect(prompt, `${name}: missing ses continuation contract`).toContain('task(task_id="ses_..."')
|
||||
}
|
||||
})
|
||||
})
|
||||
|
||||
describe("Atlas prompts no-excuses retry policy", () => {
|
||||
|
||||
@@ -243,6 +243,6 @@ export const DEFAULT_ATLAS_CRITICAL_RULES = `<critical_overrides>
|
||||
- Run lsp_diagnostics after every delegation
|
||||
- Pass inherited wisdom to every subagent
|
||||
- Verify with your own tools
|
||||
- **Store task_id from every delegation output**
|
||||
- **Use \`task_id="{task_id}"\` for retries, fixes, and follow-ups**
|
||||
- **Store continuation task_id (\`ses_...\`) from every delegation output**
|
||||
- **Use \`task(task_id="ses_...", prompt="...")\` for retries, fixes, and follow-ups**
|
||||
</critical_overrides>`
|
||||
|
||||
@@ -216,6 +216,6 @@ export const KIMI_ATLAS_CRITICAL_RULES = `<critical_overrides>
|
||||
- Run lsp_diagnostics after every delegation
|
||||
- Pass inherited wisdom to every subagent
|
||||
- Verify with your own tools
|
||||
- **Store task_id from every delegation output**
|
||||
- **Use \`task_id="{task_id}"\` for retries, fixes, and follow-ups**
|
||||
- **Store continuation task_id (\`ses_...\`) from every delegation output**
|
||||
- **Use \`task(task_id="ses_...", prompt="...")\` for retries, fixes, and follow-ups**
|
||||
</critical_overrides>`
|
||||
|
||||
@@ -232,6 +232,6 @@ export const OPUS_47_ATLAS_CRITICAL_RULES = `<critical_overrides>
|
||||
- Run lsp_diagnostics after every delegation
|
||||
- Pass inherited wisdom to every subagent
|
||||
- Verify with your own tools
|
||||
- **Store task_id from every delegation output**
|
||||
- **Use \`task_id="{task_id}"\` for retries, fixes, and follow-ups**
|
||||
- **Store continuation task_id (\`ses_...\`) from every delegation output**
|
||||
- **Use \`task(task_id="ses_...", prompt="...")\` for retries, fixes, and follow-ups**
|
||||
</critical_overrides>`
|
||||
|
||||
@@ -120,7 +120,8 @@ task(category="quick", load_skills=[], run_in_background=false, prompt="...task
|
||||
- **Task execution** (\`category="..."\`): \`run_in_background=false\` — blocks for verification
|
||||
|
||||
**Background management:**
|
||||
- Collect: \`background_output(task_id="...")\`
|
||||
- Collect with background task IDs (\`bg_...\`): \`background_output(task_id="bg_...")\`
|
||||
- Continue follow-ups with continuation task IDs (\`ses_...\`): \`task(task_id="ses_...")\`
|
||||
- Cancel DISPOSABLE background tasks individually before final answer: \`background_cancel(taskId="bg_explore_xxx")\`
|
||||
- **NEVER \`background_cancel(all=true)\`** — it kills tasks whose output you have not collected.
|
||||
</parallel_by_default>`
|
||||
|
||||
@@ -148,7 +148,7 @@ When you need the delegated results but they're not ready:
|
||||
|
||||
1. **End your response** - do NOT continue with work that depends on those results
|
||||
2. **Wait for the completion notification** - the system will trigger your next turn
|
||||
3. **Then** collect results via \`background_output(task_id="...")\`
|
||||
3. **Then** collect results via \`background_output(task_id="bg_...")\`
|
||||
4. **Do NOT** impatiently re-search the same topics while waiting
|
||||
|
||||
### Why This Matters:
|
||||
|
||||
@@ -0,0 +1,30 @@
|
||||
/// <reference types="bun-types" />
|
||||
|
||||
import { describe, expect, test } from "bun:test"
|
||||
import { buildHephaestusPrompt as buildGptHephaestusPrompt } from "./hephaestus/gpt"
|
||||
import { buildHephaestusPrompt as buildGpt53CodexHephaestusPrompt } from "./hephaestus/gpt-5-3-codex"
|
||||
import { buildHephaestusPrompt as buildGpt54HephaestusPrompt } from "./hephaestus/gpt-5-4"
|
||||
import { buildGpt55HephaestusPrompt } from "./hephaestus/gpt-5-5"
|
||||
|
||||
describe("Hephaestus background task ID guidance", () => {
|
||||
const promptBuilders = [
|
||||
["gpt", () => buildGptHephaestusPrompt()],
|
||||
["gpt-5.3-codex", () => buildGpt53CodexHephaestusPrompt()],
|
||||
["gpt-5.4", () => buildGpt54HephaestusPrompt()],
|
||||
["gpt-5.5", () => buildGpt55HephaestusPrompt([])],
|
||||
] as const
|
||||
|
||||
for (const [name, buildPrompt] of promptBuilders) {
|
||||
test(`#given ${name} prompt #when describing task follow-ups #then bg ids and continuation ids are disambiguated`, () => {
|
||||
// given, when
|
||||
const prompt = buildPrompt()
|
||||
|
||||
// then
|
||||
expect(prompt).toContain("background task IDs (`bg_...`)")
|
||||
expect(prompt).toContain("continuation IDs (`ses_...`)")
|
||||
expect(prompt).toContain("background_output(task_id=\"bg_...\")")
|
||||
expect(prompt).toContain("task(task_id=\"ses_...\")")
|
||||
expect(prompt).not.toContain("returns a task_id")
|
||||
})
|
||||
}
|
||||
})
|
||||
@@ -299,7 +299,7 @@ Prompt structure for each agent:
|
||||
- Parallelize independent file reads - don't read files one at a time
|
||||
- NEVER use \`run_in_background=false\` for explore/librarian
|
||||
- Continue only with non-overlapping work after launching background agents
|
||||
- Collect results with \`background_output(task_id="...")\` when needed
|
||||
- Keep IDs separate: collect results with background task IDs (\`bg_...\`) via \`background_output(task_id="bg_...")\`; continue follow-up sessions with continuation IDs (\`ses_...\`) via \`task(task_id="ses_...")\`
|
||||
- BEFORE final answer, cancel DISPOSABLE tasks individually: \`background_cancel(taskId="bg_explore_xxx")\`, \`background_cancel(taskId="bg_librarian_xxx")\`
|
||||
- **NEVER use \`background_cancel(all=true)\`** - it kills tasks whose results you haven't collected yet
|
||||
|
||||
@@ -407,11 +407,11 @@ After delegation, ALWAYS verify: works as expected? follows codebase pattern? MU
|
||||
|
||||
### Session Continuity
|
||||
|
||||
Every \`task()\` output includes a session_id. **USE IT for follow-ups.**
|
||||
Every \`task()\` output includes a continuation ID (\`ses_...\`). **USE IT for follow-ups.**
|
||||
|
||||
- **Task failed/incomplete** - \`task_id="{id}", prompt="Fix: {error}"\`
|
||||
- **Follow-up on result** - \`task_id="{id}", prompt="Also: {question}"\`
|
||||
- **Verification failed** - \`task_id="{id}", prompt="Failed: {error}. Fix."\`
|
||||
- **Task failed/incomplete** - \`task(task_id="ses_...", prompt="Fix: {error}")\`
|
||||
- **Follow-up on result** - \`task(task_id="ses_...", prompt="Also: {question}")\`
|
||||
- **Verification failed** - \`task(task_id="ses_...", prompt="Failed: {error}. Fix.")\`
|
||||
|
||||
${
|
||||
oracleSection
|
||||
|
||||
@@ -111,6 +111,8 @@ export function buildHephaestusPrompt(
|
||||
const identityBlock = `<identity>
|
||||
You are Hephaestus, an autonomous deep worker for software engineering.
|
||||
|
||||
ID contract: background task IDs (\`bg_...\`) use \`background_output(task_id="bg_...")\`; continuation IDs (\`ses_...\`) use \`task(task_id="ses_...")\`.
|
||||
|
||||
You communicate warmly and directly, like a senior colleague walking through a problem together. You explain the why behind decisions, not just the what. You stay concise in volume but generous in clarity - every sentence carries meaning.
|
||||
|
||||
You build context by examining the codebase first without assumptions. You think through the nuances of the code you encounter. You persist until the task is fully handled end-to-end, even when tool calls fail. You only end your turn when the problem is solved and verified.
|
||||
@@ -234,7 +236,7 @@ Agent prompt structure:
|
||||
- [REQUEST]: What to find, format to return, what to skip
|
||||
|
||||
Background task management:
|
||||
- Collect results with \`background_output(task_id="...")\` when completed
|
||||
- Keep IDs separate: collect results with background task IDs (\`bg_...\`) via \`background_output(task_id="bg_...")\`; continue follow-up sessions with continuation IDs (\`ses_...\`) via \`task(task_id="ses_...")\`
|
||||
- Before final answer, cancel disposable tasks individually: \`background_cancel(taskId="...")\`
|
||||
- Never use \`background_cancel(all=true)\` - it kills tasks whose results you have not collected yet
|
||||
|
||||
@@ -312,10 +314,10 @@ Every delegation prompt needs these 6 sections:
|
||||
After delegation, verify by reading every file the subagent touched. Check: works as expected? follows codebase pattern? Do not trust self-reports.
|
||||
|
||||
<session_continuity>
|
||||
Every \`task()\` returns a task_id. Use it for all follow-ups:
|
||||
- Task failed/incomplete: \`task_id="{id}", prompt="Fix: {error}"\`
|
||||
- Follow-up on result: \`task_id="{id}", prompt="Also: {question}"\`
|
||||
- Verification failed: \`task_id="{id}", prompt="Failed: {error}. Fix."\`
|
||||
Every \`task()\` output includes a continuation ID (\`ses_...\`). Use it for all follow-ups:
|
||||
- Task failed/incomplete: \`task(task_id="ses_...", prompt="Fix: {error}")\`
|
||||
- Follow-up on result: \`task(task_id="ses_...", prompt="Also: {question}")\`
|
||||
- Verification failed: \`task(task_id="ses_...", prompt="Failed: {error}. Fix.")\`
|
||||
|
||||
This preserves full context, avoids repeated exploration, saves 70%+ tokens.
|
||||
</session_continuity>
|
||||
|
||||
@@ -22,6 +22,8 @@ function buildTaskSystemGuide(useTaskSystem: boolean): string {
|
||||
|
||||
const HEPHAESTUS_GPT_5_5_TEMPLATE = `You are Hephaestus, an autonomous deep worker based on GPT-5.5. You and the user share one workspace. You receive goals, not step-by-step instructions, and execute them end-to-end.
|
||||
|
||||
ID contract: background task IDs (\`bg_...\`) use \`background_output(task_id="bg_...")\`; continuation IDs (\`ses_...\`) use \`task(task_id="ses_...")\`.
|
||||
|
||||
# Tone
|
||||
|
||||
Warm but spare. Communicate efficiently - enough context for the user to trust the work, then stop. No flattery, no narration, no padding. Acknowledge real progress briefly; never invent it.
|
||||
@@ -172,7 +174,7 @@ AGENTS.md files in your context carry directory-scoped conventions. Obey them fo
|
||||
**\`task()\`** for both research sub-agents and category-based delegation. Allowed: \`subagent_type="explore"\`, \`"librarian"\`, \`"oracle"\`, or \`category="..."\`.
|
||||
|
||||
- Every \`task()\` call needs \`load_skills\` (an empty array \`[]\` is valid).
|
||||
- Reuse \`task_id\` for follow-ups; never start a fresh session on a continuation. Saves 70%+ of tokens and preserves the sub-agent's full context.
|
||||
- Reuse continuation IDs (\`ses_...\`) for follow-ups via \`task(task_id="ses_...")\`; never pass background task IDs (\`bg_...\`) to \`task()\`. Saves 70%+ of tokens and preserves the sub-agent's full context.
|
||||
|
||||
Each sub-agent prompt should include four fields:
|
||||
|
||||
@@ -181,7 +183,7 @@ Each sub-agent prompt should include four fields:
|
||||
- **DOWNSTREAM**: how you will use the results.
|
||||
- **REQUEST**: what to find, what format to return, what to skip.
|
||||
|
||||
**Background tasks.** Collect with \`background_output(task_id="...")\` once they complete. Before the final answer, cancel disposable tasks individually via \`background_cancel(taskId="...")\`. Never use \`background_cancel(all=true)\` - it kills tasks whose results you have not collected.
|
||||
**Background tasks.** Collect with background task IDs (\`bg_...\`) via \`background_output(task_id="bg_...")\` once they complete. Use continuation IDs (\`ses_...\`) only for \`task(task_id="ses_...")\` follow-ups. Before the final answer, cancel disposable tasks individually via \`background_cancel(taskId="bg_...")\`. Never use \`background_cancel(all=true)\` - it kills tasks whose results you have not collected.
|
||||
|
||||
**\`skill\`** loads specialized instruction packs. Load a skill whenever its declared domain even loosely connects to your current task. Loading an irrelevant skill costs almost nothing; missing a relevant one degrades the work measurably.
|
||||
|
||||
|
||||
@@ -201,7 +201,7 @@ task(subagent_type="librarian", run_in_background=true, load_skills=[], descript
|
||||
- Parallelize independent file reads - don't read files one at a time
|
||||
- NEVER use \`run_in_background=false\` for explore/librarian
|
||||
- Continue only with non-overlapping work after launching background agents
|
||||
- Collect results with \`background_output(task_id="...")\` when needed
|
||||
- Keep IDs separate: collect results with background task IDs (\`bg_...\`) via \`background_output(task_id="bg_...")\`; continue follow-up sessions with continuation IDs (\`ses_...\`) via \`task(task_id="ses_...")\`
|
||||
- BEFORE final answer, cancel DISPOSABLE tasks individually
|
||||
- **NEVER use \`background_cancel(all=true)\`**
|
||||
|
||||
@@ -277,11 +277,11 @@ After delegation, ALWAYS verify: works as expected? follows codebase pattern? MU
|
||||
|
||||
### Session Continuity
|
||||
|
||||
Every \`task()\` output includes a task_id. **USE IT for follow-ups.**
|
||||
Every \`task()\` output includes a continuation ID (\`ses_...\`). **USE IT for follow-ups.**
|
||||
|
||||
- **Task failed/incomplete** - \`task_id="{id}", prompt="Fix: {error}"\`
|
||||
- **Follow-up on result** - \`task_id="{id}", prompt="Also: {question}"\`
|
||||
- **Verification failed** - \`task_id="{id}", prompt="Failed: {error}. Fix."\`
|
||||
- **Task failed/incomplete** - \`task(task_id="ses_...", prompt="Fix: {error}")\`
|
||||
- **Follow-up on result** - \`task(task_id="ses_...", prompt="Also: {question}")\`
|
||||
- **Verification failed** - \`task(task_id="ses_...", prompt="Failed: {error}. Fix.")\`
|
||||
|
||||
${
|
||||
oracleSection
|
||||
|
||||
@@ -0,0 +1,32 @@
|
||||
/// <reference types="bun-types" />
|
||||
|
||||
import { describe, expect, test } from "bun:test"
|
||||
import { buildClaudeOpus47SisyphusPrompt } from "./sisyphus/claude-opus-4-7"
|
||||
import { buildDefaultSisyphusPrompt } from "./sisyphus/default"
|
||||
import { buildGpt54SisyphusPrompt } from "./sisyphus/gpt-5-4"
|
||||
import { buildGpt55SisyphusPrompt } from "./sisyphus/gpt-5-5"
|
||||
import { buildKimiK26SisyphusPrompt } from "./sisyphus/kimi-k2-6"
|
||||
|
||||
describe("Sisyphus background task ID guidance", () => {
|
||||
const promptBuilders = [
|
||||
["claude-opus-4-7", buildClaudeOpus47SisyphusPrompt],
|
||||
["default", buildDefaultSisyphusPrompt],
|
||||
["gpt-5.4", buildGpt54SisyphusPrompt],
|
||||
["gpt-5.5", buildGpt55SisyphusPrompt],
|
||||
["kimi-k2.6", buildKimiK26SisyphusPrompt],
|
||||
] as const
|
||||
|
||||
for (const [name, buildPrompt] of promptBuilders) {
|
||||
test(`#given ${name} prompt #when describing background tasks #then bg ids and session ids are disambiguated`, () => {
|
||||
// given, when
|
||||
const prompt = buildPrompt(name, [])
|
||||
|
||||
// then
|
||||
expect(prompt).toContain("background task IDs (`bg_...`)")
|
||||
expect(prompt).toContain("continuation session IDs (`ses_...`)")
|
||||
expect(prompt).toContain("background_output(task_id=\"bg_...\")")
|
||||
expect(prompt).toContain("task(task_id=\"ses_...\")")
|
||||
expect(prompt).not.toContain("receive task_ids")
|
||||
})
|
||||
}
|
||||
})
|
||||
+12
-9
@@ -266,14 +266,15 @@ result = task(..., run_in_background=false) // Never wait synchronously for exp
|
||||
\`\`\`
|
||||
|
||||
### Background Result Collection:
|
||||
1. Launch parallel agents \u2192 receive task_ids
|
||||
1. Launch parallel agents \u2192 receive background task IDs (\`bg_...\`) for results and continuation session IDs (\`ses_...\`) for follow-ups
|
||||
2. Continue only with non-overlapping work
|
||||
- If you have DIFFERENT independent work \u2192 do it now
|
||||
- Otherwise \u2192 **END YOUR RESPONSE.**
|
||||
3. **STOP. END YOUR RESPONSE.** The system will send \`<system-reminder>\` when tasks complete.
|
||||
4. On receiving \`<system-reminder>\` \u2192 collect results via \`background_output(task_id="...")\`
|
||||
4. On receiving \`<system-reminder>\` \u2192 collect results via \`background_output(task_id="bg_...")\`
|
||||
5. **NEVER call \`background_output\` before receiving \`<system-reminder>\`.** This is a BLOCKING anti-pattern.
|
||||
6. Cleanup: Cancel disposable tasks individually via \`background_cancel(taskId="...")\`
|
||||
7. Use \`task(task_id="ses_...")\` only to continue the same sub-agent session
|
||||
|
||||
${buildAntiDuplicationSection()}
|
||||
|
||||
@@ -328,15 +329,17 @@ AFTER THE WORK YOU DELEGATED SEEMS DONE, ALWAYS VERIFY THE RESULTS AS FOLLOWING:
|
||||
|
||||
### Session Continuity (MANDATORY)
|
||||
|
||||
Every \`task()\` output includes a task_id. **USE IT.**
|
||||
Every \`task()\` output exposes a continuation session ID (\`ses_...\`). Pass it to \`task(task_id="ses_...")\` for follow-ups. **USE IT.**
|
||||
|
||||
**ALWAYS continue when:**
|
||||
- Task failed/incomplete → \`task_id=\"{task_id}\", prompt=\"Fix: {specific error}\"\`
|
||||
- Follow-up question on result → \`task_id=\"{task_id}\", prompt=\"Also: {question}\"\`
|
||||
- Multi-turn with same agent → \`task_id=\"{task_id}\"\` - NEVER start fresh
|
||||
- Verification failed → \`task_id=\"{task_id}\", prompt=\"Failed verification: {error}. Fix.\"\`
|
||||
- Task failed/incomplete → \`task(task_id="ses_...", prompt="Fix: {specific error}")\`
|
||||
- Follow-up question on result → \`task(task_id="ses_...", prompt="Also: {question}")\`
|
||||
- Multi-turn with same agent → \`task(task_id="ses_...")\` - NEVER start fresh
|
||||
- Verification failed → \`task(task_id="ses_...", prompt="Failed verification: {error}. Fix.")\`
|
||||
|
||||
**Why task_id is CRITICAL:**
|
||||
**Keep IDs separate:** background task IDs (\`bg_...\`) are for \`background_output(task_id="bg_...")\`; continuation session IDs (\`ses_...\`) are for \`task(task_id="ses_...")\`.
|
||||
|
||||
**Why continuation is CRITICAL:**
|
||||
- Subagent has FULL conversation context preserved
|
||||
- No repeated file reads, exploration, or setup
|
||||
- Saves 70%+ tokens on follow-ups
|
||||
@@ -350,7 +353,7 @@ task(category="quick", load_skills=[], run_in_background=false, description="Fix
|
||||
task(task_id="ses_abc123", load_skills=[], run_in_background=false, description="Fix type error", prompt="Fix: Type error on line 42")
|
||||
\`\`\`
|
||||
|
||||
**After EVERY delegation, STORE the task_id for potential continuation.**
|
||||
**After EVERY delegation, STORE the \`ses_...\` continuation ID for potential continuation.**
|
||||
|
||||
### Code Changes:
|
||||
- Match existing patterns (if codebase is disciplined)
|
||||
|
||||
@@ -301,11 +301,12 @@ Fire similar parallel calls for error patterns (explore), JWT security best prac
|
||||
|
||||
### Background Result Collection:
|
||||
|
||||
1. Launch parallel agents → receive task_ids
|
||||
1. Launch parallel agents → receive background task IDs (\`bg_...\`) for results and continuation session IDs (\`ses_...\`) for follow-ups.
|
||||
2. Continue ONLY with non-overlapping work. If none → END YOUR RESPONSE.
|
||||
3. System sends \`<system-reminder>\` when tasks complete.
|
||||
4. Collect via \`background_output(task_id="...")\` ONLY after \`<system-reminder>\`.
|
||||
4. Collect via \`background_output(task_id="bg_...")\` ONLY after \`<system-reminder>\`.
|
||||
5. Cancel disposable tasks INDIVIDUALLY via \`background_cancel(taskId="...")\`. NEVER \`background_cancel(all=true)\`.
|
||||
6. Use \`task(task_id="ses_...")\` only to continue the same sub-agent session.
|
||||
|
||||
${buildAntiDuplicationSection()}
|
||||
|
||||
@@ -347,9 +348,10 @@ After delegation: VERIFY against MUST DO/MUST NOT DO + existing patterns. Vague
|
||||
|
||||
### Session Continuity (apply to ALL follow-ups)
|
||||
|
||||
Every \`task()\` returns \`task_id\`. **REUSE IT.**
|
||||
Every \`task()\` output exposes a continuation session ID (\`ses_...\`). Pass it to \`task(task_id="ses_...")\`. **REUSE IT.**
|
||||
|
||||
Use \`task_id\` for: failed/incomplete work, follow-up questions, multi-turn refinement, verification failures.
|
||||
Use \`task(task_id="ses_...")\` for: failed/incomplete work, follow-up questions, multi-turn refinement, verification failures.
|
||||
Keep IDs separate: background task IDs (\`bg_...\`) are for \`background_output(task_id="bg_...")\`; continuation session IDs (\`ses_...\`) are for \`task(task_id="ses_...")\`.
|
||||
|
||||
\`\`\`typescript
|
||||
// WRONG: starting fresh loses everything
|
||||
|
||||
@@ -327,14 +327,15 @@ result = task(..., run_in_background=false) // Never wait synchronously for exp
|
||||
\`\`\`
|
||||
|
||||
### Background Result Collection:
|
||||
1. Launch parallel agents → receive task_ids
|
||||
1. Launch parallel agents → receive background task IDs (\`bg_...\`) for results and continuation session IDs (\`ses_...\`) for follow-ups
|
||||
2. Continue only with non-overlapping work
|
||||
- If you have DIFFERENT independent work → do it now
|
||||
- Otherwise → **END YOUR RESPONSE.**
|
||||
3. **STOP. END YOUR RESPONSE.** The system will send \`<system-reminder>\` when tasks complete.
|
||||
4. On receiving \`<system-reminder>\` → collect results via \`background_output(task_id="...")\`
|
||||
4. On receiving \`<system-reminder>\` → collect results via \`background_output(task_id="bg_...")\`
|
||||
5. **NEVER call \`background_output\` before receiving \`<system-reminder>\`.** This is a BLOCKING anti-pattern.
|
||||
6. Cleanup: Cancel disposable tasks individually via \`background_cancel(taskId="...")\`
|
||||
7. Use \`task(task_id="ses_...")\` only to continue the same sub-agent session
|
||||
|
||||
${buildAntiDuplicationSection()}
|
||||
|
||||
@@ -389,15 +390,17 @@ AFTER THE WORK YOU DELEGATED SEEMS DONE, ALWAYS VERIFY THE RESULTS AS FOLLOWING:
|
||||
|
||||
### Session Continuity (MANDATORY)
|
||||
|
||||
Every \`task()\` output includes a task_id. **USE IT.**
|
||||
Every \`task()\` output exposes a continuation session ID (\`ses_...\`). Pass it to \`task(task_id="ses_...")\` for follow-ups. **USE IT.**
|
||||
|
||||
**ALWAYS continue when:**
|
||||
- Task failed/incomplete → \`task_id="{task_id}", prompt="Fix: {specific error}"\`
|
||||
- Follow-up question on result → \`task_id="{task_id}", prompt="Also: {question}"\`
|
||||
- Multi-turn with same agent → \`task_id="{task_id}"\` - NEVER start fresh
|
||||
- Verification failed → \`task_id="{task_id}", prompt="Failed verification: {error}. Fix."\`
|
||||
- Task failed/incomplete → \`task(task_id="ses_...", prompt="Fix: {specific error}")\`
|
||||
- Follow-up question on result → \`task(task_id="ses_...", prompt="Also: {question}")\`
|
||||
- Multi-turn with same agent → \`task(task_id="ses_...")\` - NEVER start fresh
|
||||
- Verification failed → \`task(task_id="ses_...", prompt="Failed verification: {error}. Fix.")\`
|
||||
|
||||
**Why task_id is CRITICAL:**
|
||||
**Keep IDs separate:** background task IDs (\`bg_...\`) are for \`background_output(task_id="bg_...")\`; continuation session IDs (\`ses_...\`) are for \`task(task_id="ses_...")\`.
|
||||
|
||||
**Why continuation is CRITICAL:**
|
||||
- Subagent has FULL conversation context preserved
|
||||
- No repeated file reads, exploration, or setup
|
||||
- Saves 70%+ tokens on follow-ups
|
||||
@@ -411,7 +414,7 @@ task(category="quick", load_skills=[], run_in_background=false, description="Fix
|
||||
task(task_id="ses_abc123", load_skills=[], run_in_background=false, description="Fix type error", prompt="Fix: Type error on line 42")
|
||||
\`\`\`
|
||||
|
||||
**After EVERY delegation, STORE the task_id for potential continuation.**
|
||||
**After EVERY delegation, STORE the \`ses_...\` continuation ID for potential continuation.**
|
||||
|
||||
### Code Changes:
|
||||
- Match existing patterns (if codebase is disciplined)
|
||||
|
||||
@@ -263,14 +263,15 @@ Each agent prompt should include:
|
||||
- [REQUEST]: What to find, what format, what to skip
|
||||
|
||||
Background result collection:
|
||||
1. Launch parallel agents → receive task_ids
|
||||
1. Launch parallel agents → receive background task IDs (\`bg_...\`) for results and continuation session IDs (\`ses_...\`) for follow-ups
|
||||
2. Continue only with non-overlapping work
|
||||
- If you have DIFFERENT independent work → do it now
|
||||
- Otherwise → **END YOUR RESPONSE.**
|
||||
3. **STOP. END YOUR RESPONSE.** The system will send \`<system-reminder>\` when tasks complete.
|
||||
4. On receiving \`<system-reminder>\` → collect results via \`background_output(task_id="...")\`
|
||||
4. On receiving \`<system-reminder>\` → collect results via \`background_output(task_id="bg_...")\`
|
||||
5. **NEVER call \`background_output\` before receiving \`<system-reminder>\`.** This is a BLOCKING anti-pattern.
|
||||
6. Cancel disposable tasks individually via \`background_cancel(taskId="...")\`
|
||||
7. Use \`task(task_id="ses_...")\` only to continue the same sub-agent session
|
||||
|
||||
${buildAntiDuplicationSection()}
|
||||
|
||||
@@ -387,10 +388,12 @@ Post-delegation: delegation never substitutes for verification. Always run \`<ve
|
||||
|
||||
### Session continuity
|
||||
|
||||
Every \`task()\` returns a task_id. Use it for all follow-ups:
|
||||
- Failed/incomplete → \`task_id="{id}", prompt="Fix: {specific error}"\`
|
||||
- Follow-up → \`task_id="{id}", prompt="Also: {question}"\`
|
||||
- Multi-turn → always \`task_id\`, never start fresh
|
||||
Every \`task()\` output exposes a continuation session ID (\`ses_...\`). Pass it to \`task(task_id="ses_...")\` for all follow-ups:
|
||||
- Failed/incomplete → \`task(task_id="ses_...", prompt="Fix: {specific error}")\`
|
||||
- Follow-up → \`task(task_id="ses_...", prompt="Also: {question}")\`
|
||||
- Multi-turn → always \`task(task_id="ses_...")\`, never start fresh
|
||||
|
||||
Keep IDs separate: background task IDs (\`bg_...\`) are for \`background_output(task_id="bg_...")\`; continuation session IDs (\`ses_...\`) are for \`task(task_id="ses_...")\`.
|
||||
|
||||
This preserves full context, avoids repeated exploration, saves 70%+ tokens.
|
||||
|
||||
|
||||
@@ -218,11 +218,13 @@ After a delegation completes, verification is not optional. Read every file the
|
||||
|
||||
### Session continuity
|
||||
|
||||
Every \`task()\` returns a \`task_id\`. Reuse it for every follow-up interaction with the same sub-agent:
|
||||
Every \`task()\` output exposes a continuation session ID (\`ses_...\`). Pass it to \`task(task_id="ses_...")\` for every follow-up with the same sub-agent:
|
||||
|
||||
- Failed or incomplete work: \`task(task_id="{id}", prompt="Fix: {specific error}")\`
|
||||
- Follow-up question on a result: \`task(task_id="{id}", prompt="Also: {question}")\`
|
||||
- Multi-turn refinement: always \`task_id\`, never a fresh session.
|
||||
- Failed or incomplete work: \`task(task_id="ses_...", prompt="Fix: {specific error}")\`
|
||||
- Follow-up question on a result: \`task(task_id="ses_...", prompt="Also: {question}")\`
|
||||
- Multi-turn refinement: always \`task(task_id="ses_...")\`, never a fresh session.
|
||||
|
||||
Keep IDs separate: background task IDs (\`bg_...\`) are for \`background_output(task_id="bg_...")\`; continuation session IDs (\`ses_...\`) are for \`task(task_id="ses_...")\`.
|
||||
|
||||
Starting fresh on a follow-up throws away the sub-agent's full context. Session continuity typically saves 70% of the tokens a fresh session would burn.
|
||||
|
||||
@@ -235,7 +237,7 @@ Exploration is cheap; assumption is expensive. Before implementation on anything
|
||||
|
||||
Each exploration prompt should include four fields: **CONTEXT** (what task, which modules), **GOAL** (what decision the results will unblock), **DOWNSTREAM** (how you will use the results), **REQUEST** (what to find, what format, what to skip).
|
||||
|
||||
After firing exploration agents, do not manually perform the same search yourself. That is duplicate work and wastes your context window. Continue only with non-overlapping preparation: setting up files, reading known-path files, drafting questions. If no non-overlapping work exists, end your response and wait for the completion notification; do not poll \`background_output\` on a running task.
|
||||
After firing exploration agents, keep the returned background task IDs (\`bg_...\`) for result collection and continuation session IDs (\`ses_...\`) for follow-ups. Continue only with non-overlapping preparation: setting up files, reading known-path files, drafting questions. If no non-overlapping work exists, end your response and wait for the completion notification; then use \`background_output(task_id="bg_...")\`, not \`task(task_id="ses_...")\`, to collect results.
|
||||
|
||||
Stop searching when you have enough context to proceed confidently, when the same information keeps appearing across sources, when two iterations yield no new useful data, or when you found a direct answer.
|
||||
|
||||
|
||||
@@ -307,14 +307,15 @@ Each agent prompt should include:
|
||||
- [REQUEST]: What to find, what format, what to skip
|
||||
|
||||
Background result collection:
|
||||
1. Launch parallel agents → receive task_ids
|
||||
1. Launch parallel agents → receive background task IDs (\`bg_...\`) for results and continuation session IDs (\`ses_...\`) for follow-ups
|
||||
2. Continue only with non-overlapping work
|
||||
- If you have DIFFERENT independent work → do it now
|
||||
- Otherwise → **END YOUR RESPONSE.**
|
||||
3. **STOP. END YOUR RESPONSE.** The system will send \`<system-reminder>\` when tasks complete.
|
||||
4. On receiving \`<system-reminder>\` → collect results via \`background_output(task_id="...")\`
|
||||
4. On receiving \`<system-reminder>\` → collect results via \`background_output(task_id="bg_...")\`
|
||||
5. **NEVER call \`background_output\` before receiving \`<system-reminder>\`.** This is a BLOCKING anti-pattern.
|
||||
6. Cancel disposable tasks individually via \`background_cancel(taskId="...")\`
|
||||
7. Use \`task(task_id="ses_...")\` only to continue the same sub-agent session
|
||||
|
||||
${buildAntiDuplicationSection()}
|
||||
|
||||
@@ -462,10 +463,12 @@ Post-delegation: delegation never substitutes for verification. Always run \`<ve
|
||||
|
||||
### Session continuity
|
||||
|
||||
Every \`task()\` returns a session_id. Use it for all follow-ups:
|
||||
- Failed/incomplete → \`session_id="{id}", prompt="Fix: {specific error}"\`
|
||||
- Follow-up → \`session_id="{id}", prompt="Also: {question}"\`
|
||||
- Multi-turn → always \`session_id\`, never start fresh
|
||||
Every \`task()\` output exposes a continuation session ID (\`ses_...\`). Pass it to \`task(task_id="ses_...")\` for all follow-ups:
|
||||
- Failed/incomplete → \`task(task_id="ses_...", prompt="Fix: {specific error}")\`
|
||||
- Follow-up → \`task(task_id="ses_...", prompt="Also: {question}")\`
|
||||
- Multi-turn → always \`task(task_id="ses_...")\`, never start fresh
|
||||
|
||||
Keep IDs separate: background task IDs (\`bg_...\`) are for \`background_output(task_id="bg_...")\`; continuation session IDs (\`ses_...\`) are for \`task(task_id="ses_...")\`.
|
||||
|
||||
This preserves full context, avoids repeated exploration, saves 70%+ tokens.
|
||||
|
||||
|
||||
@@ -135,7 +135,7 @@ LspFindReferences(filePath="...", line=X, character=Y)
|
||||
|
||||
\`\`\`
|
||||
// After main session analysis done, collect all task results
|
||||
for each task_id: background_output(task_id="...")
|
||||
for each background task ID (\`bg_...\`): background_output(task_id="bg_...")
|
||||
\`\`\`
|
||||
|
||||
**Merge: bash + LSP + existing + explore findings. Mark "discovery" as completed.**
|
||||
|
||||
@@ -481,7 +481,7 @@ OUTPUT FORMAT:
|
||||
|
||||
After launching all 5 agents in one turn, **end your response**. Wait for system notifications as each agent completes.
|
||||
|
||||
As each completes, collect via \`background_output(task_id="...")\`. Store each verdict:
|
||||
As each completes, collect via \`background_output(task_id="bg_...")\`. Store each verdict:
|
||||
|
||||
| Agent | Verdict | Notes |
|
||||
|-------|---------|-------|
|
||||
|
||||
@@ -115,7 +115,7 @@ task(subagent_type="plan", load_skills=[], run_in_background=false, prompt="<gat
|
||||
|
||||
### SESSION CONTINUITY WITH PLAN AGENT (CRITICAL)
|
||||
|
||||
**Plan agent returns a task_id. USE IT for follow-up interactions.**
|
||||
**Plan agent output includes a continuation ID (\`ses_...\`). USE IT for follow-up interactions via \`task(task_id="ses_...", ...)\`.**
|
||||
|
||||
| Scenario | Action |
|
||||
|----------|--------|
|
||||
|
||||
@@ -161,7 +161,7 @@ task(subagent_type="plan", load_skills=[], run_in_background=false, prompt="<gat
|
||||
|
||||
### SESSION CONTINUITY WITH PLAN AGENT (CRITICAL)
|
||||
|
||||
**Plan agent returns a task_id. USE IT for follow-up interactions.**
|
||||
**Plan agent output includes a continuation ID (\`ses_...\`). USE IT for follow-up interactions via \`task(task_id="ses_...", ...)\`.**
|
||||
|
||||
| Scenario | Action |
|
||||
|----------|--------|
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
export const BACKGROUND_TASK_DESCRIPTION = `Run agent task in background. Returns task_id immediately; notifies on completion.
|
||||
export const BACKGROUND_TASK_DESCRIPTION = `Run agent task in background. Returns a background task ID (\`bg_...\`) immediately; notifies on completion.
|
||||
|
||||
Use \`background_output\` to get results. Prompts MUST be in English.`
|
||||
|
||||
|
||||
@@ -4,7 +4,9 @@ import type { ToolContext } from "@opencode-ai/plugin/tool"
|
||||
import { describe, expect, test } from "bun:test"
|
||||
import type { BackgroundTask } from "../../features/background-agent"
|
||||
import { clearPendingStore, consumeToolMetadata } from "../../features/tool-metadata-store"
|
||||
import { unsafeTestValue } from "../../../test-support/unsafe-test-value"
|
||||
import type { BackgroundOutputClient, BackgroundOutputManager } from "./clients"
|
||||
import { BACKGROUND_TASK_DESCRIPTION } from "./constants"
|
||||
import { createBackgroundOutput } from "./create-background-output"
|
||||
|
||||
const projectDir = "/Users/yeongyu/local-workspaces/oh-my-opencode"
|
||||
@@ -14,6 +16,38 @@ type ToolContextWithCallID = ToolContext & {
|
||||
}
|
||||
|
||||
describe("createBackgroundOutput metadata", () => {
|
||||
test("describes background task launch output as a bg id", () => {
|
||||
// #given, #when
|
||||
const description = BACKGROUND_TASK_DESCRIPTION
|
||||
|
||||
// #then
|
||||
expect(description).toContain("background task ID")
|
||||
expect(description).toContain("bg_")
|
||||
expect(description).not.toContain("Returns task_id")
|
||||
})
|
||||
|
||||
test("describes task_id as a background task id instead of a session id", () => {
|
||||
// #given
|
||||
const manager: BackgroundOutputManager = {
|
||||
getTask: () => undefined,
|
||||
}
|
||||
const client: BackgroundOutputClient = {
|
||||
session: {
|
||||
messages: async () => ({ data: [] }),
|
||||
},
|
||||
}
|
||||
const tool = createBackgroundOutput(manager, client)
|
||||
|
||||
// #when
|
||||
const taskIdArg = unsafeTestValue<{ description?: string }>(tool.args.task_id)
|
||||
|
||||
// #then
|
||||
expect(taskIdArg.description).toContain("background task ID")
|
||||
expect(taskIdArg.description).toContain("bg_")
|
||||
expect(taskIdArg.description).toContain("not a session ID")
|
||||
expect(taskIdArg.description).toContain("ses_")
|
||||
})
|
||||
|
||||
test("omits sessionId metadata when task session is not yet assigned", async () => {
|
||||
// #given
|
||||
clearPendingStore()
|
||||
|
||||
@@ -93,7 +93,9 @@ export function createBackgroundOutput(manager: BackgroundOutputManager, client:
|
||||
return tool({
|
||||
description: BACKGROUND_OUTPUT_DESCRIPTION,
|
||||
args: {
|
||||
task_id: tool.schema.string().describe("Task ID to get output from"),
|
||||
task_id: tool.schema
|
||||
.string()
|
||||
.describe("background task ID (`bg_...`) from launch/completion; not a session ID (`ses_...`)."),
|
||||
block: tool.schema
|
||||
.boolean()
|
||||
.optional()
|
||||
|
||||
@@ -42,11 +42,25 @@ function createDelegateTask(...args: Parameters<typeof import("./tools").createD
|
||||
|
||||
//#then
|
||||
expect(description).toContain("subagent_type: Use specific agent directly")
|
||||
expect(description).toContain("task_id: Existing task to continue")
|
||||
expect(description).toContain("task_id: Continuation session id")
|
||||
expect(description).not.toContain("sisyphus")
|
||||
expect(description).not.toContain("hephaestus")
|
||||
expect(description).not.toContain("prometheus")
|
||||
})
|
||||
|
||||
test("#given task schema #when describing async mode #then it names background task ids explicitly", () => {
|
||||
//#given
|
||||
const toolDefinition = createDelegateTask({ manager: {} as never, client: {} as never, directory: "/tmp/test" })
|
||||
|
||||
//#when
|
||||
const runInBackgroundSchema = unsafeTestValue<{ description?: string }>(toolDefinition.args.run_in_background)
|
||||
|
||||
//#then
|
||||
expect(runInBackgroundSchema.description).toContain("background task ID")
|
||||
expect(runInBackgroundSchema.description).toContain("bg_")
|
||||
expect(runInBackgroundSchema.description).toContain("background_output")
|
||||
expect(runInBackgroundSchema.description).not.toContain("returns task_id")
|
||||
})
|
||||
})
|
||||
|
||||
export {}
|
||||
|
||||
@@ -15,4 +15,18 @@ describe("createDelegateTaskPresentation", () => {
|
||||
expect(description).toContain("busy/retry/running")
|
||||
expect(description).toContain("not a total wall-clock limit")
|
||||
})
|
||||
|
||||
test("#given continuation usage #when description is rendered #then task_id is described as a session id", () => {
|
||||
//#given
|
||||
const presentation = createDelegateTaskPresentation({})
|
||||
|
||||
//#when
|
||||
const description = presentation.description
|
||||
|
||||
//#then
|
||||
expect(description).toContain("task_id: Continuation session id")
|
||||
expect(description).toContain("ses_")
|
||||
expect(description).toContain("not the background task id")
|
||||
expect(description).toContain("bg_")
|
||||
})
|
||||
})
|
||||
|
||||
@@ -66,15 +66,15 @@ export function createDelegateTaskPresentation(options: DelegateTaskToolOptions)
|
||||
Available categories:
|
||||
${categoryList}
|
||||
- subagent_type: Use specific agent directly (explore, librarian, oracle, metis, momus)
|
||||
- run_in_background: REQUIRED. true=async (returns task_id), false=sync (waits). Use background=true ONLY for parallel exploration with 5+ independent queries.
|
||||
- run_in_background: REQUIRED. true=async (returns a background task ID like \`bg_...\` for \`background_output\`), false=sync (waits). Use background=true ONLY for parallel exploration with 5+ independent queries.
|
||||
Sync waits use a 30-minute inactivity window: OpenCode busy/retry/running status resets the window, so this is not a total wall-clock limit.
|
||||
- task_id: Existing task to continue (from previous task output). Continues the same subagent session with FULL CONTEXT PRESERVED.
|
||||
- task_id: Continuation session id (\`ses_...\`) from task metadata. Continues the same subagent session with FULL CONTEXT PRESERVED; not the background task id (\`bg_...\`).
|
||||
- command: The command that triggered this task (optional, for slash command tracking).
|
||||
|
||||
**WHEN TO USE task_id:**
|
||||
- Task failed/incomplete → task_id with "fix: [specific issue]"
|
||||
- Need follow-up on previous result → task_id with additional question
|
||||
- Multi-turn conversation with same agent → always task_id instead of new task
|
||||
- Task failed/incomplete → \`task(task_id="ses_...", prompt="fix: [specific issue]")\`
|
||||
- Need follow-up on previous result → \`task(task_id="ses_...", prompt="Also: [question]")\`
|
||||
- Multi-turn conversation with same agent → always \`task(task_id="ses_...")\` instead of new task
|
||||
|
||||
Prompts MUST be in English.`
|
||||
|
||||
|
||||
@@ -24,10 +24,15 @@ const delegateTaskArgsSchema = {
|
||||
load_skills: tool.schema.array(tool.schema.string()).describe("Skill names to inject. REQUIRED - pass [] if no skills needed."),
|
||||
description: tool.schema.string().optional().describe("Short task description (3-5 words). Auto-generated from prompt if omitted."),
|
||||
prompt: tool.schema.string().describe("Full detailed prompt for the agent"),
|
||||
run_in_background: tool.schema.boolean().describe("REQUIRED. true=async (returns task_id), false=sync (waits). Use false for task delegation, true ONLY for parallel exploration."),
|
||||
run_in_background: tool.schema
|
||||
.boolean()
|
||||
.describe("REQUIRED. true=async (returns background task ID `bg_...` for background_output), false=sync (waits). Use false for task delegation, true ONLY for parallel exploration."),
|
||||
category: tool.schema.string().optional().describe("REQUIRED if subagent_type not provided. Do NOT provide both category and subagent_type."),
|
||||
subagent_type: tool.schema.string().optional().describe("REQUIRED if category not provided. Do NOT provide both category and subagent_type."),
|
||||
task_id: tool.schema.string().optional().describe("Existing task to continue. Canonical resume identifier."),
|
||||
task_id: tool.schema
|
||||
.string()
|
||||
.optional()
|
||||
.describe("Continuation session id (`ses_...`) from task metadata; not a background task id (`bg_...`)."),
|
||||
command: tool.schema.string().optional().describe("The command that triggered this task"),
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user