refactor: major codebase cleanup - BDD comments, file splitting, bug fixes (#1350)

* style(tests): normalize BDD comments from '// #given' to '// given'

- Replace 4,668 Python-style BDD comments across 107 test files
- Patterns changed: // #given -> // given, // #when -> // when, // #then -> // then
- Also handles no-space variants: //#given -> // given

* fix(rules-injector): prefer output.metadata.filePath over output.title

- Extract file path resolution to dedicated output-path.ts module
- Prefer metadata.filePath which contains actual file path
- Fall back to output.title only when metadata unavailable
- Fixes issue where rules weren't injected when tool output title was a label

* feat(slashcommand): add optional user_message parameter

- Add user_message optional parameter for command arguments
- Model can now call: command='publish' user_message='patch'
- Improves error messages with clearer format guidance
- Helps LLMs understand correct parameter usage

* feat(hooks): restore compaction-context-injector hook

- Restore hook deleted in cbbc7bd0 for session compaction context
- Injects 7 mandatory sections: User Requests, Final Goal, Work Completed,
  Remaining Tasks, Active Working Context, MUST NOT Do, Agent Verification State
- Re-register in hooks/index.ts and main plugin entry

* refactor(background-agent): split manager.ts into focused modules

- Extract constants.ts for TTL values and internal types (52 lines)
- Extract state.ts for TaskStateManager class (204 lines)
- Extract spawner.ts for task creation logic (244 lines)
- Extract result-handler.ts for completion handling (265 lines)
- Reduce manager.ts from 1377 to 755 lines (45% reduction)
- Maintain backward compatible exports

* refactor(agents): split prometheus-prompt.ts into subdirectory

- Move 1196-line prometheus-prompt.ts to prometheus/ subdirectory
- Organize prompt sections into separate files for maintainability
- Update agents/index.ts exports

* refactor(delegate-task): split tools.ts into focused modules

- Extract categories.ts for category definitions and routing
- Extract executor.ts for task execution logic
- Extract helpers.ts for utility functions
- Extract prompt-builder.ts for prompt construction
- Reduce tools.ts complexity with cleaner separation of concerns

* refactor(builtin-skills): split skills.ts into individual skill files

- Move each skill to dedicated file in skills/ subdirectory
- Create barrel export for backward compatibility
- Improve maintainability with focused skill modules

* chore: update import paths and lockfile

- Update prometheus import path after refactor
- Update bun.lock

* fix(tests): complete BDD comment normalization

- Fix remaining #when/#then patterns missed by initial sed
- Affected: state.test.ts, events.test.ts

---------

Co-authored-by: justsisyphus <justsisyphus@users.noreply.github.com>
This commit is contained in:
YeonGyu-Kim
2026-02-01 16:47:50 +09:00
committed by GitHub
parent c83150d9ea
commit f146aeff0f
145 changed files with 10307 additions and 9562 deletions
+91 -91
View File
@@ -67,21 +67,21 @@ describe("atlas hook", () => {
describe("tool.execute.after handler", () => {
test("should handle undefined output gracefully (issue #1035)", async () => {
// #given - hook and undefined output (e.g., from /review command)
// given - hook and undefined output (e.g., from /review command)
const hook = createAtlasHook(createMockPluginInput())
// #when - calling with undefined output
// when - calling with undefined output
const result = await hook["tool.execute.after"](
{ tool: "delegate_task", sessionID: "session-123" },
undefined as unknown as { title: string; output: string; metadata: Record<string, unknown> }
)
// #then - returns undefined without throwing
// then - returns undefined without throwing
expect(result).toBeUndefined()
})
test("should ignore non-delegate_task tools", async () => {
// #given - hook and non-delegate_task tool
// given - hook and non-delegate_task tool
const hook = createAtlasHook(createMockPluginInput())
const output = {
title: "Test Tool",
@@ -89,18 +89,18 @@ describe("atlas hook", () => {
metadata: {},
}
// #when
// when
await hook["tool.execute.after"](
{ tool: "other_tool", sessionID: "session-123" },
output
)
// #then - output unchanged
// then - output unchanged
expect(output.output).toBe("Original output")
})
test("should not transform when caller is not Atlas", async () => {
// #given - boulder state exists but caller agent in message storage is not Atlas
// given - boulder state exists but caller agent in message storage is not Atlas
const sessionID = "session-non-orchestrator-test"
setupMessageStorage(sessionID, "other-agent")
@@ -122,20 +122,20 @@ describe("atlas hook", () => {
metadata: {},
}
// #when
// when
await hook["tool.execute.after"](
{ tool: "delegate_task", sessionID },
output
)
// #then - output unchanged because caller is not orchestrator
// then - output unchanged because caller is not orchestrator
expect(output.output).toBe("Task completed successfully")
cleanupMessageStorage(sessionID)
})
test("should append standalone verification when no boulder state but caller is Atlas", async () => {
// #given - no boulder state, but caller is Atlas
// given - no boulder state, but caller is Atlas
const sessionID = "session-no-boulder-test"
setupMessageStorage(sessionID, "atlas")
@@ -146,13 +146,13 @@ describe("atlas hook", () => {
metadata: {},
}
// #when
// when
await hook["tool.execute.after"](
{ tool: "delegate_task", sessionID },
output
)
// #then - standalone verification reminder appended
// then - standalone verification reminder appended
expect(output.output).toContain("Task completed successfully")
expect(output.output).toContain("MANDATORY:")
expect(output.output).toContain("delegate_task(session_id=")
@@ -161,7 +161,7 @@ describe("atlas hook", () => {
})
test("should transform output when caller is Atlas with boulder state", async () => {
// #given - Atlas caller with boulder state
// given - Atlas caller with boulder state
const sessionID = "session-transform-test"
setupMessageStorage(sessionID, "atlas")
@@ -183,13 +183,13 @@ describe("atlas hook", () => {
metadata: {},
}
// #when
// when
await hook["tool.execute.after"](
{ tool: "delegate_task", sessionID },
output
)
// #then - output should be transformed (original output preserved for debugging)
// then - output should be transformed (original output preserved for debugging)
expect(output.output).toContain("Task completed successfully")
expect(output.output).toContain("SUBAGENT WORK COMPLETED")
expect(output.output).toContain("test-plan")
@@ -200,7 +200,7 @@ describe("atlas hook", () => {
})
test("should still transform when plan is complete (shows progress)", async () => {
// #given - boulder state with complete plan, Atlas caller
// given - boulder state with complete plan, Atlas caller
const sessionID = "session-complete-plan-test"
setupMessageStorage(sessionID, "atlas")
@@ -222,13 +222,13 @@ describe("atlas hook", () => {
metadata: {},
}
// #when
// when
await hook["tool.execute.after"](
{ tool: "delegate_task", sessionID },
output
)
// #then - output transformed even when complete (shows 2/2 done)
// then - output transformed even when complete (shows 2/2 done)
expect(output.output).toContain("SUBAGENT WORK COMPLETED")
expect(output.output).toContain("2/2 done")
expect(output.output).toContain("0 remaining")
@@ -237,7 +237,7 @@ describe("atlas hook", () => {
})
test("should append session ID to boulder state if not present", async () => {
// #given - boulder state without session-append-test, Atlas caller
// given - boulder state without session-append-test, Atlas caller
const sessionID = "session-append-test"
setupMessageStorage(sessionID, "atlas")
@@ -259,13 +259,13 @@ describe("atlas hook", () => {
metadata: {},
}
// #when
// when
await hook["tool.execute.after"](
{ tool: "delegate_task", sessionID },
output
)
// #then - sessionID should be appended
// then - sessionID should be appended
const updatedState = readBoulderState(TEST_DIR)
expect(updatedState?.session_ids).toContain(sessionID)
@@ -273,7 +273,7 @@ describe("atlas hook", () => {
})
test("should not duplicate existing session ID", async () => {
// #given - boulder state already has session-dup-test, Atlas caller
// given - boulder state already has session-dup-test, Atlas caller
const sessionID = "session-dup-test"
setupMessageStorage(sessionID, "atlas")
@@ -295,13 +295,13 @@ describe("atlas hook", () => {
metadata: {},
}
// #when
// when
await hook["tool.execute.after"](
{ tool: "delegate_task", sessionID },
output
)
// #then - should still have only one sessionID
// then - should still have only one sessionID
const updatedState = readBoulderState(TEST_DIR)
const count = updatedState?.session_ids.filter((id) => id === sessionID).length
expect(count).toBe(1)
@@ -310,7 +310,7 @@ describe("atlas hook", () => {
})
test("should include boulder.json path and notepad path in transformed output", async () => {
// #given - boulder state, Atlas caller
// given - boulder state, Atlas caller
const sessionID = "session-path-test"
setupMessageStorage(sessionID, "atlas")
@@ -332,13 +332,13 @@ describe("atlas hook", () => {
metadata: {},
}
// #when
// when
await hook["tool.execute.after"](
{ tool: "delegate_task", sessionID },
output
)
// #then - output should contain plan name and progress
// then - output should contain plan name and progress
expect(output.output).toContain("my-feature")
expect(output.output).toContain("1/3 done")
expect(output.output).toContain("2 remaining")
@@ -347,7 +347,7 @@ describe("atlas hook", () => {
})
test("should include session_id and checkbox instructions in reminder", async () => {
// #given - boulder state, Atlas caller
// given - boulder state, Atlas caller
const sessionID = "session-resume-test"
setupMessageStorage(sessionID, "atlas")
@@ -369,13 +369,13 @@ describe("atlas hook", () => {
metadata: {},
}
// #when
// when
await hook["tool.execute.after"](
{ tool: "delegate_task", sessionID },
output
)
// #then - should include session_id instructions and verification
// then - should include session_id instructions and verification
expect(output.output).toContain("delegate_task(session_id=")
expect(output.output).toContain("[x]")
expect(output.output).toContain("MANDATORY:")
@@ -395,7 +395,7 @@ describe("atlas hook", () => {
})
test("should append delegation reminder when orchestrator writes outside .sisyphus/", async () => {
// #given
// given
const hook = createAtlasHook(createMockPluginInput())
const output = {
title: "Write",
@@ -403,20 +403,20 @@ describe("atlas hook", () => {
metadata: { filePath: "/path/to/code.ts" },
}
// #when
// when
await hook["tool.execute.after"](
{ tool: "Write", sessionID: ORCHESTRATOR_SESSION },
output
)
// #then
// then
expect(output.output).toContain("ORCHESTRATOR, not an IMPLEMENTER")
expect(output.output).toContain("delegate_task")
expect(output.output).toContain("delegate_task")
})
test("should append delegation reminder when orchestrator edits outside .sisyphus/", async () => {
// #given
// given
const hook = createAtlasHook(createMockPluginInput())
const output = {
title: "Edit",
@@ -424,18 +424,18 @@ describe("atlas hook", () => {
metadata: { filePath: "/src/components/button.tsx" },
}
// #when
// when
await hook["tool.execute.after"](
{ tool: "Edit", sessionID: ORCHESTRATOR_SESSION },
output
)
// #then
// then
expect(output.output).toContain("ORCHESTRATOR, not an IMPLEMENTER")
})
test("should NOT append reminder when orchestrator writes inside .sisyphus/", async () => {
// #given
// given
const hook = createAtlasHook(createMockPluginInput())
const originalOutput = "File written successfully"
const output = {
@@ -444,19 +444,19 @@ describe("atlas hook", () => {
metadata: { filePath: "/project/.sisyphus/plans/work-plan.md" },
}
// #when
// when
await hook["tool.execute.after"](
{ tool: "Write", sessionID: ORCHESTRATOR_SESSION },
output
)
// #then
// then
expect(output.output).toBe(originalOutput)
expect(output.output).not.toContain("ORCHESTRATOR, not an IMPLEMENTER")
})
test("should NOT append reminder when non-orchestrator writes outside .sisyphus/", async () => {
// #given
// given
const nonOrchestratorSession = "non-orchestrator-session"
setupMessageStorage(nonOrchestratorSession, "sisyphus-junior")
@@ -468,13 +468,13 @@ describe("atlas hook", () => {
metadata: { filePath: "/path/to/code.ts" },
}
// #when
// when
await hook["tool.execute.after"](
{ tool: "Write", sessionID: nonOrchestratorSession },
output
)
// #then
// then
expect(output.output).toBe(originalOutput)
expect(output.output).not.toContain("ORCHESTRATOR, not an IMPLEMENTER")
@@ -482,7 +482,7 @@ describe("atlas hook", () => {
})
test("should NOT append reminder for read-only tools", async () => {
// #given
// given
const hook = createAtlasHook(createMockPluginInput())
const originalOutput = "File content"
const output = {
@@ -491,18 +491,18 @@ describe("atlas hook", () => {
metadata: { filePath: "/path/to/code.ts" },
}
// #when
// when
await hook["tool.execute.after"](
{ tool: "Read", sessionID: ORCHESTRATOR_SESSION },
output
)
// #then
// then
expect(output.output).toBe(originalOutput)
})
test("should handle missing filePath gracefully", async () => {
// #given
// given
const hook = createAtlasHook(createMockPluginInput())
const originalOutput = "File written successfully"
const output = {
@@ -511,19 +511,19 @@ describe("atlas hook", () => {
metadata: {},
}
// #when
// when
await hook["tool.execute.after"](
{ tool: "Write", sessionID: ORCHESTRATOR_SESSION },
output
)
// #then
// then
expect(output.output).toBe(originalOutput)
})
describe("cross-platform path validation (Windows support)", () => {
test("should NOT append reminder when orchestrator writes inside .sisyphus\\ (Windows backslash)", async () => {
// #given
// given
const hook = createAtlasHook(createMockPluginInput())
const originalOutput = "File written successfully"
const output = {
@@ -532,19 +532,19 @@ describe("atlas hook", () => {
metadata: { filePath: ".sisyphus\\plans\\work-plan.md" },
}
// #when
// when
await hook["tool.execute.after"](
{ tool: "Write", sessionID: ORCHESTRATOR_SESSION },
output
)
// #then
// then
expect(output.output).toBe(originalOutput)
expect(output.output).not.toContain("ORCHESTRATOR, not an IMPLEMENTER")
})
test("should NOT append reminder when orchestrator writes inside .sisyphus with mixed separators", async () => {
// #given
// given
const hook = createAtlasHook(createMockPluginInput())
const originalOutput = "File written successfully"
const output = {
@@ -553,19 +553,19 @@ describe("atlas hook", () => {
metadata: { filePath: ".sisyphus\\plans/work-plan.md" },
}
// #when
// when
await hook["tool.execute.after"](
{ tool: "Write", sessionID: ORCHESTRATOR_SESSION },
output
)
// #then
// then
expect(output.output).toBe(originalOutput)
expect(output.output).not.toContain("ORCHESTRATOR, not an IMPLEMENTER")
})
test("should NOT append reminder for absolute Windows path inside .sisyphus\\", async () => {
// #given
// given
const hook = createAtlasHook(createMockPluginInput())
const originalOutput = "File written successfully"
const output = {
@@ -574,19 +574,19 @@ describe("atlas hook", () => {
metadata: { filePath: "C:\\Users\\test\\project\\.sisyphus\\plans\\x.md" },
}
// #when
// when
await hook["tool.execute.after"](
{ tool: "Write", sessionID: ORCHESTRATOR_SESSION },
output
)
// #then
// then
expect(output.output).toBe(originalOutput)
expect(output.output).not.toContain("ORCHESTRATOR, not an IMPLEMENTER")
})
test("should append reminder for Windows path outside .sisyphus\\", async () => {
// #given
// given
const hook = createAtlasHook(createMockPluginInput())
const output = {
title: "Write",
@@ -594,13 +594,13 @@ describe("atlas hook", () => {
metadata: { filePath: "C:\\Users\\test\\project\\src\\code.ts" },
}
// #when
// when
await hook["tool.execute.after"](
{ tool: "Write", sessionID: ORCHESTRATOR_SESSION },
output
)
// #then
// then
expect(output.output).toContain("ORCHESTRATOR, not an IMPLEMENTER")
})
})
@@ -623,7 +623,7 @@ describe("atlas hook", () => {
})
test("should inject continuation when boulder has incomplete tasks", async () => {
// #given - boulder state with incomplete plan
// given - boulder state with incomplete plan
const planPath = join(TEST_DIR, "test-plan.md")
writeFileSync(planPath, "# Plan\n- [ ] Task 1\n- [x] Task 2\n- [ ] Task 3")
@@ -638,7 +638,7 @@ describe("atlas hook", () => {
const mockInput = createMockPluginInput()
const hook = createAtlasHook(mockInput)
// #when
// when
await hook.handler({
event: {
type: "session.idle",
@@ -646,7 +646,7 @@ describe("atlas hook", () => {
},
})
// #then - should call prompt with continuation
// then - should call prompt with continuation
expect(mockInput._promptMock).toHaveBeenCalled()
const callArgs = mockInput._promptMock.mock.calls[0][0]
expect(callArgs.path.id).toBe(MAIN_SESSION_ID)
@@ -655,11 +655,11 @@ describe("atlas hook", () => {
})
test("should not inject when no boulder state exists", async () => {
// #given - no boulder state
// given - no boulder state
const mockInput = createMockPluginInput()
const hook = createAtlasHook(mockInput)
// #when
// when
await hook.handler({
event: {
type: "session.idle",
@@ -667,12 +667,12 @@ describe("atlas hook", () => {
},
})
// #then - should not call prompt
// then - should not call prompt
expect(mockInput._promptMock).not.toHaveBeenCalled()
})
test("should not inject when boulder plan is complete", async () => {
// #given - boulder state with complete plan
// given - boulder state with complete plan
const planPath = join(TEST_DIR, "complete-plan.md")
writeFileSync(planPath, "# Plan\n- [x] Task 1\n- [x] Task 2")
@@ -687,7 +687,7 @@ describe("atlas hook", () => {
const mockInput = createMockPluginInput()
const hook = createAtlasHook(mockInput)
// #when
// when
await hook.handler({
event: {
type: "session.idle",
@@ -695,12 +695,12 @@ describe("atlas hook", () => {
},
})
// #then - should not call prompt
// then - should not call prompt
expect(mockInput._promptMock).not.toHaveBeenCalled()
})
test("should skip when abort error occurred before idle", async () => {
// #given - boulder state with incomplete plan
// given - boulder state with incomplete plan
const planPath = join(TEST_DIR, "test-plan.md")
writeFileSync(planPath, "# Plan\n- [ ] Task 1")
@@ -715,7 +715,7 @@ describe("atlas hook", () => {
const mockInput = createMockPluginInput()
const hook = createAtlasHook(mockInput)
// #when - send abort error then idle
// when - send abort error then idle
await hook.handler({
event: {
type: "session.error",
@@ -732,12 +732,12 @@ describe("atlas hook", () => {
},
})
// #then - should not call prompt
// then - should not call prompt
expect(mockInput._promptMock).not.toHaveBeenCalled()
})
test("should skip when background tasks are running", async () => {
// #given - boulder state with incomplete plan
// given - boulder state with incomplete plan
const planPath = join(TEST_DIR, "test-plan.md")
writeFileSync(planPath, "# Plan\n- [ ] Task 1")
@@ -759,7 +759,7 @@ describe("atlas hook", () => {
backgroundManager: mockBackgroundManager as any,
})
// #when
// when
await hook.handler({
event: {
type: "session.idle",
@@ -767,12 +767,12 @@ describe("atlas hook", () => {
},
})
// #then - should not call prompt
// then - should not call prompt
expect(mockInput._promptMock).not.toHaveBeenCalled()
})
test("should clear abort state on message.updated", async () => {
// #given - boulder with incomplete plan
// given - boulder with incomplete plan
const planPath = join(TEST_DIR, "test-plan.md")
writeFileSync(planPath, "# Plan\n- [ ] Task 1")
@@ -787,7 +787,7 @@ describe("atlas hook", () => {
const mockInput = createMockPluginInput()
const hook = createAtlasHook(mockInput)
// #when - abort error, then message update, then idle
// when - abort error, then message update, then idle
await hook.handler({
event: {
type: "session.error",
@@ -810,12 +810,12 @@ describe("atlas hook", () => {
},
})
// #then - should call prompt because abort state was cleared
// then - should call prompt because abort state was cleared
expect(mockInput._promptMock).toHaveBeenCalled()
})
test("should include plan progress in continuation prompt", async () => {
// #given - boulder state with specific progress
// given - boulder state with specific progress
const planPath = join(TEST_DIR, "progress-plan.md")
writeFileSync(planPath, "# Plan\n- [x] Task 1\n- [x] Task 2\n- [ ] Task 3\n- [ ] Task 4")
@@ -830,7 +830,7 @@ describe("atlas hook", () => {
const mockInput = createMockPluginInput()
const hook = createAtlasHook(mockInput)
// #when
// when
await hook.handler({
event: {
type: "session.idle",
@@ -838,14 +838,14 @@ describe("atlas hook", () => {
},
})
// #then - should include progress
// then - should include progress
const callArgs = mockInput._promptMock.mock.calls[0][0]
expect(callArgs.body.parts[0].text).toContain("2/4 completed")
expect(callArgs.body.parts[0].text).toContain("2 remaining")
})
test("should not inject when last agent is not Atlas", async () => {
// #given - boulder state with incomplete plan, but last agent is NOT Atlas
// given - boulder state with incomplete plan, but last agent is NOT Atlas
const planPath = join(TEST_DIR, "test-plan.md")
writeFileSync(planPath, "# Plan\n- [ ] Task 1\n- [ ] Task 2")
@@ -857,14 +857,14 @@ describe("atlas hook", () => {
}
writeBoulderState(TEST_DIR, state)
// #given - last agent is NOT Atlas
// given - last agent is NOT Atlas
cleanupMessageStorage(MAIN_SESSION_ID)
setupMessageStorage(MAIN_SESSION_ID, "sisyphus")
const mockInput = createMockPluginInput()
const hook = createAtlasHook(mockInput)
// #when
// when
await hook.handler({
event: {
type: "session.idle",
@@ -872,12 +872,12 @@ describe("atlas hook", () => {
},
})
// #then - should NOT call prompt because agent is not Atlas
// then - should NOT call prompt because agent is not Atlas
expect(mockInput._promptMock).not.toHaveBeenCalled()
})
test("should debounce rapid continuation injections (prevent infinite loop)", async () => {
// #given - boulder state with incomplete plan
// given - boulder state with incomplete plan
const planPath = join(TEST_DIR, "test-plan.md")
writeFileSync(planPath, "# Plan\n- [ ] Task 1\n- [ ] Task 2")
@@ -892,7 +892,7 @@ describe("atlas hook", () => {
const mockInput = createMockPluginInput()
const hook = createAtlasHook(mockInput)
// #when - fire multiple idle events in rapid succession (simulating infinite loop bug)
// when - fire multiple idle events in rapid succession (simulating infinite loop bug)
await hook.handler({
event: {
type: "session.idle",
@@ -912,12 +912,12 @@ describe("atlas hook", () => {
},
})
// #then - should only call prompt ONCE due to debouncing
// then - should only call prompt ONCE due to debouncing
expect(mockInput._promptMock).toHaveBeenCalledTimes(1)
})
test("should cleanup on session.deleted", async () => {
// #given - boulder state
// given - boulder state
const planPath = join(TEST_DIR, "test-plan.md")
writeFileSync(planPath, "# Plan\n- [ ] Task 1")
@@ -932,7 +932,7 @@ describe("atlas hook", () => {
const mockInput = createMockPluginInput()
const hook = createAtlasHook(mockInput)
// #when - create abort state then delete
// when - create abort state then delete
await hook.handler({
event: {
type: "session.error",
@@ -960,7 +960,7 @@ describe("atlas hook", () => {
},
})
// #then - should call prompt because session state was cleaned
// then - should call prompt because session state was cleaned
expect(mockInput._promptMock).toHaveBeenCalled()
})
})