diff --git a/phases/results-and-pipeline/index.json b/phases/results-and-pipeline/index.json
index c292e67..71fcfdf 100644
--- a/phases/results-and-pipeline/index.json
+++ b/phases/results-and-pipeline/index.json
@@ -21,7 +21,10 @@
{
"step": 2,
"name": "linear-static-analysis",
- "status": "pending"
+ "status": "completed",
+ "started_at": "2026-08-01T00:51:34+0900",
+ "summary": "Added LinearStaticAnalysis orchestration through serial assembly, BC elimination, PARDISO with an all-constrained bypass, reaction recovery, and deterministic nodal results with equilibrium tests.",
+ "completed_at": "2026-08-01T01:01:15+0900"
},
{
"step": 3,
diff --git a/phases/results-and-pipeline/step2-output.json b/phases/results-and-pipeline/step2-output.json
new file mode 100644
index 0000000..8be4567
--- /dev/null
+++ b/phases/results-and-pipeline/step2-output.json
@@ -0,0 +1,7 @@
+{
+ "step": 2,
+ "name": "linear-static-analysis",
+ "exitCode": 0,
+ "stdout": "{\"type\":\"thread.started\",\"thread_id\":\"019fb8df-af6d-7013-8b89-1e9ef1ceab5e\"}\n{\"type\":\"item.completed\",\"item\":{\"id\":\"item_0\",\"type\":\"error\",\"message\":\"`--dangerously-bypass-hook-trust` is enabled. Enabled hooks may run without review for this invocation.\"}}\n{\"type\":\"item.completed\",\"item\":{\"id\":\"item_1\",\"type\":\"error\",\"message\":\"`--dangerously-bypass-hook-trust` is enabled. Enabled hooks may run without review for this invocation.\"}}\n{\"type\":\"turn.started\"}\n{\"type\":\"item.completed\",\"item\":{\"id\":\"item_2\",\"type\":\"agent_message\",\"text\":\"Step 2 범위만 구현하겠습니다. 먼저 `harness`와 Superpowers의 설계·계획·TDD·완료 검증 지침을 읽고, 현재 브랜치/작업 트리와 Step 0–1 산출물을 확인한 뒤 실패 테스트부터 진행하겠습니다. CLI·HDF5·다중 step 기능은 건드리지 않겠습니다.\"}}\n{\"type\":\"item.started\",\"item\":{\"id\":\"item_3\",\"type\":\"command_execution\",\"command\":\"\\\"C:\\\\\\\\WINDOWS\\\\\\\\System32\\\\\\\\WindowsPowerShell\\\\\\\\v1.0\\\\\\\\powershell.exe\\\" -Command \\\"Get-Content -Raw -LiteralPath 'C:\\\\\\\\Users\\\\\\\\baram\\\\\\\\.codex\\\\\\\\plugins\\\\\\\\cache\\\\\\\\openai-curated-remote\\\\\\\\superpowers\\\\\\\\6.2.0\\\\\\\\skills\\\\\\\\using-superpowers\\\\\\\\SKILL.md'\\nGet-Content -Raw -LiteralPath 'C:\\\\\\\\git\\\\\\\\FESA\\\\\\\\.agents\\\\\\\\skills\\\\\\\\harness\\\\\\\\SKILL.md'\\nGet-Content -Raw -LiteralPath 'C:\\\\\\\\Users\\\\\\\\baram\\\\\\\\.codex\\\\\\\\plugins\\\\\\\\cache\\\\\\\\openai-curated-remote\\\\\\\\superpowers\\\\\\\\6.2.0\\\\\\\\skills\\\\\\\\brainstorming\\\\\\\\SKILL.md'\\nGet-Content -Raw -LiteralPath 'C:\\\\\\\\Users\\\\\\\\baram\\\\\\\\.codex\\\\\\\\plugins\\\\\\\\cache\\\\\\\\openai-curated-remote\\\\\\\\superpowers\\\\\\\\6.2.0\\\\\\\\skills\\\\\\\\writing-plans\\\\\\\\SKILL.md'\\nGet-Content -Raw -LiteralPath 'C:\\\\\\\\Users\\\\\\\\baram\\\\\\\\.codex\\\\\\\\plugins\\\\\\\\cache\\\\\\\\openai-curated-remote\\\\\\\\superpowers\\\\\\\\6.2.0\\\\\\\\skills\\\\\\\\test-driven-development\\\\\\\\SKILL.md'\\nGet-Content -Raw -LiteralPath 'C:\\\\\\\\Users\\\\\\\\baram\\\\\\\\.codex\\\\\\\\plugins\\\\\\\\cache\\\\\\\\openai-curated-remote\\\\\\\\superpowers\\\\\\\\6.2.0\\\\\\\\skills\\\\\\\\verification-before-completion\\\\\\\\SKILL.md'\\\"\",\"aggregated_output\":\"\",\"exit_code\":null,\"status\":\"in_progress\"}}\n{\"type\":\"item.completed\",\"item\":{\"id\":\"item_3\",\"type\":\"command_execution\",\"command\":\"\\\"C:\\\\\\\\WINDOWS\\\\\\\\System32\\\\\\\\WindowsPowerShell\\\\\\\\v1.0\\\\\\\\powershell.exe\\\" -Command \\\"Get-Content -Raw -LiteralPath 'C:\\\\\\\\Users\\\\\\\\baram\\\\\\\\.codex\\\\\\\\plugins\\\\\\\\cache\\\\\\\\openai-curated-remote\\\\\\\\superpowers\\\\\\\\6.2.0\\\\\\\\skills\\\\\\\\using-superpowers\\\\\\\\SKILL.md'\\nGet-Content -Raw -LiteralPath 'C:\\\\\\\\git\\\\\\\\FESA\\\\\\\\.agents\\\\\\\\skills\\\\\\\\harness\\\\\\\\SKILL.md'\\nGet-Content -Raw -LiteralPath 'C:\\\\\\\\Users\\\\\\\\baram\\\\\\\\.codex\\\\\\\\plugins\\\\\\\\cache\\\\\\\\openai-curated-remote\\\\\\\\superpowers\\\\\\\\6.2.0\\\\\\\\skills\\\\\\\\brainstorming\\\\\\\\SKILL.md'\\nGet-Content -Raw -LiteralPath 'C:\\\\\\\\Users\\\\\\\\baram\\\\\\\\.codex\\\\\\\\plugins\\\\\\\\cache\\\\\\\\openai-curated-remote\\\\\\\\superpowers\\\\\\\\6.2.0\\\\\\\\skills\\\\\\\\writing-plans\\\\\\\\SKILL.md'\\nGet-Content -Raw -LiteralPath 'C:\\\\\\\\Users\\\\\\\\baram\\\\\\\\.codex\\\\\\\\plugins\\\\\\\\cache\\\\\\\\openai-curated-remote\\\\\\\\superpowers\\\\\\\\6.2.0\\\\\\\\skills\\\\\\\\test-driven-development\\\\\\\\SKILL.md'\\nGet-Content -Raw -LiteralPath 'C:\\\\\\\\Users\\\\\\\\baram\\\\\\\\.codex\\\\\\\\plugins\\\\\\\\cache\\\\\\\\openai-curated-remote\\\\\\\\superpowers\\\\\\\\6.2.0\\\\\\\\skills\\\\\\\\verification-before-completion\\\\\\\\SKILL.md'\\\"\",\"aggregated_output\":\"---\\nname: using-superpowers\\ndescription: Use when starting any conversation - establishes how to find and use skills, requiring skill invocation before ANY response including clarifying questions\\n---\\n\\n\\nIf you were dispatched as a subagent to execute a specific task, ignore this skill.\\n\\n\\n\\nIf you think there is even a 1% chance a skill might apply to what you are doing, you ABSOLUTELY MUST invoke the skill.\\n\\nIF A SKILL APPLIES TO YOUR TASK, YOU DO NOT HAVE A CHOICE. YOU MUST USE IT.\\n\\nThis is not negotiable. You cannot rationalize your way out of this.\\n\\n\\n## The Rule\\n\\n**Invoke relevant or requested skills BEFORE any response or action** ??including clarifying questions, exploring the codebase, or checking files. If it turns out wrong for the situation, you don't have to use it.\\n\\n**Before entering plan mode:** if you haven't already brainstormed, invoke the brainstorming skill first.\\n\\nThen announce \\\"Using [skill] to [purpose]\\\" and follow the skill exactly. If it has a checklist, create a todo per item.\\n\\n## Skill Priority\\n\\nWhen multiple skills apply, process skills come first ??they set the approach, then implementation skills (frontend-design, etc.) carry it out. Brainstorming and systematic-debugging are Superpowers' most common process skills, but the rule holds for any of them.\\n\\n- \\\"Let's build X\\\" ??superpowers:brainstorming first, then implementation skills.\\n- \\\"Fix this bug\\\" ??superpowers:systematic-debugging first, then domain skills.\\n\\n## Red Flags\\n\\nThese thoughts mean STOP?봸ou're rationalizing:\\n\\n| Thought | Reality |\\n|---------|---------|\\n| \\\"This is just a simple question\\\" | Questions are tasks. Check for skills. |\\n| \\\"I need more context first\\\" | Skill check comes BEFORE clarifying questions. |\\n| \\\"Let me explore the codebase first\\\" | Skills tell you HOW to explore. Check first. |\\n| \\\"I can check git/files quickly\\\" | Files lack conversation context. Check for skills. |\\n| \\\"Let me gather information first\\\" | Skills tell you HOW to gather information. |\\n| \\\"This doesn't need a formal skill\\\" | If a skill exists, use it. |\\n| \\\"I remember this skill\\\" | Skills evolve. Read current version. |\\n| \\\"This doesn't count as a task\\\" | Action = task. Check for skills. |\\n| \\\"The skill is overkill\\\" | Simple things become complex. Use it. |\\n| \\\"I'll just do this one thing first\\\" | Check BEFORE doing anything. |\\n| \\\"This feels productive\\\" | Undisciplined action wastes time. Skills prevent this. |\\n| \\\"I know what that means\\\" | Knowing the concept ??using the skill. Invoke it. |\\n\\n## Platform Adaptation\\n\\nIf your harness appears here, read its reference file for special instructions:\\n\\n- Codex: `references/codex-tools.md`\\n- Pi: `references/pi-tools.md`\\n- Antigravity: `references/antigravity-tools.md`\\n\\n## User Instructions\\n\\nUser instructions (CLAUDE.md, AGENTS.md, GEMINI.md, etc, direct requests) take precedence over skills, which in turn override default behavior. Only skip skill workflows or instructions when your human partner has explicitly told you to.\\n\\r\\n---\\r\\nname: harness\\r\\ndescription: Use when planning agentic implementation phases, creating phases/index.json and self-contained step files, or running the Harness step executor.\\r\\n---\\r\\n\\r\\n# Harness Workflow\\r\\n\\r\\n???꾨줈?앺듃??Harness ?꾨젅?꾩썙?щ? ?ъ슜?쒕떎. ?꾨옒 ?뚰겕?뚮줈???곕씪 ?묒뾽?쒕떎.\\r\\n\\r\\n## A. ?먯깋\\r\\n\\r\\n`AGENTS.md`? `docs/` ?섏쐞 臾몄꽌(PRD, ARCHITECTURE, ADR ??瑜??쎄퀬 ?꾨줈?앺듃??湲고쉷,\\r\\n?꾪궎?띿쿂, ?ㅺ퀎 ?섎룄瑜??뚯븙?쒕떎. 蹂묐젹 ?먯깋???ㅼ젣濡??좎슜?섍퀬 ?꾩옱 ?몄뀡?먯꽌 ?덉슜??\\n?뚮쭔 Codex subagent瑜??좏깮?곸쑝濡??ъ슜?쒕떎.\\r\\n\\r\\n## B. ?쇱쓽\\r\\n\\r\\n援ы쁽???꾪빐 援ъ껜?뷀븯嫄곕굹 湲곗닠?곸쑝濡?寃곗젙?댁빞 ???ы빆???덉쑝硫??ъ슜?먯뿉寃???踰덉뿉\\r\\n?섎굹???쒖떆?섍퀬 ?쇱쓽?쒕떎.\\r\\n\\r\\n## C. Step ?ㅺ퀎\\r\\n\\r\\n?ъ슜?먭? 援ы쁽 怨꾪쉷 ?묒꽦??吏?쒗븯硫??щ윭 step?쇰줈 ?섎돏 珥덉븞???묒꽦???쇰뱶諛깆쓣\\r\\n?붿껌?쒕떎.\\r\\n\\r\\n?ㅺ퀎 ?먯튃:\\r\\n\\r\\n1. **Scope 理쒖냼??* ???섎굹??step?먯꽌 ?섎굹???덉씠???먮뒗 紐⑤뱢留??ㅻ,?? ?щ윭\\r\\n 紐⑤뱢???숈떆???섏젙?댁빞 ?섎㈃ step??履쇨컿??\\r\\n2. **?먭린?꾧껐??* ??媛?step ?뚯씪? ?낅┰??Codex ?ㅽ뻾?먯꽌 ?ъ슜?쒕떎. ?몃? ???\\n 李몄“瑜?湲덉??섍퀬 ?꾩슂???뺣낫瑜?紐⑤몢 ?뚯씪 ?덉뿉 ?곷뒗??\\r\\n3. **?ъ쟾 以鍮?媛뺤젣** ??愿??臾몄꽌? ?댁쟾 step?먯꽌 ?앹꽦?섍굅???섏젙???뚯씪 寃쎈줈瑜?\\n 紐낆떆?쒕떎.\\r\\n4. **?쒓렇?덉쿂 ?섏? 吏??* ???⑥닔? ?대옒?ㅼ쓽 ?명꽣?섏씠?ㅻ? ?쒖떆?섍퀬 ?대? 援ы쁽?\\r\\n Codex ?щ웾??留↔릿?? 硫깅벑?? 蹂댁븞, ?곗씠??臾닿껐??媛숈? ?듭떖 洹쒖튃? 紐낆떆?쒕떎.\\r\\n5. **AC???ㅽ뻾 媛?ν븳 command** ??異붿긽??議곌굔 ????ㅼ젣 鍮뚮뱶? ?뚯뒪??command瑜?\\n ?ы븿?쒕떎.\\r\\n6. **二쇱쓽?ы빆? 援ъ껜?곸쑝濡?* ??\\\"X瑜??섏? 留덈씪. ?댁쑀: Y\\\" ?뺤떇?쇰줈 ?곷뒗??\\r\\n7. **?ㅼ씠諛?* ??step name? ?듭떖 ?묒뾽???쒗쁽?섎뒗 kebab-case slug濡??뺥븳??\\r\\n\\r\\n## D. ?뚯씪 ?앹꽦\\r\\n\\r\\n?ъ슜?먭? 珥덉븞???뱀씤???꾩뿉留??ㅼ쓬 ?뚯씪???앹꽦?쒕떎.\\r\\n\\r\\n### D-1. `phases/index.json`\\r\\n\\r\\n?щ윭 task瑜?愿由ы븯??top-level ?몃뜳?ㅻ떎. ?대? 議댁옱?섎㈃ `phases` 諛곗뿴??????ぉ??\\n異붽??쒕떎.\\r\\n\\r\\n```json\\r\\n{\\r\\n \\\"phases\\\": [\\r\\n {\\r\\n \\\"dir\\\": \\\"0-mvp\\\",\\r\\n \\\"status\\\": \\\"pending\\\"\\r\\n }\\r\\n ]\\r\\n}\\r\\n```\\r\\n\\r\\n- `dir`: task ?붾젆?곕━紐?\\n- `status`: `pending` | `completed` | `error` | `blocked`\\r\\n- timestamp??executor媛 ?곹깭瑜?諛붽? ??湲곕줉?섎?濡??앹꽦 ???l? ?딅뒗??\\r\\n\\r\\n### D-2. `phases/{task-name}/index.json`\\r\\n\\r\\n```json\\r\\n{\\r\\n \\\"project\\\": \\\"\\\",\\r\\n \\\"steps\\\": [\\r\\n { \\\"step\\\": 0, \\\"name\\\": \\\"project-setup\\\", \\\"status\\\": \\\"pending\\\" },\\r\\n { \\\"step\\\": 1, \\\"name\\\": \\\"core-types\\\", \\\"status\\\": \\\"pending\\\" },\\r\\n { \\\"step\\\": 2, \\\"name\\\": \\\"api-layer\\\", \\\"status\\\": \\\"pending\\\" }\\r\\n ]\\r\\n}\\r\\n```\\r\\n\\r\\n?꾨뱶 洹쒖튃:\\r\\n\\r\\n- `project`: `AGENTS.md`???뺤쓽???꾨줈?앺듃紐?\\n- `phase`: task ?대쫫?대ʼn ?붾젆?곕━紐낃낵 ?쇱튂\\r\\n- `steps[].step`: 0遺???쒖옉?섎뒗 ?쒕쾲\\r\\n- `steps[].name`: kebab-case slug\\r\\n- `steps[].status`: 珥덇린媛?`pending`\\r\\n\\r\\n?곹깭? 湲곕줉 二쇱껜:\\r\\n\\r\\n| ?꾩씠 | 湲곕줉 ?꾨뱶 | 湲곕줉 二쇱껜 |\\r\\n|------|-----------|-----------|\\r\\n| `completed` | `summary`, `completed_at` | Codex媛 summary, executor媛 timestamp |\\r\\n| `error` | `error_message`, `failed_at` | Codex媛 message, executor媛 timestamp |\\r\\n| `blocked` | `blocked_reason`, `blocked_at` | Codex媛 reason, executor媛 timestamp |\\r\\n\\r\\n`summary`?먮뒗 ?ㅼ쓬 step???좎슜???앹꽦 ?뚯씪怨??듭떖 寃곗젙????以꾨줈 ?곷뒗??\\r\\ntask `created_at`怨?step `started_at`? executor媛 湲곕줉?섎?濡??앹꽦 ???l? ?딅뒗??\\r\\n\\r\\n### D-3. `phases/{task-name}/step{N}.md`\\r\\n\\r\\n````markdown\\r\\n# Step {N}: {?대쫫}\\r\\n\\r\\n## ?쎌뼱?????뚯씪\\r\\n\\r\\n癒쇱? ?꾨옒 ?뚯씪???쎄퀬 ?꾨줈?앺듃???꾪궎?띿쿂? ?ㅺ퀎 ?섎룄瑜??뚯븙?섎씪:\\r\\n\\r\\n- `/AGENTS.md`\\r\\n- `/docs/ARCHITECTURE.md`\\r\\n- `/docs/ADR.md`\\r\\n- ?댁쟾 step?먯꽌 ?앹꽦?섍굅???섏젙???뚯씪 寃쎈줈\\r\\n\\r\\n?댁쟾 step??肄붾뱶瑜?瑗쇨세???쎄퀬 ?ㅺ퀎 ?섎룄瑜??댄빐?????묒뾽?섎씪.\\r\\n\\r\\n## ?묒뾽\\r\\n\\r\\n援ъ껜?곸씤 援ы쁽 吏?쒕? ?뚯씪 寃쎈줈, ?대옒?ㅼ? ?⑥닔 ?쒓렇?덉쿂, 濡쒖쭅 ?ㅻ챸怨??④퍡 ?곷뒗??\\r\\n援ы쁽泥대뒗 Codex??留↔린???ㅺ퀎 ?섎룄?먯꽌 踰쀬뼱?섎㈃ ???섎뒗 ?듭떖 洹쒖튃? 紐낆떆?쒕떎.\\r\\n\\r\\n## Acceptance Criteria\\r\\n\\r\\n?꾨줈?앺듃 ?뺤떇??留욌뒗 紐낅졊???ъ슜?쒕떎. `.harness/config.json`???덉쑝硫??대떦 preset,\\r\\nsolution, configuration, platform, test command瑜??곗꽑?쒕떎.\\r\\n\\r\\n```powershell\\r\\n# CMake\\r\\ncmake --build .harness/build --config Debug\\r\\nctest --test-dir .harness/build -C Debug --output-on-failure\\r\\n\\r\\n# 吏곸젒 MSBuild\\r\\nMSBuild.exe MyProject.sln /m /p:Configuration=Debug /p:Platform=x64\\r\\n.\\\\build\\\\tests\\\\Debug\\\\MyProjectTests.exe\\r\\n```\\r\\n\\r\\n## 寃利??덉감\\r\\n\\r\\n1. Acceptance Criteria command瑜??ㅽ뻾?쒕떎.\\r\\n2. ARCHITECTURE ?붾젆?곕━ 援ъ“瑜??곕Ⅴ?붿? ?뺤씤?쒕떎.\\r\\n3. ADR 湲곗닠 ?ㅽ깮怨?`AGENTS.md` CRITICAL 洹쒖튃???뺤씤?쒕떎.\\r\\n4. 寃곌낵???곕씪 task index???대떦 step??媛깆떊?쒕떎.\\r\\n - ?깃났: `status`瑜?`completed`濡?諛붽씀怨???以?`summary` 湲곕줉\\r\\n - ?섏젙 3?????ㅽ뙣: `status`瑜?`error`濡?諛붽씀怨?`error_message` 湲곕줉\\r\\n - ?ъ슜??媛쒖엯 ?꾩슂: `status`瑜?`blocked`濡?諛붽씀怨?`blocked_reason` 湲곕줉 ??以묐떒\\r\\n\\r\\n## 湲덉??ы빆\\r\\n\\r\\n- ??step??踰붿쐞 諛?湲곕뒫??異붽??섏? 留덈씪. ?댁쑀: step???낅┰?깆쓣 源⑤쑉由곕떎.\\r\\n- 湲곗〈 ?뚯뒪?몃? 源⑤쑉由ъ? 留덈씪. ?댁쑀: ?댁쟾 ?숈옉???뚭??쒗궓??\\r\\n````\\r\\n\\r\\n## E. ?ㅽ뻾\\r\\n\\r\\n```bash\\r\\npython scripts/execute.py {task-name}\\r\\npython scripts/execute.py {task-name} --push\\r\\n```\\r\\n\\r\\n?섍꼍?먯꽌 Python 3 ?ㅽ뻾 紐낅졊??`python3`?대㈃ 洹?紐낅졊??????ъ슜?쒕떎.\\r\\n\\r\\nexecutor媛 泥섎━?섎뒗 ?묒뾽:\\r\\n\\r\\n- `feat-{task-name}` 釉뚮옖移??앹꽦 ?먮뒗 checkout\\r\\n- `AGENTS.md`? `docs/*.md` guardrail 二쇱엯\\r\\n- ?꾨즺 step??summary瑜??ㅼ쓬 prompt???꾩쟻\\r\\n- ?ㅽ뙣 ??理쒕? 3???ъ떆?꾪븯硫??댁쟾 ?ㅻ쪟瑜?prompt???꾨떖\\r\\n- 肄붾뱶 蹂寃쎄낵 metadata瑜?遺꾨━??commit\\r\\n- `started_at`, `completed_at`, `failed_at`, `blocked_at` 湲곕줉\\r\\n\\r\\n?먮윭 蹂듦뎄:\\r\\n\\r\\n- `error`: ?대떦 status瑜?`pending`?쇰줈 諛붽씀怨?`error_message`瑜???젣?????ъ떎??\\n- `blocked`: ?먯씤???닿껐?섍퀬 status瑜?`pending`?쇰줈 諛붽씀怨?`blocked_reason`????젣??\\n ???ъ떎??\\n\\r\\n---\\nname: brainstorming\\ndescription: \\\"You MUST use this before any creative work - creating features, building components, adding functionality, or modifying behavior. Explores user intent, requirements and design before implementation.\\\"\\n---\\n\\n# Brainstorming Ideas Into Designs\\n\\nHelp turn ideas into fully formed designs and specs through natural collaborative dialogue.\\n\\nStart by understanding the current project context, then ask questions one at a time to refine the idea. Once you understand what you're building, present the design and get user approval.\\n\\n\\nDo NOT invoke any implementation skill, write any code, scaffold any project, or take any implementation action until you have presented a design and the user has approved it. This applies to EVERY project regardless of perceived simplicity.\\n\\n\\n## Anti-Pattern: \\\"This Is Too Simple To Need A Design\\\"\\n\\nEvery project goes through this process. A todo list, a single-function utility, a config change ??all of them. \\\"Simple\\\" projects are where unexamined assumptions cause the most wasted work. The design can be short (a few sentences for truly simple projects), but you MUST present it and get approval.\\n\\n## Checklist\\n\\nYou MUST create a task for each of these items and complete them in order:\\n\\n1. **Explore project context** ??check files, docs, recent commits\\n2. **Offer the visual companion just-in-time** ??NOT upfront. The first time a question would genuinely be clearer shown than described, offer it then (its own message); on approval its browser tab opens for you. If no visual question ever arises, never offer it. See the Visual Companion section below.\\n3. **Ask clarifying questions** ??one at a time, understand purpose/constraints/success criteria\\n4. **Propose 2-3 approaches** ??with trade-offs and your recommendation\\n5. **Present design** ??in sections scaled to their complexity, get user approval after each section\\n6. **Write design doc** ??save to `docs/superpowers/specs/YYYY-MM-DD--design.md` and commit\\n7. **Spec self-review** ??quick inline check for placeholders, contradictions, ambiguity, scope (see below)\\n8. **User reviews written spec** ??ask user to review the spec file before proceeding\\n9. **Transition to implementation** ??invoke writing-plans skill to create implementation plan\\n\\n## Process Flow\\n\\n```dot\\ndigraph brainstorming {\\n \\\"Explore project context\\\" [shape=box];\\n \\\"Ask clarifying questions\\\" [shape=box];\\n \\\"Propose 2-3 approaches\\\" [shape=box];\\n \\\"Present design sections\\\" [shape=box];\\n \\\"User approves design?\\\" [shape=diamond];\\n \\\"Write design doc\\\" [shape=box];\\n \\\"Spec self-review\\\\n(fix inline)\\\" [shape=box];\\n \\\"User reviews spec?\\\" [shape=diamond];\\n \\\"Invoke writing-plans skill\\\" [shape=doublecircle];\\n\\n \\\"Explore project context\\\" -> \\\"Ask clarifying questions\\\";\\n \\\"Ask clarifying questions\\\" -> \\\"Propose 2-3 approaches\\\";\\n \\\"Propose 2-3 approaches\\\" -> \\\"Present design sections\\\";\\n \\\"Present design sections\\\" -> \\\"User approves design?\\\";\\n \\\"User approves design?\\\" -> \\\"Present design sections\\\" [label=\\\"no, revise\\\"];\\n \\\"User approves design?\\\" -> \\\"Write design doc\\\" [label=\\\"yes\\\"];\\n \\\"Write design doc\\\" -> \\\"Spec self-review\\\\n(fix inline)\\\";\\n \\\"Spec self-review\\\\n(fix inline)\\\" -> \\\"User reviews spec?\\\";\\n \\\"User reviews spec?\\\" -> \\\"Write design doc\\\" [label=\\\"changes requested\\\"];\\n \\\"User reviews spec?\\\" -> \\\"Invoke writing-plans skill\\\" [label=\\\"approved\\\"];\\n}\\n```\\n\\n**The terminal state is invoking writing-plans.** Do NOT invoke frontend-design, mcp-builder, or any other implementation skill. The ONLY skill you invoke after brainstorming is writing-plans.\\n\\n## The Process\\n\\n**Understanding the idea:**\\n\\n- Check out the current project state first (files, docs, recent commits)\\n- Before asking detailed questions, assess scope: if the request describes multiple independent subsystems (e.g., \\\"build a platform with chat, file storage, billing, and analytics\\\"), flag this immediately. Don't spend questions refining details of a project that needs to be decomposed first.\\n- If the project is too large for a single spec, help the user decompose into sub-projects: what are the independent pieces, how do they relate, what order should they be built? Then brainstorm the first sub-project through the normal design flow. Each sub-project gets its own spec ??plan ??implementation cycle.\\n- For appropriately-scoped projects, ask questions one at a time to refine the idea\\n- Prefer multiple choice questions when possible, but open-ended is fine too\\n- Only one question per message - if a topic needs more exploration, break it into multiple questions\\n- Focus on understanding: purpose, constraints, success criteria\\n\\n**Exploring approaches:**\\n\\n- Propose 2-3 different approaches with trade-offs\\n- Present options conversationally with your recommendation and reasoning\\n- Lead with your recommended option and explain why\\n- YAGNI ruthlessly - remove unnecessary features from every approach and design\\n\\n**Presenting the design:**\\n\\n- Once you believe you understand what you're building, present the design\\n- Scale each section to its complexity: a few sentences if straightforward, up to 200-300 words if nuanced\\n- Ask after each section whether it looks right so far\\n- Cover: architecture, components, data flow, error handling, testing\\n- Be ready to go back and clarify if something doesn't make sense\\n\\n**Design for isolation and clarity:**\\n\\n- Break the system into smaller units that each have one clear purpose, communicate through well-defined interfaces, and can be understood and tested independently\\n- For each unit, you should be able to answer: what does it do, how do you use it, and what does it depend on?\\n- Can someone understand what a unit does without reading its internals? Can you change the internals without breaking consumers? If not, the boundaries need work.\\n- Smaller, well-bounded units are also easier for you to work with - you reason better about code you can hold in context at once, and your edits are more reliable when files are focused. When a file grows large, that's often a signal that it's doing too much.\\n\\n**Working in existing codebases:**\\n\\n- Explore the current structure before proposing changes. Follow existing patterns.\\n- Where existing code has problems that affect the work (e.g., a file that's grown too large, unclear boundaries, tangled responsibilities), include targeted improvements as part of the design - the way a good developer improves code they're working in.\\n- Don't propose unrelated refactoring. Stay focused on what serves the current goal.\\n\\n## After the Design\\n\\n**Documentation:**\\n\\n- Write the validated design (spec) to `docs/superpowers/specs/YYYY-MM-DD--design.md`\\n - (User preferences for spec location override this default)\\n- Use elements-of-style:writing-clearly-and-concisely skill if available\\n- Commit the design document to git\\n\\n**Spec Self-Review:**\\nAfter writing the spec document, look at it with fresh eyes:\\n\\n1. **Placeholder scan:** Any \\\"TBD\\\", \\\"TODO\\\", incomplete sections, or vague requirements? Fix them.\\n2. **Internal consistency:** Do any sections contradict each other? Does the architecture match the feature descriptions?\\n3. **Scope check:** Is this focused enough for a single implementation plan, or does it need decomposition?\\n4. **Ambiguity check:** Could any requirement be interpreted two different ways? If so, pick one and make it explicit.\\n\\nFix any issues inline. No need to re-review ??just fix and move on.\\n\\n**User Review Gate:**\\nAfter the spec review loop passes, ask the user to review the written spec before proceeding:\\n\\n> \\\"Spec written and committed to ``. Please review it and let me know if you want to make any changes before we start writing out the implementation plan.\\\"\\n\\nWait for the user's response. If they request changes, make them and re-run the spec review loop. Only proceed once the user approves.\\n\\n**Implementation:**\\n\\n- Invoke the writing-plans skill to create a detailed implementation plan\\n- Do NOT invoke any other skill. writing-plans is the next step.\\n\\n## Visual Companion\\n\\nA browser-based companion for showing mockups, diagrams, and visual options during brainstorming. Available as a tool ??not a mode. Accepting the companion means it's available for questions that benefit from visual treatment; it does NOT mean every question goes through the browser.\\n\\n**Offering the companion (just-in-time):** Do NOT offer it upfront. Wait until a question would genuinely be clearer shown than told ??a real mockup / layout / diagram question, not merely a UI *topic*. The first time that happens, offer it then, as its own message:\\n> \\\"This next part might be easier if I show you ??I can put together mockups, diagrams, and comparisons in a browser tab as we go. It's still new and can be token-intensive. Want me to? I'll open it for you.\\\"\\n\\n**This offer MUST be its own message.** Only the offer ??no clarifying question, summary, or other content. Wait for the user's response. If they accept, start the server with `--open` so their browser opens to the first screen automatically. If they decline, continue text-only and don't offer again unless they raise it.\\n\\n**Per-question decision:** Even after the user accepts, decide FOR EACH QUESTION whether to use the browser or the terminal. The test: **would the user understand this better by seeing it than reading it?**\\n\\n- **Use the browser** for content that IS visual ??mockups, wireframes, layout comparisons, architecture diagrams, side-by-side visual designs\\n- **Use the terminal** for content that is text ??requirements questions, conceptual choices, tradeoff lists, A/B/C/D text options, scope decisions\\n\\nA question about a UI topic is not automatically a visual question. \\\"What does personality mean in this context?\\\" is a conceptual question ??use the terminal. \\\"Which wizard layout works better?\\\" is a visual question ??use the browser.\\n\\nIf they agree to the companion, read the detailed guide before proceeding:\\n`skills/brainstorming/visual-companion.md`\\n\\r\\n---\\nname: writing-plans\\ndescription: Use when you have a spec or requirements for a multi-step task, before touching code\\n---\\n\\n# Writing Plans\\n\\n## Overview\\n\\nWrite comprehensive implementation plans assuming the engineer has zero context for our codebase and questionable taste. Document everything they need to know: which files to touch for each task, code, testing, docs they might need to check, how to test it. Give them the whole plan as bite-sized tasks. DRY. YAGNI. TDD. Frequent commits.\\n\\nAssume they are a skilled developer, but know almost nothing about our toolset or problem domain. Assume they don't know good test design very well.\\n\\n**Announce at start:** \\\"I'm using the writing-plans skill to create the implementation plan.\\\"\\n\\n**Context:** If working in an isolated worktree, it should have been created via the `superpowers:using-git-worktrees` skill at execution time.\\n\\n**Save plans to:** `docs/superpowers/plans/YYYY-MM-DD-.md`\\n- (User preferences for plan location override this default)\\n\\n## Scope Check\\n\\nIf the spec covers multiple independent subsystems, it should have been broken into sub-project specs during brainstorming. If it wasn't, suggest breaking this into separate plans ??one per subsystem. Each plan should produce working, testable software on its own.\\n\\n## File Structure\\n\\nBefore defining tasks, map out which files will be created or modified and what each one is responsible for. This is where decomposition decisions get locked in.\\n\\n- Design units with clear boundaries and well-defined interfaces. Each file should have one clear responsibility.\\n- You reason best about code you can hold in context at once, and your edits are more reliable when files are focused. Prefer smaller, focused files over large ones that do too much.\\n- Files that change together should live together. Split by responsibility, not by technical layer.\\n- In existing codebases, follow established patterns. If the codebase uses large files, don't unilaterally restructure - but if a file you're modifying has grown unwieldy, including a split in the plan is reasonable.\\n\\nThis structure informs the task decomposition. Each task should produce self-contained changes that make sense independently.\\n\\n## Task Right-Sizing\\n\\nA task is the smallest unit that carries its own test cycle and is worth a\\nfresh reviewer's gate. When drawing task boundaries: fold setup,\\nconfiguration, scaffolding, and documentation steps into the task whose\\ndeliverable needs them; split only where a reviewer could meaningfully\\nreject one task while approving its neighbor. Each task ends with an\\nindependently testable deliverable.\\n\\n## Bite-Sized Task Granularity\\n\\n**Each step is one action (2-5 minutes):**\\n- \\\"Write the failing test\\\" - step\\n- \\\"Run it to make sure it fails\\\" - step\\n- \\\"Implement the minimal code to make the test pass\\\" - step\\n- \\\"Run the tests and make sure they pass\\\" - step\\n- \\\"Commit\\\" - step\\n\\n## Plan Document Header\\n\\n**Every plan MUST start with this header:**\\n\\n```markdown\\n# [Feature Name] Implementation Plan\\n\\n> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking.\\n\\n**Goal:** [One sentence describing what this builds]\\n\\n**Architecture:** [2-3 sentences about approach]\\n\\n**Tech Stack:** [Key technologies/libraries]\\n\\n## Global Constraints\\n\\n[The spec's project-wide requirements ??version floors, dependency limits,\\nnaming and copy rules, platform requirements ??one line each, with exact\\nvalues copied verbatim from the spec. Every task's requirements implicitly\\ninclude this section.]\\n\\n---\\n```\\n\\n## Task Structure\\n\\n````markdown\\n### Task N: [Component Name]\\n\\n**Files:**\\n- Create: `exact/path/to/file.py`\\n- Modify: `exact/path/to/existing.py:123-145`\\n- Test: `tests/exact/path/to/test.py`\\n\\n**Interfaces:**\\n- Consumes: [what this task uses from earlier tasks ??exact signatures]\\n- Produces: [what later tasks rely on ??exact function names, parameter\\n and return types. A task's implementer sees only their own task; this\\n block is how they learn the names and types neighboring tasks use.]\\n\\n- [ ] **Step 1: Write the failing test**\\n\\n```python\\ndef test_specific_behavior():\\n result = function(input)\\n assert result == expected\\n```\\n\\n- [ ] **Step 2: Run test to verify it fails**\\n\\nRun: `pytest tests/path/test.py::test_name -v`\\nExpected: FAIL with \\\"function not defined\\\"\\n\\n- [ ] **Step 3: Write minimal implementation**\\n\\n```python\\ndef function(input):\\n return expected\\n```\\n\\n- [ ] **Step 4: Run test to verify it passes**\\n\\nRun: `pytest tests/path/test.py::test_name -v`\\nExpected: PASS\\n\\n- [ ] **Step 5: Commit**\\n\\n```bash\\ngit add tests/path/test.py src/path/file.py\\ngit commit -m \\\"feat: add specific feature\\\"\\n```\\n````\\n\\n## No Placeholders\\n\\nEvery step must contain the actual content an engineer needs. These are **plan failures** ??never write them:\\n- \\\"TBD\\\", \\\"TODO\\\", \\\"implement later\\\", \\\"fill in details\\\"\\n- \\\"Add appropriate error handling\\\" / \\\"add validation\\\" / \\\"handle edge cases\\\"\\n- \\\"Write tests for the above\\\" (without actual test code)\\n- \\\"Similar to Task N\\\" (repeat the code ??the engineer may be reading tasks out of order)\\n- Steps that describe what to do without showing how (code blocks required for code steps)\\n- References to types, functions, or methods not defined in any task\\n\\n## Self-Review\\n\\nAfter writing the complete plan, look at the spec with fresh eyes and check the plan against it. This is a checklist you run yourself ??not a subagent dispatch.\\n\\n**1. Spec coverage:** Skim each section/requirement in the spec. Can you point to a task that implements it? List any gaps.\\n\\n**2. Placeholder scan:** Search your plan for red flags ??any of the patterns from the \\\"No Placeholders\\\" section above. Fix them.\\n\\n**3. Type consistency:** Do the types, method signatures, and property names you used in later tasks match what you defined in earlier tasks? A function called `clearLayers()` in Task 3 but `clearFullLayers()` in Task 7 is a bug.\\n\\nIf you find issues, fix them inline. No need to re-review ??just fix and move on. If you find a spec requirement with no task, add the task.\\n\\n## Execution Handoff\\n\\nAfter saving the plan, offer execution choice:\\n\\n**\\\"Plan complete and saved to `docs/superpowers/plans/.md`. Two execution options:**\\n\\n**1. Subagent-Driven (recommended)** - I dispatch a fresh subagent per task, review between tasks, fast iteration\\n\\n**2. Inline Execution** - Execute tasks in this session using executing-plans, batch execution with checkpoints\\n\\n**Which approach?\\\"**\\n\\n**If Subagent-Driven chosen:**\\n- **REQUIRED SUB-SKILL:** Use superpowers:subagent-driven-development\\n- Fresh subagent per task + two-stage review\\n\\n**If Inline Execution chosen:**\\n- **REQUIRED SUB-SKILL:** Use superpowers:executing-plans\\n- Batch execution with checkpoints for review\\n\\r\\n---\\nname: test-driven-development\\ndescription: Use when implementing any feature or bugfix, before writing implementation code\\n---\\n\\n# Test-Driven Development (TDD)\\n\\n## Overview\\n\\nWrite the test first. Watch it fail. Write minimal code to pass.\\n\\n**Core principle:** If you didn't watch the test fail, you don't know if it tests the right thing.\\n\\n**Violating the letter of the rules is violating the spirit of the rules.**\\n\\n## When to Use\\n\\n**Always:**\\n- New features\\n- Bug fixes\\n- Refactoring\\n- Behavior changes\\n\\n**Exceptions (ask your human partner):**\\n- Throwaway prototypes\\n- Generated code\\n- Configuration files\\n\\nThinking \\\"skip TDD just this once\\\"? Stop. That's rationalization.\\n\\n## The Iron Law\\n\\n```\\nNO PRODUCTION CODE WITHOUT A FAILING TEST FIRST\\n```\\n\\nWrite code before the test? Delete it. Start over.\\n\\n**No exceptions:**\\n- Don't keep it as \\\"reference\\\"\\n- Don't \\\"adapt\\\" it while writing tests\\n- Don't look at it\\n- Delete means delete\\n\\nImplement fresh from tests. Period.\\n\\n## Red-Green-Refactor\\n\\n```dot\\ndigraph tdd_cycle {\\n rankdir=LR;\\n red [label=\\\"RED\\\\nWrite failing test\\\", shape=box, style=filled, fillcolor=\\\"#ffcccc\\\"];\\n verify_red [label=\\\"Verify fails\\\\ncorrectly\\\", shape=diamond];\\n green [label=\\\"GREEN\\\\nMinimal code\\\", shape=box, style=filled, fillcolor=\\\"#ccffcc\\\"];\\n verify_green [label=\\\"Verify passes\\\\nAll green\\\", shape=diamond];\\n refactor [label=\\\"REFACTOR\\\\nClean up\\\", shape=box, style=filled, fillcolor=\\\"#ccccff\\\"];\\n next [label=\\\"Next\\\", shape=ellipse];\\n\\n red -> verify_red;\\n verify_red -> green [label=\\\"yes\\\"];\\n verify_red -> red [label=\\\"wrong\\\\nfailure\\\"];\\n green -> verify_green;\\n verify_green -> refactor [label=\\\"yes\\\"];\\n verify_green -> green [label=\\\"no\\\"];\\n refactor -> verify_green [label=\\\"stay\\\\ngreen\\\"];\\n verify_green -> next;\\n next -> red;\\n}\\n```\\n\\n### RED - Write Failing Test\\n\\nWrite one minimal test showing what should happen.\\n\\n\\n```typescript\\ntest('retries failed operations 3 times', async () => {\\n let attempts = 0;\\n const operation = () => {\\n attempts++;\\n if (attempts < 3) throw new Error('fail');\\n return 'success';\\n };\\n\\n const result = await retryOperation(operation);\\n\\n expect(result).toBe('success');\\n expect(attempts).toBe(3);\\n});\\n```\\nClear name, tests real behavior, one thing\\n\\n\\n\\n```typescript\\ntest('retry works', async () => {\\n const mock = jest.fn()\\n .mockRejectedValueOnce(new Error())\\n .mockRejectedValueOnce(new Error())\\n .mockResolvedValueOnce('success');\\n await retryOperation(mock);\\n expect(mock).toHaveBeenCalledTimes(3);\\n});\\n```\\nVague name, tests mock not code\\n\\n\\n**Requirements:**\\n- One behavior\\n- Clear name\\n- Real code (no mocks unless unavoidable)\\n\\n### Verify RED - Watch It Fail\\n\\n**MANDATORY. Never skip.**\\n\\n```bash\\nnpm test path/to/test.test.ts\\n```\\n\\nConfirm:\\n- Test fails (not errors)\\n- Failure message is expected\\n- Fails because feature missing (not typos)\\n\\n**Test passes?** You're testing existing behavior. Fix test.\\n\\n**Test errors?** Fix error, re-run until it fails correctly.\\n\\n### GREEN - Minimal Code\\n\\nWrite simplest code to pass the test.\\n\\n\\n```typescript\\nasync function retryOperation(fn: () => Promise): Promise {\\n for (let i = 0; i < 3; i++) {\\n try {\\n return await fn();\\n } catch (e) {\\n if (i === 2) throw e;\\n }\\n }\\n throw new Error('unreachable');\\n}\\n```\\nJust enough to pass\\n\\n\\n\\n```typescript\\nasync function retryOperation(\\n fn: () => Promise,\\n options?: {\\n maxRetries?: number;\\n backoff?: 'linear' | 'exponential';\\n onRetry?: (attempt: number) => void;\\n }\\n): Promise {\\n // YAGNI\\n}\\n```\\nOver-engineered\\n\\n\\nDon't add features, refactor other code, or \\\"improve\\\" beyond the test.\\n\\n### Verify GREEN - Watch It Pass\\n\\n**MANDATORY.**\\n\\n```bash\\nnpm test path/to/test.test.ts\\n```\\n\\nConfirm:\\n- Test passes\\n- Other tests still pass\\n- Output pristine (no errors, warnings)\\n\\n**Test fails?** Fix code, not test.\\n\\n**Other tests fail?** Fix now.\\n\\n### REFACTOR - Clean Up\\n\\nAfter green only:\\n- Remove duplication\\n- Improve names\\n- Extract helpers\\n\\nKeep tests green. Don't add behavior.\\n\\n### Repeat\\n\\nNext failing test for next feature.\\n\\n## Good Tests\\n\\n| Quality | Good | Bad |\\n|---------|------|-----|\\n| **Minimal** | One thing. \\\"and\\\" in name? Split it. | `test('validates email and domain and whitespace')` |\\n| **Clear** | Name describes behavior | `test('test1')` |\\n| **Shows intent** | Demonstrates desired API | Obscures what code should do |\\n\\nWhen writing or changing any test, read [writing-good-tests.md](writing-good-tests.md) for the rules that keep tests honest:\\n- Name the production change that would make the test fail ??before writing it\\n- Assert on real behavior, never on mock behavior\\n- Keep test-only code in test utilities, out of production classes\\n- Understand a dependency's side effects before mocking it\\n\\n## Common Rationalizations\\n\\n| Excuse | Reality |\\n|--------|---------|\\n| \\\"Too simple to test\\\" | Simple code breaks. Test takes 30 seconds. |\\n| \\\"I'll test after\\\" | Tests written after pass immediately ??which proves nothing. They may test the wrong thing, test the implementation instead of the behavior, or miss the edge case you forgot. You never watched it fail, so you never proved it can catch the bug. Test-first forces that failure. |\\n| \\\"Tests after achieve same goals (spirit not ritual)\\\" | Tests-after answer \\\"what does this do?\\\"; tests-first answer \\\"what should this do?\\\" Tests written after are biased by the code you already wrote ??you verify the cases you remembered, not the ones you'd have discovered. Coverage without proof the tests work. |\\n| \\\"Already manually tested\\\" | Manual testing is ad-hoc: no record of what you covered, no way to re-run it when the code changes, easy to forget cases under pressure. \\\"Worked when I tried it\\\" ??comprehensive. Automated tests run the same way every time. |\\n| \\\"Deleting X hours is wasteful\\\" | Sunk cost fallacy ??that time is already spent either way. The real choice: rewrite with TDD (high confidence) vs. keep it and bolt tests on after (low confidence, likely bugs). Keeping code you can't trust is the waste. |\\n| \\\"Keep as reference, write tests first\\\" | You'll adapt it. That's testing after. Delete means delete. |\\n| \\\"Need to explore first\\\" | Fine. Throw away exploration, start with TDD. |\\n| \\\"Test hard = design unclear\\\" | Listen to test. Hard to test = hard to use. |\\n| \\\"TDD will slow me down\\\" | TDD IS the pragmatic path: catches bugs before commit, prevents regressions, lets you refactor without fear. \\\"Pragmatic\\\" shortcuts mean debugging in production ??slower, not faster. |\\n| \\\"Manual test faster\\\" | Manual doesn't prove edge cases. You'll re-test every change. |\\n| \\\"Existing code has no tests\\\" | You're improving it. Add tests for existing code. |\\n\\n## Red Flags - STOP and Start Over\\n\\n- Code before test\\n- Test after implementation\\n- Test passes immediately\\n- Can't explain why test failed\\n- Tests added \\\"later\\\"\\n- Rationalizing \\\"just this once\\\"\\n- \\\"I already manually tested it\\\"\\n- \\\"Tests after achieve the same purpose\\\"\\n- \\\"It's about spirit not ritual\\\"\\n- \\\"Keep as reference\\\" or \\\"adapt existing code\\\"\\n- \\\"Already spent X hours, deleting is wasteful\\\"\\n- \\\"TDD is dogmatic, I'm being pragmatic\\\"\\n- \\\"This is different because...\\\"\\n\\n**All of these mean: Delete code. Start over with TDD.**\\n\\n## Example: Bug Fix\\n\\n**Bug:** Empty email accepted\\n\\n**RED**\\n```typescript\\ntest('rejects empty email', async () => {\\n const result = await submitForm({ email: '' });\\n expect(result.error).toBe('Email required');\\n});\\n```\\n\\n**Verify RED**\\n```bash\\n$ npm test\\nFAIL: expected 'Email required', got undefined\\n```\\n\\n**GREEN**\\n```typescript\\nfunction submitForm(data: FormData) {\\n if (!data.email?.trim()) {\\n return { error: 'Email required' };\\n }\\n // ...\\n}\\n```\\n\\n**Verify GREEN**\\n```bash\\n$ npm test\\nPASS\\n```\\n\\n**REFACTOR**\\nExtract validation for multiple fields if needed.\\n\\n## Verification Checklist\\n\\nBefore marking work complete:\\n\\n- [ ] Every new function/method has a test\\n- [ ] Watched each test fail before implementing\\n- [ ] Each test failed for expected reason (feature missing, not typo)\\n- [ ] Wrote minimal code to pass each test\\n- [ ] All tests pass\\n- [ ] Output pristine (no errors, warnings)\\n- [ ] Tests use real code (mocks only if unavoidable)\\n- [ ] Edge cases and errors covered\\n\\nCan't check all boxes? You skipped TDD. Start over.\\n\\n## When Stuck\\n\\n| Problem | Solution |\\n|---------|----------|\\n| Don't know how to test | Write wished-for API. Write assertion first. Ask your human partner. |\\n| Test too complicated | Design too complicated. Simplify interface. |\\n| Must mock everything | Code too coupled. Use dependency injection. |\\n| Test setup huge | Extract helpers. Still complex? Simplify design. |\\n\\n## Debugging Integration\\n\\nBug found? Write failing test reproducing it. Follow TDD cycle. Test proves fix and prevents regression.\\n\\nNever fix bugs without a test.\\n\\n## Final Rule\\n\\n```\\nProduction code ??test exists and failed first\\nOtherwise ??not TDD\\n```\\n\\nNo exceptions without your human partner's permission.\\n\\r\\n---\\nname: verification-before-completion\\ndescription: Use when about to claim work is complete, fixed, or passing, before committing or creating PRs - requires running verification commands and confirming output before making any success claims; evidence before assertions always\\n---\\n\\n# Verification Before Completion\\n\\n## Overview\\n\\n**Core principle:** Evidence before claims, always.\\n\\n**Violating the letter of this rule is violating the spirit of this rule.**\\n\\n## The Iron Law\\n\\n```\\nNO COMPLETION CLAIMS WITHOUT FRESH VERIFICATION EVIDENCE\\n```\\n\\nIf you haven't run the verification command in this message, you cannot claim it passes.\\n\\n## The Gate Function\\n\\n```\\nBEFORE claiming any status or expressing satisfaction:\\n\\n1. IDENTIFY: What command proves this claim?\\n2. RUN: Execute the FULL command (fresh, complete)\\n3. READ: Full output, check exit code, count failures\\n4. VERIFY: Does output confirm the claim?\\n - If NO: State actual status with evidence\\n - If YES: State claim WITH evidence\\n5. ONLY THEN: Make the claim\\n\\nSkip any step = lying, not verifying\\n```\\n\\n## Common Failures\\n\\n| Claim | Requires | Not Sufficient |\\n|-------|----------|----------------|\\n| Tests pass | Test command output: 0 failures | Previous run, \\\"should pass\\\" |\\n| Linter clean | Linter output: 0 errors | Partial check, extrapolation |\\n| Build succeeds | Build command: exit 0 | Linter passing, logs look good |\\n| Bug fixed | Test original symptom: passes | Code changed, assumed fixed |\\n| Regression test works | Red-green cycle verified | Test passes once |\\n| Agent completed | VCS diff shows changes | Agent reports \\\"success\\\" |\\n| Requirements met | Line-by-line checklist | Tests passing |\\n\\n## Red Flags - STOP\\n\\n- Using \\\"should\\\", \\\"probably\\\", \\\"seems to\\\"\\n- Expressing satisfaction before verification (\\\"Great!\\\", \\\"Perfect!\\\", \\\"Done!\\\", etc.)\\n- About to commit/push/PR without verification\\n- Trusting agent success reports\\n- Relying on partial verification\\n- Thinking \\\"just this once\\\"\\n- Tired and wanting work over\\n- **ANY wording implying success without having run verification**\\n\\n## Rationalization Prevention\\n\\n| Excuse | Reality |\\n|--------|---------|\\n| \\\"Should work now\\\" | RUN the verification |\\n| \\\"I'm confident\\\" | Confidence ??evidence |\\n| \\\"Just this once\\\" | No exceptions |\\n| \\\"Linter passed\\\" | Linter ??compiler |\\n| \\\"Agent said success\\\" | Verify independently |\\n| \\\"I'm tired\\\" | Exhaustion ??excuse |\\n| \\\"Partial check is enough\\\" | Partial proves nothing |\\n| \\\"Different words so rule doesn't apply\\\" | Spirit over letter |\\n\\n## Key Patterns\\n\\n**Tests:**\\n```\\n??[Run test command] [See: 34/34 pass] \\\"All tests pass\\\"\\n??\\\"Should pass now\\\" / \\\"Looks correct\\\"\\n```\\n\\n**Regression tests (TDD Red-Green):**\\n```\\n??Write ??Run (pass) ??Revert fix ??Run (MUST FAIL) ??Restore ??Run (pass)\\n??\\\"I've written a regression test\\\" (without red-green verification)\\n```\\n\\n**Build:**\\n```\\n??[Run build] [See: exit 0] \\\"Build passes\\\"\\n??\\\"Linter passed\\\" (linter doesn't check compilation)\\n```\\n\\n**Requirements:**\\n```\\n??Re-read plan ??Create checklist ??Verify each ??Report gaps or completion\\n??\\\"Tests pass, phase complete\\\"\\n```\\n\\n**Agent delegation:**\\n```\\n??Agent reports success ??Check VCS diff ??Verify changes ??Report actual state\\n??Trust agent report\\n```\\n\\n## When To Apply\\n\\n**ALWAYS before:**\\n- ANY variation of success/completion claims\\n- ANY expression of satisfaction\\n- ANY positive statement about work state\\n- Committing, PR creation, task completion\\n- Moving to next task\\n- Delegating to agents\\n\\n**Rule applies to:**\\n- Exact phrases\\n- Paraphrases and synonyms\\n- Implications of success\\n- ANY communication suggesting completion/correctness\\n\\r\\n\",\"exit_code\":0,\"status\":\"completed\"}}\n{\"type\":\"item.completed\",\"item\":{\"id\":\"item_4\",\"type\":\"agent_message\",\"text\":\"요구사항이 이미 Step 문서와 인수인계에서 구체적으로 승인된 상태이므로, 별도 설계 문서 작성·커밋을 요구하는 일반 브레인스토밍 절차는 이번 요청의 “추가 파일 금지/직접 커밋 금지”와 충돌합니다. 따라서 기존 Step 2 계약을 승인된 설계로 사용하고, 인라인 계획과 실제 production 경로를 쓰는 TDD로 진행하겠습니다.\"}}\n{\"type\":\"item.started\",\"item\":{\"id\":\"item_5\",\"type\":\"todo_list\",\"items\":[{\"text\":\"저장소 상태, 필수 문서, Step 0–1 및 기존 production API 확인\",\"completed\":false},{\"text\":\"Step 2 최소 설계와 테스트 성공 기준을 기존 계약에 맞춰 확정\",\"completed\":false},{\"text\":\"LinearStaticAnalysis 실패 테스트 작성 후 예상 실패 확인\",\"completed\":false},{\"text\":\"최소 orchestration 구현 및 focused 테스트 통과\",\"completed\":false},{\"text\":\"전체 AC 검증, diff 검토, phase index 갱신\",\"completed\":false}]}}\n{\"type\":\"item.started\",\"item\":{\"id\":\"item_6\",\"type\":\"command_execution\",\"command\":\"\\\"C:\\\\\\\\WINDOWS\\\\\\\\System32\\\\\\\\WindowsPowerShell\\\\\\\\v1.0\\\\\\\\powershell.exe\\\" -Command \\\"Get-Content -Raw -Encoding utf8 -LiteralPath 'C:\\\\\\\\Users\\\\\\\\baram\\\\\\\\.codex\\\\\\\\plugins\\\\\\\\cache\\\\\\\\openai-curated-remote\\\\\\\\superpowers\\\\\\\\6.2.0\\\\\\\\skills\\\\\\\\using-superpowers\\\\\\\\references\\\\\\\\codex-tools.md'\\\"\",\"aggregated_output\":\"\",\"exit_code\":null,\"status\":\"in_progress\"}}\n{\"type\":\"item.completed\",\"item\":{\"id\":\"item_6\",\"type\":\"command_execution\",\"command\":\"\\\"C:\\\\\\\\WINDOWS\\\\\\\\System32\\\\\\\\WindowsPowerShell\\\\\\\\v1.0\\\\\\\\powershell.exe\\\" -Command \\\"Get-Content -Raw -Encoding utf8 -LiteralPath 'C:\\\\\\\\Users\\\\\\\\baram\\\\\\\\.codex\\\\\\\\plugins\\\\\\\\cache\\\\\\\\openai-curated-remote\\\\\\\\superpowers\\\\\\\\6.2.0\\\\\\\\skills\\\\\\\\using-superpowers\\\\\\\\references\\\\\\\\codex-tools.md'\\\"\",\"aggregated_output\":\"## Subagent dispatch requires multi-agent support\\n\\nAdd to your Codex config (`~/.codex/config.toml`):\\n\\n```toml\\n[features]\\nmulti_agent = true\\n```\\n\\nThis enables `spawn_agent`, `wait_agent`, and `close_agent` for skills like `dispatching-parallel-agents` and `subagent-driven-development`. When using subagent-driven-development, close reviewer subagents when their review returns. Keep each implementer subagent open until its task's review passes — the fix loop resumes the implementer — then close it. If your harness cannot send another message to a spawned agent, dispatch each fix round as a fresh implementer carrying the brief, the report file, and the findings.\\n\\n## Environment Detection\\n\\nSkills that create worktrees or finish branches should detect their\\nenvironment with read-only git commands before proceeding:\\n\\n```bash\\nGIT_DIR=$(cd \\\"$(git rev-parse --git-dir)\\\" 2>/dev/null && pwd -P)\\nGIT_COMMON=$(cd \\\"$(git rev-parse --git-common-dir)\\\" 2>/dev/null && pwd -P)\\nBRANCH=$(git branch --show-current)\\n```\\n\\n- `GIT_DIR != GIT_COMMON` → already in a linked worktree (skip creation)\\n- `BRANCH` empty → detached HEAD (cannot branch/push/PR from sandbox)\\n\\nSee `using-git-worktrees` Step 0 and `finishing-a-development-branch`\\nStep 1 for how each skill uses these signals.\\n\\n## Codex App Finishing\\n\\nWhen the sandbox blocks branch/push operations (detached HEAD in an\\nexternally managed worktree), the agent commits all work and informs\\nthe user to use the App's native controls:\\n\\n- **\\\"Create branch\\\"** — names the branch, then commit/push/PR via App UI\\n- **\\\"Hand off to local\\\"** — transfers work to the user's local checkout\\n\\nThe agent can still run tests, stage files, and output suggested branch\\nnames, commit messages, and PR descriptions for the user to copy.\\n\\r\\n\",\"exit_code\":0,\"status\":\"completed\"}}\n{\"type\":\"item.started\",\"item\":{\"id\":\"item_7\",\"type\":\"command_execution\",\"command\":\"\\\"C:\\\\\\\\WINDOWS\\\\\\\\System32\\\\\\\\WindowsPowerShell\\\\\\\\v1.0\\\\\\\\powershell.exe\\\" -Command \\\"Get-Content -Raw -Encoding utf8 -LiteralPath 'C:\\\\\\\\Users\\\\\\\\baram\\\\\\\\.codex\\\\\\\\plugins\\\\\\\\cache\\\\\\\\openai-curated-remote\\\\\\\\superpowers\\\\\\\\6.2.0\\\\\\\\skills\\\\\\\\test-driven-development\\\\\\\\writing-good-tests.md'\\\"\",\"aggregated_output\":\"\",\"exit_code\":null,\"status\":\"in_progress\"}}\n{\"type\":\"item.completed\",\"item\":{\"id\":\"item_7\",\"type\":\"command_execution\",\"command\":\"\\\"C:\\\\\\\\WINDOWS\\\\\\\\System32\\\\\\\\WindowsPowerShell\\\\\\\\v1.0\\\\\\\\powershell.exe\\\" -Command \\\"Get-Content -Raw -Encoding utf8 -LiteralPath 'C:\\\\\\\\Users\\\\\\\\baram\\\\\\\\.codex\\\\\\\\plugins\\\\\\\\cache\\\\\\\\openai-curated-remote\\\\\\\\superpowers\\\\\\\\6.2.0\\\\\\\\skills\\\\\\\\test-driven-development\\\\\\\\writing-good-tests.md'\\\"\",\"aggregated_output\":\"# Writing Good Tests\\n\\n**Load this reference when:** writing or changing tests, adding mocks, or\\nadding cleanup/helper methods for tests.\\n\\n## Overview\\n\\nA test exists to catch a specific break. Two principles govern everything\\nhere:\\n\\n```\\n1. Every test names the break it catches\\n2. Every test exercises the real thing\\n```\\n\\nStrict TDD produces both naturally: a test written first and watched\\nfailing against real code has already proven it can fail, and only earns\\na mock when the real dependency proves slow or external.\\n\\n## Principle 1: Name the Break\\n\\nBefore writing the test body, answer: **what production change should\\nmake this test fail — and is that change a bug or a decision?** A test\\nearns its place by catching a wrong branch, missing side effect, wrong\\nargument, boundary case, or broken contract.\\n\\n**Derive expectations independently.** Use literals and hand-checked\\nfixtures; table-driven tests with literal `want` values are the preferred\\nshape. An expectation computed by the code under test — or its helpers —\\npasses no matter what that code does:\\n\\n```typescript\\n// ❌ Mirror assertion: the same builder computes both sides — always true\\nconst expected = buildSearchQuery({ tag: 'urgent' });\\nexpect(buildSearchQuery({ tag: 'urgent' })).toBe(expected);\\n\\n// ✅ Hand-derived literal\\nexpect(buildSearchQuery({ tag: 'urgent' })).toBe('tag:\\\"urgent\\\"');\\n```\\n\\n**No change detectors.** If only intentional decisions can fail a test —\\na constant's value, exact message wording, private structure — it fires\\non redesign and sleeps through bugs. Test the behavior that depends on\\nthe decision: not `expect(MAX_RETRIES).toBe(5)` but \\\"a failing call is\\nretried 5 times and the 6th attempt never happens.\\\"\\n\\n**Behavior, not text.** Asserting that a script, skill, or config\\ncontains an exact line proves only that the source is the source. Run\\nscripts against controlled inputs and assert outputs, side effects, or\\nexit codes. Documents that instruct agents are tested by the consuming\\nagent's behavior (superpowers:writing-skills); prose for humans earns no\\ntest at all.\\n\\n**Your code, not the framework.** Test the contract your code makes at\\nits boundaries — the route you register, the query you emit, the payload\\nyou produce. Upstream mechanics are their maintainers' tests to write\\n(the classic: asserting your router invokes a registered handler — that\\nis the framework's test, not yours). When upstream behavior genuinely\\nsurprised you, write one narrow characterization test naming the\\nassumption. The same boundary applies inside your code: constructors,\\ngetters, constants, and trivial forwarding earn tests only when they\\nvalidate, normalize, default, derive, enforce, or cause side effects —\\notherwise assert the first consumer-visible result that depends on them.\\n\\n### Gate Function\\n\\n```\\nBEFORE writing the test body:\\n Name the production change that would make this test fail.\\n\\n Cannot name one → redesign around an observable behavior\\n \\\"The source text changed\\\" → run the artifact and assert its effects\\n Only intentional decisions → change detector; test the behavior\\n that depends on the decision\\n\\n Confirm the expected value is derived without the code under test.\\n IF it reuses the code's logic or helpers:\\n Replace it with a literal or hand-checked fixture\\n```\\n\\n## Principle 2: Exercise the Real Thing\\n\\n**The mock earns no assertions.** A mock assertion passes when the mock\\nis present and fails when it is absent — it says nothing about the\\ncomponent. Assert the real component's behavior; if the mock is what you\\nare checking, unmock it or delete the assertion.\\n\\n```typescript\\n// ✅ Real behavior\\nexpect(screen.getByRole('navigation')).toBeInTheDocument();\\n\\n// ❌ Mock existence\\nexpect(screen.getByTestId('sidebar-mock')).toBeInTheDocument();\\n```\\n\\n**your human partner's correction:** \\\"Are we testing the behavior of a\\nmock?\\\"\\n\\n**Mock at the right level.** Learn every side effect of the real method\\nbefore replacing it; mock the slow or external operation and keep what\\nthe test depends on real. When unsure, run the test against the real\\nimplementation first and observe what actually needs to happen.\\n\\n```typescript\\n// ❌ The mock swallows the config write that duplicate detection reads\\nvi.mock('ToolCatalog', () => ({\\n discoverAndCacheTools: vi.fn().mockResolvedValue(undefined)\\n}));\\n\\n// ✅ Mock only the slow server startup; the config write stays real\\nvi.mock('MCPServerManager');\\n```\\n\\n**Make doubles specific.** When arguments, call counts, or ordering are\\npart of the contract, assert them — a fake that accepts anything verifies\\nnothing. Give each branch (success, error, malformed) its own fixture or\\nspy, so the wrong branch cannot satisfy the expectation.\\n\\n**Mirror real data completely.** Mock the complete structure as it exists\\nin reality — all documented fields — not just the ones your test reads.\\nPartial mocks fail silently when downstream code reads an omitted field:\\nthe test passes while integration breaks.\\n\\n**Production classes carry production methods only.** Cleanup that only\\ntests need lives in test utilities, never as a `destroy()` on the\\nproduction class. Ask: is this method called only from tests? Does this\\nclass own this resource's lifecycle? Wrong answers → test utility.\\n\\n**Prefer real components over complex mocks.** When mock setup outgrows\\nthe test logic, mocks miss methods the real components have, or tests\\nbreak when the mock changes, switch to an integration test with real\\ncomponents. **your human partner's question:** \\\"Do we need to be using a\\nmock here?\\\"\\n\\n### Gate Function\\n\\n```\\nBEFORE adding a mock or test helper:\\n List the real method's side effects; keep the ones the test\\n depends on real — mock the slow/external level below them.\\n\\n Mock responses mirror the complete real structure.\\n\\n A method only tests call lives in test utilities, not production.\\n\\n About to assert on the mock itself?\\n Unmock it or delete the assertion.\\n```\\n\\n## Tests Ship With the Implementation\\n\\nThe TDD cycle — failing test, minimal implementation, refactor — is what\\n\\\"complete\\\" means. Ship the tests the behavior needs and only those:\\ntrivial code and human prose earn none, and a test written to satisfy\\nprocess costs maintenance forever.\\n\\n## The Mutation Check\\n\\nBefore finishing, mentally mutate the production code; at least one test\\nshould fail for each realistic mutation:\\n\\n- Wrong constant or argument\\n- Wrong branch handler\\n- Missing state change or side effect\\n- Empty or default return\\n- Missing validation for zero, empty, nil, unauthorized, or malformed input\\n\\nA mutation nothing catches marks the behavior as unprotected — or the\\ntest as tautological.\\n\\n## Quick Reference\\n\\n| When you... | Do |\\n|-------------|-----|\\n| Write any test | Name the break it catches — a bug, not a decision |\\n| Build an expected value | Derive it by hand; never with the code under test |\\n| Test a script or document | Run it / pressure-test its consumer; never grep its text |\\n| Reach for a dependency test | Test your boundary contract, not their documented mechanics |\\n| Want to assert on a mocked element | Test the real component, or unmock it |\\n| Are about to mock a method | Learn its side effects; mock the slow/external level |\\n| Build a mock response | Mirror the real structure completely |\\n| Need cleanup only tests use | Put it in test utilities |\\n| Watch mock setup balloon | Switch to an integration test with real components |\\n| Finish a test file | Run the mutation check |\\n\\n## Warning Signs\\n\\n- Setup and assertion share the same object, guaranteeing equality\\n- The test can fail only through a panic, crash, or missing selector\\n- The test fails on every intentional change, never on accidental breakage\\n- Expected values are hidden behind loops, builders, or helpers\\n- The test greps source text, or asserts a removed symbol stays removed\\n- The test would still matter if only the framework remained\\n- The test exists for coverage, checking no side effect or outcome\\n- An assertion checks a `*-mock` test ID, or fails if you remove the mock\\n- A method is called only from test files\\n- Mock setup is more than half the test, or you can't explain why the mock is needed\\n- Mocking \\\"just to be safe\\\"\\n\\r\\n\",\"exit_code\":0,\"status\":\"completed\"}}\n{\"type\":\"item.started\",\"item\":{\"id\":\"item_8\",\"type\":\"command_execution\",\"command\":\"\\\"C:\\\\\\\\WINDOWS\\\\\\\\System32\\\\\\\\WindowsPowerShell\\\\\\\\v1.0\\\\\\\\powershell.exe\\\" -Command \\\"Get-Content -Raw -Encoding utf8 -LiteralPath 'C:\\\\\\\\git\\\\\\\\FESA\\\\\\\\.agents\\\\\\\\skills\\\\\\\\harness\\\\\\\\SKILL.md'\\\"\",\"aggregated_output\":\"\",\"exit_code\":null,\"status\":\"in_progress\"}}\n{\"type\":\"item.completed\",\"item\":{\"id\":\"item_8\",\"type\":\"command_execution\",\"command\":\"\\\"C:\\\\\\\\WINDOWS\\\\\\\\System32\\\\\\\\WindowsPowerShell\\\\\\\\v1.0\\\\\\\\powershell.exe\\\" -Command \\\"Get-Content -Raw -Encoding utf8 -LiteralPath 'C:\\\\\\\\git\\\\\\\\FESA\\\\\\\\.agents\\\\\\\\skills\\\\\\\\harness\\\\\\\\SKILL.md'\\\"\",\"aggregated_output\":\"---\\r\\nname: harness\\r\\ndescription: Use when planning agentic implementation phases, creating phases/index.json and self-contained step files, or running the Harness step executor.\\r\\n---\\r\\n\\r\\n# Harness Workflow\\r\\n\\r\\n이 프로젝트는 Harness 프레임워크를 사용한다. 아래 워크플로에 따라 작업한다.\\r\\n\\r\\n## A. 탐색\\r\\n\\r\\n`AGENTS.md`와 `docs/` 하위 문서(PRD, ARCHITECTURE, ADR 등)를 읽고 프로젝트의 기획,\\r\\n아키텍처, 설계 의도를 파악한다. 병렬 탐색이 실제로 유용하고 현재 세션에서 허용될\\r\\n때만 Codex subagent를 선택적으로 사용한다.\\r\\n\\r\\n## B. 논의\\r\\n\\r\\n구현을 위해 구체화하거나 기술적으로 결정해야 할 사항이 있으면 사용자에게 한 번에\\r\\n하나씩 제시하고 논의한다.\\r\\n\\r\\n## C. Step 설계\\r\\n\\r\\n사용자가 구현 계획 작성을 지시하면 여러 step으로 나뉜 초안을 작성해 피드백을\\r\\n요청한다.\\r\\n\\r\\n설계 원칙:\\r\\n\\r\\n1. **Scope 최소화** — 하나의 step에서 하나의 레이어 또는 모듈만 다룬다. 여러\\r\\n 모듈을 동시에 수정해야 하면 step을 쪼갠다.\\r\\n2. **자기완결성** — 각 step 파일은 독립된 Codex 실행에서 사용된다. 외부 대화\\r\\n 참조를 금지하고 필요한 정보를 모두 파일 안에 적는다.\\r\\n3. **사전 준비 강제** — 관련 문서와 이전 step에서 생성하거나 수정한 파일 경로를\\r\\n 명시한다.\\r\\n4. **시그니처 수준 지시** — 함수와 클래스의 인터페이스를 제시하고 내부 구현은\\r\\n Codex 재량에 맡긴다. 멱등성, 보안, 데이터 무결성 같은 핵심 규칙은 명시한다.\\r\\n5. **AC는 실행 가능한 command** — 추상적 조건 대신 실제 빌드와 테스트 command를\\r\\n 포함한다.\\r\\n6. **주의사항은 구체적으로** — \\\"X를 하지 마라. 이유: Y\\\" 형식으로 적는다.\\r\\n7. **네이밍** — step name은 핵심 작업을 표현하는 kebab-case slug로 정한다.\\r\\n\\r\\n## D. 파일 생성\\r\\n\\r\\n사용자가 초안을 승인한 후에만 다음 파일을 생성한다.\\r\\n\\r\\n### D-1. `phases/index.json`\\r\\n\\r\\n여러 task를 관리하는 top-level 인덱스다. 이미 존재하면 `phases` 배열에 새 항목을\\r\\n추가한다.\\r\\n\\r\\n```json\\r\\n{\\r\\n \\\"phases\\\": [\\r\\n {\\r\\n \\\"dir\\\": \\\"0-mvp\\\",\\r\\n \\\"status\\\": \\\"pending\\\"\\r\\n }\\r\\n ]\\r\\n}\\r\\n```\\r\\n\\r\\n- `dir`: task 디렉터리명\\r\\n- `status`: `pending` | `completed` | `error` | `blocked`\\r\\n- timestamp는 executor가 상태를 바꿀 때 기록하므로 생성 시 넣지 않는다.\\r\\n\\r\\n### D-2. `phases/{task-name}/index.json`\\r\\n\\r\\n```json\\r\\n{\\r\\n \\\"project\\\": \\\"<프로젝트명>\\\",\\r\\n \\\"phase\\\": \\\"\\\",\\r\\n \\\"steps\\\": [\\r\\n { \\\"step\\\": 0, \\\"name\\\": \\\"project-setup\\\", \\\"status\\\": \\\"pending\\\" },\\r\\n { \\\"step\\\": 1, \\\"name\\\": \\\"core-types\\\", \\\"status\\\": \\\"pending\\\" },\\r\\n { \\\"step\\\": 2, \\\"name\\\": \\\"api-layer\\\", \\\"status\\\": \\\"pending\\\" }\\r\\n ]\\r\\n}\\r\\n```\\r\\n\\r\\n필드 규칙:\\r\\n\\r\\n- `project`: `AGENTS.md`에 정의된 프로젝트명\\r\\n- `phase`: task 이름이며 디렉터리명과 일치\\r\\n- `steps[].step`: 0부터 시작하는 순번\\r\\n- `steps[].name`: kebab-case slug\\r\\n- `steps[].status`: 초기값 `pending`\\r\\n\\r\\n상태와 기록 주체:\\r\\n\\r\\n| 전이 | 기록 필드 | 기록 주체 |\\r\\n|------|-----------|-----------|\\r\\n| `completed` | `summary`, `completed_at` | Codex가 summary, executor가 timestamp |\\r\\n| `error` | `error_message`, `failed_at` | Codex가 message, executor가 timestamp |\\r\\n| `blocked` | `blocked_reason`, `blocked_at` | Codex가 reason, executor가 timestamp |\\r\\n\\r\\n`summary`에는 다음 step에 유용한 생성 파일과 핵심 결정을 한 줄로 적는다.\\r\\ntask `created_at`과 step `started_at`은 executor가 기록하므로 생성 시 넣지 않는다.\\r\\n\\r\\n### D-3. `phases/{task-name}/step{N}.md`\\r\\n\\r\\n````markdown\\r\\n# Step {N}: {이름}\\r\\n\\r\\n## 읽어야 할 파일\\r\\n\\r\\n먼저 아래 파일을 읽고 프로젝트의 아키텍처와 설계 의도를 파악하라:\\r\\n\\r\\n- `/AGENTS.md`\\r\\n- `/docs/ARCHITECTURE.md`\\r\\n- `/docs/ADR.md`\\r\\n- 이전 step에서 생성하거나 수정한 파일 경로\\r\\n\\r\\n이전 step의 코드를 꼼꼼히 읽고 설계 의도를 이해한 뒤 작업하라.\\r\\n\\r\\n## 작업\\r\\n\\r\\n구체적인 구현 지시를 파일 경로, 클래스와 함수 시그니처, 로직 설명과 함께 적는다.\\r\\n구현체는 Codex에 맡기되 설계 의도에서 벗어나면 안 되는 핵심 규칙은 명시한다.\\r\\n\\r\\n## Acceptance Criteria\\r\\n\\r\\n프로젝트 형식에 맞는 명령을 사용한다. `.harness/config.json`이 있으면 해당 preset,\\r\\nsolution, configuration, platform, test command를 우선한다.\\r\\n\\r\\n```powershell\\r\\n# CMake\\r\\ncmake --build .harness/build --config Debug\\r\\nctest --test-dir .harness/build -C Debug --output-on-failure\\r\\n\\r\\n# 직접 MSBuild\\r\\nMSBuild.exe MyProject.sln /m /p:Configuration=Debug /p:Platform=x64\\r\\n.\\\\build\\\\tests\\\\Debug\\\\MyProjectTests.exe\\r\\n```\\r\\n\\r\\n## 검증 절차\\r\\n\\r\\n1. Acceptance Criteria command를 실행한다.\\r\\n2. ARCHITECTURE 디렉터리 구조를 따르는지 확인한다.\\r\\n3. ADR 기술 스택과 `AGENTS.md` CRITICAL 규칙을 확인한다.\\r\\n4. 결과에 따라 task index의 해당 step을 갱신한다.\\r\\n - 성공: `status`를 `completed`로 바꾸고 한 줄 `summary` 기록\\r\\n - 수정 3회 후 실패: `status`를 `error`로 바꾸고 `error_message` 기록\\r\\n - 사용자 개입 필요: `status`를 `blocked`로 바꾸고 `blocked_reason` 기록 후 중단\\r\\n\\r\\n## 금지사항\\r\\n\\r\\n- 이 step의 범위 밖 기능을 추가하지 마라. 이유: step의 독립성을 깨뜨린다.\\r\\n- 기존 테스트를 깨뜨리지 마라. 이유: 이전 동작을 회귀시킨다.\\r\\n````\\r\\n\\r\\n## E. 실행\\r\\n\\r\\n```bash\\r\\npython scripts/execute.py {task-name}\\r\\npython scripts/execute.py {task-name} --push\\r\\n```\\r\\n\\r\\n환경에서 Python 3 실행 명령이 `python3`이면 그 명령을 대신 사용한다.\\r\\n\\r\\nexecutor가 처리하는 작업:\\r\\n\\r\\n- `feat-{task-name}` 브랜치 생성 또는 checkout\\r\\n- `AGENTS.md`와 `docs/*.md` guardrail 주입\\r\\n- 완료 step의 summary를 다음 prompt에 누적\\r\\n- 실패 시 최대 3회 재시도하며 이전 오류를 prompt에 전달\\r\\n- 코드 변경과 metadata를 분리해 commit\\r\\n- `started_at`, `completed_at`, `failed_at`, `blocked_at` 기록\\r\\n\\r\\n에러 복구:\\r\\n\\r\\n- `error`: 해당 status를 `pending`으로 바꾸고 `error_message`를 삭제한 뒤 재실행\\r\\n- `blocked`: 원인을 해결하고 status를 `pending`으로 바꾸고 `blocked_reason`을 삭제한\\r\\n 뒤 재실행\\r\\n\\r\\n\",\"exit_code\":0,\"status\":\"completed\"}}\n{\"type\":\"item.started\",\"item\":{\"id\":\"item_9\",\"type\":\"command_execution\",\"command\":\"\\\"C:\\\\\\\\WINDOWS\\\\\\\\System32\\\\\\\\WindowsPowerShell\\\\\\\\v1.0\\\\\\\\powershell.exe\\\" -Command '$branch = git branch --show-current\\n$status = git status --short --branch\\n$head = git rev-parse HEAD\\n$common = git rev-parse --git-common-dir\\n$gitdir = git rev-parse --git-dir\\n[pscustomobject]@{Branch=$branch; Head=$head; GitDir=$gitdir; GitCommonDir=$common; Status=($status -join \\\"`n\\\")} | Format-List\\nrg --files include/fesa/assembly include/fesa/constraints include/fesa/solvers/linear include/fesa/results src/fesa tests/unit phases/results-and-pipeline | Sort-Object'\",\"aggregated_output\":\"\",\"exit_code\":null,\"status\":\"in_progress\"}}\n{\"type\":\"item.completed\",\"item\":{\"id\":\"item_9\",\"type\":\"command_execution\",\"command\":\"\\\"C:\\\\\\\\WINDOWS\\\\\\\\System32\\\\\\\\WindowsPowerShell\\\\\\\\v1.0\\\\\\\\powershell.exe\\\" -Command '$branch = git branch --show-current\\n$status = git status --short --branch\\n$head = git rev-parse HEAD\\n$common = git rev-parse --git-common-dir\\n$gitdir = git rev-parse --git-dir\\n[pscustomobject]@{Branch=$branch; Head=$head; GitDir=$gitdir; GitCommonDir=$common; Status=($status -join \\\"`n\\\")} | Format-List\\nrg --files include/fesa/assembly include/fesa/constraints include/fesa/solvers/linear include/fesa/results src/fesa tests/unit phases/results-and-pipeline | Sort-Object'\",\"aggregated_output\":\"\\r\\n\\r\\nBranch : feat-results-and-pipeline\\r\\nHead : d6306383a9205c000757ef62c5d60431ea878a63\\r\\nGitDir : .git\\r\\nGitCommonDir : .git\\r\\nStatus : ## feat-results-and-pipeline\\r\\n M phases/results-and-pipeline/index.json\\r\\n\\r\\n\\r\\n\\r\\ninclude/fesa/assembly\\\\equation_system.hpp\\r\\ninclude/fesa/assembly\\\\serial_assembler.hpp\\r\\ninclude/fesa/assembly\\\\symmetric_csr.hpp\\r\\ninclude/fesa/constraints\\\\essential_bc.hpp\\r\\ninclude/fesa/results\\\\result_database.hpp\\r\\ninclude/fesa/solvers/linear\\\\linear_solver.hpp\\r\\ninclude/fesa/solvers/linear\\\\pardiso_linear_solver.hpp\\r\\nphases/results-and-pipeline\\\\index.json\\r\\nphases/results-and-pipeline\\\\step0.md\\r\\nphases/results-and-pipeline\\\\step0-output.json\\r\\nphases/results-and-pipeline\\\\step1.md\\r\\nphases/results-and-pipeline\\\\step1-output.json\\r\\nphases/results-and-pipeline\\\\step2.md\\r\\nphases/results-and-pipeline\\\\step3.md\\r\\nsrc/fesa\\\\assembly\\\\serial_assembler.cpp\\r\\nsrc/fesa\\\\cli\\\\main.cpp\\r\\nsrc/fesa\\\\constraints\\\\essential_bc.cpp\\r\\nsrc/fesa\\\\core\\\\version.cpp\\r\\nsrc/fesa\\\\elements\\\\beam\\\\beam3d2.cpp\\r\\nsrc/fesa\\\\fem\\\\beam_frame.cpp\\r\\nsrc/fesa\\\\fem\\\\dof_manager.cpp\\r\\nsrc/fesa\\\\fem\\\\gauss_rule.cpp\\r\\nsrc/fesa\\\\fem\\\\line2_shape.cpp\\r\\nsrc/fesa\\\\io\\\\abaqus\\\\parser.cpp\\r\\nsrc/fesa\\\\io\\\\abaqus\\\\semantic_mapper.cpp\\r\\nsrc/fesa\\\\io\\\\hdf5\\\\writer.cpp\\r\\nsrc/fesa\\\\model\\\\domain.cpp\\r\\nsrc/fesa\\\\model\\\\domain_builder.cpp\\r\\nsrc/fesa\\\\results\\\\result_database.cpp\\r\\nsrc/fesa\\\\solvers\\\\linear\\\\pardiso_linear_solver.cpp\\r\\ntests/unit\\\\assembly\\\\serial_assembler_test.cpp\\r\\ntests/unit\\\\constraints\\\\essential_bc_test.cpp\\r\\ntests/unit\\\\core\\\\diagnostic_test.cpp\\r\\ntests/unit\\\\core\\\\entity_id_test.cpp\\r\\ntests/unit\\\\core\\\\status_test.cpp\\r\\ntests/unit\\\\core\\\\vec3_test.cpp\\r\\ntests/unit\\\\core\\\\version_test.cpp\\r\\ntests/unit\\\\dependencies\\\\dependency_smoke_test.cpp\\r\\ntests/unit\\\\elements\\\\beam3d2_test.cpp\\r\\ntests/unit\\\\fem\\\\beam_frame_test.cpp\\r\\ntests/unit\\\\fem\\\\dof_manager_test.cpp\\r\\ntests/unit\\\\fem\\\\gauss_rule_test.cpp\\r\\ntests/unit\\\\fem\\\\line2_shape_test.cpp\\r\\ntests/unit\\\\io\\\\abaqus\\\\parser_test.cpp\\r\\ntests/unit\\\\model\\\\domain_builder_test.cpp\\r\\ntests/unit\\\\model\\\\entity_origin_test.cpp\\r\\ntests/unit\\\\model\\\\model_types_test.cpp\\r\\ntests/unit\\\\results\\\\result_database_test.cpp\\r\\ntests/unit\\\\solvers\\\\linear\\\\pardiso_linear_solver_test.cpp\\r\\nwarning: could not open directory '.harness/fallback-pytest-basetemp-82aadc57b8df4a27bf47b31b7c9393d2/': Permission denied\\nwarning: could not open directory '.harness/pytest-basetemp-ac0985ddc9224918b6d0476471cc5f39/': Permission denied\\nwarning: could not open directory '.harness/pytest-basetemp-step0b/': Permission denied\\nwarning: could not open directory '.harness/pytest-cache-files-sqpzrqiw/': Permission denied\\nwarning: could not open directory '.harness/pytest-cache-files-vf1p08il/': Permission denied\\nwarning: could not open directory '.harness/pytest-cache-files-y84afu_7/': Permission denied\\nwarning: could not open directory '.harness/pytest-cache-files-zjj4_p73/': Permission denied\\nwarning: could not open directory 'gitFESA.harnessbuildpytest-basetemp/': Permission denied\\nwarning: could not open directory 'gitFESA.harnesspytest-basetemp-step0/': Permission denied\\nwarning: could not open directory 'gitFESAoutbuildpytest-basetemp-step1/': Permission denied\\nwarning: could not open directory 'pytest-cache-files-13i94o94/': Permission denied\\nwarning: could not open directory 'pytest-cache-files-pnbzv9xt/': Permission denied\\nwarning: could not open directory 'pytest-cache-files-vtfbrhqu/': Permission denied\\nwarning: could not open directory 'pytest-cache-files-wy6fhgsc/': Permission denied\\n\",\"exit_code\":0,\"status\":\"completed\"}}\n{\"type\":\"item.started\",\"item\":{\"id\":\"item_10\",\"type\":\"command_execution\",\"command\":\"\\\"C:\\\\\\\\WINDOWS\\\\\\\\System32\\\\\\\\WindowsPowerShell\\\\\\\\v1.0\\\\\\\\powershell.exe\\\" -Command \\\"Get-Content -Raw -Encoding utf8 -LiteralPath 'include/fesa/fem/dof_manager.hpp'\\\"\",\"aggregated_output\":\"\",\"exit_code\":null,\"status\":\"in_progress\"}}\n{\"type\":\"item.started\",\"item\":{\"id\":\"item_11\",\"type\":\"command_execution\",\"command\":\"\\\"C:\\\\\\\\WINDOWS\\\\\\\\System32\\\\\\\\WindowsPowerShell\\\\\\\\v1.0\\\\\\\\powershell.exe\\\" -Command \\\"Get-Content -Raw -Encoding utf8 -LiteralPath 'docs/ADR.md'\\\"\",\"aggregated_output\":\"\",\"exit_code\":null,\"status\":\"in_progress\"}}\n{\"type\":\"item.started\",\"item\":{\"id\":\"item_12\",\"type\":\"command_execution\",\"command\":\"\\\"C:\\\\\\\\WINDOWS\\\\\\\\System32\\\\\\\\WindowsPowerShell\\\\\\\\v1.0\\\\\\\\powershell.exe\\\" -Command \\\"Get-Content -Raw -Encoding utf8 -LiteralPath 'AGENTS.md'\\\"\",\"aggregated_output\":\"\",\"exit_code\":null,\"status\":\"in_progress\"}}\n{\"type\":\"item.started\",\"item\":{\"id\":\"item_13\",\"type\":\"command_execution\",\"command\":\"\\\"C:\\\\\\\\WINDOWS\\\\\\\\System32\\\\\\\\WindowsPowerShell\\\\\\\\v1.0\\\\\\\\powershell.exe\\\" -Command \\\"Get-Content -Raw -Encoding utf8 -LiteralPath 'phases/results-and-pipeline/index.json'\\\"\",\"aggregated_output\":\"\",\"exit_code\":null,\"status\":\"in_progress\"}}\n{\"type\":\"item.started\",\"item\":{\"id\":\"item_14\",\"type\":\"command_execution\",\"command\":\"\\\"C:\\\\\\\\WINDOWS\\\\\\\\System32\\\\\\\\WindowsPowerShell\\\\\\\\v1.0\\\\\\\\powershell.exe\\\" -Command \\\"Get-Content -Raw -Encoding utf8 -LiteralPath 'docs/PRD.md'\\\"\",\"aggregated_output\":\"\",\"exit_code\":null,\"status\":\"in_progress\"}}\n{\"type\":\"item.started\",\"item\":{\"id\":\"item_15\",\"type\":\"command_execution\",\"command\":\"\\\"C:\\\\\\\\WINDOWS\\\\\\\\System32\\\\\\\\WindowsPowerShell\\\\\\\\v1.0\\\\\\\\powershell.exe\\\" -Command \\\"Get-Content -Raw -Encoding utf8 -LiteralPath 'phases/results-and-pipeline/step2.md'\\\"\",\"aggregated_output\":\"\",\"exit_code\":null,\"status\":\"in_progress\"}}\n{\"type\":\"item.started\",\"item\":{\"id\":\"item_16\",\"type\":\"command_execution\",\"command\":\"\\\"C:\\\\\\\\WINDOWS\\\\\\\\System32\\\\\\\\WindowsPowerShell\\\\\\\\v1.0\\\\\\\\powershell.exe\\\" -Command \\\"Get-Content -Raw -Encoding utf8 -LiteralPath 'include/fesa/results/result_database.hpp'\\\"\",\"aggregated_output\":\"\",\"exit_code\":null,\"status\":\"in_progress\"}}\n{\"type\":\"item.started\",\"item\":{\"id\":\"item_17\",\"type\":\"command_execution\",\"command\":\"\\\"C:\\\\\\\\WINDOWS\\\\\\\\System32\\\\\\\\WindowsPowerShell\\\\\\\\v1.0\\\\\\\\powershell.exe\\\" -Command \\\"Get-Content -Raw -Encoding utf8 -LiteralPath 'docs/ARCHITECTURE.md'\\\"\",\"aggregated_output\":\"\",\"exit_code\":null,\"status\":\"in_progress\"}}\n{\"type\":\"item.started\",\"item\":{\"id\":\"item_18\",\"type\":\"command_execution\",\"command\":\"\\\"C:\\\\\\\\WINDOWS\\\\\\\\System32\\\\\\\\WindowsPowerShell\\\\\\\\v1.0\\\\\\\\powershell.exe\\\" -Command \\\"Get-Content -Raw -Encoding utf8 -LiteralPath 'include/fesa/solvers/linear/pardiso_linear_solver.hpp'\\\"\",\"aggregated_output\":\"\",\"exit_code\":null,\"status\":\"in_progress\"}}\n{\"type\":\"item.started\",\"item\":{\"id\":\"item_19\",\"type\":\"command_execution\",\"command\":\"\\\"C:\\\\\\\\WINDOWS\\\\\\\\System32\\\\\\\\WindowsPowerShell\\\\\\\\v1.0\\\\\\\\powershell.exe\\\" -Command \\\"Get-Content -Raw -Encoding utf8 -LiteralPath 'include/fesa/assembly/serial_assembler.hpp'\\\"\",\"aggregated_output\":\"\",\"exit_code\":null,\"status\":\"in_progress\"}}\n{\"type\":\"item.started\",\"item\":{\"id\":\"item_20\",\"type\":\"command_execution\",\"command\":\"\\\"C:\\\\\\\\WINDOWS\\\\\\\\System32\\\\\\\\WindowsPowerShell\\\\\\\\v1.0\\\\\\\\powershell.exe\\\" -Command \\\"Get-Content -Raw -Encoding utf8 -LiteralPath 'include/fesa/assembly/equation_system.hpp'\\\"\",\"aggregated_output\":\"\",\"exit_code\":null,\"status\":\"in_progress\"}}\n{\"type\":\"item.started\",\"item\":{\"id\":\"item_21\",\"type\":\"command_execution\",\"command\":\"\\\"C:\\\\\\\\WINDOWS\\\\\\\\System32\\\\\\\\WindowsPowerShell\\\\\\\\v1.0\\\\\\\\powershell.exe\\\" -Command \\\"Get-Content -Raw -Encoding utf8 -LiteralPath 'include/fesa/solvers/linear/linear_solver.hpp'\\\"\",\"aggregated_output\":\"\",\"exit_code\":null,\"status\":\"in_progress\"}}\n{\"type\":\"item.started\",\"item\":{\"id\":\"item_22\",\"type\":\"command_execution\",\"command\":\"\\\"C:\\\\\\\\WINDOWS\\\\\\\\System32\\\\\\\\WindowsPowerShell\\\\\\\\v1.0\\\\\\\\powershell.exe\\\" -Command \\\"Get-Content -Raw -Encoding utf8 -LiteralPath 'include/fesa/model/domain.hpp'\\\"\",\"aggregated_output\":\"\",\"exit_code\":null,\"status\":\"in_progress\"}}\n{\"type\":\"item.started\",\"item\":{\"id\":\"item_23\",\"type\":\"command_execution\",\"command\":\"\\\"C:\\\\\\\\WINDOWS\\\\\\\\System32\\\\\\\\WindowsPowerShell\\\\\\\\v1.0\\\\\\\\powershell.exe\\\" -Command \\\"Get-Content -Raw -Encoding utf8 -LiteralPath 'include/fesa/assembly/symmetric_csr.hpp'\\\"\",\"aggregated_output\":\"\",\"exit_code\":null,\"status\":\"in_progress\"}}\n{\"type\":\"item.started\",\"item\":{\"id\":\"item_24\",\"type\":\"command_execution\",\"command\":\"\\\"C:\\\\\\\\WINDOWS\\\\\\\\System32\\\\\\\\WindowsPowerShell\\\\\\\\v1.0\\\\\\\\powershell.exe\\\" -Command \\\"Get-Content -Raw -Encoding utf8 -LiteralPath 'include/fesa/constraints/essential_bc.hpp'\\\"\",\"aggregated_output\":\"\",\"exit_code\":null,\"status\":\"in_progress\"}}\n{\"type\":\"item.completed\",\"item\":{\"id\":\"item_10\",\"type\":\"command_execution\",\"command\":\"\\\"C:\\\\\\\\WINDOWS\\\\\\\\System32\\\\\\\\WindowsPowerShell\\\\\\\\v1.0\\\\\\\\powershell.exe\\\" -Command \\\"Get-Content -Raw -Encoding utf8 -LiteralPath 'include/fesa/fem/dof_manager.hpp'\\\"\",\"aggregated_output\":\"#pragma once\\r\\n\\r\\n#include \\r\\n#include \\r\\n#include \\r\\n#include