mirror of
https://github.com/obra/superpowers.git
synced 2026-07-23 18:54:07 +08:00
Compare commits
6 Commits
agentic-en
...
codex-spin
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
c686bb947a | ||
|
|
4ed49b6a41 | ||
|
|
cdcadda4be | ||
|
|
c97988d8d1 | ||
|
|
d123bde46e | ||
|
|
3921dc9998 |
@@ -59,7 +59,7 @@ digraph process {
|
|||||||
"Finding conflicts with plan text?" [shape=diamond];
|
"Finding conflicts with plan text?" [shape=diamond];
|
||||||
"Ask human partner which governs" [shape=box];
|
"Ask human partner which governs" [shape=box];
|
||||||
"Fix round R of 5: R≤3 resume implementer; R≥4 fresh implementer, more capable model" [shape=box];
|
"Fix round R of 5: R≤3 resume implementer; R≥4 fresh implementer, more capable model" [shape=box];
|
||||||
"Dispatch scoped re-review (./re-review-prompt.md)" [shape=box];
|
"Dispatch scoped fix review (./fix-review-prompt.md)" [shape=box];
|
||||||
"All findings addressed?" [shape=diamond];
|
"All findings addressed?" [shape=diamond];
|
||||||
"R = 5?" [shape=diamond];
|
"R = 5?" [shape=diamond];
|
||||||
"Adjudicate each open finding" [shape=box];
|
"Adjudicate each open finding" [shape=box];
|
||||||
@@ -72,7 +72,7 @@ digraph process {
|
|||||||
"Setup: worktree, ledger check, read plan, pre-flight review" [shape=box];
|
"Setup: worktree, ledger check, read plan, pre-flight review" [shape=box];
|
||||||
"More tasks remain?" [shape=diamond];
|
"More tasks remain?" [shape=diamond];
|
||||||
"Dispatch final code reviewer (../requesting-code-review/code-reviewer.md)" [shape=box];
|
"Dispatch final code reviewer (../requesting-code-review/code-reviewer.md)" [shape=box];
|
||||||
"Final findings? ONE fix dispatch, one scoped re-review, adjudicate residuals" [shape=box];
|
"Final findings? ONE fix dispatch, one scoped fix review, adjudicate residuals" [shape=box];
|
||||||
"Final review clean: delete this plan's workspace" [shape=box];
|
"Final review clean: delete this plan's workspace" [shape=box];
|
||||||
"Use superpowers:finishing-a-development-branch" [shape=box style=filled fillcolor=lightgreen];
|
"Use superpowers:finishing-a-development-branch" [shape=box style=filled fillcolor=lightgreen];
|
||||||
|
|
||||||
@@ -88,8 +88,8 @@ digraph process {
|
|||||||
"Finding conflicts with plan text?" -> "Ask human partner which governs" [label="yes"];
|
"Finding conflicts with plan text?" -> "Ask human partner which governs" [label="yes"];
|
||||||
"Ask human partner which governs" -> "Fix round R of 5: R≤3 resume implementer; R≥4 fresh implementer, more capable model";
|
"Ask human partner which governs" -> "Fix round R of 5: R≤3 resume implementer; R≥4 fresh implementer, more capable model";
|
||||||
"Finding conflicts with plan text?" -> "Fix round R of 5: R≤3 resume implementer; R≥4 fresh implementer, more capable model" [label="no"];
|
"Finding conflicts with plan text?" -> "Fix round R of 5: R≤3 resume implementer; R≥4 fresh implementer, more capable model" [label="no"];
|
||||||
"Fix round R of 5: R≤3 resume implementer; R≥4 fresh implementer, more capable model" -> "Dispatch scoped re-review (./re-review-prompt.md)";
|
"Fix round R of 5: R≤3 resume implementer; R≥4 fresh implementer, more capable model" -> "Dispatch scoped fix review (./fix-review-prompt.md)";
|
||||||
"Dispatch scoped re-review (./re-review-prompt.md)" -> "All findings addressed?";
|
"Dispatch scoped fix review (./fix-review-prompt.md)" -> "All findings addressed?";
|
||||||
"All findings addressed?" -> "Append completion to ledger, mark todo complete" [label="yes"];
|
"All findings addressed?" -> "Append completion to ledger, mark todo complete" [label="yes"];
|
||||||
"All findings addressed?" -> "R = 5?" [label="no"];
|
"All findings addressed?" -> "R = 5?" [label="no"];
|
||||||
"R = 5?" -> "Fix round R of 5: R≤3 resume implementer; R≥4 fresh implementer, more capable model" [label="no - next round"];
|
"R = 5?" -> "Fix round R of 5: R≤3 resume implementer; R≥4 fresh implementer, more capable model" [label="no - next round"];
|
||||||
@@ -101,8 +101,8 @@ digraph process {
|
|||||||
"Append completion to ledger, mark todo complete" -> "More tasks remain?";
|
"Append completion to ledger, mark todo complete" -> "More tasks remain?";
|
||||||
"More tasks remain?" -> "Dispatch implementer subagent (./implementer-prompt.md)" [label="yes"];
|
"More tasks remain?" -> "Dispatch implementer subagent (./implementer-prompt.md)" [label="yes"];
|
||||||
"More tasks remain?" -> "Dispatch final code reviewer (../requesting-code-review/code-reviewer.md)" [label="no"];
|
"More tasks remain?" -> "Dispatch final code reviewer (../requesting-code-review/code-reviewer.md)" [label="no"];
|
||||||
"Dispatch final code reviewer (../requesting-code-review/code-reviewer.md)" -> "Final findings? ONE fix dispatch, one scoped re-review, adjudicate residuals";
|
"Dispatch final code reviewer (../requesting-code-review/code-reviewer.md)" -> "Final findings? ONE fix dispatch, one scoped fix review, adjudicate residuals";
|
||||||
"Final findings? ONE fix dispatch, one scoped re-review, adjudicate residuals" -> "Final review clean: delete this plan's workspace";
|
"Final findings? ONE fix dispatch, one scoped fix review, adjudicate residuals" -> "Final review clean: delete this plan's workspace";
|
||||||
"Final review clean: delete this plan's workspace" -> "Use superpowers:finishing-a-development-branch";
|
"Final review clean: delete this plan's workspace" -> "Use superpowers:finishing-a-development-branch";
|
||||||
}
|
}
|
||||||
```
|
```
|
||||||
@@ -158,6 +158,12 @@ conflicts that only emerge from implementation.
|
|||||||
|
|
||||||
Use the least powerful model that can handle each role to conserve cost and increase speed.
|
Use the least powerful model that can handle each role to conserve cost and increase speed.
|
||||||
|
|
||||||
|
When your platform's reference file (using-superpowers → Platform
|
||||||
|
Adaptation) defines a dispatch role table, that table IS this section's
|
||||||
|
mapping for your harness. Follow it over the tier language below — including
|
||||||
|
for the final review and fix-loop escalation — and follow the
|
||||||
|
`dispatch:` hint lines the task-brief and review-package scripts print.
|
||||||
|
|
||||||
**Mechanical implementation tasks** (isolated functions, clear specs, 1-2 files): use a fast, cheap model. Most implementation tasks are mechanical when the plan is well-specified.
|
**Mechanical implementation tasks** (isolated functions, clear specs, 1-2 files): use a fast, cheap model. Most implementation tasks are mechanical when the plan is well-specified.
|
||||||
|
|
||||||
**Integration and judgment tasks** (multi-file coordination, pattern matching, debugging): use a standard model.
|
**Integration and judgment tasks** (multi-file coordination, pattern matching, debugging): use a standard model.
|
||||||
@@ -168,7 +174,7 @@ capable available model, not the session default.
|
|||||||
|
|
||||||
**Review tasks**: choose the model with the same judgment, scaled to the
|
**Review tasks**: choose the model with the same judgment, scaled to the
|
||||||
diff's size, complexity, and risk. A small mechanical diff does not need the
|
diff's size, complexity, and risk. A small mechanical diff does not need the
|
||||||
most capable model; a subtle concurrency change does. Scoped re-reviews of
|
most capable model; a subtle concurrency change does. Scoped fix reviews of
|
||||||
small fix diffs take a cheap-to-mid tier.
|
small fix diffs take a cheap-to-mid tier.
|
||||||
|
|
||||||
**Fix-loop escalation (rounds 4-5)**: use a model at least one tier above
|
**Fix-loop escalation (rounds 4-5)**: use a model at least one tier above
|
||||||
@@ -317,7 +323,8 @@ Before the loop starts, two routes leave it immediately:
|
|||||||
Do not dismiss the finding because the plan mandates it, and do not
|
Do not dismiss the finding because the plan mandates it, and do not
|
||||||
dispatch a fix that contradicts the plan without asking.
|
dispatch a fix that contradicts the plan without asking.
|
||||||
Everything else enters the loop. A fix round is one fix dispatch plus one
|
Everything else enters the loop. A fix round is one fix dispatch plus one
|
||||||
scoped re-review. Five rounds maximum per task:
|
scoped fix review — a review of the fix diff, not a fresh review. Five
|
||||||
|
rounds maximum per task:
|
||||||
|
|
||||||
**Rounds 1-3 — resume the original implementer.** Send it the open findings
|
**Rounds 1-3 — resume the original implementer.** Send it the open findings
|
||||||
verbatim. Its context is intact: it knows the task, the code, and its own
|
verbatim. Its context is intact: it knows the task, the code, and its own
|
||||||
@@ -336,14 +343,14 @@ own problem — fresh eyes and a capability bump in one move.
|
|||||||
covering the amended code, appends its fix report to the same report file,
|
covering the amended code, appends its fix report to the same report file,
|
||||||
and returns the short contract. Before re-dispatching the reviewer, confirm
|
and returns the short contract. Before re-dispatching the reviewer, confirm
|
||||||
the fix report contains the covering tests, the command run, and the
|
the fix report contains the covering tests, the command run, and the
|
||||||
output; dispatch the re-review once all three are present. Name the
|
output; dispatch the fix review once all three are present. Name the
|
||||||
covering test files in the fix message — a one-line fix does not need the
|
covering test files in the fix message — a one-line fix does not need the
|
||||||
whole suite.
|
whole suite.
|
||||||
|
|
||||||
**The re-review is scoped.** Run `scripts/review-package PLAN_FILE FIX_BASE HEAD`
|
**The fix review is scoped.** Run `scripts/review-package --role fix-review PLAN_FILE FIX_BASE HEAD`
|
||||||
where FIX_BASE is the head the previous review saw, and dispatch
|
where FIX_BASE is the head the previous review saw, and dispatch
|
||||||
[re-review-prompt.md](re-review-prompt.md) with the findings list, the
|
[fix-review-prompt.md](fix-review-prompt.md) with the findings list, the
|
||||||
brief, the report file, and the printed diff path. The re-reviewer verdicts
|
brief, the report file, and the printed diff path. The fix reviewer verdicts
|
||||||
each finding ADDRESSED or NOT ADDRESSED and flags new breakage in the fix
|
each finding ADDRESSED or NOT ADDRESSED and flags new breakage in the fix
|
||||||
diff only. New Critical/Important breakage in the fix diff joins the open
|
diff only. New Critical/Important breakage in the fix diff joins the open
|
||||||
findings list. Out-of-scope observations go to the ledger as deferred
|
findings list. Out-of-scope observations go to the ledger as deferred
|
||||||
@@ -355,7 +362,7 @@ minors — they never extend the loop.
|
|||||||
Never fix findings yourself in the controller session — your context stays
|
Never fix findings yourself in the controller session — your context stays
|
||||||
clean for coordination, and controller fixes skip review.
|
clean for coordination, and controller fixes skip review.
|
||||||
|
|
||||||
**The breaker.** When round 5's re-review still leaves findings open, stop
|
**The breaker.** When round 5's fix review still leaves findings open, stop
|
||||||
dispatching. Adjudicate each open finding yourself — you hold the plan and
|
dispatching. Adjudicate each open finding yourself — you hold the plan and
|
||||||
the cross-task context the reviewer lacks:
|
the cross-task context the reviewer lacks:
|
||||||
|
|
||||||
@@ -391,7 +398,7 @@ parked-with-ruling at the cap.
|
|||||||
## Final Review
|
## Final Review
|
||||||
|
|
||||||
The final whole-branch review gets a package too: run
|
The final whole-branch review gets a package too: run
|
||||||
`scripts/review-package PLAN_FILE MERGE_BASE HEAD` (MERGE_BASE = the commit the
|
`scripts/review-package --role final-review PLAN_FILE MERGE_BASE HEAD` (MERGE_BASE = the commit the
|
||||||
branch started from, e.g. `git merge-base main HEAD`) and include the
|
branch started from, e.g. `git merge-base main HEAD`) and include the
|
||||||
printed path in the final review dispatch, so the final reviewer reads
|
printed path in the final review dispatch, so the final reviewer reads
|
||||||
one file instead of re-deriving the branch diff with git commands. Dispatch
|
one file instead of re-deriving the branch diff with git commands. Dispatch
|
||||||
@@ -405,14 +412,23 @@ If the final whole-branch review returns findings, dispatch ONE fix subagent
|
|||||||
with the complete findings list — not one fixer per finding.
|
with the complete findings list — not one fixer per finding.
|
||||||
Per-finding fixers each rebuild context and re-run suites; a real
|
Per-finding fixers each rebuild context and re-run suites; a real
|
||||||
session's final-review fix wave cost more than all its tasks combined.
|
session's final-review fix wave cost more than all its tasks combined.
|
||||||
Then run exactly one scoped re-review of the fix wave
|
Then run exactly one scoped fix review of the fix wave
|
||||||
(`scripts/review-package PLAN_FILE FIX_BASE HEAD` over the fix range,
|
(`scripts/review-package --role fix-review PLAN_FILE FIX_BASE HEAD` over the fix range,
|
||||||
[re-review-prompt.md](re-review-prompt.md)).
|
[fix-review-prompt.md](fix-review-prompt.md)).
|
||||||
Adjudicate any residual findings as in the task loop's breaker: park with
|
Adjudicate any residual findings as in the task loop's breaker: park with
|
||||||
rulings, or stop on load-bearing ones. There is no second fix wave —
|
rulings, or stop on load-bearing ones. There is no second fix wave —
|
||||||
residual load-bearing findings surface to your human partner when
|
residual load-bearing findings surface to your human partner when
|
||||||
finishing-a-development-branch presents the options.
|
finishing-a-development-branch presents the options.
|
||||||
|
|
||||||
|
The wave closing is policy, not a verdict. A sufficiently strong reviewer
|
||||||
|
finds real defects indefinitely, so "review until one comes back clean"
|
||||||
|
never terminates — the completed wave is the exit, not a clean report.
|
||||||
|
New Critical/Important breakage in the final fix diff joins the residuals
|
||||||
|
for adjudication; it does not start a second wave. And review procedures
|
||||||
|
your human partner sets up for one review — competing reviewers, scoring,
|
||||||
|
extra seats — apply to that review only. Never adopt them as standing
|
||||||
|
procedure for reviews they didn't ask about.
|
||||||
|
|
||||||
## Finish
|
## Finish
|
||||||
|
|
||||||
When the final whole-branch review is clean and its fixes are merged,
|
When the final whole-branch review is clean and its fixes are merged,
|
||||||
@@ -429,11 +445,13 @@ Use superpowers:finishing-a-development-branch.
|
|||||||
| "Close enough on spec compliance" | Reviewer found spec gaps = not done. Fix or hit the cap and adjudicate — those are the only exits. |
|
| "Close enough on spec compliance" | Reviewer found spec gaps = not done. Fix or hit the cap and adjudicate — those are the only exits. |
|
||||||
| "I'll fix it myself, dispatching is overhead" | Controller fixes pollute your context and skip review. Resume the implementer. |
|
| "I'll fix it myself, dispatching is overhead" | Controller fixes pollute your context and skip review. Resume the implementer. |
|
||||||
| "One more round will converge" | Past the cap, rounds don't converge — the failure is structural. Adjudicate and route. |
|
| "One more round will converge" | Past the cap, rounds don't converge — the failure is structural. Adjudicate and route. |
|
||||||
| "The reviewer will just find something new anyway" | Scoped re-reviews verify fixes; they cannot wander. New findings on untouched code go to the ledger, not the loop. |
|
| "The reviewer will just find something new anyway" | Scoped fix reviews verify fixes; they cannot wander. New findings on untouched code go to the ledger, not the loop. |
|
||||||
| "This finding is obviously wrong, I'll drop it" | You adjudicate only at the cap, and every ruling is a ledger entry. Silent discards are forbidden. |
|
| "This finding is obviously wrong, I'll drop it" | You adjudicate only at the cap, and every ruling is a ledger entry. Silent discards are forbidden. |
|
||||||
| "The fix was small, skip the re-review" | Unreviewed fixes are how regressions land. Every round ends with a scoped re-review. |
|
| "The fix was small, skip the fix review" | Unreviewed fixes are how regressions land. Every round ends with a scoped fix review. |
|
||||||
| "Reviews slow the loop down" | The loop without reviews is just unverified churn. Reviews are the loop's brakes and steering. |
|
| "Reviews slow the loop down" | The loop without reviews is just unverified churn. Reviews are the loop's brakes and steering. |
|
||||||
| "Ledger bookkeeping is overhead" | The ledger is what survives compaction. Controllers without one have re-dispatched entire completed task sequences. |
|
| "Ledger bookkeeping is overhead" | The ledger is what survives compaction. Controllers without one have re-dispatched entire completed task sequences. |
|
||||||
|
| "This new finding is real — one more wave" | Real findings are infinite under a strong reviewer. The completed wave is the exit; adjudicate and route. |
|
||||||
|
| "They liked competing reviewers earlier, I'll run them again" | One-off review procedures apply to the review they were given for. Re-adopting them unasked is scope creep in review clothing. |
|
||||||
|
|
||||||
## Example Workflow
|
## Example Workflow
|
||||||
|
|
||||||
@@ -483,8 +501,8 @@ Task reviewer: Spec ❌:
|
|||||||
Implementer: Added progress reporting, extracted PROGRESS_INTERVAL constant.
|
Implementer: Added progress reporting, extracted PROGRESS_INTERVAL constant.
|
||||||
Re-ran test/recovery.test.js — 10/10 passing. Fix report appended.
|
Re-ran test/recovery.test.js — 10/10 passing. Fix report appended.
|
||||||
|
|
||||||
[Run review-package PLAN_FILE FIX_BASE HEAD; dispatch scoped re-review]
|
[Run review-package --role fix-review PLAN_FILE FIX_BASE HEAD; dispatch scoped fix review]
|
||||||
Re-reviewer: Missing progress reporting — ADDRESSED (src/recovery.js:41).
|
Fix reviewer: Missing progress reporting — ADDRESSED (src/recovery.js:41).
|
||||||
Magic number — ADDRESSED (src/recovery.js:7). New breakage: none.
|
Magic number — ADDRESSED (src/recovery.js:7). New breakage: none.
|
||||||
Verdict: all findings addressed.
|
Verdict: all findings addressed.
|
||||||
|
|
||||||
@@ -494,7 +512,7 @@ Re-reviewer: Missing progress reporting — ADDRESSED (src/recovery.js:41).
|
|||||||
...
|
...
|
||||||
|
|
||||||
[After all tasks]
|
[After all tasks]
|
||||||
[Run review-package PLAN_FILE MERGE_BASE HEAD; dispatch final code-reviewer, most capable model]
|
[Run review-package --role final-review PLAN_FILE MERGE_BASE HEAD; dispatch final code-reviewer, most capable model]
|
||||||
Final reviewer: All requirements met. Deferred minors triaged: none block merge.
|
Final reviewer: All requirements met. Deferred minors triaged: none block merge.
|
||||||
|
|
||||||
[Delete this plan's workspace — the record now lives in git]
|
[Delete this plan's workspace — the record now lives in git]
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
# Scoped Re-Review Prompt Template
|
# Scoped Fix Review Prompt Template
|
||||||
|
|
||||||
Use this template when dispatching a re-review after a fix round. The
|
Use this template when dispatching a fix review after a fix round. The
|
||||||
re-reviewer verifies the findings were addressed and checks the fix diff for
|
fix reviewer verifies the findings were addressed and checks the fix diff for
|
||||||
new breakage. It is not a fresh review — the full review already happened.
|
new breakage. It is not a fresh review — the full review already happened.
|
||||||
|
|
||||||
**Purpose:** Verify each finding from the previous review was addressed, and
|
**Purpose:** Verify each finding from the previous review was addressed, and
|
||||||
@@ -9,11 +9,11 @@ that the fix itself broke nothing.
|
|||||||
|
|
||||||
```
|
```
|
||||||
Subagent (general-purpose):
|
Subagent (general-purpose):
|
||||||
description: "Re-review Task N fix round R"
|
description: "Fix review Task N round R"
|
||||||
model: [MODEL — REQUIRED: choose per SKILL.md Model Selection; an omitted
|
model: [MODEL — REQUIRED: choose per SKILL.md Model Selection; an omitted
|
||||||
model silently inherits the session's most expensive one]
|
model silently inherits the session's most expensive one]
|
||||||
prompt: |
|
prompt: |
|
||||||
You are re-reviewing one task's fix round. A previous review produced
|
You are reviewing one task's fix round. A previous review produced
|
||||||
findings; an implementer has attempted to fix them. Your job is to
|
findings; an implementer has attempted to fix them. Your job is to
|
||||||
verdict each finding and inspect the fix diff — nothing else.
|
verdict each finding and inspect the fix diff — nothing else.
|
||||||
|
|
||||||
@@ -47,7 +47,7 @@ Subagent (general-purpose):
|
|||||||
|
|
||||||
Your scope is the findings list and the fix diff. Verdict every finding.
|
Your scope is the findings list and the fix diff. Verdict every finding.
|
||||||
Inspect the fix diff for new problems the fix itself introduced. Do NOT
|
Inspect the fix diff for new problems the fix itself introduced. Do NOT
|
||||||
re-review code the fix did not touch: if you notice an issue entirely
|
review code the fix did not touch: if you notice an issue entirely
|
||||||
outside the fix diff, report it under Out-of-Scope Observations — it
|
outside the fix diff, report it under Out-of-Scope Observations — it
|
||||||
does not block this task and does not extend the loop. A broad
|
does not block this task and does not extend the loop. A broad
|
||||||
whole-branch review happens after all tasks are complete.
|
whole-branch review happens after all tasks are complete.
|
||||||
@@ -93,14 +93,14 @@ Subagent (general-purpose):
|
|||||||
|
|
||||||
**Placeholders:**
|
**Placeholders:**
|
||||||
- `[MODEL]` — REQUIRED: reviewer model per SKILL.md Model Selection; scoped
|
- `[MODEL]` — REQUIRED: reviewer model per SKILL.md Model Selection; scoped
|
||||||
re-reviews of small fix diffs take a cheap-to-mid tier
|
fix reviews of small fix diffs take a cheap-to-mid tier
|
||||||
- `[BRIEF_FILE]` — the task brief file (same file the implementer worked from)
|
- `[BRIEF_FILE]` — the task brief file (same file the implementer worked from)
|
||||||
- `[FINDINGS]` — the Critical/Important findings and spec gaps from the
|
- `[FINDINGS]` — the Critical/Important findings and spec gaps from the
|
||||||
previous review, copied verbatim, one per bullet
|
previous review, copied verbatim, one per bullet
|
||||||
- `[REPORT_FILE]` — the implementer's report file (fix reports appended)
|
- `[REPORT_FILE]` — the implementer's report file (fix reports appended)
|
||||||
- `[FIX_BASE_SHA]` — the head the previous review saw
|
- `[FIX_BASE_SHA]` — the head the previous review saw
|
||||||
- `[HEAD_SHA]` — current commit
|
- `[HEAD_SHA]` — current commit
|
||||||
- `[DIFF_FILE]` — the path `scripts/review-package PLAN_FILE FIX_BASE HEAD` printed
|
- `[DIFF_FILE]` — the path `scripts/review-package --role fix-review PLAN_FILE FIX_BASE HEAD` printed
|
||||||
|
|
||||||
**Re-reviewer returns:** per-finding verdicts (ADDRESSED / NOT ADDRESSED),
|
**Fix reviewer returns:** per-finding verdicts (ADDRESSED / NOT ADDRESSED),
|
||||||
new breakage in the fix diff, out-of-scope observations, and a round verdict.
|
new breakage in the fix diff, out-of-scope observations, and a round verdict.
|
||||||
@@ -110,14 +110,21 @@ Subagent (general-purpose):
|
|||||||
Fix them, re-run the tests that cover the amended code, and append a fix
|
Fix them, re-run the tests that cover the amended code, and append a fix
|
||||||
report to your report file: what you changed, the covering tests you
|
report to your report file: what you changed, the covering tests you
|
||||||
ran, the command, and the output. Reviewers will not re-run tests for
|
ran, the command, and the output. Reviewers will not re-run tests for
|
||||||
you — your report is the test evidence. Then reply with the same short
|
you — your report is the test evidence. If your fix report claims a
|
||||||
status contract as your first report.
|
full-suite pass, that claim needs a fresh run after your last edit —
|
||||||
|
a suite run from before the findings arrived no longer counts. Then
|
||||||
|
reply with the same short status contract as your first report.
|
||||||
|
|
||||||
## Report Format
|
## Report Format
|
||||||
|
|
||||||
Write your full report to [REPORT_FILE]:
|
Write your full report to [REPORT_FILE]:
|
||||||
- What you implemented (or what you attempted, if blocked)
|
- What you implemented (or what you attempted, if blocked)
|
||||||
- What you tested and test results
|
- What you tested, and for every gate you claim — focused tests, full
|
||||||
|
suite, lint, build — the exact command and the tail of its fresh
|
||||||
|
output. Fresh means run after your final edit: if you edited anything
|
||||||
|
since your last full-suite run, that run is stale — rerun it or
|
||||||
|
report the suite as unverified. A gate claim without pasted fresh
|
||||||
|
output is itself a defect for the reviewer to flag.
|
||||||
- **TDD Evidence** (if TDD was required for this task):
|
- **TDD Evidence** (if TDD was required for this task):
|
||||||
- RED: command run, relevant failing output before implementation, and why the failure was expected
|
- RED: command run, relevant failing output before implementation, and why the failure was expected
|
||||||
- GREEN: command run and relevant passing output after implementation
|
- GREEN: command run and relevant passing output after implementation
|
||||||
|
|||||||
@@ -4,13 +4,31 @@
|
|||||||
# call. Using the recorded per-task BASE (not HEAD~1) keeps multi-commit
|
# call. Using the recorded per-task BASE (not HEAD~1) keeps multi-commit
|
||||||
# tasks intact.
|
# tasks intact.
|
||||||
#
|
#
|
||||||
# Usage: review-package PLAN_FILE BASE HEAD [OUTFILE]
|
# Usage: review-package [--role task-review|fix-review|final-review] PLAN_FILE BASE HEAD [OUTFILE]
|
||||||
# Default OUTFILE: <repo-root>/.superpowers/sdd/<plan-basename>/review-<base7>..<head7>.diff
|
# Default OUTFILE: <repo-root>/.superpowers/sdd/<plan-basename>/review-<base7>..<head7>.diff
|
||||||
# (named per range, so a re-review after fixes gets a distinct fresh file).
|
# (named per range, so a fix review after fixes gets a distinct fresh file).
|
||||||
|
#
|
||||||
|
# The trailing dispatch hint rides this output because the controller reads it
|
||||||
|
# immediately before spawning the reviewer; skill text loaded at session start
|
||||||
|
# does not survive context compaction, but this line is reprinted every round.
|
||||||
set -euo pipefail
|
set -euo pipefail
|
||||||
|
|
||||||
|
script_dir=$(cd "$(dirname "$0")" && pwd)
|
||||||
|
|
||||||
|
role=task-review
|
||||||
|
if [ "${1:-}" = "--role" ]; then
|
||||||
|
[ $# -ge 2 ] || { echo "usage: review-package [--role task-review|fix-review|final-review] PLAN_FILE BASE HEAD [OUTFILE]" >&2; exit 2; }
|
||||||
|
role=$2
|
||||||
|
shift 2
|
||||||
|
fi
|
||||||
|
|
||||||
|
case "$role" in
|
||||||
|
task-review|fix-review|final-review) hint_key=$role ;;
|
||||||
|
*) echo "bad --role: ${role} (task-review|fix-review|final-review)" >&2; exit 2 ;;
|
||||||
|
esac
|
||||||
|
|
||||||
if [ $# -lt 3 ] || [ $# -gt 4 ]; then
|
if [ $# -lt 3 ] || [ $# -gt 4 ]; then
|
||||||
echo "usage: review-package PLAN_FILE BASE HEAD [OUTFILE]" >&2
|
echo "usage: review-package [--role task-review|fix-review|final-review] PLAN_FILE BASE HEAD [OUTFILE]" >&2
|
||||||
exit 2
|
exit 2
|
||||||
fi
|
fi
|
||||||
|
|
||||||
@@ -25,7 +43,7 @@ git rev-parse --verify --quiet "$head" >/dev/null || { echo "bad HEAD: $head" >&
|
|||||||
if [ $# -eq 4 ]; then
|
if [ $# -eq 4 ]; then
|
||||||
out=$4
|
out=$4
|
||||||
else
|
else
|
||||||
dir=$("$(cd "$(dirname "$0")" && pwd)/sdd-workspace" "$plan")
|
dir=$("$script_dir/sdd-workspace" "$plan")
|
||||||
out="$dir/review-$(git rev-parse --short "$base")..$(git rev-parse --short "$head").diff"
|
out="$dir/review-$(git rev-parse --short "$base")..$(git rev-parse --short "$head").diff"
|
||||||
fi
|
fi
|
||||||
|
|
||||||
@@ -44,3 +62,16 @@ fi
|
|||||||
|
|
||||||
commits=$(git rev-list --count "${base}..${head}")
|
commits=$(git rev-list --count "${base}..${head}")
|
||||||
echo "wrote ${out}: ${commits} commit(s), $(wc -c < "$out" | tr -d ' ') bytes"
|
echo "wrote ${out}: ${commits} commit(s), $(wc -c < "$out" | tr -d ' ') bytes"
|
||||||
|
|
||||||
|
# Platform dispatch hints ride this output because the controller reads it
|
||||||
|
# immediately before spawning; the lines themselves are owned by the platform
|
||||||
|
# reference layer (using-superpowers/references/*-dispatch.hints), not this
|
||||||
|
# script. Claude Code's dispatch templates carry model selection already, so
|
||||||
|
# the relay is suppressed there and on any harness without a hints file.
|
||||||
|
hints_file="$script_dir/../../using-superpowers/references/codex-dispatch.hints"
|
||||||
|
if [ -z "${CLAUDECODE:-}" ] && [ -f "$hints_file" ]; then
|
||||||
|
hint_line=$(grep "^${hint_key}:" "$hints_file" | head -1 | cut -d: -f2- | sed 's/^ *//') || true
|
||||||
|
if [ -n "$hint_line" ]; then
|
||||||
|
echo "$hint_line"
|
||||||
|
fi
|
||||||
|
fi
|
||||||
|
|||||||
@@ -18,10 +18,12 @@ plan=$1
|
|||||||
n=$2
|
n=$2
|
||||||
[ -f "$plan" ] || { echo "no such plan file: $plan" >&2; exit 2; }
|
[ -f "$plan" ] || { echo "no such plan file: $plan" >&2; exit 2; }
|
||||||
|
|
||||||
|
script_dir=$(cd "$(dirname "$0")" && pwd)
|
||||||
|
|
||||||
if [ $# -eq 3 ]; then
|
if [ $# -eq 3 ]; then
|
||||||
out=$3
|
out=$3
|
||||||
else
|
else
|
||||||
dir=$("$(cd "$(dirname "$0")" && pwd)/sdd-workspace" "$plan")
|
dir=$("$script_dir/sdd-workspace" "$plan")
|
||||||
out="$dir/task-${n}-brief.md"
|
out="$dir/task-${n}-brief.md"
|
||||||
fi
|
fi
|
||||||
|
|
||||||
@@ -39,3 +41,16 @@ if [ ! -s "$out" ]; then
|
|||||||
fi
|
fi
|
||||||
|
|
||||||
echo "wrote ${out}: $(wc -l < "$out" | tr -d ' ') lines"
|
echo "wrote ${out}: $(wc -l < "$out" | tr -d ' ') lines"
|
||||||
|
|
||||||
|
# Platform dispatch hints ride this output because the controller reads it
|
||||||
|
# immediately before spawning; the lines themselves are owned by the platform
|
||||||
|
# reference layer (using-superpowers/references/*-dispatch.hints), not this
|
||||||
|
# script. Claude Code's dispatch templates carry model selection already, so
|
||||||
|
# the relay is suppressed there and on any harness without a hints file.
|
||||||
|
hints_file="$script_dir/../../using-superpowers/references/codex-dispatch.hints"
|
||||||
|
if [ -z "${CLAUDECODE:-}" ] && [ -f "$hints_file" ]; then
|
||||||
|
hint_line=$(grep "^implementer:" "$hints_file" | head -1 | cut -d: -f2- | sed 's/^ *//') || true
|
||||||
|
if [ -n "$hint_line" ]; then
|
||||||
|
echo "$hint_line"
|
||||||
|
fi
|
||||||
|
fi
|
||||||
|
|||||||
11
skills/using-superpowers/references/codex-dispatch.hints
Normal file
11
skills/using-superpowers/references/codex-dispatch.hints
Normal file
@@ -0,0 +1,11 @@
|
|||||||
|
# Per-role dispatch lines for Codex spawn_agent, relayed by the
|
||||||
|
# subagent-driven-development task-brief and review-package scripts at the
|
||||||
|
# moment of dispatch (skill text loaded at session start does not survive
|
||||||
|
# context compaction; these lines reprint every round).
|
||||||
|
# Model names track Codex's spawn_agent allowlist (currently gpt-5.6-sol
|
||||||
|
# and gpt-5.6-terra) — update this file when the allowlist changes.
|
||||||
|
# Format: <role>: <line printed verbatim>
|
||||||
|
implementer: dispatch (spawn_agent): fork_turns=none model=gpt-5.6-terra reasoning_effort=high
|
||||||
|
task-review: dispatch (spawn_agent): fork_turns=none model=gpt-5.6-terra reasoning_effort=high
|
||||||
|
fix-review: dispatch (spawn_agent): fork_turns=none model=gpt-5.6-terra reasoning_effort=medium
|
||||||
|
final-review: dispatch (spawn_agent): fork_turns=none model=gpt-5.6-terra reasoning_effort=high
|
||||||
@@ -9,6 +9,31 @@ multi_agent = true
|
|||||||
|
|
||||||
This enables `spawn_agent`, `wait_agent`, and `close_agent` for skills like `dispatching-parallel-agents` and `subagent-driven-development`. When using subagent-driven-development, close reviewer subagents when their review returns. Keep each implementer subagent open until its task's review passes — the fix loop resumes the implementer — then close it. If your harness cannot send another message to a spawned agent, dispatch each fix round as a fresh implementer carrying the brief, the report file, and the findings.
|
This enables `spawn_agent`, `wait_agent`, and `close_agent` for skills like `dispatching-parallel-agents` and `subagent-driven-development`. When using subagent-driven-development, close reviewer subagents when their review returns. Keep each implementer subagent open until its task's review passes — the fix loop resumes the implementer — then close it. If your harness cannot send another message to a spawned agent, dispatch each fix round as a fresh implementer carrying the brief, the report file, and the findings.
|
||||||
|
|
||||||
|
## SDD dispatch on Codex
|
||||||
|
|
||||||
|
Every SDD `spawn_agent` call sets `fork_turns: "none"` — the default
|
||||||
|
`"all"` forks your whole transcript into the child and refuses model
|
||||||
|
and effort overrides.
|
||||||
|
|
||||||
|
If your `spawn_agent` schema has `model` and `reasoning_effort`
|
||||||
|
parameters (Codex 0.145+), set both on every dispatch: task-brief and
|
||||||
|
review-package print a `dispatch:` hint line with the exact values —
|
||||||
|
copy it onto the call verbatim, every time, even late in a long
|
||||||
|
session. Those hints are the Model Selection mapping on Codex:
|
||||||
|
reviewer tier never exceeds implementer tier, no fix round gets an
|
||||||
|
effort bump, and rounds 4-5's "more capable model" means a fresh
|
||||||
|
implementer at the same tier — needing more is a BLOCKED escalation
|
||||||
|
to your human partner. Inherited frontier-tier subagents are a
|
||||||
|
measured cause of runs spinning out for hours. (Values live in
|
||||||
|
`codex-dispatch.hints` beside this file; they track the spawn_agent
|
||||||
|
model allowlist.)
|
||||||
|
|
||||||
|
Without those parameters (Codex 0.144 and earlier), children inherit
|
||||||
|
your model and effort with no override — role files in
|
||||||
|
`~/.codex/agents/` do not attach to spawns either. Tell your human
|
||||||
|
partner before starting a plan of more than a few tasks, and offer a
|
||||||
|
lower-effort session instead.
|
||||||
|
|
||||||
## Environment Detection
|
## Environment Detection
|
||||||
|
|
||||||
Skills that create worktrees or finish branches should detect their
|
Skills that create worktrees or finish branches should detect their
|
||||||
|
|||||||
@@ -165,6 +165,63 @@ PLAN
|
|||||||
echo " got: $rp_explicit"
|
echo " got: $rp_explicit"
|
||||||
fi
|
fi
|
||||||
|
|
||||||
|
# --- platform dispatch hints ride the script output (suppressed on CC) ---
|
||||||
|
local brief_hint
|
||||||
|
brief_hint="$(cd "$repo" && env -u CLAUDECODE "$SDD_SCRIPTS/task-brief" plan-a.md 1)"
|
||||||
|
if [[ "$brief_hint" == *"dispatch (spawn_agent): fork_turns=none model=gpt-5.6-terra reasoning_effort=high"* ]]; then
|
||||||
|
pass "task-brief relays the implementer dispatch hint off Claude Code"
|
||||||
|
else
|
||||||
|
fail "task-brief relays the implementer dispatch hint off Claude Code"
|
||||||
|
echo " got: $brief_hint"
|
||||||
|
fi
|
||||||
|
|
||||||
|
local rp_hint
|
||||||
|
rp_hint="$(cd "$repo" && env -u CLAUDECODE "$SDD_SCRIPTS/review-package" plan-a.md HEAD~1 HEAD)"
|
||||||
|
if [[ "$rp_hint" == *"dispatch (spawn_agent): fork_turns=none model=gpt-5.6-terra reasoning_effort=high"* ]]; then
|
||||||
|
pass "review-package relays the default-role hint off Claude Code"
|
||||||
|
else
|
||||||
|
fail "review-package relays the default-role hint off Claude Code"
|
||||||
|
echo " got: $rp_hint"
|
||||||
|
fi
|
||||||
|
|
||||||
|
local rp_fixreview
|
||||||
|
rp_fixreview="$(cd "$repo" && env -u CLAUDECODE "$SDD_SCRIPTS/review-package" --role fix-review plan-a.md HEAD~1 HEAD)"
|
||||||
|
if [[ "$rp_fixreview" == *"dispatch (spawn_agent): fork_turns=none model=gpt-5.6-terra reasoning_effort=medium"* ]]; then
|
||||||
|
pass "review-package --role fix-review relays the medium-effort hint"
|
||||||
|
else
|
||||||
|
fail "review-package --role fix-review relays the medium-effort hint"
|
||||||
|
echo " got: $rp_fixreview"
|
||||||
|
fi
|
||||||
|
|
||||||
|
local rp_final
|
||||||
|
rp_final="$(cd "$repo" && env -u CLAUDECODE "$SDD_SCRIPTS/review-package" --role final-review plan-a.md HEAD~1 HEAD)"
|
||||||
|
if [[ "$rp_final" == *"dispatch (spawn_agent): fork_turns=none model=gpt-5.6-terra reasoning_effort=high"* ]]; then
|
||||||
|
pass "review-package --role final-review relays the high-effort hint"
|
||||||
|
else
|
||||||
|
fail "review-package --role final-review relays the high-effort hint"
|
||||||
|
echo " got: $rp_final"
|
||||||
|
fi
|
||||||
|
|
||||||
|
local brief_cc rp_cc
|
||||||
|
brief_cc="$(cd "$repo" && CLAUDECODE=1 "$SDD_SCRIPTS/task-brief" plan-a.md 1)"
|
||||||
|
rp_cc="$(cd "$repo" && CLAUDECODE=1 "$SDD_SCRIPTS/review-package" plan-a.md HEAD~1 HEAD)"
|
||||||
|
if [[ "$brief_cc" != *"dispatch (spawn_agent)"* && "$rp_cc" != *"dispatch (spawn_agent)"* ]]; then
|
||||||
|
pass "dispatch hints are suppressed under Claude Code (CLAUDECODE set)"
|
||||||
|
else
|
||||||
|
fail "dispatch hints are suppressed under Claude Code (CLAUDECODE set)"
|
||||||
|
echo " brief: $brief_cc"
|
||||||
|
echo " rp: $rp_cc"
|
||||||
|
fi
|
||||||
|
|
||||||
|
rc=0
|
||||||
|
(cd "$repo" && "$SDD_SCRIPTS/review-package" --role bogus plan-a.md HEAD~1 HEAD >/dev/null 2>&1) || rc=$?
|
||||||
|
if [[ "$rc" -eq 2 ]]; then
|
||||||
|
pass "review-package rejects an unknown --role with exit 2"
|
||||||
|
else
|
||||||
|
fail "review-package rejects an unknown --role with exit 2"
|
||||||
|
echo " exit: $rc"
|
||||||
|
fi
|
||||||
|
|
||||||
# --- Worktree isolation: a linked worktree resolves its own workspace ---
|
# --- Worktree isolation: a linked worktree resolves its own workspace ---
|
||||||
local wt="$TEST_ROOT/wt"
|
local wt="$TEST_ROOT/wt"
|
||||||
( cd "$repo" && git worktree add -q "$wt" -b wt-feature )
|
( cd "$repo" && git worktree add -q "$wt" -b wt-feature )
|
||||||
|
|||||||
Reference in New Issue
Block a user