Sync third-party and MCP marketplace plugins

Constraint: Public skills are published only by explicit administrator action unless they are tracked third-party market sources.
Confidence: high
Scope-risk: narrow
Directive: Keep private/internal skills out of the public marketplace and preserve normal incremental market Git history.
Tested: Marketplace validation passed.
This commit is contained in:
KeyInfo Bot
2026-07-24 14:20:35 +08:00
parent 8745278d76
commit c9d4e644de
81 changed files with 2332 additions and 2284 deletions
+12 -12
View File
@@ -6,8 +6,8 @@
"repo": "https://github.com/obra/superpowers.git", "repo": "https://github.com/obra/superpowers.git",
"ref": "main", "ref": "main",
"adapter": "codex-plugin", "adapter": "codex-plugin",
"commit": "d884ae04edebef577e82ff7c4e143debd0bbec99", "commit": "3dcbd5c4b48e02263fbf4a3c01e3fe4f81d584d9",
"syncedAt": "2026-07-03T16:00:00Z" "syncedAt": "2026-07-24T06:18:20Z"
}, },
{ {
"id": "superpowers-zh", "id": "superpowers-zh",
@@ -15,8 +15,8 @@
"repo": "https://github.com/AreChen/superpowers-zh.git", "repo": "https://github.com/AreChen/superpowers-zh.git",
"ref": "main", "ref": "main",
"adapter": "codex-plugin", "adapter": "codex-plugin",
"commit": "c51f23adcd482fd908aa60928f2ece34d12f7768", "commit": "e1f21a28e5a32b94d35fdaaa98e298ef73260545",
"syncedAt": "2026-07-14T02:27:30Z" "syncedAt": "2026-07-24T06:18:20Z"
}, },
{ {
"id": "oh-my-codex", "id": "oh-my-codex",
@@ -51,8 +51,8 @@
"repo": "https://github.com/Leonxlnx/taste-skill.git", "repo": "https://github.com/Leonxlnx/taste-skill.git",
"ref": "main", "ref": "main",
"adapter": "skill-collection", "adapter": "skill-collection",
"commit": "1bffae64edbb6b2023d1e78402cd088e9c9f6511", "commit": "e988add20dab0fa97d7a76781c48961c8184288e",
"syncedAt": "2026-07-23T16:00:00Z" "syncedAt": "2026-07-24T06:18:20Z"
}, },
{ {
"id": "shadcn", "id": "shadcn",
@@ -60,8 +60,8 @@
"repo": "https://github.com/shadcn-ui/ui.git", "repo": "https://github.com/shadcn-ui/ui.git",
"ref": "main", "ref": "main",
"adapter": "claude-skill", "adapter": "claude-skill",
"commit": "6cd3f4c65c361ab6554e06a77e6a0af9cf8b6e37", "commit": "4baadbc6517070ae8f8feb2c97037adc2b305544",
"syncedAt": "2026-07-23T16:00:00Z" "syncedAt": "2026-07-24T06:18:20Z"
}, },
{ {
"id": "frontend-slides", "id": "frontend-slides",
@@ -96,8 +96,8 @@
"repo": "https://github.com/hugohe3/ppt-master.git", "repo": "https://github.com/hugohe3/ppt-master.git",
"ref": "main", "ref": "main",
"adapter": "claude-skill", "adapter": "claude-skill",
"commit": "10f0adc0600ff28a470d55992133b1992c56968a", "commit": "68c690bbe2e170bca657c2dffd47434056cfabcd",
"syncedAt": "2026-07-23T16:00:00Z" "syncedAt": "2026-07-24T06:18:20Z"
}, },
{ {
"id": "next-skills", "id": "next-skills",
@@ -105,8 +105,8 @@
"repo": "https://github.com/vercel/next.js.git", "repo": "https://github.com/vercel/next.js.git",
"ref": "canary", "ref": "canary",
"adapter": "skill-collection", "adapter": "skill-collection",
"commit": "aa4f46a540b7c9176c7c2b7ef22421adb4b5688e", "commit": "29b3966370420894de50c3c105397985fb366140",
"syncedAt": "2026-07-23T16:00:00Z" "syncedAt": "2026-07-24T06:18:20Z"
} }
] ]
} }
@@ -3,5 +3,5 @@
"name": "playwright浏览器自动化操作", "name": "playwright浏览器自动化操作",
"version": "20260605", "version": "20260605",
"keySource": "none", "keySource": "none",
"syncedAt": "2026-07-23T16:02:47Z" "syncedAt": "2026-07-24T06:20:34Z"
} }
@@ -2,8 +2,8 @@
"sourceId": "next-skills", "sourceId": "next-skills",
"repo": "https://github.com/vercel/next.js.git", "repo": "https://github.com/vercel/next.js.git",
"ref": "canary", "ref": "canary",
"commit": "aa4f46a540b7c9176c7c2b7ef22421adb4b5688e", "commit": "29b3966370420894de50c3c105397985fb366140",
"adapter": "skill-collection", "adapter": "skill-collection",
"sourcePath": "skills", "sourcePath": "skills",
"syncedAt": "2026-07-23T16:00:00Z" "syncedAt": "2026-07-24T06:18:20Z"
} }
@@ -2,8 +2,8 @@
"sourceId": "ppt-master", "sourceId": "ppt-master",
"repo": "https://github.com/hugohe3/ppt-master.git", "repo": "https://github.com/hugohe3/ppt-master.git",
"ref": "main", "ref": "main",
"commit": "10f0adc0600ff28a470d55992133b1992c56968a", "commit": "68c690bbe2e170bca657c2dffd47434056cfabcd",
"adapter": "claude-skill", "adapter": "claude-skill",
"sourcePath": "skills/ppt-master", "sourcePath": "skills/ppt-master",
"syncedAt": "2026-07-23T16:00:00Z" "syncedAt": "2026-07-24T06:18:20Z"
} }
@@ -33,7 +33,7 @@ Global artifact ownership rules for PPT Master projects.
| `<import_workspace>/authoring-svg/authoring_manifest.json` | Tool-only authoring provenance contract | Per-document source/authoring hashes and document-local source-ref paths | Generated atomically with the IR; materialization validates it before reusing native payload; never load it into model context or duplicate raw payload here | | `<import_workspace>/authoring-svg/authoring_manifest.json` | Tool-only authoring provenance contract | Per-document source/authoring hashes and document-local source-ref paths | Generated atomically with the IR; materialization validates it before reusing native payload; never load it into model context or duplicate raw payload here |
| `<import_workspace>/authoring-svg-flat/` | Optional complete-page verification IR | Self-contained page composition view with its own summary and provenance manifest | Generate only from an explicitly requested `svg-flat/`; use to verify composition, while layered `authoring-svg/` remains the canonical editable source | | `<import_workspace>/authoring-svg-flat/` | Optional complete-page verification IR | Self-contained page composition view with its own summary and provenance manifest | Generate only from an explicitly requested `svg-flat/`; use to verify composition, while layered `authoring-svg/` remains the canonical editable source |
| `<import_workspace>/icons/imported/` | Imported vector pool | One canonical copy of every factored vector subtree | Authoring SVGs reference `data-icon="imported/<name>"`; vector inventories retain source refs so expansion re-establishes IR identity | | `<import_workspace>/icons/imported/` | Imported vector pool | One canonical copy of every factored vector subtree | Authoring SVGs reference `data-icon="imported/<name>"`; vector inventories retain source refs so expansion re-establishes IR identity |
| `confirm_ui/recommendations.json` | Confirmation proposal | Strategist-authored confirmation payload | Confirm UI reads; rewritten between Stage 1, Stage 2, and Stage 3 | | `confirm_ui/recommendations.stage1.json`, `.stage2.json`, `.stage3.json` | Confirmation proposals | One Strategist-authored payload per confirmation stage | Confirm UI selects the active file from `result.json`. The active, unconfirmed stage may be overwritten when the user requests a new recommendation; normal progression writes the next stage file and leaves confirmed earlier stages intact. Legacy `recommendations.json` is read only when no stage-specific file exists. |
| `confirm_ui/result.json` | Confirmation result | Persisted user-confirmed input evidence | Generate Step 4 reads the final object once into active context; Strategist consumes it completely into `design_spec.md`. Normal downstream work does not reopen it; fresh recovery may read it once when no retained final state exists. | | `confirm_ui/result.json` | Confirmation result | Persisted user-confirmed input evidence | Generate Step 4 reads the final object once into active context; Strategist consumes it completely into `design_spec.md`. Normal downstream work does not reopen it; fresh recovery may read it once when no retained final state exists. |
| `svg_output/` | Page-design author source | Main-agent handwritten SVG pages containing the complete visible design | Quality checker and native PPTX export read this as the canonical visual/page-layout source; templates and locks do not add missing visible objects at export | | `svg_output/` | Page-design author source | Main-agent handwritten SVG pages containing the complete visible design | Quality checker and native PPTX export read this as the canonical visual/page-layout source; templates and locks do not add missing visible objects at export |
| `notes/total.md` | Speaker-note source | Complete notes before splitting | Step 6 writes; Step 7.1 splits | | `notes/total.md` | Speaker-note source | Complete notes before splitting | Step 6 writes; Step 7.1 splits |
@@ -131,7 +131,7 @@ Before drawing each page, look up its entry in `page_rhythm` (key format `P<NN>`
- **Proximity**: group related elements with tight spacing; separate unrelated groups - **Proximity**: group related elements with tight spacing; separate unrelated groups
- **Element grouping (Mandatory)**: wrap each logical Slide-local body unit in a descriptive, page-unique top-level `<g id>`. Every visible direct root `<g>` declares root-coordinate `data-pptx-bounds="x y width height"`; frame/native coordinates do not replace it, and placeholder bounds also supply the slot frame. Nested groups need no bounds and any such values are ignored. Checker compares root bounds with the `viewBox` and recursively checks only estimable text against its root module: through `1px` is ignored, through `5%` warns, above `5%` fails per side. Images, shapes, paths, `<use>`, effects, and object frames remain geometrically free. Flat pages use ordinary groups; structured slots already qualify, while titles, direct Master/Layout atoms, and canvas-level static framing may remain root primitives. On flat pages, give a root background image or full-canvas scrim/decoration rectangle a stable `id` plus `data-pptx-role="background"` / `"decoration"`; never wrap it only to silence the advisory. - **Element grouping (Mandatory)**: wrap each logical Slide-local body unit in a descriptive, page-unique top-level `<g id>`. Every visible direct root `<g>` declares root-coordinate `data-pptx-bounds="x y width height"`; frame/native coordinates do not replace it, and placeholder bounds also supply the slot frame. Nested groups need no bounds and any such values are ignored. Checker compares root bounds with the `viewBox` and recursively checks only estimable text against its root module: through `1px` is ignored, through `5%` warns, above `5%` fails per side. Images, shapes, paths, `<use>`, effects, and object frames remain geometrically free. Flat pages use ordinary groups; structured slots already qualify, while titles, direct Master/Layout atoms, and canvas-level static framing may remain root primitives. On flat pages, give a root background image or full-canvas scrim/decoration rectangle a stable `id` plus `data-pptx-role="background"` / `"decoration"`; never wrap it only to silence the advisory.
- **Default — size `data-pptx-bounds` as the intended module zone, not a glyph box (may skip when no text is estimable)**: an untransformed line spans `y - 0.85 × font_size` to `y + 0.35 × font_size`, and per-character width estimates undercount CJK—most heavily on serif stacks—so leave headroom. If text does not fit, reflow or adapt; correct bounds only when that zone was recorded incorrectly. Larger bounds do not repair off-canvas text. - **Default — size `data-pptx-bounds` as the intended module zone, not a glyph box (may skip when no text is estimable)**: make the zone as generous as the canvas and sibling layout allow, without overlapping another module zone. An untransformed line spans `y - 0.85 × font_size` to `y + 0.35 × font_size`; width uses the shared SVG-to-PPTX per-run estimate and safety headroom. If text does not fit, first expand a zone that has unused non-overlapping space; otherwise reflow or adapt. Larger bounds do not repair off-canvas text.
- **Spec adherence**: follow color, layout, canvas format, and typography in the spec - **Spec adherence**: follow color, layout, canvas format, and typography in the spec
- **Template structure**: inherit the native visual framework only for `template_reuse_scope: mirror|layout`; `style` uses the flat route - **Template structure**: inherit the native visual framework only for `template_reuse_scope: mirror|layout`; `style` uses the flat route
- **Main-agent ownership**: SVG generation must run in the main agent (not sub-agents) — pages share upstream context for cross-page visual continuity - **Main-agent ownership**: SVG generation must run in the main agent (not sub-agents) — pages share upstream context for cross-page visual continuity
@@ -583,7 +583,7 @@ These forms are needed only when the stated PPT behavior matters:
### 4.3 Element Grouping (Mandatory) ### 4.3 Element Grouping (Mandatory)
**Hard rule — root groups protect body-text layout**: Every visible direct root `<g>` declares positive root-coordinate `data-pptx-bounds="x y width height"`. Keep it when frame/native coordinates size one PowerPoint object; placeholder bounds also supply the slot frame. Checker validates this subcanvas against the root `viewBox`, then recursively validates only estimable `<text>` descendants against it. Nested groups and all shapes, images, paths, `<use>` instances, effects, and object frames are not content-boundary inputs. Per side, Checker ignores text/bounds overflow through `1px`, warns through `5%` of the containing boundary dimension, and fails above `5%`. Bounds do not clip or reflow. **Hard rule — root groups protect body-text layout**: Every visible direct root `<g>` declares positive root-coordinate `data-pptx-bounds="x y width height"`. Keep it when frame/native coordinates size one PowerPoint object; placeholder bounds also supply the slot frame. On flat pages, make each module zone as generous as the canvas and sibling layout allow without overlapping another module zone. Checker validates this subcanvas against the root `viewBox`, then recursively validates only estimable `<text>` descendants against it using the shared SVG-to-PPTX per-run width estimate and safety headroom. Nested groups and all shapes, images, paths, `<use>` instances, effects, and object frames are not content-boundary inputs. Per side, Checker ignores text/bounds overflow through `1px`, warns through `5%` of the containing boundary dimension, and fails above `5%`. Bounds do not clip or reflow.
Wrap each logical Slide-local body unit in one descriptive top-level `<g id>`; group count follows the page's semantic units, and each group becomes one animation step when animation is enabled. Nested implementation groups may remain anonymous and need no bounds; any nested bounds are ignored. Flat pages use ordinary groups; structured slots already qualify, while titles, direct atomic Master/Layout elements, and canvas-level static framing—including background images and full-canvas scrim/decoration rectangles—may remain root primitives. On flat pages, give such static framing a stable `id` plus `data-pptx-role="background"` / `"decoration"`; never add a `<g>` solely to silence an ungrouped-element advisory. Wrap each logical Slide-local body unit in one descriptive top-level `<g id>`; group count follows the page's semantic units, and each group becomes one animation step when animation is enabled. Nested implementation groups may remain anonymous and need no bounds; any nested bounds are ignored. Flat pages use ordinary groups; structured slots already qualify, while titles, direct atomic Master/Layout elements, and canvas-level static framing—including background images and full-canvas scrim/decoration rectangles—may remain root primitives. On flat pages, give such static framing a stable `id` plus `data-pptx-role="background"` / `"decoration"`; never add a `<g>` solely to silence an ungrouped-element advisory.
@@ -42,9 +42,9 @@ Record the resulting exporter plan internally:
| `template_adherence: strict` | Every structured page fits an existing prototype contract without changing its Layout identity or slot topology. Mandatory for `template_reuse_scope: mirror`. | | `template_adherence: strict` | Every structured page fits an existing prototype contract without changing its Layout identity or slot topology. Mandatory for `template_reuse_scope: mirror`. |
| `template_adherence: adaptive` | Structured reuse remains useful, but at least one page needs a new explicit Layout under the selected Master. | | `template_adherence: adaptive` | Structured reuse remains useful, but at least one page needs a new explicit Layout under the selected Master. |
Write only the derived values to `spec_lock.md pptx_structure`; omit `template_adherence` for `style`. Do not put these internal values in `design_spec.md`, `recommendations.json`, the Confirm UI, or `result.json`. Write only the derived values to `spec_lock.md pptx_structure`; omit `template_adherence` for `style`. Do not put these internal values in `design_spec.md`, recommendation stage files, the Confirm UI, or `result.json`.
**Mandatory — natural-language Stage-2 plan**: Summarize which prototypes are used/skipped/repeated/reordered, what stays literal, and what may be replaced or reorganized. Write it to top-level `template_application.value` in Stage-2 `recommendations.json`; omit it without an active template. After Stage 2, re-read the confirmed `result.json` value (or exact chat answer), never the initial recommendation. Blank returns the decision to Strategist. Persist the effective plan on one line as `- **Template Application**: <prose>` in `design_spec.md §I`, then derive internal reuse/adherence values and mappings; never copy the prose to `spec_lock.md`. Do not add a questionnaire, internal controls, or fixed template-use options. **Mandatory — natural-language Stage-2 plan**: Summarize which prototypes are used/skipped/repeated/reordered, what stays literal, and what may be replaced or reorganized. Write it to top-level `template_application.value` in `recommendations.stage2.json`; omit it without an active template. After Stage 2, re-read the confirmed `result.json` value (or exact chat answer), never the initial recommendation. Blank returns the decision to Strategist. Persist the effective plan on one line as `- **Template Application**: <prose>` in `design_spec.md §I`, then derive internal reuse/adherence values and mappings; never copy the prose to `spec_lock.md`. Do not add a questionnaire, internal controls, or fixed template-use options.
**Three-stage boundary**: An installed template changes the content of Stage 2, never the confirmation sequence. Run Stage 1 → Stage 2 → Stage 3 in order in both Confirm UI and chat fallback; do not skip a stage or treat template inspection as user confirmation. On browser timeout, return to the same stage in chat. **Three-stage boundary**: An installed template changes the content of Stage 2, never the confirmation sequence. Run Stage 1 → Stage 2 → Stage 3 in order in both Confirm UI and chat fallback; do not skip a stage or treat template inspection as user confirmation. On browser timeout, return to the same stage in chat.
@@ -38,7 +38,7 @@ Do not force communication intent into one catalog label; Stage 1 records compos
> >
> **One opt-in exception**: present the spec-refinement line alongside the split-mode note ([`generate-pptx.md`](../workflows/generate-pptx.md) Step 4). It is OFF by default — the above discipline holds unchanged. Only when the user *explicitly* asks to refine the spec do you hand off to the [refine-spec](../workflows/stages/refine-spec.md) stage, which produces the full spec first and stops for user review/revision of any part before generation. Never enter it unprompted. > **One opt-in exception**: present the spec-refinement line alongside the split-mode note ([`generate-pptx.md`](../workflows/generate-pptx.md) Step 4). It is OFF by default — the above discipline holds unchanged. Only when the user *explicitly* asks to refine the spec do you hand off to the [refine-spec](../workflows/stages/refine-spec.md) stage, which produces the full spec first and stops for user review/revision of any part before generation. Never enter it unprompted.
> **Default presentation surface — Confirm UI.** Write `<project>/confirm_ui/recommendations.json` and launch per Generate Step 4. Stage 2 carries ≥3 safe / shifted / bold `design_directions`; each bundles visual style, a six-role HEX palette, CJK + Latin heading/body typography, icons, and conditional image rendering. Also print the recommendations + URL in chat as fallback context. Skip launch only for an explicit chat-only request; a chat-question tool is not a substitute. Generate Step 4 reads the final confirmed `result.json` once and retains that object for Design Spec authoring. [`confirm_ui.md`](../scripts/docs/confirm_ui.md) owns schema and lifecycle. > **Default presentation surface — Confirm UI.** Use `<project>/confirm_ui/recommendations.stage1.json`, `.stage2.json`, and `.stage3.json` at their documented handoffs and launch per Generate Step 4. The active, unconfirmed stage may be overwritten when the user asks for a new recommendation; normal progression writes the next stage file and leaves confirmed earlier stages intact. Stage 2 carries ≥3 safe / shifted / bold `design_directions`; each bundles visual style, a six-role HEX palette, CJK + Latin heading/body typography, icons, and conditional image rendering. Also print the recommendations + URL in chat as fallback context. Skip launch only for an explicit chat-only request; a chat-question tool is not a substitute. Generate Step 4 reads the final confirmed `result.json` once and retains that object for Design Spec authoring. [`confirm_ui.md`](../scripts/docs/confirm_ui.md) owns schema and lifecycle.
**Confirmed-value semantics**: confirmation preserves both the value and the owning field's semantic type. Apply the type to the affected property, not automatically to the whole object: **Confirmed-value semantics**: confirmation preserves both the value and the owning field's semantic type. Apply the type to the affected property, not automatically to the whole object:
@@ -115,7 +115,7 @@ When authoring §IX, translate every purpose named in `communication_intent` int
Two independent layers, each locks one preset or `custom`. Output: `d. Mode: <mode> + Visual style: <visual_style>`. Two independent layers, each locks one preset or `custom`. Output: `d. Mode: <mode> + Visual style: <visual_style>`.
> **Mandatory AI custom candidates.** Every Stage-2 `recommendations.json` carries visible, non-empty `custom_candidates.mode` and `.visual_style`, initially unselected unless the user supplied that exact direction. If a proposal combines or borrows catalog entries, read every named entry file before authoring the synthesis and name those exact ids in the visible proposal; a genuinely novel proposal needs no catalog reference. If selected, spell the proposal out in plain language and save literal `custom` plus the edited `mode_behavior` / `visual_style_behavior`; otherwise it remains recommendation-only. Never write bespoke prose as the enum value. > **Mandatory AI custom candidates.** Every `recommendations.stage2.json` carries visible, non-empty `custom_candidates.mode` and `.visual_style`, initially unselected unless the user supplied that exact direction. If a proposal combines or borrows catalog entries, read every named entry file before authoring the synthesis and name those exact ids in the visible proposal; a genuinely novel proposal needs no catalog reference. If selected, spell the proposal out in plain language and save literal `custom` plus the edited `mode_behavior` / `visual_style_behavior`; otherwise it remains recommendation-only. Never write bespoke prose as the enum value.
#### Layer 1 — Communication mode #### Layer 1 — Communication mode
@@ -139,7 +139,7 @@ The deck's **visual aesthetic** — shape language, decoration density, whitespa
**Source**: **Source**:
- User named a style (chat / template / beautify) → it is truth: map to the closest preset (or `custom` with a `visual_style_behavior` paragraph) and lock directly. **Skip the spectrum below** — do not re-offer choice they already made. - User named a style (chat / template / beautify) → it is truth: map to the closest preset (or `custom` with a `visual_style_behavior` paragraph) and lock directly. **Skip the spectrum below** — do not re-offer choice they already made.
- No user description → **present a personality spectrum, not one safe pick** (this is the lever against "every deck looks the same" — the visual style is what most determines a deck's character, so it gets real choice, like the alternative-set rule used for image rendering). Author **≥3 distinct styles** from the index's auto-selection table spanning *safe* (the industry-norm recommendation) → *shifted* (an alternate one tick more expressive) → *bold* (a characterful style that challenges the default — `brutalist` / `zine` / `memphis` / `ink-wash` / `vintage-poster` etc., whenever the content can carry it). Give each a one-line **temperament tag + real-world analogy** (for example, "like an Economist feature"). Write the three to `recommendations.json` `visual_style_spectrum` (each `{id, tag_zh/en/ja, note_zh/en/ja}` — include the `_ja` variants whenever the page `lang` is `ja`) **and present the same three in chat** as the always-valid fallback; set `recommend.visual_style` to the *safe* pick as the pre-selected default. The user may pick any of the three or the separate full-copy Custom proposal. Honest-shortfall may reduce the preset set, never remove Custom. - No user description → **present a personality spectrum, not one safe pick** (this is the lever against "every deck looks the same" — the visual style is what most determines a deck's character, so it gets real choice, like the alternative-set rule used for image rendering). Author **≥3 distinct styles** from the index's auto-selection table spanning *safe* (the industry-norm recommendation) → *shifted* (an alternate one tick more expressive) → *bold* (a characterful style that challenges the default — `brutalist` / `zine` / `memphis` / `ink-wash` / `vintage-poster` etc., whenever the content can carry it). Give each a one-line **temperament tag + real-world analogy** (for example, "like an Economist feature"). Write the three to `recommendations.stage2.json` `visual_style_spectrum` (each `{id, tag_zh/en/ja, note_zh/en/ja}` — include the `_ja` variants whenever the page `lang` is `ja`) **and present the same three in chat** as the always-valid fallback; set `recommend.visual_style` to the *safe* pick as the pre-selected default. The user may pick any of the three or the separate full-copy Custom proposal. Honest-shortfall may reduce the preset set, never remove Custom.
**Forbidden — a non-catalog name as `visual_style`**: the value MUST be an `id` from the visual-styles catalog or literal `custom`; bespoke prose belongs only in `visual_style_behavior`. A name that is **not** in that catalog is not a visual style — most often it is an image-rendering name from the `_index` "Paired rendering" column (`flat`, `vector-illustration`, `digital-dashboard`, `3d-isometric`, `corporate-photo`, …), which names the §h *illustration* family, not the deck's layout aesthetic. Do not borrow it. (Names that are intentionally **both** a style and its paired rendering — `glassmorphism`, `blueprint`, `editorial`, `dark-tech` — are valid styles because they *are* in the catalog.) Generic baseline words — `flat` / flat-design / 扁平 / modern / clean / simple / minimal — are **not** custom-worthy either: the whole system is flat by default (shadows discouraged), so map them to the closest preset (flat + grid → `swiss-minimal`; flat + rounded → `soft-rounded`; flat + dense → `brutalist`). Reserve a custom lock for an aesthetic no preset covers; the mandatory candidate does not make it the default. **Forbidden — a non-catalog name as `visual_style`**: the value MUST be an `id` from the visual-styles catalog or literal `custom`; bespoke prose belongs only in `visual_style_behavior`. A name that is **not** in that catalog is not a visual style — most often it is an image-rendering name from the `_index` "Paired rendering" column (`flat`, `vector-illustration`, `digital-dashboard`, `3d-isometric`, `corporate-photo`, …), which names the §h *illustration* family, not the deck's layout aesthetic. Do not borrow it. (Names that are intentionally **both** a style and its paired rendering — `glassmorphism`, `blueprint`, `editorial`, `dark-tech` — are valid styles because they *are* in the catalog.) Generic baseline words — `flat` / flat-design / 扁平 / modern / clean / simple / minimal — are **not** custom-worthy either: the whole system is flat by default (shadows discouraged), so map them to the closest preset (flat + grid → `swiss-minimal`; flat + rounded → `soft-rounded`; flat + dense → `brutalist`). Reserve a custom lock for an aesthetic no preset covers; the mandatory candidate does not make it the default.
@@ -2,12 +2,12 @@
""" """
PPT Master - Strategist confirmation stage UI Server (Step 4) PPT Master - Strategist confirmation stage UI Server (Step 4)
Lightweight Flask backend for the interactive, visual Strategist confirmation stage page. Lightweight Flask backend for the interactive, visual Strategist confirmation
Strategist writes its recommendations to stage page. Strategist writes each stage to
``<project>/confirm_ui/recommendations.json``; this server renders them as a ``<project>/confirm_ui/recommendations.stageN.json``; this server selects the
clickable page (color swatches, live font previews, candidate picks). On current stage from ``result.json`` and renders it as a clickable page (color
submit it writes the user's final choices to swatches, live font previews, candidate picks). On submit it writes the user's
``<project>/confirm_ui/result.json`` for the AI to read back. choices to ``<project>/confirm_ui/result.json`` for the AI to read back.
This is the default confirmation surface. The chat fallback is used only when This is the default confirmation surface. The chat fallback is used only when
the user explicitly requests chat-only confirmation or the browser launch the user explicitly requests chat-only confirmation or the browser launch
@@ -76,7 +76,12 @@ LOCK_FILE_NAME = '.confirm_ui.lock'
# Round-trip/session files, all under <project_path>/confirm_ui/. # Round-trip/session files, all under <project_path>/confirm_ui/.
CONFIRM_DIR_NAME = 'confirm_ui' CONFIRM_DIR_NAME = 'confirm_ui'
RECOMMENDATIONS_NAME = 'recommendations.json' LEGACY_RECOMMENDATIONS_NAME = 'recommendations.json'
RECOMMENDATION_STAGE_NAMES = {
1: 'recommendations.stage1.json',
2: 'recommendations.stage2.json',
3: 'recommendations.stage3.json',
}
RESULT_NAME = 'result.json' RESULT_NAME = 'result.json'
SESSION_NAME = 'session.json' SESSION_NAME = 'session.json'
@@ -263,6 +268,7 @@ def _wait_for_result(
proc: subprocess.Popen, proc: subprocess.Popen,
started_at: float, started_at: float,
timeout: int, timeout: int,
expected_stage: str,
) -> int: ) -> int:
"""Wait until this launch writes a fresh result file or the server exits.""" """Wait until this launch writes a fresh result file or the server exits."""
logger.info('waiting for browser confirmation...') logger.info('waiting for browser confirmation...')
@@ -272,7 +278,6 @@ def _wait_for_result(
try: try:
if result_file.stat().st_mtime >= started_at: if result_file.stat().st_mtime >= started_at:
actual_stage = _result_stage(result_file) actual_stage = _result_stage(result_file)
expected_stage = _expected_result_stage(result_file.parent)
if actual_stage != expected_stage: if actual_stage != expected_stage:
logger.error( logger.error(
'confirmation stage mismatch: expected %s, found %s', 'confirmation stage mismatch: expected %s, found %s',
@@ -335,7 +340,7 @@ def _stage_key(value: object) -> Optional[str]:
def _recommendation_stage(data: dict) -> int: def _recommendation_stage(data: dict) -> int:
"""Return recommendations.json stage number, with legacy tier fallback.""" """Return a recommendation payload's stage, with legacy tier fallback."""
stage = _stage_key(data.get('stage')) stage = _stage_key(data.get('stage'))
if not stage and 'tier' in data: if not stage and 'tier' in data:
stage = _stage_key(data.get('tier')) stage = _stage_key(data.get('tier'))
@@ -370,6 +375,68 @@ def _result_stage_number(stage: Optional[str]) -> int:
return 0 return 0
def _expected_recommendation_stage(result_stage: Optional[str]) -> int:
"""Return the recommendation stage that follows the current result."""
if result_stage == 'stage1':
return 2
if result_stage in {'stage2', 'final'}:
return 3
return 1
def _stage_recommendations_path(confirm_dir: Path, stage_number: int) -> Path:
"""Return the stage-specific recommendation path for one handoff."""
return confirm_dir / RECOMMENDATION_STAGE_NAMES[stage_number]
def _active_recommendations_path(confirm_dir: Path) -> Path:
"""Resolve the recommendation file for the current confirmation stage.
Once any stage-specific file exists, the legacy single-file input is ignored.
If the expected stage is missing but a later stage exists, return that later
file so the existing stage-skip guard can report the ordering error.
"""
result_stage = _result_stage(confirm_dir / RESULT_NAME)
expected_stage = _expected_recommendation_stage(result_stage)
expected_file = _stage_recommendations_path(confirm_dir, expected_stage)
staged_files = {
stage_number: _stage_recommendations_path(confirm_dir, stage_number)
for stage_number in RECOMMENDATION_STAGE_NAMES
}
if any(path.exists() for path in staged_files.values()):
if expected_file.exists():
return expected_file
for stage_number in range(expected_stage + 1, 4):
candidate = staged_files[stage_number]
if candidate.exists():
return candidate
return expected_file
legacy_file = confirm_dir / LEGACY_RECOMMENDATIONS_NAME
return legacy_file if legacy_file.exists() else expected_file
def _read_active_recommendations(
confirm_dir: Path,
*,
retries: int = 2,
) -> tuple[Path, dict]:
"""Read and validate the recommendation payload active for this stage."""
rec_file = _active_recommendations_path(confirm_dir)
data = _read_json_object(rec_file, retries=retries)
for stage_number, filename in RECOMMENDATION_STAGE_NAMES.items():
if rec_file.name != filename:
continue
actual_stage = _recommendation_stage(data)
if actual_stage != stage_number:
raise ValueError(
f'{filename} must declare stage={_stage_name(stage_number)}, '
f'found {_stage_name(actual_stage) or "absent"}'
)
break
return rec_file, data
def _stage_skip(rec_stage_number: int, result_stage: Optional[str]) -> bool: def _stage_skip(rec_stage_number: int, result_stage: Optional[str]) -> bool:
"""Detect a staged recommendation running ahead of the confirmed progression. """Detect a staged recommendation running ahead of the confirmed progression.
@@ -386,9 +453,9 @@ def _stage_skip(rec_stage_number: int, result_stage: Optional[str]) -> bool:
def _stage_skip_error(confirm_dir: Path) -> Optional[str]: def _stage_skip_error(confirm_dir: Path) -> Optional[str]:
"""Return a directive error when recommendations.json skips a stage.""" """Return a directive error when the active recommendation skips a stage."""
try: try:
rec_data = _read_json_object(confirm_dir / RECOMMENDATIONS_NAME) rec_file, rec_data = _read_active_recommendations(confirm_dir)
except (OSError, json.JSONDecodeError, ValueError): except (OSError, json.JSONDecodeError, ValueError):
return None return None
rec_stage_number = _recommendation_stage(rec_data) rec_stage_number = _recommendation_stage(rec_data)
@@ -400,11 +467,14 @@ def _stage_skip_error(confirm_dir: Path) -> Optional[str]:
'--daemon --wait' if expected == 'stage1' '--daemon --wait' if expected == 'stage1'
else '--wait-only --wait-stage stage2' else '--wait-only --wait-stage stage2'
) )
expected_file = RECOMMENDATION_STAGE_NAMES[
_expected_recommendation_stage(result_stage)
]
return ( return (
f'stage skip detected: recommendations.json is {_stage_name(rec_stage_number)} but the last ' f'stage skip detected: {rec_file.name} is {_stage_name(rec_stage_number)} but the last '
f'confirmed result is {result_stage or "absent"} — the page will not render a skipped stage. ' f'confirmed result is {result_stage or "absent"} — the page will not render a skipped stage. '
f'Stages confirm in order and an active template does not exempt stage2 (generate-pptx Step 4). ' f'Stages confirm in order and an active template does not exempt stage2 (generate-pptx Step 4). '
f'Overwrite recommendations.json with the {expected} recommendations, then re-run with {reattach}.' f'Write the missing {expected_file} recommendations, then re-run with {reattach}.'
) )
@@ -612,9 +682,9 @@ def _submission_stage_error(
) -> Optional[str]: ) -> Optional[str]:
"""Reject a confirmation that does not match the staged recommendation.""" """Reject a confirmation that does not match the staged recommendation."""
try: try:
recommendations = _read_json_object(confirm_dir / RECOMMENDATIONS_NAME) rec_file, recommendations = _read_active_recommendations(confirm_dir)
except (OSError, json.JSONDecodeError, ValueError) as exc: except (OSError, json.JSONDecodeError, ValueError) as exc:
return f'cannot confirm without valid recommendations.json: {exc}' return f'cannot confirm without valid current recommendations: {exc}'
rec_stage_number = _recommendation_stage(recommendations) rec_stage_number = _recommendation_stage(recommendations)
template_required = _template_confirmation_required( template_required = _template_confirmation_required(
@@ -653,7 +723,7 @@ def _submission_stage_error(
if submitted_stage not in allowed_submissions[rec_stage_number]: if submitted_stage not in allowed_submissions[rec_stage_number]:
expected = 'final' if rec_stage_number == 3 else _stage_name(rec_stage_number) expected = 'final' if rec_stage_number == 3 else _stage_name(rec_stage_number)
return ( return (
f'confirmation stage mismatch: recommendations.json is ' f'confirmation stage mismatch: {rec_file.name} is '
f'{_stage_name(rec_stage_number)}, so the submitted stage must be ' f'{_stage_name(rec_stage_number)}, so the submitted stage must be '
f'{expected}' f'{expected}'
) )
@@ -747,7 +817,7 @@ def _normalize_custom_selections(result: dict) -> None:
def _expected_result_stage(confirm_dir: Path) -> str: def _expected_result_stage(confirm_dir: Path) -> str:
"""Return the result stage expected from the current recommendations.""" """Return the result stage expected from the current recommendations."""
try: try:
recommendations = _read_json_object(confirm_dir / RECOMMENDATIONS_NAME) _, recommendations = _read_active_recommendations(confirm_dir)
except (OSError, json.JSONDecodeError, ValueError): except (OSError, json.JSONDecodeError, ValueError):
return 'final' return 'final'
return { return {
@@ -784,7 +854,7 @@ def _build_session_state(
) -> dict: ) -> dict:
"""Derive the resumable Confirm UI state from disk artifacts.""" """Derive the resumable Confirm UI state from disk artifacts."""
previous = _read_session(confirm_dir) previous = _read_session(confirm_dir)
rec_file = confirm_dir / RECOMMENDATIONS_NAME rec_file = _active_recommendations_path(confirm_dir)
result_file = confirm_dir / RESULT_NAME result_file = confirm_dir / RESULT_NAME
rec_stage_number = 0 rec_stage_number = 0
@@ -792,7 +862,7 @@ def _build_session_state(
rec_error = None rec_error = None
if rec_file.exists(): if rec_file.exists():
try: try:
rec_data = _read_json_object(rec_file) rec_file, rec_data = _read_active_recommendations(confirm_dir)
rec_stage_number = _recommendation_stage(rec_data) rec_stage_number = _recommendation_stage(rec_data)
rec_stage = _stage_name(rec_stage_number) rec_stage = _stage_name(rec_stage_number)
except (OSError, json.JSONDecodeError, ValueError) as exc: except (OSError, json.JSONDecodeError, ValueError) as exc:
@@ -803,7 +873,7 @@ def _build_session_state(
stage_skip = _stage_skip(rec_stage_number, result_stage) stage_skip = _stage_skip(rec_stage_number, result_stage)
# A skipped stage is never presented: the page keeps its "deriving…" state # A skipped stage is never presented: the page keeps its "deriving…" state
# (waiting_agent) until the AI rewrites recommendations.json in order. # (waiting_agent) until the AI creates the missing stage file.
if result_stage == 'final': if result_stage == 'final':
expected_stage_number = None expected_stage_number = None
status = 'done' status = 'done'
@@ -832,6 +902,7 @@ def _build_session_state(
'expected_stage_number': expected_stage_number, 'expected_stage_number': expected_stage_number,
'recommendation_stage': rec_stage, 'recommendation_stage': rec_stage,
'recommendation_stage_number': rec_stage_number, 'recommendation_stage_number': rec_stage_number,
'recommendation_file': rec_file.name,
'recommendation_version': _file_version(rec_file), 'recommendation_version': _file_version(rec_file),
'recommendation_error': rec_error, 'recommendation_error': rec_error,
'result_stage': result_stage, 'result_stage': result_stage,
@@ -1259,12 +1330,15 @@ def create_app(
@app.route('/api/health') @app.route('/api/health')
def health(): def health():
"""Expose a cheap readiness probe for the daemon launcher.""" """Expose a cheap readiness probe for the daemon launcher."""
rec_file = confirm_dir / RECOMMENDATIONS_NAME rec_file = _active_recommendations_path(confirm_dir)
rec_ok = False rec_ok = False
stage = None stage = None
if rec_file.exists(): if rec_file.exists():
try: try:
rec_data = _read_json_object(rec_file, retries=0) rec_file, rec_data = _read_active_recommendations(
confirm_dir,
retries=0,
)
rec_ok = True rec_ok = True
stage = _recommendation_stage(rec_data) stage = _recommendation_stage(rec_data)
except (OSError, json.JSONDecodeError, ValueError): except (OSError, json.JSONDecodeError, ValueError):
@@ -1334,15 +1408,24 @@ def create_app(
@app.route('/api/recommendations') @app.route('/api/recommendations')
def get_recommendations(): def get_recommendations():
"""Serve the Strategist-authored recommendations for this project.""" """Serve the Strategist-authored recommendations for this project."""
rec_file = confirm_dir / RECOMMENDATIONS_NAME rec_file = _active_recommendations_path(confirm_dir)
if not rec_file.exists(): if not rec_file.exists():
return jsonify({'error': 'recommendations not found'}), 404 return jsonify({'error': f'{rec_file.name} not found'}), 404
try: try:
data = _read_json_object(rec_file) rec_file, data = _read_active_recommendations(confirm_dir)
except (OSError, json.JSONDecodeError, ValueError) as exc: except (OSError, json.JSONDecodeError, ValueError) as exc:
return jsonify({'error': f'invalid recommendations.json: {exc}'}), 400 return jsonify({
'error': f'invalid current recommendation file: {exc}',
}), 400
# Report whether a result already exists (re-open after confirm). # Report whether a result already exists (re-open after confirm).
result_file = confirm_dir / RESULT_NAME result_file = confirm_dir / RESULT_NAME
if _stage_skip(
_recommendation_stage(data),
_result_stage(result_file),
):
return jsonify({
'error': _stage_skip_error(confirm_dir),
}), 409
data['_already_confirmed'] = result_file.exists() data['_already_confirmed'] = result_file.exists()
# Later stages render only downstream sections, so fold earlier confirmed # Later stages render only downstream sections, so fold earlier confirmed
# choices from result.json back in. A refresh / reopen then re-inits from # choices from result.json back in. A refresh / reopen then re-inits from
@@ -1375,9 +1458,8 @@ def create_app(
recommend.pop('template_adherence', None) recommend.pop('template_adherence', None)
data.pop('template_reuse_scope', None) data.pop('template_reuse_scope', None)
data.pop('template_adherence', None) data.pop('template_adherence', None)
# The page polls this endpoint after a stage-1 confirm until the AI # The page polls this endpoint after each confirmation until the AI
# overwrites the file with the once-authored stage-2 recommendations, so it # creates the next stage file, so it must never be cached.
# must never be served from a cache.
resp = jsonify(data) resp = jsonify(data)
resp.headers['Cache-Control'] = 'no-store' resp.headers['Cache-Control'] = 'no-store'
return resp return resp
@@ -1406,10 +1488,11 @@ def create_app(
if custom_error: if custom_error:
return jsonify({'error': custom_error}), 400 return jsonify({'error': custom_error}), 400
try: try:
current_recommendations = _read_json_object( rec_file, current_recommendations = _read_active_recommendations(
confirm_dir / RECOMMENDATIONS_NAME, confirm_dir,
) )
except (OSError, json.JSONDecodeError, ValueError): except (OSError, json.JSONDecodeError, ValueError):
rec_file = _active_recommendations_path(confirm_dir)
current_recommendations = {} current_recommendations = {}
if stage == 'stage2' or _recommendation_stage(current_recommendations) == 3: if stage == 'stage2' or _recommendation_stage(current_recommendations) == 3:
solution_error = _stage2_solution_error(result) solution_error = _stage2_solution_error(result)
@@ -1418,7 +1501,7 @@ def create_app(
_normalize_custom_selections(result) _normalize_custom_selections(result)
locked_values = _apply_locked_recommendations( locked_values = _apply_locked_recommendations(
result, result,
confirm_dir / RECOMMENDATIONS_NAME, rec_file,
result_file, result_file,
) )
result.pop('template_reuse_scope', None) result.pop('template_reuse_scope', None)
@@ -1516,7 +1599,7 @@ def main(argv: Optional[list[str]] = None) -> int:
return 2 return 2
# Step 4 cleanup: stop any lingering confirm server and exit. Independent of # Step 4 cleanup: stop any lingering confirm server and exit. Independent of
# recommendations.json (the page may never have been confirmed). # recommendation files (the page may never have been confirmed).
if args.shutdown: if args.shutdown:
return _shutdown_existing(project_path / LOCK_FILE_NAME) return _shutdown_existing(project_path / LOCK_FILE_NAME)
@@ -1525,10 +1608,12 @@ def main(argv: Optional[list[str]] = None) -> int:
if args.wait_only: if args.wait_only:
lock_file = project_path / LOCK_FILE_NAME lock_file = project_path / LOCK_FILE_NAME
if not _live_lock(lock_file): if not _live_lock(lock_file):
if not (project_path / CONFIRM_DIR_NAME / RECOMMENDATIONS_NAME).exists(): confirm_dir = project_path / CONFIRM_DIR_NAME
rec_file = _active_recommendations_path(confirm_dir)
if not rec_file.exists():
logger.error( logger.error(
'%s not found — cannot recover confirm UI before wait-only', '%s not found — cannot recover confirm UI before wait-only',
project_path / CONFIRM_DIR_NAME / RECOMMENDATIONS_NAME, rec_file,
) )
return 1 return 1
recovery_port = _preferred_recovery_port(lock_file, args.port) recovery_port = _preferred_recovery_port(lock_file, args.port)
@@ -1555,10 +1640,11 @@ def main(argv: Optional[list[str]] = None) -> int:
wait_stage, wait_stage,
) )
rec_file = project_path / CONFIRM_DIR_NAME / RECOMMENDATIONS_NAME confirm_dir = project_path / CONFIRM_DIR_NAME
rec_file = _active_recommendations_path(confirm_dir)
if not rec_file.exists(): if not rec_file.exists():
logger.error( logger.error(
'%s not found — Strategist must write recommendations.json before launch', '%s not found — Strategist must write the current recommendation stage before launch',
rec_file, rec_file,
) )
return 1 return 1
@@ -1578,6 +1664,7 @@ def main(argv: Optional[list[str]] = None) -> int:
confirm_dir = project_path / CONFIRM_DIR_NAME confirm_dir = project_path / CONFIRM_DIR_NAME
result_file = confirm_dir / RESULT_NAME result_file = confirm_dir / RESULT_NAME
expected_stage = _expected_result_stage(confirm_dir)
started_at = time.time() started_at = time.time()
try: try:
proc, port, _ = _launch_background_server( proc, port, _ = _launch_background_server(
@@ -1590,7 +1677,13 @@ def main(argv: Optional[list[str]] = None) -> int:
logger.error('%s', exc) logger.error('%s', exc)
return 1 return 1
if args.wait: if args.wait:
return _wait_for_result(result_file, proc, started_at, args.wait_timeout) return _wait_for_result(
result_file,
proc,
started_at,
args.wait_timeout,
expected_stage,
)
return 0 return 0
# Per-project mutual exclusion: refuse duplicate launches. Stale locks # Per-project mutual exclusion: refuse duplicate launches. Stale locks
@@ -17,7 +17,7 @@
stage_design: "Stage 2 · Deck direction & visual system", stage_design: "Stage 2 · Deck direction & visual system",
stage_images: "Stage 3 · Resources & production", stage_images: "Stage 3 · Resources & production",
loading: "Loading…", loading: "Loading…",
load_error: "Could not load recommendations.json. The AI must write it before launch.", load_error: "Could not load the current recommendation stage. The AI must write it before launch.",
btn_confirm: "Confirm", btn_confirm: "Confirm",
btn_confirm_contract: "Confirm contract & continue →", btn_confirm_contract: "Confirm contract & continue →",
btn_confirm_solution: "Confirm solution & continue →", btn_confirm_solution: "Confirm solution & continue →",
@@ -155,7 +155,7 @@
stage_design: "ステージ 2 · 全体方針とビジュアルシステム", stage_design: "ステージ 2 · 全体方針とビジュアルシステム",
stage_images: "ステージ 3 · リソースと制作", stage_images: "ステージ 3 · リソースと制作",
loading: "読み込み中…", loading: "読み込み中…",
load_error: "recommendations.json を読み込めませんでした。起動前にAIが書き込む必要があります。", load_error: "現在の推奨ステージを読み込めませんでした。起動前にAIが書き込む必要があります。",
btn_confirm: "確定", btn_confirm: "確定",
btn_confirm_contract: "契約内容を確定して次へ →", btn_confirm_contract: "契約内容を確定して次へ →",
btn_confirm_solution: "全体方針を確定して次へ →", btn_confirm_solution: "全体方針を確定して次へ →",
@@ -548,7 +548,7 @@
// ---- state ----------------------------------------------------------- // ---- state -----------------------------------------------------------
var CAT = null; // catalogs.json — finite option universe var CAT = null; // catalogs.json — finite option universe
var REC = null; // recommendations.json — AI picks + candidates var REC = null; // current recommendation stage — AI picks + candidates
var ICON_PREVIEWS = {}; // /api/icon-previews — real SVG samples from templates/icons var ICON_PREVIEWS = {}; // /api/icon-previews — real SVG samples from templates/icons
var STATE = {}; var STATE = {};
var REC_ALIASES = { var REC_ALIASES = {
@@ -2407,7 +2407,7 @@
// Stage of the staged confirm flow: // Stage of the staged confirm flow:
// 1 = communication contract, 2 = complete deck direction, // 1 = communication contract, 2 = complete deck direction,
// 3 = resources + production execution, // 3 = resources + production execution,
// "all" = legacy single-pass (recommendations.json carried no stage). // "all" = legacy single-pass (the recommendation payload carried no stage).
var STAGE = 1; var STAGE = 1;
function stageNumber(data) { function stageNumber(data) {
@@ -2550,8 +2550,8 @@
} }
// Stage-2 fields are (re-)read from the recommendations. At boot they come from // Stage-2 fields are (re-)read from the recommendations. At boot they come from
// whatever recommendations.json carried; after a stage-1 confirm enterStage() // the active stage file; after a stage-1 confirm enterStage() calls this again
// calls this again with the newly authored candidates. Stage-1 STATE is preserved // with the newly authored candidates. Stage-1 STATE is preserved
// across the single-session transition — this never resets the contract. // across the single-session transition — this never resets the contract.
function initStage2State() { function initStage2State() {
resetTypographySizeOverrides(); resetTypographySizeOverrides();
@@ -2809,9 +2809,9 @@
l.style.display = "block"; l.style.display = "block";
} }
// Poll session state first. It is derived from recommendations.json and // Poll session state first. It is derived from recommendation stage files
// result.json, so a recovered server can tell the existing page exactly when // and result.json, so a recovered server can tell the existing page exactly when
// the next once-authored stage is ready. // the next stage is ready.
function pollForStage(nextStage) { function pollForStage(nextStage) {
fetchJson("/api/session", "session") fetchJson("/api/session", "session")
.then(function (session) { .then(function (session) {
@@ -1,5 +1,5 @@
{ {
"_comment": "Enumerable (finite) option universe for the confirm page. Served by Flask static at /static/catalogs.json; the front-end prefers /api/catalogs when the server provides it. These fields list ALL options and the AI marks one as recommended via recommendations.json `recommend`. Open/generative fields (color, typography, generated-image style) are NOT here — the AI authors >=3 candidates each for those (creative recommendations always offer real choice). 'canvas' mirrors scripts/config.py CANVAS_FORMATS — keep in sync. User-facing catalog text supports label_zh/label_en/label_ja, desc_zh/desc_en/desc_ja, group_zh/group_en/group_ja.", "_comment": "Enumerable (finite) option universe for the confirm page. Served by Flask static at /static/catalogs.json; the front-end prefers /api/catalogs when the server provides it. These fields list ALL options and the AI marks one as recommended via the active recommendation stage's `recommend` object. Open/generative fields (color, typography, generated-image style) are NOT here — the AI authors >=3 candidates each for those (creative recommendations always offer real choice). 'canvas' mirrors scripts/config.py CANVAS_FORMATS — keep in sync. User-facing catalog text supports label_zh/label_en/label_ja, desc_zh/desc_en/desc_ja, group_zh/group_en/group_ja.",
"canvas": [ "canvas": [
{ {
"id": "ppt169", "id": "ppt169",
@@ -1,6 +1,6 @@
# Confirm UI — Strategist Confirmation Stage Page # Confirm UI — Strategist Confirmation Stage Page
> The interactive, visual surface for [`generate-pptx`](../../workflows/generate-pptx.md) Step 4. Stage 1 is an open communication brief: common purpose paths are prompt text, never checkboxes or a forced single label. Stage 2 offers **≥3 coordinated design directions** and then exposes their component values for deliberate override; color, typography, icons, and generated-image rendering are one system rather than unrelated grids. When a template workspace is active, Stage 2 also shows one editable natural-language template-application plan—never internal mode controls. Generated images inherit the selected deck colors directly—there is no image-palette control. Stage 3 contains production mechanics only. The AI writes `recommendations.json`; confirmed values accumulate into `result.json`. Final confirm saves the result and shuts the server down. The chat path mirrors the same staged semantics. > The interactive, visual surface for [`generate-pptx`](../../workflows/generate-pptx.md) Step 4. Stage 1 is an open communication brief: common purpose paths are prompt text, never checkboxes or a forced single label. Stage 2 offers **≥3 coordinated design directions** and then exposes their component values for deliberate override; color, typography, icons, and generated-image rendering are one system rather than unrelated grids. When a template workspace is active, Stage 2 also shows one editable natural-language template-application plan—never internal mode controls. Generated images inherit the selected deck colors directly—there is no image-palette control. Stage 3 contains production mechanics only. The AI uses one `recommendations.stageN.json` file per stage; the active, unconfirmed stage may be overwritten when the user requests a new recommendation. Confirmed values accumulate into `result.json`. Final confirm saves the result and shuts the server down. The chat path mirrors the same staged semantics.
## Authority and Scope ## Authority and Scope
@@ -37,7 +37,7 @@ python3 scripts/confirm_ui/server.py <project_path> --shutdown # Step 4 clean
- `--daemon` starts the Flask process in the background; add `--wait` in the main pipeline so the parent command returns only after the page writes a fresh `result.json`. The `--wait` budget defaults to **590 s** (`--wait-timeout`), kept under the typical 600 s tool ceiling — run the launch with a long tool timeout (≈600000 ms). On timeout the parent returns non-zero but the detached server keeps running, so the caller must re-check `result.json` once before the chat fallback (a slow user may confirm just after the wait returns). - `--daemon` starts the Flask process in the background; add `--wait` in the main pipeline so the parent command returns only after the page writes a fresh `result.json`. The `--wait` budget defaults to **590 s** (`--wait-timeout`), kept under the typical 600 s tool ceiling — run the launch with a long tool timeout (≈600000 ms). On timeout the parent returns non-zero but the detached server keeps running, so the caller must re-check `result.json` once before the chat fallback (a slow user may confirm just after the wait returns).
- `--wait-only` attaches to the page already running from the first `--daemon --wait` and blocks until the page writes the requested stage. If the recorded server died, it automatically restarts on the recorded/default port so polling reconnects. Use `--wait-stage stage2` for the complete-solution handoff, then the default `--wait-stage final` for Stage 3. It keys on stage alone (no mtime gate), because a user may submit before the wait command starts. - `--wait-only` attaches to the page already running from the first `--daemon --wait` and blocks until the page writes the requested stage. If the recorded server died, it automatically restarts on the recorded/default port so polling reconnects. Use `--wait-stage stage2` for the complete-solution handoff, then the default `--wait-stage final` for Stage 3. It keys on stage alone (no mtime gate), because a user may submit before the wait command starts.
- `--shutdown` stops a confirm server left running for this project and exits — **idempotent** (a no-op when nothing is running). Tries a graceful `/api/shutdown`, falls back to killing the recorded pid, then clears the lock. Generate Step 4 runs this on every path (page-confirm or chat-fallback) so the page never lingers on the shared port before live preview starts. - `--shutdown` stops a confirm server left running for this project and exits — **idempotent** (a no-op when nothing is running). Tries a graceful `/api/shutdown`, falls back to killing the recorded pid, then clears the lock. Generate Step 4 runs this on every path (page-confirm or chat-fallback) so the page never lingers on the shared port before live preview starts.
- Refuses to start unless `<project_path>/confirm_ui/recommendations.json` exists (except `--shutdown`, which needs no recommendations). - Refuses to start unless the recommendation file expected from `result.json` exists (initially `<project_path>/confirm_ui/recommendations.stage1.json`; `--shutdown` needs no recommendations).
- Per-project lock at `<project_path>/.confirm_ui.lock` — duplicate launches are refused; stale locks (dead pid) are overwritten. - Per-project lock at `<project_path>/.confirm_ui.lock` — duplicate launches are refused; stale locks (dead pid) are overwritten.
- Idle auto-shutdown after 900 s by default; `/api/shutdown` exits gracefully and releases the lock. - Idle auto-shutdown after 900 s by default; `/api/shutdown` exits gracefully and releases the lock.
- `/api/recommendations` and `/api/confirm` strip legacy `template_reuse_scope` and `template_adherence` fields. Those exporter values are never user-facing controls; an active template instead exposes the editable natural-language `template_application` field in Stage 2. - `/api/recommendations` and `/api/confirm` strip legacy `template_reuse_scope` and `template_adherence` fields. Those exporter values are never user-facing controls; an active template instead exposes the editable natural-language `template_application` field in Stage 2.
@@ -51,7 +51,7 @@ pip install flask
## Field shapes ## Field shapes
- **Enumerable + custom** — canvas / icons retain blank manual inputs; mode / visual_style instead show a mandatory AI-authored proposal in full, initially unselected and editable after selection. Selected mode / style writes literal `custom` plus its behavior sibling. - **Enumerable + custom** — canvas / icons retain blank manual inputs; mode / visual_style instead show a mandatory AI-authored proposal in full, initially unselected and editable after selection. Selected mode / style writes literal `custom` plus its behavior sibling.
- **Visual examples for hard-to-name choices** — the full-screen confirmation page loads real SVG page samples from `static/style_previews/` for `visual_style`, and renders real sample SVGs from `templates/icons` for `icons`. These thumbnails make style and icon-library choices visually comparable before the user locks them. Preview copy is fixed role text (big title / section title / body / points), not project content from `recommendations.json`, so users compare visual treatment rather than copywriting. These previews are a confirmation aid only: they do not add fields to `recommendations.json` or `result.json`, and they do not replace the later Step 6 live preview. - **Visual examples for hard-to-name choices** — the full-screen confirmation page loads real SVG page samples from `static/style_previews/` for `visual_style`, and renders real sample SVGs from `templates/icons` for `icons`. These thumbnails make style and icon-library choices visually comparable before the user locks them. Preview copy is fixed role text (big title / section title / body / points), not project content from recommendation files, so users compare visual treatment rather than copywriting. These previews are a confirmation aid only: they do not add fields to recommendation stage files or `result.json`, and they do not replace the later Step 6 live preview.
- **Image usage multi-select** — image sources are selected as one or more catalog ids: `ai` = AI-generated, `web` = Web-sourced, `provided` = User-provided, `placeholder` = Placeholder, `none` = No images. `none` is exclusive. A confirmed non-`none` set is the allowed acquisition-source boundary, not a requirement to use every selected source; only explicit `image_notes` wording can require a source, asset, or page role. Recommendation and result values may be a legacy single string, but new files should use an array. When several sources are recommended, write the source ids to `recommend.image_usage` and write the actual usage strategy to `image_notes`, not a custom prose value. - **Image usage multi-select** — image sources are selected as one or more catalog ids: `ai` = AI-generated, `web` = Web-sourced, `provided` = User-provided, `placeholder` = Placeholder, `none` = No images. `none` is exclusive. A confirmed non-`none` set is the allowed acquisition-source boundary, not a requirement to use every selected source; only explicit `image_notes` wording can require a source, asset, or page role. Recommendation and result values may be a legacy single string, but new files should use an array. When several sources are recommended, write the source ids to `recommend.image_usage` and write the actual usage strategy to `image_notes`, not a custom prose value.
- **Closed enumerable** — PPT reading mode (`delivery_purpose` compatibility key), formula policy / generation mode / refine spec, plus AI source only when image usage includes `ai`. These have no Custom box; out-of-catalog values snap back to the recommended option. - **Closed enumerable** — PPT reading mode (`delivery_purpose` compatibility key), formula policy / generation mode / refine spec, plus AI source only when image usage includes `ai`. These have no Custom box; out-of-catalog values snap back to the recommended option.
- **Open prose**`audience`, `communication_intent`, `audience_outcome`, `core_message`, `delivery_context`, `artifact_afterlife`, `content_divergence`, and `page_count`. `communication_intent` may carry several purposes plus priority / sequence; common paths appear only as help text. `content_divergence` is the source-treatment axis. `page_count` may be a range here; Strategist resolves the exact §IX roster, leaving Executor no pagination latitude. - **Open prose**`audience`, `communication_intent`, `audience_outcome`, `core_message`, `delivery_context`, `artifact_afterlife`, `content_divergence`, and `page_count`. `communication_intent` may carry several purposes plus priority / sequence; common paths appear only as help text. `content_divergence` is the source-treatment axis. `page_count` may be a range here; Strategist resolves the exact §IX roster, leaving Executor no pagination latitude.
@@ -73,20 +73,20 @@ Round-trip and session files live under `<project_path>/confirm_ui/`.
### Three-stage flow ### Three-stage flow
The page runs as a **three-stage wizard in one browser session**. `recommendations.json` carries a top-level `"stage"` selector. Legacy payloads that still carry `"tier"` are accepted as read-only compatibility input, but new files must use `stage`. The page runs as a **three-stage wizard in one browser session**. Each stage has its own Strategist-authored file and top-level `"stage"` selector. The active, unconfirmed stage may be overwritten any number of times when the user asks for a better recommendation; refresh the page to load the replacement. Once the user confirms it, normal progression writes the next stage file rather than repurposing the previous one. The server derives the active filename from `result.json`.
| `recommendations.json stage` | Page renders | Button | On submit | | Recommendation file | Declared stage | Page renders | Button | On submit |
|---|---|---|---| |---|---|---|---|---|
| `"stage1"` | communication contract — audience; open `communication_intent`; audience outcome; core message / delivery context / artifact afterlife / `content_divergence` (all prose fields may be blank); canvas | **Confirm contract & continue** | writes `result.json` `{ stage: "stage1", status: "stage1-confirmed", <communication contract> }`; the page stays open and polls | | `recommendations.stage1.json` | `"stage1"` | communication contract — audience; open `communication_intent`; audience outcome; core message / delivery context / artifact afterlife / `content_divergence` (all prose fields may be blank); canvas | **Confirm contract & continue** | writes `result.json` `{ stage: "stage1", status: "stage1-confirmed", <communication contract> }`; the page stays open and polls |
| `"stage2"` | complete deck solution — conditional natural-language template application, reading mode, mode, page count, visual direction, color, icons, typography, image usage, generated-image rendering | **Confirm solution & continue** | writes `result.json` `{ stage: "stage2", status: "stage2-confirmed", <contract + solution> }`; the page stays open and polls | | `recommendations.stage2.json` | `"stage2"` | complete deck solution — conditional natural-language template application, reading mode, mode, page count, visual direction, color, icons, typography, image usage, generated-image rendering | **Confirm solution & continue** | writes `result.json` `{ stage: "stage2", status: "stage2-confirmed", <contract + solution> }`; the page stays open and polls |
| `"stage3"` | production only — confirmed image-source summary, conditional AI acquisition path, formula policy, generation mode, refine spec | **Confirm** | writes `result.json` `{ stage: "final", status: "confirmed", <all fields> }`, then shuts the page down | | `recommendations.stage3.json` | `"stage3"` | production only — confirmed image-source summary, conditional AI acquisition path, formula policy, generation mode, refine spec | **Confirm** | writes `result.json` `{ stage: "final", status: "confirmed", <all fields> }`, then shuts the page down |
| *(absent)* | legacy free-design single-pass — every section on one page | **Confirm** | single final write (`status: "confirmed"`) — backward-compatible only when no template workspace / `template_application` is active | | `recommendations.json` | stage or legacy no-stage payload | read-only compatibility when no stage-specific file exists | matching legacy behavior | preserves the former staged or single-pass behavior; new projects never create this file |
The AI launches Stage 1, authors the complete Stage-2 solution once from the user's actual contract, then authors Stage-3 production mechanics once from the confirmed solution. An edit inside the current stage never requests another recommendation. The page preserves earlier answers across transitions. `GET /api/session` is the waiting-state endpoint; `GET /api/recommendations` is `no-store`, and the server folds confirmed earlier-stage choices back into later payloads so refresh / reopen restores the user's actual values—including Stage-2 color, typography, icon, image-source, and rendering choices. The AI launches Stage 1, authors the complete Stage-2 solution once from the user's actual contract, then authors Stage-3 production mechanics once from the confirmed solution. An edit inside the current stage never requests another recommendation. The page preserves earlier answers across transitions. `GET /api/session` is the waiting-state endpoint; `GET /api/recommendations` is `no-store`, and the server folds confirmed earlier-stage choices back into later payloads so refresh / reopen restores the user's actual values—including Stage-2 color, typography, icon, image-source, and rendering choices. Once any stage-specific file exists, the server ignores legacy `recommendations.json` to prevent mixed lifecycles.
**Stage progression guard.** Stages confirm strictly in order — a staged `recommendations.json` may only run **one** stage past the last confirmed result, and `/api/confirm` accepts only the submit stage matching that file plus its required predecessor. A file that skips ahead (e.g. `"stage3"` while only Stage 1 is confirmed) is never rendered: `/api/session` keeps reporting `waiting_agent` with `stage_skip: true`, and `--wait` / `--wait-only` exit `2` if a result skips the stage being awaited. An active template does not exempt Stage 2: its Stage-2 recommendations must include `template_application.value`, and an installed workspace or that field disables the no-stage legacy single-pass path. Legacy single-pass remains available only for non-template compatibility payloads. **Stage progression guard.** Stages confirm strictly in order. `/api/confirm` accepts only the submit stage matching the active filename and its required predecessor; the declared `stage` must also match the filename. A later file that skips ahead (for example `recommendations.stage3.json` while only Stage 1 is confirmed and Stage 2 is absent) is never rendered: `/api/session` keeps reporting `waiting_agent` with `stage_skip: true`, and `--wait` / `--wait-only` exit `2` if a result skips the stage being awaited. An active template does not exempt Stage 2: its recommendations must include `template_application.value`, and an installed workspace or that field disables the no-stage legacy single-pass path. Legacy single-pass remains available only for non-template compatibility payloads.
### Input — `recommendations.json` (written by Strategist before launch) ### Input — `recommendations.stage1.json` (created by Strategist before launch)
```json ```json
{ {
@@ -119,7 +119,7 @@ All seven Stage-1 prose values may be blank and none blocks confirmation. The va
The common paths — inform / explain / persuade / decide / align / teach / report and account / mobilize / record and hand off — appear only as help text for `communication_intent`. They are not catalog ids and must not be emitted as a `primary_job` field. The common paths — inform / explain / persuade / decide / align / teach / report and account / mobilize / record and hand off — appear only as help text for `communication_intent`. They are not catalog ids and must not be emitted as a `primary_job` field.
After Stage 1 is confirmed, overwrite the file with the complete Stage-2 solution (the server folds confirmed communication fields back in when serving the page): After Stage 1 is confirmed, create `recommendations.stage2.json` with the complete solution; leave Stage 1 unchanged (the server folds confirmed communication fields back in when serving the page):
```json ```json
{ {
@@ -186,7 +186,7 @@ After Stage 1 is confirmed, overwrite the file with the complete Stage-2 solutio
The example abbreviates the required ≥3 directions. Custom mode/style candidates remain mandatory; AI usage also requires the custom image candidate. Stage 2 rejects fewer than three bundles, incomplete six-role palettes, and incomplete heading/body stacks. Legacy grids remain readable only with three complete palettes and complete typography. The example abbreviates the required ≥3 directions. Custom mode/style candidates remain mandatory; AI usage also requires the custom image candidate. Stage 2 rejects fewer than three bundles, incomplete six-role palettes, and incomplete heading/body stacks. Legacy grids remain readable only with three complete palettes and complete typography.
After Stage 2 is confirmed, overwrite it with Stage-3 production recommendations only: After Stage 2 is confirmed, create `recommendations.stage3.json` with production recommendations only; leave both earlier files unchanged:
```json ```json
{ {
@@ -260,7 +260,7 @@ The shape above is final. Selected custom values use `mode: custom` + `mode_beha
- Bespoke mode / style prose lives only in the required behavior sibling; image custom prose lives in `image_strategy.behavior`. Canvas / icons retain free-text edge cases, color / typography retain `name: "custom"`, and image usage remains a source-id array plus `image_notes`. - Bespoke mode / style prose lives only in the required behavior sibling; image custom prose lives in `image_strategy.behavior`. Canvas / icons retain free-text edge cases, color / typography retain `name: "custom"`, and image usage remains a source-id array plus `image_notes`.
- `image_ai_path` and `image_strategy` appear only with `image_usage: ai` and remain confirmed downstream. The page is default; explicit/failure chat fallback keeps identical fields. `image_ai_path` selects the Step 5 path, and [`strategist-image.md`](../../references/strategist-image.md) §2 retains the selected rendering or custom behavior as the deck-level image identity anchor; individual prompts still adapt subject, composition, and atmosphere within it. - `image_ai_path` and `image_strategy` appear only with `image_usage: ai` and remain confirmed downstream. The page is default; explicit/failure chat fallback keeps identical fields. `image_ai_path` selects the Step 5 path, and [`strategist-image.md`](../../references/strategist-image.md) §2 retains the selected rendering or custom behavior as the deck-level image identity anchor; individual prompts still adapt subject, composition, and atmosphere within it.
- After the user clicks the **final Confirm** (Stage 3, or single-pass), the page saves `result.json` and shuts the server down (auto-close). Stage-1 **Confirm contract & continue** and Stage-2 **Confirm solution & continue** keep the page open while it polls for the once-authored downstream stage. In the default flow, the first `--daemon --wait` returns on the stage-1 result, `--wait-only --wait-stage stage2` returns on the stage-2 result, and the final `--wait-only` returns on the final result; the AI reads each immediately — no extra chat confirmation is required. Chat fallback shows the same initially-unselected custom proposals. Either way, Step 4 ends with a `--shutdown` cleanup so a never-confirmed page cannot keep holding port 5050 ahead of the Step 6 live preview. - After the user clicks the **final Confirm** (Stage 3, or single-pass), the page saves `result.json` and shuts the server down (auto-close). Stage-1 **Confirm contract & continue** and Stage-2 **Confirm solution & continue** keep the page open while it polls for the downstream stage file. In the default flow, the first `--daemon --wait` returns on the stage-1 result, `--wait-only --wait-stage stage2` returns on the stage-2 result, and the final `--wait-only` returns on the final result; the AI reads each immediately — no extra chat confirmation is required. Chat fallback shows the same initially-unselected custom proposals. Either way, Step 4 ends with a `--shutdown` cleanup so a never-confirmed page cannot keep holding port 5050 ahead of the Step 6 live preview.
## Scope ## Scope
@@ -43,6 +43,8 @@ from tts_backends import (
configure_utf8_stdio() configure_utf8_stdio()
DEFAULT_EDGE_CONCURRENCY = 3
@dataclass(frozen=True) @dataclass(frozen=True)
class AudioBackend: class AudioBackend:
@@ -52,6 +54,13 @@ class AudioBackend:
voice_id: str = "" voice_id: str = ""
@dataclass(frozen=True)
class AudioJob:
note_path: Path
text: str
output_path: Path
def _load_tts_env_file() -> None: def _load_tts_env_file() -> None:
"""Load TTS-related keys from the first .env file, without overriding shell env.""" """Load TTS-related keys from the first .env file, without overriding shell env."""
load_prefixed_env_file(( load_prefixed_env_file((
@@ -78,6 +87,59 @@ def spoken_text(markdown: str) -> str:
return "\n".join(lines).strip() return "\n".join(lines).strip()
def _prepare_audio_jobs(
note_files: list[Path],
output_dir: Path,
extension: str,
) -> list[AudioJob]:
"""Read non-empty per-slide notes into ordered audio jobs."""
jobs: list[AudioJob] = []
for note_path in note_files:
text = spoken_text(note_path.read_text(encoding="utf-8"))
if not text:
print(f"[skip] {note_path.name}: empty spoken text")
continue
jobs.append(AudioJob(
note_path=note_path,
text=text,
output_path=output_dir / f"{note_path.stem}{extension}",
))
return jobs
async def _generate_edge_jobs(
jobs: list[AudioJob],
subtitle_dir: Path,
*,
voice: str,
rate: str,
subtitle_max_chars: int,
concurrency: int,
) -> list[BaseException | None]:
"""Generate ordered Edge jobs with bounded slide-level concurrency."""
semaphore = asyncio.Semaphore(concurrency)
async def generate_job(job: AudioJob) -> None:
async with semaphore:
await backend_edge.generate(
job.text,
job.output_path,
voice=voice,
rate=rate,
subtitle_path=subtitle_dir / f"{job.note_path.stem}.srt",
subtitle_max_chars=subtitle_max_chars,
)
raw_results = await asyncio.gather(
*(generate_job(job) for job in jobs),
return_exceptions=True,
)
return [
result if isinstance(result, BaseException) else None
for result in raw_results
]
def main() -> int: def main() -> int:
_load_tts_env_file() _load_tts_env_file()
@@ -108,6 +170,12 @@ def main() -> int:
default="+0%", default="+0%",
help='edge-tts speaking rate, e.g. "+0%%", "-10%%", "+15%%" (default: +0%%). Ignored by cloud providers.', help='edge-tts speaking rate, e.g. "+0%%", "-10%%", "+15%%" (default: +0%%). Ignored by cloud providers.',
) )
parser.add_argument(
"--concurrency",
type=int,
default=DEFAULT_EDGE_CONCURRENCY,
help="maximum concurrent Edge slide requests (default: 3; ignored by cloud providers)",
)
parser.add_argument( parser.add_argument(
"--subtitle-max-chars", "--subtitle-max-chars",
type=int, type=int,
@@ -256,6 +324,10 @@ def main() -> int:
parser.error("--subtitle-max-chars must be at least 1") parser.error("--subtitle-max-chars must be at least 1")
raise AssertionError("unreachable") raise AssertionError("unreachable")
if args.concurrency < 1:
parser.error("--concurrency must be at least 1")
raise AssertionError("unreachable")
if args.provider == "elevenlabs": if args.provider == "elevenlabs":
if not voice_id: if not voice_id:
parser.error("--voice-id is required for --provider elevenlabs") parser.error("--voice-id is required for --provider elevenlabs")
@@ -314,91 +386,109 @@ def main() -> int:
print(f"error: no per-slide notes found in {notes_dir}", file=sys.stderr) print(f"error: no per-slide notes found in {notes_dir}", file=sys.stderr)
return 2 return 2
jobs = _prepare_audio_jobs(note_files, output_dir, backend.extension)
generated = 0 generated = 0
for note_path in note_files: if backend.provider == "edge":
text = spoken_text(note_path.read_text(encoding="utf-8")) print(
if not text: f"[Edge] Generating {len(jobs)} audio/SRT pair(s) "
print(f"[skip] {note_path.name}: empty spoken text") f"with concurrency={args.concurrency}"
continue )
output_path = output_dir / f"{note_path.stem}{backend.extension}"
try: try:
if backend.provider == "elevenlabs": results = asyncio.run(_generate_edge_jobs(
backend_elevenlabs.generate( jobs,
text, subtitle_dir,
output_path, voice=args.voice,
api_key=backend.api_key, rate=args.rate,
voice_id=backend.voice_id, subtitle_max_chars=args.subtitle_max_chars,
model=args.elevenlabs_model, concurrency=args.concurrency,
output_format=args.elevenlabs_output_format, ))
stability=args.elevenlabs_stability, except Exception as exc:
similarity_boost=args.elevenlabs_similarity_boost, print(f"error: Edge audio generation failed: {exc}", file=sys.stderr)
style=args.elevenlabs_style, return 1
speaker_boost=args.elevenlabs_speaker_boost,
failed = False
for job, result in zip(jobs, results):
subtitle_path = subtitle_dir / f"{job.note_path.stem}.srt"
if result is not None:
print(
f"error: failed to generate {job.output_path}: {result}",
file=sys.stderr,
) )
elif backend.provider == "minimax": failed = True
backend_minimax.generate( continue
text, generated += 1
output_path, print(f"[OK] {job.output_path}")
api_key=backend.api_key, print(f" {subtitle_path}")
voice_id=backend.voice_id, if failed:
model=args.minimax_model, return 1
audio_format=args.minimax_output_format, else:
sample_rate=args.minimax_sample_rate, for job in jobs:
bitrate=args.minimax_bitrate, output_path = job.output_path
channel=args.minimax_channel, text = job.text
speed=args.minimax_speed, try:
volume=args.minimax_volume, if backend.provider == "elevenlabs":
pitch=args.minimax_pitch, backend_elevenlabs.generate(
language_boost=args.minimax_language_boost,
base_url=args.minimax_base_url,
)
elif backend.provider == "qwen":
backend_qwen.generate(
text,
output_path,
api_key=backend.api_key,
voice_id=backend.voice_id,
model=args.qwen_model,
language_type=args.qwen_language_type,
instructions=args.qwen_instructions,
optimize_instructions=args.qwen_optimize_instructions,
base_url=args.qwen_base_url,
)
elif backend.provider == "cosyvoice":
backend_cosyvoice.generate(
text,
output_path,
api_key=backend.api_key,
voice_id=backend.voice_id,
model=args.cosyvoice_model,
audio_format=args.cosyvoice_output_format,
sample_rate=args.cosyvoice_sample_rate,
volume=args.cosyvoice_volume,
rate=args.cosyvoice_rate,
pitch=args.cosyvoice_pitch,
instruction=args.cosyvoice_instruction,
language_hint=args.cosyvoice_language_hint,
base_url=args.cosyvoice_base_url,
)
else:
subtitle_path = subtitle_dir / f"{note_path.stem}.srt"
asyncio.run(
backend_edge.generate(
text, text,
output_path, output_path,
voice=args.voice, api_key=backend.api_key,
rate=args.rate, voice_id=backend.voice_id,
subtitle_path=subtitle_path, model=args.elevenlabs_model,
subtitle_max_chars=args.subtitle_max_chars, output_format=args.elevenlabs_output_format,
stability=args.elevenlabs_stability,
similarity_boost=args.elevenlabs_similarity_boost,
style=args.elevenlabs_style,
speaker_boost=args.elevenlabs_speaker_boost,
) )
) elif backend.provider == "minimax":
except Exception as exc: backend_minimax.generate(
print(f"error: failed to generate {output_path}: {exc}", file=sys.stderr) text,
return 1 output_path,
generated += 1 api_key=backend.api_key,
print(f"[OK] {output_path}") voice_id=backend.voice_id,
if backend.provider == "edge": model=args.minimax_model,
print(f" {subtitle_path}") audio_format=args.minimax_output_format,
sample_rate=args.minimax_sample_rate,
bitrate=args.minimax_bitrate,
channel=args.minimax_channel,
speed=args.minimax_speed,
volume=args.minimax_volume,
pitch=args.minimax_pitch,
language_boost=args.minimax_language_boost,
base_url=args.minimax_base_url,
)
elif backend.provider == "qwen":
backend_qwen.generate(
text,
output_path,
api_key=backend.api_key,
voice_id=backend.voice_id,
model=args.qwen_model,
language_type=args.qwen_language_type,
instructions=args.qwen_instructions,
optimize_instructions=args.qwen_optimize_instructions,
base_url=args.qwen_base_url,
)
elif backend.provider == "cosyvoice":
backend_cosyvoice.generate(
text,
output_path,
api_key=backend.api_key,
voice_id=backend.voice_id,
model=args.cosyvoice_model,
audio_format=args.cosyvoice_output_format,
sample_rate=args.cosyvoice_sample_rate,
volume=args.cosyvoice_volume,
rate=args.cosyvoice_rate,
pitch=args.cosyvoice_pitch,
instruction=args.cosyvoice_instruction,
language_hint=args.cosyvoice_language_hint,
base_url=args.cosyvoice_base_url,
)
except Exception as exc:
print(f"error: failed to generate {output_path}: {exc}", file=sys.stderr)
return 1
generated += 1
print(f"[OK] {output_path}")
if backend.provider == "edge": if backend.provider == "edge":
print( print(
@@ -3369,7 +3369,8 @@ class SVGQualityChecker:
), ),
outer=boundary, outer=boundary,
repair=( repair=(
'reflow the text or revise the root module bounds' 'expand the root module bounds into available '
'non-overlapping space; otherwise reflow the text'
), ),
) )
@@ -1946,15 +1946,14 @@ _TEXTBOX_PADDING_MIN_PX = 0.5
_TEXTBOX_PADDING_MAX_PX = 2.0 _TEXTBOX_PADDING_MAX_PX = 2.0
_TEXTBOX_PADDING_RATIO = 0.04 _TEXTBOX_PADDING_RATIO = 0.04
# Single-line auto-fit headroom interpolates between a low-caps base and an # Single-line auto-fit headroom interpolates between a low-caps base and an
# all-caps ceiling by the fraction of cased letters that are uppercase. The # all-caps ceiling for each run. The crude per-char width estimate undercounts
# crude per-char width estimate undercounts capitals most, so all-caps lines # capitals most, so all-caps runs need the ceiling to keep wrap-ignoring
# need the ceiling to keep wrap-ignoring renderers (LibreOffice) from folding; # renderers (LibreOffice) from folding. Applying headroom per run also prevents
# mixed-case titles only need the base, so they no longer inherit the worst- # a short serif label from forcing a conservative serif multiplier onto an
# case width. Values are calibrated against LibreOffice renders of all-caps # otherwise sans-serif line. Values are calibrated against LibreOffice renders
# bold lines (the case the per-char estimate undercounts most) with bases left # of all-caps bold lines, with bases left above mixed-case and CJK render
# above the mixed-case and CJK render ratios; exact ratios shift with the # ratios; exact ratios shift with font substitution, so these carry deliberate
# renderer's font substitution, so these carry deliberate margin rather than # margin rather than tracking one environment's numbers.
# tracking one environment's numbers.
_TEXT_WIDTH_HEADROOM_BASE = 1.06 _TEXT_WIDTH_HEADROOM_BASE = 1.06
_TEXT_WIDTH_HEADROOM_CAPS = 1.12 _TEXT_WIDTH_HEADROOM_CAPS = 1.12
_SERIF_TEXT_WIDTH_HEADROOM_BASE = 1.12 _SERIF_TEXT_WIDTH_HEADROOM_BASE = 1.12
@@ -2081,21 +2080,28 @@ def _estimate_text_runs_width(
``include_headroom`` is useful for single-line auto-fit boxes where a ``include_headroom`` is useful for single-line auto-fit boxes where a
renderer that measures text slightly wider would otherwise wrap. The renderer that measures text slightly wider would otherwise wrap. The
headroom scales with the line's uppercase fraction: all-caps lines (whose headroom scales independently with each run's family and uppercase
width the per-char estimate undercounts most) get the full ceiling, while fraction. This keeps mixed-font lines from inheriting the most conservative
mixed-case titles take a small base instead of inheriting the worst case. run's multiplier. Paragraph boxes use this value as a wrapping constraint,
Paragraph boxes use this value as a wrapping constraint, so adding headroom so adding headroom there stretches the merged text frame beyond the
there stretches the merged text frame beyond the author's source line width. author's source line width.
""" """
width = sum(_estimate_run_text_width(run) for run in runs)
if not include_headroom: if not include_headroom:
return width return sum(_estimate_run_text_width(run) for run in runs)
caps = _uppercase_fraction(runs)
if any(_is_serif_run(run) for run in runs): width = 0.0
base, ceiling = _SERIF_TEXT_WIDTH_HEADROOM_BASE, _SERIF_TEXT_WIDTH_HEADROOM_CAPS for run in runs:
else: if _is_serif_run(run):
base, ceiling = _TEXT_WIDTH_HEADROOM_BASE, _TEXT_WIDTH_HEADROOM_CAPS base = _SERIF_TEXT_WIDTH_HEADROOM_BASE
return width * (base + (ceiling - base) * caps) ceiling = _SERIF_TEXT_WIDTH_HEADROOM_CAPS
else:
base = _TEXT_WIDTH_HEADROOM_BASE
ceiling = _TEXT_WIDTH_HEADROOM_CAPS
caps = _uppercase_fraction([run])
width += _estimate_run_text_width(run) * (
base + (ceiling - base) * caps
)
return width
def estimate_single_line_text_frame_width( def estimate_single_line_text_frame_width(
@@ -176,25 +176,27 @@ The core first chooses the proposed Stage 2 source ids. Load the image module be
**Confirmation ownership and surface**: Only the user confirms. Default Stage 1 is `--daemon --wait`; use chat only by explicit chat-only/delegation or after launch failure/timeout plus a `result.json` re-check. Chat tools do not replace launch. The agent may write recommendations, operate the server, and read state, but MUST NOT call `/api/confirm`, automate submission, synthesize a payload, or write/replace `result.json`. Delegation applies only to this run: show the complete three-stage summary and never fabricate UI results. Silence confirms nothing. **Confirmation ownership and surface**: Only the user confirms. Default Stage 1 is `--daemon --wait`; use chat only by explicit chat-only/delegation or after launch failure/timeout plus a `result.json` re-check. Chat tools do not replace launch. The agent may write recommendations, operate the server, and read state, but MUST NOT call `/api/confirm`, automate submission, synthesize a payload, or write/replace `result.json`. Delegation applies only to this run: show the complete three-stage summary and never fabricate UI results. Silence confirms nothing.
| Stage | Strategist writes | Completion evidence | | Stage file (the active unconfirmed stage may be overwritten) | Strategist writes | Completion evidence |
|---|---|---| |---|---|---|
| `stage1` | Communication contract, `content_divergence`, and canvas only | `status: stage1-confirmed` | | `confirm_ui/recommendations.stage1.json` | Communication contract, `content_divergence`, and canvas only | `status: stage1-confirmed` |
| `stage2` | Complete deck solution from the confirmed contract; never skip for a template | `status: stage2-confirmed` | | `confirm_ui/recommendations.stage2.json` | Complete deck solution from the confirmed contract; never skip for a template | `status: stage2-confirmed` |
| `stage3` | Production mechanics only: conditional AI path, formula policy, generation mode, refine-spec | `stage: final`, `status: confirmed` | | `confirm_ui/recommendations.stage3.json` | Production mechanics only: conditional AI path, formula policy, generation mode, refine-spec | `stage: final`, `status: confirmed` |
1. Write Stage 1 `confirm_ui/recommendations.json` per the Confirm UI contract, then launch and wait: If the user rejects the current recommendation before confirming it, regenerate by overwriting that same stage file and have the page refresh; do not create revision-suffixed files. This never authorizes one stage file to carry another stage's payload.
1. Create `confirm_ui/recommendations.stage1.json` per the Confirm UI contract, then launch and wait:
```bash ```bash
python3 ${SKILL_DIR}/scripts/confirm_ui/server.py <project_path> --daemon --wait python3 ${SKILL_DIR}/scripts/confirm_ui/server.py <project_path> --daemon --wait
``` ```
2. Read the Stage 1 result. Derive proposed image sources in core and load `strategist-image.md` before constructing Stage 2 when its trigger fires; apply `strategist-template.md` when active. Overwrite the recommendations with Stage 2, then wait: 2. Read the Stage 1 result. Derive proposed image sources in core and load `strategist-image.md` before constructing Stage 2 when its trigger fires; apply `strategist-template.md` when active. Create `confirm_ui/recommendations.stage2.json` without changing Stage 1, then wait:
```bash ```bash
python3 ${SKILL_DIR}/scripts/confirm_ui/server.py <project_path> --wait-only --wait-stage stage2 python3 ${SKILL_DIR}/scripts/confirm_ui/server.py <project_path> --wait-only --wait-stage stage2
``` ```
3. Read the Stage 2 result, overwrite recommendations with Stage 3, then perform the final blocking wait: 3. Read the Stage 2 result, create `confirm_ui/recommendations.stage3.json` without changing either earlier stage, then perform the final blocking wait:
```bash ```bash
python3 ${SKILL_DIR}/scripts/confirm_ui/server.py <project_path> --wait-only python3 ${SKILL_DIR}/scripts/confirm_ui/server.py <project_path> --wait-only
@@ -16,7 +16,7 @@ Global stop/continue rules for all four top-level routes, plus concrete failure
|---|---:|---|---|---| |---|---:|---|---|---|
| Confirm UI launch failure | No | Re-check `confirm_ui/result.json` once, then use chat fallback | No | [`generate-pptx`](../generate-pptx.md) Step 4 chat confirmation | | Confirm UI launch failure | No | Re-check `confirm_ui/result.json` once, then use chat fallback | No | [`generate-pptx`](../generate-pptx.md) Step 4 chat confirmation |
| Confirm UI wait timeout | No, if no final result yet | Re-check `result.json` once; keep server cleanup mandatory | Only if user still wants the page | Step 4 same stage or chat fallback | | Confirm UI wait timeout | No, if no final result yet | Re-check `result.json` once; keep server cleanup mandatory | Only if user still wants the page | Step 4 same stage or chat fallback |
| Confirm UI Stage 1 completed then interrupted | Yes until Stage 2 is written/confirmed | Read existing Stage 1 `result.json`, write Stage 2 recommendations, then `--wait-only --wait-stage stage2` | Usually no | Step 4 Stage 2 write/wait | | Confirm UI Stage 1 completed then interrupted | Yes until Stage 2 is written/confirmed | Read existing Stage 1 `result.json`, create `recommendations.stage2.json` without changing Stage 1, then `--wait-only --wait-stage stage2` | Usually no | Step 4 Stage 2 write/wait |
| Missing final confirmation | Yes | None | User must confirm or change the values | Step 4 final confirmation | | Missing final confirmation | Yes | None | User must confirm or change the values | Step 4 final confirmation |
| Final confirmed value is missing, changed, substituted, or weakened in `design_spec.md` | Yes | Repair from the retained final-confirmation object; only a fresh recovery turn with no retained state reads persisted final evidence once | Only when the confirmed value genuinely cannot be honored | Step 4 Gate 1 — confirmation fidelity | | Final confirmed value is missing, changed, substituted, or weakened in `design_spec.md` | Yes | Repair from the retained final-confirmation object; only a fresh recovery turn with no retained state reads persisted final evidence once | Only when the confirmed value genuinely cannot be honored | Step 4 Gate 1 — confirmation fidelity |
| `spec_lock.md` changes confirmed identity or omits a required execution anchor/routing decision | Yes | Re-author the affected lock rows from the completed Design Spec and current context; do not enumerate page-local literals | No unless the Design Spec itself is incomplete | Step 4 Gate 2 — lock context fidelity | | `spec_lock.md` changes confirmed identity or omits a required execution anchor/routing decision | Yes | Re-author the affected lock rows from the completed Design Spec and current context; do not enumerate page-local literals | No unless the Design Spec itself is incomplete | Step 4 Gate 2 — lock context fidelity |
@@ -67,8 +67,8 @@ Here, **final confirmation evidence** means either the explicit final confirmati
| Last good state | Resume from | | Last good state | Resume from |
|---|---| |---|---|
| Stage 1 confirmation exists, Stage 2 missing | Write Stage 2 recommendations, then `confirm_ui/server.py <project> --wait-only --wait-stage stage2` | | Stage 1 confirmation exists, Stage 2 missing | Create `recommendations.stage2.json` without changing Stage 1, then `confirm_ui/server.py <project> --wait-only --wait-stage stage2` |
| Stage 2 confirmation exists, final confirmation missing | Resume [`generate-pptx`](../generate-pptx.md) Step 4 confirmation orchestration at Stage 3: derive production mechanics from the confirmed solution, then perform the final wait. | | Stage 2 confirmation exists, final confirmation missing | Resume [`generate-pptx`](../generate-pptx.md) Step 4 confirmation orchestration at Stage 3: derive production mechanics from the confirmed solution, create `recommendations.stage3.json` without changing earlier stages, then perform the final wait. |
| Final confirmation evidence exists; `design_spec.md` is missing, with or without a surviving `spec_lock.md` | Return to Generate Step 4 and [`strategist.md`](../../references/strategist.md) §6.2; read final evidence once into the fresh context, read [`design_spec_reference.md`](../../templates/design_spec_reference.md), author the complete `design_spec.md` from scratch using that state plus source analysis, and pass Gate 1. Then read [`spec_lock_reference.md`](../../templates/spec_lock_reference.md) and re-author the complete `spec_lock.md` from the audited Design Spec plus current context, replacing any orphan lock. Never reconstruct the Design Spec from an orphan lock or retain orphan-lock choices as authority. | | Final confirmation evidence exists; `design_spec.md` is missing, with or without a surviving `spec_lock.md` | Return to Generate Step 4 and [`strategist.md`](../../references/strategist.md) §6.2; read final evidence once into the fresh context, read [`design_spec_reference.md`](../../templates/design_spec_reference.md), author the complete `design_spec.md` from scratch using that state plus source analysis, and pass Gate 1. Then read [`spec_lock_reference.md`](../../templates/spec_lock_reference.md) and re-author the complete `spec_lock.md` from the audited Design Spec plus current context, replacing any orphan lock. Never reconstruct the Design Spec from an orphan lock or retain orphan-lock choices as authority. |
| Final confirmation evidence exists; `design_spec.md` exists and `spec_lock.md` missing | Return to Generate Step 4; in this fresh recovery context read final evidence once to audit the existing Design Spec, then read [`spec_lock_reference.md`](../../templates/spec_lock_reference.md) and author the complete lock from the audited Design Spec plus current context. | | Final confirmation evidence exists; `design_spec.md` exists and `spec_lock.md` missing | Return to Generate Step 4; in this fresh recovery context read final evidence once to audit the existing Design Spec, then read [`spec_lock_reference.md`](../../templates/spec_lock_reference.md) and author the complete lock from the audited Design Spec plus current context. |
| Final confirmation evidence and both planning artifacts exist, but Gate 1 fails | In a fresh recovery context read final evidence once, repair `design_spec.md`, then re-author every affected lock row. Do not reopen recommendations or infer a replacement from the current lock. | | Final confirmation evidence and both planning artifacts exist, but Gate 1 fails | In a fresh recovery context read final evidence once, repair `design_spec.md`, then re-author every affected lock row. Do not reopen recommendations or infer a replacement from the current lock. |
@@ -165,7 +165,7 @@ This step has two halves:
**Visual re-confirm — full confirm UI seeded from the source**: **Visual re-confirm — full confirm UI seeded from the source**:
Write `<project_path>/confirm_ui/recommendations.json` and launch the same confirm server [`generate-pptx`](../generate-pptx.md) Step 4 uses. Do **not** hide fields: seed **every** targeted-confirmation field with the inherited / source-derived default so the user sees the recommendation and keeps the place to change it. Schema → [`scripts/docs/confirm_ui.md`](../../scripts/docs/confirm_ui.md). Use the three `<project_path>/confirm_ui/recommendations.stageN.json` files at the same staged handoffs as [`generate-pptx`](../generate-pptx.md) Step 4 and launch the same confirm server. The active, unconfirmed stage may be overwritten for a requested regeneration; normal progression leaves confirmed earlier stages intact. Do **not** hide fields: seed **every** targeted-confirmation field with the inherited / source-derived default so the user sees the recommendation and keeps the place to change it. Schema → [`scripts/docs/confirm_ui.md`](../../scripts/docs/confirm_ui.md).
```json ```json
{ {
@@ -174,6 +174,11 @@ python3 skills/ppt-master/scripts/video_subtitles.py <project_path> \
--video <powerpoint_exported_video> --language <language> --force --video <powerpoint_exported_video> --language <language> --force
``` ```
**Default — bounded Edge concurrency (may override)**: Generate up to three
slide-level audio/SRT pairs concurrently. Use `--concurrency <N>` to tune the
Edge path or `--concurrency 1` for serial troubleshooting. Cloud providers
remain serial.
If `notes_to_audio.py` errors with a missing dependency or missing provider API key, fix the prerequisite and re-run — do NOT swallow the error. If `notes_to_audio.py` errors with a missing dependency or missing provider API key, fix the prerequisite and re-run — do NOT swallow the error.
The edge command writes each MP3 and its internal page SRT from the same `edge-tts` stream. SRT cues use the service's `WordBoundary` timing: sentence-ending punctuation always closes a cue; text over the default 20-visible-character limit first splits at commas, semicolons, or colons, then at the nearest word boundary. Override the limit with `--subtitle-max-chars`. Adjacent timing overlap up to 100 ms is tolerated by moving the later cue start to the previous cue end; larger overlap fails instead of silently distorting timing. Each SRT uses a page-local timeline whose origin is `00:00:00,000`, including any leading silence before the first cue. Cloud-provider commands currently write audio only. The edge command writes each MP3 and its internal page SRT from the same `edge-tts` stream. SRT cues use the service's `WordBoundary` timing: sentence-ending punctuation always closes a cue; text over the default 20-visible-character limit first splits at commas, semicolons, or colons, then at the nearest word boundary. Override the limit with `--subtitle-max-chars`. Adjacent timing overlap up to 100 ms is tolerated by moving the later cue start to the previous cue end; larger overlap fails instead of silently distorting timing. Each SRT uses a page-local timeline whose origin is `00:00:00,000`, including any leading silence before the first cue. Cloud-provider commands currently write audio only.
@@ -2,8 +2,8 @@
"sourceId": "shadcn", "sourceId": "shadcn",
"repo": "https://github.com/shadcn-ui/ui.git", "repo": "https://github.com/shadcn-ui/ui.git",
"ref": "main", "ref": "main",
"commit": "6cd3f4c65c361ab6554e06a77e6a0af9cf8b6e37", "commit": "4baadbc6517070ae8f8feb2c97037adc2b305544",
"adapter": "claude-skill", "adapter": "claude-skill",
"sourcePath": "skills/shadcn", "sourcePath": "skills/shadcn",
"syncedAt": "2026-07-23T16:00:00Z" "syncedAt": "2026-07-24T06:18:20Z"
} }
@@ -1,6 +1,6 @@
{ {
"name": "superpowers-zh", "name": "superpowers-zh",
"version": "6.1.1-zh.1", "version": "6.2.0-zh.1",
"description": "Superpowers 的简体中文本地化版本,提供规划、TDD、系统化调试、代码评审和交付工作流。", "description": "Superpowers 的简体中文本地化版本,提供规划、TDD、系统化调试、代码评审和交付工作流。",
"author": { "author": {
"name": "Jesse Vincent", "name": "Jesse Vincent",
+20 -6
View File
@@ -6,12 +6,12 @@ Superpowers 是一套面向编程 Agent 的完整软件开发方法论。它由
## 版本对齐 ## 版本对齐
- **当前中文发行版:** `v6.1.1-zh.1` - **当前中文发行版:** `v6.2.0-zh.1`
- **对齐的上游正式版:** [`obra/superpowers v6.1.1`](https://github.com/obra/superpowers/releases/tag/v6.1.1) - **对齐的上游正式版:** [`obra/superpowers v6.2.0`](https://github.com/obra/superpowers/releases/tag/v6.2.0)
- **上游基线提交:** [`d884ae0`](https://github.com/obra/superpowers/commit/d884ae04edebef577e82ff7c4e143debd0bbec99) - **上游基线提交:** [`3dcbd5c4b48e02263fbf4a3c01e3fe4f81d584d9`](https://github.com/obra/superpowers/commit/3dcbd5c4b48e02263fbf4a3c01e3fe4f81d584d9)
- **对齐日期:** 2026-07-13 - **对齐日期:** 2026-07-24
版本号中的 `zh.1` 表示:功能基线与上游 `v6.1.1` 对齐,这是该基线上的第 1 个中文发行版。后续同步新的上游版本时,会先更新前三段版本号,再从 `zh.1` 重新开始计数。 版本号中的 `zh.1` 表示:功能基线与上游 `v6.2.0` 对齐,这是该基线上的第 1 个中文发行版。后续同步新的上游版本时,会先更新前三段版本号,再从 `zh.1` 重新开始计数。
## 上游正在招聘 ## 上游正在招聘
@@ -23,7 +23,7 @@ Superpowers 上游团队正在招聘一名全职工程师,协助社区运营
## 快速开始 ## 快速开始
为你的编程 Agent 安装 Superpowers 中文版:[Claude Code](#claude-code)、[Antigravity](#antigravity)、[Codex App](#codex-app)、[Codex CLI](#codex-cli)、[Cursor](#cursor)、[Factory Droid](#factory-droid)、[GitHub Copilot CLI](#github-copilot-cli)、[Kimi Code](#kimi-code)、[OpenCode](#opencode)、[Pi](#pi)。 为你的编程 Agent 安装 Superpowers 中文版:[Claude Code](#claude-code)、[Antigravity](#antigravity)、[Codex App](#codex-app)、[Codex CLI](#codex-cli)、[Cursor](#cursor)、[Factory Droid](#factory-droid)、[Gemini CLI](#gemini-cli)、[GitHub Copilot CLI](#github-copilot-cli)、[Kimi Code](#kimi-code)、[OpenCode](#opencode)、[Pi](#pi)。
## 工作原理 ## 工作原理
@@ -142,6 +142,20 @@ droid plugin marketplace add https://github.com/AreChen/superpowers-zh
droid plugin install superpowers@superpowers-dev droid plugin install superpowers@superpowers-dev
``` ```
### Gemini CLI
安装本中文仓库提供的扩展:
```bash
gemini extensions install https://github.com/AreChen/superpowers-zh
```
后续更新:
```bash
gemini extensions update superpowers
```
### GitHub Copilot CLI ### GitHub Copilot CLI
注册本中文仓库的插件市场: 注册本中文仓库的插件市场:
@@ -2,8 +2,8 @@
"sourceId": "superpowers-zh", "sourceId": "superpowers-zh",
"repo": "https://github.com/AreChen/superpowers-zh.git", "repo": "https://github.com/AreChen/superpowers-zh.git",
"ref": "main", "ref": "main",
"commit": "c51f23adcd482fd908aa60928f2ece34d12f7768", "commit": "e1f21a28e5a32b94d35fdaaa98e298ef73260545",
"adapter": "codex-plugin", "adapter": "codex-plugin",
"sourcePath": ".", "sourcePath": ".",
"syncedAt": "2026-07-14T02:27:30Z" "syncedAt": "2026-07-24T06:18:20Z"
} }
@@ -76,6 +76,7 @@ digraph brainstorming {
- 提出 2-3 种不同的方案,并说明各自的权衡 - 提出 2-3 种不同的方案,并说明各自的权衡
- 以对话方式呈现选项,并给出你的建议和理由 - 以对话方式呈现选项,并给出你的建议和理由
- 先介绍你推荐的选项,并解释原因 - 先介绍你推荐的选项,并解释原因
- 坚决贯彻 YAGNI——从每一种方案和设计中移除不必要的功能
**呈现设计:** **呈现设计:**
@@ -127,15 +128,6 @@ digraph brainstorming {
- 调用 writing-plans 技能来创建详细的实施计划 - 调用 writing-plans 技能来创建详细的实施计划
- 不要调用任何其他技能。writing-plans 是下一步。 - 不要调用任何其他技能。writing-plans 是下一步。
## 关键原则
- **一次只问一个问题** - 不要用多个问题让用户不知所措
- **优先使用选择题** - 条件允许时,选择题比开放式问题更容易回答
- **坚决贯彻 YAGNI** - 从所有设计中移除不必要的功能
- **探索替代方案** - 在定案前始终提出 2-3 种方案
- **增量验证** - 展示设计,获得批准后再继续
- **保持灵活** - 当某些内容不合理时,返回并加以澄清
## 可视化伴侣 ## 可视化伴侣
一个基于浏览器的伴侣工具,用于在头脑风暴期间展示原型图、图表和可视化选项。它作为工具提供——而不是一种模式。同意使用该伴侣工具,意味着它可用于那些采用可视化方式会更有帮助的问题;这并不意味着每个问题都要通过浏览器进行。 一个基于浏览器的伴侣工具,用于在头脑风暴期间展示原型图、图表和可视化选项。它作为工具提供——而不是一种模式。同意使用该伴侣工具,意味着它可用于那些采用可视化方式会更有帮助的问题;这并不意味着每个问题都要通过浏览器进行。
@@ -68,6 +68,13 @@ scripts/start-server.sh --project-dir /path/to/project --open
scripts/start-server.sh --project-dir /path/to/project --open scripts/start-server.sh --project-dir /path/to/project --open
``` ```
**Gemini CLI**
```bash
# 使用 --foreground,并在 shell 工具调用中设置 is_background: true
# 使进程能够跨轮次持续运行
scripts/start-server.sh --project-dir /path/to/project --open --foreground
```
**Copilot CLI** **Copilot CLI**
```bash ```bash
# 使用 --foreground,并通过 bash 工具以 mode: "async" 启动服务器, # 使用 --foreground,并通过 bash 工具以 mode: "async" 启动服务器,
@@ -158,15 +158,6 @@ Agent 3 → 修复 tool-approval-race-conditions.test.ts
**集成:** 所有修复彼此独立,没有冲突,完整测试套件全部通过 **集成:** 所有修复彼此独立,没有冲突,完整测试套件全部通过
**节省的时间:** 3 个问题并行解决,而非依次解决
## 主要优势
1. **并行化** - 多项调查同时进行
2. **专注** - 每个 Agent 的范围都很窄,需要跟踪的上下文更少
3. **独立性** - Agent 之间互不干扰
4. **速度** - 用解决 1 个问题的时间解决了 3 个问题
## 验证 ## 验证
Agent 返回后: Agent 返回后:
@@ -174,12 +165,3 @@ Agent 返回后:
2. **检查冲突** - Agent 是否编辑了相同的代码? 2. **检查冲突** - Agent 是否编辑了相同的代码?
3. **运行完整测试套件** - 验证所有修复能否协同工作 3. **运行完整测试套件** - 验证所有修复能否协同工作
4. **抽查** - Agent 可能会犯系统性错误 4. **抽查** - Agent 可能会犯系统性错误
## 实际影响
来自调试会话(2025-10-03):
- 3 个文件中共有 6 个失败
- 并行派遣了 3 个 Agent
- 所有调查均并发完成
- 所有修复均成功集成
- Agent 变更之间零冲突
@@ -11,15 +11,16 @@ description: 当你有一份书面实施计划,需要在单独的会话中执
**开始时宣布:**“我正在使用 executing-plans 技能来实施此计划。” **开始时宣布:**“我正在使用 executing-plans 技能来实施此计划。”
**注意:**告诉你的人类伙伴,在能够使用子 Agent 的情况下,Superpowers 的运行效果会好得多。如果在支持子 Agent 的平台上运行,其工作质量将显著提高Claude Code、Codex CLI、Codex AppCopilot CLI 都符合要求;请参阅 `../using-superpowers/references/` 中各平台对应的工具参考文档)。如果子 Agent 可用,请使用 superpowers:subagent-driven-development,而不是此技能。 **注意:**告诉你的人类伙伴,在能够使用子 Agent 的情况下,Superpowers 的运行效果会好得多(Claude Code、Codex CLI、Codex AppCopilot CLI 和 Gemini CLI 都符合要求;请参阅 `../using-superpowers/references/` 中各平台对应的工具参考文档)。如果子 Agent 可用,请使用 superpowers:subagent-driven-development,而不是此技能。
## 流程 ## 流程
### 第 1 步:加载并审查计划 ### 第 1 步:加载并审查计划
1. 阅读计划文件 1. 确保使用隔离工作区:使用 superpowers:using-git-worktrees 创建一个,或验证现有工作区
2. 严格审查——找出对计划存在的任何疑问或担忧 2. 阅读计划文件
3. 如果有疑虑:在开始之前向你的人类伙伴提出 3. 严格审查——找出对计划存在的任何疑问或担忧
4. 如果有疑虑:为计划项创建待办事项,然后继续 4. 如果有疑虑:在开始之前向你的人类伙伴提出
5. 如果没有疑虑:为计划项创建待办事项,然后继续
### 第 2 步:执行任务 ### 第 2 步:执行任务
@@ -61,10 +62,3 @@ description: 当你有一份书面实施计划,需要在单独的会话中执
- 当计划要求引用技能时,引用相应技能 - 当计划要求引用技能时,引用相应技能
- 遇到阻碍时停止,不要猜测 - 遇到阻碍时停止,不要猜测
- 未经用户明确同意,绝不要在 main/master 分支上开始实施 - 未经用户明确同意,绝不要在 main/master 分支上开始实施
## 集成
**必需的工作流技能:**
- **superpowers:using-git-worktrees**——确保使用隔离的工作区(创建一个工作树或验证现有工作树)
- **superpowers:writing-plans**——创建由此技能执行的计划
- **superpowers:finishing-a-development-branch**——在所有任务完成后完成开发
@@ -1,139 +1,132 @@
--- ---
name: finishing-a-development-branch name: finishing-a-development-branch
description: 在实现完成、所有测试均通过,并且你需要决定如何集成这些工作时使用——通过提供合并、PR 或清理的结构化选项,引导完成开发工作 description: 在实现完成、所有测试均通过,并且你需要决定如何集成这些工作时使用
--- ---
# 完成开发分支 # 完成开发分支
## 概述 ## 概述
通过提供清晰的选项并处理所选择的工作流,引导完成开发工作 **核心原则:** 验证测试 → 检测环境 → 呈现选项 → 执行选择 → 清理
**核心原则:** 验证测试 → 检测环境 → 提供选项 → 执行选择 → 清理。
**开始时宣布:** “我正在使用 finishing-a-development-branch 技能来完成这项工作。” **开始时宣布:** “我正在使用 finishing-a-development-branch 技能来完成这项工作。”
## 流程 ## 第 1 步:验证测试
### 第 1 步:验证测试 运行项目的完整测试套件(`npm test` / `cargo test` / `pytest` / `go test ./...`)。
**在提供选项之前,验证测试是否通过:** **如果测试失败,**报告失败并停止——只有测试套件全绿后才能显示菜单:
```bash
# 运行项目的测试套件
npm test / cargo test / pytest / go test ./...
```
**如果测试失败:**
``` ```
测试失败(<N> 个失败)。必须先修复才能完成: 测试失败(<N> 个失败)。必须先修复才能完成:
[显示失败信息]
在测试通过之前,无法继续合并/PR。 [显示失败信息]
``` ```
停止。不要继续执行步骤 2 **如果测试通过:** 继续执行第 2 步
**如果测试通过:** 继续执行步骤 2。 ## 第 2 步:检测环境
### 步骤 2:检测环境
**在展示选项之前确定工作区状态:**
```bash ```bash
GIT_DIR=$(cd "$(git rev-parse --git-dir)" 2>/dev/null && pwd -P) GIT_DIR=$(cd "$(git rev-parse --git-dir)" 2>/dev/null && pwd -P)
GIT_COMMON=$(cd "$(git rev-parse --git-common-dir)" 2>/dev/null && pwd -P) GIT_COMMON=$(cd "$(git rev-parse --git-common-dir)" 2>/dev/null && pwd -P)
# Capture now, while still inside the workspace — Step 5 changes directory
# before cleanup (Step 6) needs this value
WORKTREE_PATH=$(git rev-parse --show-toplevel)
``` ```
这决定了要显示哪个菜单以及如何进行清理: 这决定了要显示哪个菜单以及如何清理:
| 状态 | 菜单 | 清理 | | 状态 | 菜单 | 清理 |
|-------|------|---------| |-------|------|------|
| `GIT_DIR == GIT_COMMON`(普通仓库) | 标准的 4 个选项 | 没有要清理的工作树 | | `GIT_DIR == GIT_COMMON`(普通仓库) | 标准 3 个选项 | 没有要清理的工作树 |
| `GIT_DIR != GIT_COMMON`,命名分支 | 标准的 4 个选项 | 基于来源(见步骤 6 | | `GIT_DIR != GIT_COMMON`,命名分支 | 标准 3 个选项 | 按来源清理(见第 6 步 |
| `GIT_DIR != GIT_COMMON`分离 HEAD | 精简的 3 个选项(无合并) | 不清理(由外部管理 | | `GIT_DIR != GIT_COMMON`detached HEAD | 精简 2 个选项(无合并) | 由外部管理——保留原位 |
### 步骤 3:确定基础分支 ## 第 3 步:确定基础分支
```bash 基础分支就是当前工作从中分出的分支——通常在计划、对话或该分支的 upstream 中
# 尝试常见的基础分支 已经指明。如果还不知道,请询问:“这个分支是从 <你最合理的猜测> 分出来的——
git merge-base HEAD main 2>/dev/null || git merge-base HEAD master 2>/dev/null 对吗?”合并前必须确认;合并到错误基础分支的撤销成本很高。
```
或者询问:“这个分支是从 main 分出来的——对吗?” ## 第 4 步:呈现选项
### 步骤 4:呈现选项 **普通仓库和命名分支工作树——必须原样呈现以下 3 个选项:**
**普通仓库和命名分支工作树——必须原样呈现以下 4 个选项:**
``` ```
实现已完成。想怎么 实现已完成。想怎么处理
1. 在本地合并回 <base-branch> 1. 在本地合并回 <base-branch>
2. 推送并创建拉取请求 2. 推送并创建 Pull Request
3. 保持分支原样(我稍后处理) 3. 保持分支原样(我稍后处理)
4. 丢弃此工作
请选择哪个选项? 请选择哪个选项?
``` ```
**detached HEAD——必须原样呈现以下 3 个选项:** **Detached HEAD——必须原样呈现以下 2 个选项:**
``` ```
实现已完成。当前处于 detached HEAD(外部管理的工作区)。 实现已完成。当前处于 detached HEAD外部管理的工作区)。
1. 作为新分支推送并创建拉取请求 1. 作为新分支推送并创建 Pull Request
2. 保持原样(我稍后处理) 2. 保持原样(我稍后处理)
3. 丢弃此工作
请选择哪个选项? 请选择哪个选项?
``` ```
**不要添加解释**——保持选项简洁。 严格按上述形式呈现菜单——保持简洁,每个选项都必须来自上面的列表。只有当你的人类
伙伴明确要求丢弃工作时,才进入丢弃流程(见下文“如果你的人类伙伴要求丢弃工作”)。
等待对方回答;集成决定属于他们。
### 步骤 5:执行选择 ## 第 5 步:执行选择
#### 选项 1:在本地合并 ### 选项 1:在本地合并
```bash ```bash
# 获取主仓库根目录,以确保 CWD 安全 # Get main repo root for CWD safety
MAIN_ROOT=$(git -C "$(git rev-parse --git-common-dir)/.." rev-parse --show-toplevel) MAIN_ROOT=$(git -C "$(git rev-parse --git-common-dir)/.." rev-parse --show-toplevel)
cd "$MAIN_ROOT" cd "$MAIN_ROOT"
# 先合并——在移除任何内容之前验证是否成功 # Merge first — verify success before removing anything
git checkout <base-branch> git checkout <base-branch>
git pull git pull
git merge <feature-branch> git merge <feature-branch>
# 验证合并结果上的测试 # Verify tests on merged result
<test command> <test command>
# 仅在合并成功后:清理工作树(步骤 6),然后删除分支
``` ```
然后:清理工作树(步骤 6),然后删除分支: 如果合并结果上的测试失败:停止,保留工作树和分支并调查——尚未推送任何内容,
因此本地合并仍可恢复。
合并结果全绿后:清理工作树(第 6 步),再删除分支:
```bash ```bash
git branch -d <feature-branch> git branch -d <feature-branch>
``` ```
#### 选项 2:推送并创建 PR
### 选项 2:推送并创建 PR
```bash ```bash
# 推送分支
git push -u origin <feature-branch> git push -u origin <feature-branch>
# From a detached HEAD, name the new branch on the remote:
# git push origin HEAD:refs/heads/<new-branch>
``` ```
**不要清理工作树**——用户需要保留它,以便根据 PR 反馈进行迭代。 然后使用代码托管平台的工具,以 <base-branch> 为目标创建 Pull/Merge Request——
如果有 CLI 就使用 CLI,否则使用多数平台在推送时打印的创建 URL;若仓库存在 PR
模板和约定,则遵循它们,并将 URL 报告给你的人类伙伴。
#### 选项 3:保持原样 保留工作树——你的人类伙伴会在其中处理 PR 反馈。
报告:"保留分支 <name>。工作树保留在 <path>。" ### 选项 3:保持原样
**不要清理工作树。** 报告:“保留分支 <name>。工作树保留在 <path>。”
#### 选项 4:丢弃 ### 如果你的人类伙伴要求丢弃工作
此路径仅在对方明确要求丢弃工作时存在。请先确认:
**请先确认:**
``` ```
这将永久删除: 这将永久删除:
- 分支 <name> - 分支 <name>
@@ -143,40 +136,37 @@ git push -u origin <feature-branch>
输入 'discard' 以确认。 输入 'discard' 以确认。
``` ```
等待完全一致的确认。 等待完全一致的确认。收到后:
如果已确认:
```bash ```bash
MAIN_ROOT=$(git -C "$(git rev-parse --git-common-dir)/.." rev-parse --show-toplevel) MAIN_ROOT=$(git -C "$(git rev-parse --git-common-dir)/.." rev-parse --show-toplevel)
cd "$MAIN_ROOT" cd "$MAIN_ROOT"
``` ```
然后清理工作树(步骤 6),强制删除分支: 然后清理工作树(第 6 步),强制删除分支:
```bash ```bash
git branch -D <feature-branch> git branch -D <feature-branch>
``` ```
### 步骤 6:清理工作区 ## 第 6 步:清理工作区
**仅对选项 1 和 4 运行。** 选项 2 和 3 始终保留工作树。 **仅对选项 1 和已确认的丢弃操作运行。** 选项 2 和 3 始终保留工作树。两个调用方
都已经切换到主仓库根目录——移除工作树必须在工作树外部运行——并且必须使用第 2 步
中切换目录前捕获的 `GIT_DIR`/`GIT_COMMON`/`WORKTREE_PATH` 值。
**如果 `GIT_DIR == GIT_COMMON`:** 普通仓库,没有要清理的工作树。完成。
**如果 `WORKTREE_PATH` 位于 `.worktrees/``worktrees/` 下:** 此工作树由
Superpowers 创建——清理由我们负责:
```bash ```bash
GIT_DIR=$(cd "$(git rev-parse --git-dir)" 2>/dev/null && pwd -P)
GIT_COMMON=$(cd "$(git rev-parse --git-common-dir)" 2>/dev/null && pwd -P)
WORKTREE_PATH=$(git rev-parse --show-toplevel)
```
**如果 `GIT_DIR == GIT_COMMON`:** 普通仓库,没有需要清理的工作树。完成。
**如果工作树路径位于 `.worktrees/``worktrees/` 下:** Superpowers 创建了此工作树——清理由我们负责。
```bash
MAIN_ROOT=$(git -C "$(git rev-parse --git-common-dir)/.." rev-parse --show-toplevel)
cd "$MAIN_ROOT"
git worktree remove "$WORKTREE_PATH" git worktree remove "$WORKTREE_PATH"
git worktree prune # 自修复:清理所有陈旧的注册记录 git worktree prune # Self-healing: clean up any stale registrations
``` ```
**否则:** 主机环境(运行平台)拥有此工作区。切勿移除它。如果你的平台提供 workspace-exit 工具,请使用它。否则,将工作区保留在原处。
**否则:** 此工作区属于宿主环境——保留原位。如果你的平台提供 workspace-exit
工具,请使用它。
## 快速参考 ## 快速参考
@@ -185,53 +175,18 @@ git worktree prune # 自修复:清理所有陈旧的注册记录
| 1. 本地合并 | 是 | - | - | 是 | | 1. 本地合并 | 是 | - | - | 是 |
| 2. 创建 PR | - | 是 | 是 | - | | 2. 创建 PR | - | 是 | 是 | - |
| 3. 保持原样 | - | - | 是 | - | | 3. 保持原样 | - | - | 是 | - |
| 4. 丢弃 | - | - | - | 是(强制) | | 丢弃(仅限明确请求) | - | - | - | 是(强制) |
## 常见错误 ## 常见合理化借口
**跳过测试验证** | 借口 | 现实 |
- **问题:** 合并有问题的代码,创建会失败的 PR |--------|---------|
- **修复:** 在提供选项之前始终验证测试 | “测试在本次会话早些时候通过了” | 在即将集成的代码树上运行完整套件。一次全绿只证明运行它的那棵代码树。 |
| “他们显然想要合并” | 集成决定属于你的人类伙伴。呈现菜单并等待。 |
**开放式问题** | “他们看起来已经不需要这个功能了——我来提供丢弃选项” | 菜单按原样就已经完整。只有当你的人类伙伴明确说出要丢弃时,才能进入丢弃流程。 |
- **问题:** “接下来我该怎么做?”含义不明确 | “‘嗯,把它删掉吧’也算确认” | 只有精确输入 `discard` 才授权删除。 |
- **修复:** 恰好提供 4 个结构化选项(对于 detached HEAD 则提供 3 个) | “PR 已经创建,所以工作树现在只是累赘” | PR 反馈需要在该工作树中修复。工作合入之前,它必须保留。 |
| “另一个工作树看起来陈旧——我顺便清理” | 只清理 `.worktrees/``worktrees/` 下的工作树。其他一切都属于宿主环境。 |
**为选项 2 清理工作树** | “合并结果失败可能只是偶发问题” | 合并结果一旦失败就停止一切。调查期间保留分支和工作树。 |
- **问题:** 移除用户进行 PR 迭代所需的工作树 | “基础分支显然是 main” | 确认分叉点或询问。合并到错误基础分支的撤销成本很高。 |
- **修复:** 仅对选项 1 和 4 执行清理 | “推送被拒绝——强制推送就能解决” | 推送被拒绝说明远端已变化。先调查;只有你的人类伙伴明确要求时才强制推送。 |
**在移除工作树之前删除分支**
- **问题:** `git branch -d` 失败,因为工作树仍在引用该分支
- **修复:** 先合并,再移除工作树,然后删除分支
**在工作树内部运行 git worktree remove**
- **问题:** 当 CWD 位于正被移除的工作树内时,命令会静默失败
- **修复:** 在执行 `git worktree remove` 前,始终先 `cd` 到主仓库根目录
**清理运行平台拥有的工作树**
- **问题:** 移除由运行平台创建的工作树会导致幽灵状态
- **修复:** 仅清理 `.worktrees/``worktrees/` 下的工作树
**丢弃操作没有确认**
- **问题:** 意外删除工作成果
- **修复:** 要求输入 "discard" 进行确认
## 红旗项
**绝不:**
- 在测试失败时继续
- 未验证合并结果上的测试就进行合并
- 未经确认就删除工作成果
- 未经明确请求就强制推送
- 在确认合并成功之前移除工作树
- 清理并非由你创建的工作树(来源检查)
- 从工作树内部运行 `git worktree remove`
**始终:**
- 在提供选项前验证测试
- 在显示菜单前检测环境
- 恰好提供 4 个选项(分离 HEAD 时提供 3 个)
- 对选项 4 获取输入式确认
- 仅为选项 1 和 4 清理工作树
- 在移除工作树前,先 `cd` 到主仓库根目录
- 移除后运行 `git worktree prune`
@@ -201,10 +201,3 @@ description: 在接收代码审查反馈时、实施建议之前使用,尤其
## GitHub 线程回复 ## GitHub 线程回复
在 GitHub 上回复行内审查评论时,应在评论线程中回复(`gh api repos/{owner}/{repo}/pulls/{pr}/comments/{id}/replies`),而不是作为顶层 PR 评论回复。 在 GitHub 上回复行内审查评论时,应在评论线程中回复(`gh api repos/{owner}/{repo}/pulls/{pr}/comments/{id}/replies`),而不是作为顶层 PR 评论回复。
## 底线
**外部反馈 = 需要评估的建议,而非必须遵从的命令。**
验证。质疑。然后实施。
不作表态式附和。始终保持技术严谨。
@@ -5,7 +5,7 @@ description: 在完成任务、实现重大功能或合并前使用,以验证
# 请求代码审查 # 请求代码审查
派遣一个代码审查子 Agent,在问题产生连锁效应之前将其发现。审查者会收到专为评估而精心构建的准确上下文——绝不会收到你的会话历史记录。这能让审查者专注于工作成果,而不是你的思考过程,同时也为你继续工作保留自己的上下文。 派遣一个代码审查子 Agent,在问题产生连锁效应之前将其发现。审查者会收到专为评估而精心构建的准确上下文——绝不会收到你的会话历史记录。
**核心原则:** 及早审查,频繁审查。 **核心原则:** 及早审查,频繁审查。
@@ -72,20 +72,12 @@ HEAD_SHA=$(git rev-parse HEAD)
[继续执行任务 3] [继续执行任务 3]
``` ```
## 与工作流的集成 ## 常见合理化借口
**子 Agent 驱动开发:** | 借口 | 现实 |
- 每项任务后都进行审查 |--------|---------|
- 在问题累积之前将其发现 | “我自己审查 diff 就好,不用派遣审查者” | 你是协调者——直接在当前上下文中审查 diff,会消耗你继续推动工作所需的上下文窗口。派遣审查子 Agent:diff 和评估保留在它的上下文中,只有发现项返回给你。 |
- 修复后再转到下一项任务 | “审查者需要我的完整会话历史才能理解变更” | 只向它提供精心构建的准确上下文,绝不要提供会话历史。这样审查者会专注于工作成果,而不是你的思考过程。 |
**执行计划:**
- 在每项任务后或自然检查点进行审查
- 获取反馈、应用反馈,然后继续
**临时开发:**
- 合并前进行审查
- 遇到阻碍时进行审查
## 危险信号 ## 危险信号
@@ -5,42 +5,43 @@ description: 在当前会话中执行包含独立任务的实现计划时使用
# 子 Agent 驱动的开发 # 子 Agent 驱动的开发
通过为每任务派一个全新的实现者子 Agent 来执行计划,在每任务完成后进行一次任务审查(规符合性 + 代码质量),并在最后对整个分支进行全面审查。 通过为每任务派一个全新的实现者子 Agent在每任务后进行一次任务审查(规符合性 + 代码质量),并在最后对整个分支进行全面审查来执行计划
**为什么使用子 Agent:** 你将任务委派给具有隔离上下文的专门 Agent。通过精确构建提供给它们的指令和上下文,你可以确保它们保持专注并成功完成任务。它们绝不继承你当前会话的上下文或历史记录——你要准确构建它们所需的一切。这也会保留你自己的上下文,以用于协调工作。 **为什么使用子 Agent:** 你将任务委派给具有隔离上下文的专门 Agent。通过精确设计它们的指令和上下文,确保它们保持专注并成功完成各自的任务。它们绝不继承你当前会话的上下文或历史记录——你构建它们恰好所需的内容。这也会保留你自己的上下文,以用于协调工作。
**核心原则:** 每任务使用全新的子 Agent + 任务审查(规 + 质量)+ 全面的最终审查 = 高质量、快速迭代 **核心原则:** 每任务使用全新的子 Agent + 任务审查(规 + 质量)+ 全面的最终审查 = 高质量、快速迭代
**叙述:** 在工具调用之间,最多叙述一行简短内容—— **叙述:** 在工具调用之间,最多叙述一行简短内容——
台账和工具结果承载记录 记录由台账和工具结果承载。
**持续执行:** 不要在任务之间停下来向你的人类伙伴确认。不中断地执行计划中的所有任务。只有以下情况才可停止:出现你无法解决的 BLOCKED 状态、存在确实阻碍进展的歧义,或所有任务均已完成。“我应该继续吗?”之类的询问和进度摘要会浪费人类伙伴的时间——他们已经要求你执行计划,所以就执行它。 **持续执行:** 不要在任务之间停下来向你的人类伙伴确认。不中断地执行计划中的所有任务。仅可因以下情况停止:出现你无法解决的 BLOCKED 状态、存在确实阻碍进展的歧义,或所有任务均已完成。询问“我应该继续吗?”以及提供进度摘要会浪费他们的时间——他们要求你执行计划,所以就执行它。
## 何时使用 ## 何时使用
```dot ```dot
digraph when_to_use { digraph when_to_use {
"Have implementation plan?" [shape=diamond, label="有实现计划吗?"]; "有实现计划吗?" [shape=diamond];
"Tasks mostly independent?" [shape=diamond, label="任务大多相互独立吗?"]; "任务大多相互独立吗?" [shape=diamond];
"Stay in this session?" [shape=diamond, label="继续留在此会话中吗?"]; "留在当前会话中吗?" [shape=diamond];
"subagent-driven-development" [shape=box]; "subagent-driven-development" [shape=box];
"executing-plans" [shape=box]; "executing-plans" [shape=box];
"Manual execution or brainstorm first" [shape=box, label="先手动执行或进行头脑风暴"]; "先手动执行或进行头脑风暴" [shape=box];
"Have implementation plan?" -> "Tasks mostly independent?" [label="是"]; "有实现计划吗?" -> "任务大多相互独立吗?" [label="是"];
"Have implementation plan?" -> "Manual execution or brainstorm first" [label="否"]; "有实现计划吗?" -> "先手动执行或进行头脑风暴" [label="否"];
"Tasks mostly independent?" -> "Stay in this session?" [label="是"]; "任务大多相互独立吗?" -> "留在当前会话中吗?" [label="是"];
"Tasks mostly independent?" -> "Manual execution or brainstorm first" [label="否 - 紧密耦合"]; "任务大多相互独立吗?" -> "先手动执行或进行头脑风暴" [label="否 - 紧密耦合"];
"Stay in this session?" -> "subagent-driven-development" [label="是"]; "留在当前会话中吗?" -> "subagent-driven-development" [label="是"];
"Stay in this session?" -> "executing-plans" [label="否 - 并行会话"]; "留在当前会话中吗?" -> "executing-plans" [label="否 - 并行会话"];
} }
``` ```
**与 Executing Plans(并行会话)比:** **与执行计划(并行会话)的对比:**
- 同一会话(无需切换上下文) - 同一会话(无需切换上下文)
- 每任务使用全新的子 Agent(无上下文污染) - 每任务使用全新的子 Agent(无上下文污染)
- 每任务后进行审查(规符合性 + 代码质量),最后进行全面审查 - 每任务后进行审查(规符合性 + 代码质量),最后进行全面审查
- 更快的迭代(任务之间无需人类介入) - 更快的迭代(任务之间无需人类介入)
## 流程 ## 流程
```dot ```dot
@@ -49,335 +50,344 @@ digraph process {
subgraph cluster_per_task { subgraph cluster_per_task {
label="每项任务"; label="每项任务";
"Dispatch implementer subagent (./implementer-prompt.md)" [shape=box label="派遣实现者子 Agent (./implementer-prompt.md)"]; "派遣实现者子 Agent./implementer-prompt.md" [shape=box];
"Implementer subagent asks questions?" [shape=diamond label="实现者子 Agent 是否提出问题?"]; "实现者提出问题?" [shape=diamond];
"Answer questions, provide context" [shape=box label="回答问题,提供上下文"]; "回答问题,提供上下文" [shape=box];
"Implementer subagent implements, tests, commits, self-reviews" [shape=box label="实现者子 Agent 进行实现、测试、提交并自我审查"]; "实现者进行实现、测试、提交和自审" [shape=box];
"Write diff file, dispatch task reviewer subagent (./task-reviewer-prompt.md)" [shape=box label="写入 diff 文件,派遣任务审查者子 Agent (./task-reviewer-prompt.md)"]; "生成审查包,派遣任务审查者(./task-reviewer-prompt.md" [shape=box];
"Task reviewer reports spec ✅ and quality approved?" [shape=diamond label="任务审查者是否报告规范符合要求 ✅ 且质量获批?"]; "规格 ✅ 且质量通过?" [shape=diamond];
"Dispatch fix subagent for Critical/Important findings" [shape=box label="针对“严重”或“重要”级别的发现派遣修复子 Agent"]; "发现项与计划文本冲突?" [shape=diamond];
"Mark task complete in todo list and progress ledger" [shape=box label="在待办事项列表和进度台账中将任务标记为完成"]; "询问人类伙伴以哪个为准" [shape=box];
"第 R 轮修复(共 5 轮):R≤3 时恢复实现者;R≥4 时派遣使用更强模型的新实现者" [shape=box];
"派遣作用域化复审(./re-review-prompt.md" [shape=box];
"所有发现项都已解决?" [shape=diamond];
"R = 5" [shape=diamond];
"裁决每个未解决的发现项" [shape=box];
"存在影响关键要求的发现项?" [shape=diamond];
"停止:向人类伙伴报告 BLOCKED" [shape=box];
"将发现项及裁决暂存到账本中" [shape=box];
"把完成记录追加到账本,标记待办已完成" [shape=box];
} }
"Read plan, note context and global constraints, create todos" [shape=box label="阅读计划,记录上下文和全局约束,创建待办事项"]; "设置:工作树、账本检查、读取计划、起飞前审查" [shape=box];
"More tasks remain?" [shape=diamond label="是否还有剩余任务?"]; "还有任务?" [shape=diamond];
"Dispatch final code reviewer subagent (../requesting-code-review/code-reviewer.md)" [shape=box label="派遣最终代码审查者子 Agent (../requesting-code-review/code-reviewer.md)"]; "派遣最终代码审查者(../requesting-code-review/code-reviewer.md" [shape=box];
"Use superpowers:finishing-a-development-branch" [shape=box style=filled fillcolor=lightgreen label="使用 superpowers:finishing-a-development-branch"]; "最终审查有发现项?只派遣一次修复、进行一次作用域化复审,并裁决剩余项" [shape=box];
"最终审查干净:删除本计划的工作区" [shape=box];
"使用 superpowers:finishing-a-development-branch" [shape=box style=filled fillcolor=lightgreen];
"Read plan, note context and global constraints, create todos" -> "Dispatch implementer subagent (./implementer-prompt.md)"; "设置:工作树、账本检查、读取计划、起飞前审查" -> "派遣实现者子 Agent./implementer-prompt.md";
"Dispatch implementer subagent (./implementer-prompt.md)" -> "Implementer subagent asks questions?"; "派遣实现者子 Agent./implementer-prompt.md" -> "实现者提出问题?";
"Implementer subagent asks questions?" -> "Answer questions, provide context" [label="是"]; "实现者提出问题?" -> "回答问题,提供上下文" [label="是"];
"Answer questions, provide context" -> "Dispatch implementer subagent (./implementer-prompt.md)"; "回答问题,提供上下文" -> "实现者进行实现、测试、提交和自审";
"Implementer subagent asks questions?" -> "Implementer subagent implements, tests, commits, self-reviews" [label="否"]; "实现者提出问题?" -> "实现者进行实现、测试、提交和自审" [label="否"];
"Implementer subagent implements, tests, commits, self-reviews" -> "Write diff file, dispatch task reviewer subagent (./task-reviewer-prompt.md)"; "实现者进行实现、测试、提交和自审" -> "生成审查包,派遣任务审查者(./task-reviewer-prompt.md";
"Write diff file, dispatch task reviewer subagent (./task-reviewer-prompt.md)" -> "Task reviewer reports spec ✅ and quality approved?"; "生成审查包,派遣任务审查者(./task-reviewer-prompt.md" -> "规格 ✅ 且质量通过?";
"Task reviewer reports spec ✅ and quality approved?" -> "Dispatch fix subagent for Critical/Important findings" [label=""]; "规格 ✅ 且质量通过?" -> "把完成记录追加到账本,标记待办已完成" [label=""];
"Dispatch fix subagent for Critical/Important findings" -> "Write diff file, dispatch task reviewer subagent (./task-reviewer-prompt.md)" [label="重新审查"]; "规格 ✅ 且质量通过?" -> "发现项与计划文本冲突?" [label=""];
"Task reviewer reports spec ✅ and quality approved?" -> "Mark task complete in todo list and progress ledger" [label="是"]; "发现项与计划文本冲突?" -> "询问人类伙伴以哪个为准" [label="是"];
"Mark task complete in todo list and progress ledger" -> "More tasks remain?"; "询问人类伙伴以哪个为准" -> "第 R 轮修复(共 5 轮):R≤3 时恢复实现者;R≥4 时派遣使用更强模型的新实现者";
"More tasks remain?" -> "Dispatch implementer subagent (./implementer-prompt.md)" [label=""]; "发现项与计划文本冲突?" -> "第 R 轮修复(共 5 轮):R≤3 时恢复实现者;R≥4 时派遣使用更强模型的新实现者" [label=""];
"More tasks remain?" -> "Dispatch final code reviewer subagent (../requesting-code-review/code-reviewer.md)" [label="否"]; "第 R 轮修复(共 5 轮):R≤3 时恢复实现者;R≥4 时派遣使用更强模型的新实现者" -> "派遣作用域化复审(./re-review-prompt.md";
"Dispatch final code reviewer subagent (../requesting-code-review/code-reviewer.md)" -> "Use superpowers:finishing-a-development-branch"; "派遣作用域化复审(./re-review-prompt.md" -> "所有发现项都已解决?";
"所有发现项都已解决?" -> "把完成记录追加到账本,标记待办已完成" [label="是"];
"所有发现项都已解决?" -> "R = 5" [label="否"];
"R = 5" -> "第 R 轮修复(共 5 轮):R≤3 时恢复实现者;R≥4 时派遣使用更强模型的新实现者" [label="否——进入下一轮"];
"R = 5" -> "裁决每个未解决的发现项" [label="是——触发熔断"];
"裁决每个未解决的发现项" -> "存在影响关键要求的发现项?";
"存在影响关键要求的发现项?" -> "停止:向人类伙伴报告 BLOCKED" [label="是"];
"存在影响关键要求的发现项?" -> "将发现项及裁决暂存到账本中" [label="否"];
"将发现项及裁决暂存到账本中" -> "把完成记录追加到账本,标记待办已完成";
"把完成记录追加到账本,标记待办已完成" -> "还有任务?";
"还有任务?" -> "派遣实现者子 Agent./implementer-prompt.md" [label="是"];
"还有任务?" -> "派遣最终代码审查者(../requesting-code-review/code-reviewer.md" [label="否"];
"派遣最终代码审查者(../requesting-code-review/code-reviewer.md" -> "最终审查有发现项?只派遣一次修复、进行一次作用域化复审,并裁决剩余项";
"最终审查有发现项?只派遣一次修复、进行一次作用域化复审,并裁决剩余项" -> "最终审查干净:删除本计划的工作区";
"最终审查干净:删除本计划的工作区" -> "使用 superpowers:finishing-a-development-branch";
} }
``` ```
## 执行前计划审查 ## 设置
在派遣任务 1 之前,先通读一次计划以检查冲突: 务必确保工作在隔离的工作区中进行:使用
superpowers:using-git-worktrees 创建一个工作区,或验证现有工作区。
未经你的人类伙伴明确同意,绝不要在 main/master 分支上开始实现。
- 相互矛盾或与计划的全局约束相冲突的任务 对话记忆无法在压缩后保留。在真实会话中,忘记当前进度的
- 计划明确要求、但审查准则将其视为缺陷的任何内容 控制器曾重新分派整套已经完成的任务序列——这是观察到的代价
(没有任何断言的测试、逐字重复的逻辑块) 最高的单一故障。请在台账文件中跟踪进度,而不要仅依赖待办事项。
- 每个计划都拥有自己的工作区:技能启动时,运行本技能的
`scripts/sdd-workspace PLAN_FILE`——它会输出该计划被 git 忽略的
目录(`<repo-root>/.superpowers/sdd/<plan-basename>/`),本计划的
所有产物都存放于此:台账、简报、报告、复审包。
绝不得读取或写入另一个计划的目录。
- 检查此计划位于 `<workspace>/progress.md` 的台账。如果其第一行
指明的是你的计划文件,则带有一行 `Task <N>: complete` 的任务均已完成
——不得重新分派它们;从第一个没有该行的任务继续。若某项任务的
最后一行是一个修复轮次,则该任务正处于循环中:从下一轮继续该
循环。如果台账的第一行指明的是另一个计划文件——或者旧的扁平路径
`.superpowers/sdd/progress.md` 中存在一个遗留台账——那就是另一个
计划的进度:将其留在原处,并从头新建你自己的台账。
- 创建台账,并将其身份信息写在第一行:
`# SDD ledger — plan: <plan file path>`
- 台账是你的恢复地图:其中记载的提交即使在你的上下文不再记得创建过
它们时,也仍存在于 git 中。压缩后,相比你自己的回忆,应相信台账和
`git log`
- `git clean -fdx` 会摧毁该工作区(它是被 git 忽略的临时目录);如果
发生这种情况,请从 `git log` 恢复。
通读计划一次,记下其上下文和全局约束,并为每项任务创建一个
待办事项。
在分派任务 1 之前,通读计划一次以检查冲突:
- 彼此矛盾或与计划的全局约束相矛盾的任务
- 计划明确要求、但审查量规视为缺陷的任何事项(一个没有进行任何断言的
测试、逐字重复一个逻辑块)
在执行开始前,将你发现的所有内容作为一个批量问题提交给你的人类伙伴——
将每项发现与要求该事项的计划文本并列,并询问以哪一项为准——而不是在
计划执行途中每发现一项就中断一次。如果检查未发现问题,则不作说明,继续
执行。对于只有在实现过程中才显现的冲突,复审循环仍是兜底机制。
在执行开始前,将你发现的所有内容作为一个批量问题呈现给你的人类伙伴——
每项发现都应与要求该项的计划文本并列,并询问应以哪一方为准——
而不是在计划执行中途每发现一项就打断一次。如果扫描未发现问题,
则不作说明并继续。审查循环仍是用于捕获那些只有在实现过程中
才显现的冲突的安全网。
## 模型选择 ## 模型选择
使用能够胜任各角色的能力最弱的模型,以节省成本并提高速度。 使用能够胜任各角色的最低能力模型,以节省成本并提高速度。
**机械性实现任务**立函数、明确的规、1-2 个文件):使用快速、便宜的模型。当计划定得足够明确时,大多数实现任务都是机械性的。 **机械性实现任务**立函数、明确的规、1-2 个文件):使用快速、廉价的模型。当计划定得足够明确时,大多数实现任务都是机械性的。
**集成和判断任务**(多文件协调、模式匹配、调试):使用标准模型。 **集成和判断任务**(多文件协调、模式匹配、调试):使用标准模型。
**架构和设计任务**:使用能力最强的可用模型。 **架构和设计任务**:使用当前可用的最强大模型。
最终的整个分支审查就属于此类任务之一——应使用能力最强的可用模型来派发,而不是使用会话默认模型。 整个分支的最终审查就属于此类——应将其派发给当前可用的最强大模型,而不是会话默认模型。
**审查任务**选择具备同等判断能力,并与 diff 的规模、复杂和风险相匹配的模型。小型机械性 diff 不需要能力最强的模型;微的并发变更则需要。 **审查任务**以同样的判断标准选择模型,并根据差异的规模、复杂和风险调整。小型机械性差异不需要最强的模型;微的并发变更则需要。对小型修复差异进行范围明确的复审时,使用廉价至中档模型。
**派发子 Agent 时,始终显式指定模型。** 如果省略模型,就会继承会话所使用的模型——通常是能力最强且最昂贵的模型——这会悄无声息地违背本节要求 **修复循环升级(第 4-5 轮)**:使用至少比陷入困境的实现者高一个档次的模型
**轮次数比 token 单价更重要。** 实际耗时和上下文成本会随子 Agent 所需的轮次数增加,而最便宜的模型在多步骤工作中通常需要 2-3 倍的轮次——导致总体成本反而更高。对于审查者,以及根据自然语言描述开展工作的实现者,至少使用中档模型。当任务计划文本包含需要编写的完整代码时,实现工作就是誊写加测试:该实现者应使用最便宜的档位。单文件机械性修复也使用最便宜的档位 **派发子 Agent 时,始终明确指定模型。** 省略模型时,将继承你会话所用的模型——通常是最强大且最昂贵的模型——这会在不知不觉中违背本节的目的
**轮次数比 token 单价更重要。** 实际耗时和上下文成本会随子 Agent 的轮次数量而增加,而最廉价的模型在多步骤工作中通常需要 2-3× 的轮次——导致总体成本更高。对于审查者以及根据文字描述开展工作的实现者,应至少使用中档模型。当任务的计划文本包含需要编写的完整代码时,实现工作就是誊写加测试:为该实现者使用最廉价档次的模型。单文件机械性修复也使用最廉价档次的模型。
**任务复杂度信号(实现任务):** **任务复杂度信号(实现任务):**
- 涉及 1-2 个文件且规完整 → 便宜模型 - 涉及 1-2 个文件且规完整 → 廉价模型
- 涉及多个文件且存在集成问题 → 标准模型 - 涉及多个文件且存在集成方面的考量 → 标准模型
- 需要设计判断或对代码库有广泛理解 → 能力最强的模型 - 需要设计判断或对代码库有广泛理解 → 最强的模型
## 处理实现者状态 ## 任务循环
实现者子 Agent 会报告以下四种状态之一。应根据每种状态进行恰当处理: 你粘贴到派遣提示中的所有内容——以及子 Agent 返回的所有输出——在会话余下的整个期间都会常驻于你的上下文中,并在之后的每一轮中被重新读取。请以文件形式交接产物。
**DONE:** 生成审查包(在此技能的目录中运行 `scripts/review-package BASE HEAD`——它会打印自己写入的唯一文件路径;BASE 是派发实现者之前记录的提交——绝不能使用 `HEAD~1`,因为它会悄无声息地丢弃多提交任务中除最后一次提交之外的所有提交),然后使用打印出的路径派发任务审查者。 ### 1. 派遣实现者
**DONE_WITH_CONCERNS:** 实现者已完成工作,但提出了疑虑。继续之前先阅读这些疑虑。如果疑虑与正确性或范围有关,应在审查前解决。如果只是观察意见(例如,“这个文件越来越大了”),则记录下来并继续进行审查 派遣前记录 BASE`git rev-parse HEAD`)——审查包和修复轮次的 diff 都需要它
**NEEDS_CONTEXT:** 实现者需要尚未提供的信息。提供缺失的上下文并重新派发。 - **任务简报:**派遣实现者之前,运行本技能的
`scripts/task-brief PLAN_FILE N`——它会将任务的完整文本提取到一个具有唯一名称的文件中,并打印该路径。组织派遣内容时,应让简报始终作为需求的唯一来源。你的派遣内容应包含:(1) 用一行说明此任务在项目中的位置;(2) 简报路径,并以“先阅读此文件——它就是你的需求,其中包含必须原样使用的精确值”引出;(3) 简报不可能知道的、来自先前任务的接口和决策;(4) 你对简报中所注意到的任何歧义的解决方案;(5) 报告文件路径和报告约定。精确值(数字、魔法字符串、签名、测试用例)只能出现在简报中。绝不要让子 Agent 阅读整个计划文件。
- **报告文件:**以简报为实现者的报告文件命名(简报 `…/task-N-brief.md` → 报告 `…/task-N-report.md`),并将其写入派遣提示。实现者在该文件中写入完整报告,并且只返回状态、提交记录、一行测试摘要和关注事项。
- 一条派遣提示描述的是一个任务,而不是会话的历史。不要把累积的先前任务摘要(“任务 1-3 后的状态”)粘贴到后续派遣中——某个真实会话的派遣内容达到了 42k 个字符,其中 99% 都是粘贴的历史记录。一个全新的子 Agent 需要的是它的任务、它会涉及的接口以及全局约束。除此之外什么都不需要。
- 如果先前的任务在本任务涉及的区域中暂存了一项发现,请在派遣内容中附上指向该账本条目的指针。
- 从派遣结果中记录实现者的 Agent 身份——修复循环第 1-3 轮会恢复此 Agent。
- 绝不要并行派遣多个实现子 Agent(会产生冲突)。
**BLOCKED:** 实现者无法完成任务。评估阻塞原因: 模板:[implementer-prompt.md](implementer-prompt.md)
1. 如果是上下文问题,提供更多上下文,并使用同一模型重新派发
2. 如果任务需要更强的推理能力,使用能力更强的模型重新派发
3. 如果任务过大,将其拆分成更小的部分
4. 如果计划本身有误,上报给人类
**绝不要**忽略升级请求,也不要在不作任何改变的情况下强迫同一模型重试。如果实现者表示自己卡住了,就必须做出改变。 ### 2. 处理报告
## 处理审查者的 ⚠️ 项
任务审查者可能会报告“⚠️ 无法从差异中验证”项——这些要求 实现者子 Agent 会报告以下四种状态之一。请分别妥善处理:
存在于未更改的代码中或跨越多个任务。这些项不会阻塞审查的其余部分,
但在将任务标记为完成之前,你必须自行解决每一项:你掌握着审查者
所缺少的计划和跨任务上下文。如果你确认某一项确实是缺口,请将其视为
规格审查失败——将其发回给实现者并重新审查。
## 构建审查者提示词 **DONE:** 生成审查包(`scripts/review-package PLAN_FILE BASE HEAD`,从本技能的目录运行——它会输出其写入的唯一文件路径;BASE 是你在派遣实现者之前记录的提交——绝不能使用 `HEAD~1`,因为这会悄无声息地丢弃多提交任务中除最后一个提交之外的所有提交),然后使用输出的路径派遣任务审查者。
每任务审查是任务范围内的门禁。全面审查只在最终全分支审查 **DONE_WITH_CONCERNS:** 实现者完成了工作,但提出了一些疑虑。继续之前先阅读这些疑虑。如果疑虑涉及正确性或范围,请在审查之前解决。如果只是一些观察(例如,“这个文件变得越来越大了”),请记录下来并继续审查
进行一次。当你填写审查者模板时:
- 不要添加诸如“检查所有用法”或“如果有用就运行竞态测试” **NEEDS_CONTEXT:** 实现者需要尚未提供的信息。提供缺失的上下文并重新派遣。
之类的开放式指令,除非有具体且针对该任务的理由
- 不要要求审查者在同一份代码上重新运行实现者已经运行过的测试——
实现者的报告已提供测试证据
- 不要替审查者预判发现项——绝不要指示审查者忽略或不标记某个
特定问题。如果你认为某个发现项会是误报,就让审查者提出它,并在
审查循环中裁决。如果你正在编写的提示词包含“不要标记”、“不要将 X
视为缺陷”、“最高为次要”或“计划选择了”——停下:你正在
预判,通常是为了让自己免去一轮审查。
- 你交给审查者的全局约束块是其关注视角。逐字复制计划的“全局约束”
部分或规格中的约束性要求:精确值、精确格式,以及所述的组件间关系
(“与 X 布局相同”、“与 Y 匹配”)。审查者的模板已经包含流程规则
(YAGNI、测试卫生规范、审查方法)——约束块用于说明这个特定项目的
规格所要求的内容。
- 以文件形式将差异交给审查者:运行此技能的
`scripts/review-package BASE HEAD`,并把它输出的文件路径
传给审查者(或者在没有 bash 时:对该范围运行 `git log --oneline`
`git diff --stat``git diff -U10`,并将输出重定向到一个名称唯一的
文件)。输出绝不会进入你自己的上下文,而审查者只需一次 Read
调用,就能看到提交列表、统计摘要以及带上下文的完整差异。使用你在
派遣实现者之前记录的 BASE——绝不要使用 `HEAD~1`,它会悄无声息地
截断包含多个提交的任务。
- 派遣提示词描述的是一个任务,而不是会话历史。不要把累积的先前任务
摘要(“任务 1-3 后的状态”)粘贴到后续派遣中——一次真实会话的派遣
提示词达到了 42k 个字符,其中 99% 都是粘贴的历史记录。一个新的
子 Agent 需要的是它的任务、它所涉及的接口以及全局约束,仅此而已。
- 针对“严重”和“重要”发现项派遣修复子 Agent。过程中将“次要”
发现项记录到进度账本中,并让最终全分支审查关注该列表,以便评估
哪些必须在合并前修复。无人阅读的汇总就是无声的丢弃。
- 被标记为 plan-mandated 的发现项——或任何与计划文本要求冲突的发现项——
与任何计划矛盾一样,都应由人类决定:展示该发现项和计划文本,并询问
以哪一个为准。不要因为计划规定了它就驳回该发现项,也不要在未询问的
情况下派遣会产生与计划冲突结果的修复。
- 最终全分支审查也要获得一个包:运行
`scripts/review-package MERGE_BASE HEAD`MERGE_BASE = 分支起始的
提交,例如 `git merge-base main HEAD`),并在最终审查派遣中包含
输出的路径,使最终审查者读取一个文件,而不是使用 git 命令重新推导
分支差异。
- 每次修复派遣都包含实现者契约:修复子 Agent 要重新运行覆盖其更改的
测试并报告结果。在派遣中指明覆盖该更改的测试文件——单行修复不需要
运行整个测试套件。在重新派遣审查者之前,确认修复报告包含覆盖性测试、
运行的命令和输出;三者全部具备后再派遣重新审查。
- 如果最终全分支审查返回发现项,只派遣一个修复子 Agent,并向其提供
完整的发现项列表——不要为每个发现项分别派遣一个修复者。逐项派遣的
修复者都会各自重建上下文并重新运行测试套件;一次真实会话的最终审查
修复波次耗费的成本超过了其所有任务的总和。
## 文件交接
你粘贴到派发提示词中的所有内容——以及子 Agent **BLOCKED:** 实现者无法完成任务。评估阻碍因素:
打印返回的所有内容——都会在会话剩余期间常驻于你的上下文中, 1. 如果是上下文问题,请提供更多上下文,并使用同一模型重新派遣
并在此后的每一轮中被重新读取。请以文件形式交接产物: 2. 如果任务需要更多推理,请使用能力更强的模型重新派遣
3. 如果任务过大,请将其拆分成更小的部分
4. 如果计划本身有误,请上报给人类
- **任务简报:**在派发实现者之前,运行此技能的 **绝不**忽略上报,也绝不强迫同一模型在没有任何改变的情况下重试。如果实现者表示它卡住了,就必须做出改变。
`scripts/task-brief PLAN_FILE N`——它会将任务的完整文本提取到一个
具有唯一名称的文件中,并打印其路径。编写派发提示词时,应让该 如果实现者在开始之前或任务进行期间提出问题,请清晰、完整地回答,
简报始终作为需求的唯一来源。你的派发提示词应包含:(1) 用一行说明此任务在项目中的位置;(2) 在需要时提供额外的上下文,并且不要催促它仓促进入实现阶段。
简报路径,并以“先阅读此文件——它就是你的需求,其中包含必须
原样使用的精确值”引出;(3) 简报无法获知的、来自先前任务的接口和决策;(4) 你对
简报中任何已发现歧义的裁定;(5) 报告文件路径和 ### 3. 审查任务
报告约定。精确值(数字、魔法字符串、签名、测试
用例)只出现在简报中 逐任务审查是限定于任务范围的关卡。广泛审查只进行一次,即在最终的整个分支审查时进行。绝不要跳过任务审查,也绝不要接受缺少任一结论的报告——规范符合性和任务质量两者都必须具备。实现者自我审查绝不能取代任务审查;两者都必不可少
- **报告文件:**按照简报命名实现者的报告文件
(简报 `…/task-N-brief.md` → 报告 `…/task-N-report.md`),并将其写入 - 将 diff 作为文件交给审查者:运行本技能的
派发提示词。实现者在该文件中写入完整报告, `scripts/review-package PLAN_FILE BASE HEAD`,并将它输出的文件路径传给审查者
且仅返回状态、提交、一行测试摘要和关注事项。 (或者,不使用 bash 时:针对该范围运行 `git log --oneline``git diff --stat`
- **审查者输入:**任务审查者会获得三个路径——同一个简报 `git diff -U10`,并将输出重定向到一个名称唯一的
文件)。该输出绝不会进入你自己的上下文,而审查者只需一次 Read
调用就能看到提交列表、统计摘要以及带上下文的完整 diff。
使用你在派遣实现者之前记录的 BASE——绝不要使用
`HEAD~1`,因为它会悄无声息地截断包含多个提交的任务。绝不要
在没有 diff 文件的情况下派遣任务审查者。
- **审查者输入:**任务审查者会获得三个路径——同一份简报
文件、报告文件和审查包——以及约束该任务的全局 文件、报告文件和审查包——以及约束该任务的全局
约束。 约束。
- 修复任务的派发会将其修复报告(包括测试结果)追加到同一个 - 你交给审查者的全局约束块是其关注问题的
报告文件中,并返回简短摘要;重新审查时读取更新后的文件。 视角。逐字复制计划的“全局约束”部分或规范中具有约束力的
要求:确切的值、确切的格式,以及明确说明的组件之间的
关系(“与 X 布局相同”、“与 Y
匹配”)。审查者的模板已经包含流程规则(YAGNI、
测试卫生、审查方法)——约束块用于说明这个
项目的规范要求什么。
- 不要添加诸如“检查所有使用处”或“如果有用就运行竞态测试”
之类的开放式指令,除非有具体的、特定于任务的理由
- 不要让审查者重新运行实现者已在同一份代码上运行过的测试
——实现者的报告中包含测试证据
- 不要替审查者预判发现项——绝不要指示审查者
忽略某个具体问题或不标记它。如果你认为某个发现项会是
误报,就让审查者提出它,并在审查
循环中对其作出裁定。如果你正在编写的提示包含“不要标记”、“不要将 X
视为缺陷”、“至多为 Minor”或“计划选择了”——立即停止:你正在
预判,通常是为了让自己省掉一轮审查循环。
任务审查者可能会报告“⚠️ 无法从 diff 验证”的事项——这些要求
存在于未更改的代码中或跨越多个任务。这些事项不会阻塞审查的
其余部分,但在将任务标记为完成之前,你必须自行解决每一项:
你掌握着审查者所缺少的计划和跨任务上下文。
如果你确认某项确实是缺口,就将其视为规范审查失败——它会与其他
发现项一同进入修复循环。
## 持久化进度 模板:[task-reviewer-prompt.md](task-reviewer-prompt.md)
对话记忆无法在压缩后保留。在真实会话中, ### 4. 修复循环
丢失进度位置的控制器曾重新派发整组已完成的任务
序列——这是已观察到的代价最高的单一故障。请在
账本文件中跟踪进度,而不应只在待办事项中跟踪。
- 技能启动时,检查是否存在账本: 当审查报告规格 ❌、任何严重或重要发现项,或你已确认为真实缺口的 ⚠️ 条目时,就会触发该循环。
`cat "$(git rev-parse --show-toplevel)/.superpowers/sdd/progress.md"`。其中列为
已完成的任务均为 DONE——不要重新派发它们;从第一个
未标记为已完成的任务继续。
- 当某项任务的审查结果无问题时,在进行其他记录工作的同一条消息中,向账本追加一行:
`任务 N:已完成(提交 <base7>..<head7>,审查无问题)`
- 账本是你的恢复地图:其中列出的提交存在于 git 中,即使
你的上下文已不再记得创建过它们。压缩后,
应信任账本和 `git log`,而不是你自己的回忆。
- `git clean -fdx` 会销毁账本(它是被 git 忽略的临时文件);如果
发生这种情况,请从 `git log` 恢复。
## 提示词模板
- [implementer-prompt.md](implementer-prompt.md) - 派遣实现者子 Agent 在循环开始之前,有两条路径会立即退出该循环:
- [task-reviewer-prompt.md](task-reviewer-prompt.md) - 派遣任务审查子 Agent(规格符合性 + 代码质量)
- 最终全分支审查:使用 superpowers:requesting-code-review 的 [code-reviewer.md](../requesting-code-review/code-reviewer.md)
## 工作流示例 - 在推进过程中,将次要发现项记录到进度账本中
`Task <N>: minor (deferred): <one-liner>`),并让最终的全分支审查查看该列表,以便它判定其中哪些必须在合并前修复。无人阅读的汇总就是悄无声息的丢弃。次要发现项永不进入该循环。
- 标记为计划强制要求的发现项——或任何与计划文本要求冲突的发现项——都应像任何计划矛盾一样,由人类决定:展示该发现项和计划文本,并询问以哪一个为准。不要因为计划强制要求它就驳回该发现项,也不要在未询问的情况下派发与计划相冲突的修复。
其余一切都进入该循环。一轮修复包括一次修复派发,加上一次限定范围的复审。每个任务最多五轮:
**第 1-3 轮——恢复原实现者。** 将未解决的发现项逐字发送给它。它的上下文完好无损:它了解任务、代码以及它自己的选择。如果你的运行框架无法向仍在运行的子 Agent 再发送一条消息,则派发一个新的实现者,并向其提供简报路径、报告文件路径和这些发现项——无论哪种方式,报告文件都是持久记忆。
**第 4-5 轮——在能力更强的模型上派发一个新的实现者**(按照模型选择),并向其提供简报路径、报告文件路径、未解决的发现项,以及以下说明:“先前的一名实现者已尝试此任务 [N] 次;现在由你负责。阅读报告文件,了解已经尝试过什么。”一个经历三次恢复调用后仍未结束的循环,通常意味着实现者无法看出自己的问题——一次行动同时引入新的视角并提升能力。
**每一轮,无论是哪种情况:** 实现者进行修复,重新运行覆盖已修改代码的测试,将其修复报告追加到同一个报告文件中,并返回简短契约。在重新派发审查者之前,确认修复报告包含覆盖已修改代码的测试、所运行的命令和输出;三者全部具备后,再派发复审。在修复消息中列出覆盖测试文件的名称——单行修复不需要整个测试套件。
**复审是作用域化的。** 运行 `scripts/review-package PLAN_FILE FIX_BASE HEAD`,其中 FIX_BASE 是上一次审查所看到的提交头,并派发 [re-review-prompt.md](re-review-prompt.md),同时提供发现项列表、简报、报告文件和打印出的 diff 路径。复审者将每个发现项裁定为 ADDRESSED 或 NOT ADDRESSED,并且只标记修复 diff 中的新破坏。修复 diff 中新的 Critical/Important 破坏会加入未解决的发现项列表。超出作用域的观察项会作为延期次要项记入账本——它们绝不会延长循环。
**每轮结束后,**向账本追加:
`Task <N>: fix round <R>/5 (<X> addressed, <Y> open — <finding one-liners>; commits <a7>..<b7>)`
绝不要在控制器会话中亲自修复发现项——你的上下文应保持干净以便协调,而控制器直接修复会跳过审查。
**熔断器。** 如果第 5 轮复审后仍有未解决的发现项,就停止派遣。由你亲自裁决每个未解决的发现项——你掌握审查者所缺少的计划和跨任务上下文:
- **审查者判断错误,或该问题存在争议:** 暂存它——
`Task <N>: parked — <finding> — ruling: <why the code stands>`。最终审查会看到双方理由。
- **问题真实存在,但下游没有任何内容依赖它:** 以同样方式暂存,并在裁决中说明它确实存在但已延期。
- **问题真实存在且影响关键要求**——后续任务会依赖它,或它揭示了计划缺陷:停止。追加
`Task <N>: BLOCKED — <reason>`,并向你的人类伙伴报告该发现项、与之冲突的计划文本以及修复历史。暂存结构性失败会让所有依赖任务继续在它之上构建,也会把一个最终审查同样无法修复的问题留到最后。
只在达到轮次上限时进行裁决。提前裁决来结束循环,不过是换了名称的预判。每次裁决都必须写入账本——严禁无声丢弃。
### 5. 完成任务
当审查结果无问题——或者在达到上限时,每个未关闭的发现项都已附裁定搁置——请在执行其他记账工作的同一条消息中,将完成行追加到账本:
- `Task <N>: complete (commits <base7>..<head7>, review clean)`
- 触发熔断器后使用 `Task <N>: complete (commits <base7>..<head7>, <K> parked)`
然后将待办事项标记为完成并继续。只要审查中仍有既未修复、也未在达到上限时附裁定搁置的未关闭 Critical/Important 问题,就绝不要进入下一个任务。
## 最终审查
最终的整分支审查也要有一个审查包:运行 `scripts/review-package PLAN_FILE MERGE_BASE HEAD`MERGE_BASE = 该分支起始于的提交,例如 `git merge-base main HEAD`),并在派发最终审查时包含输出的路径,这样最终审查者只需读取一个文件,而不必使用 git 命令重新推导分支差异。使用最强的可用模型进行派发(参见模型选择),并使用 superpowers:requesting-code-review 的 [code-reviewer.md](../requesting-code-review/code-reviewer.md)。让它查看账本中的 deferred-minor 行和 parked 行,以便对哪些项目必须在合并前修复进行分诊。
如果最终的整分支审查返回发现项,请派发一个且仅一个修复子 Agent,并向其提供完整的发现项列表——不要为每个发现项分别派一个修复者。按发现项分别安排的修复者都要重建上下文并重新运行套件;在一次真实会话中,最终审查的修复波次成本超过了其所有任务的总和。然后,对该修复波次恰好进行一次限定范围的复审(针对修复范围运行 `scripts/review-package PLAN_FILE FIX_BASE HEAD`[re-review-prompt.md](re-review-prompt.md))。按照任务循环中的熔断器方式裁决任何残留发现项:附裁定搁置,或在遇到承重性发现项时停止。不存在第二个修复波次——当 finishing-a-development-branch 呈现选项时,残留的承重性发现项会提交给你的人类伙伴。
## 收尾
当最终的整分支审查无问题且其修复已合并后,删除此计划的工作区(`rm -rf <workspace>`)——现在 git 历史就是记录。同级目录属于其他计划;不要动它们。
使用 superpowers:finishing-a-development-branch。
## 常见合理化
| 借口 | 事实 |
|--------|---------|
| "规范符合度已经足够接近了" | 审查者发现规范缺口 = 尚未完成。修复它,或者达到上限并进行裁决——只有这两条出路。 |
| "我会自己修复,派发会增加开销" | 控制器进行修复会污染你的上下文并跳过审查。让实现者继续。 |
| "再来一轮就会收敛" | 超过上限后,各轮不会收敛——这种失败是结构性的。进行裁决并转交。 |
| "反正审查者还是会找到新的问题" | 限定范围的复审只验证修复;不得偏离范围。未改动代码上的新发现项应进入账本,而不是进入循环。 |
| "这个发现项显然是错的,我会丢弃它" | 你只能在达到上限时进行裁决,而且每项裁定都必须成为一条账本条目。禁止静默丢弃。 |
| "修复很小,跳过复审吧" | 未经审查的修复正是回归问题混入的方式。每一轮都以限定范围的复审结束。 |
| "审查会拖慢循环" | 没有审查的循环只不过是未经验证的反复折腾。审查是循环的刹车和方向盘。 |
| "账本记账是额外开销" | 账本是在压缩后仍能保留下来的东西。没有账本的控制器曾重新派发整个已完成的任务序列。 |
## 示例工作流
``` ```
你:我正在使用子 Agent 驱动开发来执行此计划。 你:我正在使用子 Agent 驱动开发来执行此计划。
[读取一次计划文件:docs/superpowers/plans/feature-plan.md] [设置:已验证工作树]
[一次性读取计划文件:docs/superpowers/plans/feature-plan.md]
[解析工作区:scripts/sdd-workspace docs/superpowers/plans/feature-plan.md——其中没有账本,全新开始]
[为所有任务创建待办事项] [为所有任务创建待办事项]
任务 1钩子安装脚本 任务 1Hook 安装脚本
[为任务 1 运行 task-brief携带简报 + 报告路径 + 上下文派遣实现者] [为任务 1 运行 task-brief使用简报路径 + 报告路径 + 上下文派遣实现者]
实现者:“在我开始之前——钩子应该安装在用户级还是系统级?” 实现者:“开始前确认一下——Hook 应安装在用户级还是系统级?”
你:“用户级(~/.config/superpowers/hooks/)” 你:“用户级(~/.config/superpowers/hooks/)”
实现者:“明白。现在开始实现...” 实现者:[稍后]
[稍后] 实现者:
- 实现了 install-hook 命令 - 实现了 install-hook 命令
- 添加了测试,5/5 通过 - 添加了测试,5/5 通过
- 自我审查:发现遗漏了 --force 标志,已添加 - 自:发现遗漏了 --force 标志,已添加
- 已提交 - 已提交
[运行 review-package,携带打印出的路径派遣任务审查者] [运行 review-package PLAN_FILE BASE HEAD;使用打印出的路径派遣任务审查者]
任务审查者:规格 ✅ - 满足所有要求,没有额外内容。 任务审查者:规格 ✅——满足全部要求,没有额外内容。
优点:测试覆盖良好,代码简洁。问题:无。任务质量:通过。 优点:测试覆盖良好,代码干净。问题:无。任务质量:通过。
[将任务 1 标记为完成] [账本:Task 1: complete (commits a1b2c3d..d4e5f6a, review clean)]
任务 2:恢复模式 任务 2:恢复模式
[为任务 2 运行 task-brief携带简报 + 报告路径 + 上下文派遣实现者] [为任务 2 运行 task-brief使用简报路径 + 报告路径 + 上下文派遣实现者]
实现者:[没有问题,继续执行] 实现者:[没有问题]
实现者:
- 添加了 verify/repair 模式 - 添加了 verify/repair 模式
- 8/8 测试通过 - 8/8 测试通过
- 自我审查:一切良好
- 已提交 - 已提交
[运行 review-package,携带打印出的路径派遣任务审查者] [运行 review-package PLAN_FILE BASE HEAD;使用打印出的路径派遣任务审查者]
任务审查者:规格 ❌: 任务审查者:规格 ❌:
- 缺:进度报告(规格要求“每 100 个项目报告一次”) - 缺:进度报告(规格要求“每处理 100 报告一次”)
- 额外:添加了 --json 标志(未要求)
问题(重要):魔法数字(100 问题(重要):魔法数字(100
[派遣修复子 Agent,并提供所有发现的问题] [第 1 轮修复:把两个发现项都发送给原实现者并恢复它]
修复者:移除了 --json 标志,添加了进度报告,提取 PROGRESS_INTERVAL 常量 实现者:添加了进度报告,提取 PROGRESS_INTERVAL 常量
重新运行 test/recovery.test.js——10/10 通过。已追加修复报告。
[任务审查者再次审查] [运行 review-package PLAN_FILE FIX_BASE HEAD;派遣作用域化复审]
任务审查者:规格 ✅。任务质量:通过 复审者:缺少进度报告——已解决(src/recovery.js:41
魔法数字——已解决(src/recovery.js:7)。新的破坏:无。
结论:所有发现项均已解决。
[将任务 2 标记为完成] [账本:Task 2: fix round 1/5 (2 addressed, 0 open; commits d4e5f6a..b7c8d9e)]
[账本:Task 2: complete (commits d4e5f6a..b7c8d9e, review clean)]
... ……
[所有任务完成后] [所有任务完成后]
[派遣最终 code-reviewer] [运行 review-package PLAN_FILE MERGE_BASE HEAD;派遣使用最强模型的最终代码审查者]
最终审查者:所有要求均已满足,可以合并 最终审查者:满足全部要求。已分类延期的次要问题:没有阻碍合并的项目。
完成! [删除本计划的工作区——记录现在保存在 git 中]
完成!正在使用 superpowers:finishing-a-development-branch。
``` ```
## 优势
**与手动执行相比:**
- 子 Agent 会自然地遵循 TDD
- 每个任务都有全新的上下文(不会混淆)
- 可安全并行(子 Agent 互不干扰)
- 子 Agent 可以提问(工作开始前以及工作期间都可以)
**与 Executing Plans 相比:**
- 同一会话(无需交接)
- 持续推进(无需等待)
- 自动设置审查检查点
**效率提升:**
- 控制器精确筛选所需的上下文;大体量产物以文件形式传递,
而不是粘贴文本
- 子 Agent 一开始就能获得完整信息
- 在工作开始前提出问题(而不是开始后)
**质量关卡:**
- 自我审查会在交接前发现问题
- 任务审查包含两项结论:规范符合性和代码质量
- 审查循环确保修复确实有效
- 规范符合性可防止过度构建或构建不足
- 代码质量可确保实现足够完善
**成本:**
- 需要更多次子 Agent 调用(每个任务都需要实现者 + 审查者)
- 控制器需要做更多准备工作(预先提取所有任务)
- 审查循环会增加迭代次数
- 但能及早发现问题(比之后调试更便宜)
## 红旗项
**绝不要:**
- 未经用户明确同意就在 main/master 分支上开始实现
- 跳过任务审查,或接受缺少任一结论的报告(规范符合性和任务质量两者均为必需)
- 在问题尚未修复时继续推进
- 并行派遣多个实现子 Agent(会发生冲突)
- 让子 Agent 阅读整个计划文件(应把它的任务简报——
`scripts/task-brief`——交给它)
- 跳过背景铺垫上下文(子 Agent 需要理解任务在整体中的位置)
- 忽略子 Agent 的问题(先回答,再让它们继续)
- 在规范符合性上接受“差不多就行”(审查者发现规范问题 = 尚未完成)
- 跳过审查循环(审查者发现问题 = 实现者修复 = 再次审查)
- 让实现者的自我审查取代实际审查(两者都需要)
- 在派遣提示中告诉审查者哪些问题不要标记,或预先评定某项发现的严重程度
(“最多将其视为次要”)——计划中的示例代码只是
起点,并不能证明其中的弱点是有意选择的
- 在没有 diff 文件的情况下派遣任务审查者——先生成该文件
`scripts/review-package BASE HEAD`),并在提示中写明其输出的路径
- 在审查仍有未解决的“严重”或“重要”问题时转到下一个任务
- 重新派遣进度台账中已标记为完成的任务——在任何上下文压缩或恢复后,
检查台账(以及 `git log`
**如果子 Agent 提出问题:**
- 清晰、完整地回答
- 如有需要,提供额外上下文
- 不要催促它们开始实现
**如果审查者发现问题:**
- 由实现者(同一个子 Agent)修复
- 审查者再次审查
- 重复此过程,直至获批
- 不要跳过复审
**如果子 Agent 未能完成任务:**
- 派遣修复子 Agent,并提供具体指示
- 不要尝试手动修复(会污染上下文)
## 集成
**必需的工作流技能:**
- **superpowers:using-git-worktrees** - 确保工作区隔离(创建工作树或验证现有工作树)
- **superpowers:writing-plans** - 创建本技能所执行的计划
- **superpowers:requesting-code-review** - 用于最终整个分支审查的代码审查模板
- **superpowers:finishing-a-development-branch** - 在所有任务完成后完成开发工作
**子 Agent 应使用:**
@@ -1,139 +1,137 @@
# 实现子 Agent 提示模板 # 实现子 Agent 提示模板
派发实现子 Agent 时使用此模板。 派发实现子 Agent 时使用此模板。
``` ```
Subagent (general-purpose): 子 Agent(通用):
description: "实现任务 N[任务名称]" description: "实现任务 N[任务名称]"
model: [MODEL — 必填:按照 SKILL.md 的模型选择”章节选择;若省略, model: [MODEL — 必填:根据 SKILL.md 的模型选择规则选定;省略 model
将在不提示的情况下继承会话中成本最高的模型] 会悄悄继承当前会话中成本最高的模型]
prompt: | prompt: |
你正在实现任务 N: [task name] 你正在实现任务 N[任务名称]
## 任务描述 ## 任务说明
首先阅读你的任务简报:[BRIEF_FILE] 首先阅读你的任务简报:[BRIEF_FILE]
其中包含计划的完整任务文本。 其中包含计划的完整任务文本。
## 背景 ## 上下文
[背景说明:该任务所处位置、依赖和架构上下文] [背景说明:该任务所处位置、依赖关系和架构上下文]
## 开始之前 ## 开始之前
如果你对以下内容有疑问: 如果你对以下内容有疑问:
- 需求或验收标准 - 需求或验收标准
- 方法或实现策略 - 实现方法或策略
- 依赖项或假设 - 依赖项或假设
- 任务描述中任何不清楚的内容 - 任务说明中任何不清楚的地方
**现在就提问。** 开始工作前提出任何疑虑。 **现在就提问。** 开始工作前提出所有顾虑。
## 你的工作 ## 你的工作
明确需求后: 明确需求后:
1. 完全按照任务规定进行实现 1. 严格按照任务说明实现
2. 编写测试(如果任务要求,则遵循 TDD) 2. 编写测试(如果任务要求,则遵循 TDD)
3. 验证实现能够正常工作 3. 验证实现有效
4. 提交你的工作 4. 提交你的工作
5. 自我审查(见下文) 5. 自我审查(见下文)
6. 回报 6. 汇报结果
工作目录:[directory] 工作目录:[directory]
**工作期间:** 如果遇到意外情况或不清楚之处**请提问**。 **工作期间:** 如果遇到意外情况或不清楚的地方**请提问**。
随时暂停并澄清都完全没问题。不要猜测或作出假设。 随时暂停并澄清都没问题。不要猜测或自行作出假设。
迭代期间,针对你正在改的内容运行聚焦测试;提交前运行一次 迭代过程中,运行与你正在改的内容对应的聚焦测试;完整测试套件只需在提交前
完整测试套件,而不是每次编辑后都运行。 运行一次,而不是每次编辑后都运行。
## 代码组织 ## 代码组织
当代码能够一次性完整纳入你的上下文时,你对其推理的效果最佳;当文件职责 当代码规模小到可以一次纳入上下文时,你能更好地推理;文件职责集中时,你的
聚焦时,你的编辑也更可靠。请牢记以下几点 修改也会更可靠。请牢记:
- 遵循计划中定义的文件结构 - 遵循计划中定义的文件结构
- 每个文件都应只有一个明确职责,并具有定义清晰的接口 - 每个文件都应只有一项清晰职责,并提供定义明确的接口
- 如果你正在创建的文件增长程度超出计划意图,请停止并将其报告为 - 如果你正在创建的文件增长超出计划意图,请停止并以 DONE_WITH_CONCERNS
DONE_WITH_CONCERNS——没有计划指导的情况下,不要自行拆分文件 汇报——没有计划指引时,不要自行拆分文件
- 如果你正在修改的现有文件已经很大或结构混乱,请谨慎工作, - 如果你正在修改的现有文件本就庞大或纠缠,请谨慎处理,并在报告中将其记为顾虑
并在报告中将其注明为一项疑虑 - 在现有代码库中遵循既有模式。像一名优秀开发者那样改进你接触的代码,
- 在现有代码库中,遵循已确立的模式。像优秀开发者那样改进你正在接触的 但不要重构任务范围以外的内容。
代码,但不要重构任务范围之外的内容。
## 当任务超出你的能力范围 ## 当任务超出你的能力时
随时可以停下来并说“这对我来说太难了”。劣质工作还不如不做 随时可以停并说“这个任务对我来说太难了”。糟糕的工作比没有工作更糟
你不会因上报升级而受到惩罚。 升级求助不会受到惩罚。
**以下情况下必须停止并上报升级:** **遇到以下情况时,停止并升级:**
- 任务需要在多有效方之间作出架构决策 - 任务需要在多有效方之间作出架构决策
- 你需要理解超出已提供范围的代码,并且无法弄清楚 - 你需要理解未提供的代码,却无法找到清晰答案
- 你不确定自己的方法是否正确 - 你不确定自己的方法是否正确
- 任务涉及以计划未预的方式重构现有代码 - 任务涉及以计划未预的方式重构现有代码
- 你一直逐个阅读文件,试图理解系统,却没有取得进展 - 你接连阅读文件理解系统,却始终没有进展
**如何上报升级:** 回报时使用状态 BLOCKED 或 NEEDS_CONTEXT。具体说明 **升级方式** BLOCKED 或 NEEDS_CONTEXT 状态汇报。具体说明你卡在哪里、
你卡在哪里、尝试什么,以及需要哪种帮助。 尝试什么,以及需要哪种帮助。控制器可以提供更多上下文、使用能力更强的
控制器可以提供更多背景信息、使用能力更强的模型重新派发, 模型重新派发,或把任务拆分得更小。
或将任务拆分成更小的部分。
## 回报之前:自我审查 ## 汇报前:自我审查
全新的视角审查你的工作。问问自己: 全新的视角审查你的工作。问问自己:
**完整性:** **完整性:**
- 我是否完整实现了规格中的所有内容? - 我是否完整实现了规格中的所有内容?
- 我是否遗漏了任何求? - 我是否遗漏了任何求?
- 是否我未处理的边界情况? - 是否存在我未处理的边界情况?
**质量:** **质量:**
- 这是我能做到的最佳果吗? - 这是我能做到的最佳果吗?
- 名是否清晰准确(与事物所做的事情相符,而不是与其实现方式相符)? - 名是否清晰准确(反映事物做什么,而不是如何做)?
- 代码是否整洁且易于维护? - 代码是否整洁且易于维护?
**纪律:** **纪律:**
- 我是否避免了过度构建(YAGNI)? - 我是否避免了过度构建(YAGNI)?
- 我是否只构建了要求的内容? - 我是否只构建了要求的内容?
- 我是否遵循了代码库中的有模式? - 我是否遵循了代码库中的有模式?
**测试:** **测试:**
- 测试是否真正验证了行为(而不只是模拟行为)? - 测试是否真正验证了行为(而不只是模拟行为)?
- 如果要求使用 TDD,我是否遵循了 TDD - 如果要求,我是否遵循了 TDD
- 测试是否全面? - 测试是否全面?
- 测试输出是否完全干净(没有零散的警告或噪声)? - 测试输出是否干净(没有无关警告或噪声)?
如果在自我审查期间发现问题,请立即修复,然后再报。 如果在自我审查发现问题,请现在修复,然后再报。
## 审查发现问题 ## 收到审查发现
如果审查发现问题,而你修复了这些问题,请重新运行覆盖 如果任务审查发现问题,你会带着这些发现项被恢复。修复它们,重新运行覆盖
已修代码的测试,并将结果追加到报告文件中。审查者 已修代码的测试,并把修复报告追加到你的报告文件中:说明修改了什么、
不会替你重新运行测试——你的报告就是测试证据。 运行了哪些覆盖测试、所用命令以及输出。审查者不会替你重新运行测试——
你的报告就是测试证据。然后使用与首次报告相同的简短状态契约回复。
## 报告格式 ## 报告格式
将完整报告写入 [REPORT_FILE] 将完整报告写入 [REPORT_FILE]
- 实现了什么(如果被阻塞,则说明尝试了什么) - 实现了什么(如果被阻塞,则说明尝试了什么)
- 测试了什么以及测试结果 - 测试内容与测试结果
- **TDD Evidence**(如果此任务要求使用 TDD): - **TDD 证据**(如果此任务要求 TDD):
- RED行的命令、实现前相关失败输出,以及为什么该失败符合预期 - RED实现前运行的命令、相关失败输出,以及为该失败符合预期
- GREEN执行的命令以及实现后相关通过输出 - GREEN:实现后运行的命令和相关通过输出
- 改的文件 - 改的文件
- 自我审查发现的问题(如有) - 自我审查发现(如有)
- 任何问题或 - 任何问题或
然后仅回报以下内容(不超过 15 行——详细信息位于 然后仅使用以下内容回复(少于 15 行——详细内容保存在报告文件中):
报告文件中): - **Status:** DONE | DONE_WITH_CONCERNS | BLOCKED | NEEDS_CONTEXT
- **状态:** DONE | DONE_WITH_CONCERNS | BLOCKED | NEEDS_CONTEXT
- 创建的提交(短 SHA + 主题) - 创建的提交(短 SHA + 主题)
- 行测试摘要(例如“14/14 通过,输出完全干净”) - 行测试摘要(例如“14/14 通过,输出干净”)
- 你的虑(如有) - 你的虑(如有)
- 报告文件路径 - 报告文件路径
如果状态为 BLOCKED 或 NEEDS_CONTEXT,请具体情况写在最终消息 如果状态为 BLOCKED 或 NEEDS_CONTEXT,请具体情况直接写在最终消息中——
本身中——控制器会直接据此采取行动。 控制器会直接据此采取行动。
如果你完成工作但对正确性有疑虑,使用 DONE_WITH_CONCERNS。 完成工作但对正确性有疑虑,使用 DONE_WITH_CONCERNS。无法完成任务时,
如果你无法完成任务,请使用 BLOCKED。如果你需要未提供的 使用 BLOCKED。需要未提供的信息时,使用 NEEDS_CONTEXT。绝不要悄悄交付
信息,请使用 NEEDS_CONTEXT。绝不要在不说明的情况下交付你没有把握的工作。 自己没有把握的工作。
``` ```
@@ -0,0 +1,97 @@
# 限定范围复审子 Agent 提示模板
在一轮修复后派发复审时使用此模板。复审者要验证发现项是否已经解决,
并检查修复 diff 是否引入新的破坏。这不是一次全新审查——完整审查已经完成。
**目的:** 验证上一次审查的每个发现项都已解决,并确认修复本身没有破坏任何内容。
```
子 Agent(通用):
description: "复审任务 N 的第 R 轮修复"
model: [MODEL — 必填:根据 SKILL.md 的模型选择规则选定;省略 model
会悄悄继承当前会话中成本最高的模型]
prompt: |
你正在复审一个任务的一轮修复。先前的审查产生了一组发现项;实现者已经尝试
修复它们。你的工作是裁定每个发现项,并检查修复 diff——仅此而已。
## 任务
阅读任务简报:[BRIEF_FILE]
## 待验证的发现项
[FINDINGS]
## 修复
阅读实现者的报告(修复报告追加在末尾):
[REPORT_FILE]
**Fix base:** [FIX_BASE_SHA](上次审查看到的 head
**Head:** [HEAD_SHA]
**Diff 文件:** [DIFF_FILE]
只读取一次 diff 文件——它包含修复提交、统计摘要,以及带周边上下文的修复 diff。
不要重新运行 git 命令。如果 diff 文件缺失,请自行获取 diff:
`git diff --stat [FIX_BASE_SHA]..[HEAD_SHA]`
`git diff [FIX_BASE_SHA]..[HEAD_SHA]`
在此检出目录中,你的审查是只读的。不得以任何方式修改工作树、索引、HEAD
或分支状态。
## 范围
你的范围是发现项列表和修复 diff。必须裁定每一个发现项。检查修复 diff,
看修复本身是否引入了新问题。不要重新审查修复没有触及的代码:如果你注意到
一个完全位于修复 diff 之外的问题,请将其报告在“范围外观察项”下——它不会
阻塞此任务,也不会延长循环。所有任务完成后会进行一次广泛的整分支审查。
## 测试
实现者已经重新运行覆盖已修改代码的测试,并把结果追加到了报告文件中。
把报告视为未经验证的声明:确认修复报告写明了覆盖测试并展示其输出,
再根据 diff 验证这些声明。不要为了确认报告而重新运行测试套件。只有在阅读
代码时产生了现有运行结果无法解答的具体疑问,才运行测试——而且只能运行
聚焦测试,绝不运行包级完整套件。
## 输出格式
你的最终消息就是报告本身:直接从第一个发现项的裁定开始。每一行都必须是
裁定、带 file:line 的发现项,或你运行过的检查——不要写前言、过程叙述或
结尾总结。
### 发现项裁定
按“待验证的发现项”中的顺序,逐项报告:
- **[发现项单行摘要]** — ADDRESSED | NOT ADDRESSED,并附 file:line 证据。
“尝试过”不等于已解决:具体缺陷必须已经不存在。
### 修复 Diff 中的新破坏
修复本身破坏或引入的任何内容,附严重级别(Critical/Important/Minor
和 file:line。如果干净,则写“无”。
### 范围外观察项
你注意到的、完全位于修复 diff 之外的问题。它们不阻塞任务;控制器会将其
记入账本,供最终审查处理。如果没有,则写“无”。
### 裁定
**修复轮次:** [所有发现项均已解决,且没有新的 Critical/Important 破坏 |
仍有发现项未解决]——列出尚未解决的项目。
```
**占位符:**
- `[MODEL]` — 必填:根据 SKILL.md 的模型选择规则选择审查模型;对于小型
修复 diff 的限定范围复审,使用低成本至中等档位模型
- `[BRIEF_FILE]` — 任务简报文件(实现者使用的同一文件)
- `[FINDINGS]` — 上一次审查中的 Critical/Important 发现项和规格缺口,
逐字复制,每项一个项目符号
- `[REPORT_FILE]` — 实现者的报告文件(修复报告追加在其中)
- `[FIX_BASE_SHA]` — 上一次审查看到的 head
- `[HEAD_SHA]` — 当前提交
- `[DIFF_FILE]``scripts/review-package PLAN_FILE FIX_BASE HEAD` 打印出的路径
**复审者返回:** 每个发现项的裁定(ADDRESSED / NOT ADDRESSED)、修复 diff
中的新破坏、范围外观察项,以及该轮裁定。
@@ -4,26 +4,28 @@
# call. Using the recorded per-task BASE (not HEAD~1) keeps multi-commit # call. Using the recorded per-task BASE (not HEAD~1) keeps multi-commit
# tasks intact. # tasks intact.
# #
# Usage: review-package BASE HEAD [OUTFILE] # Usage: review-package PLAN_FILE BASE HEAD [OUTFILE]
# Default OUTFILE: <repo-root>/.superpowers/sdd/review-<base7>..<head7>.diff # Default OUTFILE: <repo-root>/.superpowers/sdd/<plan-basename>/review-<base7>..<head7>.diff
# (named per range, so a re-review after fixes gets a distinct fresh file). # (named per range, so a re-review after fixes gets a distinct fresh file).
set -euo pipefail set -euo pipefail
if [ $# -lt 2 ] || [ $# -gt 3 ]; then if [ $# -lt 3 ] || [ $# -gt 4 ]; then
echo "usage: review-package BASE HEAD [OUTFILE]" >&2 echo "usage: review-package PLAN_FILE BASE HEAD [OUTFILE]" >&2
exit 2 exit 2
fi fi
base=$1 plan=$1
head=$2 base=$2
head=$3
[ -f "$plan" ] || { echo "no such plan file: $plan" >&2; exit 2; }
git rev-parse --verify --quiet "$base" >/dev/null || { echo "bad BASE: $base" >&2; exit 2; } git rev-parse --verify --quiet "$base" >/dev/null || { echo "bad BASE: $base" >&2; exit 2; }
git rev-parse --verify --quiet "$head" >/dev/null || { echo "bad HEAD: $head" >&2; exit 2; } git rev-parse --verify --quiet "$head" >/dev/null || { echo "bad HEAD: $head" >&2; exit 2; }
if [ $# -eq 3 ]; then if [ $# -eq 4 ]; then
out=$3 out=$4
else else
dir=$("$(cd "$(dirname "$0")" && pwd)/sdd-workspace") dir=$("$(cd "$(dirname "$0")" && pwd)/sdd-workspace" "$plan")
out="$dir/review-$(git rev-parse --short "$base")..$(git rev-parse --short "$head").diff" out="$dir/review-$(git rev-parse --short "$base")..$(git rev-parse --short "$head").diff"
fi fi
@@ -1,22 +1,40 @@
#!/usr/bin/env bash #!/usr/bin/env bash
# Resolve and ensure the working-tree directory SDD uses for its short-lived # Resolve and ensure the working-tree directory SDD uses for one plan's
# artifacts: task briefs, implementer reports, review packages, and the # short-lived artifacts: task briefs, implementer reports, review packages,
# progress ledger. Print the directory's absolute path. # and the progress ledger. Print the plan directory's absolute path.
#
# One directory per plan (.superpowers/sdd/<plan-basename>/) so a follow-up
# plan in the same working tree can never read or overwrite another plan's
# artifacts. A stale ledger misread as current progress makes controllers
# skip whole task sequences — plan-scoping removes that failure structurally.
# #
# The workspace lives in the working tree (not under .git/) because Claude Code # The workspace lives in the working tree (not under .git/) because Claude Code
# treats .git/ as a protected path and denies agent writes there — which blocks # treats .git/ as a protected path and denies agent writes there — which blocks
# an implementer subagent from writing its report file. A self-ignoring # an implementer subagent from writing its report file. A self-ignoring
# .gitignore keeps the workspace out of `git status` and out of accidental # .gitignore at .superpowers/sdd/ keeps every plan's workspace out of
# commits without modifying any tracked file. # `git status` and out of accidental commits without modifying any tracked file.
# #
# Single source of truth for the workspace location, so task-brief and # Single source of truth for the workspace location, so task-brief and
# review-package cannot drift to different directories. # review-package cannot drift to different directories.
# #
# Usage: sdd-workspace # Usage: sdd-workspace PLAN_FILE
set -euo pipefail set -euo pipefail
if [ $# -ne 1 ]; then
echo "usage: sdd-workspace PLAN_FILE" >&2
exit 2
fi
plan=$1
[ -f "$plan" ] || { echo "no such plan file: $plan" >&2; exit 2; }
slug=$(basename "$plan" .md)
[ -n "$slug" ] && [ "$slug" != "." ] && [ "$slug" != ".." ] \
|| { echo "cannot derive a workspace name from: $plan" >&2; exit 2; }
root=$(git rev-parse --show-toplevel) root=$(git rev-parse --show-toplevel)
dir="$root/.superpowers/sdd" base="$root/.superpowers/sdd"
dir="$base/$slug"
mkdir -p "$dir" mkdir -p "$dir"
printf '*\n' > "$dir/.gitignore" printf '*\n' > "$base/.gitignore"
cd "$dir" && pwd cd "$dir" && pwd
@@ -4,8 +4,9 @@
# through the controller's context. # through the controller's context.
# #
# Usage: task-brief PLAN_FILE TASK_NUMBER [OUTFILE] # Usage: task-brief PLAN_FILE TASK_NUMBER [OUTFILE]
# Default OUTFILE: <repo-root>/.superpowers/sdd/task-<N>-brief.md # Default OUTFILE: <repo-root>/.superpowers/sdd/<plan-basename>/task-<N>-brief.md
# (per worktree; concurrent runs in the same working tree share it). # (per plan and per worktree; concurrent runs of the SAME plan in the same
# working tree share it).
set -euo pipefail set -euo pipefail
if [ $# -lt 2 ] || [ $# -gt 3 ]; then if [ $# -lt 2 ] || [ $# -gt 3 ]; then
@@ -20,7 +21,7 @@ n=$2
if [ $# -eq 3 ]; then if [ $# -eq 3 ]; then
out=$3 out=$3
else else
dir=$("$(cd "$(dirname "$0")" && pwd)/sdd-workspace") dir=$("$(cd "$(dirname "$0")" && pwd)/sdd-workspace" "$plan")
out="$dir/task-${n}-brief.md" out="$dir/task-${n}-brief.md"
fi fi
@@ -1,176 +1,155 @@
# 任务审查提示模板 # 任务审查子 Agent 提示模板
派发任务审查员子智能体时使用此模板。审查 派发任务审查子 Agent 时使用此模板。审查者只读取一次该任务的 diff
只读取一次任务的 diff并返回两个结论:规符合性和 并返回两项裁定:规符合度和代码质量。
代码质量。
**目的:** 验证单个任务的实现是否符合其要求(不多也不少),并确认实现质量良好
(整洁、有测试、可维护)。
**目的:** 验证单个任务的实现符合其要求(不多不少),
并且构建良好(整洁、经过测试、可维护)
``` ```
Subagent (general-purpose): 子 Agent(通用):
description: "审查任务 N(规格 + 质量)" description: "审查任务 N(规格 + 质量)"
model: [MODEL — 必填:按照 SKILL.md 的模型选择”章节选择;若省略, model: [MODEL — 必填:根据 SKILL.md 的模型选择规则选定;省略 model
将在不提示的情况下继承会话中成本最高的模型] 会悄悄继承当前会话中成本最高的模型]
prompt: | prompt: |
你正在审查一任务的实现:先判断它是否符合要求, 你正在审查一任务的实现:先判断它是否符合要求,再判断它的构建质量。
再判断其构建质量是否良好。这是一个任务范围内的关卡, 这是任务级门禁,而不是合并审查——所有任务完成后,会另行进行一次广泛的
而不是合并审查——所有任务完成后,会另行对整个分支进行 整分支审查。
广泛审查。
## 请求的内容 ## 要求实现的内容
使用 Read 阅读任务简报:[BRIEF_FILE] 阅读任务简报:[BRIEF_FILE]
规格/设计中此任务具有约束力的全局约束 规格/设计中约束此任务的全局要求
[GLOBAL_CONSTRAINTS] [GLOBAL_CONSTRAINTS]
## 实现者声称构建的内容 ## 实现者声称构建的内容
使用 Read 阅读实现者的报告:[REPORT_FILE] 阅读实现者的报告:[REPORT_FILE]
## 正在审查的差异 ## 待审查 Diff
**基准:** [BASE_SHA] **Base:** [BASE_SHA]
**头部:** [HEAD_SHA] **Head:** [HEAD_SHA]
**差异文件:** [DIFF_FILE] **Diff 文件:** [DIFF_FILE]
Read 一次差异文件——其中包含提交列表、统计摘要, 读取一次 diff 文件——包含提交列表、统计摘要,以及带周边上下文的完整
以及带有周围上下文的完整差异,它就是你审视此更改的依据。 diff;这就是你观察本次变更的视图。diff 的上下文行就是已更改文件的内容:
差异中的上下文行就是已更改文件本身:不要单独 Read 不要另行读取已更改文件,除非你必须判断的某个代码块恰好在函数中间被截断;
已更改文件,除非你必须判断的某个差异块在函数中途被截断—— 如需这样做,请在报告中说明。不要重新运行 git 命令。
并在报告中说明这一点。不要重新运行 git 命令。 如果 diff 文件缺失,请自行获取 diff:
如果差异文件缺失,请自行获取差异:
`git diff --stat [BASE_SHA]..[HEAD_SHA]``git diff [BASE_SHA]..[HEAD_SHA]` `git diff --stat [BASE_SHA]..[HEAD_SHA]``git diff [BASE_SHA]..[HEAD_SHA]`
不要遍历更广泛的代码库。仅为评估一个你能够明确指出的具体风险 不要遍历更广泛的代码库。只有在评估一个你能够明确指出的具体风险时,才检查
才检查差异之外的代码——每个已指出的风险只进行一次聚焦检查, diff 之外的代码——每个已命名风险只一次聚焦检查,并在报告中同时写明风险
并在报告中同时说明该风险以及你检查了什么。 和所检查的内容。横切变更是合理的已命名风险:如果 diff 改变了锁顺序、函数
横切更改是合理的具名风险:如果差异更改了 或 API 契约,或共享可变状态,检查调用点就是正确做法。
锁顺序、函数或 API 契约,或者共享可变状态,
那么检查调用点就是正确的方法。
你对此 checkout 的审查是只读的。不得以任何方式改工作树、 在此检出目录中,你的审查是只读的。不得以任何方式改工作树、索引、HEAD
索引、HEAD 或分支状态。 或分支状态。
## 不要相信报告 ## 不要相信报告
将实现者的报告视为关于代码的、未经核实的声称。该报告
可能不完整、不准确或过于乐观。请对照 diff 核实这些声称 把实现者的报告视为尚未验证的代码声明。它可能不完整、不准确或过于乐观。
报告中的设计理由同样是声称:"出于 YAGNI 考虑而保留原样" 根据 diff 验证这些声明。报告中的设计理由同样是声明:“按照 YAGNI 保持不变”
"刻意保持简单"或任何其他解,都是实现者在给自己的工作分。 刻意保持简单或任何其他解,都是实现者在给自己的工作分。根据代码本身
根据代码本身的优劣来评判——陈述理由绝不能降低发现项的严重程度 的实际质量作出判断——一条书面理由永远不会降低发现项的严重级别
## 测试 ## 测试
实现者已经运行了测试,并针对这份代码本身提供了包含 TDD 实现者已经针对这份代码运行测试,并报告了结果及 TDD 证据。不要为了确认其
证据的结果报告。不要重新运行测试套件来确认其报告。只有阅读代码时 报告而重新运行测试套件。只有阅读代码时产生了现有运行结果无法解答的具体
产生了某个具体疑问,且已有运行均未解答该疑问时,才运行测试——且只运行 疑问,才运行测试——且只运行聚焦测试,绝不运行包级完整套件、竞态检测器
有针对性的测试,绝不要运行整个包的测试套件、竞态检测器或重复/高次数循环。 或重复/高次数循环。如果看起来有必要进行重型验证,请在报告中提出建议,
如果看起来有必要进行重度验证,请在报告中建议这样做,而不是 而不是亲自运行。如果当前环境无法运行命令,请写明你本来会运行的测试。
自行运行。如果你无法在此环境中运行命令,请指出你会运行的
测试。
实现者报告的测试输出中警告或其他噪声 实现者报告的测试输出中若有警告或其他噪声,也属于发现项——测试输出应当干净。
均属于发现项——测试输出应当完全干净。
## 第 1 部分:规符合 ## 第 1 部分:规符合
将 diff 与请求内容进行比较: 将 diff 与“要求实现的内容进行比较:
- **缺失:**他们跳过、遗漏或声称完成但并未 - **Missing**跳过、遗漏或声称完成但实际没有实现的要求
实现的要求 - **Extra** 未要求的功能、过度工程、不必要的“锦上添花”
- **额外:**未被请求的功能、过度工程化、不必要的 - **Misunderstood** 用错误方式构建了正确功能,或解决了错误的问题
“锦上添花”
- **误解:**以错误方式构建了正确功能,或解决了错误的
问题
如果某项要求无法仅此 diff 验证(它位于 如果某项要求无法仅根据此 diff 验证(它位于未更改代码中,或跨越多个任务),
未变更的代码中或跨越多个任务),请将其报告为一个 ⚠️ 项,而不要 请将其报告为 ⚠️ 条目,不要扩大搜索范围。
扩大搜索范围。
## 第 2 部分:代码质量 ## 第 2 部分:代码质量
**代码质量:** **代码质量:**
- 关注点是否清晰分离? - 关注点是否清晰分离?
- 错误处理是否当? - 错误处理是否当?
- 是否遵循 DRY 原则且没有过早抽象? - 是否遵循 DRY,同时避免过早抽象?
- 是否处理了边界情况? - 是否处理了边界情况?
**测试:** **测试:**
- 新增和修改的测试是否验证真实行为,而非模拟对象 - 新增和修改的测试是否验证真实行为,而不是 mock
- 是否覆盖了任务的边界情况? - 是否覆盖了任务的边界情况?
**结构:** **结构:**
- 每个文件是否有一项明确的职责和一个定义良好的接口? - 每个文件是否有一项清晰职责,并提供定义明确的接口?
- 各单元是否经过拆分,以便能够独立理解和测试? - 各单元是否已分解到能够独立理解和测试?
- 实现是否遵循计划中的文件结构? - 实现是否遵循计划中的文件结构?
- 此更是否创建了体量已经很大的新文件,或 - 此次变更是否创建了已经很大的新文件,或显著增大了现有文件?
显著增大了现有文件?(不要因更改前就已存在的文件 (不要标记变更前就已存在的文件大小问题——只关注此次变更带来的内容。)
大小而提出问题——重点关注此次更改带来的影响。)
你的报告应指证据:每发现以及任何你原本只会用一个简单 报告应指证据:每发现项都要提供 file:line 引用;任何本来只能回答“是”
"yes." 回答的检查,要提供 file:line 引用。一份引用了具体行号的 检查,要提供引用。带有精确行号的紧凑报告能向控制器提供所需的一切。
精炼报告能为控制器提供其所需的一切。
你的最终消息就是报告本身:直接以规范符合性判定开头。 你的最终消息就是报告本身:直接从规格符合度裁定开始。每一行都必须是裁定、
每一行都应是一个判定、一项带有 file:line 的发现,或一项你执行过的 file:line 的发现,或你运行过的检查——不要写前言、过程叙述或结尾总结。
检查——不要有前言,不要叙述过程,也不要有结尾总结。
## 校准 ## 校准
按实际严重程度对问题进行分类。并非所有问题都是“严重”。
“重要”意味着,在相关问题修复之前,不能信任这项任务:存在错误 按实际严重程度对问题分类。不是所有问题都是 Critical。
或脆弱的行为、遗漏要求,或足以让你阻止合并的可维护性 Important 表示在修复之前不能信任此任务:行为错误或脆弱、遗漏要求,或你会因此
损害——原样重复某个逻辑块、 阻止合并的可维护性损害——逐字重复一块逻辑、吞掉错误、没有作出任何断言的测试。
被吞掉的错误、没有断言任何内容的测试。"覆盖范围可以更广" “覆盖可以更广”和润色建议属于 Minor。
和打磨建议属于“次要”。 如果计划或简报明确要求了本准则认定为缺陷的内容(没有断言的测试、逐字重复的
如果计划或简报明确要求了某项被本准则称为缺陷的内容 逻辑块),它仍然是发现项——应以 Important 报告,并标记为 plan-mandated。
(没有断言任何内容的测试、原样重复某个逻辑块),那确实就是一项 计划作者不能给自己的工作打分;由人类决定。
审查发现——将其作为“重要”问题报告,并标注为 列出问题前,先肯定做得好的地方——准确的肯定能帮助实现者信任其余反馈。
“计划要求”。计划作者不能给自己的工作评分;由人类
决定。
在列出问题之前,先肯定做得好的地方——准确的表扬
有助于实施者信任其余反馈。
## 输出格式 ## 输出格式
### 规格符合 ### 规格符合
- ✅ 符合规格 | ❌ 发现问题:[缺少多余误解的内容, - ✅ 符合规格 | ❌ 发现问题:[缺少/多余/误解的内容,附 file:line 引用]
附 file:line 引用] - ⚠️ 无法从 diff 验证:[无法仅从 diff 验证的要求,以及控制器应检查什么——
- ⚠️ 无法从差异中验证:[仅凭差异无法验证的要求,以及控制器应检查的 与对其他可验证内容给出的 ✅/❌ 裁定一同报告]
内容——与所有可验证内容的 ✅/❌ 结论一起报告]
### 优点 ### 优点
[哪些地方做得好?请具体说明。] [哪些地方做得好?请具体说明。]
### 问题 ### 问题
#### 严重(必须修复) #### Critical(必须修复)
#### 重要(应当修复) #### Important(应当修复)
#### 次要(建议改进) #### Minor(可选改进)
每个问题:file:line、问题所在、它为何重要、如何修复 对每个问题说明file:line、问题是什么、为何重要、如何修复(如果方法不明显)。
(如果修复方法并不显而易见)。
### 评估 ### 评估
**任务质量:** [通过 | 需要修复] **任务质量:** [已批准 | 需要修复]
**理由:** [12 句技术评估] **理由:** [1-2 句技术评估]
``` ```
**占位符:** **占位符:**
- `[MODEL]` — 必填:按照 SKILL.md 的模型选择”章节选择审查模型 - `[MODEL]` — 必填:根据 SKILL.md 的模型选择规则选择审查模型
- `[BRIEF_FILE]` — 必填:任务简报文件(`scripts/task-brief PLAN N` - `[BRIEF_FILE]` — 必填:任务简报文件(`scripts/task-brief PLAN N`
会打印路径;即实现者据以开展工作的同一文件) 会打印路径;这也是实现者使用的同一文件)
- `[GLOBAL_CONSTRAINTS]` — 从计划的 Global Constraints 章节或规范中 - `[GLOBAL_CONSTRAINTS]` — 从计划的“全局约束”部分或规格中逐字复制、
逐字复制的、具有约束力的要求:精确的值、格式,以及明确说明的组件间关系 且对此任务有约束力的要求:准确值、格式及组件间明示的关系
(不是流程规则——这些规则已包含在此模板中 (不是流程规则——本模板已经包含这些规则)
- `[REPORT_FILE]` — 必填:实现者将其详细报告写入的文件 - `[REPORT_FILE]` — 必填:实现者写入详细报告的文件
- `[BASE_SHA]` — 此任务前的提交 - `[BASE_SHA]` — 此任务开始前的提交
- `[HEAD_SHA]` — 当前提交 - `[HEAD_SHA]` — 当前提交
- `[DIFF_FILE]` — 必填:控制器审查包写入的路径 - `[DIFF_FILE]` — 必填:控制器写入审查包的路径
`scripts/review-package BASE HEAD` 会打印写入的唯一文件路径; `scripts/review-package PLAN_FILE BASE HEAD` 会打印写入的唯一
该审查包不会进入控制器的上下文) 路径;该审查包永远不会进入控制器的上下文)
**审查者返回:** 规格符合性结论(✅/❌/⚠️)、优点、问题 **审查者返回:** 规格符合度裁定(✅/❌/⚠️)、优点、问题
严重/重要/次要)、任务质量结论 Critical/Important/Minor)和任务质量裁定
一次修复派发可以同时处理规范缺口和质量发现;
修复后的复审涵盖两项裁定。
@@ -7,8 +7,6 @@ description: 在遇到任何缺陷、测试失败或意外行为时使用,且
## 概述 ## 概述
随机尝试修复既浪费时间,又会制造新的缺陷。快速打补丁会掩盖根本问题。
**核心原则:** 在尝试修复之前,始终先找出根因。只修复症状就是失败。 **核心原则:** 在尝试修复之前,始终先找出根因。只修复症状就是失败。
**违反此流程的字面规定,就是违背调试的精神。** **违反此流程的字面规定,就是违背调试的精神。**
@@ -185,6 +183,7 @@ description: 在遇到任何缺陷、测试失败或意外行为时使用,且
- 测试现在通过了吗? - 测试现在通过了吗?
- 是否没有破坏其他测试? - 是否没有破坏其他测试?
- 问题是否确实已解决? - 问题是否确实已解决?
- 在宣称成功之前,使用 `superpowers:verification-before-completion` 技能
4. **如果修复不起作用** 4. **如果修复不起作用**
- 停止 - 停止
@@ -277,15 +276,3 @@ description: 在遇到任何缺陷、测试失败或意外行为时使用,且
- **`root-cause-tracing.md`** - 沿调用栈反向追踪 bug,以找到最初的触发因素 - **`root-cause-tracing.md`** - 沿调用栈反向追踪 bug,以找到最初的触发因素
- **`defense-in-depth.md`** - 找到根本原因后,在多个层级添加验证 - **`defense-in-depth.md`** - 找到根本原因后,在多个层级添加验证
- **`condition-based-waiting.md`** - 使用条件轮询替代任意超时 - **`condition-based-waiting.md`** - 使用条件轮询替代任意超时
**相关技能:**
- **superpowers:test-driven-development** - 用于创建失败的测试用例(第 4 阶段,第 1 步)
- **superpowers:verification-before-completion** - 在宣称成功之前验证修复确实有效
## 实际影响
根据调试会话:
- 系统化方法:15-30 分钟修复
- 随机修复方法:反复折腾 2-3 小时
- 首次修复成功率:95% 对比 40%
- 引入的新 bug:几乎为零 对比 很常见
@@ -18,9 +18,18 @@ echo "🔍 Searching for test that creates: $POLLUTION_CHECK"
echo "Test pattern: $TEST_PATTERN" echo "Test pattern: $TEST_PATTERN"
echo "" echo ""
# Get list of test files # Get list of test files (find . emits ./-prefixed paths, so accept the
TEST_FILES=$(find . -path "$TEST_PATTERN" | sort) # pattern written with or without a leading ./)
TOTAL=$(echo "$TEST_FILES" | wc -l | tr -d ' ') TEST_PATTERN="${TEST_PATTERN#./}"
# find -path can't match '**/' against zero directory levels, so a pattern
# like src/**/*.test.ts would skip src/top.test.ts; also try the pattern
# with '**/' collapsed to cover files directly under the base directory.
TEST_FILES=$(find . \( -path "./$TEST_PATTERN" -o -path "./${TEST_PATTERN//\*\*\//}" \) | sort -u)
if [ -z "$TEST_FILES" ]; then
TOTAL=0
else
TOTAL=$(printf '%s\n' "$TEST_FILES" | wc -l | tr -d ' ')
fi
echo "Found $TOTAL test files" echo "Found $TOTAL test files"
echo "" echo ""
@@ -7,9 +7,9 @@ description: 在实现任何功能或修复任何错误时使用,且应在编
## 概述 ## 概述
先编写测试。看它失败。编写使其通过所需的最少代码。 先编写测试。亲眼看它失败。编写使其通过所需的最少代码。
**核心原则:** 如果你没有亲眼看到测试失败,就不知道它是否测试了正确的内容。 **核心原则:** 如果你没有亲眼看到测试失败,就不知道它是否测试了正确的内容。
**违反规则的字面要求,就是违反规则的精神。** **违反规则的字面要求,就是违反规则的精神。**
@@ -45,6 +45,7 @@ description: 在实现任何功能或修复任何错误时使用,且应在编
从测试出发重新实现。就这样。 从测试出发重新实现。就这样。
## 红-绿-重构 ## 红-绿-重构
```dot ```dot
digraph tdd_cycle { digraph tdd_cycle {
rankdir=LR; rankdir=LR;
@@ -67,17 +68,17 @@ digraph tdd_cycle {
} }
``` ```
### 红灯 - 编写失败测试 ### RED - 编写失败测试
编写一个最小测试,展示应发生什么。 编写一个最小测试,展示应发生什么。
<Good> <Good>
```typescript ```typescript
test('失败的操作重试 3 次', async () => { test('retries failed operations 3 times', async () => {
let attempts = 0; let attempts = 0;
const operation = () => { const operation = () => {
attempts++; attempts++;
if (attempts < 3) throw new Error('失败'); if (attempts < 3) throw new Error('fail');
return 'success'; return 'success';
}; };
@@ -92,7 +93,7 @@ test('失败的操作重试 3 次', async () => {
<Bad> <Bad>
```typescript ```typescript
test('重试有效', async () => { test('retry works', async () => {
const mock = jest.fn() const mock = jest.fn()
.mockRejectedValueOnce(new Error()) .mockRejectedValueOnce(new Error())
.mockRejectedValueOnce(new Error()) .mockRejectedValueOnce(new Error())
@@ -109,7 +110,8 @@ test('重试有效', async () => {
- 名称清晰 - 名称清晰
- 真实代码(除非不可避免,否则不使用 mock) - 真实代码(除非不可避免,否则不使用 mock)
### 验证红灯 - 观察它失败 ### 验证 RED - 观察它失败
**强制要求。绝不跳过。** **强制要求。绝不跳过。**
```bash ```bash
@@ -139,7 +141,7 @@ async function retryOperation<T>(fn: () => Promise<T>): Promise<T> {
if (i === 2) throw e; if (i === 2) throw e;
} }
} }
throw new Error('不可达'); throw new Error('unreachable');
} }
``` ```
恰好足以通过 恰好足以通过
@@ -155,7 +157,7 @@ async function retryOperation<T>(
onRetry?: (attempt: number) => void; onRetry?: (attempt: number) => void;
} }
): Promise<T> { ): Promise<T> {
// 你不会需要它(YAGNI // YAGNI
} }
``` ```
过度设计 过度设计
@@ -163,7 +165,7 @@ async function retryOperation<T>(
不要添加功能、重构其他代码,也不要做超出测试要求的“改进”。 不要添加功能、重构其他代码,也不要做超出测试要求的“改进”。
### 验证 GREEN - 亲眼看它通过 ### 验证 GREEN - 观察它通过
**强制要求。** **强制要求。**
@@ -194,75 +196,33 @@ npm test path/to/test.test.ts
为下一个功能编写下一个失败测试。 为下一个功能编写下一个失败测试。
## 好测试 ## 好测试
| 质量 | 好 | 差 | | 质量 | 好 | 差 |
|---------|------|-----| |---------|------|-----|
| **最小化** | 一件事。名称里有“和”?拆开它。 | `test('验证电子邮件、域和空白字符')` | | **最小化** | 一件事。名称里有“和”?拆开它。 | `test('validates email and domain and whitespace')` |
| **清晰** | 名称描述行为 | `test('test1')` | | **清晰** | 名称描述行为 | `test('test1')` |
| **体现意图** | 展示期望的 API | 掩盖代码应做什么 | | **体现意图** | 展示期望的 API | 掩盖代码应做什么 |
## 为什么顺序很重要 编写或修改任何测试时,请阅读 [writing-good-tests.md](writing-good-tests.md),了解让测试保持诚实的规则:
- 在编写测试前,指出哪项生产代码变更会使它失败
**“我会在之后编写测试来验证它是否有效”** - 断言真实行为,绝不断言 mock 行为
- 仅供测试使用的代码应放在测试工具中,不得放入生产类
代码写完后再编写的测试会立即通过。立即通过证明不了任何事情: - 在 mock 某个依赖项之前,先了解它的副作用
- 可能测试了错误的内容
- 可能测试的是实现,而不是行为
- 可能遗漏你忘记的边界情况
- 你从未看到它捕获缺陷
测试先行会迫使你看到测试失败,从而证明它确实测试了某些内容。
**“我已经手动测试了所有边界情况”**
手动测试是临时随意的。你以为自己测试了所有内容,但是:
- 没有关于测试内容的记录
- 代码变更时无法重新运行
- 在压力下很容易忘记某些情况
- “我试的时候它能用” ≠ 全面测试
自动化测试是系统化的。它们每次都以相同方式运行。
**“删除 X 小时的工作成果是一种浪费”**
沉没成本谬误。时间已经花掉了。你现在的选择是:
- 删除并使用 TDD 重写(再花 X 小时,信心高)
- 保留它并在之后添加测试(30 分钟,信心低,很可能有缺陷)
真正的“浪费”是保留你无法信任的代码。没有真正测试的可运行代码就是技术债务。
**“TDD 很教条,务实意味着灵活应变”**
TDD 本身就是务实的:
- 在提交前发现缺陷(比事后调试更快)
- 防止回归(测试会立即捕获破坏)
- 记录行为(测试展示如何使用代码)
- 支持重构(可以自由修改,测试会捕获破坏)
“务实的”捷径 = 在生产环境中调试 = 更慢。
**“事后测试能实现相同的目标——重要的是精神,而不是仪式”**
不。事后测试回答“这做了什么?”测试先行回答“这应该做什么?”
事后测试会受到你的实现的影响。你测试的是自己构建的内容,而不是需求所要求的内容。你验证的是自己记得的边界情况,而不是发现的边界情况。
测试先行会迫使你在实现之前发现边界情况。事后测试验证的是你是否记住了所有内容(你没有)。
事后编写 30 分钟的测试 ≠ TDD。你获得了覆盖率,却失去了测试确实有效的证明。
## 常见合理化借口 ## 常见合理化借口
| 借口 | 现实 | | 借口 | 现实 |
|--------|---------| |--------|---------|
| “太简单了,不值得测试” | 简单代码也会出错。测试只需 30 秒。 | | “太简单了,不值得测试” | 简单代码也会出错。测试只需 30 秒。 |
| “我之后再测试” | 测试立即通过什么也证明不了。 | | “我之后再测试” | 事后编写的测试立即通过——这什么也证明不了。它们可能测试了错误内容、测试实现而非行为,或漏掉你忘记的边界情况。你从未看过它失败,因此从未证明它能捕获缺陷。测试先行会强制出现这次失败。 |
| “事后测试也能达到相同目标” | 事后测试 = “这是做什么?” 测试先行 = “这应该做什么?” | | “事后测试能实现相同目标(重精神而非仪式)” | 事后测试回答“这做了什么?”测试先行回答“这应该做什么?”事后测试会被已写好的代码影响——你验证的是记得的情况,而不是本可发现的情况。只有覆盖率,没有测试确实有效的证明。 |
| “已经手动测试过了” | 临时测试 ≠ 系统化测试。没有记录,无法重新运行。 | | “已经手动测试过了” | 手动测试是临时随意的:没有覆盖记录,代码变更后无法重跑,压力下很容易漏掉情况。“我试的时候能用”并不等于全面。自动化测试每次都以相同方式运行。 |
| “删除 X 小时的成果太浪费了” | 沉没成本谬误。保留未经验证的代码就是技术债务。 | | “删除 X 小时的成果太浪费了” | 沉没成本谬误——无论如何,那些时间都已经花掉。真正的选择是:使用 TDD 重写(信心高),或保留代码再补测试(信心低,很可能有缺陷)。保留无法信任的代码才是浪费。 |
| “保留作为参考,先写测试” | 你会改编它。那就是事后测试。删除就是删除。 | | “保留作为参考,先写测试” | 你会改编它。那就是事后测试。删除就是删除。 |
| “需要先探索” | 可以。丢弃探索成果,从 TDD 开始。 | | “需要先探索” | 可以。丢弃探索成果,从 TDD 开始。 |
| “测试很难 = 设计不清晰” | 倾听测试。难以测试 = 难以使用。 | | “测试很难 = 设计不清晰” | 倾听测试。难以测试 = 难以使用。 |
| “TDD 会拖慢我的速度” | TDD 比调试更快。务实 = 测试先行。 | | “TDD 会拖慢我的速度” | TDD 就是务实路径:在提交前发现缺陷、防止回归,并让你无惧重构。“务实的”捷径意味着在生产环境中调试——那更慢,而不是更快。 |
| “手动测试更快” | 手动测试无法证明边情况。每次变更后你都得重新测试。 | | “手动测试更快” | 手动测试不能证明边情况。每次变更后都要重新测试。 |
| “现有代码没有测试” | 你正在改进它。为现有代码添加测试。 | | “现有代码没有测试” | 你正在改进它。为现有代码添加测试。 |
## 红旗项——停止并从头开始 ## 红旗项——停止并从头开始
@@ -284,42 +244,44 @@ TDD 本身就是务实的:
**所有这些都意味着:删除代码。使用 TDD 从头开始。** **所有这些都意味着:删除代码。使用 TDD 从头开始。**
## 示例:错误修复 ## 示例:错误修复
**缺陷:** 接受空电子邮件地址 **缺陷:** 接受空电子邮件地址
**红灯** **RED**
```typescript ```typescript
test('拒绝空电子邮件地址', async () => { test('rejects empty email', async () => {
const result = await submitForm({ email: '' }); const result = await submitForm({ email: '' });
expect(result.error).toBe('电子邮件为必填项'); expect(result.error).toBe('Email required');
}); });
``` ```
**验证红灯** **验证 RED**
```bash ```bash
$ npm test $ npm test
FAIL: 预期为 '电子邮件为必填项',实际得到 undefined FAIL: expected 'Email required', got undefined
``` ```
**绿灯** **GREEN**
```typescript ```typescript
function submitForm(data: FormData) { function submitForm(data: FormData) {
if (!data.email?.trim()) { if (!data.email?.trim()) {
return { error: '电子邮件为必填项' }; return { error: 'Email required' };
} }
// ... // ...
} }
``` ```
**验证绿灯** **验证 GREEN**
```bash ```bash
$ npm test $ npm test
PASS PASS
``` ```
**重构** **REFACTOR**
如有需要,提取针对多个字段的验证逻辑。 如有需要,提取针对多个字段的验证逻辑。
## 验证清单 ## 验证清单
在将工作标记为完成之前: 在将工作标记为完成之前:
- [ ] 每个新函数/方法都有测试 - [ ] 每个新函数/方法都有测试
@@ -328,7 +290,7 @@ PASS
- [ ] 编写了使每个测试通过所需的最少代码 - [ ] 编写了使每个测试通过所需的最少代码
- [ ] 所有测试均通过 - [ ] 所有测试均通过
- [ ] 输出干净无瑕(无错误、无警告) - [ ] 输出干净无瑕(无错误、无警告)
- [ ] 测试使用真实代码(仅在无法避免时使用模拟对象 - [ ] 测试使用真实代码(仅在无法避免时使用 mock
- [ ] 已覆盖边界情况和错误 - [ ] 已覆盖边界情况和错误
无法勾选所有复选框?你跳过了 TDD。重新开始。 无法勾选所有复选框?你跳过了 TDD。重新开始。
@@ -339,7 +301,7 @@ PASS
|---------|----------| |---------|----------|
| 不知道如何测试 | 写出期望的 API。先写断言。询问你的人类伙伴。 | | 不知道如何测试 | 写出期望的 API。先写断言。询问你的人类伙伴。 |
| 测试过于复杂 | 设计过于复杂。简化接口。 | | 测试过于复杂 | 设计过于复杂。简化接口。 |
| 必须模拟所有东西 | 代码耦合过紧。使用依赖注入。 | | 必须 mock 所有东西 | 代码耦合过紧。使用依赖注入。 |
| 测试设置过于庞大 | 提取辅助函数。仍然复杂?简化设计。 | | 测试设置过于庞大 | 提取辅助函数。仍然复杂?简化设计。 |
## 调试集成 ## 调试集成
@@ -348,13 +310,6 @@ PASS
绝不要在没有测试的情况下修复缺陷。 绝不要在没有测试的情况下修复缺陷。
## 测试反模式
添加模拟对象或测试工具时,请阅读 [testing-anti-patterns.md](testing-anti-patterns.md),以避免常见陷阱:
- 测试模拟行为而非真实行为
- 向生产类添加仅供测试使用的方法
- 在不了解依赖关系的情况下进行模拟
## 最终规则 ## 最终规则
``` ```
@@ -1,293 +0,0 @@
# 测试反模式
**在以下情况下加载此参考:** 编写或修改测试、添加 mock,或想要向生产代码中添加仅供测试使用的方法时。
## 概述
测试必须验证真实行为,而不是 mock 行为。mock 是用于隔离的手段,而不是被测试的对象。
**核心原则:** 测试代码做了什么,而不是 mock 做了什么。
**严格遵循 TDD 可防止这些反模式。**
## 铁律
```
1. 绝不测试 mock 行为
2. 绝不向生产类添加仅供测试使用的方法
3. 绝不在不了解依赖项的情况下进行 mock
```
## 反模式 1:测试 Mock 行为
**违规做法:**
```typescript
// ❌ 错误:测试 mock 是否存在
test('渲染侧边栏', () => {
render(<Page />);
expect(screen.getByTestId('sidebar-mock')).toBeInTheDocument();
});
```
**为什么这是错误的:**
- 你验证的是 mock 是否有效,而不是组件是否有效
- mock 存在时测试通过,不存在时测试失败
- 这无法告诉你任何有关真实行为的信息
**你的人类伙伴的纠正:** “我们是在测试 mock 的行为吗?”
**修复方法:**
```typescript
// ✅ 正确:测试真实组件,或者不要对其进行 mock
test('渲染侧边栏', () => {
render(<Page />); // 不要 mock 侧边栏
expect(screen.getByRole('navigation')).toBeInTheDocument();
});
// 或者,如果必须对侧边栏进行 mock 以实现隔离:
// 不要对 mock 进行断言——测试侧边栏存在时 Page 的行为
```
### 门函数
```
在对任何 mock 元素进行断言之前:
问:“我是在测试真实的组件行为,还是仅仅测试 mock 是否存在?”
如果测试的是 mock 是否存在:
停止——删除该断言,或取消对该组件的 mock
改为测试真实行为
```
## 反模式 2:生产代码中的仅供测试使用的方法
**违规:**
```typescript
// ❌ 错误:destroy() 仅在测试中使用
class Session {
async destroy() { // 看起来像生产 API
await this._workspaceManager?.destroyWorkspace(this.id);
// ... 清理
}
}
// 在测试中
afterEach(() => session.destroy());
```
**为什么这是错误的:**
- 生产类被仅供测试使用的代码污染
- 如果在生产环境中被意外调用,会很危险
- 违反 YAGNI 原则和关注点分离原则
- 混淆了对象生命周期与实体生命周期
**修复方法:**
```typescript
// ✅ 正确:由测试工具处理测试清理
// Session 没有 destroy()——它在生产环境中是无状态的
// 在 test-utils/ 中
export async function cleanupSession(session: Session) {
const workspace = session.getWorkspaceInfo();
if (workspace) {
await workspaceManager.destroyWorkspace(workspace.id);
}
}
// 在测试中
afterEach(() => cleanupSession(session));
```
### 门函数
```
在向生产类添加任何方法之前:
询问:“这是否仅供测试使用?”
如果是:
停止——不要添加它
改为将它放入测试工具中
询问:“这个类是否拥有此资源的生命周期?”
如果不是:
停止——这个方法不属于这个类
```
## 反模式 3:在不了解的情况下进行模拟
**违规做法:**
```typescript
// ❌ 不佳:模拟破坏了测试逻辑
test('检测重复服务器', () => {
// 模拟阻止了测试所依赖的配置写入!
vi.mock('ToolCatalog', () => ({
discoverAndCacheTools: vi.fn().mockResolvedValue(undefined)
}));
await addServer(config);
await addServer(config); // 应该抛出异常——但不会!
});
```
**为什么这是错误的:**
- 被模拟的方法具有测试所依赖的副作用(写入配置)
- 为了“保险”而过度模拟会破坏实际行为
- 测试会因错误的原因通过,或以令人费解的方式失败
**修复方法:**
```typescript
// ✅ 良好:在正确层级进行模拟
test('检测重复服务器', () => {
// 模拟耗时的部分,保留测试所需的行为
vi.mock('MCPServerManager'); // 只模拟耗时的服务器启动过程
await addServer(config); // 配置已写入
await addServer(config); // 检测到重复项 ✓
});
```
### 门函数
```
在模拟任何方法之前:
停下——暂时不要模拟
1. 问:“真实方法有哪些副作用?”
2. 问:“此测试是否依赖其中的任何副作用?”
3. 问:“我是否完全理解此测试需要什么?”
如果依赖副作用:
在更低层级进行模拟(实际耗时的操作/外部操作)
或使用能够保留必要行为的测试替身
不要模拟测试所依赖的高层方法
如果不确定测试依赖什么:
务必先使用真实实现运行测试
观察实际必须发生什么
然后才在正确层级添加最少量的模拟
危险信号:
- “为了保险,我把这个模拟掉”
- “这可能会很慢,最好模拟掉”
- 尚未理解依赖链就进行模拟
```
## 反模式 4:不完整的模拟
**违规行为:**
```typescript
// ❌ 错误:部分模拟——只包含你认为需要的字段
const mockResponse = {
status: 'success',
data: { userId: '123', name: '爱丽丝' }
// 缺少:下游代码使用的 metadata
};
// 稍后:当代码访问 response.metadata.requestId 时发生错误
```
**为什么这是错误的:**
- **部分模拟会掩盖结构性假设**——你只模拟了自己知道的字段
- **下游代码可能依赖你未包含的字段**——静默失败
- **测试通过,但集成失败**——模拟不完整,真实 API 是完整的
- **虚假的信心**——测试完全无法证明真实行为
**铁律:**模拟现实中实际存在的完整数据结构,而不只是当前测试使用的字段。
**修复方法:**
```typescript
// ✅ 正确:与真实 API 的完整性保持一致
const mockResponse = {
status: 'success',
data: { userId: '123', name: '爱丽丝' },
metadata: { requestId: 'req-789', timestamp: 1234567890 }
// 真实 API 返回的所有字段
};
```
### 门函数
```
创建模拟响应之前:
检查:“真实 API 响应包含哪些字段?”
操作:
1. 检查文档/示例中的实际 API 响应
2. 包含系统可能在下游使用的所有字段
3. 验证模拟是否与真实响应模式完全匹配
关键要求:
如果你正在创建模拟,就必须理解整个结构
当代码依赖被省略的字段时,部分模拟会静默失败
如果不确定:包含所有已记录在文档中的字段
```
## 反模式 5:把集成测试当作事后补充
**违规做法:**
```
✅ 实现完成
❌ 未编写测试
“已准备好进行测试”
```
**为什么这是错误的:**
- 测试是实现的一部分,而不是可选的后续工作
- TDD 本可以发现这一问题
- 没有测试就不能声称已经完成
**修正方法:**
```
TDD 循环:
1. 编写一个失败的测试
2. 实现代码以使其通过
3. 重构
4. 然后才能声称完成
```
## 当模拟对象变得过于复杂时
**警告信号:**
- 模拟对象的设置比测试逻辑还长
- 为了让测试通过而模拟一切
- 模拟对象缺少真实组件所拥有的方法
- 模拟对象发生变化时测试就会失败
**你的人类伙伴的问题:**“我们需要在这里使用模拟对象吗?”
**考虑一下:**使用真实组件的集成测试通常比复杂的模拟对象更简单
## TDD 可防止这些反模式
**TDD 为什么有帮助:**
1. **先编写测试** → 迫使你思考自己实际在测试什么
2. **观察它失败** → 确认测试检验的是真实行为,而不是模拟对象
3. **最小化实现** → 不会悄然加入仅供测试使用的方法
4. **真实依赖项** → 在进行模拟之前,你会先看到测试实际需要什么
**如果你测试的是模拟对象的行为,就违反了 TDD**——你在没有先观察测试针对真实代码失败的情况下就添加了模拟对象。
## 快速参考
| 反模式 | 修正方法 |
|--------------|-----|
| 对模拟元素进行断言 | 测试真实组件,或取消对其模拟 |
| 生产代码中存在仅供测试使用的方法 | 将其移至测试工具中 |
| 在不了解的情况下进行模拟 | 先了解依赖项,并尽可能少地模拟 |
| 不完整的模拟对象 | 完整复刻真实 API |
| 将测试视为事后补充 | TDD——测试优先 |
| 过于复杂的模拟对象 | 考虑使用集成测试 |
## 危险信号
- 断言检查 `*-mock` 测试 ID
- 仅在测试文件中调用的方法
- 模拟对象的设置占测试的 >50%
- 移除模拟对象时测试失败
- 无法解释为什么需要模拟对象
- “只是为了保险起见”而进行模拟
## 核心结论
**模拟对象是用于隔离的工具,而不是测试对象。**
如果 TDD 揭示你测试的是模拟对象的行为,那就说明你做错了。
@@ -0,0 +1,173 @@
# 编写高质量测试
**在以下情况加载此参考:** 编写或修改测试、添加 mock,或添加仅供测试使用的清理/辅助方法。
## 概述
测试存在的意义,是捕获一种具体的破坏。这里的一切都由两个原则支配:
```
1. 每个测试都指出它能捕获的破坏
2. 每个测试都验证真实事物
```
严格的 TDD 会自然地产生这两点:先编写测试,并亲眼看到它针对真实代码失败,
就已经证明该测试确实能失败;只有当真实依赖已被证明缓慢或位于外部时,才允许使用 mock。
## 原则 1:指出破坏
编写测试主体之前,先回答:**生产代码中的哪项变更应该让这个测试失败——
而该变更是缺陷,还是有意决策?** 一个测试只有在能捕获错误分支、缺失的副作用、
错误参数、边界情况或已破坏契约时,才有存在价值。
**独立推导期望值。** 使用字面量和人工核对过的夹具;带有字面量 `want` 值的
表驱动测试是首选形式。若期望值由被测代码或其辅助函数计算,那么无论被测代码
做什么,测试都会通过:
```typescript
// ❌ Mirror assertion: the same builder computes both sides — always true
const expected = buildSearchQuery({ tag: 'urgent' });
expect(buildSearchQuery({ tag: 'urgent' })).toBe(expected);
// ✅ Hand-derived literal
expect(buildSearchQuery({ tag: 'urgent' })).toBe('tag:"urgent"');
```
**拒绝变更检测器(变更检测器陷阱)。** 如果只有有意决策才能让测试失败——
例如常量值、消息的精确措辞或私有结构——它会在重新设计时报警,却对缺陷沉睡。
请测试依赖该决策的行为:不要写 `expect(MAX_RETRIES).toBe(5)`,而要验证
“失败的调用会重试 5 次,且绝不会进行第 6 次尝试”。
**验证行为,而不是文本(字符串存在性陷阱)。** 断言脚本、技能或配置中包含
某一精确行,只能证明源文件里确实有这行文字。请用受控输入运行脚本,并断言输出、
副作用或退出码。指导 Agent 的文档应通过使用该文档的 Agent 行为来测试
superpowers:writing-skills);面向人类的说明文字完全不需要测试。
**测试你的代码,而不是框架。** 测试你的代码在边界处承诺的契约——注册的路由、
发出的查询、生成的载荷。上游机制应由它们的维护者编写测试(经典例子:断言路由器
调用已注册的处理器——那是框架的测试,不是你的)。如果上游行为确实令你意外,
请编写一个狭窄的特征测试,明确指出该假设。同样的边界也适用于代码内部:构造函数、
getter、常量和简单转发,只有在执行验证、规范化、默认值、派生、约束或副作用时才值得
单独测试——否则,请断言第一个依赖它们、且消费者可观察到的结果。
### 门禁函数
```
编写测试主体之前:
指出哪项生产代码变更会使此测试失败。
无法指出 → 围绕可观察行为重新设计
“源文本发生了变化” → 运行该产物并断言其效果
只有有意决策会导致失败 → 变更检测器;测试依赖该决策的行为
确认期望值的推导没有使用被测代码。
如果它复用了被测代码的逻辑或辅助函数:
替换为字面量或人工核对过的夹具
```
## 原则 2:验证真实事物
**mock 不值得拥有断言。** 有 mock 时通过、没有 mock 时失败的断言,完全没有说明
组件的行为。断言真实组件的行为;如果你检查的是 mock,请取消 mock 或删除该断言。
```typescript
// ✅ Real behavior
expect(screen.getByRole('navigation')).toBeInTheDocument();
// ❌ Mock existence
expect(screen.getByTestId('sidebar-mock')).toBeInTheDocument();
```
**你的人类伙伴的纠正:** “我们是在测试一个 mock 的行为吗?”
**在正确层级进行 mock。** 替换真实方法之前,先了解它的每一项副作用;mock 缓慢
或外部的操作,同时保留测试所依赖的真实行为。如果不确定,请先针对真实实现运行测试,
观察究竟需要发生什么。
```typescript
// ❌ The mock swallows the config write that duplicate detection reads
vi.mock('ToolCatalog', () => ({
discoverAndCacheTools: vi.fn().mockResolvedValue(undefined)
}));
// ✅ Mock only the slow server startup; the config write stays real
vi.mock('MCPServerManager');
```
**让测试替身具体明确。** 当参数、调用次数或顺序属于契约的一部分时,请断言它们——
接受任何内容的 fake 什么也验证不了。为每个分支(成功、错误、格式错误)分别提供
独立夹具或 spy,这样错误分支就无法满足期望。
**完整镜像真实数据。** 按照真实存在的完整结构——包括所有已记录字段——构造 mock,
而不是只提供测试读取的字段。下游代码读取被省略字段时,部分 mock 会悄悄失效:
测试通过了,集成却坏了。
**生产类只承载生产方法。** 只有测试需要的清理逻辑应放在测试工具中,绝不能作为
生产类上的 `destroy()`。请问:这个方法是否只从测试中调用?这个类是否拥有该资源的
生命周期?答案不正确 → 放入测试工具。
**复杂 mock 应让位于真实组件。** 当 mock 设置比测试逻辑还长、mock 缺少真实组件
拥有的方法,或 mock 一变测试就坏时,请切换到使用真实组件的集成测试。
**你的人类伙伴的问题:** “这里真的需要使用 mock 吗?”
### 门禁函数
```
添加 mock 或测试辅助函数之前:
列出真实方法的副作用;让测试依赖的副作用保持真实——
只 mock 它们下层缓慢或外部的部分。
mock 响应应完整镜像真实结构。
只有测试调用的方法应放在测试工具中,而不是生产代码中。
准备对 mock 本身作断言?
取消 mock 或删除该断言。
```
## 测试与实现一同交付
TDD 循环——失败测试、最小实现、重构——就是“完成”的含义。交付行为所需且仅限
这些测试:琐碎代码和面向人类的说明文字不值得测试;只为满足流程而编写的测试会
永远产生维护成本。
## 变异检查
完成前,在脑中改变生产代码;对于每一种现实可能发生的变异,都应至少有一个测试失败:
- 错误常量或参数
- 错误分支处理器
- 缺失的状态变更或副作用
- 空返回值或默认返回值
- 缺少对零值、空值、nil、未授权或格式错误输入的验证
若某项变异没有任何测试能捕获,说明该行为没有保护——或者测试只是同义反复。
## 快速参考
| 当你…… | 应该做什么 |
|-------------|-----|
| 编写任何测试 | 指出它能捕获的破坏——必须是缺陷,而不是决策 |
| 构造期望值 | 手工推导;绝不使用被测代码 |
| 测试脚本或文档 | 运行它/对使用者进行压力测试;绝不要 grep 文本 |
| 想测试某个依赖项 | 测试你的边界契约,而不是它已有文档的机制 |
| 想对被 mock 的元素作断言 | 测试真实组件,或取消 mock |
| 准备 mock 一个方法 | 了解其副作用;mock 缓慢/外部的层级 |
| 构造 mock 响应 | 完整镜像真实结构 |
| 需要仅供测试使用的清理逻辑 | 放入测试工具 |
| mock 设置开始膨胀 | 切换到使用真实组件的集成测试 |
| 完成一个测试文件 | 运行变异检查 |
## 警告信号
- 设置和断言共享同一对象,保证它们必然相等
- 测试只能因为 panic、崩溃或选择器缺失而失败
- 每次有意变更都会让测试失败,意外破坏却从不触发它
- 期望值隐藏在循环、构建器或辅助函数之后
- 测试 grep 源文本,或断言已删除符号必须继续保持删除状态
- 如果只剩框架,这个测试仍然有意义
- 测试只为覆盖率而存在,没有检查任何副作用或结果
- 断言检查 `*-mock` 测试 ID,或删除 mock 后就失败
- 某个方法只从测试文件中调用
- mock 设置超过测试的一半,或你无法解释为何需要它
- “为了保险”而 mock
@@ -1,6 +1,6 @@
--- ---
name: using-git-worktrees name: using-git-worktrees
description: 在开始需要与当前工作区隔离的功能开发时,或在执行实施计划之前使用 - 通过原生工具或 git 工作树回退方案确保隔离工作区存在 description: 在开始需要与当前工作区隔离的功能开发时,或在执行实施计划之前使用——通过原生工具或 git 工作树回退方案确保隔离工作区存在
--- ---
# 使用 Git 工作树 # 使用 Git 工作树
@@ -11,11 +11,11 @@ description: 在开始需要与当前工作区隔离的功能开发时,或在
**核心原则:** 先检测现有隔离。然后使用原生工具。再回退到 git。绝不要对抗运行平台。 **核心原则:** 先检测现有隔离。然后使用原生工具。再回退到 git。绝不要对抗运行平台。
**开始时宣布:** "我正在使用 using-git-worktrees 技能来设置一个隔离工作区。" **开始时宣布:** 我正在使用 using-git-worktrees 技能来设置一个隔离工作区。
## 步骤 0:检测现有隔离 ## 第 0 步:检测现有隔离
**创建任何内容之前,检查你是否已经位于隔离工作区中。** **创建任何内容之前,检查你是否已经位于隔离工作区中。**
```bash ```bash
GIT_DIR=$(cd "$(git rev-parse --git-dir)" 2>/dev/null && pwd -P) GIT_DIR=$(cd "$(git rev-parse --git-dir)" 2>/dev/null && pwd -P)
@@ -23,60 +23,67 @@ GIT_COMMON=$(cd "$(git rev-parse --git-common-dir)" 2>/dev/null && pwd -P)
BRANCH=$(git branch --show-current) BRANCH=$(git branch --show-current)
``` ```
**子模块防护:** 在 git 子模块中,`GIT_DIR != GIT_COMMON` 同样成立。在断定“已位于工作树中”之前,验证你并非位于子模块中: **子模块防护:** 在 git 子模块中,`GIT_DIR != GIT_COMMON` 同样成立。在断定
“已经位于工作树中”之前,请验证你并非位于子模块中:
```bash ```bash
# 如果此命令返回一个路径,则你位于子模块中,而不是工作树中 — 将其视为普通仓库 # If this returns a path, you're in a submodule, not a worktree — treat as normal repo
git rev-parse --show-superproject-working-tree 2>/dev/null git rev-parse --show-superproject-working-tree 2>/dev/null
``` ```
**如果 `GIT_DIR != GIT_COMMON`(且不在子模块中):** 你已位于链接工作树中。跳到步骤 2(项目设置)。切勿创建另一个工作树。 **如果 `GIT_DIR != GIT_COMMON`(且不在子模块中):** 你已位于链接工作树中。
跳到第 2 步(项目设置)。绝不要创建另一个工作树。
报告时包含分支状态: 报告时包含分支状态:
- 位于分支上:"已位于 `<path>` 的隔离工作区中,当前分支为 `<name>`" - 位于分支上:已位于 `<path>` 的隔离工作区中,当前分支为 `<name>`
- Detached HEAD"已位于 `<path>` 的隔离工作区中(detached HEAD,由外部管理)。完成时需要创建分支。" - Detached HEAD已位于 `<path>` 的隔离工作区中(detached HEAD,由外部管理)。完成时需要创建分支。
**如果 `GIT_DIR == GIT_COMMON`(或位于子模块中):** 你位于普通仓库检出中。 **如果 `GIT_DIR == GIT_COMMON`(或位于子模块中):** 你位于普通仓库检出中。
用户是否已经在你的指令中表明了工作树偏好?如果没有,请在创建工作树之前征求同意: 用户是否已经在你的指令中表明了工作树偏好?如果没有,请在创建工作树之前征求同意:
> "你希望我设置一个隔离工作树吗?它可以保护你的当前分支不受更改影响。" > 你希望我设置一个隔离工作树吗?它可以保护你的当前分支不受更改影响。
如果已有明确声明的偏好,无需询问,遵照执行。如果用户拒绝同意,则在原位置工作并跳到步骤 2 如果已有明确声明的偏好,无需询问,直接遵照执行。如果用户拒绝,则在原位置工作并跳到第 2 步。
## 第 1 步:创建隔离工作区
## 步骤 1:创建隔离工作区
**你有两种机制。请按以下顺序尝试。** **你有两种机制。请按以下顺序尝试。**
### 1a. 原生工作树工具(首选) ### 1a. 原生工作树工具(首选)
用户已要求使用隔离工作空间(步骤 0 中的同意)。你是否已经有创建工作树的方法?它可能是名称类似 `EnterWorktree``WorktreeCreate` 的工具、`/worktree` 命令,或 `--worktree` 标志。如果有,请使用它并跳至步骤 2。 用户已要求使用隔离工作区(第 0 中的同意)。你是否已经有创建工作树的方法?
它可能是名为 `EnterWorktree``WorktreeCreate` 的工具、`/worktree` 命令,
`--worktree` 标志。如果有,请使用它并跳到第 2 步。
原生工具会自动处理目录放置、分支创建和清理。当你拥有原生工具时,使用 `git worktree add` 会创建你的运行平台无法看到或管理的幽灵状态。 原生工具会自动处理目录放置、分支创建和清理。当你拥有原生工具时,使用
`git worktree add` 会创建运行平台无法看到或管理的幽灵状态。
仅当没有可用的原生工作树工具时,才继续执行步骤 1b。 只有在没有原生工作树工具可用时,才继续执行步骤 1b。
### 1b. Git 工作树回退方案 ### 1b. Git 工作树回退方案
**仅当步骤 1a 不适用时才使用此方案**——没有可用的原生工作树工具。使用 git 手动创建工作树。 **仅当步骤 1a 不适用时才使用此方案**——也就是没有可用的原生工作树工具。
使用 git 手动创建工作树。
#### 目录选择 #### 目录选择
按以下优先级顺序执行。用户的明确偏好始终优先于观察到的文件系统状态。 按以下优先级执行。用户的明确偏好始终优先于观察到的文件系统状态。
1. **检查你的指令中是否声明了工作树目录偏好。** 如果用户已经指定目录,无需询问,直接使用。 1. **检查你的指令中是否声明了工作树目录偏好。** 如果用户已经指定目录,无需询问,直接使用。
2. **检查是否存在项目本地工作树目录:** 2. **检查是否存在项目本地工作树目录:**
```bash ```bash
ls -d .worktrees 2>/dev/null # 首选(隐藏) ls -d .worktrees 2>/dev/null # Preferred (hidden)
ls -d worktrees 2>/dev/null # 备选 ls -d worktrees 2>/dev/null # Alternative
``` ```
如果找到,使用它。如果两者都存在,优先使用 `.worktrees` 如果找到,使用它。如果两者都存在,优先使用 `.worktrees`
3. **如果没有任何其他可用指引**,则默认使用项目根目录下的 `.worktrees/` 3. **如果没有任何其他可用指引**默认使用项目根目录下的 `.worktrees/`
#### 安全验证(仅限项目本地目录) #### 安全验证(仅限项目本地目录)
**创建工作树前,必须验证该目录已被忽略:** **创建工作树前,必须验证该目录已被忽略:**
```bash ```bash
git check-ignore -q .worktrees 2>/dev/null || git check-ignore -q worktrees 2>/dev/null git check-ignore -q .worktrees 2>/dev/null || git check-ignore -q worktrees 2>/dev/null
@@ -89,17 +96,19 @@ git check-ignore -q .worktrees 2>/dev/null || git check-ignore -q worktrees 2>/d
#### 创建工作树 #### 创建工作树
```bash ```bash
# 根据所选位置确定路径 # Determine path based on chosen location
path="$LOCATION/$BRANCH_NAME" path="$LOCATION/$BRANCH_NAME"
git worktree add "$path" -b "$BRANCH_NAME" git worktree add "$path" -b "$BRANCH_NAME"
cd "$path" cd "$path"
``` ```
**沙箱回退方案:** 如果 `git worktree add` 因权限错误(沙箱拒绝)而失败,请告用户沙箱阻止了工作树创建,因此你将改为在当前目录中工作。然后就地运行设置和基线测试。 **沙箱回退方案:** 如果 `git worktree add` 因权限错误(沙箱拒绝)而失败,请告用户
沙箱阻止了工作树创建,因此你将改为在当前目录中工作。然后就地运行设置和基线测试。
## 步骤 2:项目设置 ## 第 2 步:项目设置
自动检测并执行相应的设置:
自动检测并运行相应的设置:
```bash ```bash
# Node.js # Node.js
@@ -116,20 +125,21 @@ if [ -f pyproject.toml ]; then poetry install; fi
if [ -f go.mod ]; then go mod download; fi if [ -f go.mod ]; then go mod download; fi
``` ```
## 步骤 3:验证干净基线 ## 第 3 步:验证干净基线
运行测试,确保工作区的起始状态干净: 运行测试,确保工作区的起始状态干净:
```bash ```bash
# 使用适合项目的命令 # Use project-appropriate command
npm test / cargo test / pytest / go test ./... npm test / cargo test / pytest / go test ./...
``` ```
**如果测试失败:** 报告失败情况,询问是继续还是调查。 **如果测试失败:** 报告失败情况,询问是继续还是调查。
**如果测试通过:** 报告已就绪。 **如果测试通过:** 报告已就绪。
### 报告 ### 报告
``` ```
工作树已准备就绪,位于 <full-path> 工作树已准备就绪,位于 <full-path>
测试通过(<N> 个测试,0 个失败) 测试通过(<N> 个测试,0 个失败)
@@ -140,8 +150,8 @@ npm test / cargo test / pytest / go test ./...
| 情况 | 操作 | | 情况 | 操作 |
|-----------|--------| |-----------|--------|
| 已在链接工作树中 | 跳过创建(步骤 0 | | 已在链接工作树中 | 跳过创建(第 0 步) |
| 位于子模块中 | 按普通仓库处理(步骤 0 防护) | | 位于子模块中 | 按普通仓库处理( 0 防护) |
| 有原生工作树工具可用 | 使用它(步骤 1a) | | 有原生工作树工具可用 | 使用它(步骤 1a) |
| 没有原生工具 | 使用 Git 工作树回退方案(步骤 1b) | | 没有原生工具 | 使用 Git 工作树回退方案(步骤 1b) |
| `.worktrees/` 存在 | 使用它(验证已被忽略) | | `.worktrees/` 存在 | 使用它(验证已被忽略) |
@@ -150,50 +160,15 @@ npm test / cargo test / pytest / go test ./...
| 两者都不存在 | 检查指令文件,然后默认使用 `.worktrees/` | | 两者都不存在 | 检查指令文件,然后默认使用 `.worktrees/` |
| 目录未被忽略 | 添加到 .gitignore 并提交 | | 目录未被忽略 | 添加到 .gitignore 并提交 |
| 创建时出现权限错误 | 使用沙箱回退方案,原地工作 | | 创建时出现权限错误 | 使用沙箱回退方案,原地工作 |
| 基线测试期间测试失败 | 报告失败情况并询问 | | 基线测试失败 | 报告失败情况并询问 |
| 没有 package.json/Cargo.toml | 跳过依赖项安装 | | 没有 package.json/Cargo.toml | 跳过依赖项安装 |
## 常见错误 ## 常见合理化借口
### 与运行平台对抗 | 借口 | 现实 |
|--------|---------|
- **问题:** 在平台已经提供隔离时使用 `git worktree add` | “我显然不在工作树中——没必要检查” | 运行第 0 步。运行平台创建的隔离和子模块都可能骗过肉眼判断;检测命令才能给出结论。 |
- **修复方法:** 步骤 0 会检测现有隔离。步骤 1a 优先采用原生工具。 | “`git worktree add` 比寻找原生工具更快” | 原生工具(例如 `EnterWorktree`)负责目录、分支和清理。绕过它是头号错误——会创建运行平台看不到也无法管理的幽灵状态。 |
| “工作树目录肯定已经被忽略” | 运行 `git check-ignore`。未忽略的工作树目录会把整棵工作树提交进仓库。 |
### 跳过检测 | “任何目录名都可以” | 明确指令优先于现有项目本地目录,现有目录又优先于 `.worktrees/` 默认值。 |
| “工作区是新的——基线测试可以稍后再跑” | 不干净的基线会让之后的每次失败都无法归因。现在就运行测试;是否带着失败继续应由你的人类伙伴决定。 |
- **问题:** 在现有工作树内部创建嵌套工作树
- **修复方法:** 创建任何内容之前,始终先运行步骤 0
### 跳过忽略验证
- **问题:** 工作树内容被跟踪,污染 git status
- **修复方法:** 创建项目本地工作树之前,始终使用 `git check-ignore`
### 臆断目录位置
- **问题:** 造成不一致,违反项目约定
- **修复方法:** 遵循优先级:明确指令 > 现有项目本地目录 > 默认值
### 在测试失败时继续
- **问题:** 无法区分新错误和既有问题
- **修复方法:** 报告失败情况,获得明确许可后再继续
## 红旗项
**绝不:**
- 在步骤 0 检测到现有隔离时创建工作树
- 在有原生工作树工具(例如 `EnterWorktree`)时使用 `git worktree add`。这是头号错误——如果有,就使用它。
- 跳过步骤 1a,直接执行步骤 1b 中的 git 命令
- 未验证工作树已被忽略就创建工作树(项目本地)
- 跳过基线测试验证
- 未经询问就在测试失败的情况下继续
**始终:**
- 首先运行步骤 0 检测
- 优先使用原生工具,而不是 git 回退方案
- 遵循目录优先级:明确指令 > 现有项目本地目录 > 默认值
- 对于项目本地目录,验证该目录已被忽略
- 自动检测并运行项目设置
- 验证干净的测试基线
@@ -54,6 +54,7 @@ description: 在开始任何对话时使用——规定如何查找和使用技
如果你的运行平台列在这里,请阅读其参考文件以了解特殊指令: 如果你的运行平台列在这里,请阅读其参考文件以了解特殊指令:
- Codex`references/codex-tools.md` - Codex`references/codex-tools.md`
- Gemini CLI`references/gemini-tools.md`
- Pi`references/pi-tools.md` - Pi`references/pi-tools.md`
- Antigravity`references/antigravity-tools.md` - Antigravity`references/antigravity-tools.md`
@@ -4,7 +4,7 @@
| 技能所请求的动作 | Antigravity CLI 等效方式 | | 技能所请求的动作 | Antigravity CLI 等效方式 |
|----------------------|----------------------| |----------------------|----------------------|
| 派遣一个子 Agent`Subagent (general-purpose):` 模板) | 使用带有内置 `TypeName``invoke_subagent`——`self` 用于全能力工作,`research` 用于只读工作(参见[子 Agent 支持](#subagent-support) | | 派遣一个子 Agent`Subagent (general-purpose):` 模板) | 使用带有内置 `TypeName``invoke_subagent`——`self` 用于全能力工作,`research` 用于只读工作 |
| 任务跟踪(“创建一个待办事项”“标记为完成”) | 一个**任务产物**——使用 `write_to_file`,并设置 `IsArtifact: true``ArtifactType: "task"`(参见[任务跟踪](#task-tracking))。**不是** `manage_task`,后者用于管理后台进程。 | | 任务跟踪(“创建一个待办事项”“标记为完成”) | 一个**任务产物**——使用 `write_to_file`,并设置 `IsArtifact: true``ArtifactType: "task"`(参见[任务跟踪](#task-tracking))。**不是** `manage_task`,后者用于管理后台进程。 |
## 任务跟踪 ## 任务跟踪
@@ -7,7 +7,7 @@
multi_agent = true multi_agent = true
``` ```
这将为 `dispatching-parallel-agents``subagent-driven-development` 等技能启用 `spawn_agent``wait_agent``close_agent`。使用 subagent-driven-development 时,你应始终在实现者和审查者子 Agent 完成其全部工作后将其关闭 这将为 `dispatching-parallel-agents``subagent-driven-development` 等技能启用 `spawn_agent``wait_agent``close_agent`。使用 subagent-driven-development 时,审查者返回审查结果后就关闭它。每个实现者子 Agent 应保持开启,直到其任务通过审查——修复循环会恢复该实现者——然后再关闭。如果你的运行平台无法向已派生的 Agent 再发送消息,则每一轮修复都派遣一个新的实现者,并向它提供简报、报告文件和发现项
## 环境检测 ## 环境检测
@@ -0,0 +1,63 @@
# Gemini CLI 工具映射
技能以动作来表述(“派遣一个子 Agent”“创建一个待办事项”“读取一个文件”)。在 Gemini CLI 上,这些动作对应于以下工具。
| 技能所请求的动作 | Gemini CLI 等效方式 |
|----------------------|----------------------|
| 读取文件 | `read_file` |
| 一次读取多个文件 | `read_many_files` |
| 创建新文件 | `write_file` |
| 编辑文件 | `replace` |
| 运行 Shell 命令 | `run_shell_command` |
| 搜索文件内容 | `grep_search` |
| 按名称查找文件 | `glob` |
| 列出文件和子目录 | `list_directory` |
| 获取 URL | `web_fetch` |
| 搜索网络 | `google_web_search` |
| 调用技能 | `activate_skill` |
| 派遣子 Agent`Subagent (general-purpose):` 模板) | 调用 `invoke_agent` 并设置 `agent_name: "generalist"`(也可使用 `@generalist` 聊天语法调用——参见[子 Agent 支持](#子-agent-支持) |
| 多次并行派遣 | 在同一次响应中进行多个 `invoke_agent` 调用 |
| 任务跟踪(“创建待办事项”“标记为完成”) | `write_todos`(状态:pending、in_progress、completed、cancelled、blocked |
## 指令文件
技能提到“你的指令文件”时,在 Gemini CLI 中指的是 **`GEMINI.md`**。Gemini CLI 会分层加载 `GEMINI.md`:全局文件位于 `~/.gemini/GEMINI.md`;项目级文件位于工作区目录及其祖先目录;工具访问子目录中的文件时,还会加载该子目录下的 `GEMINI.md`
## 个人技能目录
用户级技能位于 **`~/.gemini/skills/`****`~/.agents/skills/`** 则是跨运行时别名(与 Codex 和 Copilot CLI 共享)。同一作用域同时存在两个目录时,`.agents/skills/` 优先。每个技能都是一个包含 `SKILL.md` 的子目录(其 frontmatter 含 `name``description`)。
## 子 Agent 支持
Gemini CLI 通过 `invoke_agent` 工具派遣子 Agent;该工具接受 `agent_name``prompt` 参数。同一派遣能力也提供聊天语法快捷方式:输入 `@generalist <prompt>`,等同于调用 `invoke_agent` 并设置 `agent_name: "generalist"`。内置 Agent 名称包括 `generalist``cli_help``codebase_investigator`,以及启用浏览器工具后可用的 `browser_agent`
技能使用 `Subagent (general-purpose):` 进行派遣,并且要么引用提示模板文件(例如 `superpowers:subagent-driven-development``./implementer-prompt.md`),要么直接提供内联提示。在 Gemini CLI 中:
| 技能派遣形式 | Gemini CLI 等效方式 |
|---------------------|----------------------|
| 引用 `*-prompt.md` 模板(实现者、任务审查者、代码审查者等) | 填充模板,然后调用 `invoke_agent`,设置 `agent_name: "generalist"` 并传入填充后的提示 |
| 引用 `superpowers:requesting-code-review``./code-reviewer.md` | 调用 `invoke_agent`,设置 `agent_name: "generalist"` 并传入填充后的审查模板 |
| 内联提示(未引用模板) | 调用 `invoke_agent`,设置 `agent_name: "generalist"` 并传入内联提示 |
### 填充提示
技能提供的提示模板会包含 `{WHAT_WAS_IMPLEMENTED}``[FULL TEXT of task]` 等占位符。将完整提示传给 `invoke_agent` 之前,必须填充所有占位符。提示模板本身已经包含 Agent 角色、审查标准和预期输出格式——子 Agent 会遵循这些内容。
### 并行派遣
Gemini CLI 支持并行派遣子 Agent。在同一次响应中发出多个 `invoke_agent` 调用(或在一个提示中进行多次 `@generalist` 调用),即可并行运行彼此独立的子 Agent 工作。保持依赖任务顺序执行,但不要仅为了让历史记录更简单而把独立的子 Agent 任务串行化。
## 其他 Gemini CLI 工具
以下工具为 Gemini CLI 特有:
| 工具 | 用途 |
|------|---------|
| `save_memory`(旧版) | 当 `experimental.memoryV2 = false` 时跨会话保存事实 |
| `get_internal_docs` | 查阅 Gemini CLI 内置文档 |
| `ask_user` | 向用户提出结构化问题(文本/单选/多选) |
| `enter_plan_mode` / `exit_plan_mode` | 进入或退出只读计划模式 |
| `update_topic` | 更新当前对话的主题/战略意图元数据 |
| `complete_task` | 表示 Gemini 子 Agent 已完成,并将结果返回给父 Agent |
| `tracker_create_task``tracker_update_task``tracker_get_task``tracker_list_tasks``tracker_add_dependency``tracker_visualize` | 支持依赖关系与可视化的丰富任务跟踪器 |
| `read_mcp_resource``list_mcp_resources` | 访问 MCP 资源 |
@@ -7,8 +7,6 @@ description: 在即将声称工作已完成、已修复或已通过时使用,
## 概述 ## 概述
未经验证就声称工作已完成,是不诚实,而非高效。
**核心原则:** 始终先有证据,再作声明。 **核心原则:** 始终先有证据,再作声明。
**违反这条规则的字面要求,就是违背这条规则的精神实质。** **违反这条规则的字面要求,就是违背这条规则的精神实质。**
@@ -103,15 +101,6 @@ description: 在即将声称工作已完成、已修复或已通过时使用,
❌ 相信 Agent 的报告 ❌ 相信 Agent 的报告
``` ```
## 为什么这很重要
来自 24 条失败记忆:
- 你的人类伙伴说“我不相信你”——信任已破裂
- 发布了未定义的函数——会导致崩溃
- 发布时缺少需求——功能不完整
- 因虚假的完成状态而浪费时间 → 调整方向 → 返工
- 违反:“诚实是核心价值观。如果你撒谎,你将被替换。”
## 何时应用 ## 何时应用
**在以下情况之前始终应用:** **在以下情况之前始终应用:**
@@ -127,11 +116,3 @@ description: 在即将声称工作已完成、已修复或已通过时使用,
- 改述和同义词 - 改述和同义词
- 对成功的暗示 - 对成功的暗示
- 任何表明已完成/正确的沟通 - 任何表明已完成/正确的沟通
## 底线
**验证没有捷径。**
运行命令。阅读输出。然后再声明结果。
这一点没有商量余地。
@@ -125,12 +125,6 @@ git commit -m "feat: 添加特定功能"
- 只描述要做什么,却不展示如何做的步骤(代码步骤必须包含代码块) - 只描述要做什么,却不展示如何做的步骤(代码步骤必须包含代码块)
- 引用任何任务中都未定义的类型、函数或方法 - 引用任何任务中都未定义的类型、函数或方法
## 牢记
- 始终提供精确的文件路径
- 每个步骤都要提供完整代码——如果某个步骤修改代码,就展示该代码
- 提供精确的命令及预期输出
- DRY、YAGNI、TDD、频繁提交
## 自我审查 ## 自我审查
撰写完整计划后,以全新的视角审视规范,并对照规范检查计划。这是你自己执行的检查清单——不是派遣子 Agent。 撰写完整计划后,以全新的视角审视规范,并对照规范检查计划。这是你自己执行的检查清单——不是派遣子 Agent。
@@ -9,7 +9,7 @@ description: 在创建新技能、编辑现有技能,或在部署前验证技
**编写技能就是将测试驱动开发应用于流程文档。** **编写技能就是将测试驱动开发应用于流程文档。**
**个人技能位于你的运行时的技能目录中** **个人技能位于你的运行时的技能目录中**Claude Code 为 `~/.claude/skills/`)——Codex 的路径参见 [codex-tools.md](../using-superpowers/references/codex-tools.md)Gemini 的路径参见 [gemini-tools.md](../using-superpowers/references/gemini-tools.md)。Codex、Copilot CLI 和 Gemini CLI 也都识别 `~/.agents/skills/` 这一跨运行时别名。
你编写测试用例(包含子 Agent 的压力场景),观察它们失败(基线行为),编写技能(文档),观察测试通过(Agent 遵从要求),然后进行重构(堵住漏洞)。 你编写测试用例(包含子 Agent 的压力场景),观察它们失败(基线行为),编写技能(文档),观察测试通过(Agent 遵从要求),然后进行重构(堵住漏洞)。
@@ -663,13 +663,3 @@ helper1, helper2, step3, pattern4
6. **加载示例**(仅在实现时) 6. **加载示例**(仅在实现时)
**针对这一流程进行优化**——尽早并经常放入可搜索的术语。 **针对这一流程进行优化**——尽早并经常放入可搜索的术语。
## 核心结论
**创建技能就是面向流程文档的 TDD。**
同一条铁律:没有先失败的测试,就不能创建技能。
同一个循环:RED(基线)→ GREEN(编写技能)→ REFACTOR(堵住漏洞)。
同样的收益:质量更高、意外更少、结果无懈可击。
如果你对代码遵循 TDD,那么对技能也应遵循 TDD。这是将同一种纪律应用于文档。
@@ -1,6 +1,6 @@
{ {
"name": "superpowers", "name": "superpowers",
"version": "6.1.1", "version": "6.2.0",
"description": "面向编码 Agent 的规划、TDD、调试、代码评审和交付工作流集合。", "description": "面向编码 Agent 的规划、TDD、调试、代码评审和交付工作流集合。",
"author": { "author": {
"name": "Jesse Vincent", "name": "Jesse Vincent",
+15 -1
View File
@@ -11,7 +11,7 @@ If this sounds like someone you know, definitely send them our way.
## Quickstart ## Quickstart
Give your agent Superpowers: [Claude Code](#claude-code), [Antigravity](#antigravity), [Codex App](#codex-app), [Codex CLI](#codex-cli), [Cursor](#cursor), [Factory Droid](#factory-droid), [GitHub Copilot CLI](#github-copilot-cli), [Kimi Code](#kimi-code), [OpenCode](#opencode), [Pi](#pi). Give your agent Superpowers: [Claude Code](#claude-code), [Antigravity](#antigravity), [Codex App](#codex-app), [Codex CLI](#codex-cli), [Cursor](#cursor), [Factory Droid](#factory-droid), [Gemini CLI](#gemini-cli), [GitHub Copilot CLI](#github-copilot-cli), [Kimi Code](#kimi-code), [OpenCode](#opencode), [Pi](#pi).
## How it works ## How it works
@@ -122,6 +122,20 @@ Superpowers is available via the [official Codex plugin marketplace](https://git
droid plugin install superpowers@superpowers droid plugin install superpowers@superpowers
``` ```
### Gemini CLI
- Install the extension:
```bash
gemini extensions install https://github.com/obra/superpowers
```
- Update later:
```bash
gemini extensions update superpowers
```
### GitHub Copilot CLI ### GitHub Copilot CLI
- Register the marketplace: - Register the marketplace:
@@ -2,8 +2,8 @@
"sourceId": "superpowers", "sourceId": "superpowers",
"repo": "https://github.com/obra/superpowers.git", "repo": "https://github.com/obra/superpowers.git",
"ref": "main", "ref": "main",
"commit": "d884ae04edebef577e82ff7c4e143debd0bbec99", "commit": "3dcbd5c4b48e02263fbf4a3c01e3fe4f81d584d9",
"adapter": "codex-plugin", "adapter": "codex-plugin",
"sourcePath": ".", "sourcePath": ".",
"syncedAt": "2026-07-03T16:00:00Z" "syncedAt": "2026-07-24T06:18:20Z"
} }
@@ -77,6 +77,7 @@ digraph brainstorming {
- Propose 2-3 different approaches with trade-offs - Propose 2-3 different approaches with trade-offs
- Present options conversationally with your recommendation and reasoning - Present options conversationally with your recommendation and reasoning
- Lead with your recommended option and explain why - Lead with your recommended option and explain why
- YAGNI ruthlessly - remove unnecessary features from every approach and design
**Presenting the design:** **Presenting the design:**
@@ -130,15 +131,6 @@ Wait for the user's response. If they request changes, make them and re-run the
- Invoke the writing-plans skill to create a detailed implementation plan - Invoke the writing-plans skill to create a detailed implementation plan
- Do NOT invoke any other skill. writing-plans is the next step. - Do NOT invoke any other skill. writing-plans is the next step.
## Key Principles
- **One question at a time** - Don't overwhelm with multiple questions
- **Multiple choice preferred** - Easier to answer than open-ended when possible
- **YAGNI ruthlessly** - Remove unnecessary features from all designs
- **Explore alternatives** - Always propose 2-3 approaches before settling
- **Incremental validation** - Present design, get approval before moving on
- **Be flexible** - Go back and clarify when something doesn't make sense
## Visual Companion ## Visual Companion
A browser-based companion for showing mockups, diagrams, and visual options during brainstorming. Available as a tool — not a mode. Accepting the companion means it's available for questions that benefit from visual treatment; it does NOT mean every question goes through the browser. A browser-based companion for showing mockups, diagrams, and visual options during brainstorming. Available as a tool — not a mode. Accepting the companion means it's available for questions that benefit from visual treatment; it does NOT mean every question goes through the browser.
@@ -74,6 +74,13 @@ On Windows, the script auto-detects and switches to foreground mode (which block
scripts/start-server.sh --project-dir /path/to/project --open scripts/start-server.sh --project-dir /path/to/project --open
``` ```
**Gemini CLI:**
```bash
# Use --foreground and set is_background: true on your shell tool call
# so the process survives across turns
scripts/start-server.sh --project-dir /path/to/project --open --foreground
```
**Copilot CLI:** **Copilot CLI:**
```bash ```bash
# Use --foreground and start the server via the bash tool with mode: "async" # Use --foreground and start the server via the bash tool with mode: "async"
@@ -158,15 +158,6 @@ Agent 3 → Fix tool-approval-race-conditions.test.ts
**Integration:** All fixes independent, no conflicts, full suite green **Integration:** All fixes independent, no conflicts, full suite green
**Time saved:** 3 problems solved in parallel vs sequentially
## Key Benefits
1. **Parallelization** - Multiple investigations happen simultaneously
2. **Focus** - Each agent has narrow scope, less context to track
3. **Independence** - Agents don't interfere with each other
4. **Speed** - 3 problems solved in time of 1
## Verification ## Verification
After agents return: After agents return:
@@ -174,12 +165,3 @@ After agents return:
2. **Check for conflicts** - Did agents edit same code? 2. **Check for conflicts** - Did agents edit same code?
3. **Run full suite** - Verify all fixes work together 3. **Run full suite** - Verify all fixes work together
4. **Spot check** - Agents can make systematic errors 4. **Spot check** - Agents can make systematic errors
## Real-World Impact
From debugging session (2025-10-03):
- 6 failures across 3 files
- 3 agents dispatched in parallel
- All investigations completed concurrently
- All fixes integrated successfully
- Zero conflicts between agent changes
@@ -11,15 +11,16 @@ Load plan, review critically, execute all tasks, report when complete.
**Announce at start:** "I'm using the executing-plans skill to implement this plan." **Announce at start:** "I'm using the executing-plans skill to implement this plan."
**Note:** Tell your human partner that Superpowers works much better with access to subagents. The quality of its work will be significantly higher if run on a platform with subagent support (Claude Code, Codex CLI, Codex App, and Copilot CLI all qualify; see the per-platform tool refs in `../using-superpowers/references/`). If subagents are available, use superpowers:subagent-driven-development instead of this skill. **Note:** Tell your human partner that Superpowers works much better with access to subagents (Claude Code, Codex CLI, Codex App, Copilot CLI, and Gemini CLI all qualify; see the per-platform tool refs in `../using-superpowers/references/`). If subagents are available, use superpowers:subagent-driven-development instead of this skill.
## The Process ## The Process
### Step 1: Load and Review Plan ### Step 1: Load and Review Plan
1. Read plan file 1. Ensure an isolated workspace: use superpowers:using-git-worktrees to create one or verify the existing one
2. Review critically - identify any questions or concerns about the plan 2. Read plan file
3. If concerns: Raise them with your human partner before starting 3. Review critically - identify any questions or concerns about the plan
4. If no concerns: Create todos for the plan items and proceed 4. If concerns: Raise them with your human partner before starting
5. If no concerns: Create todos for the plan items and proceed
### Step 2: Execute Tasks ### Step 2: Execute Tasks
@@ -61,10 +62,3 @@ After all tasks complete and verified:
- Reference skills when plan says to - Reference skills when plan says to
- Stop when blocked, don't guess - Stop when blocked, don't guess
- Never start implementation on main/master branch without explicit user consent - Never start implementation on main/master branch without explicit user consent
## Integration
**Required workflow skills:**
- **superpowers:using-git-worktrees** - Ensures isolated workspace (creates one or verifies existing)
- **superpowers:writing-plans** - Creates the plan this skill executes
- **superpowers:finishing-a-development-branch** - Complete development after all tasks
@@ -7,65 +7,52 @@ description: "实现完成且验证通过后,用于判断如何整理、提交
## Overview ## Overview
Guide completion of development work by presenting clear options and handling chosen workflow.
**Core principle:** Verify tests → Detect environment → Present options → Execute choice → Clean up. **Core principle:** Verify tests → Detect environment → Present options → Execute choice → Clean up.
**Announce at start:** "I'm using the finishing-a-development-branch skill to complete this work." **Announce at start:** "I'm using the finishing-a-development-branch skill to complete this work."
## The Process ## Step 1: Verify Tests
### Step 1: Verify Tests Run the project's full test suite (`npm test` / `cargo test` / `pytest` / `go test ./...`).
**Before presenting options, verify tests pass:** **If tests fail**, report the failures and stop — the menu comes after a green suite:
```bash
# Run project's test suite
npm test / cargo test / pytest / go test ./...
```
**If tests fail:**
``` ```
Tests failing (<N> failures). Must fix before completing: Tests failing (<N> failures). Must fix before completing:
[Show failures] [Show failures]
Cannot proceed with merge/PR until tests pass.
``` ```
Stop. Don't proceed to Step 2. **If tests pass:** continue to Step 2.
**If tests pass:** Continue to Step 2. ## Step 2: Detect Environment
### Step 2: Detect Environment
**Determine workspace state before presenting options:**
```bash ```bash
GIT_DIR=$(cd "$(git rev-parse --git-dir)" 2>/dev/null && pwd -P) GIT_DIR=$(cd "$(git rev-parse --git-dir)" 2>/dev/null && pwd -P)
GIT_COMMON=$(cd "$(git rev-parse --git-common-dir)" 2>/dev/null && pwd -P) GIT_COMMON=$(cd "$(git rev-parse --git-common-dir)" 2>/dev/null && pwd -P)
# Capture now, while still inside the workspace — Step 5 changes directory
# before cleanup (Step 6) needs this value
WORKTREE_PATH=$(git rev-parse --show-toplevel)
``` ```
This determines which menu to show and how cleanup works: This determines which menu to show and how cleanup works:
| State | Menu | Cleanup | | State | Menu | Cleanup |
|-------|------|---------| |-------|------|---------|
| `GIT_DIR == GIT_COMMON` (normal repo) | Standard 4 options | No worktree to clean up | | `GIT_DIR == GIT_COMMON` (normal repo) | Standard 3 options | No worktree to clean up |
| `GIT_DIR != GIT_COMMON`, named branch | Standard 4 options | Provenance-based (see Step 6) | | `GIT_DIR != GIT_COMMON`, named branch | Standard 3 options | Provenance-based (see Step 6) |
| `GIT_DIR != GIT_COMMON`, detached HEAD | Reduced 3 options (no merge) | No cleanup (externally managed) | | `GIT_DIR != GIT_COMMON`, detached HEAD | Reduced 2 options (no merge) | Externally managed — leave in place |
### Step 3: Determine Base Branch ## Step 3: Determine Base Branch
```bash The base branch is whatever this work forked from — usually named in the
# Try common base branches plan, the conversation, or the branch's upstream. If it is not already
git merge-base HEAD main 2>/dev/null || git merge-base HEAD master 2>/dev/null known, ask: "This branch split from <your best guess> - is that correct?"
``` Confirm before merging: merging into the wrong base is expensive to undo.
Or ask: "This branch split from main - is that correct?" ## Step 4: Present Options
### Step 4: Present Options **Normal repo and named-branch worktree — present exactly these 3 options:**
**Normal repo and named-branch worktree — present exactly these 4 options:**
``` ```
Implementation complete. What would you like to do? Implementation complete. What would you like to do?
@@ -73,28 +60,30 @@ Implementation complete. What would you like to do?
1. Merge back to <base-branch> locally 1. Merge back to <base-branch> locally
2. Push and create a Pull Request 2. Push and create a Pull Request
3. Keep the branch as-is (I'll handle it later) 3. Keep the branch as-is (I'll handle it later)
4. Discard this work
Which option? Which option?
``` ```
**Detached HEAD — present exactly these 3 options:** **Detached HEAD — present exactly these 2 options:**
``` ```
Implementation complete. You're on a detached HEAD (externally managed workspace). Implementation complete. You're on a detached HEAD (externally managed workspace).
1. Push as new branch and create a Pull Request 1. Push as new branch and create a Pull Request
2. Keep as-is (I'll handle it later) 2. Keep as-is (I'll handle it later)
3. Discard this work
Which option? Which option?
``` ```
**Don't add explanation** - keep options concise. Present the menu exactly as written — concise, with every option coming
from the list above. Discarding the work happens only in response to your
human partner explicitly asking for it (see "If your human partner asks to
discard the work" below). Wait for their answer; the integration decision
is theirs.
### Step 5: Execute Choice ## Step 5: Execute Choice
#### Option 1: Merge Locally ### Option 1: Merge Locally
```bash ```bash
# Get main repo root for CWD safety # Get main repo root for CWD safety
@@ -108,34 +97,43 @@ git merge <feature-branch>
# Verify tests on merged result # Verify tests on merged result
<test command> <test command>
# Only after merge succeeds: cleanup worktree (Step 6), then delete branch
``` ```
Then: Cleanup worktree (Step 6), then delete branch: If tests fail on the merged result: stop, leave the worktree and branch in
place, and investigate — nothing has been pushed, so the merge is local
and recoverable.
Once the merged result is green: clean up the worktree (Step 6), then
delete the branch:
```bash ```bash
git branch -d <feature-branch> git branch -d <feature-branch>
``` ```
#### Option 2: Push and Create PR ### Option 2: Push and Create PR
```bash ```bash
# Push branch
git push -u origin <feature-branch> git push -u origin <feature-branch>
# From a detached HEAD, name the new branch on the remote:
# git push origin HEAD:refs/heads/<new-branch>
``` ```
**Do NOT clean up worktree** — user needs it alive to iterate on PR feedback. Then create the pull/merge request against <base-branch> with the forge's
tooling — its CLI if one is available, or the creation URL most forges
print when you push — following the repo's PR template and conventions if
present, and report the URL to your human partner.
#### Option 3: Keep As-Is Keep the worktree — your human partner iterates on PR feedback there.
### Option 3: Keep As-Is
Report: "Keeping branch <name>. Worktree preserved at <path>." Report: "Keeping branch <name>. Worktree preserved at <path>."
**Don't cleanup worktree.** ### If your human partner asks to discard the work
#### Option 4: Discard This path exists only as a response to an explicit request to throw the
work away. Confirm first:
**Confirm first:**
``` ```
This will permanently delete: This will permanently delete:
- Branch <name> - Branch <name>
@@ -145,41 +143,39 @@ This will permanently delete:
Type 'discard' to confirm. Type 'discard' to confirm.
``` ```
Wait for exact confirmation. Wait for that exact confirmation. When it arrives:
If confirmed:
```bash ```bash
MAIN_ROOT=$(git -C "$(git rev-parse --git-common-dir)/.." rev-parse --show-toplevel) MAIN_ROOT=$(git -C "$(git rev-parse --git-common-dir)/.." rev-parse --show-toplevel)
cd "$MAIN_ROOT" cd "$MAIN_ROOT"
``` ```
Then: Cleanup worktree (Step 6), then force-delete branch: Then clean up the worktree (Step 6) and force-delete the branch:
```bash ```bash
git branch -D <feature-branch> git branch -D <feature-branch>
``` ```
### Step 6: Cleanup Workspace ## Step 6: Cleanup Workspace
**Only runs for Options 1 and 4.** Options 2 and 3 always preserve the worktree. **Runs for Option 1 and confirmed discards.** Options 2 and 3 always
preserve the worktree. Both callers have already changed directory to the
```bash main repo root — worktree removal must run from outside the worktree —
GIT_DIR=$(cd "$(git rev-parse --git-dir)" 2>/dev/null && pwd -P) and use the `GIT_DIR`/`GIT_COMMON`/`WORKTREE_PATH` values captured in
GIT_COMMON=$(cd "$(git rev-parse --git-common-dir)" 2>/dev/null && pwd -P) Step 2, from before that directory change.
WORKTREE_PATH=$(git rev-parse --show-toplevel)
```
**If `GIT_DIR == GIT_COMMON`:** Normal repo, no worktree to clean up. Done. **If `GIT_DIR == GIT_COMMON`:** Normal repo, no worktree to clean up. Done.
**If worktree path is under `.worktrees/` or `worktrees/`:** Superpowers created this worktree — we own cleanup. **If `WORKTREE_PATH` is under `.worktrees/` or `worktrees/`:** Superpowers
created this worktree — we own cleanup:
```bash ```bash
MAIN_ROOT=$(git -C "$(git rev-parse --git-common-dir)/.." rev-parse --show-toplevel)
cd "$MAIN_ROOT"
git worktree remove "$WORKTREE_PATH" git worktree remove "$WORKTREE_PATH"
git worktree prune # Self-healing: clean up any stale registrations git worktree prune # Self-healing: clean up any stale registrations
``` ```
**Otherwise:** The host environment (harness) owns this workspace. Do NOT remove it. If your platform provides a workspace-exit tool, use it. Otherwise, leave the workspace in place. **Otherwise:** The host environment owns this workspace — leave it in
place. If your platform provides a workspace-exit tool, use it.
## Quick Reference ## Quick Reference
@@ -188,54 +184,18 @@ git worktree prune # Self-healing: clean up any stale registrations
| 1. Merge locally | yes | - | - | yes | | 1. Merge locally | yes | - | - | yes |
| 2. Create PR | - | yes | yes | - | | 2. Create PR | - | yes | yes | - |
| 3. Keep as-is | - | - | yes | - | | 3. Keep as-is | - | - | yes | - |
| 4. Discard | - | - | - | yes (force) | | Discard (explicit request only) | - | - | - | yes (force) |
## Common Mistakes ## Common Rationalizations
**Skipping test verification** | Excuse | Reality |
- **Problem:** Merge broken code, create failing PR |--------|---------|
- **Fix:** Always verify tests before offering options | "Tests passed earlier this session" | Run the suite on the tree you are about to integrate. A green run only proves the tree it ran on. |
| "They obviously want it merged" | Integration is your human partner's decision. Present the menu and wait. |
**Open-ended questions** | "They seem done with this feature — I'll offer to discard it" | The menu is complete as written. Discard happens only when your human partner asks for it in so many words. |
- **Problem:** "What should I do next?" is ambiguous | "'Yeah, get rid of it' counts as confirmation" | Only the typed word `discard` authorizes deletion. |
- **Fix:** Present exactly 4 structured options (or 3 for detached HEAD) | "The PR is up, so the worktree is clutter now" | PR feedback gets fixed in that worktree. It stays until the work lands. |
| "This other worktree looks stale — I'll clean it too" | Clean up only worktrees under `.worktrees/` or `worktrees/`. Everything else belongs to the host. |
**Cleaning up worktree for Option 2** | "The merged-result failure is probably flaky" | A failing merged result stops everything. Branch and worktree stay put while you investigate. |
- **Problem:** Remove worktree user needs for PR iteration | "The base branch is obviously main" | Confirm the fork point or ask. Merging into the wrong base is expensive to undo. |
- **Fix:** Only cleanup for Options 1 and 4 | "The push was rejected — force-push will fix it" | A rejected push means the remote moved. Investigate; force-push only on your human partner's explicit request. |
**Deleting branch before removing worktree**
- **Problem:** `git branch -d` fails because worktree still references the branch
- **Fix:** Merge first, remove worktree, then delete branch
**Running git worktree remove from inside the worktree**
- **Problem:** Command fails silently when CWD is inside the worktree being removed
- **Fix:** Always `cd` to main repo root before `git worktree remove`
**Cleaning up harness-owned worktrees**
- **Problem:** Removing a worktree the harness created causes phantom state
- **Fix:** Only clean up worktrees under `.worktrees/` or `worktrees/`
**No confirmation for discard**
- **Problem:** Accidentally delete work
- **Fix:** Require typed "discard" confirmation
## Red Flags
**Never:**
- Proceed with failing tests
- Merge without verifying tests on result
- Delete work without confirmation
- Force-push without explicit request
- Remove a worktree before confirming merge success
- Clean up worktrees you didn't create (provenance check)
- Run `git worktree remove` from inside the worktree
**Always:**
- Verify tests before offering options
- Detect environment before presenting menu
- Present exactly 4 options (or 3 for detached HEAD)
- Get typed confirmation for Option 4
- Clean up worktree for Options 1 & 4 only
- `cd` to main repo root before worktree removal
- Run `git worktree prune` after removal
@@ -203,11 +203,3 @@ You understand 1,2,3,6. Unclear on 4,5.
## GitHub Thread Replies ## GitHub Thread Replies
When replying to inline review comments on GitHub, reply in the comment thread (`gh api repos/{owner}/{repo}/pulls/{pr}/comments/{id}/replies`), not as a top-level PR comment. When replying to inline review comments on GitHub, reply in the comment thread (`gh api repos/{owner}/{repo}/pulls/{pr}/comments/{id}/replies`), not as a top-level PR comment.
## The Bottom Line
**External feedback = suggestions to evaluate, not orders to follow.**
Verify. Question. Then implement.
No performative agreement. Technical rigor always.
@@ -5,7 +5,7 @@ description: "完成较大改动后请求代码评审,重点检查需求满足
# Requesting Code Review # Requesting Code Review
Dispatch a code reviewer subagent to catch issues before they cascade. The reviewer gets precisely crafted context for evaluation — never your session's history. This keeps the reviewer focused on the work product, not your thought process, and preserves your own context for continued work. Dispatch a code reviewer subagent to catch issues before they cascade. The reviewer gets precisely crafted context for evaluation — never your session's history.
**Core principle:** Review early, review often. **Core principle:** Review early, review often.
@@ -72,20 +72,12 @@ You: [Fix progress indicators]
[Continue to Task 3] [Continue to Task 3]
``` ```
## Integration with Workflows ## Common Rationalizations
**Subagent-Driven Development:** | Excuse | Reality |
- Review after EACH task |--------|---------|
- Catch issues before they compound | "I'll just review the diff myself instead of dispatching a reviewer" | You're the coordinator — reviewing the diff inline burns the context window you need to keep driving the work. Dispatch a reviewer subagent: the diff and the evaluation live in its context, and only the findings come back to you. |
- Fix before moving to next task | "The reviewer needs my whole session history to understand the change" | Hand it precisely crafted context, never your session's history. That keeps the reviewer on the work product, not your thought process. |
**Executing Plans:**
- Review after each task or at natural checkpoints
- Get feedback, apply, continue
**Ad-Hoc Development:**
- Review before merge
- Review when stuck
## Red Flags ## Red Flags
@@ -51,38 +51,96 @@ digraph process {
subgraph cluster_per_task { subgraph cluster_per_task {
label="Per Task"; label="Per Task";
"Dispatch implementer subagent (./implementer-prompt.md)" [shape=box]; "Dispatch implementer subagent (./implementer-prompt.md)" [shape=box];
"Implementer subagent asks questions?" [shape=diamond]; "Implementer asks questions?" [shape=diamond];
"Answer questions, provide context" [shape=box]; "Answer questions, provide context" [shape=box];
"Implementer subagent implements, tests, commits, self-reviews" [shape=box]; "Implementer implements, tests, commits, self-reviews" [shape=box];
"Write diff file, dispatch task reviewer subagent (./task-reviewer-prompt.md)" [shape=box]; "Generate review package, dispatch task reviewer (./task-reviewer-prompt.md)" [shape=box];
"Task reviewer reports spec ✅ and quality approved?" [shape=diamond]; "Spec ✅ and quality approved?" [shape=diamond];
"Dispatch fix subagent for Critical/Important findings" [shape=box]; "Finding conflicts with plan text?" [shape=diamond];
"Mark task complete in todo list and progress ledger" [shape=box]; "Ask human partner which governs" [shape=box];
"Fix round R of 5: R≤3 resume implementer; R≥4 fresh implementer, more capable model" [shape=box];
"Dispatch scoped re-review (./re-review-prompt.md)" [shape=box];
"All findings addressed?" [shape=diamond];
"R = 5?" [shape=diamond];
"Adjudicate each open finding" [shape=box];
"Any load-bearing finding?" [shape=diamond];
"STOP: report BLOCKED to human partner" [shape=box];
"Park findings in ledger with rulings" [shape=box];
"Append completion to ledger, mark todo complete" [shape=box];
} }
"Read plan, note context and global constraints, create todos" [shape=box]; "Setup: worktree, ledger check, read plan, pre-flight review" [shape=box];
"More tasks remain?" [shape=diamond]; "More tasks remain?" [shape=diamond];
"Dispatch final code reviewer subagent (../requesting-code-review/code-reviewer.md)" [shape=box]; "Dispatch final code reviewer (../requesting-code-review/code-reviewer.md)" [shape=box];
"Final findings? ONE fix dispatch, one scoped re-review, adjudicate residuals" [shape=box];
"Final review clean: delete this plan's workspace" [shape=box];
"Use superpowers:finishing-a-development-branch" [shape=box style=filled fillcolor=lightgreen]; "Use superpowers:finishing-a-development-branch" [shape=box style=filled fillcolor=lightgreen];
"Read plan, note context and global constraints, create todos" -> "Dispatch implementer subagent (./implementer-prompt.md)"; "Setup: worktree, ledger check, read plan, pre-flight review" -> "Dispatch implementer subagent (./implementer-prompt.md)";
"Dispatch implementer subagent (./implementer-prompt.md)" -> "Implementer subagent asks questions?"; "Dispatch implementer subagent (./implementer-prompt.md)" -> "Implementer asks questions?";
"Implementer subagent asks questions?" -> "Answer questions, provide context" [label="yes"]; "Implementer asks questions?" -> "Answer questions, provide context" [label="yes"];
"Answer questions, provide context" -> "Dispatch implementer subagent (./implementer-prompt.md)"; "Answer questions, provide context" -> "Implementer implements, tests, commits, self-reviews";
"Implementer subagent asks questions?" -> "Implementer subagent implements, tests, commits, self-reviews" [label="no"]; "Implementer asks questions?" -> "Implementer implements, tests, commits, self-reviews" [label="no"];
"Implementer subagent implements, tests, commits, self-reviews" -> "Write diff file, dispatch task reviewer subagent (./task-reviewer-prompt.md)"; "Implementer implements, tests, commits, self-reviews" -> "Generate review package, dispatch task reviewer (./task-reviewer-prompt.md)";
"Write diff file, dispatch task reviewer subagent (./task-reviewer-prompt.md)" -> "Task reviewer reports spec ✅ and quality approved?"; "Generate review package, dispatch task reviewer (./task-reviewer-prompt.md)" -> "Spec ✅ and quality approved?";
"Task reviewer reports spec ✅ and quality approved?" -> "Dispatch fix subagent for Critical/Important findings" [label="no"]; "Spec ✅ and quality approved?" -> "Append completion to ledger, mark todo complete" [label="yes"];
"Dispatch fix subagent for Critical/Important findings" -> "Write diff file, dispatch task reviewer subagent (./task-reviewer-prompt.md)" [label="re-review"]; "Spec ✅ and quality approved?" -> "Finding conflicts with plan text?" [label="no"];
"Task reviewer reports spec ✅ and quality approved?" -> "Mark task complete in todo list and progress ledger" [label="yes"]; "Finding conflicts with plan text?" -> "Ask human partner which governs" [label="yes"];
"Mark task complete in todo list and progress ledger" -> "More tasks remain?"; "Ask human partner which governs" -> "Fix round R of 5: R≤3 resume implementer; R≥4 fresh implementer, more capable model";
"Finding conflicts with plan text?" -> "Fix round R of 5: R≤3 resume implementer; R≥4 fresh implementer, more capable model" [label="no"];
"Fix round R of 5: R≤3 resume implementer; R≥4 fresh implementer, more capable model" -> "Dispatch scoped re-review (./re-review-prompt.md)";
"Dispatch scoped re-review (./re-review-prompt.md)" -> "All findings addressed?";
"All findings addressed?" -> "Append completion to ledger, mark todo complete" [label="yes"];
"All findings addressed?" -> "R = 5?" [label="no"];
"R = 5?" -> "Fix round R of 5: R≤3 resume implementer; R≥4 fresh implementer, more capable model" [label="no - next round"];
"R = 5?" -> "Adjudicate each open finding" [label="yes - breaker trips"];
"Adjudicate each open finding" -> "Any load-bearing finding?";
"Any load-bearing finding?" -> "STOP: report BLOCKED to human partner" [label="yes"];
"Any load-bearing finding?" -> "Park findings in ledger with rulings" [label="no"];
"Park findings in ledger with rulings" -> "Append completion to ledger, mark todo complete";
"Append completion to ledger, mark todo complete" -> "More tasks remain?";
"More tasks remain?" -> "Dispatch implementer subagent (./implementer-prompt.md)" [label="yes"]; "More tasks remain?" -> "Dispatch implementer subagent (./implementer-prompt.md)" [label="yes"];
"More tasks remain?" -> "Dispatch final code reviewer subagent (../requesting-code-review/code-reviewer.md)" [label="no"]; "More tasks remain?" -> "Dispatch final code reviewer (../requesting-code-review/code-reviewer.md)" [label="no"];
"Dispatch final code reviewer subagent (../requesting-code-review/code-reviewer.md)" -> "Use superpowers:finishing-a-development-branch"; "Dispatch final code reviewer (../requesting-code-review/code-reviewer.md)" -> "Final findings? ONE fix dispatch, one scoped re-review, adjudicate residuals";
"Final findings? ONE fix dispatch, one scoped re-review, adjudicate residuals" -> "Final review clean: delete this plan's workspace";
"Final review clean: delete this plan's workspace" -> "Use superpowers:finishing-a-development-branch";
} }
``` ```
## Pre-Flight Plan Review ## Setup
Ensure the work happens in an isolated workspace: use
superpowers:using-git-worktrees to create one or verify the existing one.
Never start implementation on a main/master branch without your human
partner's explicit consent.
Conversation memory does not survive compaction. In real sessions,
controllers that lost their place have re-dispatched entire completed task
sequences — the single most expensive failure observed. Track progress in
a ledger file, not only in todos.
- Each plan owns a workspace: at skill start, run this skill's
`scripts/sdd-workspace PLAN_FILE` — it prints the plan's git-ignored
directory (`<repo-root>/.superpowers/sdd/<plan-basename>/`), home to
every artifact for THIS plan: ledger, briefs, reports, review packages.
Another plan's directory is never yours to read or write.
- Check for this plan's ledger at `<workspace>/progress.md`. If its first
line names your plan file, tasks with a `Task <N>: complete` line are DONE
— do not re-dispatch them; resume at the first task without one. A task
whose last line is a fix round is mid-loop: resume the loop at the next
round. A ledger whose first line names a different plan file — or a stray
ledger at the old flat path `.superpowers/sdd/progress.md` — is another
plan's progress: leave it in place and start your own, fresh.
- Create the ledger with its identity as the first line:
`# SDD ledger — plan: <plan file path>`.
- The ledger is your recovery map: the commits it names exist in git even
when your context no longer remembers creating them. After compaction,
trust the ledger and `git log` over your own recollection.
- `git clean -fdx` will destroy the workspace (it's git-ignored scratch); if
that happens, recover from `git log`.
Read the plan once, note its context and Global Constraints, and create a
todo per task.
Before dispatching Task 1, scan the plan once for conflicts: Before dispatching Task 1, scan the plan once for conflicts:
@@ -110,7 +168,11 @@ capable available model, not the session default.
**Review tasks**: choose the model with the same judgment, scaled to the **Review tasks**: choose the model with the same judgment, scaled to the
diff's size, complexity, and risk. A small mechanical diff does not need the diff's size, complexity, and risk. A small mechanical diff does not need the
most capable model; a subtle concurrency change does. most capable model; a subtle concurrency change does. Scoped re-reviews of
small fix diffs take a cheap-to-mid tier.
**Fix-loop escalation (rounds 4-5)**: use a model at least one tier above
the implementer that got stuck.
**Always specify the model explicitly when dispatching a subagent.** An **Always specify the model explicitly when dispatching a subagent.** An
omitted model inherits your session's model — often the most capable and omitted model inherits your session's model — often the most capable and
@@ -129,11 +191,51 @@ that implementer. Single-file mechanical fixes also take the cheapest tier.
- Touches multiple files with integration concerns → standard model - Touches multiple files with integration concerns → standard model
- Requires design judgment or broad codebase understanding → most capable model - Requires design judgment or broad codebase understanding → most capable model
## Handling Implementer Status ## The Task Loop
Everything you paste into a dispatch prompt — and everything a subagent
prints back — stays resident in your context for the rest of the session
and is re-read on every later turn. Hand artifacts over as files.
### 1. Dispatch the implementer
Record BASE (`git rev-parse HEAD`) before dispatching — the review package
and fix-round diffs need it.
- **Task brief:** before dispatching an implementer, run this skill's
`scripts/task-brief PLAN_FILE N` — it extracts the task's full text to a
uniquely named file and prints the path. Compose the dispatch so the
brief stays the single source of
requirements. Your dispatch should contain: (1) one line on where this
task fits in the project; (2) the brief path, introduced as "read this
first — it is your requirements, with the exact values to use verbatim";
(3) interfaces and decisions from earlier tasks that the brief cannot
know; (4) your resolution of any ambiguity you noticed in the brief;
(5) the report-file path and report contract. Exact values (numbers,
magic strings, signatures, test cases) appear only in the brief. Never
make a subagent read the whole plan file.
- **Report file:** name the implementer's report file after the brief
(brief `…/task-N-brief.md` → report `…/task-N-report.md`) and put it in
the dispatch prompt. The implementer writes the full report there and
returns only status, commits, a one-line test summary, and concerns.
- A dispatch prompt describes one task, not the session's history. Do not
paste accumulated prior-task summaries ("state after Tasks 1-3") into
later dispatches — a real session's dispatch hit 42k chars of which 99%
was pasted history. A fresh subagent needs its task, the interfaces it
touches, and the global constraints. Nothing else.
- If an earlier task parked a finding in the area this task touches, carry
a pointer to that ledger entry in the dispatch.
- Record the implementer's agent identity from the dispatch result —
fix-loop rounds 1-3 resume this agent.
- Never dispatch multiple implementation subagents in parallel (conflicts).
Template: [implementer-prompt.md](implementer-prompt.md)
### 2. Handle the report
Implementer subagents report one of four statuses. Handle each appropriately: Implementer subagents report one of four statuses. Handle each appropriately:
**DONE:** Generate the review package (`scripts/review-package BASE HEAD`, from this skill's directory — it prints the unique file path it wrote; BASE is the commit you recorded before dispatching the implementer — never `HEAD~1`, which silently drops all but the last commit of a multi-commit task), then dispatch the task reviewer with the printed path. **DONE:** Generate the review package (`scripts/review-package PLAN_FILE BASE HEAD`, from this skill's directory — it prints the unique file path it wrote; BASE is the commit you recorded before dispatching the implementer — never `HEAD~1`, which silently drops all but the last commit of a multi-commit task), then dispatch the task reviewer with the printed path.
**DONE_WITH_CONCERNS:** The implementer completed the work but flagged doubts. Read the concerns before proceeding. If the concerns are about correctness or scope, address them before review. If they're observations (e.g., "this file is getting large"), note them and proceed to review. **DONE_WITH_CONCERNS:** The implementer completed the work but flagged doubts. Read the concerns before proceeding. If the concerns are about correctness or scope, address them before review. If they're observations (e.g., "this file is getting large"), note them and proceed to review.
@@ -147,20 +249,37 @@ Implementer subagents report one of four statuses. Handle each appropriately:
**Never** ignore an escalation or force the same model to retry without changes. If the implementer said it's stuck, something needs to change. **Never** ignore an escalation or force the same model to retry without changes. If the implementer said it's stuck, something needs to change.
## Handling Reviewer ⚠️ Items If the implementer asks questions — before starting or mid-task — answer
clearly and completely, provide additional context if needed, and don't
rush it into implementation.
The task reviewer may report "⚠️ Cannot verify from diff" items — requirements ### 3. Review the task
that live in unchanged code or span tasks. These do not block the rest of the
review, but you must resolve each one yourself before marking the task
complete: you hold the plan and cross-task context the reviewer
lacks. If you confirm an item is a real gap, treat it as a failed spec
review — send it back to the implementer and re-review.
## Constructing Reviewer Prompts
Per-task reviews are task-scoped gates. The broad review happens once, at the Per-task reviews are task-scoped gates. The broad review happens once, at the
final whole-branch review. When you fill a reviewer template: final whole-branch review. Never skip the task review, and never accept a
report missing either verdict — spec compliance AND task quality are both
required. Implementer self-review never replaces the task review; both are
needed.
- Hand the reviewer its diff as a file: run this skill's
`scripts/review-package PLAN_FILE BASE HEAD` and pass the reviewer the file path
it prints (or, without bash: `git log --oneline`, `git diff --stat`,
and `git diff -U10` for the range, redirected to one uniquely named
file). The output never enters your own context, and the reviewer sees
the commit list, stat summary, and full diff with context in one Read
call. Use the BASE you recorded before dispatching the implementer —
never `HEAD~1`, which silently truncates multi-commit tasks. Never
dispatch a task reviewer without a diff file.
- **Reviewer inputs:** the task reviewer gets three paths — the same brief
file, the report file, and the review package — plus the global
constraints that bind the task.
- The global-constraints block you hand the reviewer is its attention
lens. Copy the binding requirements verbatim from the plan's Global
Constraints section or the spec: exact values, exact formats, and the
stated relationships between components ("same layout as X", "matches
Y"). The reviewer's template already carries the process rules (YAGNI,
test hygiene, review method) — the constraints block is for what THIS
project's spec demands.
- Do not add open-ended directives like "check all uses" or "run race tests - Do not add open-ended directives like "check all uses" or "run race tests
if useful" without a concrete, task-specific reason if useful" without a concrete, task-specific reason
- Do not ask a reviewer to re-run tests the implementer already ran on the - Do not ask a reviewer to re-run tests the implementer already ran on the
@@ -171,110 +290,159 @@ final whole-branch review. When you fill a reviewer template:
loop. If the prompt you are writing contains "do not flag," "don't treat X loop. If the prompt you are writing contains "do not flag," "don't treat X
as a defect," "at most Minor," or "the plan chose" — stop: you are as a defect," "at most Minor," or "the plan chose" — stop: you are
pre-judging, usually to spare yourself a review loop. pre-judging, usually to spare yourself a review loop.
- The global-constraints block you hand the reviewer is its attention The task reviewer may report "⚠️ Cannot verify from diff" items — requirements
lens. Copy the binding requirements verbatim from the plan's Global that live in unchanged code or span tasks. These do not block the rest of the
Constraints section or the spec: exact values, exact formats, and the review, but you must resolve each one yourself before marking the task
stated relationships between components ("same layout as X", "matches complete: you hold the plan and cross-task context the reviewer
Y"). The reviewer's template already carries the process rules (YAGNI, lacks. If you confirm an item is a real gap, treat it as a failed spec
test hygiene, review method) — the constraints block is for what THIS review — it enters the fix loop with the other findings.
project's spec demands.
- Hand the reviewer its diff as a file: run this skill's Template: [task-reviewer-prompt.md](task-reviewer-prompt.md)
`scripts/review-package BASE HEAD` and pass the reviewer the file path
it prints (or, without bash: `git log --oneline`, `git diff --stat`, ### 4. The fix loop
and `git diff -U10` for the range, redirected to one uniquely named
file). The output never enters your own context, and the reviewer sees The loop triggers when the review reports spec ❌, any Critical or Important
the commit list, stat summary, and full diff with context in one Read finding, or a ⚠️ item you confirmed as a real gap.
call. Use the BASE you recorded before dispatching the implementer —
never `HEAD~1`, which silently truncates multi-commit tasks. Before the loop starts, two routes leave it immediately:
- A dispatch prompt describes one task, not the session's history. Do not
paste accumulated prior-task summaries ("state after Tasks 1-3") into - Record Minor findings in the progress ledger as you go
later dispatches — a real session's dispatch hit 42k chars of which 99% (`Task <N>: minor (deferred): <one-liner>`), and point the final
was pasted history. A fresh subagent needs its task, the interfaces it
touches, and the global constraints. Nothing else.
- Dispatch fix subagents for Critical and Important findings. Record Minor
findings in the progress ledger as you go, and point the final
whole-branch review at that list so it can triage which must be fixed whole-branch review at that list so it can triage which must be fixed
before merge. A roll-up nobody reads is a silent discard. before merge. A roll-up nobody reads is a silent discard. Minor findings
never enter the loop.
- A finding labeled plan-mandated — or any finding that conflicts with - A finding labeled plan-mandated — or any finding that conflicts with
what the plan's text requires — is the human's decision, like any plan what the plan's text requires — is the human's decision, like any plan
contradiction: present the finding and the plan text, ask which governs. contradiction: present the finding and the plan text, ask which governs.
Do not dismiss the finding because the plan mandates it, and do not Do not dismiss the finding because the plan mandates it, and do not
dispatch a fix that contradicts the plan without asking. dispatch a fix that contradicts the plan without asking.
- The final whole-branch review gets a package too: run Everything else enters the loop. A fix round is one fix dispatch plus one
`scripts/review-package MERGE_BASE HEAD` (MERGE_BASE = the commit the scoped re-review. Five rounds maximum per task:
branch started from, e.g. `git merge-base main HEAD`) and include the
printed path in the final review dispatch, so the final reviewer reads
one file instead of re-deriving the branch diff with git commands.
- Every fix dispatch carries the implementer contract: the fix subagent
re-runs the tests covering its change and reports the results. Name the
covering test files in the dispatch — a one-line fix does not need the
whole suite. Before re-dispatching the reviewer, confirm the fix report
contains the covering tests, the command run, and the output; dispatch
the re-review once all three are present.
- If the final whole-branch review returns findings, dispatch ONE fix
subagent with the complete findings list — not one fixer per finding.
Per-finding fixers each rebuild context and re-run suites; a real
session's final-review fix wave cost more than all its tasks combined.
## File Handoffs **Rounds 1-3 — resume the original implementer.** Send it the open findings
verbatim. Its context is intact: it knows the task, the code, and its own
choices. If your harness cannot send another message to a live subagent,
dispatch a fresh implementer carrying the brief path, the report-file path,
and the findings — the report file is the persistent memory either way.
Everything you paste into a dispatch prompt — and everything a subagent **Rounds 4-5 — dispatch a fresh implementer on a more capable model** (per
prints back — stays resident in your context for the rest of the session Model Selection), with the brief path, the report-file path, the open
and is re-read on every later turn. Hand artifacts over as files: findings, and this framing: "A prior implementer attempted this task
[N] times; you own it now. Read the report file for what was tried." A loop
that survives three resumes usually means the implementer cannot see its
own problem — fresh eyes and a capability bump in one move.
- **Task brief:** before dispatching an implementer, run this skill's **Every round, either way:** the implementer fixes, re-runs the tests
`scripts/task-brief PLAN_FILE N` — it extracts the task's full text to a covering the amended code, appends its fix report to the same report file,
uniquely named file and prints the path. Compose the dispatch so the and returns the short contract. Before re-dispatching the reviewer, confirm
brief stays the single source of requirements. Your dispatch should the fix report contains the covering tests, the command run, and the
contain: (1) one line on where this task fits in the project; (2) the output; dispatch the re-review once all three are present. Name the
brief path, introduced as "read this first — it is your requirements, covering test files in the fix message — a one-line fix does not need the
with the exact values to use verbatim"; (3) interfaces and decisions whole suite.
from earlier tasks that the brief cannot know; (4) your resolution of
any ambiguity you noticed in the brief; (5) the report-file path and
report contract. Exact values (numbers, magic strings, signatures, test
cases) appear only in the brief.
- **Report file:** name the implementer's report file after the brief
(brief `…/task-N-brief.md` → report `…/task-N-report.md`) and put it in
the dispatch prompt. The implementer writes the full report there and
returns only status, commits, a one-line test summary, and concerns.
- **Reviewer inputs:** the task reviewer gets three paths — the same brief
file, the report file, and the review package — plus the global
constraints that bind the task.
- Fix dispatches append their fix report (with test results) to the same
report file and return a short summary; re-reviews read the updated file.
## Durable Progress **The re-review is scoped.** Run `scripts/review-package PLAN_FILE FIX_BASE HEAD`
where FIX_BASE is the head the previous review saw, and dispatch
[re-review-prompt.md](re-review-prompt.md) with the findings list, the
brief, the report file, and the printed diff path. The re-reviewer verdicts
each finding ADDRESSED or NOT ADDRESSED and flags new breakage in the fix
diff only. New Critical/Important breakage in the fix diff joins the open
findings list. Out-of-scope observations go to the ledger as deferred
minors — they never extend the loop.
Conversation memory does not survive compaction. In real sessions, **After each round,** append to the ledger:
controllers that lost their place have re-dispatched entire completed task `Task <N>: fix round <R>/5 (<X> addressed, <Y> open — <finding one-liners>; commits <a7>..<b7>)`
sequences — the single most expensive failure observed. Track progress in
a ledger file, not only in todos.
- At skill start, check for a ledger: Never fix findings yourself in the controller session — your context stays
`cat "$(git rev-parse --show-toplevel)/.superpowers/sdd/progress.md"`. Tasks listed there clean for coordination, and controller fixes skip review.
as complete are DONE — do not re-dispatch them; resume at the first task
not marked complete.
- When a task's review comes back clean, append one line to the ledger in
the same message as your other bookkeeping:
`Task N: complete (commits <base7>..<head7>, review clean)`.
- The ledger is your recovery map: the commits it names exist in git even
when your context no longer remembers creating them. After compaction,
trust the ledger and `git log` over your own recollection.
- `git clean -fdx` will destroy the ledger (it's git-ignored scratch); if
that happens, recover from `git log`.
## Prompt Templates **The breaker.** When round 5's re-review still leaves findings open, stop
dispatching. Adjudicate each open finding yourself — you hold the plan and
the cross-task context the reviewer lacks:
- [implementer-prompt.md](implementer-prompt.md) - Dispatch implementer subagent - **The reviewer is wrong, or the point is contestable:** park it —
- [task-reviewer-prompt.md](task-reviewer-prompt.md) - Dispatch task reviewer subagent (spec compliance + code quality) `Task <N>: parked — <finding> — ruling: <why the code stands>`. The final
- Final whole-branch review: use superpowers:requesting-code-review's [code-reviewer.md](../requesting-code-review/code-reviewer.md) review sees both sides.
- **Real, but nothing downstream builds on it:** park it the same way, with
a ruling that says it's real and deferred.
- **Real and load-bearing** — a later task builds on it, or it reveals a
plan defect: STOP. Append `Task <N>: BLOCKED — <reason>` and report to
your human partner with the finding, the plan text it collides with, and
the fix history. Parking a structural failure lets every dependent task
build on it and hands the final review a problem it cannot fix either.
Adjudicate only at the cap. Adjudicating earlier to end a loop is
pre-judging with a different name. Every adjudication is a ledger entry —
a silent discard is forbidden.
### 5. Complete the task
When the review comes back clean — or every open finding is parked with a
ruling at the cap — append the completion line to the ledger in the same
message as your other bookkeeping:
- `Task <N>: complete (commits <base7>..<head7>, review clean)`
- `Task <N>: complete (commits <base7>..<head7>, <K> parked)` after a
tripped breaker
Then mark the todo complete and move on. Never move to the next task while
the review has open Critical/Important issues that are neither fixed nor
parked-with-ruling at the cap.
## Final Review
The final whole-branch review gets a package too: run
`scripts/review-package PLAN_FILE MERGE_BASE HEAD` (MERGE_BASE = the commit the
branch started from, e.g. `git merge-base main HEAD`) and include the
printed path in the final review dispatch, so the final reviewer reads
one file instead of re-deriving the branch diff with git commands. Dispatch
on the most capable available model (see Model Selection), using
superpowers:requesting-code-review's
[code-reviewer.md](../requesting-code-review/code-reviewer.md). Point it at
the ledger's deferred-minor and parked lines so it can triage which must be
fixed before merge.
If the final whole-branch review returns findings, dispatch ONE fix subagent
with the complete findings list — not one fixer per finding.
Per-finding fixers each rebuild context and re-run suites; a real
session's final-review fix wave cost more than all its tasks combined.
Then run exactly one scoped re-review of the fix wave
(`scripts/review-package PLAN_FILE FIX_BASE HEAD` over the fix range,
[re-review-prompt.md](re-review-prompt.md)).
Adjudicate any residual findings as in the task loop's breaker: park with
rulings, or stop on load-bearing ones. There is no second fix wave —
residual load-bearing findings surface to your human partner when
finishing-a-development-branch presents the options.
## Finish
When the final whole-branch review is clean and its fixes are merged,
delete this plan's workspace (`rm -rf <workspace>`) — the git history is
the record now. Sibling directories belong to other plans; leave them
alone.
Use superpowers:finishing-a-development-branch.
## Common Rationalizations
| Excuse | Reality |
|--------|---------|
| "Close enough on spec compliance" | Reviewer found spec gaps = not done. Fix or hit the cap and adjudicate — those are the only exits. |
| "I'll fix it myself, dispatching is overhead" | Controller fixes pollute your context and skip review. Resume the implementer. |
| "One more round will converge" | Past the cap, rounds don't converge — the failure is structural. Adjudicate and route. |
| "The reviewer will just find something new anyway" | Scoped re-reviews verify fixes; they cannot wander. New findings on untouched code go to the ledger, not the loop. |
| "This finding is obviously wrong, I'll drop it" | You adjudicate only at the cap, and every ruling is a ledger entry. Silent discards are forbidden. |
| "The fix was small, skip the re-review" | Unreviewed fixes are how regressions land. Every round ends with a scoped re-review. |
| "Reviews slow the loop down" | The loop without reviews is just unverified churn. Reviews are the loop's brakes and steering. |
| "Ledger bookkeeping is overhead" | The ledger is what survives compaction. Controllers without one have re-dispatched entire completed task sequences. |
## Example Workflow ## Example Workflow
``` ```
You: I'm using Subagent-Driven Development to execute this plan. You: I'm using Subagent-Driven Development to execute this plan.
[Setup: worktree verified]
[Read plan file once: docs/superpowers/plans/feature-plan.md] [Read plan file once: docs/superpowers/plans/feature-plan.md]
[Resolve workspace: scripts/sdd-workspace docs/superpowers/plans/feature-plan.md — no ledger inside, fresh start]
[Create todos for all tasks] [Create todos for all tasks]
Task 1: Hook installation script Task 1: Hook installation script
@@ -285,134 +453,51 @@ Implementer: "Before I begin - should the hook be installed at user or system le
You: "User level (~/.config/superpowers/hooks/)" You: "User level (~/.config/superpowers/hooks/)"
Implementer: "Got it. Implementing now..." Implementer: [Later]
[Later] Implementer:
- Implemented install-hook command - Implemented install-hook command
- Added tests, 5/5 passing - Added tests, 5/5 passing
- Self-review: Found I missed --force flag, added it - Self-review: Found I missed --force flag, added it
- Committed - Committed
[Run review-package, dispatch task reviewer with the printed path] [Run review-package PLAN_FILE BASE HEAD; dispatch task reviewer with the printed path]
Task reviewer: Spec ✅ - all requirements met, nothing extra. Task reviewer: Spec ✅ - all requirements met, nothing extra.
Strengths: Good test coverage, clean. Issues: None. Task quality: Approved. Strengths: Good test coverage, clean. Issues: None. Task quality: Approved.
[Mark Task 1 complete] [Ledger: Task 1: complete (commits a1b2c3d..d4e5f6a, review clean)]
Task 2: Recovery modes Task 2: Recovery modes
[Run task-brief for Task 2; dispatch implementer with brief + report paths + context] [Run task-brief for Task 2; dispatch implementer with brief + report paths + context]
Implementer: [No questions, proceeds] Implementer: [No questions]
Implementer:
- Added verify/repair modes - Added verify/repair modes
- 8/8 tests passing - 8/8 tests passing
- Self-review: All good
- Committed - Committed
[Run review-package, dispatch task reviewer with the printed path] [Run review-package PLAN_FILE BASE HEAD; dispatch task reviewer with the printed path]
Task reviewer: Spec ❌: Task reviewer: Spec ❌:
- Missing: Progress reporting (spec says "report every 100 items") - Missing: Progress reporting (spec says "report every 100 items")
- Extra: Added --json flag (not requested)
Issues (Important): Magic number (100) Issues (Important): Magic number (100)
[Dispatch fix subagent with all findings] [Fix round 1: resume the implementer with both findings]
Fixer: Removed --json flag, added progress reporting, extracted PROGRESS_INTERVAL constant Implementer: Added progress reporting, extracted PROGRESS_INTERVAL constant.
Re-ran test/recovery.test.js — 10/10 passing. Fix report appended.
[Task reviewer reviews again] [Run review-package PLAN_FILE FIX_BASE HEAD; dispatch scoped re-review]
Task reviewer: Spec ✅. Task quality: Approved. Re-reviewer: Missing progress reporting — ADDRESSED (src/recovery.js:41).
Magic number — ADDRESSED (src/recovery.js:7). New breakage: none.
Verdict: all findings addressed.
[Mark Task 2 complete] [Ledger: Task 2: fix round 1/5 (2 addressed, 0 open; commits d4e5f6a..b7c8d9e)]
[Ledger: Task 2: complete (commits d4e5f6a..b7c8d9e, review clean)]
... ...
[After all tasks] [After all tasks]
[Dispatch final code-reviewer] [Run review-package PLAN_FILE MERGE_BASE HEAD; dispatch final code-reviewer, most capable model]
Final reviewer: All requirements met, ready to merge Final reviewer: All requirements met. Deferred minors triaged: none block merge.
Done! [Delete this plan's workspace — the record now lives in git]
Done! Using superpowers:finishing-a-development-branch.
``` ```
## Advantages
**vs. Manual execution:**
- Subagents follow TDD naturally
- Fresh context per task (no confusion)
- Parallel-safe (subagents don't interfere)
- Subagent can ask questions (before AND during work)
**vs. Executing Plans:**
- Same session (no handoff)
- Continuous progress (no waiting)
- Review checkpoints automatic
**Efficiency gains:**
- Controller curates exactly what context is needed; bulk artifacts move
as files, not pasted text
- Subagent gets complete information upfront
- Questions surfaced before work begins (not after)
**Quality gates:**
- Self-review catches issues before handoff
- Task review carries two verdicts: spec compliance and code quality
- Review loops ensure fixes actually work
- Spec compliance prevents over/under-building
- Code quality ensures implementation is well-built
**Cost:**
- More subagent invocations (implementer + reviewer per task)
- Controller does more prep work (extracting all tasks upfront)
- Review loops add iterations
- But catches issues early (cheaper than debugging later)
## Red Flags
**Never:**
- Start implementation on main/master branch without explicit user consent
- Skip task review, or accept a report missing either verdict (spec compliance AND task quality are both required)
- Proceed with unfixed issues
- Dispatch multiple implementation subagents in parallel (conflicts)
- Make a subagent read the whole plan file (hand it its task brief —
`scripts/task-brief` — instead)
- Skip scene-setting context (subagent needs to understand where task fits)
- Ignore subagent questions (answer before letting them proceed)
- Accept "close enough" on spec compliance (reviewer found spec issues = not done)
- Skip review loops (reviewer found issues = implementer fixes = review again)
- Let implementer self-review replace actual review (both are needed)
- Tell a reviewer what not to flag, or pre-rate a finding's severity in the
dispatch prompt ("treat it as Minor at most") — the plan's example code is
a starting point, not evidence that its weaknesses were chosen
- Dispatch a task reviewer without a diff file — generate it first
(`scripts/review-package BASE HEAD`) and name the printed path in the
prompt
- Move to next task while the review has open Critical/Important issues
- Re-dispatch a task the progress ledger already marks complete — check
the ledger (and `git log`) after any compaction or resume
**If subagent asks questions:**
- Answer clearly and completely
- Provide additional context if needed
- Don't rush them into implementation
**If reviewer finds issues:**
- Implementer (same subagent) fixes them
- Reviewer reviews again
- Repeat until approved
- Don't skip the re-review
**If subagent fails task:**
- Dispatch fix subagent with specific instructions
- Don't try to fix manually (context pollution)
## Integration
**Required workflow skills:**
- **superpowers:using-git-worktrees** - Ensures isolated workspace (creates one or verifies existing)
- **superpowers:writing-plans** - Creates the plan this skill executes
- **superpowers:requesting-code-review** - Code review template for the final whole-branch review
- **superpowers:finishing-a-development-branch** - Complete development after all tasks
**Subagents should use:**
- **superpowers:test-driven-development** - Subagents follow TDD for each task
**Alternative workflow:**
- **superpowers:executing-plans** - Use for parallel session instead of same-session execution
@@ -106,9 +106,12 @@ Subagent (general-purpose):
## After Review Findings ## After Review Findings
If a reviewer finds issues and you fix them, re-run the tests that cover If the task review finds issues, you will be resumed with the findings.
the amended code and append the results to your report file. Reviewers Fix them, re-run the tests that cover the amended code, and append a fix
will not re-run tests for you — your report is the test evidence. report to your report file: what you changed, the covering tests you
ran, the command, and the output. Reviewers will not re-run tests for
you — your report is the test evidence. Then reply with the same short
status contract as your first report.
## Report Format ## Report Format
@@ -0,0 +1,106 @@
# Scoped Re-Review Prompt Template
Use this template when dispatching a re-review after a fix round. The
re-reviewer verifies the findings were addressed and checks the fix diff for
new breakage. It is not a fresh review — the full review already happened.
**Purpose:** Verify each finding from the previous review was addressed, and
that the fix itself broke nothing.
```
Subagent (general-purpose):
description: "Re-review Task N fix round R"
model: [MODEL — REQUIRED: choose per SKILL.md Model Selection; an omitted
model silently inherits the session's most expensive one]
prompt: |
You are re-reviewing one task's fix round. A previous review produced
findings; an implementer has attempted to fix them. Your job is to
verdict each finding and inspect the fix diff — nothing else.
## The Task
Read the task brief: [BRIEF_FILE]
## The Findings Under Verification
[FINDINGS]
## The Fix
Read the implementer's report (fix reports are appended at the end):
[REPORT_FILE]
**Fix base:** [FIX_BASE_SHA] (the head the previous review saw)
**Head:** [HEAD_SHA]
**Diff file:** [DIFF_FILE]
Read the diff file once — it contains the fix commits, a stat summary,
and the fix diff with surrounding context. Do not re-run git commands.
If the diff file is missing, fetch the diff yourself:
`git diff --stat [FIX_BASE_SHA]..[HEAD_SHA]` and
`git diff [FIX_BASE_SHA]..[HEAD_SHA]`.
Your review is read-only on this checkout. Do not mutate the working
tree, the index, HEAD, or branch state in any way.
## Scope
Your scope is the findings list and the fix diff. Verdict every finding.
Inspect the fix diff for new problems the fix itself introduced. Do NOT
re-review code the fix did not touch: if you notice an issue entirely
outside the fix diff, report it under Out-of-Scope Observations — it
does not block this task and does not extend the loop. A broad
whole-branch review happens after all tasks are complete.
## Tests
The implementer re-ran the tests covering the amended code and appended
the results to the report file. Treat the report as unverified claims:
confirm the fix report names the covering tests and shows their output,
and verify the claims against the diff. Do not re-run the suite to
confirm their report. Run a test only when reading the code raises a
specific doubt that no existing run answers — and then a focused test,
never a package-wide suite.
## Output Format
Your final message is the report itself: begin directly with the first
finding's verdict. Every line is a verdict, a finding with file:line,
or a check you ran — no preamble, no process narration.
### Finding Verdicts
For each finding in The Findings Under Verification, in order:
- **[finding one-liner]** — ADDRESSED | NOT ADDRESSED, with file:line
evidence. "Attempted" is not addressed: the specific defect must no
longer exist.
### New Breakage in the Fix Diff
Anything the fix itself broke or introduced, with severity
(Critical/Important/Minor) and file:line. "None" if clean.
### Out-of-Scope Observations
Issues you noticed entirely outside the fix diff. Non-blocking; the
controller ledgers these for the final review. "None" if none.
### Verdict
**Fix round:** [All findings addressed, no new Critical/Important
breakage | Findings remain open] — list the open ones.
```
**Placeholders:**
- `[MODEL]` — REQUIRED: reviewer model per SKILL.md Model Selection; scoped
re-reviews of small fix diffs take a cheap-to-mid tier
- `[BRIEF_FILE]` — the task brief file (same file the implementer worked from)
- `[FINDINGS]` — the Critical/Important findings and spec gaps from the
previous review, copied verbatim, one per bullet
- `[REPORT_FILE]` — the implementer's report file (fix reports appended)
- `[FIX_BASE_SHA]` — the head the previous review saw
- `[HEAD_SHA]` — current commit
- `[DIFF_FILE]` — the path `scripts/review-package PLAN_FILE FIX_BASE HEAD` printed
**Re-reviewer returns:** per-finding verdicts (ADDRESSED / NOT ADDRESSED),
new breakage in the fix diff, out-of-scope observations, and a round verdict.
@@ -4,26 +4,28 @@
# call. Using the recorded per-task BASE (not HEAD~1) keeps multi-commit # call. Using the recorded per-task BASE (not HEAD~1) keeps multi-commit
# tasks intact. # tasks intact.
# #
# Usage: review-package BASE HEAD [OUTFILE] # Usage: review-package PLAN_FILE BASE HEAD [OUTFILE]
# Default OUTFILE: <repo-root>/.superpowers/sdd/review-<base7>..<head7>.diff # Default OUTFILE: <repo-root>/.superpowers/sdd/<plan-basename>/review-<base7>..<head7>.diff
# (named per range, so a re-review after fixes gets a distinct fresh file). # (named per range, so a re-review after fixes gets a distinct fresh file).
set -euo pipefail set -euo pipefail
if [ $# -lt 2 ] || [ $# -gt 3 ]; then if [ $# -lt 3 ] || [ $# -gt 4 ]; then
echo "usage: review-package BASE HEAD [OUTFILE]" >&2 echo "usage: review-package PLAN_FILE BASE HEAD [OUTFILE]" >&2
exit 2 exit 2
fi fi
base=$1 plan=$1
head=$2 base=$2
head=$3
[ -f "$plan" ] || { echo "no such plan file: $plan" >&2; exit 2; }
git rev-parse --verify --quiet "$base" >/dev/null || { echo "bad BASE: $base" >&2; exit 2; } git rev-parse --verify --quiet "$base" >/dev/null || { echo "bad BASE: $base" >&2; exit 2; }
git rev-parse --verify --quiet "$head" >/dev/null || { echo "bad HEAD: $head" >&2; exit 2; } git rev-parse --verify --quiet "$head" >/dev/null || { echo "bad HEAD: $head" >&2; exit 2; }
if [ $# -eq 3 ]; then if [ $# -eq 4 ]; then
out=$3 out=$4
else else
dir=$("$(cd "$(dirname "$0")" && pwd)/sdd-workspace") dir=$("$(cd "$(dirname "$0")" && pwd)/sdd-workspace" "$plan")
out="$dir/review-$(git rev-parse --short "$base")..$(git rev-parse --short "$head").diff" out="$dir/review-$(git rev-parse --short "$base")..$(git rev-parse --short "$head").diff"
fi fi
@@ -1,22 +1,40 @@
#!/usr/bin/env bash #!/usr/bin/env bash
# Resolve and ensure the working-tree directory SDD uses for its short-lived # Resolve and ensure the working-tree directory SDD uses for one plan's
# artifacts: task briefs, implementer reports, review packages, and the # short-lived artifacts: task briefs, implementer reports, review packages,
# progress ledger. Print the directory's absolute path. # and the progress ledger. Print the plan directory's absolute path.
#
# One directory per plan (.superpowers/sdd/<plan-basename>/) so a follow-up
# plan in the same working tree can never read or overwrite another plan's
# artifacts. A stale ledger misread as current progress makes controllers
# skip whole task sequences — plan-scoping removes that failure structurally.
# #
# The workspace lives in the working tree (not under .git/) because Claude Code # The workspace lives in the working tree (not under .git/) because Claude Code
# treats .git/ as a protected path and denies agent writes there — which blocks # treats .git/ as a protected path and denies agent writes there — which blocks
# an implementer subagent from writing its report file. A self-ignoring # an implementer subagent from writing its report file. A self-ignoring
# .gitignore keeps the workspace out of `git status` and out of accidental # .gitignore at .superpowers/sdd/ keeps every plan's workspace out of
# commits without modifying any tracked file. # `git status` and out of accidental commits without modifying any tracked file.
# #
# Single source of truth for the workspace location, so task-brief and # Single source of truth for the workspace location, so task-brief and
# review-package cannot drift to different directories. # review-package cannot drift to different directories.
# #
# Usage: sdd-workspace # Usage: sdd-workspace PLAN_FILE
set -euo pipefail set -euo pipefail
if [ $# -ne 1 ]; then
echo "usage: sdd-workspace PLAN_FILE" >&2
exit 2
fi
plan=$1
[ -f "$plan" ] || { echo "no such plan file: $plan" >&2; exit 2; }
slug=$(basename "$plan" .md)
[ -n "$slug" ] && [ "$slug" != "." ] && [ "$slug" != ".." ] \
|| { echo "cannot derive a workspace name from: $plan" >&2; exit 2; }
root=$(git rev-parse --show-toplevel) root=$(git rev-parse --show-toplevel)
dir="$root/.superpowers/sdd" base="$root/.superpowers/sdd"
dir="$base/$slug"
mkdir -p "$dir" mkdir -p "$dir"
printf '*\n' > "$dir/.gitignore" printf '*\n' > "$base/.gitignore"
cd "$dir" && pwd cd "$dir" && pwd
@@ -4,8 +4,9 @@
# through the controller's context. # through the controller's context.
# #
# Usage: task-brief PLAN_FILE TASK_NUMBER [OUTFILE] # Usage: task-brief PLAN_FILE TASK_NUMBER [OUTFILE]
# Default OUTFILE: <repo-root>/.superpowers/sdd/task-<N>-brief.md # Default OUTFILE: <repo-root>/.superpowers/sdd/<plan-basename>/task-<N>-brief.md
# (per worktree; concurrent runs in the same working tree share it). # (per plan and per worktree; concurrent runs of the SAME plan in the same
# working tree share it).
set -euo pipefail set -euo pipefail
if [ $# -lt 2 ] || [ $# -gt 3 ]; then if [ $# -lt 2 ] || [ $# -gt 3 ]; then
@@ -20,7 +21,7 @@ n=$2
if [ $# -eq 3 ]; then if [ $# -eq 3 ]; then
out=$3 out=$3
else else
dir=$("$(cd "$(dirname "$0")" && pwd)/sdd-workspace") dir=$("$(cd "$(dirname "$0")" && pwd)/sdd-workspace" "$plan")
out="$dir/task-${n}-brief.md" out="$dir/task-${n}-brief.md"
fi fi
@@ -178,11 +178,8 @@ Subagent (general-purpose):
- `[BASE_SHA]` — commit before this task - `[BASE_SHA]` — commit before this task
- `[HEAD_SHA]` — current commit - `[HEAD_SHA]` — current commit
- `[DIFF_FILE]` — REQUIRED: the path the controller wrote the review - `[DIFF_FILE]` — REQUIRED: the path the controller wrote the review
package to (`scripts/review-package BASE HEAD` prints the unique path it package to (`scripts/review-package PLAN_FILE BASE HEAD` prints the unique
wrote; the package never enters the controller's context) path it wrote; the package never enters the controller's context)
**Reviewer returns:** Spec Compliance verdict (✅/❌/⚠️), Strengths, Issues **Reviewer returns:** Spec Compliance verdict (✅/❌/⚠️), Strengths, Issues
(Critical/Important/Minor), Task quality verdict (Critical/Important/Minor), Task quality verdict
A fix dispatch can address spec gaps and quality findings together;
re-review after fixes covers both verdicts.
@@ -7,8 +7,6 @@ description: "遇到 bug、测试失败或异常行为时使用,通过复现
## Overview ## Overview
Random fixes waste time and create new bugs. Quick patches mask underlying issues.
**Core principle:** ALWAYS find root cause before attempting fixes. Symptom fixes are failure. **Core principle:** ALWAYS find root cause before attempting fixes. Symptom fixes are failure.
**Violating the letter of this process is violating the spirit of debugging.** **Violating the letter of this process is violating the spirit of debugging.**
@@ -188,6 +186,7 @@ You MUST complete each phase before proceeding to the next.
- Test passes now? - Test passes now?
- No other tests broken? - No other tests broken?
- Issue actually resolved? - Issue actually resolved?
- Use the `superpowers:verification-before-completion` skill before claiming success
4. **If Fix Doesn't Work** 4. **If Fix Doesn't Work**
- STOP - STOP
@@ -282,15 +281,3 @@ These techniques are part of systematic debugging and available in this director
- **`root-cause-tracing.md`** - Trace bugs backward through call stack to find original trigger - **`root-cause-tracing.md`** - Trace bugs backward through call stack to find original trigger
- **`defense-in-depth.md`** - Add validation at multiple layers after finding root cause - **`defense-in-depth.md`** - Add validation at multiple layers after finding root cause
- **`condition-based-waiting.md`** - Replace arbitrary timeouts with condition polling - **`condition-based-waiting.md`** - Replace arbitrary timeouts with condition polling
**Related skills:**
- **superpowers:test-driven-development** - For creating failing test case (Phase 4, Step 1)
- **superpowers:verification-before-completion** - Verify fix worked before claiming success
## Real-World Impact
From debugging sessions:
- Systematic approach: 15-30 minutes to fix
- Random fixes approach: 2-3 hours of thrashing
- First-time fix rate: 95% vs 40%
- New bugs introduced: Near zero vs common
@@ -18,9 +18,18 @@ echo "🔍 Searching for test that creates: $POLLUTION_CHECK"
echo "Test pattern: $TEST_PATTERN" echo "Test pattern: $TEST_PATTERN"
echo "" echo ""
# Get list of test files # Get list of test files (find . emits ./-prefixed paths, so accept the
TEST_FILES=$(find . -path "$TEST_PATTERN" | sort) # pattern written with or without a leading ./)
TOTAL=$(echo "$TEST_FILES" | wc -l | tr -d ' ') TEST_PATTERN="${TEST_PATTERN#./}"
# find -path can't match '**/' against zero directory levels, so a pattern
# like src/**/*.test.ts would skip src/top.test.ts; also try the pattern
# with '**/' collapsed to cover files directly under the base directory.
TEST_FILES=$(find . \( -path "./$TEST_PATTERN" -o -path "./${TEST_PATTERN//\*\*\//}" \) | sort -u)
if [ -z "$TEST_FILES" ]; then
TOTAL=0
else
TOTAL=$(printf '%s\n' "$TEST_FILES" | wc -l | tr -d ' ')
fi
echo "Found $TOTAL test files" echo "Found $TOTAL test files"
echo "" echo ""
@@ -203,69 +203,25 @@ Next failing test for next feature.
| **Clear** | Name describes behavior | `test('test1')` | | **Clear** | Name describes behavior | `test('test1')` |
| **Shows intent** | Demonstrates desired API | Obscures what code should do | | **Shows intent** | Demonstrates desired API | Obscures what code should do |
## Why Order Matters When writing or changing any test, read [writing-good-tests.md](writing-good-tests.md) for the rules that keep tests honest:
- Name the production change that would make the test fail — before writing it
**"I'll write tests after to verify it works"** - Assert on real behavior, never on mock behavior
- Keep test-only code in test utilities, out of production classes
Tests written after code pass immediately. Passing immediately proves nothing: - Understand a dependency's side effects before mocking it
- Might test wrong thing
- Might test implementation, not behavior
- Might miss edge cases you forgot
- You never saw it catch the bug
Test-first forces you to see the test fail, proving it actually tests something.
**"I already manually tested all the edge cases"**
Manual testing is ad-hoc. You think you tested everything but:
- No record of what you tested
- Can't re-run when code changes
- Easy to forget cases under pressure
- "It worked when I tried it" ≠ comprehensive
Automated tests are systematic. They run the same way every time.
**"Deleting X hours of work is wasteful"**
Sunk cost fallacy. The time is already gone. Your choice now:
- Delete and rewrite with TDD (X more hours, high confidence)
- Keep it and add tests after (30 min, low confidence, likely bugs)
The "waste" is keeping code you can't trust. Working code without real tests is technical debt.
**"TDD is dogmatic, being pragmatic means adapting"**
TDD IS pragmatic:
- Finds bugs before commit (faster than debugging after)
- Prevents regressions (tests catch breaks immediately)
- Documents behavior (tests show how to use code)
- Enables refactoring (change freely, tests catch breaks)
"Pragmatic" shortcuts = debugging in production = slower.
**"Tests after achieve the same goals - it's spirit not ritual"**
No. Tests-after answer "What does this do?" Tests-first answer "What should this do?"
Tests-after are biased by your implementation. You test what you built, not what's required. You verify remembered edge cases, not discovered ones.
Tests-first force edge case discovery before implementing. Tests-after verify you remembered everything (you didn't).
30 minutes of tests after ≠ TDD. You get coverage, lose proof tests work.
## Common Rationalizations ## Common Rationalizations
| Excuse | Reality | | Excuse | Reality |
|--------|---------| |--------|---------|
| "Too simple to test" | Simple code breaks. Test takes 30 seconds. | | "Too simple to test" | Simple code breaks. Test takes 30 seconds. |
| "I'll test after" | Tests passing immediately prove nothing. | | "I'll test after" | Tests written after pass immediately — which proves nothing. They may test the wrong thing, test the implementation instead of the behavior, or miss the edge case you forgot. You never watched it fail, so you never proved it can catch the bug. Test-first forces that failure. |
| "Tests after achieve same goals" | Tests-after = "what does this do?" Tests-first = "what should this do?" | | "Tests after achieve same goals (spirit not ritual)" | Tests-after answer "what does this do?"; tests-first answer "what should this do?" Tests written after are biased by the code you already wrote — you verify the cases you remembered, not the ones you'd have discovered. Coverage without proof the tests work. |
| "Already manually tested" | Ad-hoc ≠ systematic. No record, can't re-run. | | "Already manually tested" | Manual testing is ad-hoc: no record of what you covered, no way to re-run it when the code changes, easy to forget cases under pressure. "Worked when I tried it" ≠ comprehensive. Automated tests run the same way every time. |
| "Deleting X hours is wasteful" | Sunk cost fallacy. Keeping unverified code is technical debt. | | "Deleting X hours is wasteful" | Sunk cost fallacy — that time is already spent either way. The real choice: rewrite with TDD (high confidence) vs. keep it and bolt tests on after (low confidence, likely bugs). Keeping code you can't trust is the waste. |
| "Keep as reference, write tests first" | You'll adapt it. That's testing after. Delete means delete. | | "Keep as reference, write tests first" | You'll adapt it. That's testing after. Delete means delete. |
| "Need to explore first" | Fine. Throw away exploration, start with TDD. | | "Need to explore first" | Fine. Throw away exploration, start with TDD. |
| "Test hard = design unclear" | Listen to test. Hard to test = hard to use. | | "Test hard = design unclear" | Listen to test. Hard to test = hard to use. |
| "TDD will slow me down" | TDD faster than debugging. Pragmatic = test-first. | | "TDD will slow me down" | TDD IS the pragmatic path: catches bugs before commit, prevents regressions, lets you refactor without fear. "Pragmatic" shortcuts mean debugging in production — slower, not faster. |
| "Manual test faster" | Manual doesn't prove edge cases. You'll re-test every change. | | "Manual test faster" | Manual doesn't prove edge cases. You'll re-test every change. |
| "Existing code has no tests" | You're improving it. Add tests for existing code. | | "Existing code has no tests" | You're improving it. Add tests for existing code. |
@@ -354,13 +310,6 @@ Bug found? Write failing test reproducing it. Follow TDD cycle. Test proves fix
Never fix bugs without a test. Never fix bugs without a test.
## Testing Anti-Patterns
When adding mocks or test utilities, read [testing-anti-patterns.md](testing-anti-patterns.md) to avoid common pitfalls:
- Testing mock behavior instead of real behavior
- Adding test-only methods to production classes
- Mocking without understanding dependencies
## Final Rule ## Final Rule
``` ```
@@ -1,299 +0,0 @@
# Testing Anti-Patterns
**Load this reference when:** writing or changing tests, adding mocks, or tempted to add test-only methods to production code.
## Overview
Tests must verify real behavior, not mock behavior. Mocks are a means to isolate, not the thing being tested.
**Core principle:** Test what the code does, not what the mocks do.
**Following strict TDD prevents these anti-patterns.**
## The Iron Laws
```
1. NEVER test mock behavior
2. NEVER add test-only methods to production classes
3. NEVER mock without understanding dependencies
```
## Anti-Pattern 1: Testing Mock Behavior
**The violation:**
```typescript
// ❌ BAD: Testing that the mock exists
test('renders sidebar', () => {
render(<Page />);
expect(screen.getByTestId('sidebar-mock')).toBeInTheDocument();
});
```
**Why this is wrong:**
- You're verifying the mock works, not that the component works
- Test passes when mock is present, fails when it's not
- Tells you nothing about real behavior
**your human partner's correction:** "Are we testing the behavior of a mock?"
**The fix:**
```typescript
// ✅ GOOD: Test real component or don't mock it
test('renders sidebar', () => {
render(<Page />); // Don't mock sidebar
expect(screen.getByRole('navigation')).toBeInTheDocument();
});
// OR if sidebar must be mocked for isolation:
// Don't assert on the mock - test Page's behavior with sidebar present
```
### Gate Function
```
BEFORE asserting on any mock element:
Ask: "Am I testing real component behavior or just mock existence?"
IF testing mock existence:
STOP - Delete the assertion or unmock the component
Test real behavior instead
```
## Anti-Pattern 2: Test-Only Methods in Production
**The violation:**
```typescript
// ❌ BAD: destroy() only used in tests
class Session {
async destroy() { // Looks like production API!
await this._workspaceManager?.destroyWorkspace(this.id);
// ... cleanup
}
}
// In tests
afterEach(() => session.destroy());
```
**Why this is wrong:**
- Production class polluted with test-only code
- Dangerous if accidentally called in production
- Violates YAGNI and separation of concerns
- Confuses object lifecycle with entity lifecycle
**The fix:**
```typescript
// ✅ GOOD: Test utilities handle test cleanup
// Session has no destroy() - it's stateless in production
// In test-utils/
export async function cleanupSession(session: Session) {
const workspace = session.getWorkspaceInfo();
if (workspace) {
await workspaceManager.destroyWorkspace(workspace.id);
}
}
// In tests
afterEach(() => cleanupSession(session));
```
### Gate Function
```
BEFORE adding any method to production class:
Ask: "Is this only used by tests?"
IF yes:
STOP - Don't add it
Put it in test utilities instead
Ask: "Does this class own this resource's lifecycle?"
IF no:
STOP - Wrong class for this method
```
## Anti-Pattern 3: Mocking Without Understanding
**The violation:**
```typescript
// ❌ BAD: Mock breaks test logic
test('detects duplicate server', () => {
// Mock prevents config write that test depends on!
vi.mock('ToolCatalog', () => ({
discoverAndCacheTools: vi.fn().mockResolvedValue(undefined)
}));
await addServer(config);
await addServer(config); // Should throw - but won't!
});
```
**Why this is wrong:**
- Mocked method had side effect test depended on (writing config)
- Over-mocking to "be safe" breaks actual behavior
- Test passes for wrong reason or fails mysteriously
**The fix:**
```typescript
// ✅ GOOD: Mock at correct level
test('detects duplicate server', () => {
// Mock the slow part, preserve behavior test needs
vi.mock('MCPServerManager'); // Just mock slow server startup
await addServer(config); // Config written
await addServer(config); // Duplicate detected ✓
});
```
### Gate Function
```
BEFORE mocking any method:
STOP - Don't mock yet
1. Ask: "What side effects does the real method have?"
2. Ask: "Does this test depend on any of those side effects?"
3. Ask: "Do I fully understand what this test needs?"
IF depends on side effects:
Mock at lower level (the actual slow/external operation)
OR use test doubles that preserve necessary behavior
NOT the high-level method the test depends on
IF unsure what test depends on:
Run test with real implementation FIRST
Observe what actually needs to happen
THEN add minimal mocking at the right level
Red flags:
- "I'll mock this to be safe"
- "This might be slow, better mock it"
- Mocking without understanding the dependency chain
```
## Anti-Pattern 4: Incomplete Mocks
**The violation:**
```typescript
// ❌ BAD: Partial mock - only fields you think you need
const mockResponse = {
status: 'success',
data: { userId: '123', name: 'Alice' }
// Missing: metadata that downstream code uses
};
// Later: breaks when code accesses response.metadata.requestId
```
**Why this is wrong:**
- **Partial mocks hide structural assumptions** - You only mocked fields you know about
- **Downstream code may depend on fields you didn't include** - Silent failures
- **Tests pass but integration fails** - Mock incomplete, real API complete
- **False confidence** - Test proves nothing about real behavior
**The Iron Rule:** Mock the COMPLETE data structure as it exists in reality, not just fields your immediate test uses.
**The fix:**
```typescript
// ✅ GOOD: Mirror real API completeness
const mockResponse = {
status: 'success',
data: { userId: '123', name: 'Alice' },
metadata: { requestId: 'req-789', timestamp: 1234567890 }
// All fields real API returns
};
```
### Gate Function
```
BEFORE creating mock responses:
Check: "What fields does the real API response contain?"
Actions:
1. Examine actual API response from docs/examples
2. Include ALL fields system might consume downstream
3. Verify mock matches real response schema completely
Critical:
If you're creating a mock, you must understand the ENTIRE structure
Partial mocks fail silently when code depends on omitted fields
If uncertain: Include all documented fields
```
## Anti-Pattern 5: Integration Tests as Afterthought
**The violation:**
```
✅ Implementation complete
❌ No tests written
"Ready for testing"
```
**Why this is wrong:**
- Testing is part of implementation, not optional follow-up
- TDD would have caught this
- Can't claim complete without tests
**The fix:**
```
TDD cycle:
1. Write failing test
2. Implement to pass
3. Refactor
4. THEN claim complete
```
## When Mocks Become Too Complex
**Warning signs:**
- Mock setup longer than test logic
- Mocking everything to make test pass
- Mocks missing methods real components have
- Test breaks when mock changes
**your human partner's question:** "Do we need to be using a mock here?"
**Consider:** Integration tests with real components often simpler than complex mocks
## TDD Prevents These Anti-Patterns
**Why TDD helps:**
1. **Write test first** → Forces you to think about what you're actually testing
2. **Watch it fail** → Confirms test tests real behavior, not mocks
3. **Minimal implementation** → No test-only methods creep in
4. **Real dependencies** → You see what the test actually needs before mocking
**If you're testing mock behavior, you violated TDD** - you added mocks without watching test fail against real code first.
## Quick Reference
| Anti-Pattern | Fix |
|--------------|-----|
| Assert on mock elements | Test real component or unmock it |
| Test-only methods in production | Move to test utilities |
| Mock without understanding | Understand dependencies first, mock minimally |
| Incomplete mocks | Mirror real API completely |
| Tests as afterthought | TDD - tests first |
| Over-complex mocks | Consider integration tests |
## Red Flags
- Assertion checks for `*-mock` test IDs
- Methods only called in test files
- Mock setup is >50% of test
- Test fails when you remove mock
- Can't explain why mock is needed
- Mocking "just to be safe"
## The Bottom Line
**Mocks are tools to isolate, not things to test.**
If TDD reveals you're testing mock behavior, you've gone wrong.
Fix: Test real behavior or question why you're mocking at all.
@@ -0,0 +1,198 @@
# Writing Good Tests
**Load this reference when:** writing or changing tests, adding mocks, or
adding cleanup/helper methods for tests.
## Overview
A test exists to catch a specific break. Two principles govern everything
here:
```
1. Every test names the break it catches
2. Every test exercises the real thing
```
Strict TDD produces both naturally: a test written first and watched
failing against real code has already proven it can fail, and only earns
a mock when the real dependency proves slow or external.
## Principle 1: Name the Break
Before writing the test body, answer: **what production change should
make this test fail — and is that change a bug or a decision?** A test
earns its place by catching a wrong branch, missing side effect, wrong
argument, boundary case, or broken contract.
**Derive expectations independently.** Use literals and hand-checked
fixtures; table-driven tests with literal `want` values are the preferred
shape. An expectation computed by the code under test — or its helpers —
passes no matter what that code does:
```typescript
// ❌ Mirror assertion: the same builder computes both sides — always true
const expected = buildSearchQuery({ tag: 'urgent' });
expect(buildSearchQuery({ tag: 'urgent' })).toBe(expected);
// ✅ Hand-derived literal
expect(buildSearchQuery({ tag: 'urgent' })).toBe('tag:"urgent"');
```
**No change detectors.** If only intentional decisions can fail a test —
a constant's value, exact message wording, private structure — it fires
on redesign and sleeps through bugs. Test the behavior that depends on
the decision: not `expect(MAX_RETRIES).toBe(5)` but "a failing call is
retried 5 times and the 6th attempt never happens."
**Behavior, not text.** Asserting that a script, skill, or config
contains an exact line proves only that the source is the source. Run
scripts against controlled inputs and assert outputs, side effects, or
exit codes. Documents that instruct agents are tested by the consuming
agent's behavior (superpowers:writing-skills); prose for humans earns no
test at all.
**Your code, not the framework.** Test the contract your code makes at
its boundaries — the route you register, the query you emit, the payload
you produce. Upstream mechanics are their maintainers' tests to write
(the classic: asserting your router invokes a registered handler — that
is the framework's test, not yours). When upstream behavior genuinely
surprised you, write one narrow characterization test naming the
assumption. The same boundary applies inside your code: constructors,
getters, constants, and trivial forwarding earn tests only when they
validate, normalize, default, derive, enforce, or cause side effects —
otherwise assert the first consumer-visible result that depends on them.
### Gate Function
```
BEFORE writing the test body:
Name the production change that would make this test fail.
Cannot name one → redesign around an observable behavior
"The source text changed" → run the artifact and assert its effects
Only intentional decisions → change detector; test the behavior
that depends on the decision
Confirm the expected value is derived without the code under test.
IF it reuses the code's logic or helpers:
Replace it with a literal or hand-checked fixture
```
## Principle 2: Exercise the Real Thing
**The mock earns no assertions.** A mock assertion passes when the mock
is present and fails when it is absent — it says nothing about the
component. Assert the real component's behavior; if the mock is what you
are checking, unmock it or delete the assertion.
```typescript
// ✅ Real behavior
expect(screen.getByRole('navigation')).toBeInTheDocument();
// ❌ Mock existence
expect(screen.getByTestId('sidebar-mock')).toBeInTheDocument();
```
**your human partner's correction:** "Are we testing the behavior of a
mock?"
**Mock at the right level.** Learn every side effect of the real method
before replacing it; mock the slow or external operation and keep what
the test depends on real. When unsure, run the test against the real
implementation first and observe what actually needs to happen.
```typescript
// ❌ The mock swallows the config write that duplicate detection reads
vi.mock('ToolCatalog', () => ({
discoverAndCacheTools: vi.fn().mockResolvedValue(undefined)
}));
// ✅ Mock only the slow server startup; the config write stays real
vi.mock('MCPServerManager');
```
**Make doubles specific.** When arguments, call counts, or ordering are
part of the contract, assert them — a fake that accepts anything verifies
nothing. Give each branch (success, error, malformed) its own fixture or
spy, so the wrong branch cannot satisfy the expectation.
**Mirror real data completely.** Mock the complete structure as it exists
in reality — all documented fields — not just the ones your test reads.
Partial mocks fail silently when downstream code reads an omitted field:
the test passes while integration breaks.
**Production classes carry production methods only.** Cleanup that only
tests need lives in test utilities, never as a `destroy()` on the
production class. Ask: is this method called only from tests? Does this
class own this resource's lifecycle? Wrong answers → test utility.
**Prefer real components over complex mocks.** When mock setup outgrows
the test logic, mocks miss methods the real components have, or tests
break when the mock changes, switch to an integration test with real
components. **your human partner's question:** "Do we need to be using a
mock here?"
### Gate Function
```
BEFORE adding a mock or test helper:
List the real method's side effects; keep the ones the test
depends on real — mock the slow/external level below them.
Mock responses mirror the complete real structure.
A method only tests call lives in test utilities, not production.
About to assert on the mock itself?
Unmock it or delete the assertion.
```
## Tests Ship With the Implementation
The TDD cycle — failing test, minimal implementation, refactor — is what
"complete" means. Ship the tests the behavior needs and only those:
trivial code and human prose earn none, and a test written to satisfy
process costs maintenance forever.
## The Mutation Check
Before finishing, mentally mutate the production code; at least one test
should fail for each realistic mutation:
- Wrong constant or argument
- Wrong branch handler
- Missing state change or side effect
- Empty or default return
- Missing validation for zero, empty, nil, unauthorized, or malformed input
A mutation nothing catches marks the behavior as unprotected — or the
test as tautological.
## Quick Reference
| When you... | Do |
|-------------|-----|
| Write any test | Name the break it catches — a bug, not a decision |
| Build an expected value | Derive it by hand; never with the code under test |
| Test a script or document | Run it / pressure-test its consumer; never grep its text |
| Reach for a dependency test | Test your boundary contract, not their documented mechanics |
| Want to assert on a mocked element | Test the real component, or unmock it |
| Are about to mock a method | Learn its side effects; mock the slow/external level |
| Build a mock response | Mirror the real structure completely |
| Need cleanup only tests use | Put it in test utilities |
| Watch mock setup balloon | Switch to an integration test with real components |
| Finish a test file | Run the mutation check |
## Warning Signs
- Setup and assertion share the same object, guaranteeing equality
- The test can fail only through a panic, crash, or missing selector
- The test fails on every intentional change, never on accidental breakage
- Expected values are hidden behind loops, builders, or helpers
- The test greps source text, or asserts a removed symbol stays removed
- The test would still matter if only the framework remained
- The test exists for coverage, checking no side effect or outcome
- An assertion checks a `*-mock` test ID, or fails if you remove the mock
- A method is called only from test files
- Mock setup is more than half the test, or you can't explain why the mock is needed
- Mocking "just to be safe"
@@ -156,47 +156,12 @@ Ready to implement <feature-name>
| Tests fail during baseline | Report failures + ask | | Tests fail during baseline | Report failures + ask |
| No package.json/Cargo.toml | Skip dependency install | | No package.json/Cargo.toml | Skip dependency install |
## Common Mistakes ## Common Rationalizations
### Fighting the harness | Excuse | Reality |
|--------|---------|
- **Problem:** Using `git worktree add` when the platform already provides isolation | "I'm obviously not in a worktree — no need to check" | Run Step 0. Harness-created isolation and submodules both fool eyeballing; the detection commands settle it. |
- **Fix:** Step 0 detects existing isolation. Step 1a defers to native tools. | "`git worktree add` is quicker than hunting for a native tool" | A native tool (e.g. `EnterWorktree`) owns placement, branching, and cleanup. Bypassing it is the #1 mistake — it creates phantom state your harness can't see or manage. |
| "The worktree directory is surely ignored already" | Run `git check-ignore`. An unignored worktree directory commits the whole tree into the repo. |
### Skipping detection | "Any directory name works" | Explicit instructions beat an existing project-local directory, which beats the `.worktrees/` default. |
| "The workspace is fresh — baseline tests can wait" | A dirty baseline makes every later failure ambiguous. Run the tests now; proceeding past failures is your human partner's call. |
- **Problem:** Creating a nested worktree inside an existing one
- **Fix:** Always run Step 0 before creating anything
### Skipping ignore verification
- **Problem:** Worktree contents get tracked, pollute git status
- **Fix:** Always use `git check-ignore` before creating project-local worktree
### Assuming directory location
- **Problem:** Creates inconsistency, violates project conventions
- **Fix:** Follow priority: explicit instructions > existing project-local directory > default
### Proceeding with failing tests
- **Problem:** Can't distinguish new bugs from pre-existing issues
- **Fix:** Report failures, get explicit permission to proceed
## Red Flags
**Never:**
- Create a worktree when Step 0 detects existing isolation
- Use `git worktree add` when you have a native worktree tool (e.g., `EnterWorktree`). This is the #1 mistake — if you have it, use it.
- Skip Step 1a by jumping straight to Step 1b's git commands
- Create worktree without verifying it's ignored (project-local)
- Skip baseline test verification
- Proceed with failing tests without asking
**Always:**
- Run Step 0 detection first
- Prefer native tools over git fallback
- Follow directory priority: explicit instructions > existing project-local directory > default
- Verify directory is ignored for project-local
- Auto-detect and run project setup
- Verify clean test baseline
@@ -4,7 +4,7 @@ Skills speak in actions ("dispatch a subagent", "create a todo", "read a file").
| Action skills request | Antigravity CLI equivalent | | Action skills request | Antigravity CLI equivalent |
|----------------------|----------------------| |----------------------|----------------------|
| Dispatch a subagent (`Subagent (general-purpose):` template) | `invoke_subagent` with a built-in `TypeName``self` for full-capability work, `research` for read-only (see [Subagent support](#subagent-support)) | | Dispatch a subagent (`Subagent (general-purpose):` template) | `invoke_subagent` with a built-in `TypeName``self` for full-capability work, `research` for read-only |
| Task tracking ("create a todo", "mark complete") | a **task artifact**`write_to_file` with `IsArtifact: true` and `ArtifactType: "task"` (see [Task tracking](#task-tracking)). **Not** `manage_task`, which manages background processes. | | Task tracking ("create a todo", "mark complete") | a **task artifact**`write_to_file` with `IsArtifact: true` and `ArtifactType: "task"` (see [Task tracking](#task-tracking)). **Not** `manage_task`, which manages background processes. |
## Task tracking ## Task tracking
@@ -7,7 +7,7 @@ Add to your Codex config (`~/.codex/config.toml`):
multi_agent = true multi_agent = true
``` ```
This enables `spawn_agent`, `wait_agent`, and `close_agent` for skills like `dispatching-parallel-agents` and `subagent-driven-development`. When using subagent-driven-development, you should always close implementer and reviewer subagents when they have finished all their work. This enables `spawn_agent`, `wait_agent`, and `close_agent` for skills like `dispatching-parallel-agents` and `subagent-driven-development`. When using subagent-driven-development, close reviewer subagents when their review returns. Keep each implementer subagent open until its task's review passes — the fix loop resumes the implementer — then close it. If your harness cannot send another message to a spawned agent, dispatch each fix round as a fresh implementer carrying the brief, the report file, and the findings.
## Environment Detection ## Environment Detection
@@ -0,0 +1,63 @@
# Gemini CLI Tool Mapping
Skills speak in actions ("dispatch a subagent", "create a todo", "read a file"). On Gemini CLI these resolve to the tools below.
| Action skills request | Gemini CLI equivalent |
|----------------------|----------------------|
| Read a file | `read_file` |
| Read multiple files at once | `read_many_files` |
| Create a new file | `write_file` |
| Edit a file | `replace` |
| Run a shell command | `run_shell_command` |
| Search file contents | `grep_search` |
| Find files by name | `glob` |
| List files and subdirectories | `list_directory` |
| Fetch a URL | `web_fetch` |
| Search the web | `google_web_search` |
| Invoke a skill | `activate_skill` |
| Dispatch a subagent (`Subagent (general-purpose):` template) | `invoke_agent` with `agent_name: "generalist"` (invocable via `@generalist` chat syntax — see [Subagent support](#subagent-support)) |
| Multiple parallel dispatches | Multiple `invoke_agent` calls in the same response |
| Task tracking ("create a todo", "mark complete") | `write_todos` (statuses: pending, in_progress, completed, cancelled, blocked) |
## Instructions file
When a skill mentions "your instructions file", on Gemini CLI this is **`GEMINI.md`**. Gemini CLI loads `GEMINI.md` hierarchically: global at `~/.gemini/GEMINI.md`, project-level files in workspace directories and their ancestors, and sub-directory `GEMINI.md` files when a tool accesses files in those directories.
## Personal skills directory
User-level skills live at **`~/.gemini/skills/`**, with **`~/.agents/skills/`** as a cross-runtime alias (shared with Codex and Copilot CLI). When both directories exist at the same scope, `.agents/skills/` takes precedence. Each skill is a subdirectory containing a `SKILL.md` (with `name` and `description` frontmatter).
## Subagent support
Gemini CLI dispatches subagents through the `invoke_agent` tool, which takes `agent_name` and `prompt` parameters. The same dispatch is also surfaced as a chat-syntax shortcut: typing `@generalist <prompt>` is equivalent to calling `invoke_agent` with `agent_name: "generalist"`. Built-in agent names include `generalist`, `cli_help`, `codebase_investigator`, and (with browser tooling enabled) `browser_agent`.
Skills dispatch with `Subagent (general-purpose):` and either reference a prompt-template file (e.g., `superpowers:subagent-driven-development`'s `./implementer-prompt.md`) or supply an inline prompt. On Gemini CLI:
| Skill dispatch form | Gemini CLI equivalent |
|---------------------|----------------------|
| References a `*-prompt.md` template (implementer, task-reviewer, code-reviewer, etc.) | Fill the template, then `invoke_agent` with `agent_name: "generalist"` and the filled prompt |
| References `superpowers:requesting-code-review`'s `./code-reviewer.md` | `invoke_agent` with `agent_name: "generalist"` and the filled review template |
| Inline prompt (no template referenced) | `invoke_agent` with `agent_name: "generalist"` and your inline prompt |
### Prompt filling
Skills provide prompt templates with placeholders like `{WHAT_WAS_IMPLEMENTED}` or `[FULL TEXT of task]`. Fill all placeholders before passing the complete prompt to `invoke_agent`. The prompt template itself contains the agent's role, review criteria, and expected output format — the subagent will follow it.
### Parallel dispatch
Gemini CLI supports parallel subagent dispatch. Issue multiple `invoke_agent` calls in the same response (or multiple `@generalist` invocations in one prompt) to run independent subagent work in parallel. Keep dependent tasks sequential, but do not serialize independent subagent tasks just to preserve a simpler history.
## Additional Gemini CLI tools
These tools are unique to Gemini CLI:
| Tool | Purpose |
|------|---------|
| `save_memory` (legacy) | Persist facts across sessions when `experimental.memoryV2 = false` |
| `get_internal_docs` | Look up Gemini CLI's bundled documentation |
| `ask_user` | Pose structured questions to the user (text / single-select / multi-select) |
| `enter_plan_mode` / `exit_plan_mode` | Switch into and out of read-only plan mode |
| `update_topic` | Update the current conversation's topic / strategic-intent metadata |
| `complete_task` | Signal that a Gemini subagent has completed and return its result to the parent agent |
| `tracker_create_task`, `tracker_update_task`, `tracker_get_task`, `tracker_list_tasks`, `tracker_add_dependency`, `tracker_visualize` | Rich task tracker with dependency and visualization support |
| `read_mcp_resource`, `list_mcp_resources` | MCP resource access |
@@ -7,8 +7,6 @@ description: "在声称任务完成、修复成功或测试通过前使用,确
## Overview ## Overview
Claiming work is complete without verification is dishonesty, not efficiency.
**Core principle:** Evidence before claims, always. **Core principle:** Evidence before claims, always.
**Violating the letter of this rule is violating the spirit of this rule.** **Violating the letter of this rule is violating the spirit of this rule.**
@@ -105,15 +103,6 @@ Skip any step = lying, not verifying
❌ Trust agent report ❌ Trust agent report
``` ```
## Why This Matters
From 24 failure memories:
- your human partner said "I don't believe you" - trust broken
- Undefined functions shipped - would crash
- Missing requirements shipped - incomplete features
- Time wasted on false completion → redirect → rework
- Violates: "Honesty is a core value. If you lie, you'll be replaced."
## When To Apply ## When To Apply
**ALWAYS before:** **ALWAYS before:**
@@ -129,11 +118,3 @@ From 24 failure memories:
- Paraphrases and synonyms - Paraphrases and synonyms
- Implications of success - Implications of success
- ANY communication suggesting completion/correctness - ANY communication suggesting completion/correctness
## The Bottom Line
**No shortcuts for verification.**
Run the command. Read the output. THEN claim the result.
This is non-negotiable.
@@ -135,12 +135,6 @@ Every step must contain the actual content an engineer needs. These are **plan f
- Steps that describe what to do without showing how (code blocks required for code steps) - Steps that describe what to do without showing how (code blocks required for code steps)
- References to types, functions, or methods not defined in any task - References to types, functions, or methods not defined in any task
## Remember
- Exact file paths always
- Complete code in every step — if a step changes code, show the code
- Exact commands with expected output
- DRY, YAGNI, TDD, frequent commits
## Self-Review ## Self-Review
After writing the complete plan, look at the spec with fresh eyes and check the plan against it. This is a checklist you run yourself — not a subagent dispatch. After writing the complete plan, look at the spec with fresh eyes and check the plan against it. This is a checklist you run yourself — not a subagent dispatch.
@@ -9,7 +9,7 @@ description: "创建、修改或验证 Agent skill 时使用,确保 skill 可
**Writing skills IS Test-Driven Development applied to process documentation.** **Writing skills IS Test-Driven Development applied to process documentation.**
**Personal skills live in your runtime's skills directory** **Personal skills live in your runtime's skills directory** (`~/.claude/skills/` on Claude Code) — see [codex-tools.md](../using-superpowers/references/codex-tools.md) or [gemini-tools.md](../using-superpowers/references/gemini-tools.md) for the path on those runtimes. Codex, Copilot CLI, and Gemini CLI all also recognize `~/.agents/skills/` as a cross-runtime alias.
You write test cases (pressure scenarios with subagents), watch them fail (baseline behavior), write the skill (documentation), watch tests pass (agents comply), and refactor (close loopholes). You write test cases (pressure scenarios with subagents), watch them fail (baseline behavior), write the skill (documentation), watch tests pass (agents comply), and refactor (close loopholes).
@@ -677,13 +677,3 @@ How future agents find your skill:
6. **Loads example** (only when implementing) 6. **Loads example** (only when implementing)
**Optimize for this flow** - put searchable terms early and often. **Optimize for this flow** - put searchable terms early and often.
## The Bottom Line
**Creating skills IS TDD for process documentation.**
Same Iron Law: No skill without failing test first.
Same cycle: RED (baseline) → GREEN (write skill) → REFACTOR (close loopholes).
Same benefits: Better quality, fewer surprises, bulletproof results.
If you follow TDD for code, follow it for skills. It's the same discipline applied to documentation.
@@ -2,8 +2,8 @@
"sourceId": "taste-skill", "sourceId": "taste-skill",
"repo": "https://github.com/Leonxlnx/taste-skill.git", "repo": "https://github.com/Leonxlnx/taste-skill.git",
"ref": "main", "ref": "main",
"commit": "1bffae64edbb6b2023d1e78402cd088e9c9f6511", "commit": "e988add20dab0fa97d7a76781c48961c8184288e",
"adapter": "skill-collection", "adapter": "skill-collection",
"sourcePath": "skills", "sourcePath": "skills",
"syncedAt": "2026-07-23T16:00:00Z" "syncedAt": "2026-07-24T06:18:20Z"
} }