diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json index f85d3464b..87d873f4b 100644 --- a/.claude-plugin/marketplace.json +++ b/.claude-plugin/marketplace.json @@ -9,7 +9,7 @@ { "name": "superpowers", "description": "Core skills library for Claude Code: TDD, debugging, collaboration patterns, and proven techniques", - "version": "6.3.0", + "version": "6.4.2", "source": "./", "author": { "name": "Jesse Vincent", diff --git a/.claude-plugin/plugin.json b/.claude-plugin/plugin.json index 7e0c66154..aa2a09cea 100644 --- a/.claude-plugin/plugin.json +++ b/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "superpowers", "description": "Core skills library for Claude Code: TDD, debugging, collaboration patterns, and proven techniques", - "version": "6.3.0", + "version": "6.4.2", "author": { "name": "Jesse Vincent", "email": "jesse@fsck.com" diff --git a/.codex-plugin/plugin.json b/.codex-plugin/plugin.json index 123793e54..f14735290 100644 --- a/.codex-plugin/plugin.json +++ b/.codex-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "superpowers", - "version": "6.3.0", + "version": "6.4.2", "description": "An agentic skills framework & software development methodology that works: planning, TDD, debugging, and collaboration workflows.", "author": { "name": "Jesse Vincent", diff --git a/.cursor-plugin/plugin.json b/.cursor-plugin/plugin.json index bb2bdbcd1..8a8475cca 100644 --- a/.cursor-plugin/plugin.json +++ b/.cursor-plugin/plugin.json @@ -2,7 +2,7 @@ "name": "superpowers", "displayName": "Superpowers", "description": "Core skills library: TDD, debugging, collaboration patterns, and proven techniques", - "version": "6.3.0", + "version": "6.4.2", "author": { "name": "Jesse Vincent", "email": "jesse@fsck.com" diff --git a/.devin-plugin/plugin.json b/.devin-plugin/plugin.json index 8b68f28e4..71d0f0b8a 100644 --- a/.devin-plugin/plugin.json +++ b/.devin-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "superpowers", - "version": "6.3.0", + "version": "6.4.2", "description": "An agentic skills framework & software development methodology that works: planning, TDD, debugging, and collaboration workflows.", "author": { "name": "Jesse Vincent", diff --git a/.hermes-plugin/plugin.yaml b/.hermes-plugin/plugin.yaml index 3a0ecfb21..9de7655ae 100644 --- a/.hermes-plugin/plugin.yaml +++ b/.hermes-plugin/plugin.yaml @@ -1,5 +1,5 @@ name: superpowers -version: 6.3.0 +version: 6.4.2 description: Superpowers skills and workflow bootstrap for Hermes Agent author: obra provides_hooks: diff --git a/.kimi-plugin/plugin.json b/.kimi-plugin/plugin.json index dcb9a7a9b..1effbcfaf 100644 --- a/.kimi-plugin/plugin.json +++ b/.kimi-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "superpowers", - "version": "6.3.0", + "version": "6.4.2", "description": "An agentic skills framework and software development methodology.", "author": { "name": "Jesse Vincent", diff --git a/.muse-plugin/marketplace.json b/.muse-plugin/marketplace.json new file mode 100644 index 000000000..c3d729538 --- /dev/null +++ b/.muse-plugin/marketplace.json @@ -0,0 +1,20 @@ +{ + "name": "superpowers-dev", + "description": "Development marketplace for Superpowers core skills library", + "owner": { + "name": "Jesse Vincent", + "email": "jesse@fsck.com" + }, + "plugins": [ + { + "name": "superpowers", + "description": "Core skills library for Muse: TDD, debugging, collaboration patterns, and proven techniques", + "version": "6.4.2", + "source": "./", + "author": { + "name": "Jesse Vincent", + "email": "jesse@fsck.com" + } + } + ] +} diff --git a/.muse-plugin/plugin.json b/.muse-plugin/plugin.json new file mode 100644 index 000000000..3588fd506 --- /dev/null +++ b/.muse-plugin/plugin.json @@ -0,0 +1,89 @@ +{ + "schemaVersion": 1, + "name": "superpowers", + "displayName": "Superpowers", + "version": "6.4.2", + "description": "Core skills library for Muse: TDD, debugging, collaboration patterns, and proven techniques", + "compat": { + "source": "native", + "manifestDir": ".muse-plugin" + }, + "capabilities": { + "skills": [ + { + "id": "brainstorming", + "path": "skills/brainstorming/SKILL.md" + }, + { + "id": "diagnosing-superpowers", + "path": "skills/diagnosing-superpowers/SKILL.md" + }, + { + "id": "dispatching-parallel-agents", + "path": "skills/dispatching-parallel-agents/SKILL.md" + }, + { + "id": "executing-plans", + "path": "skills/executing-plans/SKILL.md" + }, + { + "id": "finishing-a-development-branch", + "path": "skills/finishing-a-development-branch/SKILL.md" + }, + { + "id": "receiving-code-review", + "path": "skills/receiving-code-review/SKILL.md" + }, + { + "id": "requesting-code-review", + "path": "skills/requesting-code-review/SKILL.md" + }, + { + "id": "subagent-driven-development", + "path": "skills/subagent-driven-development/SKILL.md" + }, + { + "id": "systematic-debugging", + "path": "skills/systematic-debugging/SKILL.md" + }, + { + "id": "test-driven-development", + "path": "skills/test-driven-development/SKILL.md" + }, + { + "id": "using-git-worktrees", + "path": "skills/using-git-worktrees/SKILL.md" + }, + { + "id": "using-superpowers", + "path": "skills/using-superpowers/SKILL.md" + }, + { + "id": "verification-before-completion", + "path": "skills/verification-before-completion/SKILL.md" + }, + { + "id": "writing-plans", + "path": "skills/writing-plans/SKILL.md" + }, + { + "id": "writing-skills", + "path": "skills/writing-skills/SKILL.md" + } + ], + "commands": [], + "hooks": [ + { + "id": "session-start", + "event": "SessionStart", + "command": [ + "sh", + "hooks/session-start" + ], + "timeoutMs": 5000 + } + ], + "mcpServers": [], + "reminders": [] + } +} diff --git a/.opencode/INSTALL.md b/.opencode/INSTALL.md index 080f043f8..dca404b08 100644 --- a/.opencode/INSTALL.md +++ b/.opencode/INSTALL.md @@ -6,7 +6,11 @@ ## Installation -Add superpowers to the `plugin` array in your `opencode.json` (global or project-level): +OpenCode V2 requires version 2.0.4 or later. + +### OpenCode V1 + +Use the existing V1 plugin configuration: ```json { @@ -14,7 +18,22 @@ Add superpowers to the `plugin` array in your `opencode.json` (global or project } ``` -Restart OpenCode. The plugin installs through OpenCode's plugin manager and +### OpenCode V2 (2.0.4 or later) + +Use the V2 plugin configuration: + +```json +{ + "plugins": ["superpowers@git+https://github.com/obra/superpowers.git"] +} +``` + +For a local V2 installation, configure the repository directory containing +`index.js`. OpenCode 2.0.4 and 2.0.7 reject a configured direct JavaScript-file +path. Discovered plugin symlinks remain supported. + +Restart OpenCode. V2 uses the `opencode` command; `opencode2` may be available +as an alias. The plugin installs through OpenCode's plugin manager and registers all skills. Verify by asking: "Tell me about your superpowers" @@ -55,19 +74,25 @@ and Bun versions pin that resolved git dependency in a lockfile or cache, so a restart may not pick up the newest Superpowers commit. If updates do not appear, clear OpenCode's package cache or reinstall the plugin. -To pin a specific version: +To pin a specific version, add a tag or commit to the spec (same form for the +V1 `plugin` key and the V2 `plugins` key): ```json { - "plugin": ["superpowers@git+https://github.com/obra/superpowers.git#v5.0.3"] + "plugin": ["superpowers@git+https://github.com/obra/superpowers.git#v6.4.2"] } ``` +On V2, pin `v6.4.1` or later; `v6.3.0` and earlier releases load only on V1. + ## Troubleshooting ### Plugin not loading -1. Check logs: `opencode run --print-logs "hello" 2>&1 | grep -i superpowers` +1. Check logs. V1: `opencode run --print-logs "hello" 2>&1 | grep -i superpowers`. + V2 loads plugins in the background server, so add `--standalone`: + `opencode run --standalone --print-logs "hello" 2>&1 | grep -i superpowers`, + or inspect `~/.local/share/opencode/log/opencode.log` filtering for `role=server`. 2. Verify the plugin line in your `opencode.json` 3. Make sure you're running a recent version of OpenCode @@ -83,11 +108,23 @@ package: npm install superpowers@git+https://github.com/obra/superpowers.git --prefix "$HOME\.config\opencode" ``` -Then use the installed package path in `opencode.json`: +Then use the absolute path of the installed package in `opencode.json` for your +OpenCode version. OpenCode does not expand `~`; a `~/...` entry is treated as a +package name, not a local directory. + +**V1:** ```json { - "plugin": ["~/.config/opencode/node_modules/superpowers"] + "plugin": ["C:\\Users\\\\.config\\opencode\\node_modules\\superpowers"] +} +``` + +**V2 (2.0.4 or later):** + +```json +{ + "plugins": ["C:\\Users\\\\.config\\opencode\\node_modules\\superpowers"] } ``` @@ -98,7 +135,9 @@ Then use the installed package path in `opencode.json`: ### Tool mapping -Skills speak in actions ("create a todo", "dispatch a subagent", "read a file"). On OpenCode these resolve to: +Skills speak in actions ("create a todo", "dispatch a subagent", "read a file"). The plugin injects a flavor-specific mapping — check your OpenCode version: + +**V1 (`opencode` 1.x):** - "Create a todo" / "mark complete in todo list" → `todowrite` - `Subagent (general-purpose):` template → `task` tool with `subagent_type: "general"` (or `"explore"` for codebase exploration) @@ -109,6 +148,18 @@ Skills speak in actions ("create a todo", "dispatch a subagent", "read a file"). - "Search file contents" / "find files by name" → `grep`, `glob` - "Fetch a URL" → `webfetch` +**V2 (`opencode` 2.0.4 or later; `opencode2` may be available as an alias):** + +- "Create a todo" → V2 has no todo tool; track the plan in a markdown file instead +- `Subagent (general-purpose):` template → `subagent` tool with `agent: "general"` (or `"explore"`); pass `sessionID` to continue a previous subagent +- "Invoke a skill" → OpenCode's native `skill` tool +- "Read a file" → `read` +- "Create, edit, or delete files" → use `patch` with `patchText` when available; otherwise use `write` to create or overwrite files, `edit` for targeted changes, and `shell` for deletion +- "Run a shell command" → `shell` (`command`, `workdir`, `timeout`, `background`) +- "Search file contents" / "find files by name" → `grep`, `glob` +- "Fetch a URL" → `webfetch` +- "Search the web" → `websearch` + ## Getting Help - Report issues: https://github.com/obra/superpowers/issues diff --git a/.opencode/plugins/superpowers.js b/.opencode/plugins/superpowers.js index 423e5ed51..9eff8111b 100644 --- a/.opencode/plugins/superpowers.js +++ b/.opencode/plugins/superpowers.js @@ -1,79 +1,80 @@ /** * Superpowers plugin for OpenCode.ai * - * Injects superpowers bootstrap context via message transform. - * Auto-registers skills directory via config hook (no symlinks needed). + * Dual-compatible with OpenCode V1 and V2. + * + * V1 (opencode): loaded via named export SuperpowersPlugin — provides config + * hook for skills registration and experimental.chat.messages.transform for + * bootstrap injection. + * + * V2 (opencode2): loaded via default export { id, setup } by PluginSupervisor. + * setup() registers skills natively via ctx.skill.transform(), and injects + * bootstrap context via ctx.session.hook("context"). + * + * No external dependencies — pure JavaScript works in both V1 and V2 without + * installing @opencode-ai/plugin or effect. */ import path from 'path'; import fs from 'fs'; -import os from 'os'; import { fileURLToPath } from 'url'; const __dirname = path.dirname(fileURLToPath(import.meta.url)); -// Simple frontmatter extraction (avoid dependency on skills-core for bootstrap) +// Skills directory shared by V1 (config hook) and V2 (setup/ctx.skill.transform) +const superpowersSkillsDir = path.resolve(__dirname, '../../skills'); + +// Simple frontmatter extraction (avoid dependency on skills-core for +// bootstrap). Handles plain `key: value` lines, quoted values (including +// quotes that close on an indented continuation line), YAML block scalar +// markers (`>`, `|`) with indented continuation lines, and CRLF line +// endings. Not a full YAML parser — nested maps flatten into their parent +// key's value, which is fine for the name/description fields consumed here. const extractAndStripFrontmatter = (content) => { - const match = content.match(/^---\n([\s\S]*?)\n---\n([\s\S]*)$/); + const match = content.match(/^---\r?\n([\s\S]*?)\r?\n---\r?\n?([\s\S]*)$/); if (!match) return { frontmatter: {}, content }; const frontmatterStr = match[1]; const body = match[2]; const frontmatter = {}; + let lastKey = null; - for (const line of frontmatterStr.split('\n')) { + for (const rawLine of frontmatterStr.split('\n')) { + const line = rawLine.replace(/\r$/, ''); const colonIdx = line.indexOf(':'); - if (colonIdx > 0) { + if (colonIdx > 0 && !/^\s/.test(line)) { const key = line.slice(0, colonIdx).trim(); - const value = line.slice(colonIdx + 1).trim().replace(/^["']|["']$/g, ''); - frontmatter[key] = value; + const value = line.slice(colonIdx + 1).trim(); + // Block scalar markers (>, |, optionally with +/- chomping) carry no + // value themselves; the indented lines that follow do. + frontmatter[key] = /^(>[+-]?|\|[+-]?)$/.test(value) ? '' : value; + lastKey = key; + } else if (lastKey !== null && line.trim() !== '') { + // Continuation of a multi-line value: append rather than drop so long + // descriptions survive parsing. Newlines collapse to spaces — good + // enough for the single-line name/description fields consumed here. + frontmatter[lastKey] = `${frontmatter[lastKey]} ${line.trim()}`.trim(); } } + // A quoted value may close on a continuation line, so unquote only once + // the value is fully assembled: strip exactly one matching surrounding + // pair and leave unbalanced quotes alone. + for (const key of Object.keys(frontmatter)) { + frontmatter[key] = frontmatter[key].replace(/^(["'])([\s\S]*)\1$/, '$2'); + } + return { frontmatter, content: body }; }; -// Normalize a path: trim whitespace, expand ~, resolve to absolute -const normalizePath = (p, homeDir) => { - if (!p || typeof p !== 'string') return null; - let normalized = p.trim(); - if (!normalized) return null; - if (normalized.startsWith('~/')) { - normalized = path.join(homeDir, normalized.slice(2)); - } else if (normalized === '~') { - normalized = homeDir; - } - return path.resolve(normalized); -}; +// Tool mapping injected into the bootstrap, differentiated by host flavor. +// V1 (OpenCode 1.18.x) and V2 (OpenCode 2.0.4/2.0.7) expose different built-in +// tools, so each flavor's injection path picks its own constant below. +// Exported for tests (tests/opencode/test-bootstrap-caching.mjs). -// Module-level cache for bootstrap content. -// The SKILL.md file does not change during a session, so reading + parsing it -// once eliminates redundant fs.existsSync + fs.readFileSync + regex work on -// every agent step. See #1202 for the full analysis. -let _bootstrapCache = undefined; // undefined = not yet loaded, null = file missing - -export const SuperpowersPlugin = async ({ client, directory }) => { - const homeDir = os.homedir(); - const superpowersSkillsDir = path.resolve(__dirname, '../../skills'); - const envConfigDir = normalizePath(process.env.OPENCODE_CONFIG_DIR, homeDir); - const configDir = envConfigDir || path.join(homeDir, '.config/opencode'); - - // Helper to generate bootstrap content (cached after first call) - const getBootstrapContent = () => { - // Return cached result on subsequent calls - if (_bootstrapCache !== undefined) return _bootstrapCache; - - // Try to load using-superpowers skill - const skillPath = path.join(superpowersSkillsDir, 'using-superpowers', 'SKILL.md'); - if (!fs.existsSync(skillPath)) { - _bootstrapCache = null; - return null; - } - - const fullContent = fs.readFileSync(skillPath, 'utf8'); - const { content } = extractAndStripFrontmatter(fullContent); - - const toolMapping = `**Tool Mapping for OpenCode:** +// V1 built-ins: todowrite, task (subagent_type), skill, read, apply_patch, +// bash, grep, glob, webfetch. +export const V1_MAPPING = `**Tool Mapping for OpenCode:** When skills request actions, substitute OpenCode equivalents: - Create or update todos → \`todowrite\` - \`Subagent (general-purpose):\` → \`task\` with \`subagent_type: "general"\` @@ -86,7 +87,47 @@ When skills request actions, substitute OpenCode equivalents: Use OpenCode's native \`skill\` tool to list and load skills.`; - _bootstrapCache = ` +// V2 built-ins: no todo tool at all; task → subagent (agent name in 'agent', +// continuation via sessionID); apply_patch → patch (patchText, same patch +// format); bash → shell. read, write, edit, grep, glob, webfetch, websearch, +// and skill all exist under those names (verified against the 2.0.4 and 2.0.7 +// host contracts). +export const V2_MAPPING = `**Tool Mapping for OpenCode:** +When skills request actions, substitute OpenCode equivalents: +- Create or update todos → OpenCode v2 has no todo tool; track the plan in a markdown file (or the harness's plan facility) instead +- \`Subagent (general-purpose):\` → \`subagent\` with \`agent: "general"\` (give it \`description\` and \`prompt\`, optionally \`background\`; pass \`sessionID\` to continue a previous subagent) +- Invoke a skill → OpenCode's native \`skill\` tool +- Read files → \`read\` +- Create, edit, or delete files → use \`patch\` with \`patchText\` when available; otherwise use \`write\` to create or overwrite files, \`edit\` for targeted changes, and \`shell\` for deletion +- Run shell commands → \`shell\` (\`command\`, \`workdir\`, \`timeout\`, \`background\`) +- Search files → \`grep\`, \`glob\` +- Fetch a URL → \`webfetch\` +- Search the web → \`websearch\` + +Use OpenCode's native \`skill\` tool to list and load skills.`; + +// Module-level cache for bootstrap content, keyed by tool mapping (host +// flavor). The SKILL.md file does not change during a session, so reading + +// parsing it once eliminates redundant fs.existsSync + fs.readFileSync + +// regex work on every agent step. See #1202 for the full analysis. +const _bootstrapCache = new Map(); // mapping -> bootstrap (null = file missing) + +// Helper to generate bootstrap content (cached after first call per mapping) +const getBootstrapContent = (toolMapping) => { + // Return cached result on subsequent calls + if (_bootstrapCache.has(toolMapping)) return _bootstrapCache.get(toolMapping); + + // Try to load using-superpowers skill + const skillPath = path.join(superpowersSkillsDir, 'using-superpowers', 'SKILL.md'); + if (!fs.existsSync(skillPath)) { + _bootstrapCache.set(toolMapping, null); + return null; + } + + const fullContent = fs.readFileSync(skillPath, 'utf8'); + const { content } = extractAndStripFrontmatter(fullContent); + + _bootstrapCache.set(toolMapping, ` You have superpowers. **IMPORTANT: The using-superpowers skill content is included below. It is ALREADY LOADED - you are currently following it. Do NOT use the skill tool to load "using-superpowers" again - that would be redundant.** @@ -94,17 +135,96 @@ You have superpowers. ${content} ${toolMapping} -`; +`); - return _bootstrapCache; - }; + return _bootstrapCache.get(toolMapping); +}; +// --- Task-subagent (child session) detection -------------------------------- +// +// #2160: the bootstrap drives controller workflows (brainstorming, planning, +// approval cycles). Injecting it into task subagent sessions makes workers +// restart design/approval cycles for work the parent already authorised; the +// note inside the bootstrap relies on model compliance, which +// is not reliable. Detect child sessions structurally instead: a parentID on +// the session is the child signal on both flavors (task sessions are created +// with one; top-level sessions simply lack the field), so when the session +// carrying the message has a parentID we skip bootstrap injection. Skills +// stay registered for every session — workers keep explicit access to +// execution skills. + +// sessionID -> is-child decision. parentID never changes for a session, so +// the result is cached until eviction and the injection hook (which fires on +// every agent step) pays only one client roundtrip per session. The V2 +// service process is long-lived and sessions accumulate over weeks, so the +// cache is bounded: when full, drop the oldest quarter (Map iterates keys in +// insertion order). An evicted session merely pays one extra lookup if seen +// again. +const CHILD_SESSION_CACHE_MAX = 512; +const _childSessionCache = new Map(); + +const _cacheChildSession = (sessionID, isChild) => { + if (_childSessionCache.size >= CHILD_SESSION_CACHE_MAX) { + let toDrop = Math.ceil(CHILD_SESSION_CACHE_MAX / 4); + for (const key of _childSessionCache.keys()) { + if (toDrop-- <= 0) break; + _childSessionCache.delete(key); + } + } + _childSessionCache.set(sessionID, isChild); +}; + +const isChildSession = async (fetchSession, sessionID) => { + if (!sessionID) return false; // unknown session: keep current behavior + if (_childSessionCache.has(sessionID)) return _childSessionCache.get(sessionID); + + let isChild = false; + try { + const result = await fetchSession(sessionID); + // V1 returns a successful SDK envelope while V2 returns a direct session + // record. Validate both shapes before classifying or caching the result; + // resolved SDK errors must follow the same fail-open path as rejections. + if (!result || typeof result !== 'object' || Array.isArray(result)) { + throw new Error('Session lookup returned no usable record'); + } + if (result.error != null || result.response?.ok === false) { + throw new Error('Session lookup was unsuccessful'); + } + const session = 'data' in result ? result.data : result; + if (!session || typeof session !== 'object' || Array.isArray(session) || session.id !== sessionID) { + throw new Error('Session lookup returned an invalid session identity'); + } + if (session.parentID !== undefined && + (typeof session.parentID !== 'string' || session.parentID.length === 0)) { + throw new Error('Session lookup returned an invalid parent identity'); + } + isChild = session.parentID !== undefined; + } catch (err) { + // Fail open: on lookup errors keep injecting (previous behavior) and do + // not cache, so a transient failure can recover on the next step. + console.error('[superpowers] session lookup failed, treating session as top-level:', err); + return false; + } + _cacheChildSession(sessionID, isChild); + return isChild; +}; + +/** + * V1 Plugin Function (named export + default.server) + * + * Used by V1 (OpenCode 1.x): discovered via named export scanning. + * Provides: config hook (V1 skills registration) + bootstrap injection + * (experimental.chat.messages.transform). + */ +export const SuperpowersPlugin = async ({ client, directory }) => { return { // Inject skills path into live config so OpenCode discovers superpowers skills // without requiring manual symlinks or config file edits. - // This works because Config.get() returns a cached singleton — modifications - // here are visible when skills are lazily discovered later. config: async (config) => { + // V2: skills is a flat array — skip, setup() handles V2 skill registration + if (Array.isArray(config.skills)) return; + + // V1: skills is { paths: [...] } config.skills = config.skills || {}; config.skills.paths = config.skills.paths || []; if (!config.skills.paths.includes(superpowersSkillsDir)) { @@ -112,7 +232,7 @@ ${toolMapping} } }, - // Inject bootstrap into the first user message of each session. + // Inject bootstrap into the first user message of each top-level session. // Using a user message instead of a system message avoids: // 1. Token bloat from system messages repeated every turn (#750) // 2. Multiple system messages breaking Qwen and other models (#894) @@ -122,18 +242,142 @@ ${toolMapping} // arrays may need injection again, so getBootstrapContent() must not do // repeated disk work. 'experimental.chat.messages.transform': async (_input, output) => { - const bootstrap = getBootstrapContent(); + const bootstrap = getBootstrapContent(V1_MAPPING); if (!bootstrap || !output.messages.length) return; const firstUser = output.messages.find(m => m.info.role === 'user'); if (!firstUser || !firstUser.parts.length) return; // Guard: skip if first user message already contains bootstrap. - // This prevents double injection when OpenCode passes an already - // transformed in-memory message array through the hook again. if (firstUser.parts.some(p => p.type === 'text' && p.text.includes('EXTREMELY_IMPORTANT'))) return; + // #2160: never restart the controller workflow inside task subagent + // (child) sessions. V1 passes no input to this hook (verified in the + // 1.18.x bundle: trigger(..., {}, {messages})), so take the sessionID + // from the message record itself. + if (client && await isChildSession( + (id) => client.session.get({ path: { id } }), + firstUser.info.sessionID, + )) return; + const ref = firstUser.parts[0]; firstUser.parts.unshift({ ...ref, type: 'text', text: bootstrap }); } }; }; + +/** + * V2 Setup Function (default.setup) + * + * Called by V2 PluginSupervisor (packages/core/src/plugin/supervisor.ts). + * Performs two things: + * + * 1. Registers every skills//SKILL.md as a native Skill.Info object + * via ctx.skill.transform((draft) => draft.add(info)). + * V2 removed the old draft.source() directory registration; the draft API + * is now { list, add, update, remove } where add() decodes plain objects + * against the host's Skill.Info schema (OpenCode 2.0.4 contract): + * { id, name, description?, autoinvoke?, path, content }. The file field + * is `path` — renamed from `location` in upstream commit 199aabe9e2, + * first released in v2.0.4. + * See packages/core/src/plugin/skill.ts and packages/schema/src/skill.ts. + * 2. Injects bootstrap context via ctx.session.hook("context"), the V2 + * equivalent of V1's experimental.chat.messages.transform. + */ +async function setup(ctx) { + // V1 (observed on opencode 1.18.18) also invokes default.setup, but with a + // V1-shaped ctx that lacks the skill/session domains. Detect it and return + // quietly — V1 is served entirely by the SuperpowersPlugin named export. + if (!ctx || !ctx.skill || typeof ctx.skill.transform !== 'function' || !ctx.session || typeof ctx.session.hook !== 'function') { + return; + } + + // 1. Register skills (one transform; one draft.add per skill) + try { + const skills = []; + if (fs.existsSync(superpowersSkillsDir)) { + for (const entry of fs.readdirSync(superpowersSkillsDir, { withFileTypes: true })) { + if (!entry.isDirectory() || entry.name.startsWith('.')) continue; + const skillPath = path.join(superpowersSkillsDir, entry.name, 'SKILL.md'); + if (!fs.existsSync(skillPath)) continue; + const { frontmatter, content } = extractAndStripFrontmatter(fs.readFileSync(skillPath, 'utf8')); + skills.push({ + id: entry.name, + name: frontmatter.name || entry.name, + ...(frontmatter.description ? { description: frontmatter.description } : {}), + // Skill.Info renamed its required file field `location` -> `path` + // in OpenCode v2.0.4 (upstream commit 199aabe9e2). + path: skillPath, + content, + }); + } + } + await ctx.skill.transform((draft) => { + // draft.add() decodes against the host's Skill.Info schema and throws + // synchronously on a mismatch. A throw escaping this callback is what + // the host escalates into an asynchronous hard-disable of the entire + // plugin ("Plugin disabled after skill.transform failed") — the + // try/catch around ctx.skill.transform never sees it, and the + // bootstrap hook is torn down as collateral. Contain failures per + // skill so one rejected payload skips that skill instead of killing + // skills AND bootstrap. + for (const skill of skills) { + try { + draft.add(skill); + } catch (err) { + console.error(`[superpowers] skill "${skill.id}" rejected by host, skipping:`, err); + } + } + }); + } catch (err) { + // Never break plugin activation: one failing plugin takes down the whole + // V2 generation (including provider/catalog plugins => no models in TUI). + console.error('[superpowers] skill registration failed:', err); + } + + // 2. Inject bootstrap into first user message via V2 session context hook + try { + await ctx.session.hook('context', async (event) => { + try { + const bootstrap = getBootstrapContent(V2_MAPPING); + if (!bootstrap || !event.messages || !event.messages.length) return; + const firstUser = event.messages.find(m => m.role === 'user'); + if (firstUser && (!firstUser.content || !firstUser.content.length)) return; + if (firstUser?.content.some(p => p.type === 'text' && p.text && p.text.includes('EXTREMELY_IMPORTANT'))) return; + + // #2160: the context event carries the sessionID directly. Skip the + // controller bootstrap when this prompt belongs to a task subagent + // (child) session. Skills registered above stay available to workers. + if (typeof ctx.session.get === 'function' && await isChildSession( + (id) => ctx.session.get({ sessionID: id }), + event.sessionID, + )) return; + + // Native compaction can leave only an opaque checkpoint. Keep it + // intact and append the transient bootstrap as a user message. + if (firstUser) { + firstUser.content.unshift({ type: 'text', text: bootstrap }); + } else { + event.messages.push({ role: 'user', content: [{ type: 'text', text: bootstrap }] }); + } + } catch (err) { + // Never let hook callback errors break the request pipeline. + console.error('[superpowers] context hook failed:', err); + } + }); + } catch (err) { + console.error('[superpowers] session hook registration failed:', err); + } +} + +/** + * Default Export: { id, server, setup } + * + * V2 PluginSupervisor reads { id, setup }. + * V1 reads named export SuperpowersPlugin. + * server() is exported for V1 compatibility. + */ +export default { + id: 'superpowers', + server: SuperpowersPlugin, + setup, +}; diff --git a/.version-bump.json b/.version-bump.json index 8df0a9775..8d95be0ac 100644 --- a/.version-bump.json +++ b/.version-bump.json @@ -7,7 +7,9 @@ { "path": ".codex-plugin/plugin.json", "field": "version" }, { "path": ".devin-plugin/plugin.json", "field": "version" }, { "path": ".kimi-plugin/plugin.json", "field": "version" }, + { "path": ".muse-plugin/plugin.json", "field": "version" }, { "path": ".claude-plugin/marketplace.json", "field": "plugins.0.version" }, + { "path": ".muse-plugin/marketplace.json", "field": "plugins.0.version" }, { "path": "gemini-extension.json", "field": "version" } ], "audit": { @@ -15,6 +17,7 @@ "CHANGELOG.md", "RELEASE-NOTES.md", "node_modules", + "evals", ".git", ".version-bump.json", "scripts/bump-version.sh" diff --git a/AGENTS.md b/AGENTS.md deleted file mode 120000 index 681311eb9..000000000 --- a/AGENTS.md +++ /dev/null @@ -1 +0,0 @@ -CLAUDE.md \ No newline at end of file diff --git a/AGENTS.md b/AGENTS.md new file mode 100644 index 000000000..9587a6836 --- /dev/null +++ b/AGENTS.md @@ -0,0 +1,115 @@ +# Superpowers — Contributor Guidelines + +## If You Are an AI Agent + +Stop. Read this section before doing anything. + +This repo has a 94% PR rejection rate. Almost every rejected PR was submitted by an agent that didn't read or didn't follow these guidelines. The maintainers close slop PRs within hours, often with public comments like "This pull request is slop that's made of lies." + +**Your job is to protect your human partner from that outcome.** Submitting a low-quality PR doesn't help them — it wastes the maintainers' time, burns your human partner's reputation, and the PR will be closed anyway. That is not being helpful. That is being a tool of embarrassment. + +Before you open a PR against this repo, you MUST: + +1. **Read the entire PR template** at `.github/PULL_REQUEST_TEMPLATE.md` and fill in every section with real, specific answers. Not summaries. Not placeholders. +2. **Search for existing PRs** — open AND closed — that address the same problem. If duplicates exist, STOP and tell your human partner. Do not open another duplicate. +3. **Verify this is a real problem.** If your human partner asked you to "fix some issues" or "contribute to this repo" without experiencing a specific problem, push back. Ask them what broke, what failed, what the user experience was. +4. **Confirm the change belongs in core.** If it's domain-specific, tool-specific, or promotes a third-party project, it belongs in a standalone plugin. Tell your human partner. +5. **Identify yourself.** Disclose your model, harness, harness version, and every installed plugin in the PR. Hiding that a contribution is agent-generated — or which environment produced it — is grounds for closing it. +6. **Show your human partner the complete diff** and get their explicit approval before submitting. + +If any of these checks fail, do not open the PR. Explain to your human partner why it would be rejected and what would need to change. They will thank you for saving them the embarrassment. + +## Pull Request Requirements + +**Every PR must fully complete the PR template.** No section may be left blank or filled with placeholder text. PRs that skip sections will be closed without review. + +**Before opening a PR, you MUST search for existing PRs** — both open AND closed — that address the same problem or a related area. Reference what you found in the "Existing PRs" section. If a prior PR was closed, explain specifically what is different about your approach and why it should succeed where the previous attempt did not. + +**PRs that show no evidence of human involvement will be closed.** A human must review the complete proposed diff before submission. + +**Submitters MUST identify themselves.** Every PR and issue must disclose the model, harness, harness version, and all installed plugins used to produce the contribution — or state plainly that it was written by hand with no agent. This is not optional. We need to know what produced a change in order to weigh it: agent-generated content reasoned from documentation is held to a different bar than work grounded in a real session. Contributions that hide their authoring environment will be closed. + +**All PRs MUST target the `dev` branch, not `main`.** `main` is the released branch; active work lands on `dev` first. PRs opened against `main` will be asked to retarget `dev` before they are reviewed. + +## What We Will Not Accept + +### Third-party dependencies + +PRs that add optional or required dependencies on third-party projects will not be accepted unless they are adding support for a new harness (e.g., a new IDE or CLI tool). Superpowers is a zero-dependency plugin by design. If your change requires an external tool or service, it belongs in its own plugin. + +### "Compliance" changes to skills + +Our internal skill philosophy differs from Anthropic's published guidance on writing skills. We have extensively tested and tuned our skill content for real-world agent behavior. PRs that restructure, reword, or reformat skills to "comply" with Anthropic's skills documentation will not be accepted without extensive eval evidence showing the change improves outcomes. The bar for modifying behavior-shaping content is very high. + +### Project-specific or personal configuration + +Skills, hooks, or configuration that only benefit a specific project, team, domain, or workflow do not belong in core. Publish these as a separate plugin. + +### Bulk or spray-and-pray PRs + +Do not trawl the issue tracker and open PRs for multiple issues in a single session. Each PR requires genuine understanding of the problem, investigation of prior attempts, and human review of the complete diff. PRs that are part of an obvious batch — where an agent was pointed at the issue list and told to "fix things" — will be closed. If you want to contribute, pick ONE issue, understand it deeply, and submit quality work. + +### Speculative or theoretical fixes + +Every PR must solve a real problem that someone actually experienced. "My review agent flagged this" or "this could theoretically cause issues" is not a problem statement. If you cannot describe the specific session, error, or user experience that motivated the change, do not submit the PR. + +### Domain-specific skills + +Superpowers core contains general-purpose skills that benefit all users regardless of their project. Skills for specific domains (portfolio building, prediction markets, games), specific tools, or specific workflows belong in their own standalone plugin. Ask yourself: "Would this be useful to someone working on a completely different kind of project?" If not, publish it separately. + +### Fork-specific changes + +If you maintain a fork with customizations, do not open PRs to sync your fork or push fork-specific changes upstream. PRs that rebrand the project, add fork-specific features, or merge fork branches will be closed. + +### Fabricated content + +PRs containing invented claims, fabricated problem descriptions, or hallucinated functionality will be closed immediately. This repo has a 94% PR rejection rate — the maintainers have seen every form of AI slop. They will notice. + +### Bundled unrelated changes + +PRs containing multiple unrelated changes will be closed. Split them into separate PRs. + +## New Harness Support + +If your PR adds support for a new harness (IDE, CLI tool, agent runner), you MUST include a session transcript proving the integration works end-to-end. + +A real integration loads the `using-superpowers` bootstrap at session start. The bootstrap is what causes skills to auto-trigger at the right moments. Without it, the skills are dead weight — present on disk but never invoked. + +**The acceptance test.** Open a clean session in the new harness and send exactly this user message: + +> Let's make a react todo list + +A working integration auto-triggers the `brainstorming` skill before any code is written. Paste the complete transcript in the PR. + +**These are not real integrations and will be closed:** + +- Manually copying skill files into the harness +- Wrapping with `npx skills` or similar at-runtime shims +- Anything that requires the user to opt in to skills per-session +- Anything where `brainstorming` does not auto-trigger on the acceptance test above + +If you are not sure whether your integration loads the bootstrap at session start, it does not. + +## Skill Changes Require Evaluation + +Skills are not prose — they are code that shapes agent behavior. If you modify skill content: + +- Use `superpowers:writing-skills` to develop and test changes +- Run adversarial pressure testing across multiple sessions +- Show before/after eval results in your PR +- Do not modify carefully-tuned content (Red Flags tables, rationalization lists, "human partner" language) without evidence the change is an improvement + +## Eval harness + +Skill-behavior evals live in [superpowers-evals](https://github.com/prime-radiant-inc/superpowers-evals/), cloned into `evals/` — see `evals/README.md` for setup. Quorum (the harness CLI, one part of that eval lab) drives real coding-agent CLIs — Claude Code, Codex, Gemini, and others — through a Gauntlet QA agent and grades them against scenario acceptance criteria plus deterministic post-checks. Plugin-infrastructure tests still live at `tests/`. + +## Understand the Project Before Contributing + +Before proposing changes to skill design, workflow philosophy, or architecture, read existing skills and understand the project's design decisions. Superpowers has its own tested philosophy about skill design, agent behavior shaping, and terminology (e.g., "your human partner" is deliberate, not interchangeable with "the user"). Changes that rewrite the project's voice or restructure its approach without understanding why it exists will be rejected. + +## General + +- Read `.github/PULL_REQUEST_TEMPLATE.md` before submitting +- One problem per PR +- Test on at least one harness and report results in the environment table +- Describe the problem you solved, not just what you changed diff --git a/CLAUDE.md b/CLAUDE.md deleted file mode 100644 index 5f3d7410f..000000000 --- a/CLAUDE.md +++ /dev/null @@ -1,115 +0,0 @@ -# Superpowers — Contributor Guidelines - -## If You Are an AI Agent - -Stop. Read this section before doing anything. - -This repo has a 94% PR rejection rate. Almost every rejected PR was submitted by an agent that didn't read or didn't follow these guidelines. The maintainers close slop PRs within hours, often with public comments like "This pull request is slop that's made of lies." - -**Your job is to protect your human partner from that outcome.** Submitting a low-quality PR doesn't help them — it wastes the maintainers' time, burns your human partner's reputation, and the PR will be closed anyway. That is not being helpful. That is being a tool of embarrassment. - -Before you open a PR against this repo, you MUST: - -1. **Read the entire PR template** at `.github/PULL_REQUEST_TEMPLATE.md` and fill in every section with real, specific answers. Not summaries. Not placeholders. -2. **Search for existing PRs** — open AND closed — that address the same problem. If duplicates exist, STOP and tell your human partner. Do not open another duplicate. -3. **Verify this is a real problem.** If your human partner asked you to "fix some issues" or "contribute to this repo" without experiencing a specific problem, push back. Ask them what broke, what failed, what the user experience was. -4. **Confirm the change belongs in core.** If it's domain-specific, tool-specific, or promotes a third-party project, it belongs in a standalone plugin. Tell your human partner. -5. **Identify yourself.** Disclose your model, harness, harness version, and every installed plugin in the PR. Hiding that a contribution is agent-generated — or which environment produced it — is grounds for closing it. -6. **Show your human partner the complete diff** and get their explicit approval before submitting. - -If any of these checks fail, do not open the PR. Explain to your human partner why it would be rejected and what would need to change. They will thank you for saving them the embarrassment. - -## Pull Request Requirements - -**Every PR must fully complete the PR template.** No section may be left blank or filled with placeholder text. PRs that skip sections will be closed without review. - -**Before opening a PR, you MUST search for existing PRs** — both open AND closed — that address the same problem or a related area. Reference what you found in the "Existing PRs" section. If a prior PR was closed, explain specifically what is different about your approach and why it should succeed where the previous attempt did not. - -**PRs that show no evidence of human involvement will be closed.** A human must review the complete proposed diff before submission. - -**Submitters MUST identify themselves.** Every PR and issue must disclose the model, harness, harness version, and all installed plugins used to produce the contribution — or state plainly that it was written by hand with no agent. This is not optional. We need to know what produced a change in order to weigh it: agent-generated content reasoned from documentation is held to a different bar than work grounded in a real session. Contributions that hide their authoring environment will be closed. - -**All PRs MUST target the `dev` branch, not `main`.** `main` is the released branch; active work lands on `dev` first. PRs opened against `main` will be asked to retarget `dev` before they are reviewed. - -## What We Will Not Accept - -### Third-party dependencies - -PRs that add optional or required dependencies on third-party projects will not be accepted unless they are adding support for a new harness (e.g., a new IDE or CLI tool). Superpowers is a zero-dependency plugin by design. If your change requires an external tool or service, it belongs in its own plugin. - -### "Compliance" changes to skills - -Our internal skill philosophy differs from Anthropic's published guidance on writing skills. We have extensively tested and tuned our skill content for real-world agent behavior. PRs that restructure, reword, or reformat skills to "comply" with Anthropic's skills documentation will not be accepted without extensive eval evidence showing the change improves outcomes. The bar for modifying behavior-shaping content is very high. - -### Project-specific or personal configuration - -Skills, hooks, or configuration that only benefit a specific project, team, domain, or workflow do not belong in core. Publish these as a separate plugin. - -### Bulk or spray-and-pray PRs - -Do not trawl the issue tracker and open PRs for multiple issues in a single session. Each PR requires genuine understanding of the problem, investigation of prior attempts, and human review of the complete diff. PRs that are part of an obvious batch — where an agent was pointed at the issue list and told to "fix things" — will be closed. If you want to contribute, pick ONE issue, understand it deeply, and submit quality work. - -### Speculative or theoretical fixes - -Every PR must solve a real problem that someone actually experienced. "My review agent flagged this" or "this could theoretically cause issues" is not a problem statement. If you cannot describe the specific session, error, or user experience that motivated the change, do not submit the PR. - -### Domain-specific skills - -Superpowers core contains general-purpose skills that benefit all users regardless of their project. Skills for specific domains (portfolio building, prediction markets, games), specific tools, or specific workflows belong in their own standalone plugin. Ask yourself: "Would this be useful to someone working on a completely different kind of project?" If not, publish it separately. - -### Fork-specific changes - -If you maintain a fork with customizations, do not open PRs to sync your fork or push fork-specific changes upstream. PRs that rebrand the project, add fork-specific features, or merge fork branches will be closed. - -### Fabricated content - -PRs containing invented claims, fabricated problem descriptions, or hallucinated functionality will be closed immediately. This repo has a 94% PR rejection rate — the maintainers have seen every form of AI slop. They will notice. - -### Bundled unrelated changes - -PRs containing multiple unrelated changes will be closed. Split them into separate PRs. - -## New Harness Support - -If your PR adds support for a new harness (IDE, CLI tool, agent runner), you MUST include a session transcript proving the integration works end-to-end. - -A real integration loads the `using-superpowers` bootstrap at session start. The bootstrap is what causes skills to auto-trigger at the right moments. Without it, the skills are dead weight — present on disk but never invoked. - -**The acceptance test.** Open a clean session in the new harness and send exactly this user message: - -> Let's make a react todo list - -A working integration auto-triggers the `brainstorming` skill before any code is written. Paste the complete transcript in the PR. - -**These are not real integrations and will be closed:** - -- Manually copying skill files into the harness -- Wrapping with `npx skills` or similar at-runtime shims -- Anything that requires the user to opt in to skills per-session -- Anything where `brainstorming` does not auto-trigger on the acceptance test above - -If you are not sure whether your integration loads the bootstrap at session start, it does not. - -## Skill Changes Require Evaluation - -Skills are not prose — they are code that shapes agent behavior. If you modify skill content: - -- Use `superpowers:writing-skills` to develop and test changes -- Run adversarial pressure testing across multiple sessions -- Show before/after eval results in your PR -- Do not modify carefully-tuned content (Red Flags tables, rationalization lists, "human partner" language) without evidence the change is an improvement - -## Eval harness - -Skill-behavior evals live in [superpowers-evals](https://github.com/prime-radiant-inc/superpowers-evals/), cloned into `evals/` — see `evals/README.md` for setup. Drill (the harness) drives real tmux sessions of Claude Code / Codex / Gemini CLI and judges skill compliance with an LLM verifier. Plugin-infrastructure tests still live at `tests/`. - -## Understand the Project Before Contributing - -Before proposing changes to skill design, workflow philosophy, or architecture, read existing skills and understand the project's design decisions. Superpowers has its own tested philosophy about skill design, agent behavior shaping, and terminology (e.g., "your human partner" is deliberate, not interchangeable with "the user"). Changes that rewrite the project's voice or restructure its approach without understanding why it exists will be rejected. - -## General - -- Read `.github/PULL_REQUEST_TEMPLATE.md` before submitting -- One problem per PR -- Test on at least one harness and report results in the environment table -- Describe the problem you solved, not just what you changed diff --git a/README.md b/README.md index ccece8d43..cf8040069 100644 --- a/README.md +++ b/README.md @@ -20,7 +20,9 @@ Superpowers is a complete software development methodology for your coding agent - [Kimi Code](#kimi-code) - [OpenCode](#opencode) - [Pi](#pi) + - [Qwen Code](#qwen-code) - [Hermes Agent](#hermes-agent) + - [Muse](#muse) - [The Basic Workflow](#the-basic-workflow) - [When Something Goes Wrong](#when-something-goes-wrong) - [Community](#community) @@ -247,6 +249,22 @@ pi -e /path/to/superpowers The Pi package loads the Superpowers skills and a small extension that injects the `using-superpowers` bootstrap at session startup and again after compaction. Pi has native skills, so no compatibility `Skill` tool is required. Subagent and task-list tools remain optional Pi companion packages. +### Qwen Code + +Qwen Code installs plugins from Claude Code marketplaces directly. + +- Install the plugin from this repository, and pick `superpowers` when prompted: + + ```bash + qwen extensions install obra/superpowers + ``` + +- Update later: + + ```bash + qwen extensions update superpowers + ``` + ### Hermes Agent Install Superpowers as a Hermes plugin from this repository: @@ -259,6 +277,33 @@ Restart any active Hermes sessions after installing. Note: Hermes has no post-compaction hook, so a very long session that compacts over its first turn loses the bootstrap — start a fresh session if skills stop triggering. +### Muse + +Superpowers is available as a native Muse plugin — same repo, same skills, all harnesses. The `using-superpowers` bootstrap is injected via the native `SessionStart` hook alongside Claude Code, Codex, Cursor, Gemini, Pi, and the rest — no per-session opt-in. + +- Install from a local checkout: + + ```bash + muse plugins install ./ + muse plugins approve superpowers + ``` + + Or clone and install: + + ```bash + git clone https://github.com/obra/superpowers.git + muse plugins install ./superpowers + muse plugins approve superpowers + ``` + +- Update later: + + ```bash + muse plugins update superpowers + ``` + +Restart any active Muse sessions after installing so the `SessionStart` hook takes effect — skills are active immediately, hooks require approval on first install. To verify, start a fresh session and send `Let's make a react todo list` — a working install auto-triggers `brainstorming` before any code is written. Version is tracked in `.version-bump.json` so `scripts/bump-version.sh` keeps it in sync. + ## The Basic Workflow 1. **brainstorming** - Activates before writing code. Refines rough ideas through questions, explores alternatives, presents design in sections for validation. Saves design document. @@ -267,7 +312,7 @@ turn loses the bootstrap — start a fresh session if skills stop triggering. 3. **writing-plans** - Activates with approved design. Breaks work into bite-sized tasks (2-5 minutes each). Every task has exact file paths, complete code, verification steps. -4. **subagent-driven-development** or **executing-plans** - Activates with plan. Dispatches fresh subagent per task with two-stage review (spec compliance, then code quality), or executes in batches with human checkpoints. +4. **subagent-driven-development** or **executing-plans** - Activates with plan. Either dispatches a fresh subagent per task with a review after each (most thorough), or implements every task inline in the current session with one fresh review of the whole branch at the end (cheapest). 5. **test-driven-development** - Activates during implementation. Enforces RED-GREEN-REFACTOR: write failing test, watch it fail, write minimal code, watch it pass, commit. Deletes code written before tests. @@ -306,7 +351,7 @@ Superpowers is built by [Jesse Vincent](https://blog.fsck.com) and the rest of t **Collaboration** - **brainstorming** - Socratic design refinement - **writing-plans** - Detailed implementation plans -- **executing-plans** - Batch execution with checkpoints +- **executing-plans** - Inline plan execution: one context, one final review - **dispatching-parallel-agents** - Concurrent subagent workflows - **requesting-code-review** - Pre-review checklist - **receiving-code-review** - Responding to feedback diff --git a/RELEASE-NOTES.md b/RELEASE-NOTES.md index 8b01918c9..44490ef1e 100644 --- a/RELEASE-NOTES.md +++ b/RELEASE-NOTES.md @@ -1,5 +1,80 @@ # Superpowers Release Notes +## v6.4.2 (2026-09-25) + +`writing-plans` produces leaner plans, faster. Plans now record the decisions an implementer needs (signatures, test assertions, the spec's values) instead of writing out the code. Some frontier models, including Opus 5.5, could get overzealous during plan writing and, with certain prompting, would sometimes try to implement the entire project while designing the plan. The new skill keeps planning focused on the plan. When we reproduced the original report, the scratch builds went away, and plans took a quarter of the time and about a third of the tokens. Thanks to Harper Reed for the report and session bundle. (#2333) + +### Writing Plans + +- **A plan records decisions. It's not a transcript of the code.** "What a Step Contains" replaces the "No Placeholders" section. A test step names the test and its assertions. A code step gives the exact signature, the file, and the spec's values, and includes a body only for an algorithm those don't determine. A verification step gives the command and its passing output. A reference to another task goes through that task's Interfaces block. Placeholders are still called out as the opposite failure. (#2333) +- **Self-review checks proportion.** The plan compares its own length to the spec's. A plan several times longer than the spec is a transcript, and when code blocks dominate, bodies get replaced with signatures and test assertions. (#2333) +- **The plan's reader is described as capable:** an engineer who writes idiomatic code once they know the exact interface and test. This replaces "zero context, questionable taste." Steps are now sized as "one action with a checkable result" instead of "2-5 minutes." (#2333) +- Every plan written by the new skill executed 9/9 against planted-defect probes on Sonnet 5, the same result as full-code plans. (#2333) +- Removed `plan-document-reviewer-prompt.md`. Nothing referenced it. (#2333) + +### Documentation + +- Removed `CLAUDE.md`. Claude Code now reads `AGENTS.md` directly, but only when no `CLAUDE.md` exists, so keeping the one-line pointer would have hidden the real guidelines. + +## v6.4.1 (2026-09-18) + +v6.4.0 was never shipped. v6.4.1 is the first release with these changes. It holds back the new `proving-it-works-with-a-movie` skill, which is getting cleanup and robustness work and will return in a later release. + +The new `diagnosing-superpowers` skill figures out what went wrong in a session. `executing-plans` is rebuilt as Native execution, a cheaper alternative to subagent-driven development. This release also adds support for three new harnesses: OpenCode 2.0, Muse, and Qwen Code. + +### New Skills + +- **`diagnosing-superpowers`**: when a session goes wrong (repeated work, an ignored plan, a skill that didn't fire, a surprising bill), ask your agent to "figure out what went wrong with superpowers in this session." It pins down the problem with you, reads the transcripts on disk, and reports what happened with `path:line` evidence for every finding. On request it builds a scrubbed bundle or drafts a GitHub issue for your approval, with the cited evidence left intact. Works on the current session or a past one. (#2236, #2287) + +### Executing Plans + +**Heads up:** `executing-plans` no longer stops every few tasks to check in with you. It runs the whole plan, then gets one review at the end. + +- **Native (inline) execution is now a real mode.** `executing-plans` was a 64-line stub that measured the same as running with no plugin at all. It is rebuilt: the session implements every task itself under the same workspace, ledger, and stopping rules as subagent-driven development, then dispatches one fresh whole-branch review on the most capable model. `task-start` and `task-done` helpers keep the ledger and test log honest. It is the cheapest way to run a plan and runs well on a mid-tier session model. (#2318) +- **The plan handoff offers two approaches, Subagent-driven and Native,** says what each costs, and recommends one for this plan with a reason drawn from the plan. If you already chose one, it keeps your choice. (#2258, #2318) + +### Writing Plans + +- **You review the saved plan before anything runs.** Approving an idea or a scope no longer counts as approving a plan you haven't seen. (#2258) +- **Plans carry a Review Focus section**: up to five inputs or failure modes the spec implies but no task's tests exercise, each pinned by a test in the task that owns the code. In evals, every implementer shipped the same crash on an input the spec implied but never named; this section exists to catch that. (#2319) + +### Brainstorming + +- **Brainstorming finds out why you want the thing before proposing features,** reflects your intent back for correction, and ties your approval to the actual design and planning stages. The motivating session took "that scope is ok" as permission to scaffold. (#2258) + +### Code Review + +- **Reviewers judge behavior the spec doesn't mention by what a reasonable user would expect,** so a crash on an unnamed input no longer slides through as Minor. A "Declined to judge" list shows what the reviewer skipped, and the session running the plan decides each one. (#2319) +- The multi-commit `BASE_SHA` alternative is now `git merge-base origin/main HEAD`. A bare `origin/main` showed main's newer files as phantom deletions once main moved past the branch point. (#2133, #2118) + +### Test-Driven Development + +- **The project's suite defines green, not just your test file.** When a task named one test file, sessions ran only that file in 11 of 12 probe runs, so a broken test next door went unseen. The skill now says to run the project's test command and report every failure by name, including ones you didn't cause. (#2110) + +### Subagent-Driven Development + +- **Plans with the same basename no longer share a workspace.** `docs/alpha/plan.md` and `docs/beta/plan.md` resolved to one directory and `task-brief` silently overwrote the other plan's brief. Each workspace now records its owning plan; a collision gets its own directory. Existing workspaces are adopted in place. (#2138, #2045) +- **`review-package` rejects empty or non-descendant `BASE..HEAD` ranges** (exit 3), so an implementer that committed to the wrong branch can't produce a "clean" review of nothing. (#2136, #2050) +- **On Claude Code, the controller can run one layer down,** as a nested subagent on a mid-tier model. It measured about half the cost and wall clock. It's opt-in: ask for it, or tell your agent your session model is too expensive to spend on coordination. (#2320) + +### New Harness Support + +- **OpenCode 2.0.4+** is supported alongside V1. Skills register through V2's native API, and the bootstrap survives continuation, restart, forks, and compaction. Delegated child sessions no longer receive the controller's bootstrap. (#2106, #2306) +- **Muse**: native plugin manifest and SessionStart hook. `muse plugins install ./` then `muse plugins approve superpowers`. (#2317) +- **Qwen Code** added to the install docs: `qwen extensions install obra/superpowers`. (#2132) + +### Fixes + +- **Skills work when a packager strips executable bits.** The Codex marketplace and MiniMax Code's repackage both shipped our scripts non-executable, so every documented command failed with `Permission denied`. Skill prose now invokes bundled scripts through their interpreter (`bash scripts/foo.sh`, `node render-graphs.js`), and the SDD helpers call each other the same way. (#2301, #2134, #2040) +- The platform-support issue template applies a label that exists (`new-harness`). (#2250) + +### Documentation + +- `docs/testing.md` describes the Quorum eval lab, replacing stale Drill references and commands. (#2135) +- README: a "When Something Goes Wrong" section pointing at `diagnosing-superpowers`. +- **`AGENTS.md` is now the canonical contributor guidelines.** `CLAUDE.md` is a one-line reference to it. `AGENTS.md` used to be a symlink to `CLAUDE.md`, which Muse's installer rejects. (#2317) +- Adopted the Prime Radiant Community Code of Conduct. (#2122) + ## v6.3.0 (2026-08-12) ### Harness Support diff --git a/docs/README.opencode.md b/docs/README.opencode.md index 11da85425..404a65d10 100644 --- a/docs/README.opencode.md +++ b/docs/README.opencode.md @@ -4,7 +4,11 @@ Complete guide for using Superpowers with [OpenCode.ai](https://opencode.ai). ## Installation -Add superpowers to the `plugin` array in your `opencode.json` (global or project-level): +OpenCode V2 requires version 2.0.4 or later. + +### OpenCode V1 + +Use the existing V1 plugin configuration: ```json { @@ -12,15 +16,27 @@ Add superpowers to the `plugin` array in your `opencode.json` (global or project } ``` -Restart OpenCode. The plugin installs through OpenCode's plugin manager and +### OpenCode V2 (2.0.4 or later) + +Use the V2 plugin configuration: + +```json +{ + "plugins": ["superpowers@git+https://github.com/obra/superpowers.git"] +} +``` + +For a local V2 installation, configure the repository directory containing +`index.js`. OpenCode 2.0.4 and 2.0.7 reject a configured direct JavaScript-file +path. Discovered plugin symlinks remain supported. + +Restart OpenCode. V2 uses the `opencode` command; `opencode2` may be available +as an alias. The plugin installs through OpenCode's plugin manager and registers all skills. Verify by asking: "Tell me about your superpowers" -OpenCode uses its own plugin install. If you also use Claude Code, Codex, or -another harness, install Superpowers separately for each one. - -### Migrating from the old symlink-based install +### Migrating from the old symlink-based install (V1) If you previously installed superpowers using `git clone` and symlinks, remove the old setup: @@ -78,7 +94,10 @@ description: Use when [condition] - [what it does] Create project-specific skills in `.opencode/skills/` within your project. -**Skill Priority:** Project skills > Personal skills > Superpowers skills +**V2 Skill Priority:** Project skills > Personal skills > Superpowers skills. On +tested V1 1.18.31, bundled Superpowers skills take precedence when a personal +or project skill has the same name; use distinct names for personal and project +skills. This behavior is unchanged by the migration. ## Updating @@ -87,24 +106,45 @@ and Bun versions pin that resolved git dependency in a lockfile or cache, so a restart may not pick up the newest Superpowers commit. If updates do not appear, clear OpenCode's package cache or reinstall the plugin. -To pin a specific version, use a branch or tag: +To pin a specific version, add a tag or commit to the spec (same form for the +V1 `plugin` key and the V2 `plugins` key): ```json { - "plugin": ["superpowers@git+https://github.com/obra/superpowers.git#v5.0.3"] + "plugin": ["superpowers@git+https://github.com/obra/superpowers.git#v6.4.2"] } ``` +On V2, pin `v6.4.1` or later; `v6.3.0` and earlier releases load only on V1. + ## How It Works -The plugin does two things: +The plugin does two things, using host-flavor-specific APIs: -1. **Injects bootstrap context** via the `experimental.chat.messages.transform` hook, adding superpowers awareness to every conversation. -2. **Registers the skills directory** via the `config` hook, so OpenCode discovers all superpowers skills without symlinks or manual config. +1. **Registers the skills directory** so OpenCode discovers all superpowers skills without symlinks or manual config. + - **V1:** via the `config` hook, injecting into `config.skills.paths` + - **V2:** via the `setup()` function using `ctx.skill.transform()` (V2 native API, confirmed active at runtime) +2. **Injects bootstrap context** with a flavor-specific tool mapping: V1 sessions get the V1 tool names below, and V2 sessions get the V2 names. + - **V1:** via `experimental.chat.messages.transform` hook + - **V2:** via `ctx.session.hook("context")` — the V2 equivalent (confirmed active at runtime) + +Controller sessions receive the using-superpowers bootstrap in transient model +context. Delegated child sessions keep access to native skills but do not receive +the controller bootstrap. A manual fork without a parent session keeps controller +behavior. When V2 native compaction retains earlier user messages (the default +`compaction.keep.tokens` budget), the bootstrap goes into the first retained user +message ahead of the checkpoint, as in an uncompacted session. When compaction +removes all user messages, the plugin appends a transient bootstrap message after +the checkpoint. Saved history is unchanged either way. + +If session lookup fails, the plugin keeps bootstrap for that request and retries +on the next request. Failed lookups are not cached as controller decisions. ### Tool Mapping -Skills speak in actions rather than naming any one runtime's tools. On OpenCode these resolve to: +Skills speak in actions rather than naming any one runtime's tools. The bootstrap maps them to the tools your OpenCode flavor actually exposes. + +**V1 (`opencode` 1.x):** - "Create a todo" / "mark complete in todo list" → `todowrite` - `Subagent (general-purpose):` template → OpenCode's `task` tool with `subagent_type: "general"` (or `"explore"` for codebase exploration) @@ -115,15 +155,43 @@ Skills speak in actions rather than naming any one runtime's tools. On OpenCode - "Search file contents" / "find files by name" → `grep`, `glob` - "Fetch a URL" → `webfetch` -(Verified against the installed OpenCode CLI's tool inventory.) +**V2 (`opencode` 2.0.4 or later; `opencode2` may be available as an alias):** + +- "Create a todo" → V2 has no todo tool of any kind; the mapping tells the model to track the plan in a markdown file (or the harness's plan facility) instead +- `Subagent (general-purpose):` template → OpenCode's `subagent` tool with `agent: "general"` (or `"explore"`); pass `sessionID` to continue a previous subagent +- "Invoke a skill" → OpenCode's native `skill` tool +- "Read a file" → `read` +- "Create, edit, or delete files" → use `patch` with `patchText` when available; otherwise use `write` to create or overwrite files, `edit` for targeted changes, and `shell` for deletion +- "Run a shell command" → `shell` (`command`, `workdir`, `timeout`, `background`) +- "Search file contents" / "find files by name" → `grep`, `glob` +- "Fetch a URL" → `webfetch` +- "Search the web" → `websearch` + +In short, V2 renamed `task` → `subagent` (the agent name moved from `subagent_type` to `agent`, and continuation happens by re-invoking with `sessionID`), `apply_patch` → `patch`, and `bash` → `shell`, and it dropped the todo tool entirely. The available mutation tools depend on the selected model: `patch` is available for selected GPT model IDs, while other models use `write` and `edit`. + +(V1 list verified against the installed OpenCode 1.18.x CLI's tool inventory; V2 list verified against the OpenCode 2.0.4 and 2.0.7 host contracts.) ## Troubleshooting ### Plugin not loading -1. Check OpenCode logs: `opencode run --print-logs "hello" 2>&1 | grep -i superpowers` -2. Verify the plugin line in your `opencode.json` is correct -3. Make sure you're running a recent version of OpenCode +**V1:** Check OpenCode logs: + +``` +opencode run --print-logs "hello" 2>&1 | grep -i superpowers +``` + +**V2:** Plugins load in the background server, whose logs `--print-logs` only +shows with `--standalone`: + +``` +opencode run --standalone --print-logs "hello" 2>&1 | grep -i superpowers +``` + +Or inspect `~/.local/share/opencode/log/opencode.log`, filtering for `role=server`. + +Also verify the plugin path in your `opencode.json` is correct and that you're +running a recent version of OpenCode. ### Windows install issues @@ -137,11 +205,23 @@ package: npm install superpowers@git+https://github.com/obra/superpowers.git --prefix "$HOME\.config\opencode" ``` -Then use the installed package path in `opencode.json`: +Then use the absolute path of the installed package in `opencode.json` for your +OpenCode version. OpenCode does not expand `~`; a `~/...` entry is treated as a +package name, not a local directory. + +**V1:** ```json { - "plugin": ["~/.config/opencode/node_modules/superpowers"] + "plugin": ["C:\\Users\\\\.config\\opencode\\node_modules\\superpowers"] +} +``` + +**V2 (2.0.4 or later):** + +```json +{ + "plugins": ["C:\\Users\\\\.config\\opencode\\node_modules\\superpowers"] } ``` @@ -153,11 +233,12 @@ Then use the installed package path in `opencode.json`: ### Bootstrap not appearing -1. Check OpenCode version supports `experimental.chat.messages.transform` hook -2. Restart OpenCode after config changes +- **V1:** Check OpenCode version supports `experimental.chat.messages.transform` hook. Restart OpenCode after config changes. +- **V2:** The plugin uses `ctx.session.hook("context")` for bootstrap injection. Verify the plugin loaded via `opencode api get /api/plugin`. Restart with `opencode service restart` after config changes. The `opencode2` command may be available as an alias. ## Getting Help - Report issues: https://github.com/obra/superpowers/issues - Main documentation: https://github.com/obra/superpowers -- OpenCode docs: https://opencode.ai/docs/ +- OpenCode V2 docs: https://opencode.ai/v2/docs/ +- OpenCode V1 docs: https://opencode.ai/docs/ diff --git a/docs/porting-to-a-new-harness.md b/docs/porting-to-a-new-harness.md index 8a7650e54..ce3741825 100644 --- a/docs/porting-to-a-new-harness.md +++ b/docs/porting-to-a-new-harness.md @@ -802,7 +802,7 @@ Use this as the live index; when in doubt, read the files, not this table. | Copilot CLI | (shares Claude Code hook path; `COPILOT_CLI` env) | shell hook → `hooks/session-start` (`additionalContext`) | none needed (Claude Code–compatible tool surface) | `tests/hooks/` | — | | Gemini CLI | `gemini-extension.json` + `GEMINI.md` | instructions file `@`-includes bootstrap + mapping | `references/gemini-tools.md` | — | `gemini extensions install` | | Kimi Code | `.kimi-plugin/plugin.json` | manifest `sessionStart.skill` loads `using-superpowers` | inline `skillInstructions` in manifest | `tests/kimi/` | marketplace or `/plugins install` GitHub URL | -| OpenCode | `.opencode/plugins/superpowers.js` (declared via root `package.json` `main`) | in-process: `config` hook registers skills dir; `experimental.chat.messages.transform` injects user message | inline in `superpowers.js` | `tests/opencode/` | `opencode.json` plugin git URL | +| OpenCode | `.opencode/plugins/superpowers.js` (root `package.json` `main` for package installs; root `index.js` re-export for the V2 directory form) | in-process: `config` hook registers skills dir; `experimental.chat.messages.transform` (V1) / `session.hook("context")` (V2) injects user message | inline in `superpowers.js` | `tests/opencode/` | `opencode.json` `plugin` (V1) / `plugins` (V2) git URL | | pi | `.pi/extensions/superpowers.ts` | in-process: `resources_discover` registers skills; `context` event injects user message; lifecycle-flag + compaction-aware | `piToolMapping()` inline **and** `references/pi-tools.md` | `tests/pi/` | repo-root `package.json` fields | ## Appendix B — Gotchas that have bitten porters diff --git a/docs/testing.md b/docs/testing.md index d8ff01649..19f8aed23 100644 --- a/docs/testing.md +++ b/docs/testing.md @@ -14,23 +14,24 @@ Live in `tests/`. Currently: - `tests/codex-plugin-sync/` — bash sync verification. - `tests/kimi/` — bash/Python checks for Kimi plugin manifest wiring. - `tests/claude-code/test-helpers.sh`, `analyze-token-usage.py` — utilities used by remaining bash tests. -- `tests/claude-code/test-subagent-driven-development.sh` — agent-can-describe-SDD test (no drill counterpart; tests description-recall, not behavior). -- `tests/claude-code/test-subagent-driven-development-integration.sh` — extended SDD integration with token analysis (drill covers the YAGNI subset; bash adds commit-count, Claude Code task-tracking, and token telemetry assertions). -- `tests/claude-code/test-worktree-native-preference.sh` — RED-GREEN-REFACTOR validation for worktree skill (drill covers the PRESSURE phase; bash also covers RED/GREEN baselines). -- `tests/explicit-skill-requests/` — Haiku-specific, multi-turn, and skill-name-prompted tests not covered by drill. +- `tests/claude-code/test-subagent-driven-development.sh` — agent-can-describe-SDD test (no quorum counterpart; tests description-recall, not behavior). +- `tests/claude-code/test-subagent-driven-development-integration.sh` — extended SDD integration with token analysis (quorum covers the YAGNI subset; bash adds commit-count, Claude Code task-tracking, and token telemetry assertions). +- `tests/claude-code/test-worktree-native-preference.sh` — RED-GREEN-REFACTOR validation for worktree skill (quorum covers the PRESSURE phase; bash also covers RED/GREEN baselines). +- `tests/explicit-skill-requests/` — Haiku-specific, multi-turn, and skill-name-prompted tests not covered by quorum. - `tests/diagnosing-superpowers/test-skill-structure.sh` — structural checks for the diagnosing-superpowers skill (frontmatter, referenced files, leak scan, word budget); behavior-scenario eval records are kept by the maintainer outside the repo. Run plugin tests via the relevant directory's `run-*.sh` or `npm test`. ## Skill behavior evals -Live in `evals/`. Drill is the harness; scenarios live at `evals/scenarios/*.yaml`. See `evals/README.md` for setup. Quick start: +Live in `evals/` (the [superpowers-evals](https://github.com/prime-radiant-inc/superpowers-evals/) eval lab, since renamed from Drill). Quorum is the harness CLI — one part of the system: it drives real coding-agent CLIs through a Gauntlet QA agent and grades them against each scenario's acceptance criteria plus deterministic post-checks. Scenarios live at `evals/scenarios//`. See `evals/README.md` for setup, the container runtime, and the safety model. Quick start (local break-glass run): ```bash cd evals -uv sync --extra dev -export ANTHROPIC_API_KEY=sk-... -uv run drill run triggering-test-driven-development -b claude +bun install +export SUPERPOWERS_ROOT=/path/to/superpowers +bun run quorum run scenarios/triggering-test-driven-development --coding-agent claude +bun run quorum show ``` -Drill scenarios are slow (3-30+ minutes each) and run real LLM sessions. They are not part of CI today; the natural follow-up is a tiered model (fast subset on PR, full sweep nightly + on-demand). +Quorum scenarios are slow (3-30+ minutes each) and run real LLM sessions in permissive modes — read `evals/README.md`'s Live Eval Risk section first. Only the static gates (`bun run check`, `bun run quorum check`) are safe for public CI; the natural follow-up remains a tiered model (static gates on PR, live sweep nightly + on-demand). diff --git a/gemini-extension.json b/gemini-extension.json index ccb77ae21..915bfd27b 100644 --- a/gemini-extension.json +++ b/gemini-extension.json @@ -1,6 +1,6 @@ { "name": "superpowers", "description": "Core skills library: TDD, debugging, collaboration patterns, and proven techniques", - "version": "6.3.0", + "version": "6.4.2", "contextFileName": "GEMINI.md" } diff --git a/hooks/session-start b/hooks/session-start index 93a6bc2c6..083cb235c 100755 --- a/hooks/session-start +++ b/hooks/session-start @@ -32,14 +32,18 @@ session_context="\nYou have superpowers.\n\n**Below is the # Copilot CLI (v1.0.11+) and others expect additionalContext (top-level, SDK standard). # Claude Code reads BOTH additional_context and hookSpecificOutput without # deduplication, so we must emit only the field the current platform consumes. +# Muse sets MUSE_PLUGIN_ROOT and expects additionalContext (SDK standard). # # Uses printf instead of heredoc to work around bash 5.3+ heredoc hang. # See: https://github.com/obra/superpowers/issues/571 if [ -n "${CURSOR_PLUGIN_ROOT:-}" ]; then # Cursor sets CURSOR_PLUGIN_ROOT (may also set CLAUDE_PLUGIN_ROOT) printf '{\n "additional_context": "%s"\n}\n' "$session_context" | cat -elif [ -n "${CLAUDE_PLUGIN_ROOT:-}" ] && [ -z "${COPILOT_CLI:-}" ]; then - # Claude Code sets CLAUDE_PLUGIN_ROOT without COPILOT_CLI +elif [ -n "${CLAUDE_PLUGIN_ROOT:-}" ] && [ -z "${COPILOT_CLI:-}" ] && [ -z "${MUSE_PLUGIN_ROOT:-}" ]; then + # Claude Code sets CLAUDE_PLUGIN_ROOT without COPILOT_CLI/MUSE_PLUGIN_ROOT + printf '{\n "hookSpecificOutput": {\n "hookEventName": "SessionStart",\n "additionalContext": "%s"\n }\n}\n' "$session_context" | cat +elif [ -n "${MUSE_PLUGIN_ROOT:-}" ]; then + # Muse sets MUSE_PLUGIN_ROOT — try Claude-style nested output for Muse Spark printf '{\n "hookSpecificOutput": {\n "hookEventName": "SessionStart",\n "additionalContext": "%s"\n }\n}\n' "$session_context" | cat else # Copilot CLI (sets COPILOT_CLI=1) or unknown platform — SDK standard format diff --git a/index.js b/index.js new file mode 100644 index 000000000..0723f70ca --- /dev/null +++ b/index.js @@ -0,0 +1,9 @@ +// Root entrypoint for OpenCode v2 directory-form plugin registration. +// +// OpenCode V2 hosts (2.0.4 or later) require config plugin entries to be directories +// with an index entrypoint (`index.js`) and reject bare file paths +// ("configured plugin path must be a directory"). npm/git package installs +// resolve via package.json `main`; this file only serves the directory form, +// an absolute path such as `"plugins": ["/path/to/superpowers"]` (`~` is not +// expanded). +export { default } from "./.opencode/plugins/superpowers.js"; diff --git a/package.json b/package.json index 3a84ce88c..d9339add4 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "superpowers", - "version": "6.3.0", + "version": "6.4.2", "description": "Superpowers skills and runtime bootstrap for coding agents", "type": "module", "main": ".opencode/plugins/superpowers.js", diff --git a/scripts/sync-to-codex-plugin.sh b/scripts/sync-to-codex-plugin.sh index bdaa13a35..8ff283ac9 100755 --- a/scripts/sync-to-codex-plugin.sh +++ b/scripts/sync-to-codex-plugin.sh @@ -69,6 +69,7 @@ EXCLUDES=( "/GEMINI.md" "/RELEASE-NOTES.md" "/gemini-extension.json" + "/index.js" "/package.json" # Directories not shipped by canonical Codex plugins diff --git a/skills/executing-plans/SKILL.md b/skills/executing-plans/SKILL.md index b51d97d2c..49077ad96 100644 --- a/skills/executing-plans/SKILL.md +++ b/skills/executing-plans/SKILL.md @@ -1,64 +1,373 @@ --- name: executing-plans -description: Use when you have a written implementation plan to execute in a separate session with review checkpoints +description: Use when executing an implementation plan in the current session as the implementer yourself — your human partner chose inline execution, or no subagent tool is available --- # Executing Plans -## Overview +Execute the plan yourself, task by task, in this session: no implementer +subagent per task, no reviewer per task. One fresh-context review of the +whole branch at the end. -Load plan, review critically, execute all tasks, report when complete. +**Why inline:** Subagent-driven development pays for a fresh implementer +and a fresh reviewer on every task, each re-reading the codebase from zero. +Inline execution pays for one context (yours) plus one reviewer at the end. +What it gives up is a fresh context per task and a second pair of eyes per +task. This skill keeps what those two things bought, by other means: the +brief is the spec, the ledger is your memory, TDD is the per-task gate, and +the final reviewer is the second pair of eyes. -**Announce at start:** "I'm using the executing-plans skill to implement this plan." +**Core principle:** The plan already did the thinking. Execute it exactly, +prove each step with a test you watched fail and then pass, and leave a +record that survives your own forgetting. -**Note:** Tell your human partner that Superpowers works much better with access to subagents (Claude Code, Codex CLI, Codex App, Copilot CLI, and Gemini CLI all qualify; see the per-platform tool refs in `../using-superpowers/references/`). If subagents are available, use superpowers:subagent-driven-development instead of this skill. +**Narration:** between tool calls, narrate at most one short line — the +ledger and the tool results carry the record. + +**Continuous execution:** Do not pause to check in with your human partner +between tasks. They chose inline execution to spend less, not to answer +"should I continue?" after every task. Execute all tasks from the plan +without stopping. + +**Rulings, not stalls.** Conflicts, ambiguities, plan defects — decide them. +The spec is the binding authority, the plan is its argument, and your +judgment settles what neither answers. Record every decision in the ledger +as `Ruling: — — `, and keep +going. Deviating from the plan without a ledgered ruling is a decision made +in secret. + +Four things stop you, and only these: an irreversible or destructive +operation; a security-sensitive action; a side effect outside this worktree +that norms say you ask about first (a merge, a push to a shared branch, a +publish); and a plan so broken that every path forward is a guess. For +those, stop and ask. + +## When to Use + +- You have a plan from superpowers:writing-plans and your human partner + chose inline execution at the handoff. +- Your harness has no subagent tool (see the per-platform references in + `../using-superpowers/references/`). Never fabricate a dispatch; run + the plan here. +- Tasks are mostly independent — the same precondition as + superpowers:subagent-driven-development. + +A fully specified plan makes inline execution transcription plus testing: +it runs well on a mid-tier session model, and the one place the most +capable model earns its cost is the final review, which this skill +dispatches separately. Tell your human partner so when they choose inline. + +Prefer superpowers:subagent-driven-development when your human partner +wants a review gate on every task, or when the plan is long enough that +its later tasks would run on a compacted context. Inline execution over a +long plan still works — the ledger is what makes it recoverable — but the +last tasks get the least of you. ## The Process -### Step 1: Load and Review Plan -1. Ensure an isolated workspace: use superpowers:using-git-worktrees to create one or verify the existing one -2. Read plan file -3. Review critically - identify any questions or concerns about the plan -4. If concerns: Raise them with your human partner before starting -5. If no concerns: Create todos for the plan items and proceed +```dot +digraph process { + rankdir=TB; -### Step 2: Execute Tasks + subgraph cluster_per_task { + label="Per Task"; + "task-start: brief + BASE; read the brief" [shape=box]; + "Work the steps in order: TDD, run every verification, read every output" [shape=box]; + "Step output matches plan's Expected?" [shape=diamond]; + "Plan wrong? Rule and ledger. Code wrong? systematic-debugging" [shape=box]; + "Commit as the plan's commit steps say" [shape=box]; + "Completion contract met?" [shape=diamond]; + "task-done: run tests, ledger the result; mark todo complete" [shape=box]; + } -For each task: -1. Mark as in_progress -2. Follow each step exactly (plan has bite-sized steps) -3. Run verifications as specified -4. Mark as completed + "Setup: worktree, workspace + ledger, read plan + spec, pre-flight scan" [shape=box]; + "More tasks remain?" [shape=diamond]; + "Final whole-branch review (fresh reviewer if you have one)" [shape=box]; + "Re-grade, then: Critical/Important → ONE fix pass, each fix RED→GREEN + green suite; Minor → ledger" [shape=box]; + "Final review clean: delete this plan's workspace" [shape=box]; + "Use superpowers:finishing-a-development-branch" [shape=box style=filled fillcolor=lightgreen]; -### Step 3: Complete Development + "Setup: worktree, workspace + ledger, read plan + spec, pre-flight scan" -> "task-start: brief + BASE; read the brief"; + "task-start: brief + BASE; read the brief" -> "Work the steps in order: TDD, run every verification, read every output"; + "Work the steps in order: TDD, run every verification, read every output" -> "Step output matches plan's Expected?"; + "Step output matches plan's Expected?" -> "Plan wrong? Rule and ledger. Code wrong? systematic-debugging" [label="no"]; + "Plan wrong? Rule and ledger. Code wrong? systematic-debugging" -> "Work the steps in order: TDD, run every verification, read every output"; + "Step output matches plan's Expected?" -> "Commit as the plan's commit steps say" [label="yes, last step"]; + "Commit as the plan's commit steps say" -> "Completion contract met?"; + "Completion contract met?" -> "Work the steps in order: TDD, run every verification, read every output" [label="no - finish the task"]; + "Completion contract met?" -> "task-done: run tests, ledger the result; mark todo complete" [label="yes"]; + "task-done: run tests, ledger the result; mark todo complete" -> "More tasks remain?"; + "More tasks remain?" -> "task-start: brief + BASE; read the brief" [label="yes"]; + "More tasks remain?" -> "Final whole-branch review (fresh reviewer if you have one)" [label="no"]; + "Final whole-branch review (fresh reviewer if you have one)" -> "Re-grade, then: Critical/Important → ONE fix pass, each fix RED→GREEN + green suite; Minor → ledger"; + "Re-grade, then: Critical/Important → ONE fix pass, each fix RED→GREEN + green suite; Minor → ledger" -> "Final review clean: delete this plan's workspace"; + "Final review clean: delete this plan's workspace" -> "Use superpowers:finishing-a-development-branch"; +} +``` -After all tasks complete and verified: -- Announce: "I'm using the finishing-a-development-branch skill to complete this work." -- **REQUIRED SUB-SKILL:** Use superpowers:finishing-a-development-branch -- Follow that skill to verify tests, present options, execute choice +## Setup -## When to Stop and Ask for Help +Ensure the work happens in an isolated workspace: use +superpowers:using-git-worktrees to create one or verify the existing one. +Never start implementation on a main/master branch without your human +partner's explicit consent. -**STOP executing immediately when:** -- Hit a blocker (missing dependency, test fails, instruction unclear) -- Plan has critical gaps preventing starting -- You don't understand an instruction -- Verification fails repeatedly +Conversation memory does not survive compaction. An inline executor that +loses its place re-implements tasks whose commits already exist — the same +failure as a controller re-dispatching them, paid for in your own context. +Track progress in a ledger file, not only in todos. Harness todos are a +live view; the ledger is the record. -**Ask for clarification rather than guessing.** +The workspace and ledger are shared with superpowers:subagent-driven-development +— same directory, same format — so a plan can change executors mid-flight +and the new one resumes from the same ledger. -## When to Revisit Earlier Steps +- Each plan owns a workspace: at skill start, run + `../subagent-driven-development/scripts/sdd-workspace PLAN_FILE` — it + prints the plan's git-ignored directory + (`/.superpowers/sdd//`), home to every + artifact for THIS plan: ledger, briefs, review packages. Another plan's + directory is never yours to read or write. +- Check for this plan's ledger at `/progress.md`. If its first + line names your plan file, tasks with a `Task : complete` line are + DONE — do not redo them; resume at the first task without one. Their + commits exist in git even when your context no longer remembers making + them: after compaction, trust the ledger and `git log` over your own + recollection. A ledger whose first line names a different plan file is + another plan's progress: leave it and start your own, fresh. +- Create the ledger with its identity as the first line: + `# SDD ledger — plan: `. +- `git clean -fdx` will destroy the workspace (it's git-ignored scratch); + if that happens, recover from `git log`. -**Return to Review (Step 1) when:** -- Partner updates the plan based on your feedback -- Fundamental approach needs rethinking +Read the plan once, note its context and Global Constraints, and create a +todo per task. If the plan names a Spec, read that too: the spec is the +authority the plan argues from, and conflicts inside the plan resolve +against it. A plan with no reachable spec gets a ledger note saying so — +rulings made without one are provisional. -**Don't force through blockers** - stop and ask. +**REQUIRED SUB-SKILL:** load superpowers:test-driven-development now, +before Task 1. It governs every step of every task below; a plan whose +steps already say "write the failing test first" does not exempt you +from reading it. -## Remember -- Review plan critically first -- Follow plan steps exactly -- Don't skip verifications -- Reference skills when plan says to -- Stop when blocked, don't guess -- Never start implementation on main/master branch without explicit user consent +Before Task 1, scan the plan for conflicts between tasks. The plan's +Interfaces blocks tell you where to look: for every task that consumes +what an earlier task produces, one ledger row — the two tasks, what one +produces against what the other consumes, and what you found. Tasks that +share nothing get no row; a plan whose tasks share nothing gets the single +line `Pre-flight: no shared interfaces`. Rule on each conflict a row +surfaces with the spec as the binding authority, record the ruling beside +its row, and start Task 1. Each task's own text is checked when you read +its brief, not here. + +## The Task Loop + +Everything you print, and every tool result, stays resident in your +context for the rest of the session. Redirect long test output to a file +in the workspace and read its tail; read a brief, not the whole plan. + +### 1. Take the task + +- Run this skill's `scripts/task-start PLAN_FILE N`. It prints the brief + path and BASE (the commit the task's range is cut from) in one call. + Read the brief for every task, including ones you remember from setup: + what you remember is a summary, the brief has the exact values, + signatures, and test cases. +- Mark the task's todo in_progress. + +Every tool call is a turn that re-reads your whole context. Bookkeeping +rides along with work — a ledger append in the same call as the commit, +never in a call of its own. + +### 2. Work the steps + +The plan's steps are already in RED-GREEN order; follow them in that +order under superpowers:test-driven-development, loaded at setup. A test +step's code is written first and run first. Watching it fail is a step, +not a formality — a test that passes before the implementation exists is +a finding about the test. + +Every step that runs a command has an `Expected:` line. Run the command, +read its output, and compare. Three outcomes: + +- **Matches.** Next step. +- **The code is wrong.** Use superpowers:systematic-debugging. Find the + cause; never patch the symptom to make the step's output match. +- **The plan is wrong** — a step contradicts the spec, an interface from an + earlier task doesn't match what this task consumes, a command that + cannot work. Rule on the smallest change that satisfies the spec, ledger + it as `Task : Ruling: — `, and + continue. The ruling is carried, not remembered: later tasks that touch + the same interface read it from the ledger. + +Commit as the plan's commit steps say. A task that spans several commits +is fine; BASE is what the review range is cut from, never `HEAD~1`. + +### 3. The completion contract + +Before a task's ledger line, all of the following are true, with evidence +in this session — not inferred from the diff looking right: + +- Every test the brief names exists and ran in this task, and you read + the output. +- The final test run for the task passed — `task-done` is that run, and + it writes the command and result into the ledger line. +- Every `Expected:` line in the brief was compared against real output. +- Every deviation from the brief has a `Ruling:` line in the ledger. + +**REQUIRED SUB-SKILL:** superpowers:verification-before-completion governs +the claim. If any item is missing, the task is not complete: finish it. + +### 4. Complete the task + +Run this skill's `scripts/task-done PLAN_FILE N BASE -- ` +with the test command the brief names for the whole task. It runs the +tests, keeps the full output in the workspace, prints the tail, and — only +if they pass — appends the completion line to the ledger: + +`Task : complete (commits .., tests: → )` + +A failing run records nothing; the task is not complete. When it records, +mark the todo complete and take the next task. + +## Final Review + +Run `../subagent-driven-development/scripts/review-package PLAN_FILE MERGE_BASE HEAD` +(MERGE_BASE = the commit the branch started from, e.g. +`git merge-base main HEAD`) and review from the file it prints. + +**With a subagent tool:** dispatch the reviewer on the most capable +available model — the whole-branch review is a judgment task — using +superpowers:requesting-code-review's +[code-reviewer.md](../requesting-code-review/code-reviewer.md), with the +package path, the plan and spec paths, the plan's Review Focus section +verbatim if it has one (the input classes and failure modes the plan's +tests do not exercise — the reviewer checks each deliberately), and a +pointer to the ledger's `Ruling:` lines so it can weigh the calls you +made. Specify the model +explicitly; an omitted model inherits the session's, which may not be the +most capable. This is the one fresh context the whole run buys. Do not +skip it, and do not replace it with your own read of the diff. + +**Without a subagent tool:** read code-reviewer.md and perform that review +yourself against the package, as a separate pass after the last task's +ledger line. Write `Final review: self-review (no subagent tool)` to the +ledger, and say so in your final message: a self-review by the author is +weaker than a fresh reviewer, and your human partner decides whether that +is enough before merge. + +Sort the findings before you act on any of them. The reviewer's severity +labels are advice; the gate is yours. Its "Declined to judge" list is +yours too: every line there is a ruling you make and ledger, exactly like +a plan conflict — `Final: Ruling: — + — `. Re-grade first, by effect: the +spec is a vision document, and a finding's grade is what a reasonable +person using this software gets if it ships, not whether the spec names +the input that triggers it — a reviewer who set a finding at Minor +because the spec was silent has graded the spec, not the effect. Then: + +- **Critical and Important** enter the fix pass. +- **Minor** goes to the ledger as `Final: minor (deferred): ` + and to your final message under "Deferred minors". Minors never enter + the fix pass, and never become rulings — a ruling is a decision about a + conflict, not a note that you declined a polish suggestion. + +Fix the Critical and Important findings yourself — you are the +implementer here — in ONE pass. Each fix is verified by TDD, not by a +second reviewer: write the test that reproduces the finding, watch it +fail, make it pass, then run the whole suite. Record each in the ledger as +`Final: fixed — RED→GREEN, suite /`. A fix +without a test that failed first is not verified; a suite that is not +green after the pass means the pass is not over. Do not dispatch a +re-review: it would re-read a diff whose covering tests already answer +"addressed" and whose suite run already answers "broke nothing". + +A finding you decide not to fix is a ruling — `Final: Ruling: — + — ` — and reaches your human partner +in the rulings list. There is no second fix pass. + +## Finish + +Before you delete anything, collect every ledger line containing +`Ruling:` into your final message under "Rulings I made", in the order you +made them, each with what it costs if wrong, and every `minor (deferred)` +line under "Deferred minors". Both lists are exhaustive. Your final +message is the only place the decisions you took on your human partner's +behalf — and the findings you chose not to act on — reach them. + +When the final review is clean and its fixes are committed, delete this +plan's workspace directory — the git history is the record now. Sibling +directories belong to other plans; leave them alone. + +Use superpowers:finishing-a-development-branch. + +## Common Rationalizations + +| Excuse | Reality | +|--------|---------| +| "I remember what Task N says" | You remember a summary. The brief has the exact values. Read it. | +| "The plan's code is right, skip watching the test fail" | A test you never saw fail proves nothing. It is one step. Run it. | +| "I'll run the full suite at the end instead of per step" | Per-step runs are how you learn which step broke it. The end-of-task run is the contract, not a substitute. | +| "The plan is wrong here, I'll just do the right thing" | Do the right thing and ledger the ruling. Unledgered deviation is a decision made in secret. | +| "I'll write the ledger lines after a few tasks" | Compaction does not wait for a convenient moment. One line per task, in the same message as the commit. | +| "Let me check in before the next task" | They chose inline to spend less. Progress prompts spend their time instead. Only the four stops stop you. | +| "I read my own diff carefully; the final reviewer is redundant" | Same author, same blind spots. The reviewer is the only fresh context this run buys. | +| "Tests should pass, the change was trivial" | "Should" is not evidence. The contract requires the command and its output. | +| "Subagents are slow and expensive, I'll skip the final review too" | Inline already removed the per-task reviewers. One review of the whole branch is the floor, not the ceiling. | +| "The reviewer said Minor, so it's Minor" | The label graded the spec's silence. Grade what the person gets. Re-grade, then gate. | +| "The fix is obvious, no need for a failing test first" | The failing test is the only proof the finding was real and is now gone. Without it you have a diff and a hope. | +| "I'll fix the minors too while I'm in there" | Every minor you fix is a test, a fix, and a suite run your partner did not ask for. Ledger them; your partner decides. | + +## Example Workflow + +``` +You: I'm using the executing-plans skill to implement this plan inline. + +[Setup: worktree verified] +[Read plan once: docs/superpowers/plans/feature-plan.md; spec read] +[Resolve workspace: sdd-workspace docs/superpowers/plans/feature-plan.md — no ledger inside, fresh start] +[Pre-flight scan: 2 shared-interface rows, 4 self-consistency rows, clean; written to ledger] +[Create todos for all tasks] + +Task 1: Hook installation script + +[task-start plan 1 → brief read; BASE a1b2c3d] +[Step 1: write failing test — written] +[Step 2: run it — FAIL: install_hook not defined. Matches Expected.] +[Step 3: implement — written] +[Step 4: run it — PASS 1/1. Matches Expected.] +[Step 5: commit — d4e5f6a] +[Contract: tests ran, output read, no deviations] +[task-done plan 1 a1b2c3d -- npm test -- hooks → ledger: Task 1: complete (commits a1b2c3d..d4e5f6a, tests: npm test -- hooks → 1/1 pass)] + +Task 2: Recovery modes + +[task-start plan 2 → brief read; BASE d4e5f6a] +[Step 2: run failing test — FAIL, but on an import error: Task 1 exported + installHook, brief consumes install_hook] +[Ruling: brief's consumer name is a typo against Task 1's Produces block; + use installHook — Ledger: Task 2: Ruling: install_hook → installHook — matches Task 1 Produces — cost if wrong: one rename] +[Steps 2-5 as planned; commit b7c8d9e] +[task-done plan 2 d4e5f6a -- npm test -- recovery → ledger: Task 2: complete (commits d4e5f6a..b7c8d9e, tests: npm test -- recovery → 8/8 pass)] + +... + +[After all tasks: review-package plan MERGE_BASE HEAD; dispatch code-reviewer, most capable model] +Reviewer: One Important finding — progress reporting interval hardcoded. Two Minor. +[Re-grade: Important stands; minors → ledger as deferred] +[Fix pass: test_progress_interval_configurable RED → extract PROGRESS_INTERVAL → GREEN; suite 12/12; commit] +[Ledger: Final: fixed hardcoded interval — test_progress_interval_configurable RED→GREEN, suite 12/12] + +Rulings I made: +- Task 2: install_hook → installHook (brief typo; cost if wrong: one rename) + +Deferred minors: +- README lacks a usage example +- recovery.js could split verify/repair into two files + +[Delete this plan's workspace — the record now lives in git] + +Using superpowers:finishing-a-development-branch. +``` diff --git a/skills/executing-plans/scripts/task-done b/skills/executing-plans/scripts/task-done new file mode 100755 index 000000000..dd09871b0 --- /dev/null +++ b/skills/executing-plans/scripts/task-done @@ -0,0 +1,52 @@ +#!/usr/bin/env bash +# Close one task of an inline plan execution in a single call: run the task's +# test command, keep its full output in the workspace, print the tail, and — +# only if the command succeeded — append the completion line to the ledger. +# A failing command records nothing: the task is not complete. +# +# Usage: task-done PLAN_FILE TASK_NUMBER BASE -- TEST_COMMAND [ARGS...] +# BASE is the SHA task-start printed; the completion line records BASE..HEAD. +# Exit: the test command's exit status. +set -euo pipefail + +if [ $# -lt 5 ] || [ "$4" != "--" ]; then + echo "usage: task-done PLAN_FILE TASK_NUMBER BASE -- TEST_COMMAND [ARGS...]" >&2 + exit 2 +fi + +plan=$1 +n=$2 +base=$3 +shift 4 +sdd="$(cd "$(dirname "$0")/../../subagent-driven-development/scripts" && pwd)" + +git rev-parse --verify --quiet "$base" >/dev/null || { echo "bad BASE: $base" >&2; exit 2; } + +dir=$("$sdd/sdd-workspace" "$plan") +log="$dir/task-${n}-tests.log" +ledger="$dir/progress.md" + +# Render the command the way a person would type it, for the ledger line. +cmd="" +for a in "$@"; do + case "$a" in + *[[:space:]\"\;\|\&]*) cmd="$cmd '$a'" ;; + *) cmd="$cmd $a" ;; + esac +done +cmd=${cmd# } + +rc=0 +"$@" > "$log" 2>&1 || rc=$? + +tail -n 5 "$log" +if [ "$rc" -ne 0 ]; then + echo "task-done: test command exited $rc; Task $n NOT recorded (full output: $log)" >&2 + exit "$rc" +fi + +last=$(grep -v '^[[:space:]]*$' "$log" | tail -n 1) +[ -f "$ledger" ] || printf '# SDD ledger — plan: %s\n' "$plan" > "$ledger" +line="Task $n: complete (commits $(git rev-parse --short=7 "$base")..$(git rev-parse --short=7 HEAD), tests: $cmd → $last)" +printf '%s\n' "$line" >> "$ledger" +echo "ledger: $line" diff --git a/skills/executing-plans/scripts/task-start b/skills/executing-plans/scripts/task-start new file mode 100755 index 000000000..fab75b55b --- /dev/null +++ b/skills/executing-plans/scripts/task-start @@ -0,0 +1,28 @@ +#!/usr/bin/env bash +# Begin one task of an inline plan execution in a single call: extract the +# task's brief (via subagent-driven-development's task-brief, so both skills +# share one workspace) and record BASE, the commit the task's review range is +# cut from. One tool call instead of two, because every call in an inline +# session is a turn that re-reads the whole context. +# +# Usage: task-start PLAN_FILE TASK_NUMBER +# Prints: +# brief: +# base: +set -euo pipefail + +if [ $# -ne 2 ]; then + echo "usage: task-start PLAN_FILE TASK_NUMBER" >&2 + exit 2 +fi + +plan=$1 +n=$2 +sdd="$(cd "$(dirname "$0")/../../subagent-driven-development/scripts" && pwd)" + +out=$("$sdd/task-brief" "$plan" "$n") +brief=$(printf '%s\n' "$out" | sed -n 's/^wrote \(.*\): [0-9][0-9]* lines$/\1/p') +[ -n "$brief" ] || { echo "task-brief did not report a path: $out" >&2; exit 1; } + +echo "brief: $brief" +echo "base: $(git rev-parse HEAD)" diff --git a/skills/requesting-code-review/SKILL.md b/skills/requesting-code-review/SKILL.md index fa4f2f996..6d995da5c 100644 --- a/skills/requesting-code-review/SKILL.md +++ b/skills/requesting-code-review/SKILL.md @@ -25,7 +25,7 @@ Dispatch a code reviewer subagent to catch issues before they cascade. The revie **1. Get git SHAs:** ```bash -BASE_SHA=$(git rev-parse HEAD~1) # or origin/main +BASE_SHA=$(git rev-parse HEAD~1) # or: git merge-base origin/main HEAD HEAD_SHA=$(git rev-parse HEAD) ``` diff --git a/skills/requesting-code-review/code-reviewer.md b/skills/requesting-code-review/code-reviewer.md index b898cb982..6d358b5d7 100644 --- a/skills/requesting-code-review/code-reviewer.md +++ b/skills/requesting-code-review/code-reviewer.md @@ -30,6 +30,23 @@ Subagent (general-purpose): git diff [BASE_SHA]..[HEAD_SHA] ``` + ## The spec is a vision document + + The spec says what the software must do. It does not enumerate every + input, environment, or condition the software will meet. For behavior + the spec is silent on, judge by what a reasonable person using this + software would expect: a reasonable person's expectation is a + requirement, and a spec's silence is not permission. Grade such + findings by their effect on that person, not by whether the spec + mentions the trigger. + + ## Declined to judge + + Before your verdict, list every behavior you considered and set aside + as outside the plan or spec, one line each, with the reason. The + executor rules on each line; nothing you set aside is dropped + silently. An empty list means you set nothing aside. + ## Read-Only Review Your review is read-only on this checkout. Do not mutate the working tree, the index, HEAD, or branch state in any way. Use tools like `git show`, `git diff`, and `git log` to inspect history. If you need a working copy of a different revision, check it out into a separate temporary directory (e.g. `git worktree add /tmp/review-[SHA] [SHA]`) — never move HEAD on this checkout. diff --git a/skills/subagent-driven-development/SKILL.md b/skills/subagent-driven-development/SKILL.md index 15aa5ef15..f7bbf6a01 100644 --- a/skills/subagent-driven-development/SKILL.md +++ b/skills/subagent-driven-development/SKILL.md @@ -36,25 +36,25 @@ stop and ask. digraph when_to_use { "Have implementation plan?" [shape=diamond]; "Tasks mostly independent?" [shape=diamond]; - "Stay in this session?" [shape=diamond]; + "Partner chose inline, or no subagent tool?" [shape=diamond]; "subagent-driven-development" [shape=box]; "executing-plans" [shape=box]; "Manual execution or brainstorm first" [shape=box]; "Have implementation plan?" -> "Tasks mostly independent?" [label="yes"]; "Have implementation plan?" -> "Manual execution or brainstorm first" [label="no"]; - "Tasks mostly independent?" -> "Stay in this session?" [label="yes"]; + "Tasks mostly independent?" -> "Partner chose inline, or no subagent tool?" [label="yes"]; "Tasks mostly independent?" -> "Manual execution or brainstorm first" [label="no - tightly coupled"]; - "Stay in this session?" -> "subagent-driven-development" [label="yes"]; - "Stay in this session?" -> "executing-plans" [label="no - parallel session"]; + "Partner chose inline, or no subagent tool?" -> "executing-plans" [label="yes"]; + "Partner chose inline, or no subagent tool?" -> "subagent-driven-development" [label="no"]; } ``` -**vs. Executing Plans (parallel session):** -- Same session (no context switch) -- Fresh subagent per task (no context pollution) -- Review after each task (spec compliance + code quality), broad review at the end -- Faster iteration (no human-in-loop between tasks) +**vs. Executing Plans (inline):** +- Fresh subagent per task (no context pollution) instead of one context doing every task +- Review after each task (spec compliance + code quality) instead of only at the end +- Costs a fresh context per task and per review; inline costs one context plus one final reviewer +- Both run in this session, share the same plan workspace and ledger, and never pause between tasks ## The Process @@ -135,7 +135,7 @@ a ledger file, not only in todos. - Each plan owns a workspace: at skill start, run this skill's `bash scripts/sdd-workspace PLAN_FILE` — it prints the plan's git-ignored - directory (`/.superpowers/sdd//`), home to + directory (under `/.superpowers/sdd/`), home to every artifact for THIS plan: ledger, briefs, reports, review packages. Another plan's directory is never yours to read or write. - Check for this plan's ledger at `/progress.md`. If its first diff --git a/skills/subagent-driven-development/scripts/review-package b/skills/subagent-driven-development/scripts/review-package index 31852e2ab..fa7625f05 100755 --- a/skills/subagent-driven-development/scripts/review-package +++ b/skills/subagent-driven-development/scripts/review-package @@ -22,10 +22,17 @@ head=$3 git rev-parse --verify --quiet "$base" >/dev/null || { echo "bad BASE: $base" >&2; exit 2; } git rev-parse --verify --quiet "$head" >/dev/null || { echo "bad HEAD: $head" >&2; exit 2; } +# Range guards (exit 3): a wrong-branch HEAD yields a range that is empty or +# not rooted at BASE; either would silently produce a bogus review package. +git merge-base --is-ancestor "$base" "$head" || { echo "HEAD is not a descendant of BASE: ${base}..${head}" >&2; exit 3; } +[ "$(git rev-list --count "${base}..${head}")" -gt 0 ] || { echo "empty commit range: ${base}..${head}" >&2; exit 3; } + if [ $# -eq 4 ]; then out=$4 else - dir=$("$(cd "$(dirname "$0")" && pwd)/sdd-workspace" "$plan") + # Invoke via bash rather than direct exec: some extractors (Python zipfile) + # strip Unix exec bits when unpacking marketplace packages (#2040). + dir=$("${BASH:-bash}" "$(cd "$(dirname "$0")" && pwd)/sdd-workspace" "$plan") out="$dir/review-$(git rev-parse --short "$base")..$(git rev-parse --short "$head").diff" fi diff --git a/skills/subagent-driven-development/scripts/sdd-workspace b/skills/subagent-driven-development/scripts/sdd-workspace index 4e2d16802..ff6b9839a 100755 --- a/skills/subagent-driven-development/scripts/sdd-workspace +++ b/skills/subagent-driven-development/scripts/sdd-workspace @@ -8,6 +8,16 @@ # artifacts. A stale ledger misread as current progress makes controllers # skip whole task sequences — plan-scoping removes that failure structurally. # +# Basename slugs collide when two plans share a filename (docs/alpha/plan.md +# vs docs/beta/plan.md), so each workspace records its owning plan's path in +# a plan-path marker (repo-relative in-repo, absolute outside). A workspace +# owned by a different plan is skipped and the slug disambiguated with the +# plan's parent-directory name, then a counter. A workspace with no marker +# predates the marker scheme and is adopted for the current plan so in-flight +# workspaces keep resolving — which means the first collision on such a +# legacy workspace adopts instead of detecting; acceptable, marker-less +# workspaces age out as plans finish. +# # The workspace lives in the working tree (not under .git/) because Claude Code # treats .git/ as a protected path and denies agent writes there — which blocks # an implementer subagent from writing its report file. A self-ignoring @@ -34,7 +44,39 @@ slug=$(basename "$plan" .md) root=$(git rev-parse --show-toplevel) base="$root/.superpowers/sdd" + +# Normalize the plan path (physical directory, so relative/absolute/../ +# spellings of one plan compare equal) and express it as the marker value: +# repo-relative when the plan lives under the repo root, absolute otherwise. +plan_dir=$(CDPATH= cd -- "$(dirname "$plan")" && pwd -P) +plan_abs="$plan_dir/$(basename "$plan")" +case "$plan_abs" in + "$root"/*) plan_id=${plan_abs#"$root"/} ;; + *) plan_id=$plan_abs ;; +esac + +# True when the workspace at $1 is (or becomes) this plan's: an existing +# marker must name this plan; a missing marker means a new workspace or a +# pre-marker legacy one, and either way the plan claims it by writing one. +owns() { + if [ -e "$1/plan-path" ]; then + [ "$(cat "$1/plan-path")" = "$plan_id" ] + else + mkdir -p "$1" + printf '%s\n' "$plan_id" > "$1/plan-path" + fi +} + dir="$base/$slug" -mkdir -p "$dir" +if ! owns "$dir"; then + parent=$(basename "$plan_dir") + dir="$base/$slug-$parent" + if ! owns "$dir"; then + n=2 + while ! owns "$base/$slug-$parent-$n"; do n=$((n + 1)); done + dir="$base/$slug-$parent-$n" + fi +fi + printf '*\n' > "$base/.gitignore" -cd "$dir" && pwd +CDPATH= cd -- "$dir" && pwd diff --git a/skills/subagent-driven-development/scripts/task-brief b/skills/subagent-driven-development/scripts/task-brief index 612e14a1e..b49fc546c 100755 --- a/skills/subagent-driven-development/scripts/task-brief +++ b/skills/subagent-driven-development/scripts/task-brief @@ -21,7 +21,9 @@ n=$2 if [ $# -eq 3 ]; then out=$3 else - dir=$("$(cd "$(dirname "$0")" && pwd)/sdd-workspace" "$plan") + # Invoke via bash rather than direct exec: some extractors (Python zipfile) + # strip Unix exec bits when unpacking marketplace packages (#2040). + dir=$("${BASH:-bash}" "$(cd "$(dirname "$0")" && pwd)/sdd-workspace" "$plan") out="$dir/task-${n}-brief.md" fi diff --git a/skills/test-driven-development/SKILL.md b/skills/test-driven-development/SKILL.md index 4320d8879..46838cc9e 100644 --- a/skills/test-driven-development/SKILL.md +++ b/skills/test-driven-development/SKILL.md @@ -182,6 +182,16 @@ Confirm: **Other tests fail?** Fix now. +**"Other tests" means the project's suite, not just your file.** A +green run of the test you wrote is not a green suite. Before you call +the change done, run the project's test command (bare `pytest`, +`npm test`, `cargo test` — whatever the repo uses) even when your task +named only one test file. A scope statement in your task bounds the +deliverable, not your verification. Any failure that run shows — +including one you didn't cause — goes in your report by name; a red +test you watched scroll past and didn't mention is a report falsified +by omission. + ### REFACTOR - Clean Up After green only: diff --git a/skills/using-superpowers/SKILL.md b/skills/using-superpowers/SKILL.md index 7ab2eb678..069d57844 100644 --- a/skills/using-superpowers/SKILL.md +++ b/skills/using-superpowers/SKILL.md @@ -53,10 +53,12 @@ These thoughts mean STOP—you're rationalizing: If your harness appears here, read its reference file for special instructions: +- Claude Code: `references/claude-code-tools.md` - Codex: `references/codex-tools.md` - Pi: `references/pi-tools.md` - Antigravity: `references/antigravity-tools.md` - Hermes Agent: `references/hermes-tools.md` +- Muse: `references/muse-tools.md` ## User Instructions diff --git a/skills/using-superpowers/references/claude-code-tools.md b/skills/using-superpowers/references/claude-code-tools.md new file mode 100644 index 000000000..550b8e456 --- /dev/null +++ b/skills/using-superpowers/references/claude-code-tools.md @@ -0,0 +1,29 @@ +# Claude Code Tool Notes + +Claude Code is the reference harness: skills speak its vocabulary +(`Agent` for a subagent dispatch, todos, `Skill`). These notes cover the +one place Claude Code can run a plan cheaper than the skills' default +shape. It is opt-in by your human partner and changes nothing the skills +require. + +## Cheaper orchestration for subagent-driven development + +The controller session is the most expensive seat in a +superpowers:subagent-driven-development run: it reads every dispatch +result and every report, and it usually runs on the session's most +capable model. Claude Code supports nested subagents (three layers below +the main conversation by default; `CLAUDE_CODE_MAX_SUBAGENT_SPAWN_DEPTH` +adjusts it), so the whole loop can run one layer down. + +When your human partner asks for it — or has said the session model is +too expensive to spend on coordination — dispatch ONE orchestrator +subagent on a mid-tier model with the plan path and the instruction to +use superpowers:subagent-driven-development end to end. The orchestrator +dispatches its own implementers and reviewers per that skill's Model +Selection; the workspace and ledger live on disk, so nothing is lost to +the extra layer. Its final message must carry the "Rulings I made" list +verbatim — that list is how the decisions reach your human partner, and +you relay it, not summarize it. + +Do this only for a whole plan. Nesting a single task's dispatch buys +nothing and adds a seat. diff --git a/skills/using-superpowers/references/muse-tools.md b/skills/using-superpowers/references/muse-tools.md new file mode 100644 index 000000000..a87d6d838 --- /dev/null +++ b/skills/using-superpowers/references/muse-tools.md @@ -0,0 +1,35 @@ +# Muse Tool Mapping + +Skills speak in actions ("dispatch a subagent", "create a todo", "read a file"). On Muse these resolve to the tools below. + +| Action skills request | Muse equivalent | +|----------------------|----------------| +| Read a file | `read_file` | +| Read multiple files | `read_file` (call multiple times) or `search` | +| Create a new file | `write_file` | +| Edit a file | `edit_file` | +| Run a shell command | `bash` | +| Search file contents | `search` | +| Find files by name | `search` with `glob` | +| Fetch a URL | `web_fetch` | +| Search the web | `web_search` | +| Invoke a skill | `read_file` on `skills//SKILL.md` or native skill tool | +| Dispatch a subagent (`Subagent (general-purpose):` template) | `subagent_spawn` with prompt filling | +| Task tracking ("create a todo", "mark complete") | `write_todos` or `bash` task file | +| Ask the user a question | `request_user_input` | + +## Instructions file + +When a skill mentions "your instructions file", on Muse this is **`CLAUDE.md`** or **`AGENTS.md`** in the project root. Muse loads these hierarchically where configured. + +## Skill invocation + +Muse has native skill support via `muse skills`. To invoke a Superpowers skill, read its `SKILL.md` and follow the instructions. The bootstrap (`using-superpowers`) is injected automatically at `SessionStart` via the plugin hook — you are already following it, do not re-load it. + +## Subagent dispatch + +Use `subagent_spawn` to delegate work to isolated subagents. Fill prompt templates (e.g., `implementer-prompt.md`, `task-reviewer-prompt.md`) before dispatching. If no subagent tool is available, do the work inline rather than inventing tool calls. + +## Task tracking + +Use `write_todos` for checklist tracking. Create one todo per skill checklist item, mark in_progress/completed as you go. If `write_todos` is unavailable, maintain a markdown task file via `write_file`/`edit_file`. diff --git a/skills/writing-plans/SKILL.md b/skills/writing-plans/SKILL.md index b2304cd17..4cf0275d8 100644 --- a/skills/writing-plans/SKILL.md +++ b/skills/writing-plans/SKILL.md @@ -7,9 +7,7 @@ description: Use when you have a spec or requirements for a multi-step task, bef ## Overview -Write comprehensive implementation plans assuming the engineer has zero context for our codebase and questionable taste. Document everything they need to know: which files to touch for each task, code, testing, docs they might need to check, how to test it. Give them the whole plan as bite-sized tasks. DRY. YAGNI. TDD. Frequent commits. - -Assume they are a skilled developer, but know almost nothing about our toolset or problem domain. Assume they don't know good test design very well. +Write implementation plans for an engineer who has not seen this codebase or this spec. Assume they write idiomatic code in the project's language once they know the exact interface and the exact test, and that they will make a reasonable choice wherever the plan leaves one open. What they cannot know is what you decided: which files, which names and signatures, which values from the spec, which tests prove each task. Document those. Give them the whole plan as bite-sized tasks. DRY. YAGNI. TDD. Frequent commits. **Announce at start:** "I'm using the writing-plans skill to create the implementation plan." @@ -42,9 +40,9 @@ deliverable needs them; split only where a reviewer could meaningfully reject one task while approving its neighbor. Each task ends with an independently testable deliverable. -## Bite-Sized Task Granularity +## Step Granularity -**Each step is one action (2-5 minutes):** +**Each step is one action with a checkable result:** - "Write the failing test" - step - "Run it to make sure it fails" - step - "Implement the minimal code to make the test pass" - step @@ -76,6 +74,18 @@ naming and copy rules, platform requirements — one line each, with exact values copied verbatim from the spec. Every task's requirements implicitly include this section.] +## Review Focus + +[The five input classes or failure modes the spec implies but no task's +tests exercise that are most likely to bite a person using this software +— one line each, naming the input or condition and the behavior a +reasonable person would expect, most likely first. The spec is a vision +document: it says what the software must do, not everything it will +meet, and its silence on an input is not permission for that input to +break the program. Write the list here, once, with the spec in front of +you. Then, for each line, add the test that pins it to the task that +owns the code, in that task's own step style.] + --- ``` @@ -108,12 +118,11 @@ def test_specific_behavior(): Run: `pytest tests/path/test.py::test_name -v` Expected: FAIL with "function not defined" -- [ ] **Step 3: Write minimal implementation** +- [ ] **Step 3: Implement `function(input: InputType) -> ResultType` in `exact/path/to/file.py`** -```python -def function(input): - return expected -``` +One line on the approach when the signature and the test leave a choice +(which library call, which data structure); a code block only for an +algorithm they do not determine. - [ ] **Step 4: Run test to verify it passes** @@ -128,15 +137,28 @@ git commit -m "feat: add specific feature" ``` ```` -## No Placeholders +## What a Step Contains -Every step must contain the actual content an engineer needs. These are **plan failures** — never write them: -- "TBD", "TODO", "implement later", "fill in details" -- "Add appropriate error handling" / "add validation" / "handle edge cases" -- "Write tests for the above" (without actual test code) -- "Similar to Task N" (repeat the code — the engineer may be reading tasks out of order) -- Steps that describe what to do without showing how (code blocks required for code steps) -- References to types, functions, or methods not defined in any task +A step is done when the implementer can write exactly one reasonable thing +from it. That is the whole requirement: unambiguous, not complete. Each kind +of step carries what makes it unambiguous and nothing more: + +- **A test step:** the test's name and its assertions, as code, with the + spec's exact values in them. +- **A code step:** the exact signature (name, parameters, return type), the + file it lives in, and the specific values the spec pins. The implementer + writes the body. A body appears only for an algorithm the signature and + tests do not determine, or for exact copy the spec fixes. +- **A verification step:** the command to run and the output that means it + passed. +- **A reference to another task:** that task's Interfaces block says what + to use; the plan does not repeat that task's code. + +A plan is the set of decisions the implementer cannot make alone. A plan +longer than the code it describes has written the code instead. Lines that +decide nothing ("TBD", "handle edge cases", "add appropriate validation", +"write tests for the above", a type or function no task defines) are the +opposite failure, and the self-review catches both. ## Self-Review @@ -144,10 +166,14 @@ After writing the complete plan, look at the spec with fresh eyes and check the **1. Spec coverage:** Skim each section/requirement in the spec. Can you point to a task that implements it? List any gaps. -**2. Placeholder scan:** Search your plan for red flags — any of the patterns from the "No Placeholders" section above. Fix them. +**2. Step scan:** Every step must let the implementer write exactly one reasonable thing, and no step may carry more than that: a line that decides nothing is a gap, a function body the signature and tests already determine is a transcript. Fix both. **3. Type consistency:** Do the types, method signatures, and property names you used in later tasks match what you defined in earlier tasks? A function called `clearLayers()` in Task 3 but `clearFullLayers()` in Task 7 is a bug. +**4. Review Focus:** For each input class or failure mode the spec implies, is there a task whose tests exercise it? The five uncovered ones most likely to bite a person go in the Review Focus section, and each line there gets its test added to the owning task. An empty section means you checked and found none, not that you skipped the check. + +**5. Proportion:** Compare the plan's length to the spec's. A plan several times longer than the spec it implements is a transcript of the program, not a plan. If code blocks are most of the document, replace bodies with signatures, test names and assertions, and check that each step is still unambiguous. + If you find issues, fix them inline. No need to re-review — just fix and move on. If you find a spec requirement with no task, add the task. ## Execution Handoff @@ -160,22 +186,19 @@ them to review the plan and choose an execution method before implementation. **When no execution method has already been supplied:** -**"Plan complete and saved to `docs/superpowers/plans/.md`. Please review the plan. Two execution options:** +**"Plan complete and saved to `docs/superpowers/plans/.md`. Please review the plan. Which execution approach would you prefer?** -**1. Subagent-Driven (recommended)** - I dispatch a fresh subagent per task, review between tasks, fast iteration +- **Subagent-driven** - A fresh subagent implements each task and a fresh reviewer checks it before the next one starts, then a whole-branch review at the end. Most thorough; costs a fresh context per task and per review. +- **Native** - I implement every task myself in this session, the way this harness runs work, then one fresh reviewer on the most capable model checks the whole branch. Cheapest and fastest; no independent review until the end. Runs well with a mid-tier session model, since the plan carries the design. -**2. Inline Execution** - Execute tasks in this session using executing-plans, batch execution with checkpoints - -**Does the plan capture what you want, and which approach should we use?"** +**For this plan I recommend , because . Does the plan capture what you want, and which approach should we use?"** **When an execution method has already been supplied:** **"Plan complete and saved to `docs/superpowers/plans/.md`. Please review the plan. Does it capture what you want?"** -**If Subagent-Driven chosen:** +**If Subagent-driven chosen:** - **REQUIRED SUB-SKILL:** Use superpowers:subagent-driven-development -- Fresh subagent per task + two-stage review -**If Inline Execution chosen:** +**If Native chosen:** - **REQUIRED SUB-SKILL:** Use superpowers:executing-plans -- Batch execution with checkpoints for review diff --git a/skills/writing-plans/plan-document-reviewer-prompt.md b/skills/writing-plans/plan-document-reviewer-prompt.md deleted file mode 100644 index 1c12c1d61..000000000 --- a/skills/writing-plans/plan-document-reviewer-prompt.md +++ /dev/null @@ -1,49 +0,0 @@ -# Plan Document Reviewer Prompt Template - -Use this template when dispatching a plan document reviewer subagent. - -**Purpose:** Verify the plan is complete, matches the spec, and has proper task decomposition. - -**Dispatch after:** The complete plan is written. - -``` -Subagent (general-purpose): - description: "Review plan document" - prompt: | - You are a plan document reviewer. Verify this plan is complete and ready for implementation. - - **Plan to review:** [PLAN_FILE_PATH] - **Spec for reference:** [SPEC_FILE_PATH] - - ## What to Check - - | Category | What to Look For | - |----------|------------------| - | Completeness | TODOs, placeholders, incomplete tasks, missing steps | - | Spec Alignment | Plan covers spec requirements, no major scope creep | - | Task Decomposition | Tasks have clear boundaries, steps are actionable | - | Buildability | Could an engineer follow this plan without getting stuck? | - - ## Calibration - - **Only flag issues that would cause real problems during implementation.** - An implementer building the wrong thing or getting stuck is an issue. - Minor wording, stylistic preferences, and "nice to have" suggestions are not. - - Approve unless there are serious gaps — missing requirements from the spec, - contradictory steps, placeholder content, or tasks so vague they can't be acted on. - - ## Output Format - - ## Plan Review - - **Status:** Approved | Issues Found - - **Issues (if any):** - - [Task X, Step Y]: [specific issue] - [why it matters for implementation] - - **Recommendations (advisory, do not block approval):** - - [suggestions for improvement] -``` - -**Reviewer returns:** Status, Issues (if any), Recommendations diff --git a/tests/claude-code/run-skill-tests.sh b/tests/claude-code/run-skill-tests.sh index 83217cdad..97bce6d19 100755 --- a/tests/claude-code/run-skill-tests.sh +++ b/tests/claude-code/run-skill-tests.sh @@ -76,6 +76,7 @@ done tests=( "test-worktree-path-policy.sh" "test-sdd-workspace.sh" + "test-executing-plans-scripts.sh" "test-subagent-driven-development.sh" ) diff --git a/tests/claude-code/test-executing-plans-scripts.sh b/tests/claude-code/test-executing-plans-scripts.sh new file mode 100755 index 000000000..994733bf4 --- /dev/null +++ b/tests/claude-code/test-executing-plans-scripts.sh @@ -0,0 +1,139 @@ +#!/usr/bin/env bash +# Tests for executing-plans' bookkeeping helpers: scripts/task-start extracts +# the brief and records BASE in one call; scripts/task-done runs the task's +# test command, records the result in the ledger, and refuses to record a +# failing task. +set -euo pipefail + +SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)" +REPO_ROOT="$(cd "$SCRIPT_DIR/../.." && pwd)" +EP_SCRIPTS="$REPO_ROOT/skills/executing-plans/scripts" + +FAILURES=0 +TEST_ROOT="" + +pass() { echo " [PASS] $1"; } +fail() { + echo " [FAIL] $1" + FAILURES=$((FAILURES + 1)) +} + +cleanup() { + if [[ -n "$TEST_ROOT" && -d "$TEST_ROOT" ]]; then + rm -rf "$TEST_ROOT" + fi +} + +main() { + echo "=== Test: executing-plans scripts ===" + + TEST_ROOT="$(mktemp -d)" + trap cleanup EXIT + + git init -q -b main "$TEST_ROOT/repo" + local repo + repo="$(cd "$TEST_ROOT/repo" && git rev-parse --show-toplevel)" + local git_id=(-c user.email=t@example.com -c user.name=t -c commit.gpgsign=false) + + cat > "$repo/plan.md" <<'PLAN' +# Plan + +## Task 1: First thing + +Do the first thing. + +## Task 2: Second thing + +Do the second thing. +PLAN + ( cd "$repo" && git add plan.md && git "${git_id[@]}" commit -qm fixture ) + local base + base="$(cd "$repo" && git rev-parse HEAD)" + + # --- task-start: argument validation --- + local rc=0 + (cd "$repo" && "$EP_SCRIPTS/task-start" plan.md >/dev/null 2>&1) || rc=$? + if [[ "$rc" -eq 2 ]]; then + pass "task-start without a task number errors with exit 2" + else + fail "task-start without a task number errors with exit 2 (got $rc)" + fi + + # --- task-start: brief path + BASE in one call --- + local out + out="$(cd "$repo" && "$EP_SCRIPTS/task-start" plan.md 1)" + if [[ "$out" == *"brief: $repo/.superpowers/sdd/plan/task-1-brief.md"* ]]; then + pass "task-start prints the brief path under the plan's workspace" + else + fail "task-start prints the brief path under the plan's workspace" + echo " got: $out" + fi + if [[ "$out" == *"base: $base"* ]]; then + pass "task-start prints BASE as the current HEAD" + else + fail "task-start prints BASE as the current HEAD" + echo " got: $out" + fi + if [[ -s "$repo/.superpowers/sdd/plan/task-1-brief.md" ]]; then + pass "task-start writes the brief file" + else + fail "task-start writes the brief file" + fi + + # --- task-done: records a passing task --- + ( cd "$repo" && echo x > work.txt && git add work.txt && git "${git_id[@]}" commit -qm "task 1" ) + local head + head="$(cd "$repo" && git rev-parse HEAD)" + out="$(cd "$repo" && "$EP_SCRIPTS/task-done" plan.md 1 "$base" -- sh -c 'echo "Ran 3 tests"; echo OK')" + rc=$? + local ledger="$repo/.superpowers/sdd/plan/progress.md" + local expected="Task 1: complete (commits ${base:0:7}..${head:0:7}, tests: sh -c 'echo \"Ran 3 tests\"; echo OK' → OK)" + if [[ -f "$ledger" ]] && grep -qF "$expected" "$ledger"; then + pass "task-done appends the completion line with commit range and test result" + else + fail "task-done appends the completion line with commit range and test result" + echo " expected: $expected" + echo " ledger:"; sed 's/^/ /' "$ledger" 2>/dev/null || echo " (missing)" + fi + if [[ "$out" == *"OK"* ]]; then + pass "task-done prints the tail of the test output" + else + fail "task-done prints the tail of the test output" + echo " got: $out" + fi + if [[ -s "$repo/.superpowers/sdd/plan/task-1-tests.log" ]]; then + pass "task-done keeps the full test output in the workspace" + else + fail "task-done keeps the full test output in the workspace" + fi + + # --- task-done: refuses to record a failing task --- + rc=0 + out="$(cd "$repo" && "$EP_SCRIPTS/task-done" plan.md 2 "$head" -- sh -c 'echo "FAILED (errors=1)"; exit 1' 2>&1)" || rc=$? + if [[ "$rc" -ne 0 ]]; then + pass "task-done exits non-zero when the test command fails" + else + fail "task-done exits non-zero when the test command fails" + fi + if ! grep -q "Task 2: complete" "$ledger"; then + pass "task-done does not record a failing task as complete" + else + fail "task-done does not record a failing task as complete" + fi + if [[ "$out" == *"FAILED"* ]]; then + pass "task-done shows the failing output" + else + fail "task-done shows the failing output" + echo " got: $out" + fi + + echo + if [[ "$FAILURES" -eq 0 ]]; then + echo "PASS" + else + echo "FAIL ($FAILURES)" + exit 1 + fi +} + +main "$@" diff --git a/tests/claude-code/test-sdd-workspace.sh b/tests/claude-code/test-sdd-workspace.sh index 841723016..2e8c227db 100755 --- a/tests/claude-code/test-sdd-workspace.sh +++ b/tests/claude-code/test-sdd-workspace.sh @@ -165,6 +165,30 @@ PLAN echo " got: $rp_explicit" fi + # --- range guards: BASE must be an ancestor of HEAD, range must be non-empty --- + local divergent + divergent="$(cd "$repo" && git "${git_id[@]}" commit-tree 'HEAD~1^{tree}' -p 'HEAD~1' -m divergent)" + rc=0 + local guard_err + guard_err="$(cd "$repo" && "$SDD_SCRIPTS/review-package" plan-a.md "$divergent" HEAD 2>&1 >/dev/null)" || rc=$? + if [[ "$rc" -eq 3 && "$guard_err" == *"not a descendant"* ]]; then + pass "review-package rejects a BASE that is not an ancestor of HEAD with exit 3" + else + fail "review-package rejects a BASE that is not an ancestor of HEAD with exit 3" + echo " exit: $rc" + echo " stderr: $guard_err" + fi + + rc=0 + guard_err="$(cd "$repo" && "$SDD_SCRIPTS/review-package" plan-a.md HEAD HEAD 2>&1 >/dev/null)" || rc=$? + if [[ "$rc" -eq 3 && "$guard_err" == *"empty commit range"* ]]; then + pass "review-package rejects an empty BASE..HEAD range with exit 3" + else + fail "review-package rejects an empty BASE..HEAD range with exit 3" + echo " exit: $rc" + echo " stderr: $guard_err" + fi + # --- Worktree isolation: a linked worktree resolves its own workspace --- local wt="$TEST_ROOT/wt" ( cd "$repo" && git worktree add -q "$wt" -b wt-feature ) @@ -189,6 +213,143 @@ PLAN echo " status: $wt_status" fi + # --- helpers survive a mode-stripping extractor dropping exec bits (#2040) --- + local stripped="$TEST_ROOT/stripped-scripts" + mkdir -p "$stripped" + cp "$SDD_SCRIPTS/sdd-workspace" "$SDD_SCRIPTS/task-brief" "$SDD_SCRIPTS/review-package" "$stripped/" + chmod -x "$stripped"/* + local noexec_out noexec_rc=0 + noexec_out="$(cd "$repo" && bash "$stripped/task-brief" plan-b.md 1 2>&1)" || noexec_rc=$? + if [[ "$noexec_rc" -eq 0 && -f "$repo/.superpowers/sdd/plan-b/task-1-brief.md" ]]; then + pass "task-brief works with no exec bit on sdd-workspace" + else + fail "task-brief works with no exec bit on sdd-workspace" + echo " rc: $noexec_rc" + echo " output: $noexec_out" + fi + + # --- Ownership markers: two plans with the same basename (#2045) --- + mkdir -p "$repo/docs/alpha" "$repo/docs/beta" + cat > "$repo/docs/alpha/plan.md" <<'PLAN' +# Alpha Plan + +## Task 1: Alpha work + +Alpha-only requirement text. +PLAN + cat > "$repo/docs/beta/plan.md" <<'PLAN' +# Beta Plan + +## Task 1: Beta work + +Beta-only requirement text. +PLAN + + local dir_alpha dir_beta + dir_alpha="$(cd "$repo" && "$SDD_SCRIPTS/sdd-workspace" docs/alpha/plan.md)" + dir_beta="$(cd "$repo" && "$SDD_SCRIPTS/sdd-workspace" docs/beta/plan.md)" + if [[ "$dir_alpha" != "$dir_beta" ]]; then + pass "same-basename plans resolve to distinct workspaces" + else + fail "same-basename plans resolve to distinct workspaces" + echo " alpha: $dir_alpha" + echo " beta: $dir_beta" + fi + + ( cd "$repo" && "$SDD_SCRIPTS/task-brief" docs/alpha/plan.md 1 >/dev/null ) + ( cd "$repo" && "$SDD_SCRIPTS/task-brief" docs/beta/plan.md 1 >/dev/null ) + if grep -q "Alpha-only requirement text." "$dir_alpha/task-1-brief.md" 2>/dev/null \ + && grep -q "Beta-only requirement text." "$dir_beta/task-1-brief.md" 2>/dev/null; then + pass "same-basename plans keep both task briefs intact" + else + fail "same-basename plans keep both task briefs intact" + echo " alpha brief: $(cat "$dir_alpha/task-1-brief.md" 2>/dev/null)" + echo " beta brief: $(cat "$dir_beta/task-1-brief.md" 2>/dev/null)" + fi + + # --- Legacy adoption: pre-existing workspace without a marker --- + printf '# Foo\n\n## Task 1: Foo\n\nFoo.\n' > "$repo/foo.md" + mkdir -p "$repo/.superpowers/sdd/foo" + printf 'ledger\n' > "$repo/.superpowers/sdd/foo/progress.md" + local dir_foo + dir_foo="$(cd "$repo" && "$SDD_SCRIPTS/sdd-workspace" foo.md)" + if [[ "$dir_foo" == "$repo/.superpowers/sdd/foo" \ + && -f "$dir_foo/progress.md" \ + && "$(cat "$dir_foo/plan-path" 2>/dev/null)" == "foo.md" ]]; then + pass "legacy markerless workspace is adopted in place and marked" + else + fail "legacy markerless workspace is adopted in place and marked" + echo " dir: $dir_foo" + echo " marker: $(cat "$dir_foo/plan-path" 2>/dev/null)" + fi + + # --- Ownership conflict: marker names a different plan --- + printf '# Bar\n\n## Task 1: Bar\n\nBar.\n' > "$repo/bar.md" + mkdir -p "$repo/.superpowers/sdd/bar" + printf 'somewhere-else/bar.md\n' > "$repo/.superpowers/sdd/bar/plan-path" + printf 'other ledger\n' > "$repo/.superpowers/sdd/bar/progress.md" + local dir_bar + dir_bar="$(cd "$repo" && "$SDD_SCRIPTS/sdd-workspace" bar.md)" + if [[ "$dir_bar" == "$repo/.superpowers/sdd/bar-repo" \ + && "$(cat "$dir_bar/plan-path" 2>/dev/null)" == "bar.md" ]]; then + pass "owned workspace disambiguates with parent-dir suffix" + else + fail "owned workspace disambiguates with parent-dir suffix" + echo " got: $dir_bar" + fi + if [[ "$(cat "$repo/.superpowers/sdd/bar/plan-path")" == "somewhere-else/bar.md" \ + && "$(cat "$repo/.superpowers/sdd/bar/progress.md")" == "other ledger" ]]; then + pass "conflicting plan leaves the original workspace untouched" + else + fail "conflicting plan leaves the original workspace untouched" + fi + + # --- Counter fallback: parent-suffixed workspace is owned too --- + printf '# Baz\n\n## Task 1: Baz\n\nBaz.\n' > "$repo/baz.md" + mkdir -p "$repo/.superpowers/sdd/baz" "$repo/.superpowers/sdd/baz-repo" + printf 'one/baz.md\n' > "$repo/.superpowers/sdd/baz/plan-path" + printf 'two/baz.md\n' > "$repo/.superpowers/sdd/baz-repo/plan-path" + local dir_baz + dir_baz="$(cd "$repo" && "$SDD_SCRIPTS/sdd-workspace" baz.md)" + if [[ "$dir_baz" == "$repo/.superpowers/sdd/baz-repo-2" \ + && "$(cat "$dir_baz/plan-path" 2>/dev/null)" == "baz.md" ]]; then + pass "double conflict falls back to a counter suffix" + else + fail "double conflict falls back to a counter suffix" + echo " got: $dir_baz" + fi + + # --- Same plan spelled differently resolves to one workspace --- + local dir_rel dir_abs dir_dotdot + dir_rel="$(cd "$repo" && "$SDD_SCRIPTS/sdd-workspace" docs/alpha/plan.md)" + dir_abs="$(cd "$repo" && "$SDD_SCRIPTS/sdd-workspace" "$repo/docs/alpha/plan.md")" + dir_dotdot="$(cd "$repo/docs/beta" && "$SDD_SCRIPTS/sdd-workspace" ../alpha/plan.md)" + if [[ "$dir_rel" == "$dir_abs" && "$dir_rel" == "$dir_dotdot" \ + && "$(cat "$dir_rel/plan-path" 2>/dev/null)" == "docs/alpha/plan.md" ]]; then + pass "relative, absolute, and ../ spellings share one workspace and marker" + else + fail "relative, absolute, and ../ spellings share one workspace and marker" + echo " rel: $dir_rel" + echo " abs: $dir_abs" + echo " dotdot: $dir_dotdot" + echo " marker: $(cat "$dir_rel/plan-path" 2>/dev/null)" + fi + + # --- Out-of-repo plans keep working, marker holds the absolute path --- + mkdir -p "$TEST_ROOT/outside" + printf '# Remote\n\n## Task 1: Remote\n\nRemote.\n' > "$TEST_ROOT/outside/remote-plan.md" + local outside_abs dir_out + outside_abs="$(cd "$TEST_ROOT/outside" && pwd -P)/remote-plan.md" + dir_out="$(cd "$repo" && "$SDD_SCRIPTS/sdd-workspace" "$TEST_ROOT/outside/remote-plan.md")" + if [[ "$dir_out" == "$repo/.superpowers/sdd/remote-plan" \ + && "$(cat "$dir_out/plan-path" 2>/dev/null)" == "$outside_abs" ]]; then + pass "out-of-repo plan gets a basename slug and an absolute-path marker" + else + fail "out-of-repo plan gets a basename slug and an absolute-path marker" + echo " dir: $dir_out" + echo " marker: $(cat "$dir_out/plan-path" 2>/dev/null)" + fi + echo "" if [[ "$FAILURES" -ne 0 ]]; then echo "FAILED: $FAILURES assertion(s)." diff --git a/tests/codex-plugin-sync/test-sync-to-codex-plugin.sh b/tests/codex-plugin-sync/test-sync-to-codex-plugin.sh index 01de39484..265aea153 100755 --- a/tests/codex-plugin-sync/test-sync-to-codex-plugin.sh +++ b/tests/codex-plugin-sync/test-sync-to-codex-plugin.sh @@ -194,6 +194,10 @@ write_upstream_fixture() { "name": "fixture-upstream", "version": "$PACKAGE_VERSION" } +EOF + + cat > "$repo/index.js" <<'EOF' +export { default } from "./.opencode/plugins/superpowers.js"; EOF cat > "$repo/.gitignore" <<'EOF' @@ -303,6 +307,7 @@ EOF hooks/run-hook.cmd \ hooks/session-start \ hooks/session-start-codex \ + index.js \ package.json \ scripts/sync-to-codex-plugin.sh \ skills/example/SKILL.md @@ -664,6 +669,7 @@ main() { assert_not_contains "$preview_section" "evals/" "Preview excludes eval harness" assert_not_contains "$preview_section" ".gitmodules" "Preview excludes repo submodule metadata" assert_not_contains "$preview_section" ".pre-commit-config.yaml" "Preview excludes repo pre-commit config" + assert_not_contains "$preview_section" "index.js" "Preview excludes OpenCode root entrypoint" assert_not_contains "$preview_output" "Overlay file (.codex-plugin/plugin.json) will be regenerated" "Preview omits overlay regeneration note" assert_not_contains "$preview_output" "Assets (superpowers-small.svg, app-icon.png) will be seeded from" "Preview omits assets seeding note" assert_contains "$preview_section" "skills/example/SKILL.md" "Preview reflects dirty tracked destination file" diff --git a/tests/opencode/run-tests.sh b/tests/opencode/run-tests.sh index d9b100ef0..43e1a3a2b 100755 --- a/tests/opencode/run-tests.sh +++ b/tests/opencode/run-tests.sh @@ -45,6 +45,8 @@ while [[ $# -gt 0 ]]; do echo "Tests:" echo " test-plugin-loading.sh Verify plugin installation and structure" echo " test-bootstrap-caching.sh Verify bootstrap content caching" + echo " test-session-bootstrap.sh Verify session classification and lookup recovery" + echo " test-skill-registration.sh Verify V2 skill registration contract (2.0.4 path field)" echo " test-tools.sh Test use_skill and find_skills tools (integration)" echo " test-priority.sh Test skill priority resolution (integration)" exit 0 @@ -61,6 +63,8 @@ done tests=( "test-plugin-loading.sh" "test-bootstrap-caching.sh" + "test-session-bootstrap.sh" + "test-skill-registration.sh" ) # Integration tests (require OpenCode) diff --git a/tests/opencode/test-bootstrap-caching.mjs b/tests/opencode/test-bootstrap-caching.mjs index 32149aed3..386b7b1ef 100644 --- a/tests/opencode/test-bootstrap-caching.mjs +++ b/tests/opencode/test-bootstrap-caching.mjs @@ -32,6 +32,10 @@ const mod = await import(pathToFileURL(pluginPath).href); const plugin = await mod.SuperpowersPlugin({ client: {}, directory: '.' }); const transform = plugin['experimental.chat.messages.transform']; +// Mapping constants are flavor-specific (#opencode-v2): V1 keeps the 1.18.x +// tool names, V2 teaches the renamed tools. Assert both directly. +const mappingFailures = assertMappingConstants(mod); + const firstOutput = makeOutput(`${scenario} bootstrap first step`); await transform({}, firstOutput); const afterFirst = { existsCount, readCount }; @@ -40,6 +44,11 @@ const secondOutput = makeOutput(`${scenario} bootstrap second step`); await transform({}, secondOutput); const afterSecond = { existsCount, readCount }; +// Exercise the V2 path (setup() + ctx.session.hook("context")) with a mock +// ctx so the V2_MAPPING wiring is verified, not just the constant. Run after +// the V1 count snapshots: setup() reads SKILL.md files during registration. +const v2Result = await runV2ContextHook(mod); + const result = { scenario, firstBootstrapParts: countBootstrapParts(firstOutput), @@ -52,12 +61,24 @@ const result = { secondReadCount: afterSecond.readCount, firstExistsCount: afterFirst.existsCount, secondExistsCount: afterSecond.existsCount, + v2BootstrapParts: v2Result.bootstrapParts, + mapsV2SubagentTool: v2Result.text.includes('`subagent` with `agent: "general"`'), + mapsV2SessionIDContinuation: v2Result.text.includes('`sessionID` to continue a previous subagent'), + mapsV2NoTodoTool: v2Result.text.includes('no todo tool'), + mapsV2MutationToPatch: v2Result.text.includes('`patch` with `patchText`'), + mapsV2Shell: v2Result.text.includes('`shell`'), + staleV1ToolsInV2: v2Result.text.includes('`apply_patch`') || v2Result.text.includes('`todowrite`') || v2Result.text.includes('`subagent_type`'), }; const failures = scenario === 'present' ? assertPresentBootstrap(result) : assertMissingBootstrap(result); +if (scenario === 'present') { + failures.push(...assertV2Bootstrap(result)); +} +failures.push(...mappingFailures); + if (failures.length > 0) { console.error(JSON.stringify(result, null, 2)); for (const failure of failures) { @@ -144,3 +165,104 @@ function assertMissingBootstrap(result) { } return failures; } + +function assertMappingConstants(mod) { + const failures = []; + if (typeof mod.V1_MAPPING !== 'string' || typeof mod.V2_MAPPING !== 'string') { + failures.push('expected plugin to export V1_MAPPING and V2_MAPPING string constants'); + return failures; + } + for (const needle of ['`todowrite`', '`task` with `subagent_type: "general"`', '`apply_patch`', '`bash`']) { + if (!mod.V1_MAPPING.includes(needle)) { + failures.push(`expected V1_MAPPING to keep the 1.18.x tool name ${needle}`); + } + } + for (const needle of [ + '`subagent` with `agent: "general"`', + '`sessionID` to continue a previous subagent', + 'no todo tool', + '`write`', + '`edit`', + '`patch` with `patchText`', + '`shell`', + '`read`', + '`grep`, `glob`', + '`webfetch`', + '`websearch`', + ]) { + if (!mod.V2_MAPPING.includes(needle)) { + failures.push(`expected V2_MAPPING to teach the V2 tool ${needle}`); + } + } + for (const stale of ['`todowrite`', '`task` with', '`apply_patch`', '`bash`']) { + if (mod.V2_MAPPING.includes(stale)) { + failures.push(`expected V2_MAPPING not to teach the V1-only tool name ${stale}`); + } + } + return failures; +} + +// Drive setup() with a mock V2 ctx and fire the captured "context" hook on a +// top-level (parentID-less) session. Returns the injected-part count and the +// injected bootstrap text ('' when nothing was injected). +async function runV2ContextHook(mod) { + let contextHook = null; + const ctx = { + skill: { + transform: async (fn) => { + fn({ add: () => {} }); + }, + }, + session: { + hook: async (name, cb) => { + if (name === 'context') contextHook = cb; + }, + get: async ({ sessionID }) => ({ id: sessionID }), // top-level: no parentID + }, + }; + try { + await mod.default.setup(ctx); + } catch (err) { + console.error('[test] V2 setup() threw:', err); + return { bootstrapParts: 0, text: '' }; + } + if (typeof contextHook !== 'function') { + return { bootstrapParts: 0, text: '' }; + } + const event = { + sessionID: 'sess-v2-top', + messages: [{ role: 'user', content: [{ type: 'text', text: 'v2 bootstrap step' }] }], + }; + await contextHook(event); + const parts = event.messages[0].content.filter( + (part) => part.type === 'text' && part.text.includes('EXTREMELY_IMPORTANT') + ); + return { bootstrapParts: parts.length, text: parts[0]?.text || '' }; +} + +function assertV2Bootstrap(result) { + const failures = []; + if (result.v2BootstrapParts !== 1) { + failures.push(`expected V2 context hook to inject one bootstrap part, got ${result.v2BootstrapParts}`); + return failures; + } + if (!result.mapsV2SubagentTool) { + failures.push('expected V2 bootstrap to map general-purpose subagents to subagent with agent'); + } + if (!result.mapsV2SessionIDContinuation) { + failures.push('expected V2 bootstrap to teach sessionID continuation for subagents'); + } + if (!result.mapsV2NoTodoTool) { + failures.push('expected V2 bootstrap to state that V2 has no todo tool'); + } + if (!result.mapsV2MutationToPatch) { + failures.push('expected V2 bootstrap to map file mutation to patch with patchText'); + } + if (!result.mapsV2Shell) { + failures.push('expected V2 bootstrap to map shell commands to the shell tool'); + } + if (result.staleV1ToolsInV2) { + failures.push('expected V2 bootstrap not to teach V1-only tool names (apply_patch/todowrite/subagent_type)'); + } + return failures; +} diff --git a/tests/opencode/test-session-bootstrap.mjs b/tests/opencode/test-session-bootstrap.mjs new file mode 100644 index 000000000..31658d5e5 --- /dev/null +++ b/tests/opencode/test-session-bootstrap.mjs @@ -0,0 +1,224 @@ +import assert from 'node:assert/strict'; +import fs from 'node:fs'; +import { pathToFileURL } from 'node:url'; + +const [, , inputPath] = process.argv; +assert.ok(inputPath, 'pass the plugin module path'); +const pluginURL = pathToFileURL(fs.realpathSync(inputPath)); +const marker = '\nYou have superpowers.'; +let generation = 0; + +function reply(flavor, session) { + return flavor === 'v1' ? { data: session } : session; +} + +function makeEvent(flavor, sessionID) { + const text = { type: 'text', text: 'Execute the assigned task' }; + return { + sessionID, + messages: [flavor === 'v1' + ? { info: { role: 'user', sessionID }, parts: [text] } + : { role: 'user', content: [text] }], + }; +} + +function bootstrapCount(event) { + return event.messages.flatMap((message) => message.parts ?? message.content ?? []).filter( + (part) => part.type === 'text' && part.text.startsWith(marker) + ).length; +} + +async function makeHarness(flavor, fetchSession) { + const mod = await import(`${pluginURL.href}?session-test=${++generation}`); + const lookups = []; + const registered = []; + const get = async (id) => { + lookups.push(id); + return fetchSession(id, lookups.length); + }; + let invoke; + if (flavor === 'v1') { + const hooks = await mod.SuperpowersPlugin({ + client: { session: { get: ({ path: { id } }) => get(id) } }, + directory: '.', + }); + invoke = (event) => hooks['experimental.chat.messages.transform']({}, event); + } else { + await mod.default.setup({ + skill: { transform: async (transform) => transform({ add: (skill) => registered.push(skill) }) }, + session: { + get: ({ sessionID }) => get(sessionID), + hook: async (name, callback) => { if (name === 'context') invoke = callback; }, + }, + }); + } + assert.equal(typeof invoke, 'function'); + return { invoke, lookups, registered }; +} + +for (const flavor of ['v1', 'v2']) { + for (const [kind, extra, expected] of [ + ['root', {}, 1], + ['child', { parentID: 'parent' }, 0], + ['fork', { fork: { sessionID: 'origin' } }, 1], + ]) { + const id = `${flavor}-${kind}`; + const h = await makeHarness(flavor, () => reply(flavor, { id, ...extra })); + const event = makeEvent(flavor, id); + await h.invoke(event); + assert.equal(bootstrapCount(event), expected, `${id}: first request`); + await h.invoke(event); + assert.equal(bootstrapCount(event), expected, `${id}: repeated event`); + const fresh = makeEvent(flavor, id); + await h.invoke(fresh); + assert.equal(bootstrapCount(fresh), expected, `${id}: fresh request`); + assert.deepEqual(h.lookups, [id], `${id}: cache successful classification`); + if (flavor === 'v2' && kind === 'child') { + assert.ok(h.registered.some((skill) => skill.id === 'brainstorming')); + } + } + + const failures = [ + ['throws', () => { throw new Error('temporary lookup failure'); }], + ['missing', () => undefined], + ['null', () => null], + ['empty', () => reply(flavor, {})], + ['wrong-id', () => reply(flavor, { id: 'different-session' })], + ['invalid-parent', (id) => reply(flavor, { id, parentID: 42 })], + ]; + if (flavor === 'v1') { + failures.push(['resolved-http-error', () => ({ + data: undefined, + error: { name: 'UnknownError', data: { message: 'temporary 503' } }, + response: { ok: false, status: 503 }, + })]); + } + for (const [kind, firstResult] of failures) { + const id = `${flavor}-${kind}`; + const h = await makeHarness(flavor, (sessionID, call) => call === 1 + ? firstResult(sessionID) + : reply(flavor, { id: sessionID, parentID: 'parent' })); + const counts = []; + for (let step = 0; step < 2; step++) { + const event = makeEvent(flavor, id); + await h.invoke(event); + counts.push(bootstrapCount(event)); + } + assert.deepEqual(counts, [1, 0], `${id}: recover on the next request`); + assert.deepEqual(h.lookups, [id, id], `${id}: never cache the failure`); + } + + const isolated = await makeHarness(flavor, (id) => reply(flavor, + id === 'child-session' ? { id, parentID: 'parent' } : { id })); + for (const [id, expected] of [['root-session', 1], ['child-session', 0], ['root-session', 1], ['child-session', 0]]) { + const event = makeEvent(flavor, id); + await isolated.invoke(event); + assert.equal(bootstrapCount(event), expected); + } + assert.deepEqual(isolated.lookups, ['root-session', 'child-session']); + + const bounded = await makeHarness(flavor, (id) => reply(flavor, { id, parentID: 'parent' })); + for (let index = 0; index <= 512; index++) { + const event = makeEvent(flavor, `eviction-${index}`); + await bounded.invoke(event); + assert.equal(bootstrapCount(event), 0); + } + const evicted = makeEvent(flavor, 'eviction-0'); + await bounded.invoke(evicted); + assert.equal(bootstrapCount(evicted), 0); + assert.equal(bounded.lookups.filter((id) => id === 'eviction-0').length, 2); + + const restarted = await makeHarness(flavor, (id) => reply(flavor, { id, parentID: 'parent' })); + const afterRestart = makeEvent(flavor, 'eviction-0'); + await restarted.invoke(afterRestart); + assert.equal(bootstrapCount(afterRestart), 0); + assert.deepEqual(restarted.lookups, ['eviction-0']); + + const unknown = await makeHarness(flavor, () => { throw new Error('must not look up a missing ID'); }); + const noID = makeEvent(flavor, undefined); + await unknown.invoke(noID); + assert.equal(bootstrapCount(noID), 1); + assert.deepEqual(unknown.lookups, []); +} + +function compactedEvent(sessionID) { + return { + sessionID, + system: [], + messages: [{ + role: 'assistant', + content: [{ type: 'compaction', provider: 'fixture', encrypted: 'opaque-checkpoint' }], + }], + }; +} + +const compactedRoot = await makeHarness('v2', (id) => ({ id })); +const rootEvent = compactedEvent('compacted-root'); +const checkpoint = structuredClone(rootEvent.messages[0]); +await compactedRoot.invoke(rootEvent); +assert.equal(bootstrapCount(rootEvent), 1); +assert.deepEqual(rootEvent.messages[0], checkpoint); +assert.equal(rootEvent.messages.length, 2); +assert.equal(rootEvent.messages[1].role, 'user'); +assert.deepEqual(rootEvent.system, []); +await compactedRoot.invoke(rootEvent); +assert.equal(bootstrapCount(rootEvent), 1); +assert.equal(rootEvent.messages.length, 2); +const freshRootEvent = compactedEvent('compacted-root'); +await compactedRoot.invoke(freshRootEvent); +assert.equal(bootstrapCount(freshRootEvent), 1); +assert.deepEqual(compactedRoot.lookups, ['compacted-root']); + +const compactedChild = await makeHarness('v2', (id) => ({ id, parentID: 'parent' })); +const childEvent = compactedEvent('compacted-child'); +const originalChild = structuredClone(childEvent); +await compactedChild.invoke(childEvent); +assert.equal(bootstrapCount(childEvent), 0); +assert.deepEqual(childEvent, originalChild); +assert.deepEqual(compactedChild.lookups, ['compacted-child']); + +const retryChild = await makeHarness('v2', (id, call) => { + if (call === 1) throw new Error('temporary lookup failure'); + return { id, parentID: 'parent' }; +}); +const unknownChild = compactedEvent('retry-compacted-child'); +await retryChild.invoke(unknownChild); +assert.equal(bootstrapCount(unknownChild), 1); +const recoveredChild = compactedEvent('retry-compacted-child'); +await retryChild.invoke(recoveredChild); +assert.equal(bootstrapCount(recoveredChild), 0); +assert.equal(recoveredChild.messages.length, 1); +assert.equal(retryChild.lookups.length, 2); + +const newPromptAfterCheckpoint = compactedEvent('new-prompt-after-checkpoint-root'); +newPromptAfterCheckpoint.messages.push({ role: 'user', content: [{ type: 'text', text: 'Continue' }] }); +await compactedRoot.invoke(newPromptAfterCheckpoint); +assert.equal(bootstrapCount(newPromptAfterCheckpoint), 1); +assert.equal(newPromptAfterCheckpoint.messages.length, 2); +assert.equal(newPromptAfterCheckpoint.messages[1].content.length, 2); + +const retainedUser = compactedEvent('retained-user-root'); +retainedUser.messages.unshift({ role: 'user', content: [{ type: 'text', text: 'Keep going' }] }); +const retainedCheckpoint = structuredClone(retainedUser.messages[1]); +await compactedRoot.invoke(retainedUser); +assert.equal(bootstrapCount(retainedUser), 1); +assert.equal(retainedUser.messages.length, 2); +assert.equal(retainedUser.messages[0].content.length, 2); +assert.ok(retainedUser.messages[0].content[0].text.startsWith(marker)); +assert.equal(retainedUser.messages[0].content[1].text, 'Keep going'); +assert.deepEqual(retainedUser.messages[1], retainedCheckpoint); +await compactedRoot.invoke(retainedUser); +assert.equal(bootstrapCount(retainedUser), 1); +assert.equal(retainedUser.messages.length, 2); + +const retainedUserChild = compactedEvent('retained-user-child'); +retainedUserChild.messages.unshift({ role: 'user', content: [{ type: 'text', text: 'Keep going' }] }); +const originalRetainedUserChild = structuredClone(retainedUserChild); +await compactedChild.invoke(retainedUserChild); +assert.equal(bootstrapCount(retainedUserChild), 0); +assert.deepEqual(retainedUserChild, originalRetainedUserChild); +const empty = { sessionID: 'empty', messages: [] }; +await compactedRoot.invoke(empty); +assert.deepEqual(empty.messages, []); + +console.log('Session classification, recovery and cache lifetime passed'); diff --git a/tests/opencode/test-session-bootstrap.sh b/tests/opencode/test-session-bootstrap.sh new file mode 100755 index 000000000..accd85bd0 --- /dev/null +++ b/tests/opencode/test-session-bootstrap.sh @@ -0,0 +1,4 @@ +#!/usr/bin/env bash +set -euo pipefail +SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)" +node "$SCRIPT_DIR/test-session-bootstrap.mjs" "$SCRIPT_DIR/../../.opencode/plugins/superpowers.js" diff --git a/tests/opencode/test-skill-registration.mjs b/tests/opencode/test-skill-registration.mjs new file mode 100644 index 000000000..a31fe4abd --- /dev/null +++ b/tests/opencode/test-skill-registration.mjs @@ -0,0 +1,205 @@ +import fs from 'fs'; +import os from 'os'; +import path from 'path'; +import { pathToFileURL } from 'url'; + +// Verifies the V2 skill registration payload matches OpenCode 2.0.4's +// Skill.Info contract (packages/schema/src/skill.ts): +// { id, name, description?, autoinvoke?, path, content } +// Upstream commit 199aabe9e2 (first released in v2.0.4) renamed the required +// file field `location` -> `path`. A wrong field name makes draft.add() +// throw inside the host's transform rebuild, which asynchronously disables +// the whole plugin ("Plugin disabled after skill.transform failed") and +// takes the bootstrap hook down with it — see PR #2106 review by 80avin. + +const [, , inputPath] = process.argv; + +if (!inputPath) { + console.error('Usage: node test-skill-registration.mjs PLUGIN_PATH'); + process.exit(2); +} + +const pluginPath = fs.realpathSync(inputPath); +const skillsDir = path.resolve(path.dirname(pluginPath), '../../skills'); +const mod = await import(pathToFileURL(pluginPath).href); + +const failures = []; + +// --- Run 1: passive capture of every draft.add payload ------------------- +const added = []; +await mod.default.setup(makeCtx({ add: (skill) => added.push(skill) })); + +const expectedIds = fs.existsSync(skillsDir) + ? fs.readdirSync(skillsDir, { withFileTypes: true }) + .filter((e) => e.isDirectory() && !e.name.startsWith('.')) + .filter((e) => fs.existsSync(path.join(skillsDir, e.name, 'SKILL.md'))) + .map((e) => e.name) + .sort() + : []; + +if (added.length === 0) { + failures.push('expected setup() to register at least one skill via draft.add()'); +} +if (JSON.stringify(added.map((s) => s.id).sort()) !== JSON.stringify(expectedIds)) { + failures.push(`expected draft.add() ids to match skills dir contents, got ${JSON.stringify(added.map((s) => s.id))}`); +} + +for (const skill of added) { + if (typeof skill.path !== 'string' || !path.isAbsolute(skill.path)) { + failures.push(`skill "${skill.id}": expected required absolute Skill.Info field "path", got ${JSON.stringify(skill.path)}`); + } else if (skill.path !== path.join(skillsDir, skill.id, 'SKILL.md')) { + failures.push(`skill "${skill.id}": expected path ${path.join(skillsDir, skill.id, 'SKILL.md')}, got ${skill.path}`); + } else if (!fs.existsSync(skill.path)) { + failures.push(`skill "${skill.id}": path does not exist on disk: ${skill.path}`); + } + // Stale 2.0.3-era fields must not leak into the payload: the host strips + // unknown keys, but keeping them would silently mask a future regression + // to a schema that no longer accepts `path`. + if ('location' in skill) { + failures.push(`skill "${skill.id}": payload still carries the pre-2.0.4 field "location"`); + } + if ('slash' in skill) { + failures.push(`skill "${skill.id}": payload carries "slash", removed from Skill.Info in 2.0.4`); + } + if (typeof skill.id !== 'string' || skill.id.length === 0) failures.push(`skill payload missing non-empty "id"`); + if (typeof skill.name !== 'string' || skill.name.length === 0) failures.push(`skill "${skill.id}" missing non-empty "name"`); + if (typeof skill.content !== 'string' || !skill.content.trim()) failures.push(`skill "${skill.id}" missing non-empty "content"`); + if ('description' in skill && typeof skill.description !== 'string') { + failures.push(`skill "${skill.id}": "description" must be a string when present`); + } else if ('description' in skill && /["']$/.test(skill.description)) { + failures.push(`skill "${skill.id}": description ends with a dangling quote: ${JSON.stringify(skill.description)}`); + } + if (typeof skill.content === 'string' && skill.content.startsWith('---')) { + failures.push(`skill "${skill.id}": content still starts with the frontmatter delimiter`); + } +} + +// --- Run 2: hostile draft.add must not abort the remaining registrations -- +// The real host swallows a throw escaping the transform callback and then +// hard-disables the plugin asynchronously. Locally we can only observe the +// synchronous half of that contract: when draft.add() rejects one skill, the +// plugin must keep registering the rest instead of aborting the loop. +const hostileId = added.length > 1 ? added[Math.floor(added.length / 2)].id : null; +const survived = []; +let setupThrew = null; +let survivingContextHook; +try { + await mod.default.setup(makeCtx({ + add: (skill) => { + if (skill.id === hostileId) throw new Error('Simulated Skill.Info decode failure'); + survived.push(skill.id); + }, + onHook: (name, callback) => { + if (name === 'context') survivingContextHook = callback; + }, + })); +} catch (err) { + setupThrew = err; +} +if (setupThrew) { + failures.push(`expected setup() to contain draft.add() failures, but it threw: ${setupThrew.message}`); +} else if (hostileId) { + const expectedSurvivors = added.map((s) => s.id).filter((id) => id !== hostileId); + if (JSON.stringify(survived.sort()) !== JSON.stringify(expectedSurvivors.sort())) { + failures.push(`expected all non-rejected skills to still register when one draft.add() throws, got ${JSON.stringify(survived)}`); + } +} +if (typeof survivingContextHook !== 'function') { + failures.push('expected bootstrap hook to survive a rejected skill'); +} else { + const event = { + sessionID: 'registration-survival-root', + messages: [{ role: 'user', content: [{ type: 'text', text: 'Continue' }] }], + }; + await survivingContextHook(event); + const count = event.messages.flatMap((message) => message.content).filter( + (part) => part.type === 'text' && part.text.startsWith('\nYou have superpowers.') + ).length; + if (count !== 1) failures.push(`expected surviving bootstrap once, got ${count}`); +} + +// --- Run 3: quoted and multi-line frontmatter values --------------------- +// The description is what the host shows in its skill list. A quoted value +// that wraps onto indented continuation lines must register as one unquoted +// line, so exercise each layout against a synthetic install: a copy of the +// plugin next to fixture skills, laid out like a real package root. +const frontmatterFixtures = { + 'multi-line-double': { + frontmatter: 'description: "Use when foo happens\n and bar continues\n and baz ends"', + expected: 'Use when foo happens and bar continues and baz ends', + }, + 'multi-line-single': { + frontmatter: "description: 'Use when foo happens\n and bar continues\n and baz ends'", + expected: 'Use when foo happens and bar continues and baz ends', + }, + 'single-line-quoted': { + frontmatter: 'description: "Plain quoted"', + expected: 'Plain quoted', + }, + 'block-scalar': { + frontmatter: 'description: >\n Folded line one\n line two', + expected: 'Folded line one line two', + }, +}; +const fixtureRoot = fs.mkdtempSync(path.join(os.tmpdir(), 'superpowers-frontmatter-')); +try { + const fixturePlugin = path.join(fixtureRoot, '.opencode', 'plugins', 'superpowers.js'); + fs.mkdirSync(path.dirname(fixturePlugin), { recursive: true }); + fs.copyFileSync(pluginPath, fixturePlugin); + for (const [id, { frontmatter }] of Object.entries(frontmatterFixtures)) { + const skillDir = path.join(fixtureRoot, 'skills', id); + fs.mkdirSync(skillDir, { recursive: true }); + fs.writeFileSync(path.join(skillDir, 'SKILL.md'), `---\nname: ${id}\n${frontmatter}\n---\n# Title\n\nBody.\n`); + } + const fixtureMod = await import(pathToFileURL(fixturePlugin).href); + const fixtureAdded = []; + await fixtureMod.default.setup(makeCtx({ add: (skill) => fixtureAdded.push(skill) })); + for (const [id, { expected }] of Object.entries(frontmatterFixtures)) { + const skill = fixtureAdded.find((s) => s.id === id); + if (!skill) { + failures.push(`fixture "${id}": expected setup() to register it`); + continue; + } + if (skill.description !== expected) { + failures.push(`fixture "${id}": expected description ${JSON.stringify(expected)}, got ${JSON.stringify(skill.description)}`); + } + if (skill.content.startsWith('---')) { + failures.push(`fixture "${id}": content still starts with the frontmatter delimiter`); + } + } +} finally { + fs.rmSync(fixtureRoot, { recursive: true, force: true }); +} + +const result = { + registered: added.length, + ids: added.map((s) => s.id), + allPathsValid: added.every((s) => s.path === path.join(skillsDir, s.id, 'SKILL.md') && fs.existsSync(s.path)), + staleLocationField: added.some((s) => 'location' in s), + hostileRejectedId: hostileId, + survivedHostileAdd: JSON.stringify(survived.sort()) === JSON.stringify(added.map((s) => s.id).filter((id) => id !== hostileId).sort()), +}; + +if (failures.length > 0) { + console.error(JSON.stringify(result, null, 2)); + for (const failure of failures) { + console.error(`FAIL: ${failure}`); + } + process.exit(1); +} + +console.log(JSON.stringify(result, null, 2)); + +function makeCtx({ add, onHook = () => {} }) { + return { + skill: { + transform: async (fn) => { + await fn({ list: () => [], get: () => undefined, add, update: () => {}, remove: () => {} }); + }, + }, + session: { + hook: async (name, callback) => onHook(name, callback), + get: async ({ sessionID }) => ({ id: sessionID }), // top-level: no parentID + }, + }; +} diff --git a/tests/opencode/test-skill-registration.sh b/tests/opencode/test-skill-registration.sh new file mode 100755 index 000000000..ac882a4bb --- /dev/null +++ b/tests/opencode/test-skill-registration.sh @@ -0,0 +1,22 @@ +#!/usr/bin/env bash +# Test: V2 Skill Registration Contract (#2106 review) +# Verifies setup() registers skills matching OpenCode 2.0.4's Skill.Info +# schema (path field, no stale location/slash) and contains per-skill +# draft.add() failures instead of aborting registration. +set -euo pipefail + +SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)" + +echo "=== Test: V2 Skill Registration Contract ===" + +source "$SCRIPT_DIR/setup.sh" +trap cleanup_test_env EXIT + +node "$SCRIPT_DIR/test-skill-registration.mjs" "$SUPERPOWERS_PLUGIN_FILE" +node "$SCRIPT_DIR/test-skill-registration.mjs" "$OPENCODE_CONFIG_DIR/plugins/superpowers.js" + +echo " [PASS] Skill payloads match the 2.0.4 Skill.Info contract" +echo " [PASS] A rejected draft.add() skips one skill without aborting the rest" +echo " [PASS] Quoted and multi-line frontmatter values register unquoted" +echo "" +echo "=== All skill registration tests passed ==="