Compare commits

..

11 Commits

16 changed files with 106 additions and 583 deletions

6
.gitignore vendored
View File

@@ -11,9 +11,3 @@ triage/
# development (see CLAUDE.md / README.md). It is not part of the published
# plugin, so the whole directory is ignored here.
evals/
# Python
__pycache__/
*.pyc
*.pyo
.pytest_cache/

View File

@@ -1,104 +0,0 @@
import os
import re
from pathlib import Path
BOOTSTRAP_MARKER = "superpowers:using-superpowers bootstrap for hermes"
def _skills_dir() -> str:
"""Locate the stock skills/ tree for either supported install layout.
- git-clone install (`hermes plugins install obra/superpowers`): the plugin
dir is the repo root, so `.hermes-plugin/` and `skills/` are siblings and
this module resolves `../skills`.
- flattened install (plugin files copied to the plugin dir root): `skills/`
sits next to this module.
Raises loudly when neither matches — a bootstrap that silently skips is how
a broken install masquerades as a working one.
"""
here = os.path.dirname(os.path.realpath(__file__))
candidates = (
os.path.realpath(os.path.join(here, "..", "skills")),
os.path.realpath(os.path.join(here, "skills")),
)
for cand in candidates:
if os.path.isfile(os.path.join(cand, "using-superpowers", "SKILL.md")):
return cand
raise RuntimeError(
"superpowers plugin: cannot find the skills/ tree "
f"(looked at {candidates}). Reinstall with "
"`hermes plugins install obra/superpowers`."
)
def _strip_frontmatter(content: str) -> str:
match = re.match(r"^---\n[\s\S]*?\n---\n([\s\S]*)$", content)
return (match.group(1) if match else content).strip()
def _build_bootstrap(skills_dir: str) -> str:
with open(
os.path.join(skills_dir, "using-superpowers", "SKILL.md"),
encoding="utf-8",
) as f:
body = _strip_frontmatter(f.read())
tools_path = os.path.join(
skills_dir, "using-superpowers", "references", "hermes-tools.md"
)
with open(tools_path, encoding="utf-8") as f:
tool_mapping = f.read().strip()
return (
f"<EXTREMELY_IMPORTANT>\n"
f"{BOOTSTRAP_MARKER}\n\n"
f"You have superpowers.\n\n"
f"The using-superpowers skill content is included below and is already "
f"loaded for this Hermes session. Follow it now. "
f"Do not try to load using-superpowers again.\n\n"
f"{body}\n\n"
f"## Loading Superpowers Skills on Hermes\n\n"
f"Superpowers skills are registered with Hermes' native skill loader: "
f'invoke one with `skill_view("superpowers:skill-name")` '
f'(for example `skill_view("superpowers:brainstorming")`). '
f"If a namespaced lookup returns 'not found', read the skill file "
f"directly instead:\n"
f'`read_file("{skills_dir}/skill-name/SKILL.md")`\n\n'
f"The superpowers skills directory is: `{skills_dir}`\n\n"
f"{tool_mapping}\n"
f"</EXTREMELY_IMPORTANT>"
)
def register(ctx):
skills_dir = _skills_dir()
bootstrap = _build_bootstrap(skills_dir)
# Register every stock skill with Hermes' native loader so skill_view can
# load them on demand. Standard markdown; no conversion (plugin guide).
# register_skill requires a pathlib.Path — a str raises AttributeError and
# hermes silently disables the whole plugin (verified 2026-07-23).
for name in sorted(os.listdir(skills_dir)):
skill_md = os.path.join(skills_dir, name, "SKILL.md")
if os.path.isfile(skill_md):
ctx.register_skill(name, Path(skill_md))
# pre_llm_call returning {"context": ...} is the documented injection path
# (on_session_start return values are ignored, and ctx.inject_message
# refuses from that hook — verified empirically 2026-07-23). The context is
# appended to the first turn's user message.
def pre_llm_call(
session_id=None,
user_message=None,
conversation_history=None,
is_first_turn=None,
model=None,
platform=None,
**kwargs,
):
if is_first_turn:
return {"context": bootstrap}
return None
ctx.register_hook("pre_llm_call", pre_llm_call)

View File

@@ -1,6 +0,0 @@
name: superpowers
version: 6.1.1
description: Superpowers skills and workflow bootstrap for Hermes Agent
author: obra
provides_hooks:
- pre_llm_call

View File

@@ -11,7 +11,7 @@ If this sounds like someone you know, definitely send them our way.
## Quickstart
Give your agent Superpowers: [Claude Code](#claude-code), [Antigravity](#antigravity), [Codex App](#codex-app), [Codex CLI](#codex-cli), [Cursor](#cursor), [Factory Droid](#factory-droid), [Gemini CLI](#gemini-cli), [GitHub Copilot CLI](#github-copilot-cli), [Hermes Agent](#hermes-agent), [Kimi Code](#kimi-code), [OpenCode](#opencode), [Pi](#pi).
Give your agent Superpowers: [Claude Code](#claude-code), [Antigravity](#antigravity), [Codex App](#codex-app), [Codex CLI](#codex-cli), [Cursor](#cursor), [Factory Droid](#factory-droid), [Gemini CLI](#gemini-cli), [GitHub Copilot CLI](#github-copilot-cli), [Kimi Code](#kimi-code), [OpenCode](#opencode), [Pi](#pi).
## How it works
@@ -199,18 +199,6 @@ pi -e /path/to/superpowers
The Pi package loads the Superpowers skills and a small extension that injects the `using-superpowers` bootstrap at session startup and again after compaction. Pi has native skills, so no compatibility `Skill` tool is required. Subagent and task-list tools remain optional Pi companion packages.
### Hermes Agent
Install Superpowers as a Hermes plugin from this repository:
```bash
hermes plugins install obra/superpowers --enable
```
Restart any active Hermes sessions after installing. Note: Hermes has no
post-compaction hook, so a very long session that compacts over its first
turn loses the bootstrap — start a fresh session if skills stop triggering.
## The Basic Workflow
1. **brainstorming** - Activates before writing code. Refines rough ideas through questions, explores alternatives, presents design in sections for validation. Saves design document.

View File

@@ -133,6 +133,15 @@ a ledger file, not only in todos.
plan's progress: leave it in place and start your own, fresh.
- Create the ledger with its identity as the first line:
`# SDD ledger — plan: <plan file path>`.
- During that same Setup read, copy the plan's Global Constraints section
verbatim into `<workspace>/constraints.md`. Every dispatch's binding
constraint values paste from that file — the plan itself stays closed
after Setup, even across compaction.
- The workspace is never committed to the project repo. Do not `git add`
anything under it, and never write a commit whose purpose is to record,
correct, or tidy a workspace artifact — reports and ledgers are session
records, not deliverables. If the repo lacks a `.gitignore` entry for
`.superpowers/`, leave the directory untracked rather than committing it.
- The ledger is your recovery map: the commits it names exist in git even
when your context no longer remembers creating them. After compaction,
trust the ledger and `git log` over your own recollection.
@@ -140,7 +149,11 @@ a ledger file, not only in todos.
that happens, recover from `git log`.
Read the plan once, note its context and Global Constraints, and create a
todo per task.
todo per task. That is the plan's one full read for the whole session:
after Setup, the ledger and `scripts/task-brief` extracts are your working
memory — re-reading the plan or spec late in the run (to "double-check"
completion, to rebuild the final-review dispatch) re-buys context you
already paid for and is forbidden.
Before dispatching Task 1, scan the plan once for conflicts:
@@ -195,7 +208,10 @@ that implementer. Single-file mechanical fixes also take the cheapest tier.
Everything you paste into a dispatch prompt — and everything a subagent
prints back — stays resident in your context for the rest of the session
and is re-read on every later turn. Hand artifacts over as files.
and is re-read on every later turn. Hand artifacts over as files. The same
tax applies to your own words: checkpoint in one short line, keep
bookkeeping in the ledger file, and never paste back into the conversation
what a file already holds.
### 1. Dispatch the implementer
@@ -211,9 +227,15 @@ and fix-round diffs need it.
first — it is your requirements, with the exact values to use verbatim";
(3) interfaces and decisions from earlier tasks that the brief cannot
know; (4) your resolution of any ambiguity you noticed in the brief;
(5) the report-file path and report contract. Exact values (numbers,
magic strings, signatures, test cases) appear only in the brief. Never
make a subagent read the whole plan file.
(5) the report-file path and report contract; (6) the plan's binding
constraint values — any Global Constraint that mandates an exact
mechanical form (commit-message rules, naming rules, fixed literals) —
pasted verbatim. Task-specific exact values (numbers, magic strings,
signatures, test cases) appear only in the brief; binding constraint
values are the one exception — they ride in EVERY dispatch, because a
subagent that must recall a constraint from memory will reconstruct it
from its own priors instead. Never make a subagent read the whole plan
file.
- **Report file:** name the implementer's report file after the brief
(brief `…/task-N-brief.md` → report `…/task-N-report.md`) and put it in
the dispatch prompt. The implementer writes the full report there and
@@ -273,6 +295,10 @@ needed.
- **Reviewer inputs:** the task reviewer gets three paths — the same brief
file, the report file, and the review package — plus the global
constraints that bind the task.
- **Persist the verdict:** when the review returns, write its full text to
`<workspace>/task-<N>-review.md` before acting on it. The file is the
gate: completion checks for the artifact, not for your memory of a
verdict.
- The global-constraints block you hand the reviewer is its attention
lens. Copy the binding requirements verbatim from the plan's Global
Constraints section or the spec: exact values, exact formats, and the
@@ -323,22 +349,26 @@ scoped re-review. Five rounds maximum per task:
verbatim. Its context is intact: it knows the task, the code, and its own
choices. If your harness cannot send another message to a live subagent,
dispatch a fresh implementer carrying the brief path, the report-file path,
and the findings — the report file is the persistent memory either way.
the findings, and the binding constraint values verbatim — the report file
is the persistent memory either way.
**Rounds 4-5 — dispatch a fresh implementer on a more capable model** (per
Model Selection), with the brief path, the report-file path, the open
findings, and this framing: "A prior implementer attempted this task
[N] times; you own it now. Read the report file for what was tried." A loop
that survives three resumes usually means the implementer cannot see its
own problem — fresh eyes and a capability bump in one move.
findings, the binding constraint values verbatim, and this framing: "A
prior implementer attempted this task [N] times; you own it now. Read the
report file for what was tried." A loop that survives three resumes usually
means the implementer cannot see its own problem — fresh eyes and a
capability bump in one move.
**Every round, either way:** the implementer fixes, re-runs the tests
covering the amended code, appends its fix report to the same report file,
and returns the short contract. Before re-dispatching the reviewer, confirm
the fix report contains the covering tests, the command run, and the
output; dispatch the re-review once all three are present. Name the
covering test files in the fix message — a one-line fix does not need the
whole suite.
the fix report contains the covering tests, the command run, and the output;
dispatch the re-review once all three are present. Name the covering test
files in the fix message — a one-line fix does not need the whole suite.
Append the re-review's returned text to `<workspace>/task-<N>-review.md` as
well — the artifact accumulates every verdict, and the file's last entry is
the one completion relies on.
**The re-review is scoped.** Run `scripts/review-package PLAN_FILE FIX_BASE HEAD`
where FIX_BASE is the head the previous review saw, and dispatch
@@ -383,10 +413,20 @@ message as your other bookkeeping:
- `Task <N>: complete (commits <base7>..<head7>, review clean)`
- `Task <N>: complete (commits <base7>..<head7>, <K> parked)` after a
tripped breaker
- `Task <N>: complete (commits <base7>..<head7>, deviation parked: <rule>)`
when a landed commit violates a mechanical constraint that nothing
downstream builds on. Record the ruling in the ledger. A green build with
a parked, recorded deviation is complete — do not fail the task, and do
not rewrite landed history to chase cosmetics. A load-bearing violation
is different: that is a BLOCKED, not a deviation. This valve is not the
breaker: it needs no exhausted fix rounds — a mechanical deviation
discovered at completion parks here directly, with its ruling in the ledger.
Then mark the todo complete and move on. Never move to the next task while
the review has open Critical/Important issues that are neither fixed nor
parked-with-ruling at the cap.
Then mark the todo complete and move on — but only once
`<workspace>/task-<N>-review.md` exists; a completion line without its review
artifact is invalid, whatever you remember about the review. Never move to
the next task while the review has open Critical/Important issues that are
neither fixed nor parked-with-ruling at the cap.
## Final Review
@@ -399,17 +439,25 @@ on the most capable available model (see Model Selection), using
superpowers:requesting-code-review's
[code-reviewer.md](../requesting-code-review/code-reviewer.md). Point it at
the ledger's deferred-minor and parked lines so it can triage which must be
fixed before merge.
fixed before merge. Build that dispatch from the ledger alone — the
completion lines, parked rulings, and deferred minors are the whole-run
summary; do not re-read the plan, the spec, or per-task reports to
reconstruct what the ledger already states. Write the returned review to
`<workspace>/final-review.md` before dispatching any fix wave — the merge
decision cites the artifact, not a recollection.
If the final whole-branch review returns findings, dispatch ONE fix subagent
with the complete findings list — not one fixer per finding.
with the complete findings list and the binding constraint values
verbatim — not one fixer per finding.
Per-finding fixers each rebuild context and re-run suites; a real
session's final-review fix wave cost more than all its tasks combined.
Then run exactly one scoped re-review of the fix wave
(`scripts/review-package PLAN_FILE FIX_BASE HEAD` over the fix range,
[re-review-prompt.md](re-review-prompt.md)).
Adjudicate any residual findings as in the task loop's breaker: park with
rulings, or stop on load-bearing ones. There is no second fix wave —
rulings, or stop on load-bearing ones. The same valve applies to
mechanical-constraint misses discovered at the end: non-load-bearing means
parked with a ruling, not a failed branch. There is no second fix wave —
residual load-bearing findings surface to your human partner when
finishing-a-development-branch presents the options.

View File

@@ -15,6 +15,12 @@ Subagent (general-purpose):
Read your task brief first: [BRIEF_FILE]
It contains the full task text from the plan.
## Binding Constraint Values
[CONSTRAINT_VALUES — the plan's mechanical constraints, pasted verbatim
by the dispatcher. If a rule here mandates an exact form (commit-message
text, naming, fixed literals), reproduce it exactly — never from memory.]
## Context
[Scene-setting: where this fits, dependencies, architectural context]
@@ -36,8 +42,12 @@ Subagent (general-purpose):
2. Write tests (following TDD if task says to)
3. Verify implementation works
4. Commit your work
5. Self-review (see below)
6. Report back
5. Immediately after each commit, verify it against the Binding Constraint
Values above (commit-message rules, naming rules, fixed literals) while
history is still local. A miss is cheap now — amend or forward-fix at
once — and expensive after your work is delivered.
6. Self-review (see below)
7. Report back
Work from: [directory]
@@ -96,6 +106,10 @@ Subagent (general-purpose):
- Did I only build what was requested?
- Did I follow existing patterns in the codebase?
**Constraints:**
- Does every commit message satisfy the Binding Constraint Values exactly?
- Did I reproduce mandated literals from the constraint text, not from memory?
**Testing:**
- Do tests actually verify behavior (not just mock behavior)?
- Did I follow TDD if required?
@@ -104,6 +118,12 @@ Subagent (general-purpose):
If you find issues during self-review, fix them now before reporting.
If a constraint violation is already in a landed commit you cannot safely
amend, forward-fix it in a new commit when possible; when it is not, report
the deviation explicitly with Status DONE_WITH_CONCERNS — never report the
task incomplete solely for a cosmetic miss on an otherwise green build.
Whether the deviation parks or blocks is the dispatching controller's call.
## After Review Findings
If the task review finds issues, you will be resumed with the findings.
@@ -136,7 +156,8 @@ Subagent (general-purpose):
If BLOCKED or NEEDS_CONTEXT, put the specifics in the final message
itself — the controller acts on it directly.
Use DONE_WITH_CONCERNS if you completed the work but have doubts about correctness.
Use DONE_WITH_CONCERNS if you completed the work but have doubts about correctness, or
when you are reporting a known deviation you could not safely fix.
Use BLOCKED if you cannot complete the task. Use NEEDS_CONTEXT if you need
information that wasn't provided. Never silently produce work you're unsure about.
```

View File

@@ -18,18 +18,9 @@ echo "🔍 Searching for test that creates: $POLLUTION_CHECK"
echo "Test pattern: $TEST_PATTERN"
echo ""
# Get list of test files (find . emits ./-prefixed paths, so accept the
# pattern written with or without a leading ./)
TEST_PATTERN="${TEST_PATTERN#./}"
# find -path can't match '**/' against zero directory levels, so a pattern
# like src/**/*.test.ts would skip src/top.test.ts; also try the pattern
# with '**/' collapsed to cover files directly under the base directory.
TEST_FILES=$(find . \( -path "./$TEST_PATTERN" -o -path "./${TEST_PATTERN//\*\*\//}" \) | sort -u)
if [ -z "$TEST_FILES" ]; then
TOTAL=0
else
TOTAL=$(printf '%s\n' "$TEST_FILES" | wc -l | tr -d ' ')
fi
# Get list of test files
TEST_FILES=$(find . -path "$TEST_PATTERN" | sort)
TOTAL=$(echo "$TEST_FILES" | wc -l | tr -d ' ')
echo "Found $TOTAL test files"
echo ""

View File

@@ -56,7 +56,6 @@ If your harness appears here, read its reference file for special instructions:
- Codex: `references/codex-tools.md`
- Pi: `references/pi-tools.md`
- Antigravity: `references/antigravity-tools.md`
- Hermes Agent: `references/hermes-tools.md`
## User Instructions

View File

@@ -4,7 +4,7 @@ Skills speak in actions ("dispatch a subagent", "create a todo", "read a file").
| Action skills request | Antigravity CLI equivalent |
|----------------------|----------------------|
| Dispatch a subagent (`Subagent (general-purpose):` template) | `invoke_subagent` with a built-in `TypeName``self` for full-capability work, `research` for read-only |
| Dispatch a subagent (`Subagent (general-purpose):` template) | `invoke_subagent` with a built-in `TypeName``self` for full-capability work, `research` for read-only (see [Subagent support](#subagent-support)) |
| Task tracking ("create a todo", "mark complete") | a **task artifact**`write_to_file` with `IsArtifact: true` and `ArtifactType: "task"` (see [Task tracking](#task-tracking)). **Not** `manage_task`, which manages background processes. |
## Task tracking

View File

@@ -1,56 +0,0 @@
# Hermes Agent Tool Mapping
Skills speak in actions ("dispatch a subagent", "create a todo", "read a file"). On Hermes Agent these resolve to the tools below.
## Tools
| Action skills request | Hermes tool |
|---|---|
| Read a file | `read_file` |
| Create a new file | `write_file` |
| Edit a file (targeted patch) | `patch` |
| Run a shell command | `terminal` |
| Search file contents | `search_files` |
| Find files by name | `terminal` with `find` |
| Fetch a URL / read a webpage | `web_extract(urls=[...])` |
| Search the web | `web_search(query=...)` |
| Dispatch a subagent | `delegate_task(goal=..., context=..., toolsets=[...], role="leaf")` |
| Task tracking | `todo` tool |
| Invoke a skill | `skill_view("skill-name")` |
## Instructions file
When a skill mentions "your instructions file," on Hermes Agent this is **`AGENTS.md`** in the project directory, or **`SOUL.md`** globally at `~/.hermes/SOUL.md`.
## Invoking a skill
Hermes Agent has a `skills` toolset with `skill_view` and `skills_list` tools.
To invoke a superpowers skill, use:
```
skill_view("brainstorming")
skill_view("test-driven-development")
```
If `skill_view` cannot find a superpowers skill (it may not appear in the catalog
until the plugin fully registers it), fall back to reading the SKILL.md directly:
```
read_file(path="~/.hermes/plugins/superpowers/skills/<skill-name>/SKILL.md")
```
This fallback is the same mechanism used by other harnesses without native skill loading.
## Subagent dispatch
Use `delegate_task` to spawn isolated subagents for parallel or sequential workstreams:
```
delegate_task(goal="...", context="...", toolsets=[...], role="leaf")
```
If `delegate_task` is unavailable, do the work inline rather than inventing tool calls.
## Task tracking
Use the `todo` tool for task tracking within a session. For multi-agent task boards, use `hermes kanban` CLI if available. Treat older `TodoWrite` references as the task-tracking action.

View File

@@ -71,7 +71,11 @@ independently testable deliverable.
[The spec's project-wide requirements — version floors, dependency limits,
naming and copy rules, platform requirements — one line each, with exact
values copied verbatim from the spec. Every task's requirements implicitly
include this section.]
include this section. Three things may never enter it: cosmetic absolutes
on every commit (a fixed trailer or byline the work does not need), your
own identity or model name promoted into a rule, and environment
constraints (versions, platforms, paths) you have not verified against the
environment the plan will execute in.]
---
```
@@ -125,6 +129,8 @@ git commit -m "feat: add specific feature"
```
````
Commit messages describe the change. Never mandate session boilerplate — trailers, bylines, model names — as a per-commit rule; what your session stamps on its commits is not a requirement of the work.
## No Placeholders
Every step must contain the actual content an engineer needs. These are **plan failures** — never write them:
@@ -145,6 +151,8 @@ After writing the complete plan, look at the spec with fresh eyes and check the
**3. Type consistency:** Do the types, method signatures, and property names you used in later tasks match what you defined in earlier tasks? A function called `clearLayers()` in Task 3 but `clearFullLayers()` in Task 7 is a bug.
**4. Constraint hygiene:** Does any Global Constraint mandate a per-commit cosmetic absolute, name the authoring model or session, or assert an environment fact (version floor, platform, path) you did not verify? Cut or verify it.
If you find issues, fix them inline. No need to re-review — just fix and move on. If you find a spec requirement with no task, add the task.
## Execution Handoff

View File

@@ -1,30 +0,0 @@
from pathlib import Path
import pytest
from unittest.mock import MagicMock
@pytest.fixture
def mock_ctx():
ctx = MagicMock()
ctx._hooks = {}
ctx._skills = {}
def register_hook(event, fn):
ctx._hooks[event] = fn
def register_skill(name, path):
# Mimic hermes' real register_skill, which calls path.exists() and
# therefore breaks on a str (the bug that silently disabled the whole
# plugin, found 2026-07-23). Keeping that fidelity here means a
# regression to str paths fails these tests instead of failing
# silently inside hermes.
if not isinstance(path, Path):
raise AttributeError(
f"register_skill requires a pathlib.Path, got {type(path).__name__}"
)
ctx._skills[name] = path
ctx.register_hook.side_effect = register_hook
ctx.register_skill.side_effect = register_skill
return ctx

View File

@@ -1,98 +0,0 @@
import importlib
import os
import sys
import pytest
sys.path.insert(0, os.path.abspath(
os.path.join(os.path.dirname(__file__), "../../.hermes-plugin")
))
BOOTSTRAP_MARKER = "superpowers:using-superpowers bootstrap for hermes"
# Hermes spills injected context over 10,000 chars to a file, which breaks
# inline injection semantics. The bootstrap must stay under it with margin.
HERMES_CONTEXT_SPILL_LIMIT = 10_000
def _load():
if "__init__" in sys.modules:
del sys.modules["__init__"]
return importlib.import_module("__init__")
def _bootstrap():
m = _load()
return m._build_bootstrap(m._skills_dir())
class TestStripFrontmatter:
def test_strips_yaml_block(self):
m = _load()
content = "---\nname: foo\ndescription: bar\n---\n# Body\nContent here"
assert m._strip_frontmatter(content) == "# Body\nContent here"
def test_no_frontmatter_returns_trimmed_content(self):
m = _load()
content = "# No frontmatter\nJust content"
assert m._strip_frontmatter(content) == "# No frontmatter\nJust content"
def test_strips_surrounding_whitespace_from_body(self):
m = _load()
content = "---\nname: foo\n---\n\n\n# Body\n\n"
assert m._strip_frontmatter(content) == "# Body"
class TestSkillsDirResolution:
def test_repo_layout_resolves(self):
# The repo checkout IS the git-clone layout: .hermes-plugin/ and
# skills/ are siblings, so resolution must succeed from here.
m = _load()
skills = m._skills_dir()
assert os.path.isfile(
os.path.join(skills, "using-superpowers", "SKILL.md")
)
class TestBootstrapContent:
def test_marker_and_wrapper(self):
content = _bootstrap()
assert BOOTSTRAP_MARKER in content
assert content.startswith("<EXTREMELY_IMPORTANT>")
assert content.rstrip().endswith("</EXTREMELY_IMPORTANT>")
def test_contains_using_superpowers_body(self):
content = _bootstrap()
# A distinctive line from the skill body proves the real SKILL.md was
# embedded, not a stub.
assert "You have superpowers" in content
assert "## The Rule" in content
def test_frontmatter_stripped(self):
content = _bootstrap()
assert "---\nname:" not in content
def test_tool_mapping_sourced_from_reference_file(self):
m = _load()
content = _bootstrap()
ref = os.path.join(
m._skills_dir(), "using-superpowers", "references", "hermes-tools.md"
)
with open(ref, encoding="utf-8") as f:
ref_text = f.read().strip()
# The mapping is included verbatim from the reference file — the
# single source, not a drift-prone inline copy.
assert ref_text in content
assert "read_file" in content
def test_skill_view_guidance_present(self):
content = _bootstrap()
assert 'skill_view("superpowers:brainstorming")' in content
def test_under_hermes_context_spill_limit(self):
content = _bootstrap()
assert len(content) < HERMES_CONTEXT_SPILL_LIMIT, (
f"bootstrap is {len(content)} chars; hermes spills injected "
f"context over {HERMES_CONTEXT_SPILL_LIMIT} to a file, which "
"breaks inline injection"
)

View File

@@ -1,142 +0,0 @@
import importlib
import importlib.util
import os
import shutil
import sys
from pathlib import Path
import pytest
# Point at the plugin directory
_PLUGIN_DIR = os.path.abspath(
os.path.join(os.path.dirname(__file__), "../../.hermes-plugin")
)
sys.path.insert(0, _PLUGIN_DIR)
BOOTSTRAP_MARKER = "superpowers:using-superpowers bootstrap for hermes"
def _load_plugin():
"""Re-import plugin module fresh."""
if "__init__" in sys.modules:
del sys.modules["__init__"]
return importlib.import_module("__init__")
def _fire_pre_llm(ctx, **kwargs):
hook = ctx._hooks["pre_llm_call"]
defaults = {
"session_id": "s1",
"user_message": "hi",
"conversation_history": [],
"is_first_turn": False,
"model": "test-model",
"platform": "cli",
}
defaults.update(kwargs)
return hook(**defaults)
class TestPluginRegistration:
def test_register_attaches_only_pre_llm_call_hook(self, mock_ctx):
plugin = _load_plugin()
plugin.register(mock_ctx)
assert list(mock_ctx._hooks.keys()) == ["pre_llm_call"]
def test_register_registers_every_stock_skill_as_path(self, mock_ctx):
plugin = _load_plugin()
plugin.register(mock_ctx)
# The conftest mock raises on non-Path (mirroring hermes' real
# register_skill), so reaching these asserts proves every
# registration passed a pathlib.Path.
assert "using-superpowers" in mock_ctx._skills
assert "brainstorming" in mock_ctx._skills
for name, path in mock_ctx._skills.items():
assert isinstance(path, Path)
assert path.name == "SKILL.md"
assert path.parent.name == name
assert path.is_file()
def test_registered_skills_match_skill_directories(self, mock_ctx):
plugin = _load_plugin()
plugin.register(mock_ctx)
skills_root = plugin._skills_dir()
expected = {
entry
for entry in os.listdir(skills_root)
if os.path.isfile(os.path.join(skills_root, entry, "SKILL.md"))
}
assert set(mock_ctx._skills.keys()) == expected
class TestBootstrapInjection:
def test_first_turn_returns_bootstrap_context(self, mock_ctx):
plugin = _load_plugin()
plugin.register(mock_ctx)
result = _fire_pre_llm(mock_ctx, is_first_turn=True)
assert isinstance(result, dict)
content = result["context"]
assert BOOTSTRAP_MARKER in content
assert content.startswith("<EXTREMELY_IMPORTANT>")
assert content.rstrip().endswith("</EXTREMELY_IMPORTANT>")
def test_later_turns_return_none(self, mock_ctx):
plugin = _load_plugin()
plugin.register(mock_ctx)
assert _fire_pre_llm(mock_ctx, is_first_turn=False) is None
assert _fire_pre_llm(mock_ctx, is_first_turn=None) is None
def test_hook_tolerates_future_kwargs(self, mock_ctx):
plugin = _load_plugin()
plugin.register(mock_ctx)
result = _fire_pre_llm(
mock_ctx, is_first_turn=True, telemetry_schema_version=3
)
assert BOOTSTRAP_MARKER in result["context"]
class TestLayoutResolution:
def _stage(self, tmp_path, layout):
"""Copy the plugin module + a minimal skills tree in the given layout."""
src_skills = Path(_PLUGIN_DIR).parent / "skills"
if layout == "clone":
plugdir = tmp_path / "superpowers" / ".hermes-plugin"
else: # flat: module at the plugin dir root, skills nested inside it
plugdir = tmp_path / "superpowers"
skills = tmp_path / "superpowers" / "skills"
plugdir.mkdir(parents=True, exist_ok=True)
shutil.copy(Path(_PLUGIN_DIR) / "__init__.py", plugdir / "__init__.py")
for skill in ("using-superpowers", "brainstorming"):
shutil.copytree(src_skills / skill, skills / skill)
return plugdir
def _load_from(self, plugdir):
spec = importlib.util.spec_from_file_location(
f"hermes_plugin_test_{plugdir.parent.name}_{plugdir.name}",
plugdir / "__init__.py",
)
mod = importlib.util.module_from_spec(spec)
spec.loader.exec_module(mod)
return mod
def test_clone_layout_resolves_sibling_skills(self, tmp_path, mock_ctx):
# git-clone install: .hermes-plugin/ and skills/ are siblings.
plugdir = self._stage(tmp_path, "clone")
mod = self._load_from(plugdir)
mod.register(mock_ctx)
assert "using-superpowers" in mock_ctx._skills
def test_flat_layout_resolves_nested_skills(self, tmp_path, mock_ctx):
# flattened install: module at the plugin dir root, skills/ inside it.
plugdir = self._stage(tmp_path, "flat")
mod = self._load_from(plugdir)
mod.register(mock_ctx)
assert "using-superpowers" in mock_ctx._skills
def test_missing_skills_raises_loudly(self, tmp_path, mock_ctx):
plugdir = tmp_path / "superpowers"
plugdir.mkdir(parents=True)
shutil.copy(Path(_PLUGIN_DIR) / "__init__.py", plugdir / "__init__.py")
mod = self._load_from(plugdir)
with pytest.raises(RuntimeError, match="cannot find the skills"):
mod.register(mock_ctx)

View File

@@ -1,90 +0,0 @@
#!/usr/bin/env bash
set -euo pipefail
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
REPO_ROOT="$(cd "$SCRIPT_DIR/../.." && pwd)"
SCRIPT_UNDER_TEST="$REPO_ROOT/skills/systematic-debugging/find-polluter.sh"
FAILURES=0
TEST_ROOT="$(mktemp -d)"
cleanup() {
rm -rf "$TEST_ROOT"
}
trap cleanup EXIT
pass() {
echo " [PASS] $1"
}
fail() {
echo " [FAIL] $1"
FAILURES=$((FAILURES + 1))
}
assert_contains() {
local haystack="$1"
local needle="$2"
local description="$3"
if printf '%s' "$haystack" | grep -Fq -- "$needle"; then
pass "$description"
else
fail "$description (expected output to contain: $needle)"
fi
}
# Toy project: one top-level test, one nested test. A stubbed `npm` on PATH
# creates the pollution marker whenever any test runs, so the first test file
# executed is always identified as the polluter.
setup_project() {
PROJECT="$TEST_ROOT/project"
rm -rf "$PROJECT"
mkdir -p "$PROJECT/src/feature" "$PROJECT/bin"
echo "test('top')" > "$PROJECT/src/top.test.ts"
echo "test('nested')" > "$PROJECT/src/feature/nested.test.ts"
cat > "$PROJECT/bin/npm" <<'EOF'
#!/usr/bin/env bash
touch pollution.marker
EOF
chmod +x "$PROJECT/bin/npm"
}
# run_polluter <pattern> — runs the script in the toy project with the stub
# npm first on PATH; captures combined output, never aborts on exit code.
run_polluter() {
local pattern="$1"
rm -f "$PROJECT/pollution.marker"
(
cd "$PROJECT"
PATH="$PROJECT/bin:$PATH" "$SCRIPT_UNDER_TEST" 'pollution.marker' "$pattern" 2>&1
) || true
}
echo "Test: documented pattern finds nested test files (issue #2008)"
setup_project
OUTPUT="$(run_polluter 'src/**/*.test.ts')"
assert_contains "$OUTPUT" "FOUND POLLUTER" "documented pattern runs tests and detects pollution"
echo "Test: documented pattern also finds top-level test files"
setup_project
OUTPUT="$(run_polluter 'src/**/*.test.ts')"
assert_contains "$OUTPUT" "Found 2 test files" "src/**/*.test.ts matches src/top.test.ts and src/feature/nested.test.ts"
echo "Test: ./-prefixed pattern matches the same files"
setup_project
OUTPUT="$(run_polluter './src/**/*.test.ts')"
assert_contains "$OUTPUT" "Found 2 test files" "leading ./ on the pattern is accepted"
echo "Test: non-matching pattern reports an honest zero"
setup_project
OUTPUT="$(run_polluter 'nomatch/**/*.test.ts')"
assert_contains "$OUTPUT" "Found 0 test files" "empty result counts as 0, not 1"
assert_contains "$OUTPUT" "No polluter found" "empty result exits via the clean path"
echo ""
if [ "$FAILURES" -gt 0 ]; then
echo "$FAILURES test(s) failed"
exit 1
fi
echo "All tests passed"