mirror of
https://github.com/obra/superpowers
synced 2026-08-06 17:54:20 +00:00
Compare commits
8 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 6819b42d97 | |||
| 9be44ebf40 | |||
| 695744056e | |||
| fb518edf7b | |||
| 3ff8d15f15 | |||
| b6613057ae | |||
| 178528c03e | |||
| 7b177613c0 |
@@ -11,3 +11,9 @@ triage/
|
|||||||
# development (see CLAUDE.md / README.md). It is not part of the published
|
# development (see CLAUDE.md / README.md). It is not part of the published
|
||||||
# plugin, so the whole directory is ignored here.
|
# plugin, so the whole directory is ignored here.
|
||||||
evals/
|
evals/
|
||||||
|
|
||||||
|
# Python
|
||||||
|
__pycache__/
|
||||||
|
*.pyc
|
||||||
|
*.pyo
|
||||||
|
.pytest_cache/
|
||||||
|
|||||||
@@ -0,0 +1,104 @@
|
|||||||
|
import os
|
||||||
|
import re
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
BOOTSTRAP_MARKER = "superpowers:using-superpowers bootstrap for hermes"
|
||||||
|
|
||||||
|
|
||||||
|
def _skills_dir() -> str:
|
||||||
|
"""Locate the stock skills/ tree for either supported install layout.
|
||||||
|
|
||||||
|
- git-clone install (`hermes plugins install obra/superpowers`): the plugin
|
||||||
|
dir is the repo root, so `.hermes-plugin/` and `skills/` are siblings and
|
||||||
|
this module resolves `../skills`.
|
||||||
|
- flattened install (plugin files copied to the plugin dir root): `skills/`
|
||||||
|
sits next to this module.
|
||||||
|
|
||||||
|
Raises loudly when neither matches — a bootstrap that silently skips is how
|
||||||
|
a broken install masquerades as a working one.
|
||||||
|
"""
|
||||||
|
here = os.path.dirname(os.path.realpath(__file__))
|
||||||
|
candidates = (
|
||||||
|
os.path.realpath(os.path.join(here, "..", "skills")),
|
||||||
|
os.path.realpath(os.path.join(here, "skills")),
|
||||||
|
)
|
||||||
|
for cand in candidates:
|
||||||
|
if os.path.isfile(os.path.join(cand, "using-superpowers", "SKILL.md")):
|
||||||
|
return cand
|
||||||
|
raise RuntimeError(
|
||||||
|
"superpowers plugin: cannot find the skills/ tree "
|
||||||
|
f"(looked at {candidates}). Reinstall with "
|
||||||
|
"`hermes plugins install obra/superpowers`."
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _strip_frontmatter(content: str) -> str:
|
||||||
|
match = re.match(r"^---\n[\s\S]*?\n---\n([\s\S]*)$", content)
|
||||||
|
return (match.group(1) if match else content).strip()
|
||||||
|
|
||||||
|
|
||||||
|
def _build_bootstrap(skills_dir: str) -> str:
|
||||||
|
with open(
|
||||||
|
os.path.join(skills_dir, "using-superpowers", "SKILL.md"),
|
||||||
|
encoding="utf-8",
|
||||||
|
) as f:
|
||||||
|
body = _strip_frontmatter(f.read())
|
||||||
|
|
||||||
|
tools_path = os.path.join(
|
||||||
|
skills_dir, "using-superpowers", "references", "hermes-tools.md"
|
||||||
|
)
|
||||||
|
with open(tools_path, encoding="utf-8") as f:
|
||||||
|
tool_mapping = f.read().strip()
|
||||||
|
|
||||||
|
return (
|
||||||
|
f"<EXTREMELY_IMPORTANT>\n"
|
||||||
|
f"{BOOTSTRAP_MARKER}\n\n"
|
||||||
|
f"You have superpowers.\n\n"
|
||||||
|
f"The using-superpowers skill content is included below and is already "
|
||||||
|
f"loaded for this Hermes session. Follow it now. "
|
||||||
|
f"Do not try to load using-superpowers again.\n\n"
|
||||||
|
f"{body}\n\n"
|
||||||
|
f"## Loading Superpowers Skills on Hermes\n\n"
|
||||||
|
f"Superpowers skills are registered with Hermes' native skill loader: "
|
||||||
|
f'invoke one with `skill_view("superpowers:skill-name")` '
|
||||||
|
f'(for example `skill_view("superpowers:brainstorming")`). '
|
||||||
|
f"If a namespaced lookup returns 'not found', read the skill file "
|
||||||
|
f"directly instead:\n"
|
||||||
|
f'`read_file("{skills_dir}/skill-name/SKILL.md")`\n\n'
|
||||||
|
f"The superpowers skills directory is: `{skills_dir}`\n\n"
|
||||||
|
f"{tool_mapping}\n"
|
||||||
|
f"</EXTREMELY_IMPORTANT>"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def register(ctx):
|
||||||
|
skills_dir = _skills_dir()
|
||||||
|
bootstrap = _build_bootstrap(skills_dir)
|
||||||
|
|
||||||
|
# Register every stock skill with Hermes' native loader so skill_view can
|
||||||
|
# load them on demand. Standard markdown; no conversion (plugin guide).
|
||||||
|
# register_skill requires a pathlib.Path — a str raises AttributeError and
|
||||||
|
# hermes silently disables the whole plugin (verified 2026-07-23).
|
||||||
|
for name in sorted(os.listdir(skills_dir)):
|
||||||
|
skill_md = os.path.join(skills_dir, name, "SKILL.md")
|
||||||
|
if os.path.isfile(skill_md):
|
||||||
|
ctx.register_skill(name, Path(skill_md))
|
||||||
|
|
||||||
|
# pre_llm_call returning {"context": ...} is the documented injection path
|
||||||
|
# (on_session_start return values are ignored, and ctx.inject_message
|
||||||
|
# refuses from that hook — verified empirically 2026-07-23). The context is
|
||||||
|
# appended to the first turn's user message.
|
||||||
|
def pre_llm_call(
|
||||||
|
session_id=None,
|
||||||
|
user_message=None,
|
||||||
|
conversation_history=None,
|
||||||
|
is_first_turn=None,
|
||||||
|
model=None,
|
||||||
|
platform=None,
|
||||||
|
**kwargs,
|
||||||
|
):
|
||||||
|
if is_first_turn:
|
||||||
|
return {"context": bootstrap}
|
||||||
|
return None
|
||||||
|
|
||||||
|
ctx.register_hook("pre_llm_call", pre_llm_call)
|
||||||
@@ -0,0 +1,6 @@
|
|||||||
|
name: superpowers
|
||||||
|
version: 6.2.0
|
||||||
|
description: Superpowers skills and workflow bootstrap for Hermes Agent
|
||||||
|
author: obra
|
||||||
|
provides_hooks:
|
||||||
|
- pre_llm_call
|
||||||
@@ -2,10 +2,35 @@
|
|||||||
|
|
||||||
Superpowers is a complete software development methodology for your coding agents, built on top of a set of composable skills and some initial instructions that make sure your agent uses them.
|
Superpowers is a complete software development methodology for your coding agents, built on top of a set of composable skills and some initial instructions that make sure your agent uses them.
|
||||||
|
|
||||||
|
## Table of Contents
|
||||||
|
|
||||||
|
- [Quickstart](#quickstart)
|
||||||
|
- [How it works](#how-it-works)
|
||||||
|
- [Commercial Services](#commercial-services)
|
||||||
|
- [Installation](#installation)
|
||||||
|
- [Claude Code](#claude-code)
|
||||||
|
- [Antigravity](#antigravity)
|
||||||
|
- [Codex App](#codex-app)
|
||||||
|
- [Codex CLI](#codex-cli)
|
||||||
|
- [Cursor](#cursor)
|
||||||
|
- [Factory Droid](#factory-droid)
|
||||||
|
- [Gemini CLI](#gemini-cli)
|
||||||
|
- [GitHub Copilot CLI](#github-copilot-cli)
|
||||||
|
- [Kimi Code](#kimi-code)
|
||||||
|
- [OpenCode](#opencode)
|
||||||
|
- [Pi](#pi)
|
||||||
|
- [The Basic Workflow](#the-basic-workflow)
|
||||||
|
- [Community](#community)
|
||||||
|
- [What's Inside](#whats-inside)
|
||||||
|
- [Philosophy](#philosophy)
|
||||||
|
- [Contributing](#contributing)
|
||||||
|
- [Updating](#updating)
|
||||||
|
- [License](#license)
|
||||||
|
- [Visual companion telemetry](#visual-companion-telemetry)
|
||||||
|
|
||||||
## Quickstart
|
## Quickstart
|
||||||
|
|
||||||
Give your agent Superpowers: [Claude Code](#claude-code), [Antigravity](#antigravity), [Codex App](#codex-app), [Codex CLI](#codex-cli), [Cursor](#cursor), [Factory Droid](#factory-droid), [Gemini CLI](#gemini-cli), [GitHub Copilot CLI](#github-copilot-cli), [Kimi Code](#kimi-code), [OpenCode](#opencode), [Pi](#pi).
|
Give your agent Superpowers: [Claude Code](#claude-code), [Antigravity](#antigravity), [Codex App](#codex-app), [Codex CLI](#codex-cli), [Cursor](#cursor), [Factory Droid](#factory-droid), [Gemini CLI](#gemini-cli), [GitHub Copilot CLI](#github-copilot-cli), [Hermes Agent](#hermes-agent), [Kimi Code](#kimi-code), [OpenCode](#opencode), [Pi](#pi).
|
||||||
|
|
||||||
## How it works
|
## How it works
|
||||||
|
|
||||||
@@ -193,6 +218,18 @@ pi -e /path/to/superpowers
|
|||||||
|
|
||||||
The Pi package loads the Superpowers skills and a small extension that injects the `using-superpowers` bootstrap at session startup and again after compaction. Pi has native skills, so no compatibility `Skill` tool is required. Subagent and task-list tools remain optional Pi companion packages.
|
The Pi package loads the Superpowers skills and a small extension that injects the `using-superpowers` bootstrap at session startup and again after compaction. Pi has native skills, so no compatibility `Skill` tool is required. Subagent and task-list tools remain optional Pi companion packages.
|
||||||
|
|
||||||
|
### Hermes Agent
|
||||||
|
|
||||||
|
Install Superpowers as a Hermes plugin from this repository:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
hermes plugins install obra/superpowers --enable
|
||||||
|
```
|
||||||
|
|
||||||
|
Restart any active Hermes sessions after installing. Note: Hermes has no
|
||||||
|
post-compaction hook, so a very long session that compacts over its first
|
||||||
|
turn loses the bootstrap — start a fresh session if skills stop triggering.
|
||||||
|
|
||||||
## The Basic Workflow
|
## The Basic Workflow
|
||||||
|
|
||||||
1. **brainstorming** - Activates before writing code. Refines rough ideas through questions, explores alternatives, presents design in sections for validation. Saves design document.
|
1. **brainstorming** - Activates before writing code. Refines rough ideas through questions, explores alternatives, presents design in sections for validation. Saves design document.
|
||||||
@@ -211,6 +248,14 @@ The Pi package loads the Superpowers skills and a small extension that injects t
|
|||||||
|
|
||||||
**The agent checks for relevant skills before any task.** Mandatory workflows, not suggestions.
|
**The agent checks for relevant skills before any task.** Mandatory workflows, not suggestions.
|
||||||
|
|
||||||
|
## Community
|
||||||
|
|
||||||
|
Superpowers is built by [Jesse Vincent](https://blog.fsck.com) and the rest of the folks at [Prime Radiant](https://primeradiant.com).
|
||||||
|
|
||||||
|
- **Discord**: [Join us](https://discord.gg/35wsABTejz) for community support, questions, and sharing what you're building with Superpowers
|
||||||
|
- **Issues**: https://github.com/obra/superpowers/issues
|
||||||
|
- **Release announcements**: [Sign up](https://primeradiant.com/superpowers/) to get notified about new versions
|
||||||
|
|
||||||
## What's Inside
|
## What's Inside
|
||||||
|
|
||||||
### Skills Library
|
### Skills Library
|
||||||
@@ -271,11 +316,3 @@ MIT License - see LICENSE file for details
|
|||||||
## Visual companion telemetry
|
## Visual companion telemetry
|
||||||
|
|
||||||
Because skills and plugins don't provide any feedback to creators, we have no idea how many of you are using Superpowers. By default, the Prime Radiant logo on brainstorming's optional visual companion feature is loaded from our website. It includes the version of Superpowers in use. It does not include any details about your project, prompt, or coding agent. We don't see your clicks or anything about what you're building. This helps us have a rough idea of how many folks are using Superpowers and which version of Superpowers they're using. It's 100% optional. To disable this, set the environment variable `SUPERPOWERS_DISABLE_TELEMETRY` to any true value. Superpowers also honors Claude Code's `DISABLE_TELEMETRY` and `CLAUDE_CODE_DISABLE_NONESSENTIAL_TRAFFIC` opt-outs.
|
Because skills and plugins don't provide any feedback to creators, we have no idea how many of you are using Superpowers. By default, the Prime Radiant logo on brainstorming's optional visual companion feature is loaded from our website. It includes the version of Superpowers in use. It does not include any details about your project, prompt, or coding agent. We don't see your clicks or anything about what you're building. This helps us have a rough idea of how many folks are using Superpowers and which version of Superpowers they're using. It's 100% optional. To disable this, set the environment variable `SUPERPOWERS_DISABLE_TELEMETRY` to any true value. Superpowers also honors Claude Code's `DISABLE_TELEMETRY` and `CLAUDE_CODE_DISABLE_NONESSENTIAL_TRAFFIC` opt-outs.
|
||||||
|
|
||||||
## Community
|
|
||||||
|
|
||||||
Superpowers is built by [Jesse Vincent](https://blog.fsck.com) and the rest of the folks at [Prime Radiant](https://primeradiant.com).
|
|
||||||
|
|
||||||
- **Discord**: [Join us](https://discord.gg/35wsABTejz) for community support, questions, and sharing what you're building with Superpowers
|
|
||||||
- **Issues**: https://github.com/obra/superpowers/issues
|
|
||||||
- **Release announcements**: [Sign up](https://primeradiant.com/superpowers/) to get notified about new versions
|
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,252 @@
|
|||||||
|
# Codex Efficiency Fixes — Design
|
||||||
|
|
||||||
|
Date: 2026-07-30
|
||||||
|
Status: approved by Jesse (in-session)
|
||||||
|
Branch: `codex-efficiency-fixes` off `dev`
|
||||||
|
|
||||||
|
## Sources
|
||||||
|
|
||||||
|
- Eval campaign closeout: `superpowers-autoresearch/reports/2026-07-codex-efficiency-campaign.md`
|
||||||
|
(treatment table §4; every treatment below has a scorer and a measured
|
||||||
|
`dev` baseline).
|
||||||
|
- Codex source recon: `superpowers-autoresearch/docs/2026-07-29-codex-multiagent-v2-capabilities.md`
|
||||||
|
(file:line citations against the Codex CLI source; grounds T2, T3, T5).
|
||||||
|
- Published experiment write-ups: `superpowers-evals/docs/experiments/`.
|
||||||
|
- Drew's spinout stack (PRs #2036, #2035) is **evidence, not adopted text**:
|
||||||
|
Jesse wants to dig into those fixes in more detail before adopting any
|
||||||
|
of them; they inform the problem statements only.
|
||||||
|
|
||||||
|
## Goal
|
||||||
|
|
||||||
|
Ship the five evidence-strong treatments from the codex-efficiency eval
|
||||||
|
campaign as superpowers skill/doc changes, each graded against its
|
||||||
|
pre-registered criterion by the campaign's scorers before its PR is cut.
|
||||||
|
Phase 2 (everything else in the closeout treatment table) follows, each
|
||||||
|
item gated on new baseline work first.
|
||||||
|
|
||||||
|
## Scope decisions (settled with Jesse)
|
||||||
|
|
||||||
|
- **Phase 1 = the evidence-strong five** (T1–T5 below). Phase 2 items
|
||||||
|
each need a failing baseline before any fix ships (discrimination
|
||||||
|
rule: inconclusive-by-zero is a stop).
|
||||||
|
- **One branch, PR per treatment.** Development and batteries happen on
|
||||||
|
`codex-efficiency-fixes`; when a treatment beats its criterion, it is
|
||||||
|
cut into its own PR against `dev` with its eval evidence. No merge
|
||||||
|
without Jesse's per-PR approval.
|
||||||
|
- **T4 ships cross-harness with a global regression battery** (Claude
|
||||||
|
Code, Codex, Gemini), variant C shape: ceremony scales, approval never
|
||||||
|
does.
|
||||||
|
|
||||||
|
## The five treatments
|
||||||
|
|
||||||
|
### T1. SDD worker-review prohibition
|
||||||
|
|
||||||
|
**Evidence:** 9/9 depth-2 spawns across 4 corpora were implementer-issued
|
||||||
|
reviewers; all 9 were same-task duplicates of the review the controller
|
||||||
|
dispatches anyway. The dispatch contract never says review is not the
|
||||||
|
worker's job; "self-review" in the implementer prompt gets reified into a
|
||||||
|
reviewer subagent on harnesses where children can spawn (Codex).
|
||||||
|
|
||||||
|
**Changes:**
|
||||||
|
- `skills/subagent-driven-development/implementer-prompt.md`: an explicit
|
||||||
|
"You do not dispatch subagents" clause — self-review means reading your
|
||||||
|
own diff; the controller owns all review dispatch; a reviewer you spawn
|
||||||
|
duplicates a review the process already provides.
|
||||||
|
- `skills/subagent-driven-development/SKILL.md`: one dispatch-contract
|
||||||
|
line in the task loop, plus a Red Flags row: "An independent review
|
||||||
|
would strengthen my report" → review is the controller's next step;
|
||||||
|
your reviewer is a duplicate seat.
|
||||||
|
- Harness-agnostic wording (no-op where children cannot spawn).
|
||||||
|
|
||||||
|
**Graded by:** `score_e6.py` (depth-2 spawns by spawner role, duplicate
|
||||||
|
review families); `score_e5.py` for the same-scope variant.
|
||||||
|
**Baseline:** 9/9 worker-issued, 0 counter-examples.
|
||||||
|
**Criterion:** 0 worker-issued depth-2 spawns AND review coverage
|
||||||
|
preserved (every task still gets exactly one controller-dispatched task
|
||||||
|
review).
|
||||||
|
|
||||||
|
### T2. Event-driven waiting
|
||||||
|
|
||||||
|
**Evidence:** 60–78% of `wait_agent` calls time out in every corpus
|
||||||
|
(dev 67.1%, spinout 60.2%). Source recon: V2 waits are event
|
||||||
|
subscriptions, not polls — one long wait has the same wake latency as a
|
||||||
|
10s poll at ~1/90th the calls; a completed child's FINAL_ANSWER is pushed
|
||||||
|
into the parent's mailbox and drained into the next model request with no
|
||||||
|
wait at all.
|
||||||
|
|
||||||
|
**Changes** (`skills/using-superpowers/references/codex-tools.md`):
|
||||||
|
- Never short-timeout poll.
|
||||||
|
- While local work remains, do not wait — child results arrive with your
|
||||||
|
next turn via the mailbox.
|
||||||
|
- When genuinely idle, issue ONE `wait_agent` with a long `timeout_ms`
|
||||||
|
(900000+; harness max 3600000).
|
||||||
|
- V2 caveat stated: completion mail carries `trigger_turn=false` and will
|
||||||
|
not wake an idle controller — that is the one job `wait_agent` has.
|
||||||
|
|
||||||
|
**Graded by:** `score_e7.py` (timeout rate, inter-poll cadence,
|
||||||
|
cache-rebill estimate — the rebill figure stays labeled as an estimate).
|
||||||
|
**Baseline:** dev 67.1% timeout rate.
|
||||||
|
**Criterion:** timeout rate < 25% with no loss of task completion.
|
||||||
|
|
||||||
|
### T3. codex-tools.md corrections
|
||||||
|
|
||||||
|
**Evidence:** five claims in the current guidance are contradicted by the
|
||||||
|
Codex source (all file:line-cited in the capabilities doc):
|
||||||
|
1. `close_agent` does not exist in multi-agent V2 (V1-only). V2 LRU-evicts
|
||||||
|
finished children automatically; not closing costs nothing;
|
||||||
|
`followup_task` transparently reloads an evicted child.
|
||||||
|
2. Fix rounds can always resume the implementer via `followup_task` —
|
||||||
|
dev's "if your harness cannot send another message to a spawned agent,
|
||||||
|
dispatch each fix round as a fresh implementer" branch is dead on V2.
|
||||||
|
3. Role files (`~/.codex/agents/**.toml`) DO attach to spawns via
|
||||||
|
`agent_type` on isolated forks (0.145+).
|
||||||
|
4. Full-history forks accept `model`/`reasoning_effort` overrides; only
|
||||||
|
`agent_type` is refused. (Isolated forks remain the SDD guidance for
|
||||||
|
context-hygiene reasons, stated accurately.)
|
||||||
|
5. Dispatch guidance must never name non-V2 model presets — the V2 spawn
|
||||||
|
allowlist is v2 presets only; others hard-error.
|
||||||
|
|
||||||
|
**Changes:** rewrite the multi-agent paragraph of
|
||||||
|
`skills/using-superpowers/references/codex-tools.md` to be
|
||||||
|
version-honest (V1 vs V2 behavior labeled where they differ).
|
||||||
|
|
||||||
|
**Graded by:** source citation (already verified); no scorer regressions
|
||||||
|
on the shared battery. `score_e8.py` is retained as a V1/V2 schema
|
||||||
|
detector, not a hygiene grader — no `close_agent` checklist ships.
|
||||||
|
|
||||||
|
### T4. Brainstorming three-path router (variant C: approval always)
|
||||||
|
|
||||||
|
**Evidence:** micro — the current HARD-GATE text pushes a bounded task to
|
||||||
|
FULL ceremony 5/5, while Z-null (no guidance) and a three-path router
|
||||||
|
both differentiate 5/5: the absolute wording suppresses discrimination
|
||||||
|
the model draws natively. FULL battery — ceremony volume scales
|
||||||
|
moderately (16.7 vs 24.0 tool calls, bounded vs arch), but the
|
||||||
|
two-document ritual (spec file → plan file) ran unconditionally in every
|
||||||
|
rep. The measured waste is the unconditional artifact ritual, not the
|
||||||
|
approval gate.
|
||||||
|
|
||||||
|
**Design (variant C):** three paths scale the ARTIFACT; every path keeps
|
||||||
|
human approval before implementation:
|
||||||
|
- **Spike** (feasibility question, explicitly throwaway): present the
|
||||||
|
question and the intended probe in 2–3 sentences, get a nod, go. No
|
||||||
|
docs. Findings return as a recommendation; anything built stays labeled
|
||||||
|
throwaway.
|
||||||
|
- **Bounded** (well-scoped change to an existing, understood flow):
|
||||||
|
present a short design in chat, get approval, implement. No spec file,
|
||||||
|
no writing-plans invocation.
|
||||||
|
- **Architectural** (restructures components, new subsystem, public
|
||||||
|
interface change): the full current flow — spec doc, review,
|
||||||
|
writing-plans.
|
||||||
|
|
||||||
|
**Guards (all ship with the router):**
|
||||||
|
- Classification is said out loud ("this looks bounded, so I'll present a
|
||||||
|
short design here rather than write a spec") so the human can override.
|
||||||
|
- When in doubt between two paths, take the heavier one.
|
||||||
|
- One-way ratchet: hidden complexity discovered mid-path upgrades the
|
||||||
|
path; never downgrade mid-task.
|
||||||
|
- New Red Flags rows targeting classification-as-escape-hatch ("I'll call
|
||||||
|
it bounded to skip the doc").
|
||||||
|
|
||||||
|
**Changes** (`skills/brainstorming/SKILL.md`): HARD-GATE keeps "no
|
||||||
|
implementation before approval" and drops "regardless of perceived
|
||||||
|
simplicity" as the ceremony driver; anti-pattern section reframed (the
|
||||||
|
sin is skipping approval, not skipping documents); checklist steps 6–9
|
||||||
|
become the architectural path; process-flow graph gains the router; Red
|
||||||
|
Flags rows added. This is carefully-tuned content — the edit follows
|
||||||
|
writing-skills methodology and ships only with the full eval evidence
|
||||||
|
below.
|
||||||
|
|
||||||
|
**Graded by (three layers):**
|
||||||
|
1. **Micro** (`ceremony-path-micro.py`, adapted): variant C literal text,
|
||||||
|
plus adversarially ambiguous briefs the campaign never tested (a task
|
||||||
|
that pattern-matches bounded but hides a public interface change).
|
||||||
|
Criteria: spike/bounded/arch differentiate (≥4/5 per cell); ambiguous
|
||||||
|
briefs escalate to FULL (≥4/5); arch never downgrades (5/5).
|
||||||
|
2. **Codex ceremony battery:** `cx-ceremony-{spike,bounded,arch}` on the
|
||||||
|
fix arm, 3 reps each, `score_e4.py` census. Criteria: bounded reps
|
||||||
|
show an approval turn but zero committed spec files and zero
|
||||||
|
writing-plans ritual; arch reps keep the full two-doc flow; spike reps
|
||||||
|
stay minimal.
|
||||||
|
3. **Global regression battery:** the same three ceremony scenarios on
|
||||||
|
Claude Code and Gemini (rig work: those scenarios are currently
|
||||||
|
codex-gated), 3 reps each; plus the triggering acceptance check
|
||||||
|
("Let's make a react todo list" auto-triggers brainstorming into the
|
||||||
|
full/architectural path) on all three harnesses.
|
||||||
|
|
||||||
|
### T5. Explicit model on child-issued spawns
|
||||||
|
|
||||||
|
**Evidence:** root spawns are 100% explicit-model at CLI 0.146 (dev
|
||||||
|
14/14); the live gap is depth-2 — 2/2 child-issued spawns omitted
|
||||||
|
`model`. Source recon: `model` without `reasoning_effort` resets effort
|
||||||
|
to the MODEL's default, not the parent's.
|
||||||
|
|
||||||
|
**Changes** (`skills/using-superpowers/references/codex-tools.md`):
|
||||||
|
- Every spawn you issue — including as a child — sets `model` AND
|
||||||
|
`reasoning_effort`; the effort-reset trap is named.
|
||||||
|
- Advise `[agents].default_subagent_model` and
|
||||||
|
`[agents].default_subagent_reasoning_effort` in `~/.codex/config.toml`
|
||||||
|
as the machine-level backstop for anything that slips through.
|
||||||
|
|
||||||
|
**Graded by:** `score_e1.py` (per-spawn explicit-model rate, by depth) on
|
||||||
|
the shared battery.
|
||||||
|
**Baseline:** depth-2: 0/2 explicit.
|
||||||
|
**Criterion:** every spawn at every depth carries explicit model +
|
||||||
|
effort. Pre-registered caveat: if T1 eliminates depth-2 spawns entirely,
|
||||||
|
T5 grades as root-spawn regression (hold 100%) plus doc correctness and
|
||||||
|
is recorded inconclusive-by-zero at depth-2 — the config backstop is then
|
||||||
|
the operative mechanism.
|
||||||
|
|
||||||
|
## Grading plan
|
||||||
|
|
||||||
|
- **Shared SDD battery** carries T1, T2, T5: `cx-sdd-small`, fix-branch
|
||||||
|
arm (`/tmp/sp-arm-fix`), 8 reps across both container lanes. Dev
|
||||||
|
baselines are already measured; no baseline re-runs.
|
||||||
|
- **T4 batteries** as listed above (micro + codex ceremony + global
|
||||||
|
regression).
|
||||||
|
- **Pre-registration:** every battery gets a hypothesis-log entry
|
||||||
|
(prediction, scorer, criterion) in
|
||||||
|
`superpowers-autoresearch/logs/2026-07-30-codex-efficiency-fixes.md`
|
||||||
|
BEFORE it runs. Standing rules carry over: append-only log, manual
|
||||||
|
inspection of scorer matches on fix-arm runs (non-circular
|
||||||
|
verification), no raw rollouts committed, correctness rides beside
|
||||||
|
cost in every verdict.
|
||||||
|
- **Attribution:** orthogonal scorers on one combined branch; unexpected
|
||||||
|
regressions bisect by treatment commit.
|
||||||
|
- **Budget:** shared battery ~$40, codex ceremony ~$40, global
|
||||||
|
regression ~$40–80, micros ~$5 → phase 1 ≈ $150–200 of the ~$850
|
||||||
|
remaining from the campaign's $1000.
|
||||||
|
|
||||||
|
## Process
|
||||||
|
|
||||||
|
- Work happens in the `codex-efficiency-fixes` worktree (branched off
|
||||||
|
`dev`); execution via subagent-driven-development from a written plan.
|
||||||
|
- Skill-text changes follow writing-skills methodology.
|
||||||
|
- Scenario/rig changes (un-gating ceremony scenarios for Claude
|
||||||
|
Code/Gemini, adversarial micro briefs) land in `superpowers-evals`
|
||||||
|
main, as authorized.
|
||||||
|
- PR-per-treatment against `dev`, each with its eval evidence and the
|
||||||
|
standard identification block; merges only on Jesse's per-PR approval.
|
||||||
|
|
||||||
|
## Phase 2 queue (baseline-first; not in this plan's tasks)
|
||||||
|
|
||||||
|
Each item requires a failing baseline before any fix ships:
|
||||||
|
1. **Dispatch routing / long-session drift** — needs a long-session
|
||||||
|
elicitation rig (fresh sessions don't reproduce the pathology at CLI
|
||||||
|
0.146). Drew's stack informs the treatment shape.
|
||||||
|
2. **Verification leases / evidence receipts** — needs the
|
||||||
|
substring-aware duplicate counter added to `score_e3.py` first
|
||||||
|
(current baseline 1/23 exact-string pairs is too weak).
|
||||||
|
3. **Remediation cap** — small-n baseline (2/3 reps) needs more reps.
|
||||||
|
4. **Cross-task-race probe redesign** — `score_e5.py`'s probe is
|
||||||
|
inconclusive-by-zero by design tradeoff; needs a stronger probe.
|
||||||
|
5. **E5 D4 shell-command parser** — fix-review-scope classifier cannot
|
||||||
|
parse compound commands; scorer work, not skill work.
|
||||||
|
|
||||||
|
## Out of scope
|
||||||
|
|
||||||
|
- Adopting Drew's spinout stack (#2036/#2035) or its text.
|
||||||
|
- RoboRev, Codex token telemetry (separate codebases).
|
||||||
|
- A `close_agent` hygiene checklist (V2 has no such tool — closed as
|
||||||
|
do-not-ship in the campaign).
|
||||||
|
- Claude Code/Gemini-specific efficiency treatments beyond the T4
|
||||||
|
regression battery.
|
||||||
@@ -56,6 +56,7 @@ If your harness appears here, read its reference file for special instructions:
|
|||||||
- Codex: `references/codex-tools.md`
|
- Codex: `references/codex-tools.md`
|
||||||
- Pi: `references/pi-tools.md`
|
- Pi: `references/pi-tools.md`
|
||||||
- Antigravity: `references/antigravity-tools.md`
|
- Antigravity: `references/antigravity-tools.md`
|
||||||
|
- Hermes Agent: `references/hermes-tools.md`
|
||||||
|
|
||||||
## User Instructions
|
## User Instructions
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,56 @@
|
|||||||
|
# Hermes Agent Tool Mapping
|
||||||
|
|
||||||
|
Skills speak in actions ("dispatch a subagent", "create a todo", "read a file"). On Hermes Agent these resolve to the tools below.
|
||||||
|
|
||||||
|
## Tools
|
||||||
|
|
||||||
|
| Action skills request | Hermes tool |
|
||||||
|
|---|---|
|
||||||
|
| Read a file | `read_file` |
|
||||||
|
| Create a new file | `write_file` |
|
||||||
|
| Edit a file (targeted patch) | `patch` |
|
||||||
|
| Run a shell command | `terminal` |
|
||||||
|
| Search file contents | `search_files` |
|
||||||
|
| Find files by name | `terminal` with `find` |
|
||||||
|
| Fetch a URL / read a webpage | `web_extract(urls=[...])` |
|
||||||
|
| Search the web | `web_search(query=...)` |
|
||||||
|
| Dispatch a subagent | `delegate_task(goal=..., context=..., toolsets=[...], role="leaf")` |
|
||||||
|
| Task tracking | `todo` tool |
|
||||||
|
| Invoke a skill | `skill_view("skill-name")` |
|
||||||
|
|
||||||
|
## Instructions file
|
||||||
|
|
||||||
|
When a skill mentions "your instructions file," on Hermes Agent this is **`AGENTS.md`** in the project directory, or **`SOUL.md`** globally at `~/.hermes/SOUL.md`.
|
||||||
|
|
||||||
|
## Invoking a skill
|
||||||
|
|
||||||
|
Hermes Agent has a `skills` toolset with `skill_view` and `skills_list` tools.
|
||||||
|
To invoke a superpowers skill, use:
|
||||||
|
|
||||||
|
```
|
||||||
|
skill_view("brainstorming")
|
||||||
|
skill_view("test-driven-development")
|
||||||
|
```
|
||||||
|
|
||||||
|
If `skill_view` cannot find a superpowers skill (it may not appear in the catalog
|
||||||
|
until the plugin fully registers it), fall back to reading the SKILL.md directly:
|
||||||
|
|
||||||
|
```
|
||||||
|
read_file(path="~/.hermes/plugins/superpowers/skills/<skill-name>/SKILL.md")
|
||||||
|
```
|
||||||
|
|
||||||
|
This fallback is the same mechanism used by other harnesses without native skill loading.
|
||||||
|
|
||||||
|
## Subagent dispatch
|
||||||
|
|
||||||
|
Use `delegate_task` to spawn isolated subagents for parallel or sequential workstreams:
|
||||||
|
|
||||||
|
```
|
||||||
|
delegate_task(goal="...", context="...", toolsets=[...], role="leaf")
|
||||||
|
```
|
||||||
|
|
||||||
|
If `delegate_task` is unavailable, do the work inline rather than inventing tool calls.
|
||||||
|
|
||||||
|
## Task tracking
|
||||||
|
|
||||||
|
Use the `todo` tool for task tracking within a session. For multi-agent task boards, use `hermes kanban` CLI if available. Treat older `TodoWrite` references as the task-tracking action.
|
||||||
@@ -0,0 +1,30 @@
|
|||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
from unittest.mock import MagicMock
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.fixture
|
||||||
|
def mock_ctx():
|
||||||
|
ctx = MagicMock()
|
||||||
|
ctx._hooks = {}
|
||||||
|
ctx._skills = {}
|
||||||
|
|
||||||
|
def register_hook(event, fn):
|
||||||
|
ctx._hooks[event] = fn
|
||||||
|
|
||||||
|
def register_skill(name, path):
|
||||||
|
# Mimic hermes' real register_skill, which calls path.exists() and
|
||||||
|
# therefore breaks on a str (the bug that silently disabled the whole
|
||||||
|
# plugin, found 2026-07-23). Keeping that fidelity here means a
|
||||||
|
# regression to str paths fails these tests instead of failing
|
||||||
|
# silently inside hermes.
|
||||||
|
if not isinstance(path, Path):
|
||||||
|
raise AttributeError(
|
||||||
|
f"register_skill requires a pathlib.Path, got {type(path).__name__}"
|
||||||
|
)
|
||||||
|
ctx._skills[name] = path
|
||||||
|
|
||||||
|
ctx.register_hook.side_effect = register_hook
|
||||||
|
ctx.register_skill.side_effect = register_skill
|
||||||
|
return ctx
|
||||||
@@ -0,0 +1,98 @@
|
|||||||
|
import importlib
|
||||||
|
import os
|
||||||
|
import sys
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
sys.path.insert(0, os.path.abspath(
|
||||||
|
os.path.join(os.path.dirname(__file__), "../../.hermes-plugin")
|
||||||
|
))
|
||||||
|
|
||||||
|
BOOTSTRAP_MARKER = "superpowers:using-superpowers bootstrap for hermes"
|
||||||
|
|
||||||
|
# Hermes spills injected context over 10,000 chars to a file, which breaks
|
||||||
|
# inline injection semantics. The bootstrap must stay under it with margin.
|
||||||
|
HERMES_CONTEXT_SPILL_LIMIT = 10_000
|
||||||
|
|
||||||
|
|
||||||
|
def _load():
|
||||||
|
if "__init__" in sys.modules:
|
||||||
|
del sys.modules["__init__"]
|
||||||
|
return importlib.import_module("__init__")
|
||||||
|
|
||||||
|
|
||||||
|
def _bootstrap():
|
||||||
|
m = _load()
|
||||||
|
return m._build_bootstrap(m._skills_dir())
|
||||||
|
|
||||||
|
|
||||||
|
class TestStripFrontmatter:
|
||||||
|
def test_strips_yaml_block(self):
|
||||||
|
m = _load()
|
||||||
|
content = "---\nname: foo\ndescription: bar\n---\n# Body\nContent here"
|
||||||
|
assert m._strip_frontmatter(content) == "# Body\nContent here"
|
||||||
|
|
||||||
|
def test_no_frontmatter_returns_trimmed_content(self):
|
||||||
|
m = _load()
|
||||||
|
content = "# No frontmatter\nJust content"
|
||||||
|
assert m._strip_frontmatter(content) == "# No frontmatter\nJust content"
|
||||||
|
|
||||||
|
def test_strips_surrounding_whitespace_from_body(self):
|
||||||
|
m = _load()
|
||||||
|
content = "---\nname: foo\n---\n\n\n# Body\n\n"
|
||||||
|
assert m._strip_frontmatter(content) == "# Body"
|
||||||
|
|
||||||
|
|
||||||
|
class TestSkillsDirResolution:
|
||||||
|
def test_repo_layout_resolves(self):
|
||||||
|
# The repo checkout IS the git-clone layout: .hermes-plugin/ and
|
||||||
|
# skills/ are siblings, so resolution must succeed from here.
|
||||||
|
m = _load()
|
||||||
|
skills = m._skills_dir()
|
||||||
|
assert os.path.isfile(
|
||||||
|
os.path.join(skills, "using-superpowers", "SKILL.md")
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class TestBootstrapContent:
|
||||||
|
def test_marker_and_wrapper(self):
|
||||||
|
content = _bootstrap()
|
||||||
|
assert BOOTSTRAP_MARKER in content
|
||||||
|
assert content.startswith("<EXTREMELY_IMPORTANT>")
|
||||||
|
assert content.rstrip().endswith("</EXTREMELY_IMPORTANT>")
|
||||||
|
|
||||||
|
def test_contains_using_superpowers_body(self):
|
||||||
|
content = _bootstrap()
|
||||||
|
# A distinctive line from the skill body proves the real SKILL.md was
|
||||||
|
# embedded, not a stub.
|
||||||
|
assert "You have superpowers" in content
|
||||||
|
assert "## The Rule" in content
|
||||||
|
|
||||||
|
def test_frontmatter_stripped(self):
|
||||||
|
content = _bootstrap()
|
||||||
|
assert "---\nname:" not in content
|
||||||
|
|
||||||
|
def test_tool_mapping_sourced_from_reference_file(self):
|
||||||
|
m = _load()
|
||||||
|
content = _bootstrap()
|
||||||
|
ref = os.path.join(
|
||||||
|
m._skills_dir(), "using-superpowers", "references", "hermes-tools.md"
|
||||||
|
)
|
||||||
|
with open(ref, encoding="utf-8") as f:
|
||||||
|
ref_text = f.read().strip()
|
||||||
|
# The mapping is included verbatim from the reference file — the
|
||||||
|
# single source, not a drift-prone inline copy.
|
||||||
|
assert ref_text in content
|
||||||
|
assert "read_file" in content
|
||||||
|
|
||||||
|
def test_skill_view_guidance_present(self):
|
||||||
|
content = _bootstrap()
|
||||||
|
assert 'skill_view("superpowers:brainstorming")' in content
|
||||||
|
|
||||||
|
def test_under_hermes_context_spill_limit(self):
|
||||||
|
content = _bootstrap()
|
||||||
|
assert len(content) < HERMES_CONTEXT_SPILL_LIMIT, (
|
||||||
|
f"bootstrap is {len(content)} chars; hermes spills injected "
|
||||||
|
f"context over {HERMES_CONTEXT_SPILL_LIMIT} to a file, which "
|
||||||
|
"breaks inline injection"
|
||||||
|
)
|
||||||
@@ -0,0 +1,142 @@
|
|||||||
|
import importlib
|
||||||
|
import importlib.util
|
||||||
|
import os
|
||||||
|
import shutil
|
||||||
|
import sys
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
# Point at the plugin directory
|
||||||
|
_PLUGIN_DIR = os.path.abspath(
|
||||||
|
os.path.join(os.path.dirname(__file__), "../../.hermes-plugin")
|
||||||
|
)
|
||||||
|
sys.path.insert(0, _PLUGIN_DIR)
|
||||||
|
|
||||||
|
BOOTSTRAP_MARKER = "superpowers:using-superpowers bootstrap for hermes"
|
||||||
|
|
||||||
|
|
||||||
|
def _load_plugin():
|
||||||
|
"""Re-import plugin module fresh."""
|
||||||
|
if "__init__" in sys.modules:
|
||||||
|
del sys.modules["__init__"]
|
||||||
|
return importlib.import_module("__init__")
|
||||||
|
|
||||||
|
|
||||||
|
def _fire_pre_llm(ctx, **kwargs):
|
||||||
|
hook = ctx._hooks["pre_llm_call"]
|
||||||
|
defaults = {
|
||||||
|
"session_id": "s1",
|
||||||
|
"user_message": "hi",
|
||||||
|
"conversation_history": [],
|
||||||
|
"is_first_turn": False,
|
||||||
|
"model": "test-model",
|
||||||
|
"platform": "cli",
|
||||||
|
}
|
||||||
|
defaults.update(kwargs)
|
||||||
|
return hook(**defaults)
|
||||||
|
|
||||||
|
|
||||||
|
class TestPluginRegistration:
|
||||||
|
def test_register_attaches_only_pre_llm_call_hook(self, mock_ctx):
|
||||||
|
plugin = _load_plugin()
|
||||||
|
plugin.register(mock_ctx)
|
||||||
|
assert list(mock_ctx._hooks.keys()) == ["pre_llm_call"]
|
||||||
|
|
||||||
|
def test_register_registers_every_stock_skill_as_path(self, mock_ctx):
|
||||||
|
plugin = _load_plugin()
|
||||||
|
plugin.register(mock_ctx)
|
||||||
|
# The conftest mock raises on non-Path (mirroring hermes' real
|
||||||
|
# register_skill), so reaching these asserts proves every
|
||||||
|
# registration passed a pathlib.Path.
|
||||||
|
assert "using-superpowers" in mock_ctx._skills
|
||||||
|
assert "brainstorming" in mock_ctx._skills
|
||||||
|
for name, path in mock_ctx._skills.items():
|
||||||
|
assert isinstance(path, Path)
|
||||||
|
assert path.name == "SKILL.md"
|
||||||
|
assert path.parent.name == name
|
||||||
|
assert path.is_file()
|
||||||
|
|
||||||
|
def test_registered_skills_match_skill_directories(self, mock_ctx):
|
||||||
|
plugin = _load_plugin()
|
||||||
|
plugin.register(mock_ctx)
|
||||||
|
skills_root = plugin._skills_dir()
|
||||||
|
expected = {
|
||||||
|
entry
|
||||||
|
for entry in os.listdir(skills_root)
|
||||||
|
if os.path.isfile(os.path.join(skills_root, entry, "SKILL.md"))
|
||||||
|
}
|
||||||
|
assert set(mock_ctx._skills.keys()) == expected
|
||||||
|
|
||||||
|
|
||||||
|
class TestBootstrapInjection:
|
||||||
|
def test_first_turn_returns_bootstrap_context(self, mock_ctx):
|
||||||
|
plugin = _load_plugin()
|
||||||
|
plugin.register(mock_ctx)
|
||||||
|
result = _fire_pre_llm(mock_ctx, is_first_turn=True)
|
||||||
|
assert isinstance(result, dict)
|
||||||
|
content = result["context"]
|
||||||
|
assert BOOTSTRAP_MARKER in content
|
||||||
|
assert content.startswith("<EXTREMELY_IMPORTANT>")
|
||||||
|
assert content.rstrip().endswith("</EXTREMELY_IMPORTANT>")
|
||||||
|
|
||||||
|
def test_later_turns_return_none(self, mock_ctx):
|
||||||
|
plugin = _load_plugin()
|
||||||
|
plugin.register(mock_ctx)
|
||||||
|
assert _fire_pre_llm(mock_ctx, is_first_turn=False) is None
|
||||||
|
assert _fire_pre_llm(mock_ctx, is_first_turn=None) is None
|
||||||
|
|
||||||
|
def test_hook_tolerates_future_kwargs(self, mock_ctx):
|
||||||
|
plugin = _load_plugin()
|
||||||
|
plugin.register(mock_ctx)
|
||||||
|
result = _fire_pre_llm(
|
||||||
|
mock_ctx, is_first_turn=True, telemetry_schema_version=3
|
||||||
|
)
|
||||||
|
assert BOOTSTRAP_MARKER in result["context"]
|
||||||
|
|
||||||
|
|
||||||
|
class TestLayoutResolution:
|
||||||
|
def _stage(self, tmp_path, layout):
|
||||||
|
"""Copy the plugin module + a minimal skills tree in the given layout."""
|
||||||
|
src_skills = Path(_PLUGIN_DIR).parent / "skills"
|
||||||
|
if layout == "clone":
|
||||||
|
plugdir = tmp_path / "superpowers" / ".hermes-plugin"
|
||||||
|
else: # flat: module at the plugin dir root, skills nested inside it
|
||||||
|
plugdir = tmp_path / "superpowers"
|
||||||
|
skills = tmp_path / "superpowers" / "skills"
|
||||||
|
plugdir.mkdir(parents=True, exist_ok=True)
|
||||||
|
shutil.copy(Path(_PLUGIN_DIR) / "__init__.py", plugdir / "__init__.py")
|
||||||
|
for skill in ("using-superpowers", "brainstorming"):
|
||||||
|
shutil.copytree(src_skills / skill, skills / skill)
|
||||||
|
return plugdir
|
||||||
|
|
||||||
|
def _load_from(self, plugdir):
|
||||||
|
spec = importlib.util.spec_from_file_location(
|
||||||
|
f"hermes_plugin_test_{plugdir.parent.name}_{plugdir.name}",
|
||||||
|
plugdir / "__init__.py",
|
||||||
|
)
|
||||||
|
mod = importlib.util.module_from_spec(spec)
|
||||||
|
spec.loader.exec_module(mod)
|
||||||
|
return mod
|
||||||
|
|
||||||
|
def test_clone_layout_resolves_sibling_skills(self, tmp_path, mock_ctx):
|
||||||
|
# git-clone install: .hermes-plugin/ and skills/ are siblings.
|
||||||
|
plugdir = self._stage(tmp_path, "clone")
|
||||||
|
mod = self._load_from(plugdir)
|
||||||
|
mod.register(mock_ctx)
|
||||||
|
assert "using-superpowers" in mock_ctx._skills
|
||||||
|
|
||||||
|
def test_flat_layout_resolves_nested_skills(self, tmp_path, mock_ctx):
|
||||||
|
# flattened install: module at the plugin dir root, skills/ inside it.
|
||||||
|
plugdir = self._stage(tmp_path, "flat")
|
||||||
|
mod = self._load_from(plugdir)
|
||||||
|
mod.register(mock_ctx)
|
||||||
|
assert "using-superpowers" in mock_ctx._skills
|
||||||
|
|
||||||
|
def test_missing_skills_raises_loudly(self, tmp_path, mock_ctx):
|
||||||
|
plugdir = tmp_path / "superpowers"
|
||||||
|
plugdir.mkdir(parents=True)
|
||||||
|
shutil.copy(Path(_PLUGIN_DIR) / "__init__.py", plugdir / "__init__.py")
|
||||||
|
mod = self._load_from(plugdir)
|
||||||
|
with pytest.raises(RuntimeError, match="cannot find the skills"):
|
||||||
|
mod.register(mock_ctx)
|
||||||
Reference in New Issue
Block a user