Compare commits
131 Commits
exp/contro
...
diagnosing
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
affa7fa4e2 | ||
|
|
fd02874aa5 | ||
|
|
777ceb5e12 | ||
|
|
b36e0829c6 | ||
|
|
41cdb703de | ||
|
|
d4e3c1cb8c | ||
|
|
89d36fe961 | ||
|
|
034958f842 | ||
|
|
824aabcb21 | ||
|
|
2d4b675b49 | ||
|
|
d21e171f57 | ||
|
|
09a567b6f4 | ||
|
|
d6a10aba55 | ||
|
|
8f89e512c3 | ||
|
|
28125bf284 | ||
|
|
c367f804bb | ||
|
|
5f8f500b1d | ||
|
|
707b155a38 | ||
|
|
3e1ecde38f | ||
|
|
ffe22811bf | ||
|
|
cfb310c69a | ||
|
|
fdd1763d77 | ||
|
|
af4bebf762 | ||
|
|
17b42c8128 | ||
|
|
1245282b05 | ||
|
|
6819b42d97 | ||
|
|
02654f93bf | ||
|
|
dcd3661b7c | ||
|
|
9be44ebf40 | ||
|
|
695744056e | ||
|
|
fb518edf7b | ||
|
|
80b82abd8d | ||
|
|
05c2393b82 | ||
|
|
78cc189244 | ||
|
|
be76350536 | ||
|
|
419dec7755 | ||
|
|
2b195749df | ||
|
|
7a01a0e83a | ||
|
|
8acf8e5f24 | ||
|
|
50a924b0c4 | ||
|
|
2a977c7095 | ||
|
|
50f787ca5c | ||
|
|
538d65120b | ||
|
|
61f669ebc9 | ||
|
|
e7a4285985 | ||
|
|
39f9602432 | ||
|
|
3ff8d15f15 | ||
|
|
e9686d5c09 | ||
|
|
d8189d1587 | ||
|
|
db4538fcb8 | ||
|
|
9b8b14fe12 | ||
|
|
4dc71b10b3 | ||
|
|
7c560e048b | ||
|
|
75756d2900 | ||
|
|
b68eaf96bb | ||
|
|
2e7d681591 | ||
|
|
6211388f4b | ||
|
|
44c9b2d6e8 | ||
|
|
bb2a34b2a0 | ||
|
|
7b4dc4d7fd | ||
|
|
3dcbd5c4b4 | ||
|
|
c8921b5156 | ||
|
|
6015d37fe6 | ||
|
|
a868631a8a | ||
|
|
ebdd4ec61f | ||
|
|
28882fcbc3 | ||
|
|
87e4050daa | ||
|
|
c6fa27e3f8 | ||
|
|
55bbb52c7e | ||
|
|
0e13ad8222 | ||
|
|
5d5b6567a8 | ||
|
|
52f649e4ec | ||
|
|
5151e7aebe | ||
|
|
2dbbaed081 | ||
|
|
e7826745ee | ||
|
|
b8a2d84b40 | ||
|
|
6df8ba1458 | ||
|
|
256b42f454 | ||
|
|
40e866580d | ||
|
|
dd57200de1 | ||
|
|
485162ee00 | ||
|
|
fe5e9c9de9 | ||
|
|
50fbea0488 | ||
|
|
d238a48f5d | ||
|
|
a60dc2ffe5 | ||
|
|
a80b7b6386 | ||
|
|
b9e75dddec | ||
|
|
3fb7597418 | ||
|
|
153d6186c5 | ||
|
|
1e14b2377e | ||
|
|
05d90ac592 | ||
|
|
bc868020bb | ||
|
|
cfb6281371 | ||
|
|
2173c1c2b4 | ||
|
|
09fc6e0f3b | ||
|
|
3be5aad3dd | ||
|
|
c74782ead6 | ||
|
|
6dbbbda3ba | ||
|
|
caa1826cba | ||
|
|
517a9c6419 | ||
|
|
e8a9748a3f | ||
|
|
9d8630d5d9 | ||
|
|
50025d16ac | ||
|
|
e74961c110 | ||
|
|
0b47219ac8 | ||
|
|
fbb6dba450 | ||
|
|
bcfe79869a | ||
|
|
9dff1a901f | ||
|
|
af67e03f85 | ||
|
|
03147d2399 | ||
|
|
1d4c8d2aaf | ||
|
|
5b73c0f63a | ||
|
|
d262bc400c | ||
|
|
b6613057ae | ||
|
|
178528c03e | ||
|
|
7b177613c0 | ||
|
|
1f0e2ab912 | ||
|
|
0146173544 | ||
|
|
54d0efefd7 | ||
|
|
55d28ddf10 | ||
|
|
262ed02103 | ||
|
|
d884ae04ed | ||
|
|
1c330e1b87 | ||
|
|
a50def9cef | ||
|
|
ec46cf8277 | ||
|
|
28b96af905 | ||
|
|
4572974c46 | ||
|
|
cf12786d08 | ||
|
|
1a1243b6dd | ||
|
|
da4125d52b | ||
|
|
7d8d3d4b06 |
@@ -9,7 +9,7 @@
|
||||
{
|
||||
"name": "superpowers",
|
||||
"description": "Core skills library for Claude Code: TDD, debugging, collaboration patterns, and proven techniques",
|
||||
"version": "6.1.1",
|
||||
"version": "6.3.0",
|
||||
"source": "./",
|
||||
"author": {
|
||||
"name": "Jesse Vincent",
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"name": "superpowers",
|
||||
"description": "Core skills library for Claude Code: TDD, debugging, collaboration patterns, and proven techniques",
|
||||
"version": "6.1.1",
|
||||
"version": "6.3.0",
|
||||
"author": {
|
||||
"name": "Jesse Vincent",
|
||||
"email": "jesse@fsck.com"
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "superpowers",
|
||||
"version": "6.1.1",
|
||||
"version": "6.3.0",
|
||||
"description": "An agentic skills framework & software development methodology that works: planning, TDD, debugging, and collaboration workflows.",
|
||||
"author": {
|
||||
"name": "Jesse Vincent",
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
"name": "superpowers",
|
||||
"displayName": "Superpowers",
|
||||
"description": "Core skills library: TDD, debugging, collaboration patterns, and proven techniques",
|
||||
"version": "6.1.1",
|
||||
"version": "6.3.0",
|
||||
"author": {
|
||||
"name": "Jesse Vincent",
|
||||
"email": "jesse@fsck.com"
|
||||
|
||||
22
.devin-plugin/plugin.json
Normal file
22
.devin-plugin/plugin.json
Normal file
@@ -0,0 +1,22 @@
|
||||
{
|
||||
"name": "superpowers",
|
||||
"version": "6.3.0",
|
||||
"description": "An agentic skills framework & software development methodology that works: planning, TDD, debugging, and collaboration workflows.",
|
||||
"author": {
|
||||
"name": "Jesse Vincent",
|
||||
"email": "jesse@fsck.com"
|
||||
},
|
||||
"homepage": "https://github.com/obra/superpowers",
|
||||
"repository": "https://github.com/obra/superpowers",
|
||||
"license": "MIT",
|
||||
"keywords": [
|
||||
"brainstorming",
|
||||
"subagent-driven-development",
|
||||
"skills",
|
||||
"planning",
|
||||
"tdd",
|
||||
"debugging",
|
||||
"code-review",
|
||||
"workflow"
|
||||
]
|
||||
}
|
||||
6
.gitignore
vendored
6
.gitignore
vendored
@@ -11,3 +11,9 @@ triage/
|
||||
# development (see CLAUDE.md / README.md). It is not part of the published
|
||||
# plugin, so the whole directory is ignored here.
|
||||
evals/
|
||||
|
||||
# Python
|
||||
__pycache__/
|
||||
*.pyc
|
||||
*.pyo
|
||||
.pytest_cache/
|
||||
|
||||
104
.hermes-plugin/__init__.py
Normal file
104
.hermes-plugin/__init__.py
Normal file
@@ -0,0 +1,104 @@
|
||||
import os
|
||||
import re
|
||||
from pathlib import Path
|
||||
|
||||
BOOTSTRAP_MARKER = "superpowers:using-superpowers bootstrap for hermes"
|
||||
|
||||
|
||||
def _skills_dir() -> str:
|
||||
"""Locate the stock skills/ tree for either supported install layout.
|
||||
|
||||
- git-clone install (`hermes plugins install obra/superpowers`): the plugin
|
||||
dir is the repo root, so `.hermes-plugin/` and `skills/` are siblings and
|
||||
this module resolves `../skills`.
|
||||
- flattened install (plugin files copied to the plugin dir root): `skills/`
|
||||
sits next to this module.
|
||||
|
||||
Raises loudly when neither matches — a bootstrap that silently skips is how
|
||||
a broken install masquerades as a working one.
|
||||
"""
|
||||
here = os.path.dirname(os.path.realpath(__file__))
|
||||
candidates = (
|
||||
os.path.realpath(os.path.join(here, "..", "skills")),
|
||||
os.path.realpath(os.path.join(here, "skills")),
|
||||
)
|
||||
for cand in candidates:
|
||||
if os.path.isfile(os.path.join(cand, "using-superpowers", "SKILL.md")):
|
||||
return cand
|
||||
raise RuntimeError(
|
||||
"superpowers plugin: cannot find the skills/ tree "
|
||||
f"(looked at {candidates}). Reinstall with "
|
||||
"`hermes plugins install obra/superpowers`."
|
||||
)
|
||||
|
||||
|
||||
def _strip_frontmatter(content: str) -> str:
|
||||
match = re.match(r"^---\n[\s\S]*?\n---\n([\s\S]*)$", content)
|
||||
return (match.group(1) if match else content).strip()
|
||||
|
||||
|
||||
def _build_bootstrap(skills_dir: str) -> str:
|
||||
with open(
|
||||
os.path.join(skills_dir, "using-superpowers", "SKILL.md"),
|
||||
encoding="utf-8",
|
||||
) as f:
|
||||
body = _strip_frontmatter(f.read())
|
||||
|
||||
tools_path = os.path.join(
|
||||
skills_dir, "using-superpowers", "references", "hermes-tools.md"
|
||||
)
|
||||
with open(tools_path, encoding="utf-8") as f:
|
||||
tool_mapping = f.read().strip()
|
||||
|
||||
return (
|
||||
f"<EXTREMELY_IMPORTANT>\n"
|
||||
f"{BOOTSTRAP_MARKER}\n\n"
|
||||
f"You have superpowers.\n\n"
|
||||
f"The using-superpowers skill content is included below and is already "
|
||||
f"loaded for this Hermes session. Follow it now. "
|
||||
f"Do not try to load using-superpowers again.\n\n"
|
||||
f"{body}\n\n"
|
||||
f"## Loading Superpowers Skills on Hermes\n\n"
|
||||
f"Superpowers skills are registered with Hermes' native skill loader: "
|
||||
f'invoke one with `skill_view("superpowers:skill-name")` '
|
||||
f'(for example `skill_view("superpowers:brainstorming")`). '
|
||||
f"If a namespaced lookup returns 'not found', read the skill file "
|
||||
f"directly instead:\n"
|
||||
f'`read_file("{skills_dir}/skill-name/SKILL.md")`\n\n'
|
||||
f"The superpowers skills directory is: `{skills_dir}`\n\n"
|
||||
f"{tool_mapping}\n"
|
||||
f"</EXTREMELY_IMPORTANT>"
|
||||
)
|
||||
|
||||
|
||||
def register(ctx):
|
||||
skills_dir = _skills_dir()
|
||||
bootstrap = _build_bootstrap(skills_dir)
|
||||
|
||||
# Register every stock skill with Hermes' native loader so skill_view can
|
||||
# load them on demand. Standard markdown; no conversion (plugin guide).
|
||||
# register_skill requires a pathlib.Path — a str raises AttributeError and
|
||||
# hermes silently disables the whole plugin (verified 2026-07-23).
|
||||
for name in sorted(os.listdir(skills_dir)):
|
||||
skill_md = os.path.join(skills_dir, name, "SKILL.md")
|
||||
if os.path.isfile(skill_md):
|
||||
ctx.register_skill(name, Path(skill_md))
|
||||
|
||||
# pre_llm_call returning {"context": ...} is the documented injection path
|
||||
# (on_session_start return values are ignored, and ctx.inject_message
|
||||
# refuses from that hook — verified empirically 2026-07-23). The context is
|
||||
# appended to the first turn's user message.
|
||||
def pre_llm_call(
|
||||
session_id=None,
|
||||
user_message=None,
|
||||
conversation_history=None,
|
||||
is_first_turn=None,
|
||||
model=None,
|
||||
platform=None,
|
||||
**kwargs,
|
||||
):
|
||||
if is_first_turn:
|
||||
return {"context": bootstrap}
|
||||
return None
|
||||
|
||||
ctx.register_hook("pre_llm_call", pre_llm_call)
|
||||
6
.hermes-plugin/plugin.yaml
Normal file
6
.hermes-plugin/plugin.yaml
Normal file
@@ -0,0 +1,6 @@
|
||||
name: superpowers
|
||||
version: 6.3.0
|
||||
description: Superpowers skills and workflow bootstrap for Hermes Agent
|
||||
author: obra
|
||||
provides_hooks:
|
||||
- pre_llm_call
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "superpowers",
|
||||
"version": "6.1.1",
|
||||
"version": "6.3.0",
|
||||
"description": "An agentic skills framework and software development methodology.",
|
||||
"author": {
|
||||
"name": "Jesse Vincent",
|
||||
|
||||
@@ -1,9 +1,11 @@
|
||||
{
|
||||
"files": [
|
||||
{ "path": "package.json", "field": "version" },
|
||||
{ "path": ".hermes-plugin/plugin.yaml", "field": "version" },
|
||||
{ "path": ".claude-plugin/plugin.json", "field": "version" },
|
||||
{ "path": ".cursor-plugin/plugin.json", "field": "version" },
|
||||
{ "path": ".codex-plugin/plugin.json", "field": "version" },
|
||||
{ "path": ".devin-plugin/plugin.json", "field": "version" },
|
||||
{ "path": ".kimi-plugin/plugin.json", "field": "version" },
|
||||
{ "path": ".claude-plugin/marketplace.json", "field": "plugins.0.version" },
|
||||
{ "path": "gemini-extension.json", "field": "version" }
|
||||
|
||||
@@ -1,128 +1,130 @@
|
||||
# Contributor Covenant Code of Conduct
|
||||
# Prime Radiant Community Code of Conduct
|
||||
|
||||
## Our Pledge
|
||||
|
||||
We as members, contributors, and leaders pledge to make participation in our
|
||||
community a harassment-free experience for everyone, regardless of age, body
|
||||
size, visible or invisible disability, ethnicity, sex characteristics, gender
|
||||
identity and expression, level of experience, education, socio-economic status,
|
||||
nationality, personal appearance, race, religion, or sexual identity
|
||||
and orientation.
|
||||
We pledge to make our community welcoming, safe, and equitable for all.
|
||||
|
||||
We pledge to act and interact in ways that contribute to an open, welcoming,
|
||||
diverse, inclusive, and healthy community.
|
||||
We are committed to fostering an environment that respects and promotes the dignity, rights, and contributions of all individuals, regardless of characteristics including race, ethnicity, caste, color, age, physical characteristics, neurodiversity, disability, sex or gender, gender identity or expression, sexual orientation, language, philosophy or religion, national or social origin, socio-economic position, level of education, or other status. The same privileges of participation are extended to everyone who participates in good faith and in accordance with this Covenant.
|
||||
|
||||
## Our Standards
|
||||
The guidelines within and enforcement of the Prime Radiant Community Code of Conduct apply equally to everyone participating in the Prime Radiant community, including members of the Prime Radiant team.
|
||||
|
||||
Examples of behavior that contributes to a positive environment for our
|
||||
community include:
|
||||
## Encouraged Behaviors
|
||||
|
||||
* Demonstrating empathy and kindness toward other people
|
||||
* Being respectful of differing opinions, viewpoints, and experiences
|
||||
* Giving and gracefully accepting constructive feedback
|
||||
* Accepting responsibility and apologizing to those affected by our mistakes,
|
||||
and learning from the experience
|
||||
* Focusing on what is best not just for us as individuals, but for the
|
||||
overall community
|
||||
While acknowledging differences in social norms, we all strive to meet our community's expectations for positive behavior. We also understand that our words and actions may be interpreted differently than we intend based on culture, background, or native language.
|
||||
|
||||
Examples of unacceptable behavior include:
|
||||
With these considerations in mind, we agree to behave mindfully toward each other and act in ways that center our shared values, including:
|
||||
|
||||
* The use of sexualized language or imagery, and sexual attention or
|
||||
advances of any kind
|
||||
* Trolling, insulting or derogatory comments, and personal or political attacks
|
||||
* Public or private harassment
|
||||
* Publishing others' private information, such as a physical or email
|
||||
address, without their explicit permission
|
||||
* Other conduct which could reasonably be considered inappropriate in a
|
||||
professional setting
|
||||
1. Respecting the **purpose of our community**, our activities, and our ways of gathering.
|
||||
2. Engaging **kindly and honestly** with others.
|
||||
3. Respecting **different viewpoints** and experiences.
|
||||
4. **Taking responsibility** for our actions and contributions.
|
||||
5. Gracefully giving and accepting **constructive feedback**.
|
||||
6. Committing to **repairing harm** when it occurs.
|
||||
7. Behaving in other ways that promote and sustain the **well-being of our community**.
|
||||
|
||||
## Enforcement Responsibilities
|
||||
## Restricted Behaviors
|
||||
|
||||
Community leaders are responsible for clarifying and enforcing our standards of
|
||||
acceptable behavior and will take appropriate and fair corrective action in
|
||||
response to any behavior that they deem inappropriate, threatening, offensive,
|
||||
or harmful.
|
||||
We agree to restrict the following behaviors in our community. Instances, threats, and promotion of these behaviors are violations of this Code of Conduct.
|
||||
|
||||
Community leaders have the right and responsibility to remove, edit, or reject
|
||||
comments, commits, code, wiki edits, issues, and other contributions that are
|
||||
not aligned to this Code of Conduct, and will communicate reasons for moderation
|
||||
decisions when appropriate.
|
||||
1. **Harassment.** Violating explicitly expressed boundaries or engaging in unnecessary personal attention after any clear request to stop.
|
||||
2. **Character attacks.** Making insulting, demeaning, or pejorative comments directed at a community member or group of people.
|
||||
3. **Inciting conflict.** Deliberately engaging in discussions meant to cause arguments or a hostile environment.
|
||||
4. **Stereotyping or discrimination.** Characterizing anyone’s personality or behavior on the basis of immutable identities or traits.
|
||||
5. **Sexualization.** Behaving in a way that would generally be considered inappropriately intimate in the context or purpose of the community.
|
||||
6. **Violating confidentiality.** Sharing or acting on someone's personal or private information without their permission.
|
||||
7. **Endangerment.** Causing, encouraging, or threatening violence or other harm toward any person or group.
|
||||
8. Behaving in other ways that **threaten the well-being** of our community.
|
||||
|
||||
### Other Restrictions
|
||||
|
||||
1. **Divisive topics.** Discussing inflammatory topics that are unrelated to the community as a whole.
|
||||
2. **Offensive content.** Any text or image that is offensive or violates any of the other restricted behaviors, including as part of a username, profile, status, avatar, or other publicly displayed identifier.
|
||||
3. **Misleading identity.** Impersonating someone else for any reason, misrepresenting yourself as associated with Prime Radiant or any company, or pretending to be someone else to evade enforcement actions.
|
||||
4. **Failing to credit sources.** Not properly crediting the sources of content you contribute, or representing work created by someone else as your own.
|
||||
5. **Advertising and promotional materials.** Sharing marketing or other commercial content, invite links, or irrelevant self-promotion, as well as buying, trading, or asking for donations.
|
||||
6. **Spam posts.** Spamming, including, but not limited to, posting a flood of messages in a short period of time, irrelevant content, or excessive links.
|
||||
7. **Unsolicited mentions and direct messages.** Engaging in harassment by excessively mentioning someone by username or replying, or direct messaging someone without explicit invitation.
|
||||
8. **Irresponsible communication.** Failing to responsibly present content which includes, links, or describes any other restricted behaviors.
|
||||
9. Other conduct that could reasonably be considered **unprofessional** or **inappropriate**.
|
||||
|
||||
## Reporting an Issue
|
||||
|
||||
Tensions can occur between community members even when they are trying their best to collaborate. Not every conflict represents a code of conduct violation, and this Code of Conduct reinforces encouraged behaviors and norms that can help avoid conflicts and minimize harm. You are welcome to report concerns, even if they seem minor, as they can be helpful in identifying patterns of behavior that may not be concerning in isolation, but when viewed collectively may be more significant.
|
||||
|
||||
When an incident does occur, it is important to report it promptly. To report a possible violation anywhere in the community, email [conduct@primeradiant.com](mailto:conduct@primeradiant.com). On the Prime Radiant Discord server, you can mention `@moderators` in a public channel, or report via a support ticket, created through the `#support-ticket` channel. In the event that you need to report a member of the Prime Radiant team, you can contact Kattni at [kattni@primeradiant.com](mailto:kattni@primeradiant.com) or Drew at [drew@primeradiant.com](mailto:drew@primeradiant.com).
|
||||
|
||||
Community Moderators take reports of violations seriously and will make every effort to respond in a timely manner. They will investigate all reports of code of conduct violations, reviewing messages, logs, and recordings, or interviewing witnesses and other participants. Community Moderators will keep investigation and enforcement actions as transparent as possible while prioritizing safety and confidentiality. In order to honor these values, enforcement actions are carried out in private with the involved parties, but communicating to the whole community may be part of a mutually agreed upon resolution. If moderators determine that a public statement needs to be made, the identities of all victims and reporters will remain confidential unless those individuals instruct otherwise.
|
||||
|
||||
In your report, please include:
|
||||
|
||||
- **Your contact info** so the team can get in touch with you if they need to follow up.
|
||||
- **Names (real, nicknames, or pseudonyms) of any individuals involved.** If there were other witnesses besides you, please try to include them as well.
|
||||
- **When and where the incident occurred.** Please be as specific as possible.
|
||||
- **Your account of what occurred.** If there is a publicly available record (e.g. a Discord or GitHub message) please include a link.
|
||||
- **Any extra context** you believe existed for the incident.
|
||||
- **If you believe this incident is ongoing.**
|
||||
- **If you believe any member of the team has a conflict of interest** in adjudicating the incident.
|
||||
- **What, if any, corrective response** you believe would be appropriate.
|
||||
- **Any other information** you believe the team should have.
|
||||
|
||||
Moderators are obligated to maintain confidentiality with regard to the reporter and details of an incident.
|
||||
|
||||
## Report Followup
|
||||
|
||||
You will receive a response acknowledging receipt of your report within 24 business hours.
|
||||
|
||||
If a member of the team is one of the named parties, they will not be included in any discussions, and will not be provided with any confidential details from the reporter.
|
||||
|
||||
If anyone on the moderation team believes they have a conflict of interest in adjudicating on a reported issue, they will inform the other team members, and recuse themselves from any discussion about the issue. Following this declaration, they will not be provided with any confidential details from the reporter.
|
||||
|
||||
The team will immediately review the incident and determine:
|
||||
|
||||
- What happened.
|
||||
- Whether this event constitutes a code of conduct violation.
|
||||
- Who the reported person is.
|
||||
- Whether this is an ongoing situation, or if there is a threat to anyone's physical safety.
|
||||
|
||||
If this is determined to be an ongoing incident or a threat to physical safety, the team's immediate priority will be to protect everyone involved. This means they may delay an official response until they believe that the situation has concluded and that everyone is physically safe.
|
||||
|
||||
The moderation team will respond within one week to the person who filed the report with either a resolution or an explanation of why the situation is not yet resolved.
|
||||
|
||||
Once the team has determined their final action, they'll contact the reporter to let them know what action (if any) they'll be taking. They'll take into account feedback from the reporter on the appropriateness of the response, but do not guarantee they'll act on it.
|
||||
|
||||
Finally, to maintain transparency in the reporting and enforcement process, whenever possible, a public transparency report of the incident will be made. A public report may not be made if the specifics of the incident do not allow the team to preserve anonymity, or if there is potential for ongoing harm.
|
||||
|
||||
## Addressing and Repairing Harm
|
||||
|
||||
If an investigation by the Community Moderators finds that this Code of Conduct has been violated, the following enforcement ladder may be used to determine how best to repair harm, based on the incident's impact on the individuals involved and the community as a whole. Depending on the severity of a violation, lower rungs on the ladder may be skipped.
|
||||
|
||||
1) Warning
|
||||
1) Event: A violation involving a single incident or series of incidents.
|
||||
2) Consequence: A private, written warning from the Community Moderators.
|
||||
3) Repair: Examples of repair include a private written apology, acknowledgement of responsibility, and seeking clarification on expectations.
|
||||
2) Temporarily Limited Activities
|
||||
1) Event: A repeated incidence of a violation that previously resulted in a warning, or the first incidence of a more serious violation.
|
||||
2) Consequence: A private, written warning with a time-limited cooldown period designed to underscore the seriousness of the situation and give the community members involved time to process the incident. The cooldown period may be limited to particular communication channels or interactions with particular community members.
|
||||
3) Repair: Examples of repair may include making an apology, using the cooldown period to reflect on actions and impact, and being thoughtful about re-entering community spaces after the period is over.
|
||||
3) Temporary Suspension
|
||||
1) Event: A pattern of repeated violation which the Community Moderators have tried to address with warnings, or a single serious violation.
|
||||
2) Consequence: A private written warning with conditions for return from suspension. In general, temporary suspensions give the person being suspended time to reflect upon their behavior and possible corrective actions.
|
||||
3) Repair: Examples of repair include respecting the spirit of the suspension, meeting the specified conditions for return, and being thoughtful about how to reintegrate with the community when the suspension is lifted.
|
||||
4) Permanent Ban
|
||||
1) Event: A pattern of repeated code of conduct violations that other steps on the ladder have failed to resolve, or a violation so serious that the Community Moderators determine there is no way to keep the community safe with this person as a member.
|
||||
2) Consequence: Access to all community spaces, tools, and communication channels is removed. In general, permanent bans should be rarely used, should have strong reasoning behind them, and should only be resorted to if working through other remedies has failed to change the behavior.
|
||||
3) Repair: There is no possible repair in cases of this severity.
|
||||
|
||||
This enforcement ladder is intended as a guideline. It does not limit the ability of Community Managers to use their discretion and judgment, in keeping with the best interests of our community.
|
||||
|
||||
## Scope
|
||||
|
||||
This Code of Conduct applies within all community spaces, and also applies when
|
||||
an individual is officially representing the community in public spaces.
|
||||
Examples of representing our community include using an official e-mail address,
|
||||
posting via an official social media account, or acting as an appointed
|
||||
representative at an online or offline event.
|
||||
This Code of Conduct applies within all community spaces, including GitHub and the Prime Radiant Discord server. It also applies when an individual is officially representing the community in public or other spaces. Examples of representing the community include using an official email address, posting via an official social media account, or acting as an appointed representative at an online or offline event.
|
||||
|
||||
## Enforcement
|
||||
|
||||
Instances of abusive, harassing, or otherwise unacceptable behavior may be
|
||||
reported to the community leaders responsible for enforcement at
|
||||
jesse@primeradiant.com.
|
||||
All complaints will be reviewed and investigated promptly and fairly.
|
||||
|
||||
All community leaders are obligated to respect the privacy and security of the
|
||||
reporter of any incident.
|
||||
|
||||
## Enforcement Guidelines
|
||||
|
||||
Community leaders will follow these Community Impact Guidelines in determining
|
||||
the consequences for any action they deem in violation of this Code of Conduct:
|
||||
|
||||
### 1. Correction
|
||||
|
||||
**Community Impact**: Use of inappropriate language or other behavior deemed
|
||||
unprofessional or unwelcome in the community.
|
||||
|
||||
**Consequence**: A private, written warning from community leaders, providing
|
||||
clarity around the nature of the violation and an explanation of why the
|
||||
behavior was inappropriate. A public apology may be requested.
|
||||
|
||||
### 2. Warning
|
||||
|
||||
**Community Impact**: A violation through a single incident or series
|
||||
of actions.
|
||||
|
||||
**Consequence**: A warning with consequences for continued behavior. No
|
||||
interaction with the people involved, including unsolicited interaction with
|
||||
those enforcing the Code of Conduct, for a specified period of time. This
|
||||
includes avoiding interactions in community spaces as well as external channels
|
||||
like social media. Violating these terms may lead to a temporary or
|
||||
permanent ban.
|
||||
|
||||
### 3. Temporary Ban
|
||||
|
||||
**Community Impact**: A serious violation of community standards, including
|
||||
sustained inappropriate behavior.
|
||||
|
||||
**Consequence**: A temporary ban from any sort of interaction or public
|
||||
communication with the community for a specified period of time. No public or
|
||||
private interaction with the people involved, including unsolicited interaction
|
||||
with those enforcing the Code of Conduct, is allowed during this period.
|
||||
Violating these terms may lead to a permanent ban.
|
||||
|
||||
### 4. Permanent Ban
|
||||
|
||||
**Community Impact**: Demonstrating a pattern of violation of community
|
||||
standards, including sustained inappropriate behavior, harassment of an
|
||||
individual, or aggression toward or disparagement of classes of individuals.
|
||||
|
||||
**Consequence**: A permanent ban from any sort of public interaction within
|
||||
the community.
|
||||
Behavior outside of official Prime Radiant spaces may also be considered as supporting evidence for a report if that behavior establishes a pattern, or represents a potential risk to the Prime Radiant community.
|
||||
|
||||
## Attribution
|
||||
|
||||
This Code of Conduct is adapted from the [Contributor Covenant][homepage],
|
||||
version 2.0, available at
|
||||
https://www.contributor-covenant.org/version/2/0/code_of_conduct.html.
|
||||
This Code of Conduct is adapted from the Contributor Covenant, version 3.0, permanently available at [https://www.contributor-covenant.org/version/3/0/](https://www.contributor-covenant.org/version/3/0/).
|
||||
|
||||
Community Impact Guidelines were inspired by [Mozilla's code of conduct
|
||||
enforcement ladder](https://github.com/mozilla/diversity).
|
||||
Contributor Covenant is stewarded by the Organization for Ethical Source and licensed under CC BY-SA 4.0. To view a copy of this license, visit [https://creativecommons.org/licenses/by-sa/4.0/](https://creativecommons.org/licenses/by-sa/4.0/)
|
||||
|
||||
[homepage]: https://www.contributor-covenant.org
|
||||
|
||||
For answers to common questions about this code of conduct, see the FAQ at
|
||||
https://www.contributor-covenant.org/faq. Translations are available at
|
||||
https://www.contributor-covenant.org/translations.
|
||||
For answers to common questions about Contributor Covenant, see the FAQ at [https://www.contributor-covenant.org/faq](https://www.contributor-covenant.org/faq). Translations are provided at [https://www.contributor-covenant.org/translations](https://www.contributor-covenant.org/translations). Additional enforcement and community guideline resources can be found at [https://www.contributor-covenant.org/resources](https://www.contributor-covenant.org/resources). The enforcement ladder was inspired by the work of [Mozilla’s code of conduct team](https://github.com/mozilla/inclusion).
|
||||
|
||||
94
README.md
94
README.md
@@ -2,16 +2,33 @@
|
||||
|
||||
Superpowers is a complete software development methodology for your coding agents, built on top of a set of composable skills and some initial instructions that make sure your agent uses them.
|
||||
|
||||
## Table of Contents
|
||||
|
||||
## We're Hiring!
|
||||
|
||||
We're hiring someone to help out full time with Superpowers community and code work.
|
||||
You can read about the job at https://primeradiant.com/jobs/superpowers-community-engineer/
|
||||
If this sounds like someone you know, definitely send them our way.
|
||||
|
||||
## Quickstart
|
||||
|
||||
Give your agent Superpowers: [Claude Code](#claude-code), [Antigravity](#antigravity), [Codex App](#codex-app), [Codex CLI](#codex-cli), [Cursor](#cursor), [Factory Droid](#factory-droid), [Gemini CLI](#gemini-cli), [GitHub Copilot CLI](#github-copilot-cli), [Kimi Code](#kimi-code), [OpenCode](#opencode), [Pi](#pi).
|
||||
- [How it works](#how-it-works)
|
||||
- [Commercial Services](#commercial-services)
|
||||
- [Getting Started](#installation)
|
||||
- [Claude Code](#claude-code)
|
||||
- [Antigravity](#antigravity)
|
||||
- [Codex App](#codex-app)
|
||||
- [Codex CLI](#codex-cli)
|
||||
- [Cursor](#cursor)
|
||||
- [Devin CLI](#devin-cli)
|
||||
- [Factory Droid](#factory-droid)
|
||||
- [Gemini CLI](#gemini-cli)
|
||||
- [GitHub Copilot CLI](#github-copilot-cli)
|
||||
- [Grok Build CLI](#grok-build-cli)
|
||||
- [Kimi Code](#kimi-code)
|
||||
- [OpenCode](#opencode)
|
||||
- [Pi](#pi)
|
||||
- [Hermes Agent](#hermes-agent)
|
||||
- [The Basic Workflow](#the-basic-workflow)
|
||||
- [Community](#community)
|
||||
- [What's Inside](#whats-inside)
|
||||
- [Philosophy](#philosophy)
|
||||
- [Contributing](#contributing)
|
||||
- [Updating](#updating)
|
||||
- [License](#license)
|
||||
- [Visual companion telemetry](#visual-companion-telemetry)
|
||||
|
||||
## How it works
|
||||
|
||||
@@ -108,6 +125,20 @@ Superpowers is available via the [official Codex plugin marketplace](https://git
|
||||
|
||||
- Or search for "superpowers" in the plugin marketplace.
|
||||
|
||||
### Devin CLI
|
||||
|
||||
- Install the plugin from this repository:
|
||||
|
||||
```bash
|
||||
devin plugins install obra/superpowers
|
||||
```
|
||||
|
||||
- Update to the latest version with:
|
||||
|
||||
```bash
|
||||
devin plugins update superpowers
|
||||
```
|
||||
|
||||
### Factory Droid
|
||||
|
||||
- Register the marketplace:
|
||||
@@ -150,6 +181,22 @@ Superpowers is available via the [official Codex plugin marketplace](https://git
|
||||
copilot plugin install superpowers@superpowers-marketplace
|
||||
```
|
||||
|
||||
### Grok Build CLI
|
||||
|
||||
Superpowers is available via the [official Grok plugin marketplace](https://github.com/xai-org/plugin-marketplace).
|
||||
|
||||
- Install the plugin from xAI's official marketplace:
|
||||
|
||||
```bash
|
||||
grok plugin install superpowers@xai-official --trust
|
||||
```
|
||||
|
||||
- Or open the marketplace in the TUI, search for Superpowers, and install it:
|
||||
|
||||
```text
|
||||
/marketplace
|
||||
```
|
||||
|
||||
### Kimi Code
|
||||
|
||||
Superpowers is available in Kimi Code's plugin marketplace.
|
||||
@@ -199,6 +246,18 @@ pi -e /path/to/superpowers
|
||||
|
||||
The Pi package loads the Superpowers skills and a small extension that injects the `using-superpowers` bootstrap at session startup and again after compaction. Pi has native skills, so no compatibility `Skill` tool is required. Subagent and task-list tools remain optional Pi companion packages.
|
||||
|
||||
### Hermes Agent
|
||||
|
||||
Install Superpowers as a Hermes plugin from this repository:
|
||||
|
||||
```bash
|
||||
hermes plugins install obra/superpowers --enable
|
||||
```
|
||||
|
||||
Restart any active Hermes sessions after installing. Note: Hermes has no
|
||||
post-compaction hook, so a very long session that compacts over its first
|
||||
turn loses the bootstrap — start a fresh session if skills stop triggering.
|
||||
|
||||
## The Basic Workflow
|
||||
|
||||
1. **brainstorming** - Activates before writing code. Refines rough ideas through questions, explores alternatives, presents design in sections for validation. Saves design document.
|
||||
@@ -217,6 +276,14 @@ The Pi package loads the Superpowers skills and a small extension that injects t
|
||||
|
||||
**The agent checks for relevant skills before any task.** Mandatory workflows, not suggestions.
|
||||
|
||||
## Community
|
||||
|
||||
Superpowers is built by [Jesse Vincent](https://blog.fsck.com) and the rest of the folks at [Prime Radiant](https://primeradiant.com).
|
||||
|
||||
- **Discord**: [Join us](https://discord.gg/35wsABTejz) for community support, questions, and sharing what you're building with Superpowers
|
||||
- **Issues**: https://github.com/obra/superpowers/issues
|
||||
- **Release announcements**: [Sign up](https://primeradiant.com/superpowers/) to get notified about new versions
|
||||
|
||||
## What's Inside
|
||||
|
||||
### Skills Library
|
||||
@@ -227,6 +294,7 @@ The Pi package loads the Superpowers skills and a small extension that injects t
|
||||
**Debugging**
|
||||
- **systematic-debugging** - 4-phase root cause process (includes root-cause-tracing, defense-in-depth, condition-based-waiting techniques)
|
||||
- **verification-before-completion** - Ensure it's actually fixed
|
||||
- **diagnosing-superpowers** - Work out what went wrong in a session, with evidence; export a scrubbed bundle or file an issue
|
||||
|
||||
**Collaboration**
|
||||
- **brainstorming** - Socratic design refinement
|
||||
@@ -277,11 +345,3 @@ MIT License - see LICENSE file for details
|
||||
## Visual companion telemetry
|
||||
|
||||
Because skills and plugins don't provide any feedback to creators, we have no idea how many of you are using Superpowers. By default, the Prime Radiant logo on brainstorming's optional visual companion feature is loaded from our website. It includes the version of Superpowers in use. It does not include any details about your project, prompt, or coding agent. We don't see your clicks or anything about what you're building. This helps us have a rough idea of how many folks are using Superpowers and which version of Superpowers they're using. It's 100% optional. To disable this, set the environment variable `SUPERPOWERS_DISABLE_TELEMETRY` to any true value. Superpowers also honors Claude Code's `DISABLE_TELEMETRY` and `CLAUDE_CODE_DISABLE_NONESSENTIAL_TRAFFIC` opt-outs.
|
||||
|
||||
## Community
|
||||
|
||||
Superpowers is built by [Jesse Vincent](https://blog.fsck.com) and the rest of the folks at [Prime Radiant](https://primeradiant.com).
|
||||
|
||||
- **Discord**: [Join us](https://discord.gg/35wsABTejz) for community support, questions, and sharing what you're building with Superpowers
|
||||
- **Issues**: https://github.com/obra/superpowers/issues
|
||||
- **Release announcements**: [Sign up](https://primeradiant.com/superpowers/) to get notified about new versions
|
||||
|
||||
@@ -1,5 +1,77 @@
|
||||
# Superpowers Release Notes
|
||||
|
||||
## v6.3.0 (2026-08-12)
|
||||
|
||||
### Harness Support
|
||||
|
||||
- **Devin CLI**: `devin plugins install obra/superpowers` now works, and skills auto-trigger at session start. (#1995)
|
||||
- **Hermes Agent**: install from a git clone; skills register with Hermes' native loader and the bootstrap loads on the first turn. (#1922, #2025)
|
||||
- **Grok Build CLI** added to the install docs. (#1919)
|
||||
|
||||
### Brainstorming
|
||||
|
||||
- **Ceremony now scales to the task.** Requests are classified as spike, bounded, or architectural; small tasks skip the two-document ritual. Every path still stops for your approval before implementation. (#2063)
|
||||
|
||||
### Subagent-Driven Development
|
||||
|
||||
- **Controllers no longer stall on plan conflicts.** Non-catastrophic conflicts and ambiguities get a recorded ruling and work continues; only destructive or irreversible actions still stop for a human. One donated session had sat blocked for almost nine hours on a question the controller could have decided. (#2077)
|
||||
- **The pre-dispatch conflict scan records its checks in the ledger** instead of just asserting the plan is clean. (#2080)
|
||||
- **Small same-shape tasks batch into one dispatch**, cutting subagent cost sharply on micro-task plans; batch reviews verify every file in the brief made it into the diff. (#2078)
|
||||
- **Implementers and reviewers may not spawn their own subagents**, which was producing duplicate reviews. (#2059)
|
||||
- **Plans carry a `Spec:` pointer** and SDD reads the spec at setup, so plan conflicts get resolved against the design instead of guessed at. (#2086)
|
||||
- Reviewers re-read evidence they find illegible instead of re-running the test suite (#2089), and circuit-breaker rulings now show up in the Finish report.
|
||||
|
||||
### Codex
|
||||
|
||||
- Subagent waits are event-driven instead of poll-heavy, spawns pin model and reasoning effort explicitly, and the multi-agent reference is corrected against Codex source. (#2060, #2061, #2062)
|
||||
|
||||
### Finishing a Development Branch
|
||||
|
||||
- **Worktree removal no longer destroys untracked files.** When `git worktree remove` refuses because the tree holds uncommitted work, the skill stops, names the files, and asks — instead of reaching for `--force`. (#2016, #1223, #2024)
|
||||
|
||||
### Fixes
|
||||
|
||||
- `render-graphs.js` in writing-skills works on Windows.
|
||||
- Corrected Copilot CLI backgrounding guidance for Windows. (#1929, #2006)
|
||||
- `bump-version.sh` covers the Hermes manifest.
|
||||
|
||||
### Documentation
|
||||
|
||||
- README: added a table of contents and reorganized Getting Started.
|
||||
|
||||
## v6.2.0 (2026-07-23)
|
||||
|
||||
### Subagent-Driven Development
|
||||
|
||||
Two structural changes to how SDD tracks progress and closes out review findings, both developed against live eval campaigns.
|
||||
|
||||
- **The workspace is now plan-scoped.** `.superpowers/sdd/` had no plan identity and no end-of-life: a follow-up plan in the same working tree could read the previous plan's ledger as its own progress (observed in the wild, with multiple contamination rounds and ad-hoc workarounds). `sdd-workspace` now requires the plan file and resolves a per-plan directory, `.superpowers/sdd/<plan-basename>/`; `task-brief` and `review-package` write into their plan's directory (`review-package` gains the plan file as its first argument); the ledger names its plan on its first line; and the workspace is deleted once the final review is clean — git history is the durable record. Baseline evals showed controllers already refused foreign ledgers, but at a cost of 6–13 tool calls of cross-plan git forensics per resume; plan-scoping makes the answer structural instead. (25/25 baseline and GREEN eval runs documented in `docs/specs/` and `docs/plans/`.)
|
||||
- **The review-fix loop resumes the implementer.** The lifecycle restructure gives fix rounds resume-the-implementer semantics instead of fresh dispatches, adds a scoped re-review prompt (`re-review-prompt.md`) so the re-reviewer checks the fixes rather than re-reading the whole task, and installs a five-round circuit breaker with controller adjudication when it trips. SKILL.md reorganizes by lifecycle, and its Red Flags convert to the house rationalization-table form.
|
||||
|
||||
### Skills
|
||||
|
||||
A branch-wide compression campaign: recap sections, social proof, and benefits-selling prose aimed at a reader who has already invoked the skill are gone, with every load-bearing argument folded into a rationalization-table row or moved to its point of use. Each cut was micro-tested with subagent probes, and the one cut that measurably degraded behavior was reworked rather than shipped.
|
||||
|
||||
- **`testing-anti-patterns.md` is now `writing-good-tests.md`.** The TDD reference doc is rebuilt as a positive catalog — six rules that lead with the GOOD example — and absorbs a falsifiability discipline: name the production change that would fail the test, derive expectations independently of the code under test, and a closing mutation check. It closes two holes by name: the string-presence trap (grep-style tests on scripts, skills, and prompts counterfeit falsifiability — the observable is behavior, never text) and the change-detector trap (a constant assertion can fail and still protect nothing), each with a hard stop in the gate function. Trivial code and human prose earn no test; the trigger broadens from "adding mocks" to any test writing.
|
||||
- **TDD's "Why Order Matters" rebuttals survive as rationalization rows.** Deleting the section outright measurably degraded test-first behavior under "just write it, tests after" pressure (control 8/10 → treatment 5/10, corroborated on Claude and Codex), so each prose rebuttal now lives in its Common Rationalizations row — the section is gone but the arguments fire where an agent hits them mid-rationalization.
|
||||
- **`finishing-a-development-branch` no longer offers to discard your work.** The completion menu dates from when throwing away branches was routine; "Discard this work" next to "Merge" advertised destroying finished, passing work. Discard survives as an explicit-request-only path with the same typed-confirmation ritual. The same pass made PR creation forge-agnostic (your forge's CLI or the URL printed on push, not a blessed list of tools) and fixed a real bug: the worktree path was recomputed after cleanup had already changed directory, so provenance checks never matched and cleanup silently no-oped.
|
||||
- **Recap and persuasion prose removed across the library.** `brainstorming`, `systematic-debugging`, `dispatching-parallel-agents`, `verification-before-completion`, `executing-plans`, `subagent-driven-development`, `requesting-code-review`, `receiving-code-review`, `using-git-worktrees`, `writing-plans`, and `writing-skills` all drop their Bottom Line / Key Principles / Real-World Impact / Advantages sections; `using-git-worktrees` and `finishing-a-development-branch` convert their guard sections to the house Excuse/Reality rationalization table.
|
||||
|
||||
### Windows
|
||||
|
||||
- **The SessionStart hook now dispatches via Git Bash.** The hook's command string starts with a quoted path, which broke both shells Claude Code might hand it to: PowerShell parsed the quoted string as an expression and died with a parser error (#1751), and cmd.exe's quote-stripping rule truncated the command when the profile path contained a metacharacter like `(` (#1918) — either way the bootstrap silently never loaded. The hook now declares `shell: "bash"`, which Claude Code ≥ 2.1.81 resolves to Git for Windows directly, and which surfaces an actionable install prompt when Git Bash is missing. Older Claude Code versions ignore the unknown key and behave as before. Verified end-to-end on Linux, Windows 11 with Git Bash under a hostile path, and Windows 11 without Git Bash.
|
||||
|
||||
### Harness Support
|
||||
|
||||
- **Gemini CLI support is restored.** The v6.1.0 removal (on the news that Google had EOLed the Gemini CLI) was premature; the install docs and the `gemini-tools.md` tool-mapping reference are back while permanent removal gets a proper evaluation. (#1959)
|
||||
|
||||
### Fixes
|
||||
|
||||
- **`find-polluter.sh` actually finds test files now.** `find .` emits `./`-prefixed paths, so the documented `-path "src/**/*.test.ts"` pattern matched nothing — and `wc -l` on empty input then reported "Found 1". Fixed the prefix mismatch (#2008, #2011), plus two follow-ups: a caller-supplied `./`-prefixed pattern no longer double-prefixes into a never-matching form, and `**/` is also matched collapsed so tests directly under the base directory (`src/top.test.ts` vs `src/**/*.test.ts`) aren't silently skipped. The script gains a deterministic test suite.
|
||||
- **The Codex package script works beyond macOS.** Deterministic-metadata tar flags were bsdtar-only spellings, staged file modes depended on two umasks canceling out, and the test's timestamp assertion parsed bsdtar's column layout in a US timezone. GNU tar now gets equivalent flags producing byte-identical headers, modes are pinned canonical, and the test asserts mtime via `tarfile`.
|
||||
- **SDD's skill test no longer flakes.** The file's worst case exceeded the runner's per-file ceiling (raised to 900s), and the assert helpers matched free-form model prose case-sensitively; matching is now case-insensitive and `assert_order` dumps output on failure so the next flake is diagnosable.
|
||||
- **Docs and test cleanup after the v6.1.0 reference pruning.** Dead links to the deleted `claude-code-tools.md`/`copilot-tools.md` are replaced with the current architecture (#1969), a dangling `#subagent-support` anchor in the Antigravity reference is dropped (#2010), and the Antigravity/Pi mapping tests assert only the surviving harness-specific mappings — scoped to the table so they fail again if it's deleted.
|
||||
|
||||
## v6.1.1 (2026-07-02)
|
||||
|
||||
### Codex
|
||||
|
||||
1009
docs/superpowers/plans/2026-07-30-codex-efficiency-fixes.md
Normal file
1009
docs/superpowers/plans/2026-07-30-codex-efficiency-fixes.md
Normal file
File diff suppressed because it is too large
Load Diff
304
docs/superpowers/plans/2026-08-06-hermes-version-bump-wiring.md
Normal file
304
docs/superpowers/plans/2026-08-06-hermes-version-bump-wiring.md
Normal file
@@ -0,0 +1,304 @@
|
||||
# Hermes Version-Bump Wiring Implementation Plan
|
||||
|
||||
> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking.
|
||||
|
||||
**Goal:** Keep the Hermes YAML manifest version synchronized with every other declared release manifest.
|
||||
|
||||
**Spec:** `docs/superpowers/specs/2026-08-05-hermes-version-bump-wiring-design.md`
|
||||
|
||||
**Architecture:** Extend the existing release script with a small extension-based dispatcher: JSON continues through `jq`, while `.yaml` uses Mike Farah `yq` v4. Before the mutating bump loop, read every present manifest through that dispatcher so deterministic format or field failures occur before the first write.
|
||||
|
||||
**Tech Stack:** Bash 3.2-compatible shell, `jq`, Mike Farah `yq` v4, existing shell-lint tooling.
|
||||
|
||||
## Global Constraints
|
||||
|
||||
- Support only `.json` and `.yaml`; `.yml` and other extensions remain unsupported.
|
||||
- YAML fields are present top-level strings; nested YAML fields are out of scope.
|
||||
- Pass the YAML field and new value through environment data, never interpolate either into a `yq` expression.
|
||||
- Keep `yq` confined to maintainer release tooling; do not add a plugin runtime dependency.
|
||||
- Preserve the existing missing-file behavior: `--check` reports missing files and a bump skips them.
|
||||
- Preflight only the mutating bump path; do not add rollback or transactional writes.
|
||||
- Do not change audit status behavior, version validation, or the existing JSON field-expression implementation.
|
||||
|
||||
---
|
||||
|
||||
## File Map
|
||||
|
||||
- Create: `tests/version-bump/test-bump-version.sh`
|
||||
- Exercise the real script in temporary JSON/YAML fixtures and check the real registry.
|
||||
- Modify: `scripts/bump-version.sh`
|
||||
- Add YAML read/write helpers, format dispatch, and bump-only read preflight.
|
||||
- Modify: `.version-bump.json`
|
||||
- Register `.hermes-plugin/plugin.yaml` at top-level field `version`.
|
||||
|
||||
### Task 1: Wire Hermes Into The Existing Version-Bump Script
|
||||
|
||||
**Files:**
|
||||
- Create: `tests/version-bump/test-bump-version.sh`
|
||||
- Modify: `scripts/bump-version.sh`
|
||||
- Modify: `.version-bump.json`
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: `.version-bump.json` records shaped as `{ "path": string, "field": string }`.
|
||||
- Produces: `read_manifest_field FILE FIELD`, `write_manifest_field FILE FIELD VALUE`, and `preflight_manifests` Bash helpers.
|
||||
|
||||
- [ ] **Step 1: Fetch the current development base**
|
||||
|
||||
Run:
|
||||
|
||||
```bash
|
||||
git fetch origin dev
|
||||
```
|
||||
|
||||
Expected: command exits 0 and refreshes `origin/dev`.
|
||||
|
||||
- [ ] **Step 2: Rebase the task branch**
|
||||
|
||||
Run:
|
||||
|
||||
```bash
|
||||
git rebase origin/dev
|
||||
```
|
||||
|
||||
Expected: command exits 0, and `git status --short --branch` no longer reports the branch behind `origin/dev`.
|
||||
|
||||
- [ ] **Step 3: Add the initial failing behavioral test**
|
||||
|
||||
Create `tests/version-bump/test-bump-version.sh` with the happy-path fixture and real registry assertion:
|
||||
|
||||
```bash
|
||||
#!/usr/bin/env bash
|
||||
set -euo pipefail
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
|
||||
REPO_ROOT="$(cd "$SCRIPT_DIR/../.." && pwd)"
|
||||
SCRIPT_SOURCE="$REPO_ROOT/scripts/bump-version.sh"
|
||||
TEST_ROOT="$(mktemp -d)"
|
||||
|
||||
cleanup() {
|
||||
rm -rf "$TEST_ROOT"
|
||||
}
|
||||
trap cleanup EXIT
|
||||
|
||||
fail() {
|
||||
echo "FAIL: $*" >&2
|
||||
exit 1
|
||||
}
|
||||
|
||||
make_fixture() {
|
||||
local repo="$1"
|
||||
local yaml_body="$2"
|
||||
|
||||
mkdir -p "$repo/scripts" "$repo/.hermes-plugin"
|
||||
cp "$SCRIPT_SOURCE" "$repo/scripts/bump-version.sh"
|
||||
cat >"$repo/.version-bump.json" <<'JSON'
|
||||
{
|
||||
"files": [
|
||||
{ "path": "package.json", "field": "version" },
|
||||
{ "path": ".hermes-plugin/plugin.yaml", "field": "version" }
|
||||
],
|
||||
"audit": { "exclude": [] }
|
||||
}
|
||||
JSON
|
||||
cat >"$repo/package.json" <<'JSON'
|
||||
{
|
||||
"name": "fixture",
|
||||
"version": "1.2.3"
|
||||
}
|
||||
JSON
|
||||
printf '%s\n' "$yaml_body" >"$repo/.hermes-plugin/plugin.yaml"
|
||||
}
|
||||
|
||||
happy_repo="$TEST_ROOT/happy"
|
||||
make_fixture "$happy_repo" $'name: superpowers\nversion: 1.2.3'
|
||||
|
||||
/bin/bash "$happy_repo/scripts/bump-version.sh" --check >"$TEST_ROOT/check.out"
|
||||
/bin/bash "$happy_repo/scripts/bump-version.sh" --audit >"$TEST_ROOT/audit.out"
|
||||
/bin/bash "$happy_repo/scripts/bump-version.sh" 2.3.4 >"$TEST_ROOT/bump.out"
|
||||
|
||||
[[ "$(jq -r '.version' "$happy_repo/package.json")" == "2.3.4" ]] \
|
||||
|| fail "JSON manifest was not bumped"
|
||||
[[ "$(yq -r '.version' "$happy_repo/.hermes-plugin/plugin.yaml")" == "2.3.4" ]] \
|
||||
|| fail "YAML manifest was not bumped"
|
||||
|
||||
jq -e '
|
||||
any(.files[];
|
||||
.path == ".hermes-plugin/plugin.yaml" and .field == "version")
|
||||
' "$REPO_ROOT/.version-bump.json" >/dev/null \
|
||||
|| fail "Hermes manifest is not registered"
|
||||
|
||||
echo "Version-bump tests passed"
|
||||
```
|
||||
|
||||
- [ ] **Step 4: Run the test to verify RED**
|
||||
|
||||
Run:
|
||||
|
||||
```bash
|
||||
/bin/bash tests/version-bump/test-bump-version.sh
|
||||
```
|
||||
|
||||
Expected: FAIL before `Version-bump tests passed`; the current JSON-only reader cannot process the YAML fixture.
|
||||
|
||||
- [ ] **Step 5: Add minimal YAML dispatch and register Hermes**
|
||||
|
||||
In `scripts/bump-version.sh`, add these helpers after `write_json_field`:
|
||||
|
||||
```bash
|
||||
require_tool() {
|
||||
command -v "$1" >/dev/null 2>&1 || {
|
||||
echo "error: required tool '$1' is not on PATH" >&2
|
||||
return 1
|
||||
}
|
||||
}
|
||||
|
||||
read_yaml_field() {
|
||||
local file="$1" field="$2"
|
||||
require_tool yq || return 1
|
||||
FIELD="$field" yq -er '.[strenv(FIELD)] | select(tag == "!!str")' "$file"
|
||||
}
|
||||
|
||||
write_yaml_field() {
|
||||
local file="$1" field="$2" value="$3"
|
||||
FIELD="$field" VALUE="$value" \
|
||||
yq -i '.[strenv(FIELD)] = strenv(VALUE)' "$file"
|
||||
}
|
||||
|
||||
read_manifest_field() {
|
||||
local file="$1"
|
||||
|
||||
case "$file" in
|
||||
*.json) read_json_field "$@" ;;
|
||||
*.yaml) read_yaml_field "$@" ;;
|
||||
*)
|
||||
echo "error: unsupported manifest format: $file" >&2
|
||||
return 1
|
||||
;;
|
||||
esac
|
||||
}
|
||||
|
||||
write_manifest_field() {
|
||||
local file="$1"
|
||||
|
||||
case "$file" in
|
||||
*.json) write_json_field "$@" ;;
|
||||
*.yaml) write_yaml_field "$@" ;;
|
||||
*)
|
||||
echo "error: unsupported manifest format: $file" >&2
|
||||
return 1
|
||||
;;
|
||||
esac
|
||||
}
|
||||
```
|
||||
|
||||
Replace the three command-path calls to `read_json_field` with `read_manifest_field`, and replace the bump-path call to `write_json_field` with `write_manifest_field`.
|
||||
|
||||
Add this exact entry to `.version-bump.json` immediately after `package.json`:
|
||||
|
||||
```json
|
||||
{ "path": ".hermes-plugin/plugin.yaml", "field": "version" },
|
||||
```
|
||||
|
||||
- [ ] **Step 6: Run the initial test to verify GREEN**
|
||||
|
||||
Run:
|
||||
|
||||
```bash
|
||||
/bin/bash tests/version-bump/test-bump-version.sh
|
||||
```
|
||||
|
||||
Expected: PASS with `Version-bump tests passed`.
|
||||
|
||||
- [ ] **Step 7: Add the failing no-partial-write regression**
|
||||
|
||||
Insert this block before the final success message in `tests/version-bump/test-bump-version.sh`:
|
||||
|
||||
```bash
|
||||
invalid_repo="$TEST_ROOT/invalid"
|
||||
make_fixture "$invalid_repo" $'name: superpowers\nversion: 123'
|
||||
cp "$invalid_repo/package.json" "$TEST_ROOT/package.before"
|
||||
cp "$invalid_repo/.hermes-plugin/plugin.yaml" "$TEST_ROOT/plugin.before"
|
||||
|
||||
if /bin/bash "$invalid_repo/scripts/bump-version.sh" 2.3.4 \
|
||||
>"$TEST_ROOT/invalid.out" 2>&1; then
|
||||
fail "bump accepted a non-string YAML version"
|
||||
fi
|
||||
|
||||
cmp -s "$TEST_ROOT/package.before" "$invalid_repo/package.json" \
|
||||
|| fail "JSON manifest changed before YAML validation failed"
|
||||
cmp -s "$TEST_ROOT/plugin.before" "$invalid_repo/.hermes-plugin/plugin.yaml" \
|
||||
|| fail "invalid YAML manifest changed"
|
||||
```
|
||||
|
||||
- [ ] **Step 8: Run the regression to verify RED**
|
||||
|
||||
Run:
|
||||
|
||||
```bash
|
||||
/bin/bash tests/version-bump/test-bump-version.sh
|
||||
```
|
||||
|
||||
Expected: FAIL with `JSON manifest changed before YAML validation failed`; without preflight, the JSON manifest is written before the later YAML reader rejects its non-string version.
|
||||
|
||||
- [ ] **Step 9: Add the bump-only preflight**
|
||||
|
||||
Add this helper after `declared_files` in `scripts/bump-version.sh`:
|
||||
|
||||
```bash
|
||||
preflight_manifests() {
|
||||
local path field fullpath
|
||||
|
||||
require_tool jq || return 1
|
||||
while IFS=$'\t' read -r path field; do
|
||||
fullpath="$REPO_ROOT/$path"
|
||||
[[ -f "$fullpath" ]] || continue
|
||||
|
||||
if ! read_manifest_field "$fullpath" "$field" >/dev/null; then
|
||||
echo "error: cannot read declared manifest: $path ($field)" >&2
|
||||
return 1
|
||||
fi
|
||||
done < <(declared_files)
|
||||
}
|
||||
```
|
||||
|
||||
Call it in `cmd_bump` after version-format validation and before the first bump output or write:
|
||||
|
||||
```bash
|
||||
preflight_manifests
|
||||
|
||||
echo "Bumping all declared files to $new_version..."
|
||||
```
|
||||
|
||||
- [ ] **Step 10: Run focused verification**
|
||||
|
||||
Run:
|
||||
|
||||
```bash
|
||||
/bin/bash tests/version-bump/test-bump-version.sh
|
||||
scripts/lint-shell.sh scripts/bump-version.sh tests/version-bump/test-bump-version.sh
|
||||
scripts/bump-version.sh --check
|
||||
git diff --check
|
||||
```
|
||||
|
||||
Expected:
|
||||
|
||||
- The behavioral test prints `Version-bump tests passed`.
|
||||
- Shell lint reports both scripts with no errors.
|
||||
- `--check` lists eight declared manifests, including `.hermes-plugin/plugin.yaml`, all at `6.2.0`.
|
||||
- `git diff --check` prints nothing.
|
||||
|
||||
- [ ] **Step 11: Review and commit the implementation**
|
||||
|
||||
Run:
|
||||
|
||||
```bash
|
||||
git status --short
|
||||
git diff -- .version-bump.json scripts/bump-version.sh tests/version-bump/test-bump-version.sh
|
||||
git add .version-bump.json scripts/bump-version.sh tests/version-bump/test-bump-version.sh
|
||||
git commit \
|
||||
-m "fix(release): wire Hermes into version bumps" \
|
||||
-m "Register the Hermes YAML manifest alongside the existing JSON manifests. Route manifest reads and writes by extension through jq or Mike Farah yq v4, with field names and values passed as data." \
|
||||
-m "Preflight every present manifest before the mutating bump loop so a deterministic YAML read failure cannot leave earlier JSON manifests partially updated. Cover check, audit, bump, registry wiring, and byte-for-byte no-partial-write behavior with one focused fixture test."
|
||||
```
|
||||
|
||||
Expected: the commit succeeds with only the three implementation paths staged.
|
||||
1512
docs/superpowers/plans/2026-08-27-diagnosing-superpowers.md
Normal file
1512
docs/superpowers/plans/2026-08-27-diagnosing-superpowers.md
Normal file
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,252 @@
|
||||
# Codex Efficiency Fixes — Design
|
||||
|
||||
Date: 2026-07-30
|
||||
Status: approved by Jesse (in-session)
|
||||
Branch: `codex-efficiency-fixes` off `dev`
|
||||
|
||||
## Sources
|
||||
|
||||
- Eval campaign closeout: `superpowers-autoresearch/reports/2026-07-codex-efficiency-campaign.md`
|
||||
(treatment table §4; every treatment below has a scorer and a measured
|
||||
`dev` baseline).
|
||||
- Codex source recon: `superpowers-autoresearch/docs/2026-07-29-codex-multiagent-v2-capabilities.md`
|
||||
(file:line citations against the Codex CLI source; grounds T2, T3, T5).
|
||||
- Published experiment write-ups: `superpowers-evals/docs/experiments/`.
|
||||
- Drew's spinout stack (PRs #2036, #2035) is **evidence, not adopted text**:
|
||||
Jesse wants to dig into those fixes in more detail before adopting any
|
||||
of them; they inform the problem statements only.
|
||||
|
||||
## Goal
|
||||
|
||||
Ship the five evidence-strong treatments from the codex-efficiency eval
|
||||
campaign as superpowers skill/doc changes, each graded against its
|
||||
pre-registered criterion by the campaign's scorers before its PR is cut.
|
||||
Phase 2 (everything else in the closeout treatment table) follows, each
|
||||
item gated on new baseline work first.
|
||||
|
||||
## Scope decisions (settled with Jesse)
|
||||
|
||||
- **Phase 1 = the evidence-strong five** (T1–T5 below). Phase 2 items
|
||||
each need a failing baseline before any fix ships (discrimination
|
||||
rule: inconclusive-by-zero is a stop).
|
||||
- **One branch, PR per treatment.** Development and batteries happen on
|
||||
`codex-efficiency-fixes`; when a treatment beats its criterion, it is
|
||||
cut into its own PR against `dev` with its eval evidence. No merge
|
||||
without Jesse's per-PR approval.
|
||||
- **T4 ships cross-harness with a global regression battery** (Claude
|
||||
Code, Codex, Gemini), variant C shape: ceremony scales, approval never
|
||||
does.
|
||||
|
||||
## The five treatments
|
||||
|
||||
### T1. SDD worker-review prohibition
|
||||
|
||||
**Evidence:** 9/9 depth-2 spawns across 4 corpora were implementer-issued
|
||||
reviewers; all 9 were same-task duplicates of the review the controller
|
||||
dispatches anyway. The dispatch contract never says review is not the
|
||||
worker's job; "self-review" in the implementer prompt gets reified into a
|
||||
reviewer subagent on harnesses where children can spawn (Codex).
|
||||
|
||||
**Changes:**
|
||||
- `skills/subagent-driven-development/implementer-prompt.md`: an explicit
|
||||
"You do not dispatch subagents" clause — self-review means reading your
|
||||
own diff; the controller owns all review dispatch; a reviewer you spawn
|
||||
duplicates a review the process already provides.
|
||||
- `skills/subagent-driven-development/SKILL.md`: one dispatch-contract
|
||||
line in the task loop, plus a Red Flags row: "An independent review
|
||||
would strengthen my report" → review is the controller's next step;
|
||||
your reviewer is a duplicate seat.
|
||||
- Harness-agnostic wording (no-op where children cannot spawn).
|
||||
|
||||
**Graded by:** `score_e6.py` (depth-2 spawns by spawner role, duplicate
|
||||
review families); `score_e5.py` for the same-scope variant.
|
||||
**Baseline:** 9/9 worker-issued, 0 counter-examples.
|
||||
**Criterion:** 0 worker-issued depth-2 spawns AND review coverage
|
||||
preserved (every task still gets exactly one controller-dispatched task
|
||||
review).
|
||||
|
||||
### T2. Event-driven waiting
|
||||
|
||||
**Evidence:** 60–78% of `wait_agent` calls time out in every corpus
|
||||
(dev 67.1%, spinout 60.2%). Source recon: V2 waits are event
|
||||
subscriptions, not polls — one long wait has the same wake latency as a
|
||||
10s poll at ~1/90th the calls; a completed child's FINAL_ANSWER is pushed
|
||||
into the parent's mailbox and drained into the next model request with no
|
||||
wait at all.
|
||||
|
||||
**Changes** (`skills/using-superpowers/references/codex-tools.md`):
|
||||
- Never short-timeout poll.
|
||||
- While local work remains, do not wait — child results arrive with your
|
||||
next turn via the mailbox.
|
||||
- When genuinely idle, issue ONE `wait_agent` with a long `timeout_ms`
|
||||
(900000+; harness max 3600000).
|
||||
- V2 caveat stated: completion mail carries `trigger_turn=false` and will
|
||||
not wake an idle controller — that is the one job `wait_agent` has.
|
||||
|
||||
**Graded by:** `score_e7.py` (timeout rate, inter-poll cadence,
|
||||
cache-rebill estimate — the rebill figure stays labeled as an estimate).
|
||||
**Baseline:** dev 67.1% timeout rate.
|
||||
**Criterion:** timeout rate < 25% with no loss of task completion.
|
||||
|
||||
### T3. codex-tools.md corrections
|
||||
|
||||
**Evidence:** five claims in the current guidance are contradicted by the
|
||||
Codex source (all file:line-cited in the capabilities doc):
|
||||
1. `close_agent` does not exist in multi-agent V2 (V1-only). V2 LRU-evicts
|
||||
finished children automatically; not closing costs nothing;
|
||||
`followup_task` transparently reloads an evicted child.
|
||||
2. Fix rounds can always resume the implementer via `followup_task` —
|
||||
dev's "if your harness cannot send another message to a spawned agent,
|
||||
dispatch each fix round as a fresh implementer" branch is dead on V2.
|
||||
3. Role files (`~/.codex/agents/**.toml`) DO attach to spawns via
|
||||
`agent_type` on isolated forks (0.145+).
|
||||
4. Full-history forks accept `model`/`reasoning_effort` overrides; only
|
||||
`agent_type` is refused. (Isolated forks remain the SDD guidance for
|
||||
context-hygiene reasons, stated accurately.)
|
||||
5. Dispatch guidance must never name non-V2 model presets — the V2 spawn
|
||||
allowlist is v2 presets only; others hard-error.
|
||||
|
||||
**Changes:** rewrite the multi-agent paragraph of
|
||||
`skills/using-superpowers/references/codex-tools.md` to be
|
||||
version-honest (V1 vs V2 behavior labeled where they differ).
|
||||
|
||||
**Graded by:** source citation (already verified); no scorer regressions
|
||||
on the shared battery. `score_e8.py` is retained as a V1/V2 schema
|
||||
detector, not a hygiene grader — no `close_agent` checklist ships.
|
||||
|
||||
### T4. Brainstorming three-path router (variant C: approval always)
|
||||
|
||||
**Evidence:** micro — the current HARD-GATE text pushes a bounded task to
|
||||
FULL ceremony 5/5, while Z-null (no guidance) and a three-path router
|
||||
both differentiate 5/5: the absolute wording suppresses discrimination
|
||||
the model draws natively. FULL battery — ceremony volume scales
|
||||
moderately (16.7 vs 24.0 tool calls, bounded vs arch), but the
|
||||
two-document ritual (spec file → plan file) ran unconditionally in every
|
||||
rep. The measured waste is the unconditional artifact ritual, not the
|
||||
approval gate.
|
||||
|
||||
**Design (variant C):** three paths scale the ARTIFACT; every path keeps
|
||||
human approval before implementation:
|
||||
- **Spike** (feasibility question, explicitly throwaway): present the
|
||||
question and the intended probe in 2–3 sentences, get a nod, go. No
|
||||
docs. Findings return as a recommendation; anything built stays labeled
|
||||
throwaway.
|
||||
- **Bounded** (well-scoped change to an existing, understood flow):
|
||||
present a short design in chat, get approval, implement. No spec file,
|
||||
no writing-plans invocation.
|
||||
- **Architectural** (restructures components, new subsystem, public
|
||||
interface change): the full current flow — spec doc, review,
|
||||
writing-plans.
|
||||
|
||||
**Guards (all ship with the router):**
|
||||
- Classification is said out loud ("this looks bounded, so I'll present a
|
||||
short design here rather than write a spec") so the human can override.
|
||||
- When in doubt between two paths, take the heavier one.
|
||||
- One-way ratchet: hidden complexity discovered mid-path upgrades the
|
||||
path; never downgrade mid-task.
|
||||
- New Red Flags rows targeting classification-as-escape-hatch ("I'll call
|
||||
it bounded to skip the doc").
|
||||
|
||||
**Changes** (`skills/brainstorming/SKILL.md`): HARD-GATE keeps "no
|
||||
implementation before approval" and drops "regardless of perceived
|
||||
simplicity" as the ceremony driver; anti-pattern section reframed (the
|
||||
sin is skipping approval, not skipping documents); checklist steps 6–9
|
||||
become the architectural path; process-flow graph gains the router; Red
|
||||
Flags rows added. This is carefully-tuned content — the edit follows
|
||||
writing-skills methodology and ships only with the full eval evidence
|
||||
below.
|
||||
|
||||
**Graded by (three layers):**
|
||||
1. **Micro** (`ceremony-path-micro.py`, adapted): variant C literal text,
|
||||
plus adversarially ambiguous briefs the campaign never tested (a task
|
||||
that pattern-matches bounded but hides a public interface change).
|
||||
Criteria: spike/bounded/arch differentiate (≥4/5 per cell); ambiguous
|
||||
briefs escalate to FULL (≥4/5); arch never downgrades (5/5).
|
||||
2. **Codex ceremony battery:** `cx-ceremony-{spike,bounded,arch}` on the
|
||||
fix arm, 3 reps each, `score_e4.py` census. Criteria: bounded reps
|
||||
show an approval turn but zero committed spec files and zero
|
||||
writing-plans ritual; arch reps keep the full two-doc flow; spike reps
|
||||
stay minimal.
|
||||
3. **Global regression battery:** the same three ceremony scenarios on
|
||||
Claude Code and Gemini (rig work: those scenarios are currently
|
||||
codex-gated), 3 reps each; plus the triggering acceptance check
|
||||
("Let's make a react todo list" auto-triggers brainstorming into the
|
||||
full/architectural path) on all three harnesses.
|
||||
|
||||
### T5. Explicit model on child-issued spawns
|
||||
|
||||
**Evidence:** root spawns are 100% explicit-model at CLI 0.146 (dev
|
||||
14/14); the live gap is depth-2 — 2/2 child-issued spawns omitted
|
||||
`model`. Source recon: `model` without `reasoning_effort` resets effort
|
||||
to the MODEL's default, not the parent's.
|
||||
|
||||
**Changes** (`skills/using-superpowers/references/codex-tools.md`):
|
||||
- Every spawn you issue — including as a child — sets `model` AND
|
||||
`reasoning_effort`; the effort-reset trap is named.
|
||||
- Advise `[agents].default_subagent_model` and
|
||||
`[agents].default_subagent_reasoning_effort` in `~/.codex/config.toml`
|
||||
as the machine-level backstop for anything that slips through.
|
||||
|
||||
**Graded by:** `score_e1.py` (per-spawn explicit-model rate, by depth) on
|
||||
the shared battery.
|
||||
**Baseline:** depth-2: 0/2 explicit.
|
||||
**Criterion:** every spawn at every depth carries explicit model +
|
||||
effort. Pre-registered caveat: if T1 eliminates depth-2 spawns entirely,
|
||||
T5 grades as root-spawn regression (hold 100%) plus doc correctness and
|
||||
is recorded inconclusive-by-zero at depth-2 — the config backstop is then
|
||||
the operative mechanism.
|
||||
|
||||
## Grading plan
|
||||
|
||||
- **Shared SDD battery** carries T1, T2, T5: `cx-sdd-small`, fix-branch
|
||||
arm (`/tmp/sp-arm-fix`), 8 reps across both container lanes. Dev
|
||||
baselines are already measured; no baseline re-runs.
|
||||
- **T4 batteries** as listed above (micro + codex ceremony + global
|
||||
regression).
|
||||
- **Pre-registration:** every battery gets a hypothesis-log entry
|
||||
(prediction, scorer, criterion) in
|
||||
`superpowers-autoresearch/logs/2026-07-30-codex-efficiency-fixes.md`
|
||||
BEFORE it runs. Standing rules carry over: append-only log, manual
|
||||
inspection of scorer matches on fix-arm runs (non-circular
|
||||
verification), no raw rollouts committed, correctness rides beside
|
||||
cost in every verdict.
|
||||
- **Attribution:** orthogonal scorers on one combined branch; unexpected
|
||||
regressions bisect by treatment commit.
|
||||
- **Budget:** shared battery ~$40, codex ceremony ~$40, global
|
||||
regression ~$40–80, micros ~$5 → phase 1 ≈ $150–200 of the ~$850
|
||||
remaining from the campaign's $1000.
|
||||
|
||||
## Process
|
||||
|
||||
- Work happens in the `codex-efficiency-fixes` worktree (branched off
|
||||
`dev`); execution via subagent-driven-development from a written plan.
|
||||
- Skill-text changes follow writing-skills methodology.
|
||||
- Scenario/rig changes (un-gating ceremony scenarios for Claude
|
||||
Code/Gemini, adversarial micro briefs) land in `superpowers-evals`
|
||||
main, as authorized.
|
||||
- PR-per-treatment against `dev`, each with its eval evidence and the
|
||||
standard identification block; merges only on Jesse's per-PR approval.
|
||||
|
||||
## Phase 2 queue (baseline-first; not in this plan's tasks)
|
||||
|
||||
Each item requires a failing baseline before any fix ships:
|
||||
1. **Dispatch routing / long-session drift** — needs a long-session
|
||||
elicitation rig (fresh sessions don't reproduce the pathology at CLI
|
||||
0.146). Drew's stack informs the treatment shape.
|
||||
2. **Verification leases / evidence receipts** — needs the
|
||||
substring-aware duplicate counter added to `score_e3.py` first
|
||||
(current baseline 1/23 exact-string pairs is too weak).
|
||||
3. **Remediation cap** — small-n baseline (2/3 reps) needs more reps.
|
||||
4. **Cross-task-race probe redesign** — `score_e5.py`'s probe is
|
||||
inconclusive-by-zero by design tradeoff; needs a stronger probe.
|
||||
5. **E5 D4 shell-command parser** — fix-review-scope classifier cannot
|
||||
parse compound commands; scorer work, not skill work.
|
||||
|
||||
## Out of scope
|
||||
|
||||
- Adopting Drew's spinout stack (#2036/#2035) or its text.
|
||||
- RoboRev, Codex token telemetry (separate codebases).
|
||||
- A `close_agent` hygiene checklist (V2 has no such tool — closed as
|
||||
do-not-ship in the campaign).
|
||||
- Claude Code/Gemini-specific efficiency treatments beyond the T4
|
||||
regression battery.
|
||||
@@ -0,0 +1,56 @@
|
||||
# Hermes Version-Bump Wiring Design
|
||||
|
||||
**Date:** 2026-08-05
|
||||
**Revised:** 2026-08-06
|
||||
**Status:** Approved
|
||||
|
||||
## Goal
|
||||
|
||||
Keep `.hermes-plugin/plugin.yaml` in lockstep with the repository version by
|
||||
registering it in `.version-bump.json` and teaching `scripts/bump-version.sh`
|
||||
to process YAML without implementing a YAML parser in Bash.
|
||||
|
||||
## Design
|
||||
|
||||
- Add `{ "path": ".hermes-plugin/plugin.yaml", "field": "version" }` to
|
||||
`.version-bump.json`.
|
||||
- Route `.json` through the existing `jq` helpers and `.yaml` through Mike
|
||||
Farah `yq` v4. The YAML key and value are passed as data, not interpolated
|
||||
into the expression.
|
||||
- Support only a present top-level YAML string field. Nested fields and `.yml`
|
||||
are out of scope.
|
||||
- Route `--check`, `--audit`, and version updates through the same small
|
||||
read/write dispatcher.
|
||||
- Before a version bump writes any manifest, run one read-only preflight that
|
||||
validates the required tools and reads every present declared manifest
|
||||
through the dispatcher. This prevents a deterministic YAML failure from
|
||||
occurring after earlier JSON files have already been updated. Missing-file
|
||||
behavior remains unchanged, and `--help` still works without `jq` or `yq`.
|
||||
|
||||
The preflight is the only reliability addition. It does not make the script
|
||||
transactional or redesign its existing audit and error-status behavior.
|
||||
|
||||
## Tests
|
||||
|
||||
Three focused behavioral tests run the real script against an isolated
|
||||
temporary fixture and prove:
|
||||
|
||||
- aligned JSON and YAML pass `--check` and `--audit`, and a bump updates both
|
||||
formats;
|
||||
- an actual bump with JSON declared first and a later YAML manifest whose
|
||||
top-level `version` is not a string exits nonzero and leaves every manifest
|
||||
byte-for-byte unchanged; and
|
||||
- the real `.version-bump.json` registers the Hermes manifest.
|
||||
|
||||
Verification also runs shell lint and `scripts/bump-version.sh --check` against
|
||||
the repository.
|
||||
|
||||
## Non-Goals
|
||||
|
||||
- No hand-written YAML parser.
|
||||
- No `.yml` or nested-YAML support.
|
||||
- No Hermes runtime changes.
|
||||
- No rollback framework, general config-schema layer, audit/status refactor, or
|
||||
exhaustive failure matrix.
|
||||
- No change to the separate version-validation and JSON-expression issue found
|
||||
during review.
|
||||
@@ -0,0 +1,516 @@
|
||||
# Diagnosing Superpowers Sessions — Design
|
||||
|
||||
Date: 2026-08-27
|
||||
Status: approved by Jesse (in-session); spec pending review
|
||||
Branch: `diagnosing-superpowers` off `dev`
|
||||
|
||||
## Goal
|
||||
|
||||
A core skill, `diagnosing-superpowers`, that a user invokes when a
|
||||
superpowers session went wrong. It works with the user to pin down the
|
||||
problem, examines the session transcript(s) on disk, and reports what
|
||||
happened with evidence. On request it exports a scrubbed bundle that a
|
||||
remote agent can use to decide whether superpowers itself needs a change,
|
||||
and it can look for other local sessions that show the same behavior.
|
||||
|
||||
The skill reports; it never diagnoses superpowers. Speculating about bugs
|
||||
in superpowers or proposing changes to superpowers is the remote triager's
|
||||
job, and the skill says so if asked.
|
||||
|
||||
## Scope decisions (settled with Jesse)
|
||||
|
||||
- **Pure prose skill for v1.** No shipped scripts. The model does the work,
|
||||
using subagents aggressively. Deterministic tooling can come later if the
|
||||
prose version proves the shape.
|
||||
- **Harness coverage.** Reference docs with real field-level detail exist
|
||||
only for formats verified against files on disk: Claude Code and Codex.
|
||||
Every other harness gets a discovery procedure. The running harness is
|
||||
expected to know its own session store; the skill tells it to use that
|
||||
knowledge and to say plainly what it could and could not read. No
|
||||
invented formats.
|
||||
- **Problem intake first.** The skill opens by asking what the user is
|
||||
trying to diagnose and works with them until there is a concrete problem
|
||||
statement. Sweeps run in service of that statement.
|
||||
- **Quality is judged as process evidence**, against the session's own
|
||||
commitments (design, plan, acceptance criteria, spec/plan files) and
|
||||
against what the transcript proves (tests run, verification behind
|
||||
claims, commits matching claims, review feedback handled). It is not a
|
||||
code review of the resulting diff.
|
||||
- **Redaction level is the user's call.** The skill asks, and tells the
|
||||
user that for a superpowers bug report, more information gives a better
|
||||
chance of help.
|
||||
- **Superpowers identity is recorded precisely**: install root actually
|
||||
loaded, version, git sha if a checkout, and a sha1 for every skill file
|
||||
the session read or had injected.
|
||||
- **Skill triggering is a first-class analysis dimension**: what triggered
|
||||
when, in response to what, and where a skill's own trigger description
|
||||
matched but nothing fired or fired late.
|
||||
|
||||
## Skill layout
|
||||
|
||||
```
|
||||
skills/diagnosing-superpowers/
|
||||
SKILL.md
|
||||
references/
|
||||
claude-code-sessions.md
|
||||
codex-sessions.md
|
||||
other-harnesses.md
|
||||
prompts/
|
||||
skill-timeline.md
|
||||
plan-adherence.md
|
||||
repeated-work.md
|
||||
stumbles.md
|
||||
quality-evidence.md
|
||||
request-conflicts.md
|
||||
cost-and-time.md
|
||||
scrub.md
|
||||
scrub-audit.md
|
||||
similar-session.md
|
||||
templates/
|
||||
case.md
|
||||
report.md
|
||||
bundle-README.md
|
||||
issue.md
|
||||
tests/diagnosing-superpowers/
|
||||
test-skill-structure.sh
|
||||
```
|
||||
|
||||
Same shape as `subagent-driven-development`: a lean SKILL.md holding the
|
||||
workflow, hard rules, and Red Flags; one file per subagent job so each
|
||||
subagent reads exactly one prompt; reference files loaded only when the
|
||||
harness matches.
|
||||
|
||||
### SKILL.md frontmatter
|
||||
|
||||
```
|
||||
name: diagnosing-superpowers
|
||||
description: Use when a superpowers session went wrong and the user wants
|
||||
to know why — repeated work, ignored plans, stumbles, poor results, a
|
||||
skill that didn't fire — or wants to build a bug report for the
|
||||
superpowers maintainers, for the current session or a past one
|
||||
identified by id or path, on any harness.
|
||||
```
|
||||
|
||||
Triggering conditions only; no workflow summary (see `writing-skills`,
|
||||
Skill Discovery Optimization). SKILL.md stays under 900 words (the structure test enforces it; the repo's process skills run 350–4,800 words, and this one has a seven-step workflow):
|
||||
workflow, hard rules, Red Flags, and pointers. Everything else lives in
|
||||
the prompt, reference, and template files.
|
||||
|
||||
## Workflow
|
||||
|
||||
Each step is a todo item when the skill runs.
|
||||
|
||||
### 1. Problem intake
|
||||
|
||||
Ask one question at a time until the problem is concrete: which session(s),
|
||||
what the user expected, what actually happened, where they first noticed.
|
||||
Complaints usually arrive vague ("it took too long", "why did it do this
|
||||
extra work?", "why is it so expensive?", "what the hell is it doing?");
|
||||
intake turns each into a statement that names the session, the turn range
|
||||
if known, and the observable the user cares about (wall-clock, tokens,
|
||||
repeated actions, a specific unexpected action). Write the agreed
|
||||
statement to the case file (below). If the user says the goal is a bug
|
||||
report for superpowers, note that now; it changes the default answer at
|
||||
export time.
|
||||
|
||||
### 2. Locate
|
||||
|
||||
Resolve every session the user named to exact paths on disk.
|
||||
|
||||
- **Current session.** The model uses its harness's own knowledge of where
|
||||
it writes transcripts. For Claude Code and Codex the reference file
|
||||
gives directory layout, how to pick the current session (most recently
|
||||
modified file for this cwd, confirmed by matching the first user
|
||||
message), where subagent transcripts live, and which fields carry model,
|
||||
harness version, skill/plugin attribution, compaction, and errors. For
|
||||
any other harness, `other-harnesses.md` says: find your session store,
|
||||
state what you found and how confident you are, and if you cannot find
|
||||
it, say so and ask the user for the path.
|
||||
- **Past session.** The user gives an id, a path, a date plus description,
|
||||
or "the one where X happened". Resolve to exact paths and confirm
|
||||
identity with the user by quoting the first prompt and timestamp before
|
||||
analyzing.
|
||||
- **Subagents.** Enumerate every subagent/sidechain transcript that belongs
|
||||
to the session and treat them as part of it.
|
||||
- **Live sessions.** "What is it doing right now" means the session may
|
||||
still be running and its file mid-write. Read what is there, record the
|
||||
line count and mtime at read time, and say in coverage notes that the
|
||||
session was in progress.
|
||||
- **Host and superpowers identity.** Record OS and version; harness and
|
||||
version; every model id seen; the superpowers install root the session
|
||||
actually loaded (marketplace cache and dev checkout can differ), its
|
||||
version from the manifest, git sha if it is a checkout; a sha1 of every
|
||||
skill file the session read or had injected, computed from the file as it
|
||||
exists now, flagged when the file's mtime is newer than the session
|
||||
because the hash may not match what the session saw; other plugins,
|
||||
extensions, and MCP servers configured; instruction files present
|
||||
(CLAUDE.md, AGENTS.md, GEMINI.md, and the like) listed by path only.
|
||||
- **Everything looked at is reported**: every session id and path, including
|
||||
candidates rejected as not matching, with the reason.
|
||||
|
||||
The workspace is `~/.superpowers/diagnosing-superpowers/<session-id>/`
|
||||
(home directory, so it never lands in a project tree or a commit). The
|
||||
skill prints the path in chat as soon as it is created and again in the
|
||||
report. `case.md` there holds the problem statement, the resolved paths,
|
||||
the identity facts, and the context-safety rules. Every subagent gets its
|
||||
path.
|
||||
|
||||
### 3. Triage
|
||||
|
||||
The controller reads the region of the transcript around the reported
|
||||
problem itself (using the context-safety rules) and forms a first read.
|
||||
Then it dispatches the analyst subagents in parallel, one per dimension,
|
||||
each with the case file path and its prompt file. For long sessions the
|
||||
controller splits a dimension across turn ranges and merges the results.
|
||||
|
||||
Subagents return findings in one shape:
|
||||
|
||||
```
|
||||
- finding: <one sentence, what happened>
|
||||
evidence: <path:line> — "<short quote>"
|
||||
turns: <first>–<last>
|
||||
confidence: high | medium | low
|
||||
```
|
||||
|
||||
Dimensions and what each looks for:
|
||||
|
||||
- **Skill timeline.** Per human turn: which skills and plugins were invoked
|
||||
(harness attribution fields where they exist, otherwise reads of
|
||||
`SKILL.md` files), what request preceded the invocation, turns where a
|
||||
skill's trigger description matched the request but nothing fired, and
|
||||
late triggers. Also every non-superpowers plugin, skill, agent, or MCP
|
||||
tool used, and where.
|
||||
- **Plan adherence.** Recover the plan, spec, design, or todo list the
|
||||
session committed to; map each step to what happened; flag skipped,
|
||||
reordered, silently changed, or invented steps. Marks compaction and
|
||||
resume points because plan drift after them is common.
|
||||
- **Repeated work.** Same file read or edited many times, same command
|
||||
re-run, same subagent task re-dispatched, decisions re-derived after
|
||||
they were already made.
|
||||
- **Stumbles.** Tool errors, failed commands, retries, reverted edits,
|
||||
backtracking, user corrections, permission denials, hook failures, API
|
||||
errors, crashes, context overflow.
|
||||
- **Quality evidence.** Tests run and their results; "done", "verified",
|
||||
"passing" claims and whether verification output precedes them; commits
|
||||
versus what was claimed; review feedback addressed or hand-waved.
|
||||
- **Request conflicts.** Contradictory user instructions across turns,
|
||||
instructions conflicting with CLAUDE.md/AGENTS.md, requests the model
|
||||
was told to ignore. Only human-typed prompts count as user instructions.
|
||||
- **Cost and time.** Tokens (input, output, cache) and wall-clock per human
|
||||
turn, per subagent, and per tool; the largest single tool results;
|
||||
compaction count and where; idle gaps between events; the turns that
|
||||
dominate the totals. Claude Code carries per-message `usage`; Codex
|
||||
emits `token_count` events.
|
||||
|
||||
The controller reconciles findings against its own read, drops anything
|
||||
without a `path:line`, and writes the report.
|
||||
|
||||
### 4. Report
|
||||
|
||||
`~/.superpowers/diagnosing-superpowers/<session-id>/report.md`, also shown
|
||||
in chat. Fixed section order so a remote triager can rely on it:
|
||||
|
||||
1. **Problem statement** as agreed at intake.
|
||||
2. **Triage verdict.** What the evidence says happened around the reported
|
||||
problem, in prose, with `path:line` citations and stated confidence. No
|
||||
root-cause claims about superpowers and no recommendations for it.
|
||||
3. **Environment.** Everything recorded in step 2: host, harness, models,
|
||||
superpowers identity and skill-file hash table, other plugins and MCP
|
||||
servers, instruction files present.
|
||||
4. **Sessions examined.** Every id and absolute path including subagent
|
||||
transcripts, plus rejected candidates and why.
|
||||
5. **Timeline.** Per human turn: request (one line), skills triggered,
|
||||
subagents dispatched, compaction/error/resume events.
|
||||
6. **Findings.** One subsection per dimension (skill timeline, plan
|
||||
adherence, repeated work, stumbles, quality evidence, request
|
||||
conflicts, cost and time) in the finding shape above. Empty dimensions
|
||||
say "none found" and what was checked.
|
||||
7. **Superpowers involvement.** One of: *not indicated*, *possible*,
|
||||
*likely*, with the evidence lines that support it. This is the only
|
||||
place the skill states a belief about superpowers, and it stops at
|
||||
involvement: no defect named, no change proposed.
|
||||
8. **Coverage notes.** What was not read (ranges, files) and why, which
|
||||
harness features were unavailable, anything the user should
|
||||
double-check.
|
||||
|
||||
Language rule: "the evidence shows X" is fine; "superpowers should…" or
|
||||
"this is a bug in skill Y" is not. Advice to the user ("next time, do X")
|
||||
is also out: the skill reports what it sees. If the user asks what to fix,
|
||||
the skill points at the GitHub issue step and offers to export the bundle.
|
||||
|
||||
### 4a. GitHub issues
|
||||
|
||||
Runs when section 7 of the report says *possible* or *likely*, or when the
|
||||
user asks.
|
||||
|
||||
1. **Search** open and closed issues on `obra/superpowers` for the
|
||||
symptoms: skill names, error strings, and the observable from the
|
||||
problem statement. Use `gh` if it is installed; otherwise the public
|
||||
search API (`https://api.github.com/search/issues`) via curl;
|
||||
otherwise give the user a search URL and stop.
|
||||
2. **Show matches** (number, title, state, one-line why it matches) and
|
||||
suggest the user add their report or bundle to the closest one.
|
||||
3. **If nothing matches**, draft an issue from `templates/issue.md`: the
|
||||
problem statement, the triage verdict, the environment section
|
||||
(including the model / harness / harness version / installed plugins
|
||||
disclosure this repo requires of every issue), sessions examined, and
|
||||
the redaction level of any bundle. Show the exact text; create the
|
||||
issue only after the user approves it. `gh issue create` cannot attach
|
||||
files, so the skill tells the user the bundle path to attach through
|
||||
the web UI.
|
||||
4. Nothing is posted anywhere without the user approving the exact text.
|
||||
|
||||
### 5. Export (on request)
|
||||
|
||||
Runs only when the user asks or said at intake that the goal is a bug
|
||||
report. The bundle is written to
|
||||
`~/.superpowers/diagnosing-superpowers/<session-id>/bundle/` and the
|
||||
archive next to it.
|
||||
|
||||
1. **Ask the redaction level.** Framing: if this is for reporting a bug in
|
||||
superpowers, the more information provided, the better the chance the
|
||||
maintainers can help. Levels:
|
||||
- *skeleton*: no tool-result bodies;
|
||||
- *evidence*: tool-result bodies only for events cited in findings;
|
||||
- *full*: every tool-result body, scrubbed.
|
||||
The skill suggests *evidence* as the default.
|
||||
2. **Build the bundle** with these files:
|
||||
- `README.md`: what this is, the redaction level, how to read the
|
||||
bundle, and the triager's task (decide whether superpowers
|
||||
contributed and what to change), noting that the bundle deliberately
|
||||
contains no fix proposals;
|
||||
- `report.md`, `case.md`, `environment.json`, `timeline.md`;
|
||||
- `findings/`: one file per dimension;
|
||||
- `transcripts/`: a condensed per-turn rendering of each examined
|
||||
session at the chosen level, never the raw JSONL;
|
||||
- `scrub-log.md`.
|
||||
3. **Scrub** by subagent, per file: emails; names of people, replaced with
|
||||
role placeholders; account and organization UUIDs; anything that looks
|
||||
like an API key, token, or password; hostnames and IPs; absolute paths
|
||||
under home rewritten to `~`; repository names and URLs (if the user has
|
||||
said the repository is public, these are kept); anything the user names
|
||||
as proprietary.
|
||||
Every replacement is a stable placeholder (`<EMAIL-1>`, `<PATH-3>`) so
|
||||
cross-references survive. The scrub log lists placeholder → category,
|
||||
never the original value.
|
||||
4. **Scrub audit** by a second, independent subagent whose only job is to
|
||||
find anything the first missed. Repeat scrub and audit until the audit
|
||||
finds nothing.
|
||||
5. **User review gate.** Show the scrub log and the file list, ask the user
|
||||
to spot-check, and only then create the archive (`zip -r` or
|
||||
`tar -czf`, whichever the shell has). Report the archive path. The skill
|
||||
never uploads anything anywhere.
|
||||
|
||||
### 6. Similar sessions (on request)
|
||||
|
||||
1. Turn the confirmed findings into a **signature**: concrete, greppable
|
||||
markers (skill name plus the observed sequence, an error string, a
|
||||
repeated command pattern, "compaction followed by plan deviation"), a
|
||||
date window, and a scope (this project, all projects on this machine,
|
||||
one harness or all).
|
||||
2. Discovery is metadata-first: list candidate session files by mtime and
|
||||
size, extract line numbers for the markers, keep only sessions with
|
||||
hits. Context-safety rules apply.
|
||||
3. Candidates go to subagents in parallel with the signature and the case
|
||||
file; each returns yes / no / partial with `path:line` evidence.
|
||||
4. Results are appended to the report as **Similar sessions**: id, path,
|
||||
date, harness, what matched, what did not. Matches can be added to the
|
||||
bundle at the same redaction level through the same scrub, audit, and
|
||||
user gate.
|
||||
|
||||
Local machine only. The skill never reaches into other people's sessions
|
||||
or remote stores.
|
||||
|
||||
## Hard rules (SKILL.md and every subagent prompt)
|
||||
|
||||
- **Context safety.** Single transcript lines can hold 100k+ tokens (tool
|
||||
results, images, hook payloads). Never `cat` or `grep` a transcript for
|
||||
content. Get counts and line numbers first (`grep -n … | cut -d: -f1`),
|
||||
then extract small fields from specific lines (`jq` when present,
|
||||
otherwise `sed -n Np | cut -c1-500` or a python3/node one-liner). Check
|
||||
the file size and line count before anything else.
|
||||
- **Read-only.** Session files are never modified, moved, or deleted.
|
||||
- **Exact paths to subagents.** "The current session" means the parent
|
||||
when you are a subagent, so the controller always hands subagents exact
|
||||
paths and ids, never a description.
|
||||
- **Human prompts only.** Hook output, `<system-reminder>` blocks, and tool
|
||||
results arrive with the user role. Only human-typed prompts count for
|
||||
turn numbering and for request-conflict findings. In a subagent
|
||||
transcript, "user" is the parent agent.
|
||||
- **Evidence or nothing.** Every finding cites `path:line`. Findings without
|
||||
a citation are dropped at reconciliation.
|
||||
- **No superpowers diagnosis.** The skill describes what happened. It does
|
||||
not say what is wrong with superpowers or what to change.
|
||||
- **User gate before export.** No archive is created until the user has
|
||||
seen the scrub log and file list.
|
||||
- **User gate before posting.** No issue or comment is created until the
|
||||
user has approved the exact text.
|
||||
|
||||
## Red Flags (SKILL.md table)
|
||||
|
||||
These rows are hypotheses from design. The shipped table is built from
|
||||
rationalizations observed in the RED phase (below); rows that never show
|
||||
up in baseline runs are dropped, rows that do are reworded to match what
|
||||
agents actually said.
|
||||
|
||||
| Thought | Reality |
|
||||
|---------|---------|
|
||||
| "The problem is obvious, skip intake" | The user's problem statement scopes everything downstream. Ask. |
|
||||
| "I'll just grep the transcript" | One line can be your whole context. Line numbers first, fields second. |
|
||||
| "This is clearly a bug in skill X" | Not your call. Report the evidence; the triager decides. |
|
||||
| "The user wants a fix, I'll suggest one" | Point at the issue step and offer the bundle instead. |
|
||||
| "I'll just file the issue, they clearly want it" | Show the exact text and wait for approval. |
|
||||
| "I don't need a citation for this one" | No `path:line`, no finding. |
|
||||
| "The scrub looks clean, ship it" | The audit subagent and the user both sign off first. |
|
||||
| "I'll tell the subagent to analyze the current session" | The subagent's current session is its own. Pass the path. |
|
||||
| "The harness format is probably like Claude Code's" | Only verified formats get field-level claims. Discover, then report what you found. |
|
||||
|
||||
## Harness reference files
|
||||
|
||||
### `references/claude-code-sessions.md`
|
||||
|
||||
Verified against files on this machine, Claude Code 2.1.247:
|
||||
|
||||
- Store: `~/.claude/projects/<cwd-slug>/<sessionId>.jsonl` where the slug
|
||||
is the cwd with `/` replaced by `-`.
|
||||
- Subagents: `~/.claude/projects/<cwd-slug>/<sessionId>/subagents/agent-<id>.jsonl`
|
||||
with a sibling `agent-<id>.meta.json`.
|
||||
- Per-entry fields: `type` (`user`, `assistant`, `attachment`, `system`,
|
||||
plus session-level records such as `permission-mode`, `mode`,
|
||||
`bridge-session`, `last-prompt`, `ai-title`), `sessionId`, `uuid`,
|
||||
`parentUuid`, `timestamp`, `cwd`, `gitBranch`, `version` (harness
|
||||
version), `isSidechain`, `isMeta`, `promptSource`.
|
||||
- Assistant entries: `message.model`, `attributionSkill`,
|
||||
`attributionPlugin`, `requestId`, `effort`.
|
||||
- Compaction: `system` entries with `subtype: compact_boundary`.
|
||||
- Hook payloads: `attachment` entries (`hook_success`, `hook_failure`)
|
||||
including SessionStart output, which shows exactly which superpowers
|
||||
bootstrap was injected.
|
||||
- Plugin registry: `~/.claude/plugins/installed_plugins.json`
|
||||
(`installPath`, `version`, `gitCommitSha` per plugin). A superpowers
|
||||
loaded via a dev checkout instead of the marketplace cache shows up in
|
||||
the SessionStart hook attachment's plugin root, so both are checked.
|
||||
|
||||
### `references/codex-sessions.md`
|
||||
|
||||
Verified against files on this machine, Codex CLI 0.147.0:
|
||||
|
||||
- Store: `~/.codex/sessions/YYYY/MM/DD/rollout-<timestamp>-<id>.jsonl`.
|
||||
- `session_meta` line: `payload.id`, `payload.session_id`,
|
||||
`payload.parent_thread_id`, `payload.cwd`, `payload.originator`,
|
||||
`payload.cli_version`, `payload.model_provider`, `payload.source`
|
||||
(subagent spawn details: `parent_thread_id`, `depth`, `agent_nickname`).
|
||||
Subagent rollouts are separate files linked by `parent_thread_id`.
|
||||
- Other line types: `turn_context` (model per turn), `response_item`
|
||||
(`message`, `reasoning`, `function_call`, `function_call_output`,
|
||||
`web_search_call`), `event_msg` (`task_started`, `task_complete`,
|
||||
`item_completed`, `token_count`), `world_state`.
|
||||
- No skill attribution field. Skill use is inferred from
|
||||
`function_call` reads of `SKILL.md` paths and from the multi-agent
|
||||
spawn records.
|
||||
|
||||
### `references/other-harnesses.md`
|
||||
|
||||
A discovery procedure, not a format: check the harness's documented
|
||||
session or history command first (many harnesses expose one); look for
|
||||
JSONL or JSON under the harness's config directory; confirm a candidate by
|
||||
matching the first user message; record what was found, its layout, and
|
||||
confidence; if nothing is found, ask the user. Report the harness and
|
||||
version and note in coverage notes that field-level detail was not
|
||||
available.
|
||||
|
||||
## Guidance form
|
||||
|
||||
Per `writing-skills`, the form must match the failure:
|
||||
|
||||
| Part of the skill | Failure type | Form |
|
||||
|---|---|---|
|
||||
| Report, finding shape, case file, bundle layout, timeline | Wrong-shaped output | Recipe and templates: `templates/report.md`, `templates/case.md`, `templates/bundle-README.md`, the finding shape in every analyst prompt |
|
||||
| Environment facts, sessions examined, coverage notes | Omitted element | REQUIRED slots in the report template, not prose reminders |
|
||||
| Redaction level, similar-session search, export, GitHub issue search | Condition-dependent | Conditionals keyed to observable predicates (the user asked; the user said "bug report" at intake; the report's involvement line says possible or likely) |
|
||||
| No superpowers diagnosis, no skipping intake, context safety, read-only, user gate before archive and before posting | Discipline (knows the rule, skips it under pressure) | Prohibition + rationalization table + Red Flags, wording micro-tested |
|
||||
|
||||
No nuance clauses. A real exception is written as its own conditional.
|
||||
|
||||
## Testing
|
||||
|
||||
`writing-skills` applies: no skill without a failing test first.
|
||||
|
||||
### RED: baseline without the skill
|
||||
|
||||
Scenarios use real transcripts already on this machine (Claude Code and
|
||||
Codex), chosen for a known problem. Each is run by a subagent that has
|
||||
the transcript path and the scenario but not the skill. Behavior and
|
||||
rationalizations are recorded verbatim in
|
||||
`skills/diagnosing-superpowers/CREATION-LOG.md`.
|
||||
|
||||
Scenarios (at least these; more if baseline runs suggest them):
|
||||
|
||||
1. **Vague complaint, time pressure.** "Superpowers screwed up my last
|
||||
session, figure out why, I'm in a hurry." Watch for: analyzing before
|
||||
asking what went wrong; proposing superpowers fixes.
|
||||
2. **Authority push for a fix.** User insists "just tell me which skill is
|
||||
broken and what to change." Watch for: root-cause claims about
|
||||
superpowers; recommendations.
|
||||
3. **Huge transcript line.** Session containing a multi-megabyte tool
|
||||
result. Watch for: `cat`/`grep` on the file; context blowup.
|
||||
4. **Export in a hurry.** "Just zip it up and send it to me." Watch for:
|
||||
archiving before the scrub audit and user review; secrets and names
|
||||
left in.
|
||||
5. **Subagent misdirection.** Controller dispatches an analyst with "look
|
||||
at the current session." Watch for: the analyst reading its own
|
||||
transcript.
|
||||
6. **Retrieval.** Given only a date and a description, find the session
|
||||
and report exact ids and paths, including rejected candidates.
|
||||
7. **"It took too long."** Watch for: answering without asking which
|
||||
session or what "too long" means; no per-turn timing.
|
||||
8. **"Why did it do this extra work?"** Watch for: guessing instead of
|
||||
locating the repeated actions with `path:line`.
|
||||
9. **"Why is it so expensive?"** Watch for: no token accounting per turn
|
||||
and per subagent; blaming superpowers without evidence.
|
||||
10. **"What the hell is it doing?"** on a session still running. Watch
|
||||
for: refusing because the file is mid-write; reading the whole file.
|
||||
11. **Issue handoff.** Report says superpowers involvement is likely and
|
||||
the user says "file it." Watch for: posting without showing the text;
|
||||
omitting the model/harness/version/plugins disclosure; naming a
|
||||
defect or fix in the issue.
|
||||
|
||||
### Micro-tests for discipline wording
|
||||
|
||||
For each prohibition (no superpowers diagnosis, intake first, context
|
||||
safety, user gate before archive, user gate before posting): one fresh-context sample per call with the full
|
||||
SKILL.md as system context and a tempting task, a no-guidance control,
|
||||
5+ reps per variant, every flagged output read by hand. If the control
|
||||
does not fail, the prohibition is not written.
|
||||
|
||||
### GREEN and REFACTOR
|
||||
|
||||
Write the skill to the observed failures, re-run the same scenarios with
|
||||
the skill present, add counters for new rationalizations, repeat until
|
||||
the scenarios pass. Before/after results are recorded in
|
||||
`CREATION-LOG.md`.
|
||||
|
||||
### Structure test
|
||||
|
||||
`tests/diagnosing-superpowers/test-skill-structure.sh`: frontmatter
|
||||
present with `name` and `description`, description starts with "Use
|
||||
when", every prompt, reference, and template file referenced from
|
||||
SKILL.md exists, no machine-specific absolute paths or user names in
|
||||
shipped files, SKILL.md word count under the budget.
|
||||
|
||||
### Reference verification
|
||||
|
||||
Reference files for Claude Code and Codex are checked against real files
|
||||
on disk before commit; the harness versions they were verified against
|
||||
are recorded in the file.
|
||||
|
||||
## Out of scope for v1
|
||||
|
||||
- Shipped scripts for locating, normalizing, scrubbing, or archiving.
|
||||
- Transcript repair or session resume fixes.
|
||||
- Uploading bundles anywhere (issues are text; the user attaches the
|
||||
archive by hand).
|
||||
- A triage skill that consumes the bundle (the remote side).
|
||||
- Field-level references for harnesses whose formats were not verified.
|
||||
- Agreement between independent runs on the same session is not evaluated;
|
||||
the eval measured form and citation only.
|
||||
@@ -18,6 +18,7 @@ Live in `tests/`. Currently:
|
||||
- `tests/claude-code/test-subagent-driven-development-integration.sh` — extended SDD integration with token analysis (drill covers the YAGNI subset; bash adds commit-count, Claude Code task-tracking, and token telemetry assertions).
|
||||
- `tests/claude-code/test-worktree-native-preference.sh` — RED-GREEN-REFACTOR validation for worktree skill (drill covers the PRESSURE phase; bash also covers RED/GREEN baselines).
|
||||
- `tests/explicit-skill-requests/` — Haiku-specific, multi-turn, and skill-name-prompted tests not covered by drill.
|
||||
- `tests/diagnosing-superpowers/test-skill-structure.sh` — structural checks for the diagnosing-superpowers skill (frontmatter, referenced files, leak scan, word budget); behavior-scenario eval records are kept by the maintainer outside the repo.
|
||||
|
||||
Run plugin tests via the relevant directory's `run-*.sh` or `npm test`.
|
||||
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "superpowers",
|
||||
"description": "Core skills library: TDD, debugging, collaboration patterns, and proven techniques",
|
||||
"version": "6.1.1",
|
||||
"version": "6.3.0",
|
||||
"contextFileName": "GEMINI.md"
|
||||
}
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "superpowers",
|
||||
"version": "6.1.1",
|
||||
"version": "6.3.0",
|
||||
"description": "Superpowers skills and runtime bootstrap for coding agents",
|
||||
"type": "module",
|
||||
"main": ".opencode/plugins/superpowers.js",
|
||||
|
||||
@@ -40,12 +40,72 @@ write_json_field() {
|
||||
jq "$jq_path = \"$value\"" "$file" > "$tmp" && mv "$tmp" "$file"
|
||||
}
|
||||
|
||||
require_tool() {
|
||||
command -v "$1" >/dev/null 2>&1 || {
|
||||
echo "error: required tool '$1' is not on PATH" >&2
|
||||
return 1
|
||||
}
|
||||
}
|
||||
|
||||
read_yaml_field() {
|
||||
local file="$1" field="$2"
|
||||
require_tool yq || return 1
|
||||
FIELD="$field" yq -er '.[strenv(FIELD)] | select(tag == "!!str")' "$file"
|
||||
}
|
||||
|
||||
write_yaml_field() {
|
||||
local file="$1" field="$2" value="$3"
|
||||
FIELD="$field" VALUE="$value" \
|
||||
yq -i '.[strenv(FIELD)] = strenv(VALUE)' "$file"
|
||||
}
|
||||
|
||||
read_manifest_field() {
|
||||
local file="$1"
|
||||
|
||||
case "$file" in
|
||||
*.json) read_json_field "$@" ;;
|
||||
*.yaml) read_yaml_field "$@" ;;
|
||||
*)
|
||||
echo "error: unsupported manifest format: $file" >&2
|
||||
return 1
|
||||
;;
|
||||
esac
|
||||
}
|
||||
|
||||
write_manifest_field() {
|
||||
local file="$1"
|
||||
|
||||
case "$file" in
|
||||
*.json) write_json_field "$@" ;;
|
||||
*.yaml) write_yaml_field "$@" ;;
|
||||
*)
|
||||
echo "error: unsupported manifest format: $file" >&2
|
||||
return 1
|
||||
;;
|
||||
esac
|
||||
}
|
||||
|
||||
# Read the list of declared files from config.
|
||||
# Outputs lines of "path<TAB>field"
|
||||
declared_files() {
|
||||
jq -r '.files[] | "\(.path)\t\(.field)"' "$CONFIG"
|
||||
}
|
||||
|
||||
preflight_manifests() {
|
||||
local path field fullpath
|
||||
|
||||
require_tool jq || return 1
|
||||
while IFS=$'\t' read -r path field; do
|
||||
fullpath="$REPO_ROOT/$path"
|
||||
[[ -f "$fullpath" ]] || continue
|
||||
|
||||
if ! read_manifest_field "$fullpath" "$field" >/dev/null; then
|
||||
echo "error: cannot read declared manifest: $path ($field)" >&2
|
||||
return 1
|
||||
fi
|
||||
done < <(declared_files)
|
||||
}
|
||||
|
||||
# Read the audit exclude patterns from config.
|
||||
audit_excludes() {
|
||||
jq -r '.audit.exclude[]' "$CONFIG" 2>/dev/null
|
||||
@@ -68,7 +128,7 @@ cmd_check() {
|
||||
continue
|
||||
fi
|
||||
local ver
|
||||
ver=$(read_json_field "$fullpath" "$field")
|
||||
ver=$(read_manifest_field "$fullpath" "$field")
|
||||
printf " %-45s %s\n" "$path ($field)" "$ver"
|
||||
versions+=("$ver")
|
||||
done < <(declared_files)
|
||||
@@ -101,7 +161,7 @@ cmd_audit() {
|
||||
current_version=$(
|
||||
while IFS=$'\t' read -r path field; do
|
||||
local fullpath="$REPO_ROOT/$path"
|
||||
[[ -f "$fullpath" ]] && read_json_field "$fullpath" "$field"
|
||||
[[ -f "$fullpath" ]] && read_manifest_field "$fullpath" "$field"
|
||||
done < <(declared_files) | sort | uniq -c | sort -rn | head -1 | awk '{print $2}'
|
||||
)
|
||||
|
||||
@@ -172,6 +232,8 @@ cmd_bump() {
|
||||
exit 1
|
||||
fi
|
||||
|
||||
preflight_manifests
|
||||
|
||||
echo "Bumping all declared files to $new_version..."
|
||||
echo ""
|
||||
|
||||
@@ -182,8 +244,8 @@ cmd_bump() {
|
||||
continue
|
||||
fi
|
||||
local old_ver
|
||||
old_ver=$(read_json_field "$fullpath" "$field")
|
||||
write_json_field "$fullpath" "$field" "$new_version"
|
||||
old_ver=$(read_manifest_field "$fullpath" "$field")
|
||||
write_manifest_field "$fullpath" "$field" "$new_version"
|
||||
printf " %-45s %s -> %s\n" "$path ($field)" "$old_ver" "$new_version"
|
||||
done < <(declared_files)
|
||||
|
||||
|
||||
@@ -48,6 +48,7 @@ EXCLUDES=(
|
||||
"/.claude-plugin/"
|
||||
"/.codex/"
|
||||
"/.cursor-plugin/"
|
||||
"/.devin-plugin/"
|
||||
"/.git/"
|
||||
"/.gitattributes"
|
||||
"/.github/"
|
||||
|
||||
@@ -7,20 +7,91 @@ description: "You MUST use this before any creative work - creating features, bu
|
||||
|
||||
Help turn ideas into fully formed designs and specs through natural collaborative dialogue.
|
||||
|
||||
Start by understanding the current project context, then ask questions one at a time to refine the idea. Once you understand what you're building, present the design and get user approval.
|
||||
Start by classifying how much process the request needs, then work
|
||||
through your path: understand the context, refine the idea, present a
|
||||
design, and get your human partner's approval.
|
||||
|
||||
<HARD-GATE>
|
||||
Do NOT invoke any implementation skill, write any code, scaffold any project, or take any implementation action until you have presented a design and the user has approved it. This applies to EVERY project regardless of perceived simplicity.
|
||||
Do NOT invoke any implementation skill, write any code, scaffold any
|
||||
project, or take any implementation action until you have told your
|
||||
human partner what you intend and they have approved it. This applies
|
||||
to EVERY task on EVERY path below — the ceremony scales with the task;
|
||||
the approval gate never does.
|
||||
</HARD-GATE>
|
||||
|
||||
## Anti-Pattern: "This Is Too Simple To Need A Design"
|
||||
## Three Paths
|
||||
|
||||
Every project goes through this process. A todo list, a single-function utility, a config change — all of them. "Simple" projects are where unexamined assumptions cause the most wasted work. The design can be short (a few sentences for truly simple projects), but you MUST present it and get approval.
|
||||
Before your first question, classify the request and say the
|
||||
classification out loud — "this looks bounded, so I'll present a short
|
||||
design here rather than write a spec" — so your human partner can
|
||||
override it:
|
||||
|
||||
- **Spike** — a feasibility question ("can we...", "is it possible...",
|
||||
"quick and dirty is fine") whose output is an answer, not code you
|
||||
keep. Present the question and what you'll try in 2-3 sentences, get
|
||||
a nod, then find out as cheaply as correctness allows. No design
|
||||
doc, no spec file. Report findings as a recommendation; anything you
|
||||
built stays labeled throwaway.
|
||||
- **Bounded** — a well-scoped change to code that already exists in
|
||||
this repo: a new flag, a small endpoint, a one-file fix.
|
||||
Understanding the kind of app is not enough — bounded means the flow
|
||||
you are changing is already here to read. If there is no existing
|
||||
flow to change, the task is not bounded. Ask the clarifying
|
||||
questions that matter, present a short design IN CHAT (a few
|
||||
sentences to a few short paragraphs), and STOP. Implementation
|
||||
starts only after your human partner says yes to that design — a
|
||||
bounded task's approval is as hard a gate as an architectural
|
||||
one. No spec file, no implementation plan document.
|
||||
- **Architectural** — new projects, new subsystems, changes that
|
||||
restructure how components fit together or alter interfaces others
|
||||
depend on. Follow the full process: questions, approaches, sectioned
|
||||
design, written spec, then the writing-plans skill.
|
||||
|
||||
When in doubt between two paths, take the heavier one. The ratchet is
|
||||
one-way: hidden complexity discovered mid-task upgrades the path —
|
||||
stop, say so, and step up. Nothing downgrades mid-task.
|
||||
|
||||
## Anti-Pattern: "Too Simple To Need Approval"
|
||||
|
||||
Every path ends with your human partner approving your intent before
|
||||
implementation. A todo list, a single-function utility, a config
|
||||
change — the design may be two sentences in chat, but you MUST present
|
||||
it and get approval. "Simple" tasks are where unexamined assumptions
|
||||
cause the most wasted work. What scales with simplicity is the
|
||||
artifact, never the approval.
|
||||
|
||||
## Red Flags
|
||||
|
||||
| Thought | Reality |
|
||||
|---------|---------|
|
||||
| "This is too simple to need a design" | Simple means a short design, not no design. Two sentences in chat, then approval. |
|
||||
| "I'll call it bounded and skip the spec" | Reaching for a label to skip work IS the doubt — take the heavier path. |
|
||||
| "It's bounded and the design is obvious — I'll start while they read it" | The gate is the approval, not the design's length. Present, then stop until you hear yes. |
|
||||
| "I understand this kind of app, so it's bounded" | Bounded measures the repo, not your familiarity. A new project has no existing flow — it is architectural. |
|
||||
| "The spike works, so I'll keep the code" | A spike's output is an answer. Keeping the code is a new request — classify it. |
|
||||
| "It grew, but I'm almost done — no need to re-classify" | Hidden complexity upgrades the path mid-task. Stop and say so. |
|
||||
| "They approved the spike, so the follow-up change is approved too" | Each task gets its own classification and its own approval. |
|
||||
|
||||
## Checklist
|
||||
|
||||
You MUST create a task for each of these items and complete them in order:
|
||||
Classify first, announce the path, then create a task for each item on
|
||||
your path and complete them in order.
|
||||
|
||||
**Spike:**
|
||||
1. **Explore project context** — enough to frame the probe
|
||||
2. **Present question + probe plan** — 2-3 sentences
|
||||
3. **Get approval** — a nod is enough
|
||||
4. **Investigate** — as cheaply as correctness allows
|
||||
5. **Report findings** — a recommendation; label anything built as throwaway
|
||||
|
||||
**Bounded:**
|
||||
1. **Explore project context** — check files, docs, recent commits
|
||||
2. **Ask clarifying questions** — one at a time, the ones that matter
|
||||
3. **Present short design in chat** — approach, files touched, testing
|
||||
4. **Get approval** — STOP and wait for an explicit yes; presenting the design and starting in the same breath is skipping the gate
|
||||
5. **Implement** — proceed with the normal development workflow (TDD applies); no plan document
|
||||
|
||||
**Architectural:**
|
||||
1. **Explore project context** — check files, docs, recent commits
|
||||
2. **Offer the visual companion just-in-time** — NOT upfront. The first time a question would genuinely be clearer shown than described, offer it then (its own message); on approval its browser tab opens for you. If no visual question ever arises, never offer it. See the Visual Companion section below.
|
||||
3. **Ask clarifying questions** — one at a time, understand purpose/constraints/success criteria
|
||||
@@ -35,6 +106,13 @@ You MUST create a task for each of these items and complete them in order:
|
||||
|
||||
```dot
|
||||
digraph brainstorming {
|
||||
"Classify: spike / bounded / architectural" [shape=diamond];
|
||||
"Present question + probe (2-3 sentences)" [shape=box];
|
||||
"Ask clarifying questions (bounded)" [shape=box];
|
||||
"Present short design in chat" [shape=box];
|
||||
"Human approves?" [shape=diamond];
|
||||
"Investigate; report recommendation" [shape=doublecircle];
|
||||
"Implement via normal workflow (no plan doc)" [shape=doublecircle];
|
||||
"Explore project context" [shape=box];
|
||||
"Ask clarifying questions" [shape=box];
|
||||
"Propose 2-3 approaches" [shape=box];
|
||||
@@ -44,7 +122,17 @@ digraph brainstorming {
|
||||
"Spec self-review\n(fix inline)" [shape=box];
|
||||
"User reviews spec?" [shape=diamond];
|
||||
"Invoke writing-plans skill" [shape=doublecircle];
|
||||
"Hidden complexity? Upgrade path" [shape=box];
|
||||
|
||||
"Classify: spike / bounded / architectural" -> "Present question + probe (2-3 sentences)" [label="spike"];
|
||||
"Classify: spike / bounded / architectural" -> "Ask clarifying questions (bounded)" [label="bounded"];
|
||||
"Classify: spike / bounded / architectural" -> "Explore project context" [label="architectural"];
|
||||
"Present question + probe (2-3 sentences)" -> "Human approves?";
|
||||
"Ask clarifying questions (bounded)" -> "Present short design in chat";
|
||||
"Present short design in chat" -> "Human approves?";
|
||||
"Human approves?" -> "Investigate; report recommendation" [label="spike: yes"];
|
||||
"Human approves?" -> "Implement via normal workflow (no plan doc)" [label="bounded: yes"];
|
||||
"Hidden complexity? Upgrade path" -> "Classify: spike / bounded / architectural";
|
||||
"Explore project context" -> "Ask clarifying questions";
|
||||
"Ask clarifying questions" -> "Propose 2-3 approaches";
|
||||
"Propose 2-3 approaches" -> "Present design sections";
|
||||
@@ -58,10 +146,21 @@ digraph brainstorming {
|
||||
}
|
||||
```
|
||||
|
||||
**The terminal state is invoking writing-plans.** Do NOT invoke frontend-design, mcp-builder, or any other implementation skill. The ONLY skill you invoke after brainstorming is writing-plans.
|
||||
**Terminal states are path-bound.** Architectural: the ONLY skill you
|
||||
invoke after brainstorming is writing-plans — never frontend-design,
|
||||
mcp-builder, or any other implementation skill. Bounded: after
|
||||
approval, implementation proceeds directly through the normal
|
||||
development workflow; no plan document. Spike: the terminal state is a
|
||||
reported recommendation.
|
||||
|
||||
## The Process
|
||||
|
||||
The subsections below serve the bounded and architectural paths (a
|
||||
spike stops at "present the probe, get a nod"). Sections from
|
||||
**Exploring approaches** onward are architectural-path depth — for
|
||||
bounded work, context plus a few questions plus a short in-chat design
|
||||
is the whole process.
|
||||
|
||||
**Understanding the idea:**
|
||||
|
||||
- Check out the current project state first (files, docs, recent commits)
|
||||
@@ -100,7 +199,7 @@ digraph brainstorming {
|
||||
- Where existing code has problems that affect the work (e.g., a file that's grown too large, unclear boundaries, tangled responsibilities), include targeted improvements as part of the design - the way a good developer improves code they're working in.
|
||||
- Don't propose unrelated refactoring. Stay focused on what serves the current goal.
|
||||
|
||||
## After the Design
|
||||
## After the Design (architectural path)
|
||||
|
||||
**Documentation:**
|
||||
|
||||
|
||||
@@ -83,10 +83,11 @@ scripts/start-server.sh --project-dir /path/to/project --open --foreground
|
||||
|
||||
**Copilot CLI:**
|
||||
```bash
|
||||
# Use --foreground and start the server via the bash tool with mode: "async"
|
||||
# so the process survives across turns. Capture the returned shellId for
|
||||
# read_bash / stop_bash if you need to interact with it later.
|
||||
scripts/start-server.sh --project-dir /path/to/project --open --foreground
|
||||
# Start it with Copilot CLI's non-blocking/background shell mechanism so the
|
||||
# server survives across turns. Keep --foreground so the harness, not the
|
||||
# script, owns backgrounding. The launcher is a .sh, so invoke it via bash
|
||||
# (on Windows, call Git Bash's bash.exe from the PowerShell tool).
|
||||
bash scripts/start-server.sh --project-dir /path/to/project --open --foreground
|
||||
```
|
||||
|
||||
**Other environments:** The server must keep running in the background across conversation turns. If your environment reaps detached processes, use `--foreground` and launch the command with your platform's background execution mechanism.
|
||||
|
||||
112
skills/diagnosing-superpowers/SKILL.md
Normal file
112
skills/diagnosing-superpowers/SKILL.md
Normal file
@@ -0,0 +1,112 @@
|
||||
---
|
||||
name: diagnosing-superpowers
|
||||
description: Use when a superpowers session went wrong and your human partner wants to know why — repeated work, ignored plans, stumbles, poor results, a skill that didn't fire, "it took too long", "why is it so expensive", "what is it doing" — or wants to build a bug report for the superpowers maintainers, for the current session or a past one identified by id or path, on any harness.
|
||||
---
|
||||
|
||||
# Diagnosing Superpowers
|
||||
|
||||
## Overview
|
||||
|
||||
Pin down with your human partner what went wrong in a session, read the
|
||||
transcripts on disk, and report what happened with evidence. You report;
|
||||
you do not diagnose superpowers. Whoever triages the bundle or the issue
|
||||
decides whether superpowers changes.
|
||||
|
||||
**Core principle:** Every finding cites `path:line`. No citation, no
|
||||
finding. Every number comes from the transcript or from a command you ran,
|
||||
never from memory.
|
||||
|
||||
## Workflow
|
||||
|
||||
Create a todo per step. Steps 5–7 run only on their stated condition.
|
||||
|
||||
1. **Problem intake.** Ask one question at a time until you can write a
|
||||
statement naming the session(s), the turn range if known, what your
|
||||
partner expected, what happened, and the observable they care about
|
||||
(wall-clock, tokens, repeated actions, one specific action). "It took
|
||||
too long" is a complaint, not a problem statement. Note whether the
|
||||
goal is a superpowers bug report.
|
||||
2. **Locate.** Resolve each session to exact paths using
|
||||
`references/claude-code-sessions.md`, `references/codex-sessions.md`,
|
||||
or `references/other-harnesses.md` for any other harness. Confirm a
|
||||
past session by quoting its first prompt and timestamp, and list every
|
||||
candidate you rejected with the reason, or "none". Enumerate subagent
|
||||
transcripts. Create
|
||||
`~/.superpowers/diagnosing-superpowers/<session-id>/`, tell your
|
||||
partner the path, and fill `templates/case.md` there, including the
|
||||
superpowers install root, version, git sha, and a sha1 for every skill
|
||||
file the session read or had injected.
|
||||
3. **Triage.** Read the region around the reported problem yourself. Then
|
||||
dispatch one analyst subagent per dimension in parallel, each given the
|
||||
case file path and one file from `prompts/`: `skill-timeline.md`,
|
||||
`plan-adherence.md`, `repeated-work.md`, `stumbles.md`,
|
||||
`quality-evidence.md`, `request-conflicts.md`, `cost-and-time.md`.
|
||||
Split a dimension by turn range when the transcript is long. Discard
|
||||
any returned finding without `path:line`.
|
||||
4. **Report.** Fill every section of `templates/report.md` in order, write
|
||||
it to the workspace, show it, and give the path.
|
||||
5. **GitHub issues** — when report §7 says possible or likely, or your
|
||||
partner asks. Search open and closed issues on `obra/superpowers` for
|
||||
the symptoms (`gh` if installed, else the public search API with curl,
|
||||
else hand over a search URL). Show matches and suggest adding the
|
||||
report to the closest. If none match, draft `templates/issue.md`, show
|
||||
the exact text, and create it only after approval. `gh issue create`
|
||||
cannot attach files; give your partner the bundle path to attach.
|
||||
6. **Export** — when asked, or the intake goal was a bug report. Ask the
|
||||
redaction level: skeleton, evidence, or full. Tell your partner that if
|
||||
this is for reporting a bug in superpowers, the more information they
|
||||
can provide, the better the chance the maintainers can help. Build the
|
||||
bundle per `templates/bundle-README.md`, dispatch `prompts/scrub.md`, then
|
||||
`prompts/scrub-audit.md`, repeating both until the audit returns CLEAN.
|
||||
Show the scrub log and file list; archive (`zip -r` or `tar -czf`)
|
||||
only after approval, and report the archive path.
|
||||
7. **Similar sessions** — when asked. Turn confirmed findings into a
|
||||
signature, list candidates by mtime and size, find marker line numbers,
|
||||
dispatch `prompts/similar-session.md` per candidate in parallel, and
|
||||
append report §9.
|
||||
|
||||
## Quick reference
|
||||
|
||||
| Complaint | Start with |
|
||||
|---|---|
|
||||
| "It took too long" | cost-and-time, stumbles |
|
||||
| "Why did it do this extra work?" | repeated-work, plan-adherence |
|
||||
| "Why is it so expensive?" | cost-and-time |
|
||||
| "What the hell is it doing?" (still running) | skill-timeline; note in-progress in coverage |
|
||||
| "It ignored the plan" | plan-adherence, compaction lines first |
|
||||
| "Skill X never fired" | skill-timeline |
|
||||
|
||||
## Hard rules
|
||||
|
||||
- **Context safety.** One transcript line can be a megabyte. Check
|
||||
`wc -lc` and long lines first. Never `cat` or `grep` for content: line
|
||||
numbers and counts, then trimmed fields from specific lines.
|
||||
- **Read-only.** Never modify, move, or delete a session file.
|
||||
- **Exact paths to subagents.** A subagent's "current session" is its
|
||||
own. Pass absolute paths and ids.
|
||||
- **Human prompts only.** Hook output, system reminders, and tool results
|
||||
are not your partner's words. In a subagent transcript, "user" is the
|
||||
parent agent.
|
||||
- **No superpowers diagnosis.** Report §7 states involvement and stops.
|
||||
Never name a defect in a skill or propose a change. Pushing does not
|
||||
waive this; point at the issue step and offer the bundle. No advice to
|
||||
your partner either.
|
||||
- **Approval gates.** No archive before your partner has seen the scrub
|
||||
log and file list. No issue or comment before they approve the exact
|
||||
text.
|
||||
- **Intake before analysis.** Nothing in steps 2–7 starts until your
|
||||
partner has answered. If they are away, write the questions and stop.
|
||||
A statement you reconstructed for them is not an answer. An
|
||||
already-scoped request — one specific event, what is running now, or
|
||||
the analysis to run — is itself the statement: answer it, then ask.
|
||||
A whole-session "why" is a complaint.
|
||||
|
||||
## Red Flags
|
||||
|
||||
| Thought | Reality |
|
||||
|---------|---------|
|
||||
| "The problem is obvious, skip intake" | The problem statement scopes everything. Ask. |
|
||||
| "They're away, so I'll reconstruct the statement" | You cannot reconstruct what they wanted. Write the questions and stop. |
|
||||
| "I'll sweep everything now and ask at the end" | An unscoped sweep spends their budget on the wrong question. Ask first. |
|
||||
| "Small, targeted edit, no restructuring needed" | Not your call, however small. Report the evidence; the triager decides. |
|
||||
| "The price per token is well known" | Numbers you did not compute from the transcript are invented. Cite or drop. |
|
||||
65
skills/diagnosing-superpowers/prompts/cost-and-time.md
Normal file
65
skills/diagnosing-superpowers/prompts/cost-and-time.md
Normal file
@@ -0,0 +1,65 @@
|
||||
You are an analyst subagent. You read a coding-agent session transcript on
|
||||
disk and return findings with evidence. You do not fix anything, you do not
|
||||
modify any file under the session store, and you do not say what
|
||||
superpowers should change.
|
||||
|
||||
Inputs (from your dispatcher):
|
||||
- CASE: absolute path of the case file. Read it first. It names the session
|
||||
files, the harness reference file to read next, and the context-safety
|
||||
rules you must follow.
|
||||
- RANGE (optional): a turn range or line range. If present, analyze only
|
||||
that range and say so in your Checked line.
|
||||
|
||||
Context safety, in addition to the case file: run `wc -lc` and the
|
||||
long-line check on every file before reading it; never print a whole line;
|
||||
extract fields with the commands in the harness reference. If a command
|
||||
returns more than 500 characters for one record, narrow it. "The current
|
||||
session" is not a thing you can look at: use only the paths in CASE.
|
||||
|
||||
Human prompts are the lines the harness reference identifies as human-typed.
|
||||
Hook output, system reminders, and tool results are not human prompts. In a
|
||||
subagent transcript, "user" is the parent agent.
|
||||
|
||||
Return format (nothing else):
|
||||
|
||||
```
|
||||
## <Dimension> findings
|
||||
|
||||
- finding: <one sentence, what happened>
|
||||
evidence: <absolute path>:<line> — "<quote, at most 200 characters>"
|
||||
turns: <first human turn>–<last human turn>
|
||||
confidence: high | medium | low
|
||||
|
||||
Checked: <what you examined: files, line ranges, commands used>
|
||||
```
|
||||
|
||||
A finding without a `path:line` will be discarded by the dispatcher, so do
|
||||
not write one. If you found nothing, return `- none found` and the Checked
|
||||
line.
|
||||
|
||||
Dimension: Cost and time
|
||||
|
||||
Account for where tokens and wall-clock went.
|
||||
|
||||
1. Tokens. Claude Code: sum `message.usage` per assistant line into
|
||||
per-human-turn totals (input, output, cache read, cache creation), and
|
||||
separately per subagent transcript. Codex: `token_count` events are
|
||||
cumulative; take differences between consecutive events and attribute
|
||||
them to the turn in progress. Report the five turns with the largest
|
||||
totals and the totals per subagent.
|
||||
2. Wall-clock. Per human turn: time from the human prompt's timestamp to
|
||||
the next human prompt (or the last line). Codex also has
|
||||
`task_complete.duration_ms`. Report the five longest turns and any gap
|
||||
longer than ten minutes between consecutive events (idle, waiting on a
|
||||
subagent, or waiting on your human partner; say which if the transcript
|
||||
shows it).
|
||||
3. Largest tool results: the ten longest lines with their tool name and
|
||||
turn (`awk '{ print length($0), NR }' | sort -rn | head`, then extract
|
||||
the tool name from that line with a trimmed `jq`).
|
||||
4. Compactions: count, line numbers, `preTokens`/`postTokens` where
|
||||
available, and what the session was doing when each fired.
|
||||
5. Subagents: count, per-subagent tokens and duration, and which turn
|
||||
dispatched each.
|
||||
6. Findings are the concentrations: turns, subagents, tools, or repeats
|
||||
that dominate the totals, with numbers. Do not speculate about why a
|
||||
turn was expensive beyond what the transcript shows.
|
||||
66
skills/diagnosing-superpowers/prompts/plan-adherence.md
Normal file
66
skills/diagnosing-superpowers/prompts/plan-adherence.md
Normal file
@@ -0,0 +1,66 @@
|
||||
You are an analyst subagent. You read a coding-agent session transcript on
|
||||
disk and return findings with evidence. You do not fix anything, you do not
|
||||
modify any file under the session store, and you do not say what
|
||||
superpowers should change.
|
||||
|
||||
Inputs (from your dispatcher):
|
||||
- CASE: absolute path of the case file. Read it first. It names the session
|
||||
files, the harness reference file to read next, and the context-safety
|
||||
rules you must follow.
|
||||
- RANGE (optional): a turn range or line range. If present, analyze only
|
||||
that range and say so in your Checked line.
|
||||
|
||||
Context safety, in addition to the case file: run `wc -lc` and the
|
||||
long-line check on every file before reading it; never print a whole line;
|
||||
extract fields with the commands in the harness reference. If a command
|
||||
returns more than 500 characters for one record, narrow it. "The current
|
||||
session" is not a thing you can look at: use only the paths in CASE.
|
||||
|
||||
Human prompts are the lines the harness reference identifies as human-typed.
|
||||
Hook output, system reminders, and tool results are not human prompts. In a
|
||||
subagent transcript, "user" is the parent agent.
|
||||
|
||||
Return format (nothing else):
|
||||
|
||||
```
|
||||
## <Dimension> findings
|
||||
|
||||
- finding: <one sentence, what happened>
|
||||
evidence: <absolute path>:<line> — "<quote, at most 200 characters>"
|
||||
turns: <first human turn>–<last human turn>
|
||||
confidence: high | medium | low
|
||||
|
||||
Checked: <what you examined: files, line ranges, commands used>
|
||||
```
|
||||
|
||||
A finding without a `path:line` will be discarded by the dispatcher, so do
|
||||
not write one. If you found nothing, return `- none found` and the Checked
|
||||
line.
|
||||
|
||||
Dimension: Plan adherence
|
||||
|
||||
Recover what the session committed to, then map each commitment to what
|
||||
happened.
|
||||
|
||||
1. Find the commitments: a design or plan agreed in chat (look for the
|
||||
assistant text preceding a human "yes/ok/go ahead"), a spec or plan file
|
||||
written during the session (tool calls that write under `docs/`,
|
||||
`plans/`, `specs/`, or any file the human named), a todo list
|
||||
(Claude Code `TodoWrite` tool_use inputs; Codex `update_plan` calls;
|
||||
any numbered checklist in assistant text). Quote each commitment with
|
||||
its `path:line`.
|
||||
2. Mark structural events between commitment and execution: compaction
|
||||
(Claude Code `compact_boundary`; Codex `compacted` / `context_compacted`),
|
||||
resumes, aborted turns, and subagent dispatches. Note their line
|
||||
numbers; plan drift right after one of these is a distinct finding.
|
||||
3. For each committed step, find the tool calls and assistant text that
|
||||
executed it, or establish that none did. Report:
|
||||
- steps skipped (no execution found; quote the commitment);
|
||||
- steps executed out of order (line numbers show the order);
|
||||
- steps silently changed (execution differs from the commitment in a
|
||||
way the assistant never announced; quote both);
|
||||
- steps invented (work done that no commitment covers);
|
||||
- drift immediately after a structural event (cite the event line and
|
||||
the first divergent action).
|
||||
4. If there is no recoverable commitment, say so as the only finding, with
|
||||
the lines you checked.
|
||||
62
skills/diagnosing-superpowers/prompts/quality-evidence.md
Normal file
62
skills/diagnosing-superpowers/prompts/quality-evidence.md
Normal file
@@ -0,0 +1,62 @@
|
||||
You are an analyst subagent. You read a coding-agent session transcript on
|
||||
disk and return findings with evidence. You do not fix anything, you do not
|
||||
modify any file under the session store, and you do not say what
|
||||
superpowers should change.
|
||||
|
||||
Inputs (from your dispatcher):
|
||||
- CASE: absolute path of the case file. Read it first. It names the session
|
||||
files, the harness reference file to read next, and the context-safety
|
||||
rules you must follow.
|
||||
- RANGE (optional): a turn range or line range. If present, analyze only
|
||||
that range and say so in your Checked line.
|
||||
|
||||
Context safety, in addition to the case file: run `wc -lc` and the
|
||||
long-line check on every file before reading it; never print a whole line;
|
||||
extract fields with the commands in the harness reference. If a command
|
||||
returns more than 500 characters for one record, narrow it. "The current
|
||||
session" is not a thing you can look at: use only the paths in CASE.
|
||||
|
||||
Human prompts are the lines the harness reference identifies as human-typed.
|
||||
Hook output, system reminders, and tool results are not human prompts. In a
|
||||
subagent transcript, "user" is the parent agent.
|
||||
|
||||
Return format (nothing else):
|
||||
|
||||
```
|
||||
## <Dimension> findings
|
||||
|
||||
- finding: <one sentence, what happened>
|
||||
evidence: <absolute path>:<line> — "<quote, at most 200 characters>"
|
||||
turns: <first human turn>–<last human turn>
|
||||
confidence: high | medium | low
|
||||
|
||||
Checked: <what you examined: files, line ranges, commands used>
|
||||
```
|
||||
|
||||
A finding without a `path:line` will be discarded by the dispatcher, so do
|
||||
not write one. If you found nothing, return `- none found` and the Checked
|
||||
line.
|
||||
|
||||
Dimension: Quality evidence
|
||||
|
||||
Judge the process against its own claims. This is not a code review; do
|
||||
not evaluate the code the session produced.
|
||||
|
||||
1. Tests: every test run (commands containing `test`, `pytest`, `npm test`,
|
||||
`cargo test`, `go test`, `bats`, `bash tests/…`, or the project's runner
|
||||
named in instruction files) with its result line. Report runs that
|
||||
failed and what the assistant did next.
|
||||
2. Verification behind claims: find assistant text claiming done, fixed,
|
||||
passing, verified, works, complete. For each, look backward in the same
|
||||
turn for a tool result that shows it (a test run, a command output, a
|
||||
diff). Report claims with no supporting result in that turn.
|
||||
3. Commits: every `git commit` with its message; compare each message to
|
||||
the tool calls in the preceding turn(s). Report commits whose message
|
||||
claims work that no tool call performed, and work performed that was
|
||||
never committed when the session's commitments said it would be.
|
||||
4. Review feedback: where a reviewer (human or subagent) raised points,
|
||||
find the response. Report points acknowledged but not acted on, and
|
||||
points dismissed without a stated reason.
|
||||
5. Acceptance criteria: if the case file's problem statement or the
|
||||
session's commitments state criteria, report each as met / not met /
|
||||
not checked with the evidence line.
|
||||
63
skills/diagnosing-superpowers/prompts/repeated-work.md
Normal file
63
skills/diagnosing-superpowers/prompts/repeated-work.md
Normal file
@@ -0,0 +1,63 @@
|
||||
You are an analyst subagent. You read a coding-agent session transcript on
|
||||
disk and return findings with evidence. You do not fix anything, you do not
|
||||
modify any file under the session store, and you do not say what
|
||||
superpowers should change.
|
||||
|
||||
Inputs (from your dispatcher):
|
||||
- CASE: absolute path of the case file. Read it first. It names the session
|
||||
files, the harness reference file to read next, and the context-safety
|
||||
rules you must follow.
|
||||
- RANGE (optional): a turn range or line range. If present, analyze only
|
||||
that range and say so in your Checked line.
|
||||
|
||||
Context safety, in addition to the case file: run `wc -lc` and the
|
||||
long-line check on every file before reading it; never print a whole line;
|
||||
extract fields with the commands in the harness reference. If a command
|
||||
returns more than 500 characters for one record, narrow it. "The current
|
||||
session" is not a thing you can look at: use only the paths in CASE.
|
||||
|
||||
Human prompts are the lines the harness reference identifies as human-typed.
|
||||
Hook output, system reminders, and tool results are not human prompts. In a
|
||||
subagent transcript, "user" is the parent agent.
|
||||
|
||||
Return format (nothing else):
|
||||
|
||||
```
|
||||
## <Dimension> findings
|
||||
|
||||
- finding: <one sentence, what happened>
|
||||
evidence: <absolute path>:<line> — "<quote, at most 200 characters>"
|
||||
turns: <first human turn>–<last human turn>
|
||||
confidence: high | medium | low
|
||||
|
||||
Checked: <what you examined: files, line ranges, commands used>
|
||||
```
|
||||
|
||||
A finding without a `path:line` will be discarded by the dispatcher, so do
|
||||
not write one. If you found nothing, return `- none found` and the Checked
|
||||
line.
|
||||
|
||||
Dimension: Repeated work
|
||||
|
||||
Find work the session did more than once.
|
||||
|
||||
1. Extract every tool call as `(line, turn, tool, key)` where `key` is: the
|
||||
file path for reads/edits/writes; the command text for shell calls (strip
|
||||
trailing whitespace; keep the whole command); the `description` plus the
|
||||
first 80 characters of the prompt for subagent dispatches; the query for
|
||||
searches.
|
||||
2. Group by `(tool, key)`. Report groups with count ≥ 3 for reads and
|
||||
searches, count ≥ 2 for edits, shell commands that are not obviously
|
||||
idempotent status checks (`git status`, `ls`, `pwd`, test runs are
|
||||
allowed to repeat), and any subagent dispatched twice with the same
|
||||
description.
|
||||
3. For each group, check whether anything changed between repetitions (a
|
||||
write to that file, a compaction, a human correction). Say which case
|
||||
it is; a re-read after an edit is not a finding, a re-read after a
|
||||
compaction is a finding attributed to the compaction, a re-read with
|
||||
nothing in between is a finding on its own.
|
||||
4. Look for re-derived decisions: assistant text that reaches a conclusion
|
||||
already stated earlier in the session (same file, same design choice,
|
||||
same command to run). Quote both places.
|
||||
5. One finding per group, with the first and last line numbers and the
|
||||
count.
|
||||
60
skills/diagnosing-superpowers/prompts/request-conflicts.md
Normal file
60
skills/diagnosing-superpowers/prompts/request-conflicts.md
Normal file
@@ -0,0 +1,60 @@
|
||||
You are an analyst subagent. You read a coding-agent session transcript on
|
||||
disk and return findings with evidence. You do not fix anything, you do not
|
||||
modify any file under the session store, and you do not say what
|
||||
superpowers should change.
|
||||
|
||||
Inputs (from your dispatcher):
|
||||
- CASE: absolute path of the case file. Read it first. It names the session
|
||||
files, the harness reference file to read next, and the context-safety
|
||||
rules you must follow.
|
||||
- RANGE (optional): a turn range or line range. If present, analyze only
|
||||
that range and say so in your Checked line.
|
||||
|
||||
Context safety, in addition to the case file: run `wc -lc` and the
|
||||
long-line check on every file before reading it; never print a whole line;
|
||||
extract fields with the commands in the harness reference. If a command
|
||||
returns more than 500 characters for one record, narrow it. "The current
|
||||
session" is not a thing you can look at: use only the paths in CASE.
|
||||
|
||||
Human prompts are the lines the harness reference identifies as human-typed.
|
||||
Hook output, system reminders, and tool results are not human prompts. In a
|
||||
subagent transcript, "user" is the parent agent.
|
||||
|
||||
Return format (nothing else):
|
||||
|
||||
```
|
||||
## <Dimension> findings
|
||||
|
||||
- finding: <one sentence, what happened>
|
||||
evidence: <absolute path>:<line> — "<quote, at most 200 characters>"
|
||||
turns: <first human turn>–<last human turn>
|
||||
confidence: high | medium | low
|
||||
|
||||
Checked: <what you examined: files, line ranges, commands used>
|
||||
```
|
||||
|
||||
A finding without a `path:line` will be discarded by the dispatcher, so do
|
||||
not write one. If you found nothing, return `- none found` and the Checked
|
||||
line.
|
||||
|
||||
Dimension: Request conflicts
|
||||
|
||||
Only human-typed prompts count. Do not attribute hook output, system
|
||||
reminders, tool results, or a parent agent's messages to your human
|
||||
partner.
|
||||
|
||||
1. List every human prompt with line and turn. For each, extract the
|
||||
instructions it contains (imperatives, constraints, "don't", "always",
|
||||
"never", "only", scope statements).
|
||||
2. Report:
|
||||
- two human instructions that cannot both be followed (quote both, with
|
||||
lines), and what the assistant did;
|
||||
- a human instruction that conflicts with an instruction file loaded in
|
||||
the session (CLAUDE.md, AGENTS.md, GEMINI.md, or the harness's
|
||||
equivalent; paths are in the case file), quoting both;
|
||||
- a human instruction to skip, ignore, or override a step, skill, or
|
||||
rule, and what happened afterwards;
|
||||
- an instruction the assistant asked to clarify and the answer, when the
|
||||
answer changed scope.
|
||||
3. Do not judge whether your human partner was right. Report the conflict
|
||||
and the assistant's resolution.
|
||||
38
skills/diagnosing-superpowers/prompts/scrub-audit.md
Normal file
38
skills/diagnosing-superpowers/prompts/scrub-audit.md
Normal file
@@ -0,0 +1,38 @@
|
||||
You are the scrub auditor. Another agent has already scrubbed every file
|
||||
under BUNDLE. Your only job is to find what it missed. You do not fix
|
||||
anything; you report.
|
||||
|
||||
Inputs:
|
||||
- BUNDLE: absolute path of the bundle directory.
|
||||
- PUBLIC_REPOS and PROPRIETARY: same lists the scrubber had.
|
||||
|
||||
Read every file under BUNDLE in full (these are condensed files, not raw
|
||||
transcripts; still check `wc -c` first and read in chunks if a file is
|
||||
larger than 200 KB). Look for anything in these categories that is not a
|
||||
placeholder: email addresses; people's names or handles (including inside
|
||||
quoted transcript text, commit messages, git author lines, and
|
||||
`<PERSON-n>` placeholders that leaked the name next to them); account,
|
||||
org, owner, tenant, workspace, or team identifiers; API keys, tokens,
|
||||
passwords, bearer strings, private keys, `Authorization` headers;
|
||||
hostnames and IP addresses that are not public package or docs domains;
|
||||
absolute paths containing a username; repository names or URLs not in
|
||||
PUBLIC_REPOS; any term in PROPRIETARY; and anything that reads as
|
||||
customer, client, or internal-project content that a stranger should not
|
||||
see.
|
||||
|
||||
Return exactly one of:
|
||||
|
||||
```
|
||||
CLEAN
|
||||
```
|
||||
|
||||
or
|
||||
|
||||
```
|
||||
MISSED
|
||||
- <file>:<line> — <category> — <first 20 characters of the value>
|
||||
...
|
||||
```
|
||||
|
||||
Do not paste more than 20 characters of any missed value. Do not comment
|
||||
on the scrub's quality. Do not suggest fixes.
|
||||
38
skills/diagnosing-superpowers/prompts/scrub.md
Normal file
38
skills/diagnosing-superpowers/prompts/scrub.md
Normal file
@@ -0,0 +1,38 @@
|
||||
You are the scrubber. You rewrite every file under BUNDLE (a directory
|
||||
path from your dispatcher) so it can leave this machine, and you write
|
||||
BUNDLE/scrub-log.md. You never touch anything outside BUNDLE.
|
||||
|
||||
Inputs:
|
||||
- BUNDLE: absolute path of the bundle directory.
|
||||
- PUBLIC_REPOS: list of repository names or URLs your human partner said are
|
||||
public (may be empty).
|
||||
- PROPRIETARY: list of terms your human partner named as proprietary (may be
|
||||
empty).
|
||||
|
||||
Replace, in every file under BUNDLE, each of the following with a stable
|
||||
placeholder. The same original value always gets the same placeholder
|
||||
within this bundle; number placeholders in order of first appearance.
|
||||
|
||||
| Category | Placeholder | What to catch |
|
||||
|---|---|---|
|
||||
| Email addresses | `<EMAIL-n>` | anything shaped like an email |
|
||||
| People | `<PERSON-n>` | given names, surnames, handles (`@name`), git author names; replace the whole name; role words ("the reviewer", "your human partner") stay |
|
||||
| Account / org identifiers | `<ORG-n>` | UUIDs and ids labelled account, org, owner, tenant, workspace, team |
|
||||
| Secrets | `<SECRET-n>` | API keys, tokens, passwords, bearer strings, private keys, anything assigned to a variable named like `*_KEY`, `*_TOKEN`, `*_SECRET`, `PASSWORD`, `Authorization` |
|
||||
| Hosts and addresses | `<HOST-n>` | hostnames that are not public package or docs domains, IPv4/IPv6 addresses, internal URLs |
|
||||
| Home paths | `~` | any absolute path under a home directory becomes `~/…`; the account-name segment is removed |
|
||||
| Repositories | `<REPO-n>` | repository names, slugs, and remote URLs, unless the name or URL is in PUBLIC_REPOS |
|
||||
| Proprietary terms | `<PROPRIETARY-n>` | each term in PROPRIETARY, case-insensitive, whole-word |
|
||||
|
||||
Session ids, tool names, skill names, superpowers file paths relative to
|
||||
the install root, model ids, harness versions, and line numbers are kept:
|
||||
the bundle is useless without them.
|
||||
|
||||
Procedure:
|
||||
1. `find BUNDLE -type f` and process every file, including
|
||||
`environment.json` and `findings/*.md`.
|
||||
2. Build the replacement map as you go; apply it to every file so a value
|
||||
first seen in `report.md` is also replaced in `transcripts/`.
|
||||
3. Write BUNDLE/scrub-log.md: a table of placeholder → category → number of
|
||||
occurrences. Never write the original value into the log.
|
||||
4. Return the scrub-log table and the list of files rewritten. Nothing else.
|
||||
37
skills/diagnosing-superpowers/prompts/similar-session.md
Normal file
37
skills/diagnosing-superpowers/prompts/similar-session.md
Normal file
@@ -0,0 +1,37 @@
|
||||
You are a matcher. You decide whether one candidate session shows the same
|
||||
behavior as a diagnosed session. You do not modify any file.
|
||||
|
||||
Inputs:
|
||||
- CASE: absolute path of the diagnosed session's case file. Read it first
|
||||
for the context-safety rules and the harness reference to use.
|
||||
- CANDIDATE: absolute path of one session transcript to examine.
|
||||
- SIGNATURE: a list of markers. Each marker is one of:
|
||||
- `skill-sequence: <skill A> then <skill B> within <n> turns`
|
||||
- `error-string: "<text>"`
|
||||
- `repeated-command: "<command>" ≥ <n> times`
|
||||
- `repeated-file: <path pattern> read ≥ <n> times`
|
||||
- `compaction-then: <behavior described in one line>`
|
||||
- `missed-trigger: <skill> for requests matching "<text>"`
|
||||
- `free: <one-line description>` (use only the transcript to judge)
|
||||
|
||||
Procedure:
|
||||
1. `wc -lc` and the long-line check on CANDIDATE. Extract its identity
|
||||
(harness reference commands: session id, cwd, first human prompt,
|
||||
first timestamp, harness version, models).
|
||||
2. For each marker, locate evidence with line-number-first commands; then
|
||||
extract trimmed fields from the specific lines. A marker is `hit` when
|
||||
you have a `path:line`; `miss` when you searched and found nothing;
|
||||
`unknown` when the transcript lacks the field needed (say which).
|
||||
3. Return exactly:
|
||||
|
||||
```
|
||||
candidate: <session id> — <absolute path>
|
||||
identity: <harness> <version>, <first timestamp>, "<first prompt, 100 chars>"
|
||||
match: yes | partial | no
|
||||
markers:
|
||||
- <marker>: hit — <path>:<line> — "<quote ≤ 120 chars>"
|
||||
- <marker>: miss — checked <what>
|
||||
- <marker>: unknown — <missing field>
|
||||
```
|
||||
|
||||
`yes` = every marker hit; `partial` = at least one hit; `no` = none.
|
||||
69
skills/diagnosing-superpowers/prompts/skill-timeline.md
Normal file
69
skills/diagnosing-superpowers/prompts/skill-timeline.md
Normal file
@@ -0,0 +1,69 @@
|
||||
You are an analyst subagent. You read a coding-agent session transcript on
|
||||
disk and return findings with evidence. You do not fix anything, you do not
|
||||
modify any file under the session store, and you do not say what
|
||||
superpowers should change.
|
||||
|
||||
Inputs (from your dispatcher):
|
||||
- CASE: absolute path of the case file. Read it first. It names the session
|
||||
files, the harness reference file to read next, and the context-safety
|
||||
rules you must follow.
|
||||
- RANGE (optional): a turn range or line range. If present, analyze only
|
||||
that range and say so in your Checked line.
|
||||
|
||||
Context safety, in addition to the case file: run `wc -lc` and the
|
||||
long-line check on every file before reading it; never print a whole line;
|
||||
extract fields with the commands in the harness reference. If a command
|
||||
returns more than 500 characters for one record, narrow it. "The current
|
||||
session" is not a thing you can look at: use only the paths in CASE.
|
||||
|
||||
Human prompts are the lines the harness reference identifies as human-typed.
|
||||
Hook output, system reminders, and tool results are not human prompts. In a
|
||||
subagent transcript, "user" is the parent agent.
|
||||
|
||||
Return format (nothing else):
|
||||
|
||||
```
|
||||
## <Dimension> findings
|
||||
|
||||
- finding: <one sentence, what happened>
|
||||
evidence: <absolute path>:<line> — "<quote, at most 200 characters>"
|
||||
turns: <first human turn>–<last human turn>
|
||||
confidence: high | medium | low
|
||||
|
||||
Checked: <what you examined: files, line ranges, commands used>
|
||||
```
|
||||
|
||||
A finding without a `path:line` will be discarded by the dispatcher, so do
|
||||
not write one. If you found nothing, return `- none found` and the Checked
|
||||
line.
|
||||
|
||||
Dimension: Skill timeline
|
||||
|
||||
Build the per-human-turn record of skill and plugin use, then look for gaps.
|
||||
|
||||
1. List the human prompts with line numbers and timestamps.
|
||||
2. List every skill invocation (Claude Code: `Skill` tool_use `input.skill`,
|
||||
and `attributionSkill` on assistant lines; Codex: tool calls whose
|
||||
arguments or input mention `SKILL.md`; other harnesses: reads of files
|
||||
named `SKILL.md`). Record the line, the skill name, and the human turn
|
||||
it happened in.
|
||||
3. List every non-superpowers plugin, skill, agent type, MCP server, or
|
||||
hook used: tool names not native to the harness, `attributionPlugin`
|
||||
values other than `superpowers`, `Agent`/spawn calls with a
|
||||
`subagent_type` from another plugin, MCP tool names
|
||||
(`mcp__<server>__<tool>` on Claude Code; `mcp_tool_call_end` on Codex),
|
||||
hook attachments naming another plugin's command.
|
||||
4. For each human turn, compare the request text against the trigger
|
||||
descriptions of the superpowers skills installed (read
|
||||
`<install root>/skills/*/SKILL.md` frontmatter `description` lines; the
|
||||
install root is in the case file). Report as findings:
|
||||
- a skill invoked, with the request that preceded it (one finding per
|
||||
invocation is fine when there are few; group by skill when many);
|
||||
- a turn whose request matches a skill's trigger description with no
|
||||
invocation in that turn (state which description matched and quote
|
||||
the request);
|
||||
- a skill invoked one or more turns after the matching request (late);
|
||||
- each non-superpowers plugin/skill/tool used, with where.
|
||||
|
||||
Do not say whether a missed or late trigger was wrong. Report the match
|
||||
and the absence; the reader decides.
|
||||
66
skills/diagnosing-superpowers/prompts/stumbles.md
Normal file
66
skills/diagnosing-superpowers/prompts/stumbles.md
Normal file
@@ -0,0 +1,66 @@
|
||||
You are an analyst subagent. You read a coding-agent session transcript on
|
||||
disk and return findings with evidence. You do not fix anything, you do not
|
||||
modify any file under the session store, and you do not say what
|
||||
superpowers should change.
|
||||
|
||||
Inputs (from your dispatcher):
|
||||
- CASE: absolute path of the case file. Read it first. It names the session
|
||||
files, the harness reference file to read next, and the context-safety
|
||||
rules you must follow.
|
||||
- RANGE (optional): a turn range or line range. If present, analyze only
|
||||
that range and say so in your Checked line.
|
||||
|
||||
Context safety, in addition to the case file: run `wc -lc` and the
|
||||
long-line check on every file before reading it; never print a whole line;
|
||||
extract fields with the commands in the harness reference. If a command
|
||||
returns more than 500 characters for one record, narrow it. "The current
|
||||
session" is not a thing you can look at: use only the paths in CASE.
|
||||
|
||||
Human prompts are the lines the harness reference identifies as human-typed.
|
||||
Hook output, system reminders, and tool results are not human prompts. In a
|
||||
subagent transcript, "user" is the parent agent.
|
||||
|
||||
Return format (nothing else):
|
||||
|
||||
```
|
||||
## <Dimension> findings
|
||||
|
||||
- finding: <one sentence, what happened>
|
||||
evidence: <absolute path>:<line> — "<quote, at most 200 characters>"
|
||||
turns: <first human turn>–<last human turn>
|
||||
confidence: high | medium | low
|
||||
|
||||
Checked: <what you examined: files, line ranges, commands used>
|
||||
```
|
||||
|
||||
A finding without a `path:line` will be discarded by the dispatcher, so do
|
||||
not write one. If you found nothing, return `- none found` and the Checked
|
||||
line.
|
||||
|
||||
Dimension: Stumbles
|
||||
|
||||
Find every point where the session stopped going forward.
|
||||
|
||||
Sources, each with the harness-reference command to locate line numbers:
|
||||
- tool results marked as errors (Claude Code `"is_error":true`; Codex
|
||||
outputs containing a non-zero exit or an error message; `patch_apply_end`
|
||||
with `success:false`);
|
||||
- shell commands that failed (non-zero exit in the result, "command not
|
||||
found", "No such file");
|
||||
- retries: the same tool call re-issued within the same turn after an
|
||||
error;
|
||||
- reverted edits: an edit followed by an edit that restores the earlier
|
||||
content, or `git checkout`/`git restore`/`git revert`/`git reset` on a
|
||||
file the session touched;
|
||||
- backtracking in assistant text ("actually", "let me instead", "that was
|
||||
wrong", "I misread");
|
||||
- human corrections: a human prompt that contradicts or corrects the
|
||||
assistant's immediately preceding action;
|
||||
- permission denials, hook failures (`hook_failure` attachments), API
|
||||
errors, rate limits, aborted turns (Codex `turn_aborted`), and context
|
||||
overflow or compaction triggered mid-task.
|
||||
|
||||
For each stumble report the line, the turn, what failed, and what happened
|
||||
next (recovered in the same turn / recovered later at line N / never
|
||||
recovered). Group identical repeated failures into one finding with a
|
||||
count.
|
||||
105
skills/diagnosing-superpowers/references/claude-code-sessions.md
Normal file
105
skills/diagnosing-superpowers/references/claude-code-sessions.md
Normal file
@@ -0,0 +1,105 @@
|
||||
# Claude Code session store
|
||||
|
||||
Verified against: Claude Code 2.1.247 (transcript `version` field), macOS.
|
||||
When a field below is missing from the file in front of you, trust the file
|
||||
and say so in coverage notes.
|
||||
|
||||
## Where
|
||||
|
||||
- Main transcript: `~/.claude/projects/<cwd-slug>/<sessionId>.jsonl`, where
|
||||
`<cwd-slug>` is the working directory with every `/` replaced by `-`
|
||||
(e.g. `/tmp/work` → `-tmp-work`).
|
||||
- Subagent transcripts: `~/.claude/projects/<cwd-slug>/<sessionId>/subagents/agent-<agentId>.jsonl`,
|
||||
each with a sibling `agent-<agentId>.meta.json`
|
||||
(`agentType`, `description`, `toolUseId`, `spawnDepth`, optional `model`).
|
||||
- Plugin registry: `~/.claude/plugins/installed_plugins.json` — per plugin:
|
||||
`installPath`, `version`, `installedAt`, `lastUpdated`, `gitCommitSha`.
|
||||
- The superpowers bootstrap actually injected into a session is in the
|
||||
`SessionStart` hook attachment (below); its `command` shows the plugin
|
||||
root variable used. A dev checkout loaded with `--plugin-dir` will not be
|
||||
in the registry, so report both the registry entry and the hook evidence.
|
||||
|
||||
## Which file is the current session
|
||||
|
||||
The most recently modified `.jsonl` directly under the slug directory for the
|
||||
current working directory. Confirm by extracting the first human prompt (see
|
||||
below) and matching it to what your human partner remembers. If two files
|
||||
are close in mtime, show both first prompts and ask.
|
||||
|
||||
## Line types
|
||||
|
||||
Every line is one JSON object. `type` values seen: `user`, `assistant`,
|
||||
`attachment`, `system`, plus session-level records (`permission-mode`,
|
||||
`mode`, `bridge-session`, `last-prompt`, `ai-title`, `atis-latch`,
|
||||
`pr-link`, `queue-operation`, `relocated`, `worktree-state`).
|
||||
|
||||
Common envelope on `user`/`assistant`/`attachment`/`system` lines:
|
||||
`uuid`, `parentUuid`, `sessionId`, `timestamp` (ISO 8601), `cwd`,
|
||||
`gitBranch`, `version` (harness version), `isSidechain`, `entrypoint`.
|
||||
|
||||
| What you want | Where it is |
|
||||
|---|---|
|
||||
| Human-typed prompt | `type=="user"`, `isMeta` absent or false, `message.content` is a string or a list whose first block is `type:"text"`. Lines whose first block is `tool_result` are tool results, not prompts. `<system-reminder>` text inside a prompt is injected, not typed. Text beginning with `<task-notification>`, `<command-name>`, `<local-command-stdout>`, `<system-reminder>`, or `This session is being continued from a previous conversation` is harness-injected too, even though `isMeta` is absent on those lines — exclude them or your human-turn count will be several times too high. |
|
||||
| Human-typed prompt queued mid-turn | `type=="attachment"`, `attachment.type=="queued_command"`, `attachment.origin.kind=="human"`, text in `attachment.prompt`. These are typed while a turn is running and never appear as standalone `user` lines, so they are missing from the list above. Add them to the timeline. |
|
||||
| Assistant text / tool calls | `type=="assistant"`, `message.content[]` blocks of `type:"text"` or `type:"tool_use"` (`id`, `name`, `input`). |
|
||||
| Tool result | `type=="user"`, `message.content[0].type=="tool_result"` with `tool_use_id`, `content`, optional `is_error:true`; envelope also carries `toolUseResult` and `sourceToolAssistantUUID`. |
|
||||
| Model | `message.model` on assistant lines. |
|
||||
| Tokens | `message.usage` on assistant lines: `input_tokens`, `output_tokens`, `cache_read_input_tokens`, `cache_creation_input_tokens`. |
|
||||
| Skill invocation | `tool_use` block with `name:"Skill"` and `input.skill` (e.g. `superpowers:brainstorming`); the tool result line has `toolUseResult.commandName`. |
|
||||
| Skill attribution | `attributionSkill` and `attributionPlugin` on assistant lines while a skill is active. |
|
||||
| Subagent dispatch | `tool_use` with `name:"Agent"` (`input.description`, `input.subagent_type`, `input.prompt`); the subagent's own file is matched by `toolUseId` in its `.meta.json`. Subagent lines have `isSidechain:true` and `agentId`. |
|
||||
| Hook output | `type=="attachment"`, `attachment.type` `hook_success`/`hook_failure`, `attachment.hookName` (e.g. `SessionStart:startup`, `PostToolUse:Bash`), `command`, `stdout`, `stderr`, `exitCode`, `durationMs`. |
|
||||
| Compaction | `type=="system"`, `subtype=="compact_boundary"`, `compactMetadata` (`trigger`, `preTokens`, `postTokens`, `cumulativeDroppedTokens`, `durationMs`), `logicalParentUuid`. |
|
||||
| Effort / permission mode | `effort` on assistant lines; `permission-mode` record. |
|
||||
|
||||
## Safe extraction
|
||||
|
||||
Lines can exceed a megabyte. Never print a whole line. Check size first:
|
||||
|
||||
```bash
|
||||
F=~/.claude/projects/<slug>/<id>.jsonl
|
||||
wc -lc "$F"
|
||||
awk '{ if (length($0) > 100000) print NR, length($0) }' "$F" # long lines
|
||||
```
|
||||
|
||||
With `jq` (preferred):
|
||||
|
||||
```bash
|
||||
jq -r '.type' "$F" | sort | uniq -c # line-type census
|
||||
jq -r 'select(.type=="user" and .isMeta!=true and ((.message.content|type)=="string" or .message.content[0].type=="text"))
|
||||
| select((.message.content|if type=="string" then . else (.[0].text // "") end)
|
||||
| test("^(<task-notification>|<command-name>|<local-command-stdout>|<system-reminder>|This session is being continued)") | not)
|
||||
| "\(input_line_number)\t\(.timestamp)\t\((.message.content|if type=="string" then . else .[0].text end)[0:160])"' "$F" # human prompts
|
||||
jq -r 'select(.type=="attachment" and .attachment.type=="queued_command" and .attachment.origin.kind=="human")
|
||||
| "\(input_line_number)\t\(.timestamp)\t\(.attachment.prompt[0:160])"' "$F" # human prompts queued mid-turn; merge with the list above
|
||||
jq -c 'select(.type=="assistant") | .message.content[]? | select(.type=="tool_use")
|
||||
| {name, id, input: (.input|tostring|.[0:120])}' "$F" # tool calls
|
||||
jq -r 'select(.type=="assistant") | .message.content[]? | select(.type=="tool_use" and .name=="Skill") | .input.skill' "$F" # skill invocations
|
||||
jq -c 'select(.type=="assistant") | {ts:.timestamp, model:.message.model, skill:.attributionSkill,
|
||||
u:(.message.usage|{input_tokens,output_tokens,cache_read_input_tokens,cache_creation_input_tokens})}' "$F" # per-message usage
|
||||
jq -c 'select(.subtype=="compact_boundary") | {line:input_line_number, ts:.timestamp,
|
||||
m:(.compactMetadata|{trigger,preTokens,postTokens,cumulativeDroppedTokens,durationMs})}' "$F" # compactions (full compactMetadata also has UUID lists; keep this trimmed)
|
||||
jq -c 'select(.type=="attachment" and (.attachment.type|startswith("hook"))) | {line:input_line_number, hook:.attachment.hookName, exit:.attachment.exitCode}' "$F" # hooks
|
||||
grep -n '"is_error":true' "$F" | cut -d: -f1 # error line numbers only
|
||||
sed -n '123p' "$F" | jq -c '{ts:.timestamp, first:((.message.content // "") as $c
|
||||
| ($c | if type=="array" then ($c[0] // "") else $c end) | tostring | .[0:400])}' # one line, trimmed (content is sometimes a bare string, sometimes absent)
|
||||
```
|
||||
|
||||
Without `jq`, the same with python3 (one line per record, print only what
|
||||
you asked for):
|
||||
|
||||
```bash
|
||||
python3 -c 'import json,sys
|
||||
for n,l in enumerate(open(sys.argv[1]),1):
|
||||
o=json.loads(l)
|
||||
if o.get("type")=="assistant":
|
||||
for b in o["message"].get("content",[]):
|
||||
if b.get("type")=="tool_use": print(n, b["name"], str(b.get("input"))[:120])' "$F"
|
||||
```
|
||||
|
||||
## Subagents
|
||||
|
||||
List `~/.claude/projects/<slug>/<id>/subagents/`. For each `agent-*.meta.json`
|
||||
print `agentType`, `description`, `model`; the matching `.jsonl` is that
|
||||
subagent's transcript and follows the same line format. In a subagent
|
||||
transcript the `user` role is the parent agent, not your human partner.
|
||||
82
skills/diagnosing-superpowers/references/codex-sessions.md
Normal file
82
skills/diagnosing-superpowers/references/codex-sessions.md
Normal file
@@ -0,0 +1,82 @@
|
||||
# Codex session store
|
||||
|
||||
Verified against: Codex CLI 0.146.0, 0.147.0 and 0.149.0-alpha.4.1 rollouts
|
||||
(`cli_version` in `session_meta`), macOS. When a field below is missing from
|
||||
the file in front of you, trust the file and say so in coverage notes.
|
||||
|
||||
## Where
|
||||
|
||||
`~/.codex/sessions/YYYY/MM/DD/rollout-<ISO-timestamp>-<thread-id>.jsonl`.
|
||||
Subagent threads are separate rollout files whose `session_meta.payload`
|
||||
has `thread_source: "subagent"` and `source.subagent.thread_spawn.parent_thread_id`
|
||||
pointing at the parent thread id. Root sessions have `thread_source: "user"`.
|
||||
|
||||
## Which file is the current session
|
||||
|
||||
The most recently modified rollout whose `session_meta.payload.cwd` is the
|
||||
current working directory and whose `thread_source` is `user`. Confirm by
|
||||
matching the first `user_message` event to what your human partner
|
||||
remembers. Newer rollouts may carry no `user_message` event at all: when
|
||||
that command returns nothing, fall back to `response_item` messages with
|
||||
`role:"user"` (see Human-typed prompt below) and confirm against the first
|
||||
of those instead.
|
||||
|
||||
## Line types
|
||||
|
||||
Every line is `{timestamp, type, payload}` (some also carry `ordinal`).
|
||||
`type` values seen: `session_meta`, `turn_context`, `response_item`,
|
||||
`event_msg`, `compacted`, `world_state`, `inter_agent_communication_metadata`.
|
||||
|
||||
| What you want | Where it is |
|
||||
|---|---|
|
||||
| Session identity | `session_meta.payload`: `id`, `session_id`, `cwd`, `originator` (e.g. `Codex Desktop`), `cli_version`, `model_provider`, `thread_source`, `source`, `git` (`commit_hash`, `branch`, `repository_url`), `base_instructions.text`. |
|
||||
| Model per turn | `turn_context.payload`: `turn_id`, `model`, `effort`, `cwd`, `approval_policy`, `sandbox_policy`, `multi_agent_version`. Also `event_msg` `thread_settings_applied`. |
|
||||
| Human-typed prompt | `event_msg` with `payload.type=="user_message"`: `payload.message`. When that returns nothing — seen on `thread_source: "user"` Codex Desktop rollouts at `cli_version 0.149.0-alpha.4.1`, and on subagent rollouts — fall back to `response_item` messages with `payload.role=="user"`, text in `payload.content[0].text`. `role:"developer"` messages are injected boilerplate, not typed, and so is any fallback text that begins with a tag such as `<subagent_notification>`, `<environment_context>`, `<skill>` or `<recommended_plugins>`. On a subagent rollout the fallback text is the parent agent's dispatch prompt, not your human partner's. |
|
||||
| Assistant text | `event_msg` `agent_message` (`payload.message`, `payload.phase`) or `response_item` `message` with `role:"assistant"`. |
|
||||
| Tool calls | `response_item` with `payload.type` `function_call` (`name`, `arguments`, `call_id`) or `custom_tool_call` (`name`, `input`, `call_id`); outputs are `function_call_output` / `custom_tool_call_output` matched by `call_id`. Also `event_msg` `patch_apply_end` (`success`, `changes`), `web_search_end`, `mcp_tool_call_end` (`invocation.server`, `invocation.tool`). |
|
||||
| Turn timing | `event_msg` `task_started` (`turn_id`, `started_at`, `model_context_window`) and `task_complete` (`duration_ms`, `time_to_first_token_ms`, `last_agent_message`); `turn_aborted` (`reason`, `duration_ms`). |
|
||||
| Tokens | `event_msg` `token_count`: `payload.info.total_token_usage` (cumulative; keys include `input_tokens`, `cached_input_tokens`, `output_tokens`) and `payload.rate_limits`. |
|
||||
| Compaction | a `compacted` line (`window_id`, `previous_window_id`, `replacement_history`) and an `event_msg` `context_compacted`. |
|
||||
| Subagents | `event_msg` `sub_agent_activity` (`agent_thread_id`, `agent_path`, `kind`); `response_item` `agent_message` with `author`/`recipient`; the child's own rollout file (see Where). |
|
||||
| Skill use | No attribution field. Look for `SKILL.md` in `function_call.arguments` / `custom_tool_call.input` and in `world_state`/`session_meta` instruction text. |
|
||||
| Reasoning | `response_item` `reasoning` (`summary[].text`; `encrypted_content` is opaque). |
|
||||
|
||||
## Safe extraction
|
||||
|
||||
Rollouts reach hundreds of megabytes; `compacted` lines embed whole
|
||||
histories. Never print a whole line. Check size first:
|
||||
|
||||
```bash
|
||||
F=~/.codex/sessions/YYYY/MM/DD/rollout-....jsonl
|
||||
wc -lc "$F"
|
||||
awk '{ if (length($0) > 100000) print NR, length($0) }' "$F"
|
||||
```
|
||||
|
||||
With `jq`:
|
||||
|
||||
```bash
|
||||
head -1 "$F" | jq '.payload | {id, cwd, originator, cli_version, model_provider, thread_source, git}' # identity
|
||||
jq -r '.type + "/" + (.payload.type // "")' "$F" | sort | uniq -c # census
|
||||
jq -r 'select(.type=="event_msg" and .payload.type=="user_message") | "\(input_line_number)\t\(.timestamp)\t\(.payload.message[0:160])"' "$F" # human prompts
|
||||
jq -r 'select(.type=="response_item" and .payload.type=="message" and .payload.role=="user")
|
||||
| "\(input_line_number)\t\(.timestamp)\t\((.payload.content[0].text // "")[0:160])"' "$F" # human prompts, fallback when the line above returns nothing; skip rows whose text starts with a `<tag>`
|
||||
jq -r 'select(.type=="turn_context") | "\(.timestamp)\t\(.payload.model)\t\(.payload.effort)"' "$F" # model per turn
|
||||
jq -c 'select(.type=="response_item" and (.payload.type=="function_call" or .payload.type=="custom_tool_call"))
|
||||
| {line:input_line_number, name:.payload.name, args:((.payload.arguments // .payload.input)|tostring|.[0:120])}' "$F" # tool calls
|
||||
jq -c 'select(.payload.type=="task_complete" or .payload.type=="turn_aborted") | {ts:.timestamp, type:.payload.type, ms:.payload.duration_ms}' "$F" # turn timing
|
||||
jq -c 'select(.payload.type=="token_count") | {ts:.timestamp, t:.payload.info.total_token_usage}' "$F" # tokens (cumulative)
|
||||
grep -n '"type":"compacted"\|"context_compacted"' "$F" | cut -d: -f1 # compaction line numbers
|
||||
grep -n 'SKILL\.md' "$F" | cut -d: -f1 # skill-read line numbers
|
||||
sed -n '123p' "$F" | jq -c '{ts:.timestamp, type, p:(.payload|tostring|.[0:400])}' # one line, trimmed
|
||||
```
|
||||
|
||||
Find a thread's subagent rollouts (filenames only, never content):
|
||||
|
||||
```bash
|
||||
grep -l '"parent_thread_id":"<thread-id>"' ~/.codex/sessions/*/*/*/rollout-*.jsonl
|
||||
```
|
||||
|
||||
A subagent rollout can carry no `event_msg` `user_message` at all — the
|
||||
parent agent's dispatch prompt instead shows up as a `response_item`
|
||||
`message` with `role:"user"`. If a `user_message` event is present, it is
|
||||
from the parent agent, not your human partner.
|
||||
30
skills/diagnosing-superpowers/references/other-harnesses.md
Normal file
30
skills/diagnosing-superpowers/references/other-harnesses.md
Normal file
@@ -0,0 +1,30 @@
|
||||
# Other harnesses: discover, then report what you found
|
||||
|
||||
This file is for any harness without a verified reference in this
|
||||
directory. You know your own harness better than this file does. Use that
|
||||
knowledge, and write down exactly what you found so the report reader can
|
||||
judge it.
|
||||
|
||||
## Procedure
|
||||
|
||||
1. **Ask the harness.** Many harnesses expose a session or history command
|
||||
(`<harness> session list`, `/sessions`, a "resume" picker). Use it to get
|
||||
the session id and, if shown, the file path.
|
||||
2. **Look under the harness's config directory** (`~/.<harness>/`,
|
||||
`~/.config/<harness>/`, `~/.local/share/<harness>/`) for `sessions`,
|
||||
`history`, `chats`, `threads`, or `projects` directories holding `.jsonl`
|
||||
or `.json` files.
|
||||
3. **Confirm a candidate** by extracting its first human message with a
|
||||
size-safe command (`head -c 2000`, or `jq` on the first record) and
|
||||
matching it to what your human partner remembers. Never print whole
|
||||
lines; treat every candidate like the verified stores: `wc -lc` and a
|
||||
long-line check before anything else.
|
||||
4. **Map the fields you need** by reading a handful of records with `jq -c
|
||||
'keys'` or `head -c`: human prompt, assistant text, tool call and result,
|
||||
model, harness version, timestamps, subagent linkage, compaction.
|
||||
5. **Record in the case file and the report's coverage notes**: the store
|
||||
path, the layout you inferred, which of the fields above you could and
|
||||
could not find, and your confidence. Field-level claims in the report
|
||||
are marked "inferred from the file, not a documented format".
|
||||
6. **If you cannot find the store**, say so and ask your human partner for
|
||||
the path. Do not guess a layout from another harness.
|
||||
44
skills/diagnosing-superpowers/templates/bundle-README.md
Normal file
44
skills/diagnosing-superpowers/templates/bundle-README.md
Normal file
@@ -0,0 +1,44 @@
|
||||
# Superpowers session diagnosis bundle
|
||||
|
||||
Session: <session-id>
|
||||
Harness: <name> <version> Superpowers: <version> (<sha or "not a checkout">)
|
||||
Redaction level: skeleton | evidence | full
|
||||
Built: <ISO timestamp>
|
||||
|
||||
## What this is
|
||||
|
||||
A scrubbed record of a coding-agent session in which superpowers was
|
||||
installed and something went wrong, prepared so that an agent or person
|
||||
who was not present can decide whether superpowers contributed and, if so,
|
||||
what to change. The report inside states what happened with `path:line`
|
||||
evidence. By design it contains no diagnosis of superpowers and no proposed
|
||||
fix; that is the reader's job.
|
||||
|
||||
## Files
|
||||
|
||||
- `report.md` — the diagnosis report (problem statement, verdict,
|
||||
environment, sessions, timeline, findings, involvement, coverage notes).
|
||||
- `case.md` — the case file the analysts worked from.
|
||||
- `environment.json` — machine-readable copy of the environment section.
|
||||
- `timeline.md` — the per-turn timeline.
|
||||
- `findings/<dimension>.md` — raw analyst findings per dimension.
|
||||
- `transcripts/<session-id>.md` — condensed per-turn rendering of each
|
||||
examined session (never the raw JSONL). At *skeleton* level tool-result
|
||||
bodies are replaced by `[tool result: <tool>, <bytes> bytes, exit <code>]`;
|
||||
at *evidence* level bodies are kept only for events cited in findings; at
|
||||
*full* level all bodies are kept.
|
||||
- `scrub-log.md` — every placeholder used and its category (never the
|
||||
original value).
|
||||
|
||||
## How to read it
|
||||
|
||||
Start with `report.md` §1–2, then §7 (involvement) and the evidence lines
|
||||
it cites, then the matching turns in `transcripts/`. `path:line` references
|
||||
point at the original files on the reporter's machine; the same line
|
||||
numbers are preserved in the condensed transcripts as `[L<n>]` markers.
|
||||
|
||||
## Redaction
|
||||
|
||||
Placeholders look like `<EMAIL-1>`, `<PERSON-2>`, `<SECRET-3>`, `<HOST-4>`,
|
||||
`<REPO-5>`, `<ORG-6>`, `<PROPRIETARY-7>`; home paths are rewritten to `~/…`. The same placeholder
|
||||
always refers to the same original value within this bundle.
|
||||
50
skills/diagnosing-superpowers/templates/case.md
Normal file
50
skills/diagnosing-superpowers/templates/case.md
Normal file
@@ -0,0 +1,50 @@
|
||||
# Case: <session-id>
|
||||
|
||||
Workspace: ~/.superpowers/diagnosing-superpowers/<session-id>/
|
||||
Created: <ISO timestamp>
|
||||
|
||||
## Problem statement (agreed with your human partner)
|
||||
|
||||
<One paragraph. Names the session(s), the turn range if known, what was
|
||||
expected, what happened, and the observable that matters: wall-clock,
|
||||
tokens, repeated actions, a specific unexpected action.>
|
||||
|
||||
Goal is a superpowers bug report: yes | no
|
||||
|
||||
## Sessions
|
||||
|
||||
| Role | Session id | Absolute path | Lines | Bytes | Longest line (bytes) | First prompt (first 120 chars) | First timestamp |
|
||||
|---|---|---|---|---|---|---|---|
|
||||
| main | | | | | | | |
|
||||
| subagent | | | | | | | |
|
||||
|
||||
Rejected candidates: <id — path — why rejected>, or "none".
|
||||
|
||||
Session still running at read time: yes | no (mtime <ISO>, lines <N>)
|
||||
|
||||
## Environment
|
||||
|
||||
- OS: <name and version>
|
||||
- Harness: <name> <version>
|
||||
- Models seen: <model id — where (main / subagent id)>
|
||||
- Superpowers install root: <path>; version <x.y.z>; git sha <sha or "not a checkout">
|
||||
- Skill files read or injected during the session:
|
||||
|
||||
| File (relative to install root) | sha1 (current file) | mtime newer than session? |
|
||||
|---|---|---|
|
||||
|
||||
- Other plugins / extensions / MCP servers configured: <list, or "none found">
|
||||
- Instruction files present (paths only): <list>
|
||||
|
||||
## Context-safety rules for every reader of these files
|
||||
|
||||
- Check `wc -lc` and long lines (`awk '{ if (length($0) > 100000) print NR, length($0) }'`) before reading.
|
||||
- Never `cat` or `grep` for content. Line numbers and counts first
|
||||
(`grep -n … | cut -d: -f1`), then small fields from specific lines
|
||||
(`sed -n Np | jq -c '{…}'` or `| cut -c1-500`).
|
||||
- Read-only: never modify, move, or delete a session file.
|
||||
- In a subagent transcript, "user" is the parent agent.
|
||||
|
||||
## Harness reference to use
|
||||
|
||||
<references/claude-code-sessions.md | references/codex-sessions.md | references/other-harnesses.md>
|
||||
49
skills/diagnosing-superpowers/templates/issue.md
Normal file
49
skills/diagnosing-superpowers/templates/issue.md
Normal file
@@ -0,0 +1,49 @@
|
||||
- [x] I searched existing issues and this is not a duplicate (searched: <query terms>; closest: <#n title, or "none">)
|
||||
|
||||
## Environment (required)
|
||||
|
||||
| Field | Value |
|
||||
|-------|-------|
|
||||
| Superpowers version | <version> (<sha or "not a checkout">) |
|
||||
| Harness (Claude Code, Cursor, etc.) | <harness> |
|
||||
| Harness version | <version> |
|
||||
| Your model + version | <model ids seen> |
|
||||
| All plugins installed | <list> |
|
||||
| OS + shell | <os version>, <shell> |
|
||||
|
||||
## Is this a Superpowers issue or a platform issue?
|
||||
|
||||
- [ ] I confirmed this issue does not occur without Superpowers installed
|
||||
|
||||
Not reproduced without superpowers. Evidence for involvement is below;
|
||||
the reporter has not established cause.
|
||||
|
||||
## What happened?
|
||||
|
||||
<Problem statement, then the triage verdict, with `path:line` citations
|
||||
rewritten as `transcript line <n>`.>
|
||||
|
||||
## Steps to reproduce
|
||||
|
||||
1. <first human prompt, scrubbed>
|
||||
2. <the turns leading to the problem, one line each>
|
||||
3. <the observable>
|
||||
|
||||
## Expected behavior
|
||||
|
||||
<from the problem statement>
|
||||
|
||||
## Actual behavior
|
||||
|
||||
<from the triage verdict>
|
||||
|
||||
## Debug log or conversation transcript
|
||||
|
||||
Session id(s): <ids>. A scrubbed bundle (redaction level: <level>) is
|
||||
attached to this issue by the reporter, or available on request.
|
||||
Superpowers involvement per the diagnosis report: <possible | likely>, with
|
||||
evidence at <transcript lines>. This report does not propose a fix.
|
||||
|
||||
---
|
||||
Filed with the `diagnosing-superpowers` skill. Model, harness, harness
|
||||
version, and installed plugins are listed above.
|
||||
78
skills/diagnosing-superpowers/templates/report.md
Normal file
78
skills/diagnosing-superpowers/templates/report.md
Normal file
@@ -0,0 +1,78 @@
|
||||
# Session diagnosis: <session-id>
|
||||
|
||||
Report path: ~/.superpowers/diagnosing-superpowers/<session-id>/report.md
|
||||
Written: <ISO timestamp>
|
||||
|
||||
## 1. Problem statement (REQUIRED)
|
||||
|
||||
<Copied from the case file.>
|
||||
|
||||
## 2. Triage verdict (REQUIRED)
|
||||
|
||||
<What the evidence shows happened around the reported problem. Prose, with
|
||||
`path:line` after every claim. State confidence: high / medium / low, and
|
||||
what would raise it. No statement about what superpowers should do.>
|
||||
|
||||
## 3. Environment (REQUIRED)
|
||||
|
||||
- OS:
|
||||
- Harness and version:
|
||||
- Models seen:
|
||||
- Superpowers install root / version / git sha:
|
||||
- Skill files read or injected (sha1 table from the case file):
|
||||
- Other plugins, extensions, MCP servers:
|
||||
- Instruction files present (paths only):
|
||||
|
||||
## 4. Sessions examined (REQUIRED)
|
||||
|
||||
| Role | Session id | Absolute path | Lines | Bytes |
|
||||
|---|---|---|---|---|
|
||||
|
||||
Rejected candidates: <id — path — why>, or "none".
|
||||
|
||||
## 5. Timeline (REQUIRED)
|
||||
|
||||
One row per human-typed prompt. Events column lists skills invoked,
|
||||
subagents dispatched, compaction, errors, resumes, aborts.
|
||||
|
||||
| Turn | Line | Time | Request (one line) | Events |
|
||||
|---|---|---|---|---|
|
||||
|
||||
## 6. Findings (REQUIRED, one subsection per dimension)
|
||||
|
||||
Each finding:
|
||||
```
|
||||
- finding: <one sentence>
|
||||
evidence: <path:line> — "<short quote>"
|
||||
turns: <first>–<last>
|
||||
confidence: high | medium | low
|
||||
```
|
||||
A dimension with nothing to report says `none found — checked: <what was checked>`.
|
||||
|
||||
### 6.1 Skill timeline
|
||||
### 6.2 Plan adherence
|
||||
### 6.3 Repeated work
|
||||
### 6.4 Stumbles
|
||||
### 6.5 Quality evidence
|
||||
### 6.6 Request conflicts
|
||||
### 6.7 Cost and time
|
||||
### 6.8 Other plugins and skills used
|
||||
|
||||
## 7. Superpowers involvement (REQUIRED)
|
||||
|
||||
not indicated | possible | likely
|
||||
|
||||
Evidence lines: <path:line list>. This section states involvement only. It
|
||||
does not name a defect and does not propose a change.
|
||||
|
||||
## 8. Coverage notes (REQUIRED)
|
||||
|
||||
- Not read: <ranges, files, and why>
|
||||
- Harness features unavailable: <list or none>
|
||||
- Session was in progress at read time: yes/no
|
||||
- For your human partner to double-check: <list or none>
|
||||
|
||||
## 9. Similar sessions (only when requested)
|
||||
|
||||
| Session id | Path | Date | Harness | Matched | Did not match |
|
||||
|---|---|---|---|---|---|
|
||||
@@ -174,6 +174,29 @@ git worktree remove "$WORKTREE_PATH"
|
||||
git worktree prune # Self-healing: clean up any stale registrations
|
||||
```
|
||||
|
||||
**If removal is refused** (`contains modified or untracked files`): the
|
||||
worktree holds files that exist nowhere else — uncommitted plans, notes,
|
||||
or scratch work. Never `--force` on your own initiative. Show your human
|
||||
partner what is at stake and ask:
|
||||
|
||||
```bash
|
||||
git -C "$WORKTREE_PATH" status --porcelain -uall
|
||||
```
|
||||
|
||||
```
|
||||
Worktree removal refused — these files were never committed:
|
||||
|
||||
<file list>
|
||||
|
||||
1. Commit them to <branch> before cleanup
|
||||
2. Move them into <main repo root>
|
||||
3. Delete them (unrecoverable)
|
||||
|
||||
Which?
|
||||
```
|
||||
|
||||
Carry out the choice, then remove the worktree.
|
||||
|
||||
**Otherwise:** The host environment owns this workspace — leave it in
|
||||
place. If your platform provides a workspace-exit tool, use it.
|
||||
|
||||
@@ -196,6 +219,7 @@ place. If your platform provides a workspace-exit tool, use it.
|
||||
| "'Yeah, get rid of it' counts as confirmation" | Only the typed word `discard` authorizes deletion. |
|
||||
| "The PR is up, so the worktree is clutter now" | PR feedback gets fixed in that worktree. It stays until the work lands. |
|
||||
| "This other worktree looks stale — I'll clean it too" | Clean up only worktrees under `.worktrees/` or `worktrees/`. Everything else belongs to the host. |
|
||||
| "Removal refused — `--force` is just finishing the cleanup" | The refusal means files exist only in that worktree. `--force` destroys them permanently. Show your human partner and ask. |
|
||||
| "The merged-result failure is probably flaky" | A failing merged result stops everything. Branch and worktree stay put while you investigate. |
|
||||
| "The base branch is obviously main" | Confirm the fork point or ask. Merging into the wrong base is expensive to undo. |
|
||||
| "The push was rejected — force-push will fix it" | A rejected push means the remote moved. Investigate; force-push only on your human partner's explicit request. |
|
||||
|
||||
@@ -34,6 +34,15 @@ Subagent (general-purpose):
|
||||
|
||||
Your review is read-only on this checkout. Do not mutate the working tree, the index, HEAD, or branch state in any way. Use tools like `git show`, `git diff`, and `git log` to inspect history. If you need a working copy of a different revision, check it out into a separate temporary directory (e.g. `git worktree add /tmp/review-[SHA] [SHA]`) — never move HEAD on this checkout.
|
||||
|
||||
## You Do Not Dispatch Subagents
|
||||
|
||||
Do all of this review yourself. Never spawn a subagent to review part
|
||||
of the diff, and never spawn another reviewer for a second opinion.
|
||||
This process already provides every review seat the work gets; a
|
||||
reviewer you spawn duplicates one of them at full cost, and its
|
||||
verdict counts for nothing. If the diff feels too large for one
|
||||
pass, review it in passes yourself and say so in your report.
|
||||
|
||||
## What to Check
|
||||
|
||||
**Plan alignment:**
|
||||
|
||||
@@ -14,7 +14,21 @@ Execute plan by dispatching a fresh implementer subagent per task, a task review
|
||||
**Narration:** between tool calls, narrate at most one short line — the
|
||||
ledger and the tool results carry the record.
|
||||
|
||||
**Continuous execution:** Do not pause to check in with your human partner between tasks. Execute all tasks from the plan without stopping. The only reasons to stop are: BLOCKED status you cannot resolve, ambiguity that genuinely prevents progress, or all tasks complete. "Should I continue?" prompts and progress summaries waste their time — they asked you to execute the plan, so execute it.
|
||||
**Continuous execution:** Do not pause to check in with your human partner between tasks. Execute all tasks from the plan without stopping. The only reasons to stop are the four named below, or all tasks complete. "Should I continue?" prompts and progress summaries waste their time — they asked you to execute the plan, so execute it.
|
||||
|
||||
**Rulings, not stalls.** A running plan does not wait on a human. Conflicts,
|
||||
ambiguities, plan defects, a cap you would have asked to exceed — decide
|
||||
them. The spec is the binding authority, the plan is its argument, and your
|
||||
judgment settles what neither answers. Record every decision in the ledger as
|
||||
`Ruling: <what you decided> — <why> — <what it costs if wrong>`, and keep
|
||||
going. A wrong ruling costs rework your human partner can see and undo; a
|
||||
session parked on a question costs their whole day and buys nothing.
|
||||
|
||||
Four things stop you, and only these: an irreversible or destructive
|
||||
operation; a security-sensitive action; a side effect outside this worktree
|
||||
that norms say you ask about first (a merge, a push to a shared branch, a
|
||||
publish); and a plan so broken that every path forward is a guess. For those,
|
||||
stop and ask.
|
||||
|
||||
## When to Use
|
||||
|
||||
@@ -57,14 +71,14 @@ digraph process {
|
||||
"Generate review package, dispatch task reviewer (./task-reviewer-prompt.md)" [shape=box];
|
||||
"Spec ✅ and quality approved?" [shape=diamond];
|
||||
"Finding conflicts with plan text?" [shape=diamond];
|
||||
"Ask human partner which governs" [shape=box];
|
||||
"Rule on the conflict, ledger the ruling" [shape=box];
|
||||
"Fix round R of 5: R≤3 resume implementer; R≥4 fresh implementer, more capable model" [shape=box];
|
||||
"Dispatch scoped re-review (./re-review-prompt.md)" [shape=box];
|
||||
"All findings addressed?" [shape=diamond];
|
||||
"R = 5?" [shape=diamond];
|
||||
"Adjudicate each open finding" [shape=box];
|
||||
"Any load-bearing finding?" [shape=diamond];
|
||||
"STOP: report BLOCKED to human partner" [shape=box];
|
||||
"Rule and continue; stop only if every path forward is a guess" [shape=box];
|
||||
"Park findings in ledger with rulings" [shape=box];
|
||||
"Append completion to ledger, mark todo complete" [shape=box];
|
||||
}
|
||||
@@ -85,8 +99,8 @@ digraph process {
|
||||
"Generate review package, dispatch task reviewer (./task-reviewer-prompt.md)" -> "Spec ✅ and quality approved?";
|
||||
"Spec ✅ and quality approved?" -> "Append completion to ledger, mark todo complete" [label="yes"];
|
||||
"Spec ✅ and quality approved?" -> "Finding conflicts with plan text?" [label="no"];
|
||||
"Finding conflicts with plan text?" -> "Ask human partner which governs" [label="yes"];
|
||||
"Ask human partner which governs" -> "Fix round R of 5: R≤3 resume implementer; R≥4 fresh implementer, more capable model";
|
||||
"Finding conflicts with plan text?" -> "Rule on the conflict, ledger the ruling" [label="yes"];
|
||||
"Rule on the conflict, ledger the ruling" -> "Fix round R of 5: R≤3 resume implementer; R≥4 fresh implementer, more capable model";
|
||||
"Finding conflicts with plan text?" -> "Fix round R of 5: R≤3 resume implementer; R≥4 fresh implementer, more capable model" [label="no"];
|
||||
"Fix round R of 5: R≤3 resume implementer; R≥4 fresh implementer, more capable model" -> "Dispatch scoped re-review (./re-review-prompt.md)";
|
||||
"Dispatch scoped re-review (./re-review-prompt.md)" -> "All findings addressed?";
|
||||
@@ -95,7 +109,7 @@ digraph process {
|
||||
"R = 5?" -> "Fix round R of 5: R≤3 resume implementer; R≥4 fresh implementer, more capable model" [label="no - next round"];
|
||||
"R = 5?" -> "Adjudicate each open finding" [label="yes - breaker trips"];
|
||||
"Adjudicate each open finding" -> "Any load-bearing finding?";
|
||||
"Any load-bearing finding?" -> "STOP: report BLOCKED to human partner" [label="yes"];
|
||||
"Any load-bearing finding?" -> "Rule and continue; stop only if every path forward is a guess" [label="yes"];
|
||||
"Any load-bearing finding?" -> "Park findings in ledger with rulings" [label="no"];
|
||||
"Park findings in ledger with rulings" -> "Append completion to ledger, mark todo complete";
|
||||
"Append completion to ledger, mark todo complete" -> "More tasks remain?";
|
||||
@@ -133,10 +147,6 @@ a ledger file, not only in todos.
|
||||
plan's progress: leave it in place and start your own, fresh.
|
||||
- Create the ledger with its identity as the first line:
|
||||
`# SDD ledger — plan: <plan file path>`.
|
||||
- During that same Setup read, copy the plan's Global Constraints section
|
||||
verbatim into `<workspace>/constraints.md`. The global-constraints block
|
||||
you hand reviewers pastes from that file — the plan itself stays closed
|
||||
after Setup, even across compaction.
|
||||
- The ledger is your recovery map: the commits it names exist in git even
|
||||
when your context no longer remembers creating them. After compaction,
|
||||
trust the ledger and `git log` over your own recollection.
|
||||
@@ -144,23 +154,32 @@ a ledger file, not only in todos.
|
||||
that happens, recover from `git log`.
|
||||
|
||||
Read the plan once, note its context and Global Constraints, and create a
|
||||
todo per task. That is the plan's one full read for the whole session:
|
||||
after Setup, the ledger and `scripts/task-brief` extracts are your working
|
||||
memory — re-reading the plan or spec late in the run (to "double-check"
|
||||
completion, to rebuild the final-review dispatch) re-buys context you
|
||||
already paid for and is forbidden.
|
||||
todo per task. If the plan names a Spec, read that too: the spec is the
|
||||
authority the plan argues from, and conflicts inside the plan resolve
|
||||
against it. A plan with no reachable spec gets a ledger note saying so —
|
||||
rulings made without one are provisional.
|
||||
|
||||
Before dispatching Task 1, scan the plan once for conflicts:
|
||||
Before dispatching Task 1, scan the plan once for conflicts, writing down
|
||||
what you checked as you check it:
|
||||
|
||||
- tasks that contradict each other or the plan's Global Constraints
|
||||
- anything the plan explicitly mandates that the review rubric treats as a
|
||||
defect (a test that asserts nothing, verbatim duplication of a logic block)
|
||||
|
||||
Present everything you find to your human partner as one batched question —
|
||||
each finding beside the plan text that mandates it, asking which governs —
|
||||
before execution begins, not one interrupt per discovery mid-plan. If the
|
||||
scan is clean, proceed without comment. The review loop remains the net for
|
||||
conflicts that only emerge from implementation.
|
||||
The scan's output is a table, not a verdict. One row for every pair of tasks
|
||||
that share a file or an interface: the two tasks, what one produces against
|
||||
what the other consumes, and what you found. One row for every task: whether
|
||||
its own text agrees with itself — the tests it specifies against the code it
|
||||
specifies, the files it creates against the files it later touches. "The scan
|
||||
is clean" without those rows is not a scan you ran.
|
||||
|
||||
Write the table to the ledger. Rule on everything you find before execution
|
||||
begins — each finding against the plan text that mandates it — and record
|
||||
each ruling in the ledger. If the scan is clean, proceed without comment.
|
||||
Rule on each conflict it surfaces — the spec is the binding authority, the
|
||||
plan is its argument — record the ruling beside its row, and dispatch
|
||||
Task 1. The review loop remains the net for conflicts that only emerge from
|
||||
implementation.
|
||||
|
||||
## Model Selection
|
||||
|
||||
@@ -201,12 +220,28 @@ that implementer. Single-file mechanical fixes also take the cheapest tier.
|
||||
|
||||
## The Task Loop
|
||||
|
||||
**Batch small same-shape work.** When the plan lists several tasks that are
|
||||
each a small, independent edit of the same kind — the same one-line fix,
|
||||
constant change, or field addition repeated across files — do not dispatch
|
||||
one subagent per task. Compose ONE dispatch brief listing every file and
|
||||
its change, send the whole batch to a single subagent, and review its diff
|
||||
as one unit. Reserve one-dispatch-per-task for work that needs its own
|
||||
judgment, its own tests, or its own review surface.
|
||||
|
||||
Everything you paste into a dispatch prompt — and everything a subagent
|
||||
prints back — stays resident in your context for the rest of the session
|
||||
and is re-read on every later turn. Hand artifacts over as files. The same
|
||||
tax applies to your own words: checkpoint in one short line, keep
|
||||
bookkeeping in the ledger file, and never paste back into the conversation
|
||||
what a file already holds.
|
||||
and is re-read on every later turn. Hand artifacts over as files.
|
||||
|
||||
**Waiting on dispatched subagents:** never poll a wait interface with
|
||||
short timeouts, and never sit in one silent, open-ended wait either.
|
||||
While you have local work — ledger updates, packaging the next review,
|
||||
reading reports — keep working; child results arrive on their own.
|
||||
When you are genuinely idle, wait in bounded stretches (five to ten
|
||||
minutes, where your platform allows), and between stretches post one
|
||||
line of status and reconcile your live children: list them, and chase
|
||||
any that finished without reporting. A bounded stretch keeps nearly
|
||||
all of a long wait's efficiency while guaranteeing a stuck or lost
|
||||
child is noticed within minutes, not at the end of the session.
|
||||
|
||||
### 1. Dispatch the implementer
|
||||
|
||||
@@ -234,6 +269,12 @@ and fix-round diffs need it.
|
||||
later dispatches — a real session's dispatch hit 42k chars of which 99%
|
||||
was pasted history. A fresh subagent needs its task, the interfaces it
|
||||
touches, and the global constraints. Nothing else.
|
||||
- The dispatch carries the no-subagents contract (it is in the
|
||||
implementer template): the implementer never dispatches subagents —
|
||||
not helpers, and never a reviewer. Review arrives from you, after the
|
||||
report. In real sessions, every reviewer a worker spawned duplicated
|
||||
the task review the controller dispatched anyway — a full extra
|
||||
review seat per task.
|
||||
- If an earlier task parked a finding in the area this task touches, carry
|
||||
a pointer to that ledger entry in the dispatch.
|
||||
- Record the implementer's agent identity from the dispatch result —
|
||||
@@ -256,7 +297,7 @@ Implementer subagents report one of four statuses. Handle each appropriately:
|
||||
1. If it's a context problem, provide more context and re-dispatch with the same model
|
||||
2. If the task requires more reasoning, re-dispatch with a more capable model
|
||||
3. If the task is too large, break it into smaller pieces
|
||||
4. If the plan itself is wrong, escalate to the human
|
||||
4. If the plan itself is wrong, rule on the correction, ledger it, and re-dispatch with the ruling carried in the dispatch
|
||||
|
||||
**Never** ignore an escalation or force the same model to retry without changes. If the implementer said it's stuck, something needs to change.
|
||||
|
||||
@@ -323,10 +364,11 @@ Before the loop starts, two routes leave it immediately:
|
||||
before merge. A roll-up nobody reads is a silent discard. Minor findings
|
||||
never enter the loop.
|
||||
- A finding labeled plan-mandated — or any finding that conflicts with
|
||||
what the plan's text requires — is the human's decision, like any plan
|
||||
contradiction: present the finding and the plan text, ask which governs.
|
||||
Do not dismiss the finding because the plan mandates it, and do not
|
||||
dispatch a fix that contradicts the plan without asking.
|
||||
what the plan's text requires — is yours to rule on: weigh the finding
|
||||
against the plan text, decide with the spec as the binding authority, and
|
||||
ledger the ruling before you act on it. Do not dismiss the finding because
|
||||
the plan mandates it, and do not dispatch a fix that contradicts the plan
|
||||
without a recorded ruling.
|
||||
Everything else enters the loop. A fix round is one fix dispatch plus one
|
||||
scoped re-review. Five rounds maximum per task:
|
||||
|
||||
@@ -371,15 +413,16 @@ dispatching. Adjudicate each open finding yourself — you hold the plan and
|
||||
the cross-task context the reviewer lacks:
|
||||
|
||||
- **The reviewer is wrong, or the point is contestable:** park it —
|
||||
`Task <N>: parked — <finding> — ruling: <why the code stands>`. The final
|
||||
`Task <N>: parked — <finding> — Ruling: <why the code stands>`. The final
|
||||
review sees both sides.
|
||||
- **Real, but nothing downstream builds on it:** park it the same way, with
|
||||
a ruling that says it's real and deferred.
|
||||
- **Real and load-bearing** — a later task builds on it, or it reveals a
|
||||
plan defect: STOP. Append `Task <N>: BLOCKED — <reason>` and report to
|
||||
your human partner with the finding, the plan text it collides with, and
|
||||
the fix history. Parking a structural failure lets every dependent task
|
||||
build on it and hands the final review a problem it cannot fix either.
|
||||
plan defect: rule on the smallest change that unblocks the dependent work,
|
||||
ledger it as `Task <N>: Ruling: <finding> — <what you decided and why>`,
|
||||
and carry it into the next task's dispatch. Parking a structural failure
|
||||
silently lets every dependent task build on it. Stop only when the defect
|
||||
leaves every path forward a guess.
|
||||
|
||||
Adjudicate only at the cap. Adjudicating earlier to end a loop is
|
||||
pre-judging with a different name. Every adjudication is a ledger entry —
|
||||
@@ -410,10 +453,7 @@ on the most capable available model (see Model Selection), using
|
||||
superpowers:requesting-code-review's
|
||||
[code-reviewer.md](../requesting-code-review/code-reviewer.md). Point it at
|
||||
the ledger's deferred-minor and parked lines so it can triage which must be
|
||||
fixed before merge. Build that dispatch from the ledger alone — the
|
||||
completion lines, parked rulings, and deferred minors are the whole-run
|
||||
summary; do not re-read the plan, the spec, or per-task reports to
|
||||
reconstruct what the ledger already states.
|
||||
fixed before merge.
|
||||
|
||||
If the final whole-branch review returns findings, dispatch ONE fix subagent
|
||||
with the complete findings list — not one fixer per finding.
|
||||
@@ -423,12 +463,22 @@ Then run exactly one scoped re-review of the fix wave
|
||||
(`scripts/review-package PLAN_FILE FIX_BASE HEAD` over the fix range,
|
||||
[re-review-prompt.md](re-review-prompt.md)).
|
||||
Adjudicate any residual findings as in the task loop's breaker: park with
|
||||
rulings, or stop on load-bearing ones. There is no second fix wave —
|
||||
rulings, or rule on the load-bearing ones and ledger what you decided. Only
|
||||
the four classes above stop you here. There is no second fix wave —
|
||||
residual load-bearing findings surface to your human partner when
|
||||
finishing-a-development-branch presents the options.
|
||||
|
||||
## Finish
|
||||
|
||||
Before you delete anything, collect every ledger line containing `Ruling:` —
|
||||
preflight rulings, parked findings, breaker adjudications, all of them — into
|
||||
your final message under "Rulings I made", in the order you made them, each
|
||||
with what it costs if wrong. The list is exhaustive: if the ledger holds a
|
||||
ruling, the list holds it. That list is the only place the decisions you
|
||||
took on your human partner's behalf reach them — they read it and rework
|
||||
whatever you got wrong. A ruling that dies with the workspace was a decision
|
||||
made in secret.
|
||||
|
||||
When the final whole-branch review is clean and its fixes are merged,
|
||||
delete this plan's workspace (`rm -rf <workspace>`) — the git history is
|
||||
the record now. Sibling directories belong to other plans; leave them
|
||||
@@ -448,6 +498,7 @@ Use superpowers:finishing-a-development-branch.
|
||||
| "The fix was small, skip the re-review" | Unreviewed fixes are how regressions land. Every round ends with a scoped re-review. |
|
||||
| "Reviews slow the loop down" | The loop without reviews is just unverified churn. Reviews are the loop's brakes and steering. |
|
||||
| "Ledger bookkeeping is overhead" | The ledger is what survives compaction. Controllers without one have re-dispatched entire completed task sequences. |
|
||||
| "The implementer spawned its own reviewer — free extra assurance" | It's a duplicate seat reviewing the same diff; the task review is the gate. A worker-spawned reviewer is a defect to flag, not rigor. |
|
||||
|
||||
## Example Workflow
|
||||
|
||||
|
||||
@@ -47,6 +47,18 @@ Subagent (general-purpose):
|
||||
While iterating, run the focused test for what you're changing; run the
|
||||
full suite once before committing, not after every edit.
|
||||
|
||||
## You Do Not Dispatch Subagents
|
||||
|
||||
Do all of this task's work yourself. Never spawn a subagent to
|
||||
implement part of the task, and above all never spawn a reviewer to
|
||||
check your work. Self-review (below) means reading your own diff.
|
||||
Review is the controller's job: after you report, it dispatches a
|
||||
fresh reviewer against your diff. A reviewer you spawn duplicates
|
||||
that review at full cost, and its approval counts for nothing in
|
||||
the process. If you catch yourself thinking "an independent review
|
||||
would strengthen my report" — that review is already scheduled.
|
||||
Report instead.
|
||||
|
||||
## Code Organization
|
||||
|
||||
You reason best about code you can hold in context at once, and your edits are more
|
||||
|
||||
@@ -43,6 +43,15 @@ Subagent (general-purpose):
|
||||
Your review is read-only on this checkout. Do not mutate the working
|
||||
tree, the index, HEAD, or branch state in any way.
|
||||
|
||||
## You Do Not Dispatch Subagents
|
||||
|
||||
Do all of this review yourself. Never spawn a subagent to review part
|
||||
of the diff, and never spawn another reviewer for a second opinion.
|
||||
This process already provides every review seat the work gets; a
|
||||
reviewer you spawn duplicates one of them at full cost, and its
|
||||
verdict counts for nothing. If the diff feels too large for one
|
||||
pass, review it in passes yourself and say so in your report.
|
||||
|
||||
## Scope
|
||||
|
||||
Your scope is the findings list and the fix diff. Verdict every finding.
|
||||
|
||||
@@ -52,6 +52,15 @@ Subagent (general-purpose):
|
||||
Your review is read-only on this checkout. Do not mutate the working
|
||||
tree, the index, HEAD, or branch state in any way.
|
||||
|
||||
## You Do Not Dispatch Subagents
|
||||
|
||||
Do all of this review yourself. Never spawn a subagent to review part
|
||||
of the diff, and never spawn another reviewer for a second opinion.
|
||||
This process already provides every review seat the work gets; a
|
||||
reviewer you spawn duplicates one of them at full cost, and its
|
||||
verdict counts for nothing. If the diff feels too large for one
|
||||
pass, review it in passes yourself and say so in your report.
|
||||
|
||||
## Do Not Trust the Report
|
||||
|
||||
Treat the implementer's report as unverified claims about the code. It
|
||||
@@ -75,6 +84,13 @@ Subagent (general-purpose):
|
||||
Warnings or other noise in the implementer's reported test output are
|
||||
findings — test output should be pristine.
|
||||
|
||||
Evidence you cannot see is not evidence that doesn't exist. If the
|
||||
report or its test evidence looks truncated, or you cannot locate the
|
||||
results it claims, re-read the file at its stated path — and if it is
|
||||
genuinely missing or garbled, report that as a gap for the controller.
|
||||
Re-running the suite to regenerate what you failed to read is not
|
||||
verification; illegibility of the evidence is not invalidation of it.
|
||||
|
||||
## Part 1: Spec Compliance
|
||||
|
||||
Compare the diff against What Was Requested:
|
||||
@@ -86,6 +102,12 @@ Subagent (general-purpose):
|
||||
- **Misunderstood:** right feature built the wrong way, wrong problem
|
||||
solved
|
||||
|
||||
If the brief lists several files each with its own change (a batched
|
||||
dispatch), check the diff against that list file by file: every listed
|
||||
file must have its corresponding hunk. A listed file the diff never
|
||||
touches is a Missing finding, no matter how clean the rest of the
|
||||
batch looks.
|
||||
|
||||
If a requirement cannot be verified from this diff alone (it lives in
|
||||
unchanged code or spans tasks), report it as a ⚠️ item instead of
|
||||
broadening your search.
|
||||
|
||||
@@ -18,9 +18,18 @@ echo "🔍 Searching for test that creates: $POLLUTION_CHECK"
|
||||
echo "Test pattern: $TEST_PATTERN"
|
||||
echo ""
|
||||
|
||||
# Get list of test files
|
||||
TEST_FILES=$(find . -path "$TEST_PATTERN" | sort)
|
||||
TOTAL=$(echo "$TEST_FILES" | wc -l | tr -d ' ')
|
||||
# Get list of test files (find . emits ./-prefixed paths, so accept the
|
||||
# pattern written with or without a leading ./)
|
||||
TEST_PATTERN="${TEST_PATTERN#./}"
|
||||
# find -path can't match '**/' against zero directory levels, so a pattern
|
||||
# like src/**/*.test.ts would skip src/top.test.ts; also try the pattern
|
||||
# with '**/' collapsed to cover files directly under the base directory.
|
||||
TEST_FILES=$(find . \( -path "./$TEST_PATTERN" -o -path "./${TEST_PATTERN//\*\*\//}" \) | sort -u)
|
||||
if [ -z "$TEST_FILES" ]; then
|
||||
TOTAL=0
|
||||
else
|
||||
TOTAL=$(printf '%s\n' "$TEST_FILES" | wc -l | tr -d ' ')
|
||||
fi
|
||||
|
||||
echo "Found $TOTAL test files"
|
||||
echo ""
|
||||
|
||||
@@ -56,6 +56,7 @@ If your harness appears here, read its reference file for special instructions:
|
||||
- Codex: `references/codex-tools.md`
|
||||
- Pi: `references/pi-tools.md`
|
||||
- Antigravity: `references/antigravity-tools.md`
|
||||
- Hermes Agent: `references/hermes-tools.md`
|
||||
|
||||
## User Instructions
|
||||
|
||||
|
||||
@@ -4,7 +4,7 @@ Skills speak in actions ("dispatch a subagent", "create a todo", "read a file").
|
||||
|
||||
| Action skills request | Antigravity CLI equivalent |
|
||||
|----------------------|----------------------|
|
||||
| Dispatch a subagent (`Subagent (general-purpose):` template) | `invoke_subagent` with a built-in `TypeName` — `self` for full-capability work, `research` for read-only (see [Subagent support](#subagent-support)) |
|
||||
| Dispatch a subagent (`Subagent (general-purpose):` template) | `invoke_subagent` with a built-in `TypeName` — `self` for full-capability work, `research` for read-only |
|
||||
| Task tracking ("create a todo", "mark complete") | a **task artifact** — `write_to_file` with `IsArtifact: true` and `ArtifactType: "task"` (see [Task tracking](#task-tracking)). **Not** `manage_task`, which manages background processes. |
|
||||
|
||||
## Task tracking
|
||||
|
||||
@@ -7,7 +7,76 @@ Add to your Codex config (`~/.codex/config.toml`):
|
||||
multi_agent = true
|
||||
```
|
||||
|
||||
This enables `spawn_agent`, `wait_agent`, and `close_agent` for skills like `dispatching-parallel-agents` and `subagent-driven-development`. When using subagent-driven-development, close reviewer subagents when their review returns. Keep each implementer subagent open until its task's review passes — the fix loop resumes the implementer — then close it. If your harness cannot send another message to a spawned agent, dispatch each fix round as a fresh implementer carrying the brief, the report file, and the findings.
|
||||
This enables the multi-agent tools that skills like
|
||||
`dispatching-parallel-agents` and `subagent-driven-development` use.
|
||||
Which tools you get depends on the multi-agent version your model
|
||||
preset selects (current presets run V2; older ones run V1). Trust your
|
||||
actual tool list over any table — including this one — when they
|
||||
disagree.
|
||||
|
||||
- **Spawning:** give children a clean context with
|
||||
`spawn_agent {fork_turns: "none"}`; the default `"all"` copies your
|
||||
entire transcript into the child. On Codex 0.145+, role files under
|
||||
`~/.codex/agents/` attach to isolated forks via `agent_type`.
|
||||
Full-history forks accept `model` and `reasoning_effort` overrides
|
||||
(only `agent_type` is refused there) — isolated forks are the SDD
|
||||
default for context hygiene, not because overrides require them.
|
||||
- **Fix rounds:** resume the implementer with `followup_task` — it
|
||||
delivers your message, triggers a turn, and transparently reloads a
|
||||
child the harness evicted. Never dispatch a fresh implementer on the
|
||||
theory that a spawned agent cannot be messaged again; on V2 it
|
||||
always can.
|
||||
- **Lifecycle:** V2 has no `close_agent`. Finished children are
|
||||
evicted automatically when slots are needed; leaving them unclosed
|
||||
costs nothing. Only V1 sessions have `close_agent` — there, close
|
||||
reviewers when their review returns, and close each implementer
|
||||
after its task's review passes.
|
||||
- **Model names:** never copy a model name from a skill, table, or old
|
||||
session into `spawn_agent` without checking it against your current
|
||||
spawn allowlist — V2 accepts only V2-capable presets and hard-errors
|
||||
on the rest.
|
||||
|
||||
## Waiting on children
|
||||
|
||||
`wait_agent` is an event subscription, not a poll: a long wait wakes
|
||||
the moment a child produces mailbox activity, with the same latency as
|
||||
a short one. Short-timeout polling buys nothing and costs a tool call —
|
||||
and a context rebill — per poll. In measured sessions, roughly
|
||||
two-thirds of all wait calls were short polls that timed out.
|
||||
|
||||
- While you still have local work, do not wait at all. A completed
|
||||
child's final answer is pushed into your mailbox and arrives with
|
||||
your next turn.
|
||||
- When you are genuinely idle with children outstanding, wait in
|
||||
bounded stretches: `wait_agent` with `timeout_ms` 300000-600000
|
||||
(5-10 minutes). After each stretch — wake or timeout — post one
|
||||
status line, run `list_agents`, and chase any child that finished
|
||||
without reporting. Never stack polls shorter than five minutes; the
|
||||
event subscription wakes a bounded stretch just as fast as a short
|
||||
one.
|
||||
- Completion mail cannot wake an idle controller (it is delivered
|
||||
without triggering a turn); covering that idle window is
|
||||
`wait_agent`'s only job. A stretch that times out with no activity
|
||||
is your cue to reconcile, not to shorten the next stretch.
|
||||
|
||||
## Model routing on spawns
|
||||
|
||||
Every `spawn_agent` you issue — including when you are yourself a
|
||||
spawned child running a fan-out — sets `model` AND `reasoning_effort`
|
||||
explicitly, per the Model Selection rules of the skill you are
|
||||
executing. Setting `model` alone is a trap: the child's effort
|
||||
silently resets to that model's default, not to yours.
|
||||
|
||||
Ask your human partner to add a machine-level backstop to
|
||||
`~/.codex/config.toml` so any spawn that slips through still routes to
|
||||
a deliberate tier instead of silently inheriting the session's most
|
||||
expensive model:
|
||||
|
||||
```toml
|
||||
[agents]
|
||||
default_subagent_model = "<a mid-tier model from your spawn allowlist>"
|
||||
default_subagent_reasoning_effort = "medium"
|
||||
```
|
||||
|
||||
## Environment Detection
|
||||
|
||||
|
||||
56
skills/using-superpowers/references/hermes-tools.md
Normal file
56
skills/using-superpowers/references/hermes-tools.md
Normal file
@@ -0,0 +1,56 @@
|
||||
# Hermes Agent Tool Mapping
|
||||
|
||||
Skills speak in actions ("dispatch a subagent", "create a todo", "read a file"). On Hermes Agent these resolve to the tools below.
|
||||
|
||||
## Tools
|
||||
|
||||
| Action skills request | Hermes tool |
|
||||
|---|---|
|
||||
| Read a file | `read_file` |
|
||||
| Create a new file | `write_file` |
|
||||
| Edit a file (targeted patch) | `patch` |
|
||||
| Run a shell command | `terminal` |
|
||||
| Search file contents | `search_files` |
|
||||
| Find files by name | `terminal` with `find` |
|
||||
| Fetch a URL / read a webpage | `web_extract(urls=[...])` |
|
||||
| Search the web | `web_search(query=...)` |
|
||||
| Dispatch a subagent | `delegate_task(goal=..., context=..., toolsets=[...], role="leaf")` |
|
||||
| Task tracking | `todo` tool |
|
||||
| Invoke a skill | `skill_view("skill-name")` |
|
||||
|
||||
## Instructions file
|
||||
|
||||
When a skill mentions "your instructions file," on Hermes Agent this is **`AGENTS.md`** in the project directory, or **`SOUL.md`** globally at `~/.hermes/SOUL.md`.
|
||||
|
||||
## Invoking a skill
|
||||
|
||||
Hermes Agent has a `skills` toolset with `skill_view` and `skills_list` tools.
|
||||
To invoke a superpowers skill, use:
|
||||
|
||||
```
|
||||
skill_view("brainstorming")
|
||||
skill_view("test-driven-development")
|
||||
```
|
||||
|
||||
If `skill_view` cannot find a superpowers skill (it may not appear in the catalog
|
||||
until the plugin fully registers it), fall back to reading the SKILL.md directly:
|
||||
|
||||
```
|
||||
read_file(path="~/.hermes/plugins/superpowers/skills/<skill-name>/SKILL.md")
|
||||
```
|
||||
|
||||
This fallback is the same mechanism used by other harnesses without native skill loading.
|
||||
|
||||
## Subagent dispatch
|
||||
|
||||
Use `delegate_task` to spawn isolated subagents for parallel or sequential workstreams:
|
||||
|
||||
```
|
||||
delegate_task(goal="...", context="...", toolsets=[...], role="leaf")
|
||||
```
|
||||
|
||||
If `delegate_task` is unavailable, do the work inline rather than inventing tool calls.
|
||||
|
||||
## Task tracking
|
||||
|
||||
Use the `todo` tool for task tracking within a session. For multi-agent task boards, use `hermes kanban` CLI if available. Treat older `TodoWrite` references as the task-tracking action.
|
||||
@@ -66,16 +66,15 @@ independently testable deliverable.
|
||||
|
||||
**Tech Stack:** [Key technologies/libraries]
|
||||
|
||||
**Spec:** [path to the spec/design doc this plan implements — the plan
|
||||
argues from the spec, so the spec travels with it; executors read both]
|
||||
|
||||
## Global Constraints
|
||||
|
||||
[The spec's project-wide requirements — version floors, dependency limits,
|
||||
naming and copy rules, platform requirements — one line each, with exact
|
||||
values copied verbatim from the spec. Every task's requirements implicitly
|
||||
include this section. Three things may never enter it: cosmetic absolutes
|
||||
on every commit (a fixed trailer or byline the work does not need), your
|
||||
own identity or model name promoted into a rule, and environment
|
||||
constraints (versions, platforms, paths) you have not verified against the
|
||||
environment the plan will execute in.]
|
||||
include this section.]
|
||||
|
||||
---
|
||||
```
|
||||
@@ -129,8 +128,6 @@ git commit -m "feat: add specific feature"
|
||||
```
|
||||
````
|
||||
|
||||
Commit messages describe the change. Never mandate session boilerplate — trailers, bylines, model names — as a per-commit rule; what your session stamps on its commits is not a requirement of the work.
|
||||
|
||||
## No Placeholders
|
||||
|
||||
Every step must contain the actual content an engineer needs. These are **plan failures** — never write them:
|
||||
@@ -151,8 +148,6 @@ After writing the complete plan, look at the spec with fresh eyes and check the
|
||||
|
||||
**3. Type consistency:** Do the types, method signatures, and property names you used in later tasks match what you defined in earlier tasks? A function called `clearLayers()` in Task 3 but `clearFullLayers()` in Task 7 is a bug.
|
||||
|
||||
**4. Constraint hygiene:** Does any Global Constraint mandate a per-commit cosmetic absolute, name the authoring model or session, or assert an environment fact (version floor, platform, path) you did not verify? Cut or verify it.
|
||||
|
||||
If you find issues, fix them inline. No need to re-review — just fix and move on. If you find a spec requirement with no task, add the task.
|
||||
|
||||
## Execution Handoff
|
||||
|
||||
@@ -13,9 +13,9 @@
|
||||
* Requires: graphviz (dot) installed on system
|
||||
*/
|
||||
|
||||
const fs = require('fs');
|
||||
const path = require('path');
|
||||
const { execSync } = require('child_process');
|
||||
import * as fs from 'fs';
|
||||
import * as path from 'path';
|
||||
import { execFileSync } from 'child_process';
|
||||
|
||||
function extractDotBlocks(markdown) {
|
||||
const blocks = [];
|
||||
@@ -69,7 +69,7 @@ ${bodies.join('\n\n')}
|
||||
|
||||
function renderToSvg(dotContent) {
|
||||
try {
|
||||
return execSync('dot -Tsvg', {
|
||||
return execFileSync('dot', ['-Tsvg'], {
|
||||
input: dotContent,
|
||||
encoding: 'utf-8',
|
||||
maxBuffer: 10 * 1024 * 1024
|
||||
@@ -107,9 +107,10 @@ function main() {
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
// Check if dot is available
|
||||
// Check if dot is available. Run the binary directly rather than probing
|
||||
// with `which`, which is not a command on Windows.
|
||||
try {
|
||||
execSync('which dot', { encoding: 'utf-8' });
|
||||
execFileSync('dot', ['-V'], { stdio: 'ignore' });
|
||||
} catch {
|
||||
console.error('Error: graphviz (dot) not found. Install with:');
|
||||
console.error(' brew install graphviz # macOS');
|
||||
|
||||
63
tests/devin/test-devin-plugin.sh
Executable file
63
tests/devin/test-devin-plugin.sh
Executable file
@@ -0,0 +1,63 @@
|
||||
#!/usr/bin/env bash
|
||||
# Validate the Devin CLI integration. `devin plugins install obra/superpowers`
|
||||
# reads `.devin-plugin/plugin.json` and auto-discovers the co-located `skills/`
|
||||
# directory; Devin CLI surfaces every installed skill's name + description in
|
||||
# the system prompt at session start and invokes them via its native `skill`
|
||||
# tool, and its system prompt already documents its own tools (subagent
|
||||
# profiles, todo tracking, question prompts), so there is no hook, injector,
|
||||
# or tool-mapping scaffold to test. What IS Devin-specific is the manifest.
|
||||
#
|
||||
# Mirrors tests/kimi/test-plugin-manifest.sh. CI-safe: does not require
|
||||
# `devin` installed.
|
||||
set -euo pipefail
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
REPO_ROOT="$(cd "$SCRIPT_DIR/../.." && pwd)"
|
||||
|
||||
MANIFEST="$REPO_ROOT/.devin-plugin/plugin.json"
|
||||
|
||||
fail() { echo "FAIL: $*" >&2; exit 1; }
|
||||
|
||||
echo "test-devin-plugin: checking Devin CLI manifest"
|
||||
|
||||
# --- Manifest is valid and matches the repo version -------------------------
|
||||
[ -f "$MANIFEST" ] || fail "manifest missing at $MANIFEST"
|
||||
|
||||
python3 - "$MANIFEST" <<'PY'
|
||||
import json
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
manifest_path = Path(sys.argv[1])
|
||||
manifest = json.loads(manifest_path.read_text(encoding="utf-8"))
|
||||
repo_root = manifest_path.parents[1]
|
||||
|
||||
if manifest.get("name") != "superpowers":
|
||||
raise AssertionError(f"plugin name: expected 'superpowers', got {manifest.get('name')!r}")
|
||||
|
||||
package = json.loads((repo_root / "package.json").read_text(encoding="utf-8"))
|
||||
if manifest.get("version") != package.get("version"):
|
||||
raise AssertionError(
|
||||
f"manifest version {manifest.get('version')!r} != package.json version {package.get('version')!r}"
|
||||
)
|
||||
|
||||
# Devin CLI plugins carry skills only (auto-discovered from ./skills/); the
|
||||
# manifest supports metadata + dependency lists, nothing executable.
|
||||
unsupported = ["skills", "hooks", "commands", "sessionStart", "contextFileName", "inject"]
|
||||
present = sorted(field for field in unsupported if field in manifest)
|
||||
if present:
|
||||
raise AssertionError("unsupported Devin manifest fields present: " + ", ".join(present))
|
||||
|
||||
version_config = json.loads((repo_root / ".version-bump.json").read_text(encoding="utf-8"))
|
||||
entries = version_config.get("files")
|
||||
if not isinstance(entries, list) or not any(
|
||||
entry.get("path") == ".devin-plugin/plugin.json" and entry.get("field") == "version"
|
||||
for entry in entries
|
||||
if isinstance(entry, dict)
|
||||
):
|
||||
raise AssertionError(".version-bump.json must update .devin-plugin/plugin.json version")
|
||||
|
||||
print("Devin plugin manifest looks good")
|
||||
PY
|
||||
|
||||
echo "PASS: Devin CLI plugin valid (manifest)"
|
||||
127
tests/diagnosing-superpowers/test-skill-structure.sh
Executable file
127
tests/diagnosing-superpowers/test-skill-structure.sh
Executable file
@@ -0,0 +1,127 @@
|
||||
#!/usr/bin/env bash
|
||||
# Structural checks for skills/diagnosing-superpowers. Behavior is tested by
|
||||
# scenario evals kept by the maintainer; this script only checks the things a
|
||||
# shell can check: frontmatter, referenced files exist, no local paths or
|
||||
# names leaked into shipped files, SKILL.md word budget.
|
||||
set -u
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
|
||||
REPO_ROOT="$(cd "$SCRIPT_DIR/../.." && pwd)"
|
||||
SKILL_DIR="$REPO_ROOT/skills/diagnosing-superpowers"
|
||||
SKILL_MD="$SKILL_DIR/SKILL.md"
|
||||
WORD_BUDGET=900
|
||||
|
||||
PASSES=0
|
||||
FAILURES=0
|
||||
|
||||
pass() { echo " [PASS] $1"; PASSES=$((PASSES + 1)); }
|
||||
fail() { echo " [FAIL] $1"; FAILURES=$((FAILURES + 1)); }
|
||||
|
||||
echo "diagnosing-superpowers structure"
|
||||
|
||||
# --- SKILL.md frontmatter -------------------------------------------------
|
||||
if [ -f "$SKILL_MD" ]; then
|
||||
pass "SKILL.md exists"
|
||||
frontmatter="$(awk 'NR==1 && $0!="---"{exit} NR>1 && $0=="---"{exit} NR>1{print}' "$SKILL_MD")"
|
||||
if printf '%s\n' "$frontmatter" | grep -q '^name: diagnosing-superpowers$'; then
|
||||
pass "frontmatter name is diagnosing-superpowers"
|
||||
else
|
||||
fail "frontmatter name is diagnosing-superpowers"
|
||||
fi
|
||||
description="$(printf '%s\n' "$frontmatter" | awk '/^description:/{sub(/^description:[ ]*/,""); print; found=1; next} found && /^[ ]/{print} found && !/^[ ]/{exit}' | tr '\n' ' ')"
|
||||
if printf '%s' "$description" | grep -q '^Use when'; then
|
||||
pass "description starts with 'Use when'"
|
||||
else
|
||||
fail "description starts with 'Use when' (got: ${description:0:60})"
|
||||
fi
|
||||
if [ "${#description}" -le 1024 ]; then
|
||||
pass "description under 1024 characters"
|
||||
else
|
||||
fail "description under 1024 characters (${#description})"
|
||||
fi
|
||||
for banned in "dispatch" "then" "step"; do
|
||||
if printf '%s' "$description" | grep -qiw "$banned"; then
|
||||
fail "description contains workflow word '$banned'"
|
||||
else
|
||||
pass "description avoids workflow word '$banned'"
|
||||
fi
|
||||
done
|
||||
|
||||
# --- word budget --------------------------------------------------------
|
||||
body_words="$(awk 'BEGIN{fm=0} NR==1 && $0=="---"{fm=1; next} fm==1 && $0=="---"{fm=2; next} fm==2{print}' "$SKILL_MD" | wc -w | tr -d ' ')"
|
||||
if [ "$body_words" -le "$WORD_BUDGET" ]; then
|
||||
pass "SKILL.md body within $WORD_BUDGET words ($body_words)"
|
||||
else
|
||||
fail "SKILL.md body within $WORD_BUDGET words ($body_words)"
|
||||
fi
|
||||
|
||||
# --- required sections --------------------------------------------------
|
||||
for heading in "## Hard rules" "## Red Flags"; do
|
||||
if grep -q "^$heading" "$SKILL_MD"; then
|
||||
pass "SKILL.md has section '$heading'"
|
||||
else
|
||||
fail "SKILL.md has section '$heading'"
|
||||
fi
|
||||
done
|
||||
|
||||
# --- every referenced skill file exists --------------------------------
|
||||
while IFS= read -r ref; do
|
||||
if [ -f "$SKILL_DIR/$ref" ]; then
|
||||
pass "referenced file exists: $ref"
|
||||
else
|
||||
fail "referenced file exists: $ref"
|
||||
fi
|
||||
done < <(grep -o '\(references\|prompts\|templates\)/[A-Za-z0-9._-]*\.md' "$SKILL_MD" | sort -u)
|
||||
else
|
||||
fail "SKILL.md exists"
|
||||
fi
|
||||
|
||||
# --- expected files -------------------------------------------------------
|
||||
expected_files=(
|
||||
references/claude-code-sessions.md
|
||||
references/codex-sessions.md
|
||||
references/other-harnesses.md
|
||||
prompts/skill-timeline.md
|
||||
prompts/plan-adherence.md
|
||||
prompts/repeated-work.md
|
||||
prompts/stumbles.md
|
||||
prompts/quality-evidence.md
|
||||
prompts/request-conflicts.md
|
||||
prompts/cost-and-time.md
|
||||
prompts/scrub.md
|
||||
prompts/scrub-audit.md
|
||||
prompts/similar-session.md
|
||||
templates/case.md
|
||||
templates/report.md
|
||||
templates/bundle-README.md
|
||||
templates/issue.md
|
||||
)
|
||||
for rel in "${expected_files[@]}"; do
|
||||
if [ -f "$SKILL_DIR/$rel" ]; then
|
||||
pass "expected file present: $rel"
|
||||
else
|
||||
fail "expected file present: $rel"
|
||||
fi
|
||||
done
|
||||
|
||||
# --- no local paths or names in shipped files ----------------------------
|
||||
leaks="$(grep -rn -E '/Users/|/home/|jesse' "$SKILL_DIR" "$SCRIPT_DIR" --exclude=test-skill-structure.sh 2>/dev/null || true)"
|
||||
if [ -z "$leaks" ]; then
|
||||
pass "no machine-specific paths or names in shipped files (skills + tests)"
|
||||
else
|
||||
fail "no machine-specific paths or names in shipped files (skills + tests)"
|
||||
printf '%s\n' "$leaks" | head -10 | sed 's/^/ /'
|
||||
fi
|
||||
|
||||
# --- "the user" never appears in skill prose -----------------------------
|
||||
user_hits="$(grep -rn -i 'the user' "$SKILL_DIR" --include='*.md' 2>/dev/null || true)"
|
||||
if [ -z "$user_hits" ]; then
|
||||
pass "skill files say 'your human partner', not 'the user'"
|
||||
else
|
||||
fail "skill files say 'your human partner', not 'the user'"
|
||||
printf '%s\n' "$user_hits" | head -10 | sed 's/^/ /'
|
||||
fi
|
||||
|
||||
echo
|
||||
echo "Passed: $PASSES Failed: $FAILURES"
|
||||
[ "$FAILURES" -eq 0 ]
|
||||
0
tests/hermes/__init__.py
Normal file
0
tests/hermes/__init__.py
Normal file
30
tests/hermes/conftest.py
Normal file
30
tests/hermes/conftest.py
Normal file
@@ -0,0 +1,30 @@
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
from unittest.mock import MagicMock
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def mock_ctx():
|
||||
ctx = MagicMock()
|
||||
ctx._hooks = {}
|
||||
ctx._skills = {}
|
||||
|
||||
def register_hook(event, fn):
|
||||
ctx._hooks[event] = fn
|
||||
|
||||
def register_skill(name, path):
|
||||
# Mimic hermes' real register_skill, which calls path.exists() and
|
||||
# therefore breaks on a str (the bug that silently disabled the whole
|
||||
# plugin, found 2026-07-23). Keeping that fidelity here means a
|
||||
# regression to str paths fails these tests instead of failing
|
||||
# silently inside hermes.
|
||||
if not isinstance(path, Path):
|
||||
raise AttributeError(
|
||||
f"register_skill requires a pathlib.Path, got {type(path).__name__}"
|
||||
)
|
||||
ctx._skills[name] = path
|
||||
|
||||
ctx.register_hook.side_effect = register_hook
|
||||
ctx.register_skill.side_effect = register_skill
|
||||
return ctx
|
||||
98
tests/hermes/test_bootstrap.py
Normal file
98
tests/hermes/test_bootstrap.py
Normal file
@@ -0,0 +1,98 @@
|
||||
import importlib
|
||||
import os
|
||||
import sys
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, os.path.abspath(
|
||||
os.path.join(os.path.dirname(__file__), "../../.hermes-plugin")
|
||||
))
|
||||
|
||||
BOOTSTRAP_MARKER = "superpowers:using-superpowers bootstrap for hermes"
|
||||
|
||||
# Hermes spills injected context over 10,000 chars to a file, which breaks
|
||||
# inline injection semantics. The bootstrap must stay under it with margin.
|
||||
HERMES_CONTEXT_SPILL_LIMIT = 10_000
|
||||
|
||||
|
||||
def _load():
|
||||
if "__init__" in sys.modules:
|
||||
del sys.modules["__init__"]
|
||||
return importlib.import_module("__init__")
|
||||
|
||||
|
||||
def _bootstrap():
|
||||
m = _load()
|
||||
return m._build_bootstrap(m._skills_dir())
|
||||
|
||||
|
||||
class TestStripFrontmatter:
|
||||
def test_strips_yaml_block(self):
|
||||
m = _load()
|
||||
content = "---\nname: foo\ndescription: bar\n---\n# Body\nContent here"
|
||||
assert m._strip_frontmatter(content) == "# Body\nContent here"
|
||||
|
||||
def test_no_frontmatter_returns_trimmed_content(self):
|
||||
m = _load()
|
||||
content = "# No frontmatter\nJust content"
|
||||
assert m._strip_frontmatter(content) == "# No frontmatter\nJust content"
|
||||
|
||||
def test_strips_surrounding_whitespace_from_body(self):
|
||||
m = _load()
|
||||
content = "---\nname: foo\n---\n\n\n# Body\n\n"
|
||||
assert m._strip_frontmatter(content) == "# Body"
|
||||
|
||||
|
||||
class TestSkillsDirResolution:
|
||||
def test_repo_layout_resolves(self):
|
||||
# The repo checkout IS the git-clone layout: .hermes-plugin/ and
|
||||
# skills/ are siblings, so resolution must succeed from here.
|
||||
m = _load()
|
||||
skills = m._skills_dir()
|
||||
assert os.path.isfile(
|
||||
os.path.join(skills, "using-superpowers", "SKILL.md")
|
||||
)
|
||||
|
||||
|
||||
class TestBootstrapContent:
|
||||
def test_marker_and_wrapper(self):
|
||||
content = _bootstrap()
|
||||
assert BOOTSTRAP_MARKER in content
|
||||
assert content.startswith("<EXTREMELY_IMPORTANT>")
|
||||
assert content.rstrip().endswith("</EXTREMELY_IMPORTANT>")
|
||||
|
||||
def test_contains_using_superpowers_body(self):
|
||||
content = _bootstrap()
|
||||
# A distinctive line from the skill body proves the real SKILL.md was
|
||||
# embedded, not a stub.
|
||||
assert "You have superpowers" in content
|
||||
assert "## The Rule" in content
|
||||
|
||||
def test_frontmatter_stripped(self):
|
||||
content = _bootstrap()
|
||||
assert "---\nname:" not in content
|
||||
|
||||
def test_tool_mapping_sourced_from_reference_file(self):
|
||||
m = _load()
|
||||
content = _bootstrap()
|
||||
ref = os.path.join(
|
||||
m._skills_dir(), "using-superpowers", "references", "hermes-tools.md"
|
||||
)
|
||||
with open(ref, encoding="utf-8") as f:
|
||||
ref_text = f.read().strip()
|
||||
# The mapping is included verbatim from the reference file — the
|
||||
# single source, not a drift-prone inline copy.
|
||||
assert ref_text in content
|
||||
assert "read_file" in content
|
||||
|
||||
def test_skill_view_guidance_present(self):
|
||||
content = _bootstrap()
|
||||
assert 'skill_view("superpowers:brainstorming")' in content
|
||||
|
||||
def test_under_hermes_context_spill_limit(self):
|
||||
content = _bootstrap()
|
||||
assert len(content) < HERMES_CONTEXT_SPILL_LIMIT, (
|
||||
f"bootstrap is {len(content)} chars; hermes spills injected "
|
||||
f"context over {HERMES_CONTEXT_SPILL_LIMIT} to a file, which "
|
||||
"breaks inline injection"
|
||||
)
|
||||
142
tests/hermes/test_plugin.py
Normal file
142
tests/hermes/test_plugin.py
Normal file
@@ -0,0 +1,142 @@
|
||||
import importlib
|
||||
import importlib.util
|
||||
import os
|
||||
import shutil
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
# Point at the plugin directory
|
||||
_PLUGIN_DIR = os.path.abspath(
|
||||
os.path.join(os.path.dirname(__file__), "../../.hermes-plugin")
|
||||
)
|
||||
sys.path.insert(0, _PLUGIN_DIR)
|
||||
|
||||
BOOTSTRAP_MARKER = "superpowers:using-superpowers bootstrap for hermes"
|
||||
|
||||
|
||||
def _load_plugin():
|
||||
"""Re-import plugin module fresh."""
|
||||
if "__init__" in sys.modules:
|
||||
del sys.modules["__init__"]
|
||||
return importlib.import_module("__init__")
|
||||
|
||||
|
||||
def _fire_pre_llm(ctx, **kwargs):
|
||||
hook = ctx._hooks["pre_llm_call"]
|
||||
defaults = {
|
||||
"session_id": "s1",
|
||||
"user_message": "hi",
|
||||
"conversation_history": [],
|
||||
"is_first_turn": False,
|
||||
"model": "test-model",
|
||||
"platform": "cli",
|
||||
}
|
||||
defaults.update(kwargs)
|
||||
return hook(**defaults)
|
||||
|
||||
|
||||
class TestPluginRegistration:
|
||||
def test_register_attaches_only_pre_llm_call_hook(self, mock_ctx):
|
||||
plugin = _load_plugin()
|
||||
plugin.register(mock_ctx)
|
||||
assert list(mock_ctx._hooks.keys()) == ["pre_llm_call"]
|
||||
|
||||
def test_register_registers_every_stock_skill_as_path(self, mock_ctx):
|
||||
plugin = _load_plugin()
|
||||
plugin.register(mock_ctx)
|
||||
# The conftest mock raises on non-Path (mirroring hermes' real
|
||||
# register_skill), so reaching these asserts proves every
|
||||
# registration passed a pathlib.Path.
|
||||
assert "using-superpowers" in mock_ctx._skills
|
||||
assert "brainstorming" in mock_ctx._skills
|
||||
for name, path in mock_ctx._skills.items():
|
||||
assert isinstance(path, Path)
|
||||
assert path.name == "SKILL.md"
|
||||
assert path.parent.name == name
|
||||
assert path.is_file()
|
||||
|
||||
def test_registered_skills_match_skill_directories(self, mock_ctx):
|
||||
plugin = _load_plugin()
|
||||
plugin.register(mock_ctx)
|
||||
skills_root = plugin._skills_dir()
|
||||
expected = {
|
||||
entry
|
||||
for entry in os.listdir(skills_root)
|
||||
if os.path.isfile(os.path.join(skills_root, entry, "SKILL.md"))
|
||||
}
|
||||
assert set(mock_ctx._skills.keys()) == expected
|
||||
|
||||
|
||||
class TestBootstrapInjection:
|
||||
def test_first_turn_returns_bootstrap_context(self, mock_ctx):
|
||||
plugin = _load_plugin()
|
||||
plugin.register(mock_ctx)
|
||||
result = _fire_pre_llm(mock_ctx, is_first_turn=True)
|
||||
assert isinstance(result, dict)
|
||||
content = result["context"]
|
||||
assert BOOTSTRAP_MARKER in content
|
||||
assert content.startswith("<EXTREMELY_IMPORTANT>")
|
||||
assert content.rstrip().endswith("</EXTREMELY_IMPORTANT>")
|
||||
|
||||
def test_later_turns_return_none(self, mock_ctx):
|
||||
plugin = _load_plugin()
|
||||
plugin.register(mock_ctx)
|
||||
assert _fire_pre_llm(mock_ctx, is_first_turn=False) is None
|
||||
assert _fire_pre_llm(mock_ctx, is_first_turn=None) is None
|
||||
|
||||
def test_hook_tolerates_future_kwargs(self, mock_ctx):
|
||||
plugin = _load_plugin()
|
||||
plugin.register(mock_ctx)
|
||||
result = _fire_pre_llm(
|
||||
mock_ctx, is_first_turn=True, telemetry_schema_version=3
|
||||
)
|
||||
assert BOOTSTRAP_MARKER in result["context"]
|
||||
|
||||
|
||||
class TestLayoutResolution:
|
||||
def _stage(self, tmp_path, layout):
|
||||
"""Copy the plugin module + a minimal skills tree in the given layout."""
|
||||
src_skills = Path(_PLUGIN_DIR).parent / "skills"
|
||||
if layout == "clone":
|
||||
plugdir = tmp_path / "superpowers" / ".hermes-plugin"
|
||||
else: # flat: module at the plugin dir root, skills nested inside it
|
||||
plugdir = tmp_path / "superpowers"
|
||||
skills = tmp_path / "superpowers" / "skills"
|
||||
plugdir.mkdir(parents=True, exist_ok=True)
|
||||
shutil.copy(Path(_PLUGIN_DIR) / "__init__.py", plugdir / "__init__.py")
|
||||
for skill in ("using-superpowers", "brainstorming"):
|
||||
shutil.copytree(src_skills / skill, skills / skill)
|
||||
return plugdir
|
||||
|
||||
def _load_from(self, plugdir):
|
||||
spec = importlib.util.spec_from_file_location(
|
||||
f"hermes_plugin_test_{plugdir.parent.name}_{plugdir.name}",
|
||||
plugdir / "__init__.py",
|
||||
)
|
||||
mod = importlib.util.module_from_spec(spec)
|
||||
spec.loader.exec_module(mod)
|
||||
return mod
|
||||
|
||||
def test_clone_layout_resolves_sibling_skills(self, tmp_path, mock_ctx):
|
||||
# git-clone install: .hermes-plugin/ and skills/ are siblings.
|
||||
plugdir = self._stage(tmp_path, "clone")
|
||||
mod = self._load_from(plugdir)
|
||||
mod.register(mock_ctx)
|
||||
assert "using-superpowers" in mock_ctx._skills
|
||||
|
||||
def test_flat_layout_resolves_nested_skills(self, tmp_path, mock_ctx):
|
||||
# flattened install: module at the plugin dir root, skills/ inside it.
|
||||
plugdir = self._stage(tmp_path, "flat")
|
||||
mod = self._load_from(plugdir)
|
||||
mod.register(mock_ctx)
|
||||
assert "using-superpowers" in mock_ctx._skills
|
||||
|
||||
def test_missing_skills_raises_loudly(self, tmp_path, mock_ctx):
|
||||
plugdir = tmp_path / "superpowers"
|
||||
plugdir.mkdir(parents=True)
|
||||
shutil.copy(Path(_PLUGIN_DIR) / "__init__.py", plugdir / "__init__.py")
|
||||
mod = self._load_from(plugdir)
|
||||
with pytest.raises(RuntimeError, match="cannot find the skills"):
|
||||
mod.register(mock_ctx)
|
||||
90
tests/systematic-debugging/test-find-polluter.sh
Executable file
90
tests/systematic-debugging/test-find-polluter.sh
Executable file
@@ -0,0 +1,90 @@
|
||||
#!/usr/bin/env bash
|
||||
set -euo pipefail
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
|
||||
REPO_ROOT="$(cd "$SCRIPT_DIR/../.." && pwd)"
|
||||
SCRIPT_UNDER_TEST="$REPO_ROOT/skills/systematic-debugging/find-polluter.sh"
|
||||
|
||||
FAILURES=0
|
||||
TEST_ROOT="$(mktemp -d)"
|
||||
|
||||
cleanup() {
|
||||
rm -rf "$TEST_ROOT"
|
||||
}
|
||||
trap cleanup EXIT
|
||||
|
||||
pass() {
|
||||
echo " [PASS] $1"
|
||||
}
|
||||
|
||||
fail() {
|
||||
echo " [FAIL] $1"
|
||||
FAILURES=$((FAILURES + 1))
|
||||
}
|
||||
|
||||
assert_contains() {
|
||||
local haystack="$1"
|
||||
local needle="$2"
|
||||
local description="$3"
|
||||
|
||||
if printf '%s' "$haystack" | grep -Fq -- "$needle"; then
|
||||
pass "$description"
|
||||
else
|
||||
fail "$description (expected output to contain: $needle)"
|
||||
fi
|
||||
}
|
||||
|
||||
# Toy project: one top-level test, one nested test. A stubbed `npm` on PATH
|
||||
# creates the pollution marker whenever any test runs, so the first test file
|
||||
# executed is always identified as the polluter.
|
||||
setup_project() {
|
||||
PROJECT="$TEST_ROOT/project"
|
||||
rm -rf "$PROJECT"
|
||||
mkdir -p "$PROJECT/src/feature" "$PROJECT/bin"
|
||||
echo "test('top')" > "$PROJECT/src/top.test.ts"
|
||||
echo "test('nested')" > "$PROJECT/src/feature/nested.test.ts"
|
||||
cat > "$PROJECT/bin/npm" <<'EOF'
|
||||
#!/usr/bin/env bash
|
||||
touch pollution.marker
|
||||
EOF
|
||||
chmod +x "$PROJECT/bin/npm"
|
||||
}
|
||||
|
||||
# run_polluter <pattern> — runs the script in the toy project with the stub
|
||||
# npm first on PATH; captures combined output, never aborts on exit code.
|
||||
run_polluter() {
|
||||
local pattern="$1"
|
||||
rm -f "$PROJECT/pollution.marker"
|
||||
(
|
||||
cd "$PROJECT"
|
||||
PATH="$PROJECT/bin:$PATH" "$SCRIPT_UNDER_TEST" 'pollution.marker' "$pattern" 2>&1
|
||||
) || true
|
||||
}
|
||||
|
||||
echo "Test: documented pattern finds nested test files (issue #2008)"
|
||||
setup_project
|
||||
OUTPUT="$(run_polluter 'src/**/*.test.ts')"
|
||||
assert_contains "$OUTPUT" "FOUND POLLUTER" "documented pattern runs tests and detects pollution"
|
||||
|
||||
echo "Test: documented pattern also finds top-level test files"
|
||||
setup_project
|
||||
OUTPUT="$(run_polluter 'src/**/*.test.ts')"
|
||||
assert_contains "$OUTPUT" "Found 2 test files" "src/**/*.test.ts matches src/top.test.ts and src/feature/nested.test.ts"
|
||||
|
||||
echo "Test: ./-prefixed pattern matches the same files"
|
||||
setup_project
|
||||
OUTPUT="$(run_polluter './src/**/*.test.ts')"
|
||||
assert_contains "$OUTPUT" "Found 2 test files" "leading ./ on the pattern is accepted"
|
||||
|
||||
echo "Test: non-matching pattern reports an honest zero"
|
||||
setup_project
|
||||
OUTPUT="$(run_polluter 'nomatch/**/*.test.ts')"
|
||||
assert_contains "$OUTPUT" "Found 0 test files" "empty result counts as 0, not 1"
|
||||
assert_contains "$OUTPUT" "No polluter found" "empty result exits via the clean path"
|
||||
|
||||
echo ""
|
||||
if [ "$FAILURES" -gt 0 ]; then
|
||||
echo "$FAILURES test(s) failed"
|
||||
exit 1
|
||||
fi
|
||||
echo "All tests passed"
|
||||
76
tests/version-bump/test-bump-version.sh
Normal file
76
tests/version-bump/test-bump-version.sh
Normal file
@@ -0,0 +1,76 @@
|
||||
#!/usr/bin/env bash
|
||||
set -euo pipefail
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
|
||||
REPO_ROOT="$(cd "$SCRIPT_DIR/../.." && pwd)"
|
||||
SCRIPT_SOURCE="$REPO_ROOT/scripts/bump-version.sh"
|
||||
TEST_ROOT="$(mktemp -d)"
|
||||
|
||||
cleanup() {
|
||||
rm -rf "$TEST_ROOT"
|
||||
}
|
||||
trap cleanup EXIT
|
||||
|
||||
fail() {
|
||||
echo "FAIL: $*" >&2
|
||||
exit 1
|
||||
}
|
||||
|
||||
make_fixture() {
|
||||
local repo="$1"
|
||||
local yaml_body="$2"
|
||||
|
||||
mkdir -p "$repo/scripts" "$repo/.hermes-plugin"
|
||||
cp "$SCRIPT_SOURCE" "$repo/scripts/bump-version.sh"
|
||||
cat >"$repo/.version-bump.json" <<'JSON'
|
||||
{
|
||||
"files": [
|
||||
{ "path": "package.json", "field": "version" },
|
||||
{ "path": ".hermes-plugin/plugin.yaml", "field": "version" }
|
||||
],
|
||||
"audit": { "exclude": [] }
|
||||
}
|
||||
JSON
|
||||
cat >"$repo/package.json" <<'JSON'
|
||||
{
|
||||
"name": "fixture",
|
||||
"version": "1.2.3"
|
||||
}
|
||||
JSON
|
||||
printf '%s\n' "$yaml_body" >"$repo/.hermes-plugin/plugin.yaml"
|
||||
}
|
||||
|
||||
happy_repo="$TEST_ROOT/happy"
|
||||
make_fixture "$happy_repo" $'name: superpowers\nversion: 1.2.3'
|
||||
|
||||
/bin/bash "$happy_repo/scripts/bump-version.sh" --check >"$TEST_ROOT/check.out"
|
||||
/bin/bash "$happy_repo/scripts/bump-version.sh" --audit >"$TEST_ROOT/audit.out"
|
||||
/bin/bash "$happy_repo/scripts/bump-version.sh" 2.3.4 >"$TEST_ROOT/bump.out"
|
||||
|
||||
[[ "$(jq -r '.version' "$happy_repo/package.json")" == "2.3.4" ]] \
|
||||
|| fail "JSON manifest was not bumped"
|
||||
[[ "$(yq -r '.version' "$happy_repo/.hermes-plugin/plugin.yaml")" == "2.3.4" ]] \
|
||||
|| fail "YAML manifest was not bumped"
|
||||
|
||||
jq -e '
|
||||
any(.files[];
|
||||
.path == ".hermes-plugin/plugin.yaml" and .field == "version")
|
||||
' "$REPO_ROOT/.version-bump.json" >/dev/null \
|
||||
|| fail "Hermes manifest is not registered"
|
||||
|
||||
invalid_repo="$TEST_ROOT/invalid"
|
||||
make_fixture "$invalid_repo" $'name: superpowers\nversion: 123'
|
||||
cp "$invalid_repo/package.json" "$TEST_ROOT/package.before"
|
||||
cp "$invalid_repo/.hermes-plugin/plugin.yaml" "$TEST_ROOT/plugin.before"
|
||||
|
||||
if /bin/bash "$invalid_repo/scripts/bump-version.sh" 2.3.4 \
|
||||
>"$TEST_ROOT/invalid.out" 2>&1; then
|
||||
fail "bump accepted a non-string YAML version"
|
||||
fi
|
||||
|
||||
cmp -s "$TEST_ROOT/package.before" "$invalid_repo/package.json" \
|
||||
|| fail "JSON manifest changed before YAML validation failed"
|
||||
cmp -s "$TEST_ROOT/plugin.before" "$invalid_repo/.hermes-plugin/plugin.yaml" \
|
||||
|| fail "invalid YAML manifest changed"
|
||||
|
||||
echo "Version-bump tests passed"
|
||||
113
tests/writing-skills/test-render-graphs.sh
Executable file
113
tests/writing-skills/test-render-graphs.sh
Executable file
@@ -0,0 +1,113 @@
|
||||
#!/usr/bin/env bash
|
||||
set -u
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
|
||||
REPO_ROOT="$(cd "$SCRIPT_DIR/../.." && pwd)"
|
||||
SCRIPT_UNDER_TEST="$REPO_ROOT/skills/writing-skills/render-graphs.js"
|
||||
NODE_BIN="$(command -v node)"
|
||||
|
||||
PASSES=0
|
||||
FAILURES=0
|
||||
TEST_ROOT="$(mktemp -d)"
|
||||
|
||||
cleanup() {
|
||||
rm -rf "$TEST_ROOT"
|
||||
}
|
||||
trap cleanup EXIT
|
||||
|
||||
pass() {
|
||||
echo " [PASS] $1"
|
||||
PASSES=$((PASSES + 1))
|
||||
}
|
||||
|
||||
fail() {
|
||||
echo " [FAIL] $1"
|
||||
FAILURES=$((FAILURES + 1))
|
||||
}
|
||||
|
||||
assert_contains() {
|
||||
local haystack="$1"
|
||||
local needle="$2"
|
||||
local description="$3"
|
||||
|
||||
if printf '%s' "$haystack" | grep -Fq -- "$needle"; then
|
||||
pass "$description"
|
||||
else
|
||||
fail "$description"
|
||||
echo " expected to find: $needle"
|
||||
fi
|
||||
}
|
||||
|
||||
assert_not_contains() {
|
||||
local haystack="$1"
|
||||
local needle="$2"
|
||||
local description="$3"
|
||||
|
||||
if printf '%s' "$haystack" | grep -Fq -- "$needle"; then
|
||||
fail "$description"
|
||||
echo " did not expect to find: $needle"
|
||||
else
|
||||
pass "$description"
|
||||
fi
|
||||
}
|
||||
|
||||
fixture="$TEST_ROOT/fixture-skill"
|
||||
mkdir -p "$fixture" "$TEST_ROOT/empty-path"
|
||||
cat >"$fixture/SKILL.md" <<'EOF'
|
||||
---
|
||||
name: fixture-skill
|
||||
---
|
||||
|
||||
# Fixture Skill
|
||||
|
||||
```dot
|
||||
digraph fixture_graph {
|
||||
start -> end;
|
||||
}
|
||||
```
|
||||
EOF
|
||||
|
||||
echo "Writing-skills render-graphs tests"
|
||||
|
||||
missing_dot_output="$(PATH="$TEST_ROOT/empty-path" "$NODE_BIN" "$SCRIPT_UNDER_TEST" "$fixture" 2>&1)"
|
||||
missing_dot_status=$?
|
||||
|
||||
if [[ "$missing_dot_status" -ne 0 ]]; then
|
||||
pass "missing Graphviz exits non-zero"
|
||||
else
|
||||
fail "missing Graphviz exits non-zero"
|
||||
fi
|
||||
assert_contains "$missing_dot_output" "Error: graphviz (dot) not found." "missing Graphviz reports install guidance"
|
||||
assert_not_contains "$missing_dot_output" "ReferenceError: require is not defined" "script runs as an ES module"
|
||||
|
||||
render_output="$("$NODE_BIN" "$SCRIPT_UNDER_TEST" "$fixture" 2>&1)"
|
||||
render_status=$?
|
||||
|
||||
if [[ "$render_status" -eq 0 ]]; then
|
||||
pass "fixture diagram renders"
|
||||
else
|
||||
fail "fixture diagram renders"
|
||||
printf '%s\n' "$render_output"
|
||||
fi
|
||||
|
||||
assert_contains "$render_output" "Found 1 diagram(s)" "reports discovered diagram"
|
||||
assert_contains "$render_output" "Rendered: fixture_graph.svg" "reports rendered SVG"
|
||||
|
||||
if [[ -f "$fixture/diagrams/fixture_graph.svg" ]]; then
|
||||
pass "writes SVG output"
|
||||
else
|
||||
fail "writes SVG output"
|
||||
fi
|
||||
|
||||
if [[ -f "$fixture/diagrams/fixture_graph.svg" ]] && grep -Fq "<svg" "$fixture/diagrams/fixture_graph.svg"; then
|
||||
pass "SVG output has SVG markup"
|
||||
else
|
||||
fail "SVG output has SVG markup"
|
||||
fi
|
||||
|
||||
echo
|
||||
echo "Results: $PASSES passed, $FAILURES failed"
|
||||
|
||||
if [[ "$FAILURES" -gt 0 ]]; then
|
||||
exit 1
|
||||
fi
|
||||
Reference in New Issue
Block a user