Compare commits
8 Commits
a42d6b76f4
...
master
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
c0f813b792 | ||
|
|
8f231d49e7 | ||
|
|
545c213f88 | ||
|
|
04f24bb16d | ||
|
|
26585a86fc | ||
|
|
cd32841847 | ||
|
|
1bc7711f38 | ||
|
|
27aaabf305 |
@@ -26,7 +26,7 @@ Copy the skills into a specific project (`<repo>/.agents/skills` and `<repo>/.cl
|
|||||||
make project TARGET=/path/to/repo
|
make project TARGET=/path/to/repo
|
||||||
```
|
```
|
||||||
|
|
||||||
Both commands also enable the caveman and ponytail plugins in the target's `.claude/settings.json` so their modes auto-activate on session start; existing settings are preserved.
|
Both commands also patch the target's `.claude/settings.json`: the caveman and ponytail plugins are enabled so their modes auto-activate on session start, and the `autotune` reminder hook is registered so the agent points you at `/autotune` at most once an hour. Existing settings are preserved.
|
||||||
|
|
||||||
Restart your agent afterwards so it loads the new skills.
|
Restart your agent afterwards so it loads the new skills.
|
||||||
|
|
||||||
|
|||||||
@@ -1,11 +1,12 @@
|
|||||||
#!/usr/bin/env node
|
#!/usr/bin/env node
|
||||||
// Enable the caveman and ponytail plugins in a Claude Code settings.json so
|
// Enable the caveman and ponytail plugins in a Claude Code settings.json so
|
||||||
// their SessionStart hooks auto-activate both modes. Merges non-destructively:
|
// their SessionStart hooks auto-activate both modes, and register the autotune
|
||||||
// existing marketplaces and enabled plugins are preserved, unparseable files
|
// reminder hook. Merges non-destructively: existing marketplaces, plugins and
|
||||||
// are left untouched.
|
// hooks are preserved, unparseable files are left untouched.
|
||||||
"use strict";
|
"use strict";
|
||||||
|
|
||||||
const fs = require("fs");
|
const fs = require("fs");
|
||||||
|
const path = require("path");
|
||||||
|
|
||||||
const settingsPath = process.argv[2];
|
const settingsPath = process.argv[2];
|
||||||
if (!settingsPath) {
|
if (!settingsPath) {
|
||||||
@@ -38,5 +39,15 @@ for (const [name, entry] of Object.entries(MARKETPLACES)) {
|
|||||||
}
|
}
|
||||||
data.enabledPlugins = Object.assign(data.enabledPlugins || {}, PLUGINS);
|
data.enabledPlugins = Object.assign(data.enabledPlugins || {}, PLUGINS);
|
||||||
|
|
||||||
|
const reminder = path.join(path.dirname(settingsPath), "skills", "autotune", "hooks", "reminder.sh");
|
||||||
|
data.hooks = data.hooks || {};
|
||||||
|
data.hooks.UserPromptSubmit = data.hooks.UserPromptSubmit || [];
|
||||||
|
const registered = JSON.stringify(data.hooks.UserPromptSubmit).includes("autotune/hooks/reminder.sh");
|
||||||
|
if (!registered) {
|
||||||
|
data.hooks.UserPromptSubmit.push({
|
||||||
|
hooks: [{ type: "command", command: `bash ${JSON.stringify(reminder)}` }],
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
fs.writeFileSync(settingsPath, JSON.stringify(data, null, 2) + "\n");
|
fs.writeFileSync(settingsPath, JSON.stringify(data, null, 2) + "\n");
|
||||||
console.log(`skills: enabled caveman + ponytail plugins in ${settingsPath}`);
|
console.log(`skills: enabled caveman + ponytail plugins and the autotune reminder in ${settingsPath}`);
|
||||||
|
|||||||
92
skills/autotune/SKILL.md
Normal file
92
skills/autotune/SKILL.md
Normal file
@@ -0,0 +1,92 @@
|
|||||||
|
---
|
||||||
|
name: autotune
|
||||||
|
description: >
|
||||||
|
Generate new agent skills and optimize existing ones so the agent works more
|
||||||
|
efficiently: fewer input and output tokens, fewer errors, more focus, and
|
||||||
|
learnings that persist across sessions. Every proposal is confirmed with the
|
||||||
|
operator through single-select (radio button) questions. Trigger ONLY on an
|
||||||
|
explicit operator request (/autotune, "autotune", "tune the skills"); never
|
||||||
|
run it on your own initiative. Portable across projects.
|
||||||
|
---
|
||||||
|
|
||||||
|
Autotune turns observed friction into durable skills. It runs on demand only,
|
||||||
|
it changes nothing without an explicit answer from the operator, and every
|
||||||
|
skill it writes or rewrites lands in the operator's skills repository.
|
||||||
|
|
||||||
|
## Trigger discipline
|
||||||
|
|
||||||
|
- Run **only** when the operator asks: `/autotune`, "autotune", "tune the
|
||||||
|
skills", or an equivalent instruction.
|
||||||
|
- The 60-minute reminder hook is a hint for the operator, not a trigger. When it
|
||||||
|
fires, print the hint and continue the operator's actual request.
|
||||||
|
- Never propose, write, or edit a skill as a side effect of unrelated work.
|
||||||
|
|
||||||
|
## Step 1: collect evidence
|
||||||
|
|
||||||
|
Before asking anything, gather concrete friction from the current session and,
|
||||||
|
where readable, from the project's memory and agent instruction files:
|
||||||
|
|
||||||
|
- **Token waste**: repeated file re-reads, whole-file reads where a range was
|
||||||
|
enough, long tool output pasted back, re-running a command instead of
|
||||||
|
grepping a saved log, verbose report prose.
|
||||||
|
- **Errors**: rejected tool calls, permission prompts, retried commands,
|
||||||
|
corrections the operator had to give twice.
|
||||||
|
- **Focus loss**: work that drifted beyond the request, unrequested tests or
|
||||||
|
deploys, questions that stalled an autonomous run.
|
||||||
|
- **Lost learnings**: facts re-derived this session that a memory file or a
|
||||||
|
skill should already have carried.
|
||||||
|
|
||||||
|
List each finding as one line: `symptom -> cost -> candidate fix`. Skip
|
||||||
|
anything that happened once and is not a pattern.
|
||||||
|
|
||||||
|
## Step 2: ask, do not decide
|
||||||
|
|
||||||
|
Use `AskUserQuestion` with `multiSelect: false` so every question renders as
|
||||||
|
radio buttons. Never assume an answer, never batch several decisions into one
|
||||||
|
option, and never write a file before the answers are in.
|
||||||
|
|
||||||
|
Ask in this order, at most four questions per call:
|
||||||
|
|
||||||
|
1. **Scope** - "What should autotune do this run?" Options: `Create a new
|
||||||
|
skill`, `Optimize an existing skill`, `Both`, plus the ranked findings if
|
||||||
|
more than one candidate exists.
|
||||||
|
2. **Target** - which finding becomes a skill, or which existing skill gets
|
||||||
|
optimized. One option per candidate, each with its measured cost in the
|
||||||
|
description.
|
||||||
|
3. **Shape** - for a new skill: `Standalone skill`, `Row in the shortcuts
|
||||||
|
skill`, `Extend an existing skill`, `Memory entry instead of a skill`.
|
||||||
|
For an optimization: `Tighten wording only`, `Change the trigger`, `Add
|
||||||
|
rules`, `Split into two skills`, `Merge into another skill`.
|
||||||
|
4. **Placement** - `Operator skills repository (portable)` or
|
||||||
|
`This project's skills directory (project-specific)`.
|
||||||
|
|
||||||
|
Present the drafted skill text (name, trigger phrasing, body outline) and ask
|
||||||
|
for a final `Write it` / `Revise it` / `Discard` confirmation before touching
|
||||||
|
disk. Rewrites of an existing skill additionally show a before/after diff of
|
||||||
|
the changed lines in that confirmation.
|
||||||
|
|
||||||
|
## Step 3: write
|
||||||
|
|
||||||
|
- Portable skills go to `skills/<name>/SKILL.md` in the operator's skills
|
||||||
|
repository; project-specific ones to that project's skills directory.
|
||||||
|
- Frontmatter carries `name` and a `description` that states what the skill
|
||||||
|
does **and** when to trigger it - the description is the only part loaded into
|
||||||
|
every session, so it decides whether the skill ever fires.
|
||||||
|
- Body is thin and single-purpose: the smallest instruction that changes
|
||||||
|
behaviour. Route to an authoritative doc instead of duplicating it.
|
||||||
|
- Optimizing an existing skill means the file gets shorter or sharper, not
|
||||||
|
longer. Delete rules the agent already follows, merge duplicated ones, and
|
||||||
|
cut every sentence that does not change an action.
|
||||||
|
- Report what changed in at most three lines, then refresh the reminder stamp
|
||||||
|
so the next hint is an hour out: `date +%s > ~/.claude/.autotune-stamp`.
|
||||||
|
|
||||||
|
## Rules
|
||||||
|
|
||||||
|
- One skill per run unless the operator picked `Both`; a broad rewrite of the
|
||||||
|
whole collection is not what autotune is for.
|
||||||
|
- Never write a skill for one-off work, for secrets, or for behaviour an
|
||||||
|
existing skill already covers - propose extending that skill instead.
|
||||||
|
- A skill that saves tokens but loses correctness is a regression; validation,
|
||||||
|
error handling, and security steps are never the thing that gets trimmed.
|
||||||
|
- Efficiency claims stay honest: name the observed friction the skill removes,
|
||||||
|
never a guessed percentage.
|
||||||
32
skills/autotune/hooks/reminder.sh
Executable file
32
skills/autotune/hooks/reminder.sh
Executable file
@@ -0,0 +1,32 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
# UserPromptSubmit hook: emit the autotune hint at most once per interval.
|
||||||
|
# Stdout lands in the agent's context, which is why the payload tells the agent
|
||||||
|
# to show the hint and keep going instead of running autotune by itself.
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
interval="${AUTOTUNE_INTERVAL_SECONDS:-3600}"
|
||||||
|
stamp="${AUTOTUNE_STAMP:-${HOME}/.claude/.autotune-stamp}"
|
||||||
|
|
||||||
|
now="$(date +%s)"
|
||||||
|
last=0
|
||||||
|
if [[ -f "${stamp}" ]]; then
|
||||||
|
read -r last <"${stamp}" || true
|
||||||
|
[[ "${last}" =~ ^[0-9]+$ ]] || last=0
|
||||||
|
fi
|
||||||
|
|
||||||
|
if ((now - last < interval)); then
|
||||||
|
exit 0
|
||||||
|
fi
|
||||||
|
|
||||||
|
mkdir -p "$(dirname "${stamp}")"
|
||||||
|
printf '%s\n' "${now}" >"${stamp}"
|
||||||
|
|
||||||
|
cat <<'HINT'
|
||||||
|
AUTOTUNE HINT: over an hour since the last hint. Show the operator this notice
|
||||||
|
verbatim as a short block, then continue with their request. Do NOT run
|
||||||
|
autotune, do not propose skills, do not ask questions about it.
|
||||||
|
|
||||||
|
💡 /autotune - generate a new skill or optimize an existing one
|
||||||
|
(fewer tokens, fewer errors, more focus, learnings that persist).
|
||||||
|
Runs only when you trigger it.
|
||||||
|
HINT
|
||||||
66
skills/plan/SKILL.md
Normal file
66
skills/plan/SKILL.md
Normal file
@@ -0,0 +1,66 @@
|
|||||||
|
---
|
||||||
|
name: plan
|
||||||
|
description: >
|
||||||
|
Turn a request into an objective-oriented plan and hand it to the robot skill
|
||||||
|
for autonomous execution. Trigger on /plan or when the operator asks for a
|
||||||
|
plan, a breakdown, or a strategy for a task before any work starts. Portable
|
||||||
|
across projects.
|
||||||
|
---
|
||||||
|
|
||||||
|
Produce a plan whose every item is an outcome with a check, then execute it
|
||||||
|
autonomously. A plan that lists activities instead of objectives cannot be
|
||||||
|
verified, and an unverifiable plan cannot be handed to a robot.
|
||||||
|
|
||||||
|
## Procedure
|
||||||
|
|
||||||
|
1. **Interview first.** Open every plan with the `active-listening` skill: ask
|
||||||
|
until scope, constraints, success criteria, and the operator-only facts are
|
||||||
|
all pinned, then reflect the understanding back. This is the only point in
|
||||||
|
the run where questions are allowed, so leave nothing open here.
|
||||||
|
2. **Pin the objective.** State the end state in one sentence, as a condition
|
||||||
|
that is either true or false, never as an activity. Name the command or
|
||||||
|
observation that proves it.
|
||||||
|
3. **Inspect before decomposing.** Read the code, config, tests, and history the
|
||||||
|
objective touches. Every plan item must rest on something you have seen, not
|
||||||
|
on an assumption about how the project works.
|
||||||
|
4. **Decompose into sub-objectives.** Each item gets: the outcome it reaches,
|
||||||
|
the verification that proves it, and the items it depends on. Split an item
|
||||||
|
whenever it needs more than one verification. Order by dependency and mark
|
||||||
|
the items that are independent, so they can run in parallel.
|
||||||
|
5. **Resolve the open decisions now.** Any root cause, design choice, or
|
||||||
|
trade-off the plan rests on gets settled before execution, escalating to the
|
||||||
|
`dialectic` skill where being wrong is expensive. The robot does not ask, so
|
||||||
|
an unresolved decision becomes a guess at runtime.
|
||||||
|
6. **Record the plan as a todo list.** Write the sub-objectives into the
|
||||||
|
harness's own todo tracking, one entry per item, phrased as the outcome. Keep
|
||||||
|
a plan file only when the operator asks for one.
|
||||||
|
7. **Present it and wait.** Show the objective, the ordered sub-objectives with
|
||||||
|
their verifications, what is deliberately out of scope, and the assumptions
|
||||||
|
the plan rests on. Stop here: the plan is the deliverable of this step.
|
||||||
|
8. **Implement it statically, in full.** On the operator's go, write every code,
|
||||||
|
config, test, and doc change the plan calls for, across all sub-objectives,
|
||||||
|
before running anything that needs live infrastructure. Verify statically:
|
||||||
|
read the diff, run the linters, the unit tests, and whatever dry-run or
|
||||||
|
syntax check the project offers. This step ends only when the whole plan
|
||||||
|
exists on disk, with no sub-objective left unwritten.
|
||||||
|
9. **Then iterate under `robot`.** With the static implementation complete,
|
||||||
|
invoke the `robot` skill with the objective as its goal for the dynamic part:
|
||||||
|
run, deploy, observe, fix, repeat until every verification passes. It drives
|
||||||
|
the loop without check-ins, verifies each outcome instead of assuming it, and
|
||||||
|
reports the items that fall outside its clearance for the operator to run.
|
||||||
|
|
||||||
|
## Rules
|
||||||
|
|
||||||
|
- One objective per item, in outcome form: "endpoint returns 200 for an expired
|
||||||
|
token", not "fix the auth middleware".
|
||||||
|
- No item without a verification. If you cannot name what proves it, the item is
|
||||||
|
still a wish, not a plan.
|
||||||
|
- Plan only what the request covers. Speculative future-proofing, refactors
|
||||||
|
nobody asked for, and abstractions with one caller belong in the out-of-scope
|
||||||
|
list, not in the plan.
|
||||||
|
- Static before dynamic: never start the robot loop on a half-written plan. A
|
||||||
|
deploy that fails on code you had not written yet costs a full cycle and
|
||||||
|
proves nothing.
|
||||||
|
- Re-plan on contradicted evidence: when execution disproves an assumption the
|
||||||
|
plan rests on, revise the plan instead of forcing the original steps through.
|
||||||
|
- Cite `file:line` for every claim about the current state of the code.
|
||||||
108
skills/refactor/SKILL.md
Normal file
108
skills/refactor/SKILL.md
Normal file
@@ -0,0 +1,108 @@
|
|||||||
|
---
|
||||||
|
name: refactor
|
||||||
|
description: >
|
||||||
|
Refactor every file that is currently uncommitted in the git working tree -
|
||||||
|
staged, unstaged and untracked - up to the project's coding standards,
|
||||||
|
without changing behaviour. Trigger on /refactor or when the operator asks to
|
||||||
|
clean up, tidy, or polish the current changes before committing. Portable
|
||||||
|
across projects.
|
||||||
|
---
|
||||||
|
|
||||||
|
The unit of work is the uncommitted change set, not the repository. Behaviour
|
||||||
|
stays identical: a refactor that changes what the code does is a bug, not a
|
||||||
|
cleanup.
|
||||||
|
|
||||||
|
## Step 1: fix the file set
|
||||||
|
|
||||||
|
Read the set from git, never from memory of what was edited:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
git status --porcelain
|
||||||
|
```
|
||||||
|
|
||||||
|
Take modified, added, renamed and untracked files. Drop deleted files, binaries,
|
||||||
|
lockfiles, vendored trees and generated output. Restate the resulting list in
|
||||||
|
one line before touching anything; if it is empty, say so and stop.
|
||||||
|
|
||||||
|
Never `git stash` and never revert a working-tree file to compare - other agents
|
||||||
|
and the operator share this tree. Read the committed version with
|
||||||
|
`git show HEAD:<path>` into a temporary file instead.
|
||||||
|
|
||||||
|
## Step 2: get a safety net first
|
||||||
|
|
||||||
|
A refactor is only provably behaviour-preserving if something failed before it
|
||||||
|
and passes after it. Before editing, check what covers the change set:
|
||||||
|
|
||||||
|
- **Coverage exists**: run it now and record the green baseline. A suite that
|
||||||
|
was already red gets reported, not refactored around.
|
||||||
|
- **No coverage and the code is testable logic**: write the characterization
|
||||||
|
tests first, against the current behaviour - including the behaviour you
|
||||||
|
consider wrong, so the refactor cannot silently change it. Keep them small
|
||||||
|
and framework-free unless the project already has a framework. Fix the
|
||||||
|
captured wrongness in a separate change, never in this one.
|
||||||
|
- **Not sensibly testable**: pure glue, declarative config, a thin wrapper over
|
||||||
|
a service the test would have to mock into meaninglessness, or a project with
|
||||||
|
no test setup at all. Say so in one line and refactor without a net; do not
|
||||||
|
invent a test harness the project never asked for.
|
||||||
|
|
||||||
|
The new tests are part of the deliverable and must still pass unchanged after
|
||||||
|
the refactor. A test that had to be edited to stay green means the behaviour
|
||||||
|
moved: revert the refactor, not the test.
|
||||||
|
|
||||||
|
## Step 3: refactor each file
|
||||||
|
|
||||||
|
Read the whole file first, then apply the standards that the surrounding code
|
||||||
|
and the project's own config already establish (formatter config, linter rules,
|
||||||
|
existing idioms) over any generic preference of your own.
|
||||||
|
|
||||||
|
Four principles govern the result:
|
||||||
|
|
||||||
|
- **SRP**: one file, one responsibility; one function, one reason to change. A
|
||||||
|
unit that needs an "and" to describe it is two units.
|
||||||
|
- **SPOT**: every fact - a constant, a path, a schema, a rule - lives in exactly
|
||||||
|
one place and everything else references it.
|
||||||
|
- **KISS**: the simplest construct that works. No abstraction without a second
|
||||||
|
caller today, no configuration for a value that never varies.
|
||||||
|
- **DRY**: the same logic is written once. A literal repeated three times is a
|
||||||
|
missing constant.
|
||||||
|
|
||||||
|
Concretely:
|
||||||
|
|
||||||
|
- **File size**: no file containing program code exceeds 250 lines. Split the
|
||||||
|
overflow along the SRP seam - it is the symptom, not the cause. Description
|
||||||
|
and markup languages are exempt (Markdown, YAML, JSON, HTML, translation and
|
||||||
|
data files), as is the rare file the language or framework genuinely forbids
|
||||||
|
splitting: name that reason in the report, "it was easier" does not qualify.
|
||||||
|
- **Duplication**: fold repeated logic into the helper that already exists;
|
||||||
|
create a new one only when two or more call sites need it now.
|
||||||
|
- **Dead weight**: unused imports, variables, parameters, branches that cannot
|
||||||
|
be reached, flags with a single value, abstractions with one implementation.
|
||||||
|
- **Naming**: names that state what a thing is, matching the file's convention.
|
||||||
|
- **Shape**: split a function that needs a scroll to read; collapse nesting by
|
||||||
|
returning early; replace magic values with a named constant at its use site.
|
||||||
|
- **Errors**: never widen a caught exception, never swallow one, never remove a
|
||||||
|
validation or a security check to make the code shorter.
|
||||||
|
- **Comments**: invoke the `comments-clean` skill rather than reimplementing its
|
||||||
|
rules here.
|
||||||
|
|
||||||
|
Do not reformat regions you have no substantive fix for - cosmetic churn buries
|
||||||
|
the real change in the diff. If a file's problem is architectural and cannot be
|
||||||
|
fixed inside the change set, report it instead of half-doing it.
|
||||||
|
|
||||||
|
## Step 4: keep behaviour provable
|
||||||
|
|
||||||
|
- Re-run the step 2 net and require the same green result as the baseline.
|
||||||
|
- Never fold a bug fix, a feature, or a dependency change into a refactor.
|
||||||
|
Found a real bug? Name it in the report and leave it for a separate change.
|
||||||
|
- Run the cheap checks the project already has for the touched files (formatter,
|
||||||
|
linter, the module's own unit tests). Do not start a full suite, a build, or a
|
||||||
|
deploy on your own initiative - propose it and let the operator trigger it.
|
||||||
|
- State plainly what was verified and what was not. An unverified refactor is
|
||||||
|
reported as unverified.
|
||||||
|
|
||||||
|
## Step 5: report
|
||||||
|
|
||||||
|
One line per file: what changed and why it is safe. Then one line naming what
|
||||||
|
was deliberately left alone (architectural findings, bugs, files skipped) and
|
||||||
|
the exact command the operator can run to validate. Do not commit unless the
|
||||||
|
operator asked for a commit.
|
||||||
95
skills/test-fix/SKILL.md
Normal file
95
skills/test-fix/SKILL.md
Normal file
@@ -0,0 +1,95 @@
|
|||||||
|
---
|
||||||
|
name: test-fix
|
||||||
|
description: >
|
||||||
|
Run the project's Makefile test targets in a loop - run, diagnose, fix, re-run -
|
||||||
|
until every selected target is green or the loop is provably stuck. Trigger on
|
||||||
|
/test-fix or when the operator asks to make the tests pass, fix the failing
|
||||||
|
tests, or get the suite green. Portable across projects.
|
||||||
|
---
|
||||||
|
|
||||||
|
Same target discovery and selection as the `test` skill - invoke it for step 1
|
||||||
|
and step 2 rather than reimplementing the grep and the selection list. This
|
||||||
|
skill owns what happens after the first red result: the fix loop.
|
||||||
|
|
||||||
|
The contract is narrow: the code becomes correct, the tests stay honest.
|
||||||
|
|
||||||
|
## The loop
|
||||||
|
|
||||||
|
One iteration is: run -> read the failure -> one root cause -> one fix ->
|
||||||
|
re-run. Never batch several speculative fixes into one iteration; a green run
|
||||||
|
after three simultaneous changes proves nothing about which one mattered.
|
||||||
|
|
||||||
|
### Run
|
||||||
|
|
||||||
|
Run the failing target alone, not the whole selection - the fastest command
|
||||||
|
that reproduces the failure. Widen back to the full selection only when that
|
||||||
|
target is green.
|
||||||
|
|
||||||
|
### Diagnose
|
||||||
|
|
||||||
|
Read the actual error output before touching a file. Apply the `triage` skill
|
||||||
|
to reach a verified root cause; for a failure whose cause is genuinely unclear
|
||||||
|
after one look, apply `dialectic` rather than guessing twice.
|
||||||
|
|
||||||
|
Decide explicitly which side is wrong, and say so in one line before editing:
|
||||||
|
|
||||||
|
- **Code is wrong**: the test states the intended behaviour. Fix the code.
|
||||||
|
- **Test is wrong**: the test encodes an outdated or incorrect expectation.
|
||||||
|
Fixing it is allowed *only* with a stated reason for why the old expectation
|
||||||
|
was wrong - never because the code disagrees with it.
|
||||||
|
- **Environment is wrong**: missing dependency, stale container, absent env var,
|
||||||
|
unbuilt artefact. Fix the environment or report it; do not patch code around
|
||||||
|
a broken environment.
|
||||||
|
|
||||||
|
### Fix
|
||||||
|
|
||||||
|
Smallest change that addresses the root cause. Follow the project's existing
|
||||||
|
idioms and the `no-defaults` and `comments-clean` rules.
|
||||||
|
|
||||||
|
Forbidden, in every iteration, without explicit operator approval:
|
||||||
|
|
||||||
|
- deleting, skipping, `xfail`-ing, or commenting out a failing test
|
||||||
|
- loosening an assertion to whatever the code currently produces
|
||||||
|
- catching or swallowing the exception the test was written to surface
|
||||||
|
- adding retries, sleeps, or reruns to paper over a flaky failure
|
||||||
|
- disabling a linter rule, a type check, or a test target from the selection
|
||||||
|
|
||||||
|
Any of these is a report item, not a fix. A suite that is green because the
|
||||||
|
failing test no longer runs is a regression disguised as success.
|
||||||
|
|
||||||
|
### Re-run
|
||||||
|
|
||||||
|
Re-run the same target. Then, once it passes, re-run every target that was
|
||||||
|
already green - a fix that breaks a neighbouring suite is not a fix. Only
|
||||||
|
after the full selection is green is the loop done.
|
||||||
|
|
||||||
|
## Stopping
|
||||||
|
|
||||||
|
Stop and report, without asking for permission to stop:
|
||||||
|
|
||||||
|
- **Green**: full selection passes. Report and stop.
|
||||||
|
- **No progress**: the same failure survives 3 iterations, or the failure count
|
||||||
|
stops falling across 3 iterations. Report the root cause reached so far, what
|
||||||
|
was tried, and why each attempt failed.
|
||||||
|
- **Fix exceeds the mandate**: the real fix is an API change, a dependency bump,
|
||||||
|
a schema migration, or a redesign. Name it, show the minimal diff it would
|
||||||
|
need, and let the operator decide.
|
||||||
|
- **Oscillation**: fixing A re-breaks B and vice versa. Report both with the
|
||||||
|
conflict between them - that is a design problem, not a test problem.
|
||||||
|
- **Flaky**: a target passes and fails without any change between runs. Report
|
||||||
|
it as flaky with both outputs; never "fix" it by rerunning until green.
|
||||||
|
|
||||||
|
Never loop silently. Emit one line per iteration: target, failure, hypothesis,
|
||||||
|
change made.
|
||||||
|
|
||||||
|
## Report
|
||||||
|
|
||||||
|
- Per iteration: what failed, the root cause, the fix, the result.
|
||||||
|
- Final state of every selected target, pass or fail, with the verbatim output
|
||||||
|
of anything still failing.
|
||||||
|
- Everything deliberately not done: tests judged wrong but left alone,
|
||||||
|
environment issues, out-of-mandate fixes.
|
||||||
|
- The exact command reproducing the final state.
|
||||||
|
|
||||||
|
State calibrated confidence in the fixes per the `confidence` skill. Do not
|
||||||
|
commit unless the operator asked for a commit.
|
||||||
77
skills/test/SKILL.md
Normal file
77
skills/test/SKILL.md
Normal file
@@ -0,0 +1,77 @@
|
|||||||
|
---
|
||||||
|
name: test
|
||||||
|
description: >
|
||||||
|
Discover the test targets of the project's Makefile, let the operator pick
|
||||||
|
which ones to run, then run them and report. Trigger on /test or when the
|
||||||
|
operator asks to run the tests, the test suite, or a specific test target
|
||||||
|
without naming the exact command. Portable across projects.
|
||||||
|
---
|
||||||
|
|
||||||
|
The Makefile is the source of truth for how this project runs tests. Never
|
||||||
|
invent a command (`pytest`, `npm test`, `go test`) while a Makefile target
|
||||||
|
exists - the target carries the project's env vars, build dependencies and
|
||||||
|
container setup.
|
||||||
|
|
||||||
|
## Step 1: find the targets
|
||||||
|
|
||||||
|
From the repository root:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
grep -nE '^test[A-Za-z0-9_.-]*:' Makefile
|
||||||
|
```
|
||||||
|
|
||||||
|
Read the matched rules plus their prerequisites, so the selection list can say
|
||||||
|
what each one actually does (delegated script, container build, sub-targets).
|
||||||
|
|
||||||
|
Edge cases, handled explicitly rather than guessed around:
|
||||||
|
|
||||||
|
- **No Makefile**: say so in one line, name the test runner the project does
|
||||||
|
use (from `pyproject.toml`, `package.json`, `tox.ini`, CI workflow), and ask
|
||||||
|
before running anything.
|
||||||
|
- **No `test*` target**: report the targets that do exist and stop.
|
||||||
|
- **Included makefiles** (`include foo.mk`): grep those too.
|
||||||
|
- **Aggregate targets**: a target whose recipe is only other test targets (e.g.
|
||||||
|
`test: test-unit test-integration`) is the "run everything" entry - mark it as
|
||||||
|
such in the list, do not expand it into its parts silently.
|
||||||
|
|
||||||
|
## Step 2: offer the selection
|
||||||
|
|
||||||
|
Ask with `AskUserQuestion`, `multiSelect: true`, one question. Options are the
|
||||||
|
discovered targets, each with a one-line description of what it runs. Order the
|
||||||
|
aggregate/full target first and label it as the complete suite.
|
||||||
|
|
||||||
|
The tool takes at most 4 options. With more targets than that:
|
||||||
|
|
||||||
|
- print the **complete** discovered list as text first, one line per target, so
|
||||||
|
nothing is hidden, then
|
||||||
|
- offer the 4 most useful entries (aggregate first, then the ones matching the
|
||||||
|
operator's stated intent or the files currently uncommitted), and note that
|
||||||
|
"Other" accepts any target name from the printed list.
|
||||||
|
|
||||||
|
Never silently truncate. If the operator already named a target in their
|
||||||
|
request, skip the question and run it.
|
||||||
|
|
||||||
|
## Step 3: run
|
||||||
|
|
||||||
|
Run the selected targets in the order the operator listed them, one `make`
|
||||||
|
invocation per target, each as its own command so a failure is attributable:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
make <target>
|
||||||
|
```
|
||||||
|
|
||||||
|
Do not add `-k`, do not reorder, do not substitute a faster equivalent. If a
|
||||||
|
target needs a long timeout (container builds, e2e), set it on the Bash call
|
||||||
|
rather than backgrounding blindly.
|
||||||
|
|
||||||
|
Stop after the first failing target unless the operator asked for all of them -
|
||||||
|
a later suite running against a broken build produces noise, not information.
|
||||||
|
|
||||||
|
## Step 4: report
|
||||||
|
|
||||||
|
Per target: pass/fail plus the failing test names and the exact error output,
|
||||||
|
quoted verbatim. Never paraphrase a failure. Then one line with the exact
|
||||||
|
command to reproduce the failure alone.
|
||||||
|
|
||||||
|
Do not fix what failed unless the operator asks - report first. If the operator
|
||||||
|
does ask for a fix, apply the `triage` skill to it.
|
||||||
@@ -17,20 +17,33 @@ root cause is proven.
|
|||||||
whose conclusion is failure, cancelled, or timed-out. A downstream aggregate
|
whose conclusion is failure, cancelled, or timed-out. A downstream aggregate
|
||||||
job that fails only because an upstream job did is not a separate root cause —
|
job that fails only because an upstream job did is not a separate root cause —
|
||||||
note it and move on.
|
note it and move on.
|
||||||
2. **Dialectic per failing job.** For EACH failing job, invoke the `dialectic`
|
2. **Cluster the failures by error message.** Group the failing jobs by their
|
||||||
skill: form a thesis about the root cause from evidence (the job log, its
|
actual failing line, not by job name. Jobs whose messages differ in anything
|
||||||
artifacts, the code at the run's commit, git history), attack it with
|
but the interpolated identifiers (host, app, port, path) belong in separate
|
||||||
independent skeptics, and iterate to a ~99% thesis. The jobs are independent,
|
clusters.
|
||||||
so run their investigations in parallel where the tooling allows.
|
3. **Download artifacts per cluster.** For each cluster, fetch every artifact of
|
||||||
3. **Distinguish shared vs hidden causes.** When several jobs share one root
|
one representative job: reports, rescue diagnostics, inventories, container
|
||||||
cause, fix it once. When a job hides a second failure behind the first, keep
|
logs. Pull a second member's artifacts only when the representative's
|
||||||
going until the job is actually green, not just past the first error.
|
evidence does not explain the whole cluster. The job log usually names an
|
||||||
4. **Fix at the root.** Apply the real fix in the repository for each proven root
|
artifact path and nothing more; the assertion text, the server-side
|
||||||
|
exception and the state dumps are inside the artifact. Do this even when the
|
||||||
|
log looks conclusive, because a log that explains the symptom rarely
|
||||||
|
explains the cause.
|
||||||
|
4. **Dialectic per cluster.** For EACH cluster, invoke the `dialectic` skill:
|
||||||
|
form a thesis about the root cause from evidence (the job log, its artifacts,
|
||||||
|
the code at the run's commit, git history), attack it with independent
|
||||||
|
skeptics, and iterate to a ~99% thesis. The clusters are independent, so run
|
||||||
|
their investigations in parallel where the tooling allows.
|
||||||
|
5. **Distinguish shared vs hidden causes.** A cluster is a hypothesis, not a
|
||||||
|
proof: if the dialectic shows one cluster splitting into two causes, split it.
|
||||||
|
When a job hides a second failure behind the first, keep going until the job
|
||||||
|
is actually green, not just past the first error.
|
||||||
|
6. **Fix at the root.** Apply the real fix in the repository for each proven root
|
||||||
cause. Never mask a failure — no retry-until-pass, no disabling the check, no
|
cause. Never mask a failure — no retry-until-pass, no disabling the check, no
|
||||||
soft-skip. If a failure is genuinely external (upstream outage, flaky infra),
|
soft-skip. If a failure is genuinely external (upstream outage, flaky infra),
|
||||||
confirm that with evidence and surface it honestly instead of fixing around
|
confirm that with evidence and surface it honestly instead of fixing around
|
||||||
it.
|
it.
|
||||||
5. **Follow the run.** While the run is still in progress, re-check it
|
7. **Follow the run.** While the run is still in progress, re-check it
|
||||||
periodically and triage each newly failed job as it appears. The run is done
|
periodically and triage each newly failed job as it appears. The run is done
|
||||||
only when it has finished and every failure has a verified fix.
|
only when it has finished and every failure has a verified fix.
|
||||||
|
|
||||||
|
|||||||
92
tests/test_autotune.py
Normal file
92
tests/test_autotune.py
Normal file
@@ -0,0 +1,92 @@
|
|||||||
|
"""Validate the autotune reminder hook and its settings registration."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import json
|
||||||
|
import shutil
|
||||||
|
import subprocess
|
||||||
|
import tempfile
|
||||||
|
import unittest
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
REPO_ROOT = Path(__file__).resolve().parents[1]
|
||||||
|
SKILL = REPO_ROOT / "skills" / "autotune" / "SKILL.md"
|
||||||
|
HOOK = REPO_ROOT / "skills" / "autotune" / "hooks" / "reminder.sh"
|
||||||
|
PATCHER = REPO_ROOT / "scripts" / "enable-plugins.js"
|
||||||
|
|
||||||
|
|
||||||
|
def _run_hook(stamp: Path, interval: str = "3600") -> subprocess.CompletedProcess:
|
||||||
|
return subprocess.run(
|
||||||
|
["bash", str(HOOK)],
|
||||||
|
capture_output=True,
|
||||||
|
text=True,
|
||||||
|
check=True,
|
||||||
|
env={
|
||||||
|
"PATH": "/usr/bin:/bin",
|
||||||
|
"HOME": str(stamp.parent),
|
||||||
|
"AUTOTUNE_STAMP": str(stamp),
|
||||||
|
"AUTOTUNE_INTERVAL_SECONDS": interval,
|
||||||
|
},
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class TestAutotuneSkill(unittest.TestCase):
|
||||||
|
def test_skill_is_trigger_only(self):
|
||||||
|
text = SKILL.read_text(encoding="utf-8")
|
||||||
|
self.assertIn("name: autotune", text)
|
||||||
|
self.assertIn("multiSelect: false", text)
|
||||||
|
|
||||||
|
|
||||||
|
class TestAutotuneHook(unittest.TestCase):
|
||||||
|
def setUp(self):
|
||||||
|
self.tmp = tempfile.TemporaryDirectory()
|
||||||
|
self.addCleanup(self.tmp.cleanup)
|
||||||
|
self.stamp = Path(self.tmp.name) / ".claude" / ".autotune-stamp"
|
||||||
|
|
||||||
|
def test_first_run_hints_and_stamps(self):
|
||||||
|
result = _run_hook(self.stamp)
|
||||||
|
self.assertIn("/autotune", result.stdout)
|
||||||
|
self.assertTrue(self.stamp.is_file())
|
||||||
|
|
||||||
|
def test_second_run_is_silent_within_interval(self):
|
||||||
|
_run_hook(self.stamp)
|
||||||
|
self.assertEqual(_run_hook(self.stamp).stdout, "")
|
||||||
|
|
||||||
|
def test_hint_returns_after_the_interval(self):
|
||||||
|
_run_hook(self.stamp)
|
||||||
|
self.assertIn("/autotune", _run_hook(self.stamp, interval="0").stdout)
|
||||||
|
|
||||||
|
def test_corrupt_stamp_does_not_crash(self):
|
||||||
|
self.stamp.parent.mkdir(parents=True, exist_ok=True)
|
||||||
|
self.stamp.write_text("not-a-timestamp\n", encoding="utf-8")
|
||||||
|
self.assertIn("/autotune", _run_hook(self.stamp).stdout)
|
||||||
|
|
||||||
|
|
||||||
|
@unittest.skipUnless(shutil.which("node"), "node not installed")
|
||||||
|
class TestSettingsRegistration(unittest.TestCase):
|
||||||
|
def setUp(self):
|
||||||
|
self.tmp = tempfile.TemporaryDirectory()
|
||||||
|
self.addCleanup(self.tmp.cleanup)
|
||||||
|
self.settings = Path(self.tmp.name) / "settings.json"
|
||||||
|
|
||||||
|
def _patch(self) -> dict:
|
||||||
|
subprocess.run(["node", str(PATCHER), str(self.settings)], check=True, capture_output=True)
|
||||||
|
return json.loads(self.settings.read_text(encoding="utf-8"))
|
||||||
|
|
||||||
|
def test_hook_registered_once(self):
|
||||||
|
self._patch()
|
||||||
|
data = self._patch()
|
||||||
|
entries = json.dumps(data["hooks"]["UserPromptSubmit"])
|
||||||
|
self.assertEqual(entries.count("autotune/hooks/reminder.sh"), 1)
|
||||||
|
|
||||||
|
def test_existing_hooks_preserved(self):
|
||||||
|
self.settings.write_text(
|
||||||
|
json.dumps({"hooks": {"UserPromptSubmit": [{"hooks": [{"type": "command", "command": "true"}]}]}}),
|
||||||
|
encoding="utf-8",
|
||||||
|
)
|
||||||
|
data = self._patch()
|
||||||
|
self.assertEqual(len(data["hooks"]["UserPromptSubmit"]), 2)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
unittest.main()
|
||||||
Reference in New Issue
Block a user