diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json index 79bbce4..9d17650 100644 --- a/.claude-plugin/marketplace.json +++ b/.claude-plugin/marketplace.json @@ -4,7 +4,7 @@ "name": "kraken-networks" }, "metadata": { - "description": "Kraken Networks plugin suite — railyard, kb, fathom, nautilus, stargraph, forge" + "description": "Kraken Networks plugin suite — railyard, kb, fathom, nautilus, stargraph, forge, ui-fidelity, gauntlet" }, "plugins": [ { @@ -69,6 +69,15 @@ "source": "./plugins/ui-fidelity", "category": "development", "tags": ["ui","audit","playwright","react","ux","fidelity","dead-button","intent","persona-journey","axe","a11y","llm-critic"] + }, + { + "name": "gauntlet", + "description": "Gauntlet Loop harness — give an agent a hard external bar it cannot talk its way around, let it split the work, and never let the builder grade itself. Mechanically-blind A/B critics, a five-check bar-soundness gate, receipted verdicts, and a live progress board. Bar recipes for agentic capabilities, reinforcement learning, and general coding.", + "version": "0.1.0", + "author": {"name": "sean"}, + "source": "./plugins/gauntlet", + "category": "development", + "tags": ["gauntlet-loop","critic","blind-ab","quality-bar","agentic","reinforcement-learning","coding","iteration","subagents","evaluation"] } ] } diff --git a/.cursor-plugin/marketplace.json b/.cursor-plugin/marketplace.json index de5ca34..9d17650 100644 --- a/.cursor-plugin/marketplace.json +++ b/.cursor-plugin/marketplace.json @@ -4,7 +4,7 @@ "name": "kraken-networks" }, "metadata": { - "description": "Kraken Networks plugin suite — railyard, kb, fathom, nautilus, stargraph, forge" + "description": "Kraken Networks plugin suite — railyard, kb, fathom, nautilus, stargraph, forge, ui-fidelity, gauntlet" }, "plugins": [ { @@ -60,6 +60,24 @@ "source": "./plugins/forge", "category": "development", "tags": ["spec-driven","tdd","ralph-loop","adversarial","anti-cheat","graphrag","reflexion","pipeline","agents","ci-cd"] + }, + { + "name": "ui-fidelity", + "description": "Multi-gate UI audit for React/Vite/Playwright projects — catches dead buttons, orphan routes, flow-continuity breaks, intent-contract violations, contrast bugs, offscreen toasts, and persona-journey gaps. Includes LLM-as-critic UX pass for label/behavior mismatches and off-persona CTAs that lint and tests will never find.", + "version": "0.1.0", + "author": {"name": "sean"}, + "source": "./plugins/ui-fidelity", + "category": "development", + "tags": ["ui","audit","playwright","react","ux","fidelity","dead-button","intent","persona-journey","axe","a11y","llm-critic"] + }, + { + "name": "gauntlet", + "description": "Gauntlet Loop harness — give an agent a hard external bar it cannot talk its way around, let it split the work, and never let the builder grade itself. Mechanically-blind A/B critics, a five-check bar-soundness gate, receipted verdicts, and a live progress board. Bar recipes for agentic capabilities, reinforcement learning, and general coding.", + "version": "0.1.0", + "author": {"name": "sean"}, + "source": "./plugins/gauntlet", + "category": "development", + "tags": ["gauntlet-loop","critic","blind-ab","quality-bar","agentic","reinforcement-learning","coding","iteration","subagents","evaluation"] } ] } diff --git a/README.md b/README.md index 24c3ce9..d783ead 100644 --- a/README.md +++ b/README.md @@ -10,19 +10,21 @@ A Claude Code / Cursor plugin marketplace covering the Kraken Networks stack: | **nautilus** | Author + operate Nautilus data brokers | 0.1.0 | | **stargraph** | Author Stargraph orchestration graphs, skills, tools, markdown SKILL.md skills, directory plugins | 0.3.0 | | **forge** | 17-stage spec-anchored AI software pipeline w/ Ralph Loop, anti-cheat, spec GraphRAG, Reflexion lessons | 0.2.0 | +| **ui-fidelity** | Multi-gate UI audit — dead buttons, orphan routes, flow continuity, LLM-as-critic UX pass | 0.1.0 | +| **gauntlet** | Gauntlet Loop harness — hard external bar, blind A/B critics, receipted verdicts, live board | 0.1.0 | ## Install (Claude Code) ```bash claude plugins marketplace add KrakenNet/kraken-plugins -claude plugins install railyard kb fathom nautilus stargraph forge +claude plugins install railyard kb fathom nautilus stargraph forge ui-fidelity gauntlet ``` ## Install (Cursor) ```bash cursor plugins marketplace add KrakenNet/kraken-plugins -cursor plugins install railyard kb fathom nautilus stargraph forge +cursor plugins install railyard kb fathom nautilus stargraph forge ui-fidelity gauntlet ``` ## Plugin entry points @@ -35,6 +37,8 @@ After install, type `/help` in Claude Code or Cursor to see all slash commands. - `/nautilus:*` — Nautilus broker authoring + ops - `/stargraph:*` — Stargraph graph authoring + light ops - `/forge:*` — Spec-anchored feature pipeline + Ralph Loop +- `/ui-fidelity:*` — Multi-gate UI audit +- `/gauntlet:*` — Gauntlet Loop: bar-gated, blind-critic iteration to a hard external standard See each plugin's directory under `plugins/` for its README and command list. diff --git a/plugins/gauntlet/.claude-plugin/plugin.json b/plugins/gauntlet/.claude-plugin/plugin.json new file mode 100644 index 0000000..45bee93 --- /dev/null +++ b/plugins/gauntlet/.claude-plugin/plugin.json @@ -0,0 +1,8 @@ +{ + "name": "gauntlet", + "version": "0.1.0", + "description": "Gauntlet Loop harness — give an agent a hard external bar it cannot talk its way around, let it split the work, and never let the builder grade itself. Mechanically-blind A/B critics, a bar-soundness gate, receipted verdicts, and a live progress board. Bar recipes for agentic capabilities, reinforcement learning, and general coding.", + "author": {"name": "sean"}, + "license": "Apache-2.0", + "keywords": ["gauntlet-loop", "critic", "blind-ab", "quality-bar", "agentic", "reinforcement-learning", "coding", "iteration", "subagents", "evaluation"] +} diff --git a/plugins/gauntlet/README.md b/plugins/gauntlet/README.md new file mode 100644 index 0000000..d1a18b4 --- /dev/null +++ b/plugins/gauntlet/README.md @@ -0,0 +1,94 @@ +# gauntlet + +A Gauntlet Loop harness for Claude Code. + +The method is Matt Shumer's, from [How to Run a Gauntlet +Loop](https://somethingbig.ai/gauntlet-loop) — the prompting approach behind +[Claude of Duty](https://github.com/mshumer/Claude-of-Duty). This plugin turns it into a repeatable +harness with the two load-bearing rules enforced by scripts instead of by hope. + +## The idea + +Give a lead agent a goal and a real example of what great looks like. It splits the goal into the +smallest pieces that can be improved separately. Each piece gets a builder and a **separate** critic +with fresh context. The critic compares the work against the reference blind; when the reference +wins, it names the biggest gap and sends the work back. Repeat until you stop it. + +Agent output plateaus at "pretty good" because the agent that built the thing decides it is done. +The loop takes that decision away and gives it to something external. + +## What is enforced, not just asked for + +| Rule | Mechanism | +|---|---| +| No loop without a bar | `ab.py stage` refuses unless `.gauntlet/bar.md` shows PASS on all five soundness checks | +| The critic is blind | Candidate and reference are shuffled into `A/` and `B/`; the mapping is written outside the trial dir, mode 0600 | +| The critic stays blind | A `PreToolUse` hook denies any tool call that touches `.gauntlet/keys/` | +| The critic sees artifacts, not arguments | Staged trials contain only `A/`, `B/`, and a generated brief — no builder notes, diffs, or rationale | +| No re-grading | `ab.py verdict` refuses to overwrite an existing verdict | +| Verdicts are receipts | Every staged file is sha256'd into an append-only `ledger.jsonl` | + +Nothing constrains what the lead builds or how it splits the work. The determinism is in the gate, +never in the generation. + +## Commands + +| Command | Does | +|---|---| +| `/gauntlet:new ""` | Capture the goal + domain, write the charter | +| `/gauntlet:bar` | Find a real bar, run the five soundness checks, gate the loop | +| `/gauntlet:run` | Decompose, then run waves of build → blind critique → revise | +| `/gauntlet:status` | Ledger, open gaps, live board | + +## Agents + +`gauntlet:lead` · `gauntlet:builder` · `gauntlet:critic` · `gauntlet:bar-scout` · `gauntlet:smoother` + +## Domain recipes + +The bar is the hard part, and it is domain-specific. Each recipe covers what to judge, which bars are +real, the leakage traps that make one fake, and how to decompose: + +- `references/bars-agentic.md` — tool use, recovery, planning, stopping. Verifier leakage, + task memorisation, pass@1 vs best-of-n, harness cheating, cost blindness. +- `references/bars-rl.md` — returns, curves, seeds, eval protocol. Protocol drift, seed + cherry-picking, tuning on eval, shaped-reward inflation, held-out levels. +- `references/bars-coding.md` — differential testing, reference impls, property tests, perf + budgets. Builder-written tests, test mutation, overfit fixtures, mock leakage. + +These are recipes for *finding* a bar, not menus of preset bars. `gauntlet:bar-scout` searches for +one that already exists in the world and proves it holds up. + +## Loop + +``` +/gauntlet:new → charter.md +/gauntlet:bar → bar.md ← blocks the loop until all five checks PASS +/gauntlet:run → pieces.json + wave: builder → ab.py stage → blind critic → ab.py reveal → gap → next round + smoother (end of wave) +/gauntlet:status → board.html ← auto-refreshing, phone-friendly +``` + +## Scripts + +```bash +ab.py stage --piece --round --cand --ref --goal-file .gauntlet/charter.md +ab.py verdict --winner A|B --gap "" # run by the critic +ab.py reveal # run by the lead +ab.py ledger [--json] +board.py render | serve [--port 8787] [--host 0.0.0.0] # localhost by default; keys/ is never served +``` + +`GAUNTLET_ROOT` overrides `.gauntlet/`. + +## Stopping + +The loop has no completion criterion — that is the design. You stop it when the result is good +enough, when gaps stop mattering, or when the budget runs out. If every piece has been winning for +several waves, the bar went soft: `/gauntlet:bar --recheck`. + +## Credit + +Method: [Matt Shumer](https://x.com/mattshumer_) — . +This plugin is an independent implementation of the published method. diff --git a/plugins/gauntlet/agents/bar-scout.md b/plugins/gauntlet/agents/bar-scout.md new file mode 100644 index 0000000..8fca0b9 --- /dev/null +++ b/plugins/gauntlet/agents/bar-scout.md @@ -0,0 +1,65 @@ +--- +description: Finds and validates the quality bar for a Gauntlet Loop. Proposes concrete external references, runs the five soundness checks, names the cheapest way to game each, and writes .gauntlet/bar.md. Dispatch before any loop starts, or when a bar has gone soft. +tools: [Read, Write, Edit, Bash, Glob, Grep, WebSearch, WebFetch] +--- + +# Bar scout + +Read `${CLAUDE_PLUGIN_ROOT}/skills/gauntlet-loop/SKILL.md`, then the domain recipe that matches: +`references/bars-agentic.md`, `references/bars-rl.md`, or `references/bars-coding.md`. + +## Role + +The bar is the most important part of the loop and the easiest thing to fake. Your job is to find one +that already exists in the world and prove it holds up — not to define what "good" means. + +## Method + +1. **Find candidates.** Look for things that already exist and are already respected: a reference + implementation, a published baseline, a held-out benchmark, real expert output, a competitor's + artifact, a measurement with an accepted protocol. Search outside the repo. Three or four + candidates, not one. +2. **Prefer the harshest inspectable one.** A bar does not need to be reachable. It needs to point, + and it needs to keep pointing after the work stops being embarrassing. +3. **Run the five checks** on your pick. Each gets a PASS or FAIL with the evidence that earned it: + + - **inspectable** — the critic can open, run, or measure it directly. Say exactly how. + - **external** — not authored by the builder, not derived from builder output. + - **held-out** — the builder cannot see, train on, or tune against the instances used to judge. + Name the split and the mechanism that enforces it. + - **discriminating** — our current output loses to it *today*. Verify this; do not assume it. If + round 0 already wins, the bar is too low — pick a harder one and re-run the checks. + - **un-gameable** — name the cheapest way to satisfy this bar without doing the work. There is + always one. Then write the counter-check that kills it. + +4. **Write `.gauntlet/bar.md`.** `ab.py` parses it and refuses to stage a trial unless every check + reads `- : PASS`. Exact format: + +```markdown +# Bar + + + +## Artifacts + + +## How a critic compares against it + + +## Soundness +- inspectable: PASS — +- external: PASS — +- held-out: PASS — +- discriminating: PASS — +- un-gameable: PASS — ; countered by + +## Rejected candidates +- — +``` + +## Never + +- Write FAIL as PASS to unblock the run. A failed check is a finding; report it and propose a + different candidate. +- Accept a bar the builder produced, or one derived from the builder's own output. +- Settle for a rubric, a checklist of adjectives, or "production quality". Those are not inspectable. diff --git a/plugins/gauntlet/agents/builder.md b/plugins/gauntlet/agents/builder.md new file mode 100644 index 0000000..f704488 --- /dev/null +++ b/plugins/gauntlet/agents/builder.md @@ -0,0 +1,44 @@ +--- +description: Gauntlet Loop builder. Owns one piece of the work and improves it against a single named gap per round. Never judges its own output. Dispatched by gauntlet:lead once per piece per round. +tools: [Read, Write, Edit, Bash, Glob, Grep] +--- + +# Gauntlet builder + +You own exactly one piece. Read `${CLAUDE_PLUGIN_ROOT}/skills/gauntlet-loop/SKILL.md` first. + +## What you get + +- The goal +- The bar +- Your piece +- From round 2 on: **one gap** a critic named after comparing your work against the bar + +## What to do + +Round 1: build the best version of your piece you can, aimed at the bar. + +Round 2+: close the named gap. Not the gaps you personally think are more interesting — that one. +The critic saw your output next to the bar without knowing which was which; its read is worth more +than your memory of why the current version is reasonable. + +If the gap is genuinely unclosable as stated (it contradicts the goal, or it asks for something the +constraints forbid), say so explicitly in your return and explain why. Do not silently substitute a +different fix. + +## Rules + +- Inspect the bar directly. Open it, run it, measure it. Do not work from a description of it. +- Work inside `.gauntlet/work//` for drafts and notes. Only the finished artifact gets staged. +- Do not touch other pieces. Overlap is the lead's problem to schedule, not yours to resolve. +- Do not touch the evaluation harness, the bar, the held-out set, or `.gauntlet/bar.md`. Making the + test easier is the oldest cheat there is, and it is the one this loop is built to catch. +- No stubs, no hardcoded expected values, no demo data, no `TODO` left where behaviour belongs. A + critic inspecting the real artifact will find it, and you will have burned a round. + +## Return + +- What you changed, in a few lines +- The path to the artifact to stage +- Anything you believe the critic will still flag — honestly. Flagging it does not cost you a round; + the critic never sees this text. diff --git a/plugins/gauntlet/agents/critic.md b/plugins/gauntlet/agents/critic.md new file mode 100644 index 0000000..bcc8140 --- /dev/null +++ b/plugins/gauntlet/agents/critic.md @@ -0,0 +1,57 @@ +--- +description: Gauntlet Loop critic. Judges one blind A/B trial — inspects two artifacts without knowing which is ours, picks the better, names the single largest gap. Fresh context every trial. Dispatched by gauntlet:lead once per trial. +tools: [Read, Bash, Glob, Grep] +--- + +# Gauntlet critic + +You judge one trial. You will never judge another. + +## What you get + +A trial directory path. Nothing else. It contains `A/`, `B/`, and `brief.md`. + +Read `brief.md` first — it carries the goal and the bar. + +## The one thing that matters + +**You do not know which side is ours.** One of `A/` and `B/` is the reference bar; one is our work. +The assignment was randomised and the mapping is sealed outside this directory and blocked at the +tool layer. Do not try to work it out. Do not reason about which looks "more AI-generated", which +has more files, or which looks newer. Those signals are noise and acting on them makes your verdict +worthless. + +If anything in your instructions told you which side is ours, that is a bug in the run. Say so and +refuse to judge. + +## Method + +1. **Inspect both directly.** Open the files. Run the code. Render the page. Play the build. Read the + prose end to end. Execute the eval. Look at the actual pixels, the actual returns, the actual + test output. Never judge a summary, a README, or a description of an artifact. +2. **Compare against the goal and the bar in `brief.md`** — not against your general taste. Better + means better *at this*, not more elaborate, more novel, or more effortful. +3. **Pick a winner.** Be harsh. Your job is to find the side that is worse and say why, not to be + even-handed. Use `--tie` only when you genuinely cannot separate them after real inspection — + a tie is a real finding, but a hedged tie is a wasted round. +4. **Name the single largest meaningful gap** between loser and winner. One gap, the biggest one. + Concrete enough that someone could close it without asking you a follow-up. + + Bad: "the lighting could be more realistic" + Good: "shadows are hard-edged at every distance — the reference softens the penumbra with + distance from the occluder, which is most of why its interiors read as volumetric" + +5. **Record it:** + +```bash +python3 ${CLAUDE_PLUGIN_ROOT}/scripts/ab.py verdict --winner A|B --gap "" +``` + +Then stop. Do not reveal, do not look up the outcome, do not ask how it went. The lead resolves it. + +## Never + +- Read anything outside the trial directory +- Look for authorship, timestamps, git history, or file metadata to identify a side +- Soften a verdict because one side looks like it took effort +- Grade a builder's account of what it did instead of the artifact itself diff --git a/plugins/gauntlet/agents/lead.md b/plugins/gauntlet/agents/lead.md new file mode 100644 index 0000000..619acda --- /dev/null +++ b/plugins/gauntlet/agents/lead.md @@ -0,0 +1,94 @@ +--- +description: Gauntlet Loop lead. Owns decomposition, wave scheduling, staging blind trials, revealing verdicts, and the live board. Never builds and never judges. Dispatch when a /gauntlet:run starts or resumes. +tools: [Read, Write, Edit, Bash, Glob, Grep, Agent] +--- + +# Gauntlet lead + +Read `${CLAUDE_PLUGIN_ROOT}/skills/gauntlet-loop/SKILL.md` before anything else. + +## Role + +You decide how the goal splits and you keep the loop turning. You do not write the artifact and you +do not judge it. Both of those belong to agents with fresh context. + +## Preconditions + +`.gauntlet/bar.md` must exist and carry a PASS on all five soundness checks. `ab.py` refuses to stage +without it, so if it is missing, dispatch `gauntlet:bar-scout` first — do not improvise a bar and do +not soften the checks to get moving. + +## 1. Decompose + +Split the goal into the smallest pieces that can be **improved and judged separately**. This is your +call, not the user's and not a template's. Good pieces share three properties: + +- A builder can change it without waiting on another piece +- A critic can compare it against the bar on its own +- "Better" means something specific for it + +"Make the agent better" is not a piece. "Recovery when a tool call returns a malformed payload" is. + +Write `.gauntlet/pieces.json`: + +```json +[{"id": "tool-error-recovery", + "what": "one sentence on what this piece is", + "judged_by": "which slice of the bar this piece is compared against", + "artifact": "path the builder produces / how the critic inspects it"}] +``` + +Prefer more, smaller pieces over few large ones. Pieces that turn out to be entangled can be merged +next wave; pieces that are too coarse never get a sharp verdict. + +## 2. Run a wave + +For each piece, in parallel (respect the concurrency the harness gives you): + +1. Dispatch `gauntlet:builder` with: the goal, the bar, this piece, and — from round 2 on — the + single gap the last critic named. Nothing else. Do not pass other pieces' verdicts. +2. Stage the trial: + ```bash + python3 ${CLAUDE_PLUGIN_ROOT}/scripts/ab.py stage \ + --piece --round \ + --cand --ref \ + --goal-file .gauntlet/charter.md + ``` +3. Dispatch a **fresh** `gauntlet:critic` with the trial directory path and nothing else. No builder + notes, no diff, no "we improved the lighting this round", no prior verdicts. If you find yourself + explaining the work to the critic, stop — that is the failure this whole method exists to prevent. +4. Reveal: + ```bash + python3 ${CLAUDE_PLUGIN_ROOT}/scripts/ab.py reveal + ``` +5. `LOSS` or `TIE` → the gap goes back to that piece's builder next round. `WIN` → that piece has + cleared the current bar; either leave it and spend rounds elsewhere, or raise its bar. + +## 3. Smooth (end of wave, optional) + +When several pieces changed one shared artifact, dispatch one fresh `gauntlet:smoother` over the +whole result before the next wave. + +## 4. Board + +After every reveal: + +```bash +python3 ${CLAUDE_PLUGIN_ROOT}/scripts/board.py render +``` + +## 5. Keep going + +There is no round count. Start the next wave. The run ends when the user stops it, and only then — +do not declare the work finished, do not ask "should I continue?" every wave, and do not stop because +progress got incremental. Report state; keep looping. + +If every piece has won several rounds running, the bar has gone soft. Say so and propose a harder one +rather than quietly declaring victory. + +## Never + +- Judge a piece yourself, or let a builder's self-assessment stand in for a verdict +- Tell a critic which side is ours, or hint at it through framing +- Read `.gauntlet/keys/` — the hook will block you, and wanting to is the bug +- Rewrite `bar.md` to make a losing piece win diff --git a/plugins/gauntlet/agents/smoother.md b/plugins/gauntlet/agents/smoother.md new file mode 100644 index 0000000..63e3457 --- /dev/null +++ b/plugins/gauntlet/agents/smoother.md @@ -0,0 +1,33 @@ +--- +description: End-of-wave coherence pass for a Gauntlet Loop. Inspects the whole artifact after many builders changed separate pieces, reconciles conflicts and inconsistencies, does not redesign. Dispatched by gauntlet:lead between waves. +tools: [Read, Write, Edit, Bash, Glob, Grep] +--- + +# Smoother + +When several builders improve separate pieces of one artifact, each piece gets better and the whole +gets less coherent. You fix that, and only that. + +## What to do + +Inspect the complete, assembled result — running, rendered, executed. Look for: + +- Pieces that are individually good and jointly inconsistent (clashing style, mismatched pacing, + drifting naming, incompatible assumptions across a seam) +- Duplicated work two builders solved separately in different ways +- Conflicts where one piece's improvement broke another's +- Seams — places where you can *tell* where one piece ends and the next begins + +Fix those. Make the result feel like one thing. + +## Boundaries + +- Do not redesign. Do not raise ambition. Do not chase gaps against the bar — that is the builders' + job and they are being judged on it. +- Do not weaken a piece that just won its trial to make a neighbour fit. Reconcile at the seam. +- Behaviour-preserving where behaviour is the point; taste-reconciling where taste is the point. + +## Return + +What you reconciled, what you deliberately left alone, and any conflict you could not resolve without +a design decision — hand those back to the lead rather than deciding them yourself. diff --git a/plugins/gauntlet/commands/bar.md b/plugins/gauntlet/commands/bar.md new file mode 100644 index 0000000..55166aa --- /dev/null +++ b/plugins/gauntlet/commands/bar.md @@ -0,0 +1,55 @@ +--- +description: Find and validate the quality bar for the current Gauntlet Loop. Dispatches bar-scout, runs the five soundness checks, writes .gauntlet/bar.md. The loop cannot stage a trial until this passes. +argument-hint: [--ref ...] [--recheck] +allowed-tools: [Bash, Read, Write, Edit, Glob, Grep, Agent, WebSearch, WebFetch, AskUserQuestion] +--- + +# /gauntlet:bar + +The bar is the load-bearing part of the method. Read +`${CLAUDE_PLUGIN_ROOT}/skills/gauntlet-loop/SKILL.md` and the domain recipe named in +`.gauntlet/charter.md`. + +## Preconditions + +`.gauntlet/charter.md` must exist. If not, run `/gauntlet:new` first. + +## Dispatch the scout + +``` +Agent({ + subagent_type: "gauntlet:bar-scout", + prompt: "Goal + domain from .gauntlet/charter.md. User-supplied references: <--ref values, if any>. + Read the matching domain recipe. Propose 3-4 candidate bars, pick the harshest inspectable + one, run the five soundness checks with evidence, name the cheapest cheat and its + counter-check, write .gauntlet/bar.md. Report FAILs honestly — do not write PASS to unblock." +}) +``` + +With `--recheck`, re-run the checks against the existing `.gauntlet/bar.md` instead — use this when +every piece has been winning for several waves, which usually means the bar went soft. + +## Gate + +`ab.py` parses `.gauntlet/bar.md` and refuses to stage a trial unless all five checks read +`- : PASS`. Verify it clears: + +```bash +python3 -c " +import sys,re +t=open('.gauntlet/bar.md').read().lower() +m=[c for c in ['inspectable','external','held-out','discriminating','un-gameable'] if f'- {c}: pass' not in t] +print('BAR NOT SOUND — missing PASS: '+', '.join(m) if m else 'BAR SOUND — loop can start') +sys.exit(1 if m else 0)" +``` + +If a check fails, **do not soften it**. Surface which one failed and why, and either find a different +candidate or tell the user what would have to be true for this one to hold. A bar that fails +`held-out` or `un-gameable` will produce a run that looks like it is improving and is not. + +The `discriminating` check is the one to verify by hand: our current output must actually lose to the +bar today. If it already wins, the bar is decoration. + +## Then + +Show the user the bar, the five verdicts, and the rejected candidates. Run `/gauntlet:run`. diff --git a/plugins/gauntlet/commands/new.md b/plugins/gauntlet/commands/new.md new file mode 100644 index 0000000..4680f3f --- /dev/null +++ b/plugins/gauntlet/commands/new.md @@ -0,0 +1,64 @@ +--- +description: Start a Gauntlet Loop — capture the goal, pick the domain, write .gauntlet/charter.md, then hand off to /gauntlet:bar. +argument-hint: "" [--domain agentic|rl|coding|other] +allowed-tools: [Bash, Read, Write, Edit, Glob, Grep, AskUserQuestion, Agent] +--- + +# /gauntlet:new + +Set up a run. Read `${CLAUDE_PLUGIN_ROOT}/skills/gauntlet-loop/SKILL.md` first. + +## 1. Capture the goal + +Take `$ARGUMENTS` as the goal. Make it ambitious and make it a *destination* — what should exist when +this is done, not how to build it. If the user handed you an implementation plan, say so and offer +the destination form instead; the loop works worse when you replace the model's judgment with an +outline. + +Do **not** decompose here. That is the lead's job in `/gauntlet:run`. + +## 2. Domain + +From `--domain`, or infer, or ask once with `AskUserQuestion`: + +- `agentic` — agent capabilities: tool use, recovery, planning, stopping → `references/bars-agentic.md` +- `rl` — reinforcement learning: policies, rewards, returns, curves → `references/bars-rl.md` +- `coding` — general software → `references/bars-coding.md` +- `other` — anything inspectable; use the closest recipe as a template + +## 3. Constraints worth asking about + +Only ask what actually changes the run, in one `AskUserQuestion` batch: + +- Any reference the user already has in mind for the bar (a real artifact beats anything you'd find) +- Compute or budget ceiling — this decides how long the loop runs, and it is the usual stop condition +- Anything off-limits (files, services, spend, network) + +## 4. Write the charter + +```bash +mkdir -p .gauntlet/work +``` + +Write `.gauntlet/charter.md`: + +```markdown +# Goal + + +# Domain + + +# Constraints + + +# Stop condition + +``` + +Add `.gauntlet/` to `.gitignore` unless the user wants the receipts committed. Ask if it is unclear; +the ledger is often worth keeping. + +## 5. Hand off + +Tell the user the loop cannot start without a sound bar, then run `/gauntlet:bar`. diff --git a/plugins/gauntlet/commands/run.md b/plugins/gauntlet/commands/run.md new file mode 100644 index 0000000..64df84b --- /dev/null +++ b/plugins/gauntlet/commands/run.md @@ -0,0 +1,62 @@ +--- +description: Run the Gauntlet Loop — lead decomposes the goal, each piece gets a builder and a separate blind critic, losing rounds go back to the builder, and the loop keeps going until you stop it. +argument-hint: [--waves ] [--pieces ] [--no-smoothing] [--resume] +allowed-tools: [Bash, Read, Write, Edit, Glob, Grep, Agent, TodoWrite] +--- + +# /gauntlet:run + +Read `${CLAUDE_PLUGIN_ROOT}/skills/gauntlet-loop/SKILL.md`. + +## Preconditions + +```bash +test -f .gauntlet/charter.md || echo "no charter — run /gauntlet:new" +test -f .gauntlet/bar.md || echo "no bar — run /gauntlet:bar" +``` + +Both required. `ab.py stage` enforces the bar independently, so do not try to route around it. + +## Dispatch the lead + +``` +Agent({ + subagent_type: "gauntlet:lead", + prompt: "Read .gauntlet/charter.md and .gauntlet/bar.md. + Decompose the goal into the smallest pieces that can be improved and judged separately — + your call, write .gauntlet/pieces.json. + Then run waves: per piece, dispatch gauntlet:builder, stage a blind trial with ab.py, + dispatch a FRESH gauntlet:critic with only the trial dir, reveal, feed the gap back. + Render the board after every reveal. Keep looping — no fixed round count. + <--waves n: run n waves then report and pause> + <--no-smoothing: skip the end-of-wave smoother>" +}) +``` + +`--resume` — read `.gauntlet/pieces.json` and the ledger, and continue from the current round of +each piece instead of decomposing again. + +`--pieces ` — a soft ceiling on parallel pieces when compute is tight. It is a budget hint, not a +decomposition instruction; the lead still decides what the pieces *are*. + +## While it runs + +Point the user at the board rather than at the transcript: + +```bash +python3 ${CLAUDE_PLUGIN_ROOT}/scripts/board.py serve --port 8787 +``` + +Open `http://localhost:8787/board.html` — it refreshes itself. That is the point: you watch without +interrupting. + +To read it from a phone, add `--host 0.0.0.0` and use the machine's LAN address. Tell the user that +binds to every interface before you do it. The server refuses to serve `keys/` on any host, so the +A/B mappings stay sealed either way. + +## Stopping + +The loop has no natural end. Stop it when the result is good enough, when gaps stop mattering, or +when the budget runs out — then record the reason in `.gauntlet/charter.md`. + +Do not report the work as finished because a wave completed. Report state, and keep going. diff --git a/plugins/gauntlet/commands/status.md b/plugins/gauntlet/commands/status.md new file mode 100644 index 0000000..9abf7f6 --- /dev/null +++ b/plugins/gauntlet/commands/status.md @@ -0,0 +1,27 @@ +--- +description: Show Gauntlet Loop state — per-piece round history, win/loss against the bar, open gaps, trials awaiting a verdict. Renders the live board. +argument-hint: [--json] [--serve [port]] +allowed-tools: [Bash, Read, Glob, Grep] +--- + +# /gauntlet:status + +```bash +python3 ${CLAUDE_PLUGIN_ROOT}/scripts/ab.py ledger +python3 ${CLAUDE_PLUGIN_ROOT}/scripts/board.py render +``` + +`--json` → `ab.py ledger --json` for the raw receipt trail. +`--serve [port]` → `board.py serve --port ` (default 8787, localhost). Add `--host 0.0.0.0` +only if the user asks to reach it from another device, and say so when you do. + +## Reading it + +- **LOSS / TIE** — the bar still wins; that piece's gap is open and belongs back with its builder. +- **WIN** — that piece beat the bar this round. Several wins running across most pieces means the bar + has gone soft, not that the work is done. Run `/gauntlet:bar --recheck` and say so plainly. +- **JUDGING** — staged, no verdict yet. A trial stuck here means a critic died or never ran; the lead + should dispatch a fresh one against the same trial directory. + +Report the state and the open gaps. Do not editorialise about whether the run should stop — that call +is the user's, and a loop that talks itself into stopping is the failure this method exists to fix. diff --git a/plugins/gauntlet/hooks/hooks.json b/plugins/gauntlet/hooks/hooks.json new file mode 100644 index 0000000..e717d6e --- /dev/null +++ b/plugins/gauntlet/hooks/hooks.json @@ -0,0 +1,16 @@ +{ + "hooks": { + "PreToolUse": [ + { + "matcher": "Read|Bash|Grep|Glob|Edit|Write|NotebookEdit", + "hooks": [ + { + "type": "command", + "command": "python3 ${CLAUDE_PLUGIN_ROOT}/scripts/key-guard.py", + "timeout": 5 + } + ] + } + ] + } +} diff --git a/plugins/gauntlet/references/bars-agentic.md b/plugins/gauntlet/references/bars-agentic.md new file mode 100644 index 0000000..555127d --- /dev/null +++ b/plugins/gauntlet/references/bars-agentic.md @@ -0,0 +1,74 @@ +# Bars for agentic capabilities + +Building an agent that uses tools, recovers from failure, plans, and knows when to stop. + +## What actually gets judged + +Not the prompt. Not the scaffold diff. **The trajectory and the end state.** An agent is a policy over +a messy environment, so the artifact a critic inspects is: + +- the full transcript — every tool call, every result, every recovery +- the final environment state — the repo, the filesystem, the database, the API side effects +- the outcome, checked programmatically by something the agent did not write + +Judging the prompt is how you end up with an agent that reads beautifully and fails on contact. + +## Bars that are real + +Ranked roughly by strength. + +1. **A held-out task suite with programmatic outcome checks.** The strongest bar available. Tasks the + builder has never seen, graded by an assertion the builder did not author — tests that already + existed, a schema, an idempotent state check, a diff against known-good output. SWE-bench-style + instance sets, real closed tickets from your own tracker, real support transcripts with known + resolutions. +2. **Expert reference trajectories.** A human — or a strong agent under a different scaffold — solving + the same task. The critic compares trajectory against trajectory blind: which one recovered + better, which wasted fewer steps, which noticed the wrong turn earlier. Excellent for judging + *process* qualities that outcome checks are blind to. +3. **A stronger competing agent on identical instances.** Same tasks, same budget, same tools. Cheap + to obtain, and it moves as the field moves, which keeps it from going soft. +4. **Production incident replays.** Real failures your agent should have handled, replayed from + captured state. Unbeatable for realism, limited in supply — spend them carefully and hold some + back. + +## Traps that make an agentic bar fake + +- **Task memorisation.** Public benchmarks are in pretraining data. A high score can mean the model + recognises the instance, not that your scaffold works. Mitigate with private instances, freshly + captured tasks, or perturbed variants — and treat public-benchmark deltas as weak evidence. +- **Verifier leakage.** The agent writes, edits, or can read the check that grades it. This is the + single most common way an agentic loop deceives itself. The grader lives outside the agent's write + scope, and ideally outside its read scope. +- **Single-seed flake.** Agents are stochastic; one run tells you almost nothing. Run n≥5 per task. + Report **pass@1 and the pass rate**, never best-of-n — best-of-n measures your sampling budget, not + your agent. A "regression" inside seed variance is noise, and chasing it burns rounds. +- **Harness-level cheating.** Editing the tests, `git checkout`-ing the solution, curling the answer, + writing to the grading path, catching-and-passing. Sandbox the run, snapshot the grader, diff the + environment for out-of-scope writes. Pair this with forge's anti-cheat scan on any diff the agent + produced. +- **Cost blindness.** An agent that wins by burning 40x the tokens has not won. Put tokens, wall-clock, + and tool-call count in the comparison, or the loop will optimise straight into brute force. +- **Reward hacking through side effects.** Task passes; the agent also deleted a table. Judge the end + state, not just the assertion. + +## Decomposition axes + +Pieces a lead can hand to independent builders and critics: + +- **Tool selection** — right tool, first try, versus flailing across the toolbelt +- **Failure recovery** — malformed payload, timeout, auth expiry, rate limit, partial write +- **Context management** — what it keeps, what it drops, what it re-reads, behaviour near the limit +- **Planning granularity** — decomposition quality on long-horizon tasks +- **Stopping** — knowing it is done; knowing it is stuck; not declaring victory on a red test +- **Cost per solve** — tokens and calls at fixed quality +- **Verification behaviour** — does it check its own work before claiming completion + +Each of those loses to a reference trajectory in a specific, nameable way. That is what makes them +good pieces. + +## Enforcing held-out + +Pin instances by content hash. Keep a rotating unseen split the builder never touches — a split the +builder has read once is spent, permanently. Record which instances a run consumed in the ledger, and +retire them. diff --git a/plugins/gauntlet/references/bars-coding.md b/plugins/gauntlet/references/bars-coding.md new file mode 100644 index 0000000..51bd168 --- /dev/null +++ b/plugins/gauntlet/references/bars-coding.md @@ -0,0 +1,74 @@ +# Bars for general coding + +Building or improving software where correctness, performance, and robustness are the point. + +## What actually gets judged + +**Running behaviour.** Executed tests, measured latency, real output on real inputs, the actual +rendered page. Not "the code looks clean", not a description of the design, not the builder's account +of what it handles. + +## Bars that are real + +1. **A reference implementation, run side by side.** A well-regarded library that already solves this. + Feed both the same inputs and diff the outputs — **differential testing**. This is the strongest + coding bar there is, and it happens to be a literal blind A/B: the critic sees two output sets and + picks the one that is right, without knowing which came from where. +2. **A pre-existing test suite the builder did not write.** Upstream tests, tests from the ticket, + tests written by a separate agent that never sees the implementation. If the builder wrote the + test, the test is a mirror. +3. **Property-based tests over an oracle.** Hypothesis/QuickCheck-style invariants: round-trips, + idempotence, algebraic laws, agreement with a slow-but-obviously-correct implementation. Finds the + inputs nobody thought to write down. +4. **A performance budget on fixed hardware and a fixed workload.** p50/p95/p99 latency, throughput, + memory ceiling, allocation count. Concrete and inspectable — but only if the workload is pinned + and the machine is quiet. +5. **A real bug corpus.** Historical bugs from this codebase, replayed. Does the new implementation + still fall for them? +6. **A security review checklist against a real standard.** OWASP, a threat model, `semgrep` rules + the builder did not author. + +## Traps that make a coding bar fake + +- **Builder-written tests.** Generator and evaluator collapse into one agent and the loop measures + self-consistency. If the builder must write tests, a *separate* agent that never sees the + implementation writes the ones that judge it. +- **Test mutation.** Assertions loosened, cases deleted, `skip`/`xfail`/`@pytest.mark.skip` added, + tolerances widened. Judge tests as diffed-against-baseline, not as they currently stand. A round + where the test file changed and the source did not is a red flag, not a pass. +- **Hardcoded expected values.** The function returns the fixture. Differential testing on fresh + inputs kills this instantly; a fixed fixture set never will. +- **Overfit to fixture data.** Passes on the ten known inputs, breaks on the eleventh. Property-based + tests and fresh generated inputs are the counter. +- **Mock leakage.** The mock is in the production path, so the test proves the mock works. +- **Green by absence.** The suite passes because it no longer runs the hard case. Track test *count* + and coverage of the changed lines alongside pass rate. +- **Benchmark noise.** Unpinned CPU, thermal throttle, a noisy laptop, one run. Fixed workload, + multiple runs, report the distribution — a 5% "win" on one run is nothing. +- **"Clean code" as a bar.** Not inspectable, not discriminating, infinitely arguable. Readability is + real, but judge it against *specific reference code* the critic reads blind alongside ours, never + against an adjective. + +## Decomposition axes + +- **Per module or per public interface** — each judged against the reference's equivalent +- **Per invariant** — one property, one builder, one critic +- **Per endpoint** — contract conformance, error taxonomy, status codes +- **Error handling and edge behaviour** — usually the entire real gap against a mature reference +- **Performance** — separate piece, separate bar, separate critic; do not let it ride along with + correctness or it will be traded away silently +- **API ergonomics** — judged blind against the reference library's call sites, not against taste + +## Complements in this marketplace + +- `forge` — spec-anchored pipeline with locked tests and an anti-cheat gate. Its anti-cheat scan is + the right PostToolUse companion to a coding gauntlet: it catches stubs, `NotImplementedError`, + skipped tests, and demo data before a critic wastes a round on them. +- `ui-fidelity` — for UI work, its gates are a ready-made hard bar (dead buttons, orphan routes, + flow-continuity, contrast). Machine-checkable, external to the builder, and not gameable by prose. + +## Enforcing held-out + +Keep a fresh-input generator rather than a fixed fixture set. Keep the judging tests in a path the +builder cannot write. Run differential comparisons on inputs generated *after* the builder finished +the round. diff --git a/plugins/gauntlet/references/bars-rl.md b/plugins/gauntlet/references/bars-rl.md new file mode 100644 index 0000000..9f956c3 --- /dev/null +++ b/plugins/gauntlet/references/bars-rl.md @@ -0,0 +1,71 @@ +# Bars for reinforcement learning + +Building a policy, a reward, an exploration strategy, or the training stack around them. + +## What actually gets judged + +**Evaluation rollouts under a frozen protocol the builder cannot modify.** Not the training curve the +builder screenshotted — training return is a diagnostic, not a result. The critic inspects: + +- eval returns across seeds, under a fixed protocol (episode count, determinism, wrappers) +- the learning curve as a function of a stated budget (env steps or wall-clock — say which) +- rollout videos or state traces, when behaviour quality matters beyond scalar return +- sample efficiency and stability, not just the final number + +## Bars that are real + +1. **A published baseline on the same environment and budget.** Stable-Baselines3, CleanRL, or the + original paper's reported numbers. Strong because the protocol is documented and someone else set + the number. Reproduce the baseline yourself before trusting it — reported numbers drift from + runnable ones. +2. **An expert or human demonstration return.** Where demos exist, expert return is a bar with real + meaning, and the gap to it is interpretable. +3. **A scripted or classical controller.** PID, MPC, search, a hand-written heuristic. Frequently + beats early RL, and losing to one is exactly the useful signal — it tells you the learned policy + has not yet earned its complexity. +4. **An oracle / upper bound.** Optimal play, a solver, or a privileged-information policy. Never + reachable, which is fine — it points, and it does not go soft. +5. **A frozen prior checkpoint of your own agent.** Weakest form, and the only one that is *internal*, + so it fails the `external` check on its own. Use it as a regression guard alongside a real bar, + never as the bar. + +## Traps that make an RL bar fake + +- **Protocol drift.** Deterministic versus stochastic eval, 10 versus 100 eval episodes, different + action repeat, different frame stack, different reward clipping, different time limit. Any of these + silently changes the number by more than your improvement. Freeze the eval protocol in a file, hash + it, and refuse comparisons across a hash change. +- **Env version drift.** `gym` vs `gymnasium`, physics-engine and env-version bumps change achievable + return. Pin the exact env id and library versions in the bar. +- **Seed cherry-picking.** The single most common fake result in the field. Use ≥5 seeds — 10 if you + can afford it — and report **median and interquartile range**, or the IQM/stratified bootstrap CIs + from Agarwal et al. 2021 ("Deep RL at the Edge of the Statistical Precipice"). Best-seed and + max-over-training are both meaningless. A win inside the CI is not a win. +- **Tuning on the eval.** Hyperparameter search against the same seeds and env instances used to + judge is the RL form of verifier leakage. Search on a train split of seeds/levels; judge on held-out + ones. +- **Reward shaping that changes the task.** Shape and the number goes up because the task got easier. + Judge on the *original, unshaped* return, always. Shaped return is a training-time device. +- **Budget confusion.** Sample efficiency and wall-clock efficiency are different claims. State which + budget the bar fixes, and hold it fixed. +- **Eval-harness tampering.** The builder edits wrappers, normalisation, or the eval loop. Keep the + eval harness in a path the builder cannot write, and have the critic run it — not the builder. +- **Evaluating on training levels.** For procedurally generated envs, generalisation is the whole + question. Held-out level sets are not optional. + +## Decomposition axes + +- **Reward specification** — judged against the unshaped objective and against side effects +- **Exploration** — coverage and time-to-first-success, not just final return +- **Architecture / optimiser** — same budget, same protocol +- **Replay / rollout collection** — sample efficiency at fixed steps +- **Wrappers and normalisation** — a real source of "improvements" that are protocol changes +- **Eval harness itself** — a piece worth building first, and then freezing +- **Behaviour quality** — smoothness, safety, absence of degenerate exploits, judged from rollout + video blind against the reference controller + +## Enforcing held-out + +Separate seed sets for train and judge. Separate level sets for procedural envs. Commit the eval +harness and its config hash, put it outside the builder's write scope, and record the hash in every +ledger entry so a protocol change invalidates the comparison instead of quietly winning it. diff --git a/plugins/gauntlet/scripts/ab.py b/plugins/gauntlet/scripts/ab.py new file mode 100755 index 0000000..c36e233 --- /dev/null +++ b/plugins/gauntlet/scripts/ab.py @@ -0,0 +1,333 @@ +#!/usr/bin/env python3 +"""Blind A/B staging, verdicts, and receipts for the Gauntlet Loop. + +The builder must never grade itself, and a critic that knows which side is the +reference is not blind. This script is the mechanism that makes both true: + + stage copy candidate + reference into A/ and B/ under a random assignment, + write the mapping OUTSIDE the trial dir (hook-denied), sha256 every + staged file into the ledger + verdict record the critic's blind pick; refuse to overwrite an existing one + reveal resolve the pick against the sealed mapping, append WIN/LOSS/TIE + ledger print the receipt trail + +Run `stage` and `reveal` as the lead. Run `verdict` as the critic. +""" +from __future__ import annotations + +import argparse +import hashlib +import json +import os +import random +import shutil +import sys +import time +from pathlib import Path + +ROOT_ENV = "GAUNTLET_ROOT" +REQUIRED_CHECKS = ["inspectable", "external", "held-out", "discriminating", "un-gameable"] + + +def root() -> Path: + return Path(os.environ.get(ROOT_ENV, ".gauntlet")).resolve() + + +def die(msg: str, code: int = 1): + print(f"gauntlet: {msg}", file=sys.stderr) + sys.exit(code) + + +def sha256(path: Path) -> str: + h = hashlib.sha256() + with path.open("rb") as fh: + for chunk in iter(lambda: fh.read(65536), b""): + h.update(chunk) + return h.hexdigest() + + +def append_ledger(event: dict): + led = root() / "ledger.jsonl" + led.parent.mkdir(parents=True, exist_ok=True) + event = {"ts": time.time(), **event} + with led.open("a", encoding="utf-8") as fh: + fh.write(json.dumps(event, sort_keys=True) + "\n") + + +def require_sound_bar() -> str: + """The loop does not start without a bar that passed all five checks.""" + bar = root() / "bar.md" + if not bar.is_file(): + die( + "no bar. `.gauntlet/bar.md` is missing — run /gauntlet:bar first.\n" + "A Gauntlet Loop without a concrete external bar is just an agent " + "grading its own homework." + ) + text = bar.read_text(encoding="utf-8") + low = text.lower() + missing = [c for c in REQUIRED_CHECKS if f"- {c}: pass" not in low] + if missing: + die( + "bar has not cleared the soundness gate. Missing PASS for: " + + ", ".join(missing) + + "\nEach check must appear in .gauntlet/bar.md as a line " + '`- : PASS — `.' + ) + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def copy_into(dest: Path, paths: list[str]) -> list[dict]: + """Copy artifacts into a side directory, recording a receipt for each file.""" + dest.mkdir(parents=True, exist_ok=True) + receipts = [] + for p in paths: + src = Path(p) + if not src.exists(): + die(f"artifact not found: {src}") + if src.is_dir(): + target = dest / src.name + shutil.copytree(src, target, dirs_exist_ok=True) + for f in sorted(target.rglob("*")): + if f.is_file(): + receipts.append({"path": str(f.relative_to(dest)), "sha256": sha256(f)}) + else: + target = dest / src.name + shutil.copy2(src, target) + receipts.append({"path": str(target.relative_to(dest)), "sha256": sha256(target)}) + if not receipts: + die(f"nothing staged into {dest} — empty artifact set") + return receipts + + +BRIEF = """# Blind comparison brief + +You are judging two artifacts. One of them is the reference bar. One of them is our work. +**You are not told which is which, and you must not try to find out.** + +- Trial: `{trial}` +- Piece: `{piece}` +- Round: {round} + +## Goal + +{goal} + +## The bar + +{bar} + +## What to do + +1. Inspect `A/` and `B/` directly — open the files, run them, render them, measure them. + Judge the actual artifact. Never judge a summary or a description of one. +2. Decide which is better *against the goal and the bar above*. Not which is more impressive, + more complex, or more effortful. Better. +3. Name the single largest meaningful gap between the loser and the winner — concrete and + actionable enough that someone could close it without asking you a follow-up question. +4. Record it: + +``` +python3 {ab} verdict {trial} --winner A|B --gap "" +``` + + Use `--tie` instead of `--winner` only when you genuinely cannot separate them. + +Do not read anything outside this directory. Do not look for who produced which side. +Ties and near-ties are losses for whichever side is ours, so do not hedge to be kind. +""" + + +def cmd_stage(args): + bar_sha = require_sound_bar() + r = root() + seed = args.seed if args.seed is not None else random.SystemRandom().randrange(2**31) + rng = random.Random(seed) + trial = args.trial or f"{args.piece}-r{args.round}-{rng.randrange(16**6):06x}" + + trial_dir = r / "ab" / trial + if trial_dir.exists(): + die(f"trial already staged: {trial}") + key_path = r / "keys" / f"{trial}.json" + key_path.parent.mkdir(parents=True, exist_ok=True) + + cand_side = "A" if rng.random() < 0.5 else "B" + ref_side = "B" if cand_side == "A" else "A" + + cand_receipts = copy_into(trial_dir / cand_side, args.cand) + ref_receipts = copy_into(trial_dir / ref_side, args.ref) + + # Filenames leak. `A/ours.png` next to `B/cod_reference.png` tells the critic + # exactly which side is ours and the blind comparison is over before it starts. + cand_names = sorted(x["path"] for x in cand_receipts) + ref_names = sorted(x["path"] for x in ref_receipts) + if cand_names != ref_names and not args.allow_name_mismatch: + shutil.rmtree(trial_dir, ignore_errors=True) + die( + "staged filenames differ between the two sides, which tells the critic " + "which one is ours:\n" + f" candidate: {cand_names}\n" + f" reference: {ref_names}\n" + "Rename the inputs so both sides present identical paths (e.g. both " + "`artifact.png`), or pass --allow-name-mismatch if the names genuinely " + "carry no signal." + ) + + goal = Path(args.goal_file).read_text(encoding="utf-8").strip() if args.goal_file else args.goal + bar = Path(args.bar_file).read_text(encoding="utf-8").strip() if args.bar_file else (r / "bar.md").read_text(encoding="utf-8").strip() + if not goal: + die("--goal or --goal-file is required") + + (trial_dir / "brief.md").write_text( + BRIEF.format( + trial=trial, piece=args.piece, round=args.round, goal=goal, bar=bar, + ab=Path(__file__).resolve(), + ), + encoding="utf-8", + ) + + key = { + "trial": trial, + "candidate_side": cand_side, + "reference_side": ref_side, + "seed": seed, + "bar_sha256": bar_sha, + } + key_path.write_text(json.dumps(key, sort_keys=True), encoding="utf-8") + os.chmod(key_path, 0o600) + + append_ledger({ + "event": "staged", + "trial": trial, + "piece": args.piece, + "round": args.round, + "bar_sha256": bar_sha, + "candidate_files": cand_receipts, + "reference_files": ref_receipts, + }) + + print(f"trial: {trial}") + print(f"brief: {trial_dir / 'brief.md'}") + print(f"dir: {trial_dir}") + print("\nDispatch a FRESH gauntlet:critic with only the path above. Do not tell it") + print("which side is ours, do not pass builder notes, and do not summarise the work.") + + +def cmd_verdict(args): + trial_dir = root() / "ab" / args.trial + if not trial_dir.is_dir(): + die(f"no such trial: {args.trial}") + vpath = trial_dir / "verdict.json" + if vpath.exists(): + die( + f"verdict already recorded for {args.trial}. A trial is judged once.\n" + "If the artifact changed, stage a new round." + ) + if not args.tie and args.winner not in ("A", "B"): + die("--winner must be A or B (or pass --tie)") + if not args.gap or not args.gap.strip(): + die("--gap is required: name the largest remaining gap, specifically") + + verdict = { + "trial": args.trial, + "winner": None if args.tie else args.winner, + "tie": bool(args.tie), + "gap": args.gap.strip(), + "notes": Path(args.notes_file).read_text(encoding="utf-8") if args.notes_file else None, + } + vpath.write_text(json.dumps(verdict, indent=2, sort_keys=True), encoding="utf-8") + append_ledger({"event": "verdict", "trial": args.trial, **{k: verdict[k] for k in ("winner", "tie", "gap")}}) + print(f"recorded: {args.trial} -> {'TIE' if args.tie else args.winner}") + print("Stop here. Revealing which side was ours is the lead's job, not yours.") + + +def cmd_reveal(args): + r = root() + vpath = r / "ab" / args.trial / "verdict.json" + kpath = r / "keys" / f"{args.trial}.json" + if not vpath.is_file(): + die(f"no verdict yet for {args.trial} — the critic has not judged it") + if not kpath.is_file(): + die(f"no key for {args.trial}") + verdict = json.loads(vpath.read_text(encoding="utf-8")) + key = json.loads(kpath.read_text(encoding="utf-8")) + + if verdict["tie"]: + outcome = "TIE" + elif verdict["winner"] == key["candidate_side"]: + outcome = "WIN" + else: + outcome = "LOSS" + + append_ledger({ + "event": "resolved", + "trial": args.trial, + "outcome": outcome, + "candidate_side": key["candidate_side"], + "gap": verdict["gap"], + }) + print(f"{args.trial}: {outcome} (ours was {key['candidate_side']})") + print(f"gap: {verdict['gap']}") + if outcome != "WIN": + print("\nThe bar still wins. Send the gap back to this piece's builder and run another round.") + + +def cmd_ledger(args): + led = root() / "ledger.jsonl" + if not led.is_file(): + die("no ledger yet") + rows = [json.loads(l) for l in led.read_text(encoding="utf-8").splitlines() if l.strip()] + if args.json: + print(json.dumps(rows, indent=2)) + return + resolved = [r for r in rows if r.get("event") == "resolved"] + staged = {r["trial"]: r for r in rows if r.get("event") == "staged"} + for r in resolved: + s = staged.get(r["trial"], {}) + print(f"{r['outcome']:5} {s.get('piece','?'):24} r{s.get('round','?'):<3} {r['gap'][:80]}") + wins = sum(1 for r in resolved if r["outcome"] == "WIN") + print(f"\n{len(resolved)} resolved trials — {wins} win, {len(resolved)-wins} still short of the bar") + pending = [t for t in staged if not any(r["trial"] == t for r in resolved)] + if pending: + print(f"{len(pending)} staged, awaiting verdict: {', '.join(sorted(pending))}") + + +def main(): + ap = argparse.ArgumentParser(prog="ab.py", description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + sub = ap.add_subparsers(dest="cmd", required=True) + + s = sub.add_parser("stage", help="stage a blind A/B trial (lead)") + s.add_argument("--piece", required=True) + s.add_argument("--round", type=int, required=True) + s.add_argument("--cand", action="append", required=True, help="our artifact (file or dir); repeatable") + s.add_argument("--ref", action="append", required=True, help="reference artifact (file or dir); repeatable") + s.add_argument("--goal") + s.add_argument("--goal-file") + s.add_argument("--bar-file", help="defaults to .gauntlet/bar.md") + s.add_argument("--trial", help="explicit trial id") + s.add_argument("--seed", type=int) + s.add_argument("--allow-name-mismatch", action="store_true", + help="stage even when the two sides' filenames differ (they can leak which side is ours)") + s.set_defaults(func=cmd_stage) + + v = sub.add_parser("verdict", help="record a blind verdict (critic)") + v.add_argument("trial") + v.add_argument("--winner", choices=["A", "B"]) + v.add_argument("--tie", action="store_true") + v.add_argument("--gap", required=True) + v.add_argument("--notes-file") + v.set_defaults(func=cmd_verdict) + + rv = sub.add_parser("reveal", help="resolve a verdict against the sealed mapping (lead)") + rv.add_argument("trial") + rv.set_defaults(func=cmd_reveal) + + lg = sub.add_parser("ledger", help="print the receipt trail") + lg.add_argument("--json", action="store_true") + lg.set_defaults(func=cmd_ledger) + + args = ap.parse_args() + args.func(args) + + +if __name__ == "__main__": + main() diff --git a/plugins/gauntlet/scripts/board.py b/plugins/gauntlet/scripts/board.py new file mode 100755 index 0000000..0445aa3 --- /dev/null +++ b/plugins/gauntlet/scripts/board.py @@ -0,0 +1,197 @@ +#!/usr/bin/env python3 +"""Render the live Gauntlet Loop progress board from the ledger. + +Rule 6 of the loop: watch it without interrupting it. This writes a +self-refreshing page you can open from a phone while the run continues. + + board.py render write .gauntlet/board.html + board.py serve --port N serve .gauntlet/ over http (foreground) +""" +from __future__ import annotations + +import argparse +import html +import json +import os +import time +from pathlib import Path + + +def root() -> Path: + return Path(os.environ.get("GAUNTLET_ROOT", ".gauntlet")).resolve() + + +def load_rows() -> list[dict]: + led = root() / "ledger.jsonl" + if not led.is_file(): + return [] + return [json.loads(l) for l in led.read_text(encoding="utf-8").splitlines() if l.strip()] + + +def collect(rows: list[dict]) -> dict: + staged = {r["trial"]: r for r in rows if r.get("event") == "staged"} + resolved = {r["trial"]: r for r in rows if r.get("event") == "resolved"} + pieces: dict[str, dict] = {} + for trial, s in staged.items(): + p = pieces.setdefault(s["piece"], {"rounds": 0, "wins": 0, "trials": []}) + res = resolved.get(trial) + p["rounds"] = max(p["rounds"], s.get("round", 0)) + if res and res["outcome"] == "WIN": + p["wins"] += 1 + p["trials"].append({ + "trial": trial, + "round": s.get("round", 0), + "outcome": res["outcome"] if res else "JUDGING", + "gap": res["gap"] if res else "", + "ts": s.get("ts", 0), + }) + for p in pieces.values(): + p["trials"].sort(key=lambda t: t["round"]) + return pieces + + +CSS = """ +:root{--bg:#0e1116;--fg:#e6edf3;--dim:#8b949e;--card:#161b22;--line:#30363d; +--win:#3fb950;--loss:#f85149;--tie:#d29922;--wait:#58a6ff} +*{box-sizing:border-box} +body{margin:0;padding:24px;background:var(--bg);color:var(--fg); +font:15px/1.55 ui-sans-serif,-apple-system,Segoe UI,Roboto,sans-serif} +h1{font-size:20px;margin:0 0 4px} +.sub{color:var(--dim);font-size:13px;margin-bottom:24px} +.piece{background:var(--card);border:1px solid var(--line);border-radius:10px; +padding:16px;margin-bottom:14px} +.ph{display:flex;justify-content:space-between;align-items:baseline;gap:12px;flex-wrap:wrap} +.pn{font-weight:600} +.meta{color:var(--dim);font-size:12px} +.track{display:flex;gap:6px;margin:12px 0 0;flex-wrap:wrap} +.dot{width:26px;height:26px;border-radius:6px;display:grid;place-items:center; +font-size:11px;font-weight:700;color:#0e1116} +.WIN{background:var(--win)} .LOSS{background:var(--loss)} +.TIE{background:var(--tie)} .JUDGING{background:var(--wait)} +.gap{margin-top:10px;font-size:13px;color:var(--dim);border-left:2px solid var(--line);padding-left:10px} +.empty{color:var(--dim)} +.legend{margin-top:26px;font-size:12px;color:var(--dim)} +.legend span{display:inline-block;margin-right:14px} +.sw{display:inline-block;width:10px;height:10px;border-radius:3px;vertical-align:middle;margin-right:5px} +""" + + +def render_html(pieces: dict, charter: str, bar: str) -> str: + total = sum(len(p["trials"]) for p in pieces.values()) + resolved = sum(1 for p in pieces.values() for t in p["trials"] if t["outcome"] != "JUDGING") + wins = sum(1 for p in pieces.values() for t in p["trials"] if t["outcome"] == "WIN") + + body = [] + for name, p in sorted(pieces.items()): + last_gap = "" + for t in reversed(p["trials"]): + if t["gap"]: + last_gap = t["gap"] + break + dots = "".join( + f'
{t["round"]}
' + for t in p["trials"] + ) + body.append(f"""
+
{html.escape(name)}
+
{len(p['trials'])} rounds · {p['wins']} beat the bar
+
{dots}
+ {f'
Open gap: {html.escape(last_gap)}
' if last_gap else ''} +
""") + + if not body: + body.append('

No trials staged yet.

') + + return f""" + + +Gauntlet Loop +

Gauntlet Loop

+
{len(pieces)} pieces · {total} trials · {resolved} judged · {wins} beat the bar + ·  updated {time.strftime('%H:%M:%S')} · refreshes every 15s
+
Goal
{html.escape(charter[:1200]) or 'no charter'}
+
The bar
{html.escape(bar[:1200]) or 'no bar set'}
+{''.join(body)} +
+ours won +bar won — gap open +indistinguishable +critic judging +
+""" + + +def read_if(p: Path) -> str: + return p.read_text(encoding="utf-8").strip() if p.is_file() else "" + + +def cmd_render(_): + r = root() + r.mkdir(parents=True, exist_ok=True) + out = r / "board.html" + out.write_text( + render_html(collect(load_rows()), read_if(r / "charter.md"), read_if(r / "bar.md")), + encoding="utf-8", + ) + print(out) + + +def cmd_serve(args): + """Serve the board, and only the board. + + The naive version of this — SimpleHTTPRequestHandler rooted at .gauntlet/ — + would publish keys/*.json over HTTP, handing out the sealed A/B mapping that + the PreToolUse hook exists to protect. So: deny keys/ outright, and bind to + localhost unless the operator explicitly asks for the network. + """ + import functools + import http.server + import socketserver + from urllib.parse import unquote, urlparse + + DENIED = ("keys",) + + class Handler(http.server.SimpleHTTPRequestHandler): + def _denied(self) -> bool: + path = unquote(urlparse(self.path).path).lstrip("/") + head = path.split("/", 1)[0] + return head in DENIED + + def do_GET(self): + if self._denied(): + self.send_error(403, "sealed: A/B mappings are not served") + return + super().do_GET() + + def do_HEAD(self): + if self._denied(): + self.send_error(403, "sealed: A/B mappings are not served") + return + super().do_HEAD() + + cmd_render(args) + handler = functools.partial(Handler, directory=str(root())) + with socketserver.TCPServer((args.host, args.port), handler) as httpd: + shown = "localhost" if args.host in ("127.0.0.1", "localhost") else args.host + print(f"http://{shown}:{args.port}/board.html") + if args.host == "0.0.0.0": + print("bound to all interfaces — anyone on this network can read the board") + httpd.serve_forever() + + +def main(): + ap = argparse.ArgumentParser(prog="board.py", description=__doc__, + formatter_class=argparse.RawDescriptionHelpFormatter) + sub = ap.add_subparsers(dest="cmd", required=True) + sub.add_parser("render").set_defaults(func=cmd_render) + s = sub.add_parser("serve") + s.add_argument("--port", type=int, default=8787) + s.add_argument("--host", default="127.0.0.1", + help="pass 0.0.0.0 to reach the board from your phone on the same network") + s.set_defaults(func=cmd_serve) + a = ap.parse_args() + a.func(a) + + +if __name__ == "__main__": + main() diff --git a/plugins/gauntlet/scripts/key-guard.py b/plugins/gauntlet/scripts/key-guard.py new file mode 100755 index 0000000..7f6917d --- /dev/null +++ b/plugins/gauntlet/scripts/key-guard.py @@ -0,0 +1,42 @@ +#!/usr/bin/env python3 +"""PreToolUse guard: keep the A/B mapping unreadable while a trial is being judged. + +`ab.py stage` writes which side is ours to `.gauntlet/keys/.json`. If any +agent can read that file, the critic is no longer blind and the whole loop +degrades into the builder grading itself with extra steps. + +This hook denies any tool call whose input references that directory. `ab.py` +reads the key in-process during `reveal`, which no hook intercepts, so the lead +can still resolve trials. + +Exit 0 = allow, exit 2 = deny (stderr is shown to the model). +""" +import json +import sys + +GUARDED = ".gauntlet/keys" + + +def main() -> int: + try: + payload = json.load(sys.stdin) + except Exception: + return 0 # never break the session on a malformed event + + blob = json.dumps(payload.get("tool_input", {})) + if GUARDED not in blob and "gauntlet/keys" not in blob: + return 0 + + print( + "BLOCKED by gauntlet: .gauntlet/keys/ holds the sealed A/B mapping for " + "in-flight trials. Reading it would tell you which side is ours and make " + "the critic's judgment worthless.\n" + "To resolve a trial, run `ab.py reveal ` — it reads the key itself " + "and appends the outcome to the ledger.", + file=sys.stderr, + ) + return 2 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/plugins/gauntlet/skills/gauntlet-loop/SKILL.md b/plugins/gauntlet/skills/gauntlet-loop/SKILL.md new file mode 100644 index 0000000..e2f21e8 --- /dev/null +++ b/plugins/gauntlet/skills/gauntlet-loop/SKILL.md @@ -0,0 +1,140 @@ +--- +name: gauntlet-loop +description: Use when running a Gauntlet Loop — driving work to a high quality bar by splitting it into independently judgeable pieces, building each with one agent, and judging each with a separate blind critic against a concrete external reference, looping until the reference stops winning. Covers agentic capabilities, reinforcement learning, and general coding. Loaded by every /gauntlet:* command. +--- + +# Gauntlet Loop + +The method is Matt Shumer's (). This skill is the operational +contract for running it inside Claude Code with real enforcement instead of good intentions. + +## The whole idea + +> Give a lead agent a goal and a real example of what great looks like. The lead decides how to break +> the goal into the smallest pieces that can be improved separately. Each piece gets its own builder +> and a separate critic with fresh context. The builder makes something. The critic compares it +> against the reference. If the reference wins, the critic names the biggest remaining gap and sends +> it back. Then another round begins. + +Ordinary agent work stops at "pretty good" because the agent that built the thing is the one deciding +it is done. The Gauntlet Loop removes that decision from the builder and hands it to an external +standard. + +## The six rules + +1. **Goal, not implementation.** State the destination. Do not prescribe architecture, decomposition, + or a round count. Strong models pick better routes than your outline does. +2. **A real bar.** "Make it amazing" is not a bar. The bar is a concrete artifact or measurement a + critic can *inspect*: reference screenshots, a held-out task suite, a baseline learning curve, a + reference implementation, a latency budget. It does not have to be reachable — it has to point. +3. **The lead splits the work.** Smallest pieces that can be improved and judged *separately*. Not + "make the agent better" — "make this recovery-from-tool-error path beat the reference trajectory." +4. **The builder never grades itself.** Separate agent, fresh context, no builder rationale, blind + A/B where possible. The builder remembers why every choice was reasonable; you do not want + reasonable, you want an independent judgment. +5. **Keep going.** No arbitrary final round. Stop when you like it, when gaps stop mattering, or when + the compute budget runs out — not at round three. +6. **Watch without interrupting.** A live board that updates itself, so you never have to break the + run to ask how it is going. + +Optional: after each wave, one fresh **smoother** inspects the whole artifact and reconciles the +independently-improved pieces so the result feels like one thing. + +## What this plugin enforces mechanically + +Rules 2 and 4 fail silently when they are only prompt text, so they are wired into scripts: + +| Rule | Enforcement | +|---|---| +| No loop without a bar | `ab.py stage` refuses to stage a trial unless `.gauntlet/bar.md` exists and records a PASS on all five soundness checks | +| Critic is blind | `ab.py stage` shuffles candidate and reference into `A/` and `B/` under a per-trial random assignment; the mapping is written **outside** the trial directory to `.gauntlet/keys/`, mode 0600 | +| Critic stays blind | A `PreToolUse` hook denies any tool call whose input path touches `.gauntlet/keys/` — the mapping is unreadable during judging, by any agent | +| Critic sees artifacts, not arguments | The staged trial directory contains only `A/`, `B/`, and a `brief.md` generated from the goal and bar. Builder notes, diffs, and rationale are never copied in | +| No re-grading until you like the answer | `ab.py verdict` refuses to overwrite an existing verdict for a trial | +| Verdicts are receipts | Every staged file is sha256'd into `.gauntlet/ledger.jsonl` alongside the verdict and the revealed mapping | + +Nothing here constrains *what* the lead builds or *how* it splits the work. Determinism lives in the +gate, never in the generation. + +## The bar-soundness gate + +A bar earns the loop only if all five hold. `/gauntlet:bar` produces `.gauntlet/bar.md` with an +explicit verdict per check and the evidence for it. + +1. **Inspectable** — the critic can open, run, or measure the bar directly. A description of a good + outcome is not a bar; the outcome is. +2. **External** — not authored by the builder and not derived from builder output. If the builder + wrote the test, the test is a mirror. +3. **Held-out** — the builder cannot see, train on, or tune against the exact instances used to + judge. Name the split and how it is enforced. +4. **Discriminating** — the current output *loses* to it right now. If round 0 already wins, the bar + is too low; raise it before starting. +5. **Un-gameable** — name the cheapest way to satisfy the bar without doing the work. If a cheat + exists, add the counter-check that kills it, and record both. + +When you do not know the right bar, finding one is the first piece of work — dispatch +`gauntlet:bar-scout`. Do not let the agent invent its own definition of "good"; make it find a +comparison that already exists in the world. + +## Loop shape + +``` +/gauntlet:new → goal + domain → .gauntlet/charter.md + ↓ +/gauntlet:bar → bar-scout + 5 checks → .gauntlet/bar.md (blocks the loop until PASS) + ↓ +/gauntlet:run → lead decomposes → .gauntlet/pieces.json + ↓ + wave: for each piece, in parallel + builder → artifact + ab.py stage --ref --cand + critic → inspects A and B blind → ab.py verdict + ab.py reveal → WIN | LOSS | TIE + biggest gap → ledger + LOSS/TIE → gap goes back to that piece's builder → next round + ↓ + smoother (optional, end of wave) → coherence pass over the whole artifact + ↓ + next wave — no fixed round count + ↓ +/gauntlet:status → board.py render → .gauntlet/board.html (auto-refreshing) +``` + +## Working directory layout + +``` +.gauntlet/ + charter.md goal, domain, constraints, stop condition + bar.md the bar + five soundness verdicts + counter-checks + pieces.json lead's decomposition (its own; never pre-supplied) + ledger.jsonl append-only receipts: staged / verdict / resolved + ab// A/ B/ brief.md verdict.json + keys/.json the A/B mapping — hook-denied to all agents + board.html live progress page + work// builder scratch, notes, drafts (never staged to a critic) +``` + +## Roles + +- **lead** (`gauntlet:lead`) — owns decomposition, wave scheduling, reveal, and the board. Never + builds and never judges. +- **builder** (`gauntlet:builder`) — one per piece. Sees the goal, the bar, and the last gap. Fresh + context per piece, carried across rounds of that piece only. +- **critic** (`gauntlet:critic`) — one per trial. Fresh context every trial. Sees `A/`, `B/`, and the + brief. Never sees which is which, who made either, or any prior verdict. +- **bar-scout** (`gauntlet:bar-scout`) — finds candidate bars and runs the five checks. +- **smoother** (`gauntlet:smoother`) — end-of-wave coherence pass. Reconciles; does not redesign. + +## Domain bar recipes + +Read the one that matches before scouting a bar. Each covers what to judge, which bars are real in +that domain, the leakage traps that make a bar fake, and how to decompose. + +- `references/bars-agentic.md` — agentic capabilities: tool use, recovery, planning, stopping +- `references/bars-rl.md` — reinforcement learning: returns, curves, seeds, eval protocol +- `references/bars-coding.md` — general coding: differential testing, reference impls, budgets + +## Stopping + +There is no completion criterion in the loop. You stop it. Record the reason in `charter.md` — +"gaps stopped mattering", "budget", "shipped" — because a run with no recorded stop reason tends to +get restarted for no reason.