diff --git a/.agents/plugins/marketplace.json b/.agents/plugins/marketplace.json index fa304fb..2a510a9 100644 --- a/.agents/plugins/marketplace.json +++ b/.agents/plugins/marketplace.json @@ -1,14 +1,14 @@ { - "name": "beyond10x", + "name": "b10x", "interface": { "displayName": "Beyond10x" }, "plugins": [ { - "name": "beyond10x", + "name": "b10x", "source": { "source": "local", - "path": "./plugins/beyond10x" + "path": "./plugins/b10x" }, "policy": { "installation": "AVAILABLE", @@ -17,10 +17,10 @@ "category": "Productivity" }, { - "name": "aep-plan", + "name": "aep", "source": { "source": "local", - "path": "./plugins/aep-plan" + "path": "./plugins/aep" }, "policy": { "installation": "AVAILABLE", @@ -29,10 +29,10 @@ "category": "Productivity" }, { - "name": "aep-drive", + "name": "connectors", "source": { "source": "local", - "path": "./plugins/aep-drive" + "path": "./plugins/connectors" }, "policy": { "installation": "AVAILABLE", @@ -41,10 +41,10 @@ "category": "Productivity" }, { - "name": "ess-specify", + "name": "worktree", "source": { "source": "local", - "path": "./plugins/ess-specify" + "path": "./plugins/worktree" }, "policy": { "installation": "AVAILABLE", @@ -53,22 +53,16 @@ "category": "Productivity" }, { - "name": "workspace-hygiene", + "name": "ess", "source": { "source": "local", - "path": "./plugins/workspace-hygiene" + "path": "./plugins/ess" }, "policy": { "installation": "AVAILABLE", "authentication": "ON_INSTALL" }, "category": "Productivity" - }, - { - "name": "connectors", - "source": { "source": "local", "path": "./plugins/connectors" }, - "policy": { "installation": "AVAILABLE", "authentication": "ON_INSTALL" }, - "category": "Productivity" } ] } diff --git a/.agents/skills/improving-by-trial/SKILL.md b/.agents/skills/improving-by-trial/SKILL.md new file mode 100644 index 0000000..e5f20cc --- /dev/null +++ b/.agents/skills/improving-by-trial/SKILL.md @@ -0,0 +1,180 @@ +--- +name: improving-by-trial +description: Improve this repository's plugins by trial — run fresh, isolated agents that get only a user's sentence and this repository's link, collect where they got stuck, fix what belongs here, file what belongs elsewhere, and repeat until the trial passes. Use when asked to trial, dogfood, test the onboarding or a plugin end to end, run a fresh agent against the plugins, or iterate on trial findings before a release. +--- + +# Improving the plugins by trial + +A trial is a fresh agent with nothing but a user's sentence and a link. It shows what the skills +fail to say. The gate cannot show that: every check passes while an agent still guesses. + +## 1. Build a sandbox from this checkout + +```console +task trial:sandbox NAME= +``` + +This creates `/var/tmp/b10x-trials-$USER//` (the Taskfile's `TRIALS`) with its own `home/`, a `b10x` built from this +checkout, and an `env` file. The `env` file sets `HOME` to the sandbox and `B10X_MARKETPLACE` to +this checkout, so `b10x init` installs the plugins as they are in the working tree, before any +release. It also unsets `ANTHROPIC_API_KEY`, `TMPDIR` and `TMPPREFIX`. A defined trial +(`task trial:run TRIAL=`, § 2) builds its own sandbox and fixture; for an ad-hoc one, put the +fixture it needs (a small service, a `TODO.md`, an OpenAPI document) under `work/`, and commit it +there with `git` when the trial needs history or a remote. + +To trial the released version instead, remove `home/.local/bin/b10x` and unset `B10X_MARKETPLACE` +in `env`; the agent then follows `SETUP.md` from the release. + +## 2. Run it headless, never as a sub-agent + +The round's trials are defined in `trials/`, one directory each: + +| file | holds | +|---|---| +| `trials//trial.yaml` | `name`, `kind`, `prompt`, optional `dir` (the run's subdirectory of `work/`), `fixture` (a directory beside it), `remote` (a bare `origin` in the sandbox), `seeded`, `setup` (shell lines run from the sandbox root with its `env` before the run), `measures`, and `outputs` (name → path below the run's directory) | +| `trials//fixture/` | the small service or backlog the trial starts from | +| `trials/baseline.json` | the measures of the last accepted run of each trial | + +`task check` validates every definition. Run one by name: + +```console +task trial:run TRIAL= +``` + +This builds a fresh sandbox (`trial:sandbox`), copies the fixture into `work/` and commits it +there (`agentplugins-check trial-prepare`), runs the definition's `setup` (installing the plugins +under test with `b10x init … --out plan.json` and `b10x setup apply --plan plan.json --yes`, or +seeding an older release), then runs the prompt. An ad-hoc trial still runs in an existing sandbox: + +```console +task trial:run NAME= PROMPT='' [DIR=] [SEEDED=true] +``` + +The task copies the operator's credentials in (mode 600), runs `claude -p` with the sandbox's +`env`, `--strict-mcp-config` (no MCP server, including the account's claude.ai connectors) and +stream-json output into `run.jsonl`, and deletes the credentials when it finishes. The sandbox +`PATH` has `cargo` and `go`, and `go` is an allowed tool, so a trial can build and test an +implementation. + +**Never run a trial as a sub-agent of the working session.** A sub-agent gets the session's own +agents and plugins whatever `HOME` its shell uses. In trial 3 the aep critics that ran were this +machine's, not the plugin's under test. + +The prompt is what a user would type, plus the link when the trial is about finding the +repository. Ask the agent to end by listing the skills and agents it used, quoting any text that +confused it, and pasting the final command output verbatim. Give it no other hints. + +## 3. Only an isolated run counts + +`trial:run` checks isolation before it measures: + +```console +agentplugins-check trial-isolation /run.jsonl --sandbox --version +``` + +It fails when a plugin loaded from outside the sandbox home and the checkout, when a plugin is not +in the sandbox's own registry as it was before the run, when a plugin is not at the version under +test, or when the run used an agent or skill from a plugin the sandbox did not load. One that was +only offered (claude.ai account skills reach sub-agent sessions) is printed as a note. A failing run +is discarded, not interpreted. + +**A sandbox under the home directory is not isolated.** Claude Code reads `CLAUDE.md`, +`.claude/CLAUDE.md` and `.claude/settings*.json` in every directory above its working directory. In +trial 3, sandboxes under `~/.cache` loaded the operator's `~/.claude/CLAUDE.md` (one agent followed +its commit rules) and a `.claude/settings.local.json` an earlier trial left in `~/.cache`, which +enabled `aep@b10x`. `trial:sandbox` refuses to build below any of those files; that is why `TRIALS` +is outside `$HOME`. Neither leak shows in the `init` event, so the placement is the only guard. An upgrade trial seeds older plugins on +purpose (`seeded: true` in its definition, or `SEEDED=true`), so plugins must match what the sandbox +was seeded with; a registry that did not exist before the run is read after it. + +## 4. Read the run + +`trial:run` ends with the numbers, one line each: + +```console +agentplugins-check trial-report /run.jsonl --trial [--baseline trials/baseline.json] [--write-baseline] +``` + +| measure | read from | +|---|---| +| `tool_calls` | `tool_use` blocks in the run | +| `validate` | the last `ess specify validate` the run ran: `valid`, not valid, or not run | +| `synthesis` | `N scenario(s) … M refusal(s)` in the last `ess verify conform synthesize` output | +| `unmapped` | `UNMAPPED:` markers in the YAML files the run wrote, read from disk, not from its prose | +| `outputs` | which of the definition's `outputs` exist (a directory counts when it is not empty) | +| `go_test` | passed, failed and skipped tests of the last `go test` (`-v` or `-json`); a package that does not build counts as a failure | + +A trial reports the measures its definition lists; without `--trial`, an ad-hoc run gets every +measure but `outputs`. With `--baseline` it exits 1 when a measure got worse than the trial's entry: +validate stops passing, refusals or `go test` failures go up, a synthesis or `go test` that ran no +longer runs, an output the baseline had is missing, or tool calls rise by more than 50%. Other +changes print and pass. `--write-baseline` records the run as the trial's entry; do that for an +isolated run the round accepts, and commit `trials/baseline.json` with the fixes it led to. + +The numbers say what happened, not why. From `run.jsonl`, also collect: + +| collect | how | +|---|---| +| stuck points | every refusal and error in `tool_result`s, verbatim | +| guesses | what the final report says it inferred or could not find | +| waste | calls repeated, files read twice, commands that failed and were retried unchanged | +| the outcome | the verbatim `validate` / `generate` / plan output it pasted | + +Before writing a finding into a skill, reproduce each claimed behaviour with the released CLI. A +trial agent's explanation of a refusal is a hypothesis. + +## 5. Triage every finding to one owner + +| the cause | where the fix goes | +|---|---| +| a skill, agent, `SETUP.md`, the catalog or the `b10x` CLI | here: fix it on a branch, with a test for CLI behaviour | +| a product CLI or language (`ess`, `aep`, `worktree`) | an issue in that repository, created through the bot (`AGENTS.md`); the skill here documents the workaround until it is fixed | +| both | both: the workaround here, the issue there, each naming the other | + +The issue body has the trial name, the exact command and output, the expected behaviour and the +workaround the skill now documents, and ends with the line *Found by an agentplugins trial*. The +bot creates it with the label `trial-finding` (`"labels": ["trial-finding"]` in the request). The +label is the ledger: + +```console +gh search issues --owner beyond10x --label trial-finding --state open # still broken +gh search issues --owner beyond10x --label trial-finding --state closed # fixed: released yet? +``` + +Check both lists before each round. Delete a workaround from the skills once its issue is fixed +and the fix is in a release, not when the issue closes. + +## 6. Fix, gate, re-run + +1. Fix on a managed worktree branch. +2. Run `task check`, and `agentplugins-check tools` when a CLI command or the ESS example changed. +3. Rebuild the sandbox (`task trial:sandbox`) and re-run the trial that found the problem. It + passes when the finding does not recur. New findings go back to step 5. +4. Release when every trial of the round passes. + +Each round runs every trial in `trials/`: 4 ESS trials (`ess-new`, a new specification; +`ess-retrofit`, an existing service; `ess-pipeline`, generation plus a synthesized suite; +`ess-full-package`, every output plus a Go implementation held to the synthesized suite), and +`aep-backlog`, `worktree-onboarding` and `upgrade-seeded`. Change the domains and fixtures each +round so the agents cannot copy the previous answer from the skills; a changed trial starts a new +baseline entry. + +### Every product release is re-verified + +`verified.json` names, per CLI (`aep`, `ess`, `worktree`), the release the skills were last +verified against. The daily `agentplugins-check tools` run fails with one line per CLI whose newest +release is newer. Then: + +1. Run `agentplugins-check tools` and fix every command it reports. +2. Run an ESS trial round (at least `ess-full-package`) against the new release. +3. Set the CLI to the new release in `verified.json` in the same pull request. + +## 7. Clean up + +`trial:run` deletes the credentials it copied. When the round is released, check that no +credentials are left in any sandbox and remove the sandboxes: + +```console +ls /var/tmp/b10x-trials-$USER/*/home/.claude/.credentials.json +rm -rf /var/tmp/b10x-trials-$USER/ +``` diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json index 0792150..4a0d744 100644 --- a/.claude-plugin/marketplace.json +++ b/.claude-plugin/marketplace.json @@ -1,36 +1,36 @@ { - "name": "beyond10x", - "owner": { "name": "beyond10x" }, + "name": "b10x", + "owner": { + "name": "beyond10x" + }, + "metadata": { + "description": "Beyond10x agent plugins: the plugins this repository carries, and product plugins pinned at their own releases." + }, "plugins": [ { - "name": "beyond10x", - "source": "./plugins/beyond10x", - "description": "Navigate Beyond10x resources and create portable Codex and Claude Code plugins." + "name": "b10x", + "source": "./plugins/b10x", + "description": "Set up, upgrade and check the Beyond10x plugins and binaries, route work to them, and create portable plugins." }, { - "name": "aep-plan", - "source": "./plugins/aep-plan", - "description": "Govern AEP planning artifacts, decomposition, review, and reverse engineering." + "name": "aep", + "source": "./plugins/aep", + "description": "Plan governed work in the AEP artifact store and deliver it in reviewed waves: decomposition, plan critique, reverse engineering, story scoping, implementation and adversarial review." }, { - "name": "aep-drive", - "source": "./plugins/aep-drive", - "description": "Coordinate AEP development waves, story scoping, implementation, and adversarial review." - }, - { - "name": "ess-specify", - "source": "./plugins/ess-specify", - "description": "Validate ESS models and guide deterministic schema and OpenAPI projections." + "name": "connectors", + "source": "./plugins/connectors", + "description": "Set up, inspect, and invoke governed integrations through the connectors CLI." }, { - "name": "workspace-hygiene", - "source": "./plugins/workspace-hygiene", - "description": "Manage isolated Git worktrees with explicit ownership and safe cleanup proof." + "name": "worktree", + "source": "./plugins/worktree", + "description": "Create, lease, finish and safely clean isolated Git worktrees through the worktree CLI." }, { - "name": "connectors", - "source": "./plugins/connectors", - "description": "Set up, inspect, and invoke governed integrations through the connectors CLI." + "name": "ess", + "source": "./plugins/ess", + "description": "Write, retrofit, validate and project ESS specifications, and hold implementations to them with conformance suites." } ] } diff --git a/.claude/skills/improving-by-trial b/.claude/skills/improving-by-trial new file mode 120000 index 0000000..4517d5b --- /dev/null +++ b/.claude/skills/improving-by-trial @@ -0,0 +1 @@ +../../.agents/skills/improving-by-trial \ No newline at end of file diff --git a/.engineering/planning/journal.jsonl b/.engineering/planning/journal.jsonl index d97ab78..0b2f00e 100644 --- a/.engineering/planning/journal.jsonl +++ b/.engineering/planning/journal.jsonl @@ -179,3 +179,11 @@ {"entity":"task","version":1,"id":"release-0-9-2-connectors-skill-v1-line","revision":4,"type":"aep.status.move/v1","from_state":"active","to_state":"implemented","changed":{},"args":{"command":"move-status","decided_on":"{\"recorded\":{\"test_result\":1}}","target":"01MEM0000000000000088","to":"implemented"},"payload":{"actor":"human:timo","at":1789491785669,"causation":"cmd-move-task-release-0-9-2-connectors-skill-v1-line-1789491785669","change":{"change":"moved","decided_on":{"recorded":{"test_result":1}},"from":"active","to":"implemented"},"correlation":"protocol-artifact-move","event_id":"task:release-0-9-2-connectors-skill-v1-line@4#0~964338e589eac631","executor":null,"recorded_at":"2026-09-15T17:03:05Z"}} {"entity":"story","version":1,"id":"shared-docs-toolchain","revision":1,"type":"aep.entity.create/v1","from_state":null,"to_state":"draft","changed":{"body":"## Outcome\n\nUse the reviewed shared documentation runtime for this repository's passive source bundle producer as part of the coordinated Mandate contract viewer publication.\n\n## Acceptance\n\nThe Atlas-generated producer caller pins Docs System commit 1d4c0262911761118ffdd7037890f541a0688714. Source allowlists, credential-free execution and repository-owned validation remain intact. Atlas reconciliation reports no caller drift. Required source and common checks pass on the exact published commit, and its passive producer yields a valid immutable source bundle.\n\n## Scope\n\n.github/workflows/b10x-docs-bundle.yml and this planning record. The coordinating authority is the shared documentation rollout; this repository's product/runtime contracts do not change.\n","title":"Align the passive producer with the shared contract viewer runtime"},"args":{"command":"create-entity","data":{"body":"## Outcome\n\nUse the reviewed shared documentation runtime for this repository's passive source bundle producer as part of the coordinated Mandate contract viewer publication.\n\n## Acceptance\n\nThe Atlas-generated producer caller pins Docs System commit 1d4c0262911761118ffdd7037890f541a0688714. Source allowlists, credential-free execution and repository-owned validation remain intact. Atlas reconciliation reports no caller drift. Required source and common checks pass on the exact published commit, and its passive producer yields a valid immutable source bundle.\n\n## Scope\n\n.github/workflows/b10x-docs-bundle.yml and this planning record. The coordinating authority is the shared documentation rollout; this repository's product/runtime contracts do not change.\n","status":"draft","title":"Align the passive producer with the shared contract viewer runtime"},"entity_type":"aep.story/v1","locator":"ep://planning/store/story/shared-docs-toolchain"},"payload":{"actor":"agent:mandate-docs-toolchain","at":1789627861573,"causation":"cmd-new-story-shared-docs-toolchain","change":{"change":"created","status":"draft"},"correlation":"protocol-artifact-new","event_id":"story:shared-docs-toolchain@1#0~db0fa97c13705108","executor":null,"recorded_at":"2026-09-17T06:51:01Z"}} {"entity":"story","version":1,"id":"shared-docs-toolchain","revision":2,"type":"aep.entity.update/v1","from_state":"draft","to_state":"draft","changed":{"scope":[{"confidence":"cited","path":".github/workflows/b10x-docs-bundle.yml"}]},"args":{"changes":{"scope":[{"confidence":"cited","path":".github/workflows/b10x-docs-bundle.yml"}]},"command":"update-entity","target":"01MEM0000000000000067"},"payload":{"actor":"agent:mandate-docs-toolchain","at":1789627861652,"causation":"cmd-protocol-artifact-scope-story-shared-docs-toolchain-1789627861652","change":{"change":"body_replaced"},"correlation":"protocol-artifact-scope","event_id":"story:shared-docs-toolchain@2#0~27ac65189db9b60f","executor":null,"recorded_at":"2026-09-17T06:51:01Z"}} +{"entity":"story","version":1,"id":"retire-ess-specify","revision":1,"type":"aep.entity.create/v1","from_state":null,"to_state":"draft","changed":{"body":"# Retire `ess-specify`; spec-driven work routes to the ESS repository's own plugin\n\n## Why\n\nESS now ships its agent plugin from `beyond10x/ess` (`plugins/ess/`, marketplace `ess`), at the\nsame version as the `ess` binary, and the binary prints the same skills through `ess skill`. The\ncopy here released on its own cadence and drifted: `0.9.2` told adopters to install ESS `0.22.1`\nwhile ESS was at `0.29.0`.\n\n## Acceptance\n\n`plugins/ess-specify` and its marketplace entries are gone; `task check` prints\n`valid: marketplace beyond10x, 5 focused plugin(s)`; `ess-specify` is a retired name the gate\nrefuses in authored files; the `beyond10x` front door and the install guide send spec-driven work to\nthe ESS marketplace and `ess skill`.\n\n## Scope\n\n- `plugins/ess-specify/**` (deleted), `.claude-plugin/marketplace.json`, `.agents/plugins/marketplace.json`\n- `crates/agentplugins-check/src/{main.rs,evals.rs}`\n- `evals/ess-specify-new-entity/**` (deleted), `evals/golden-path-end-to-end/**`, `evals/README.md`\n- `plugins/beyond10x/skills/beyond10x/**`, `plugins/aep-plan/skills/planning/SKILL.md`\n- `README.md`, `website/**`, `CHANGELOG.md`\n\n## Not in scope\n\n- An eval case for the ESS plugin. ESS runs no eval corpus; the dropped case is not re-homed.\n","title":"Retire ess-specify; spec-driven work routes to the ESS repository's own plugin"},"args":{"command":"create-entity","data":{"body":"# Retire `ess-specify`; spec-driven work routes to the ESS repository's own plugin\n\n## Why\n\nESS now ships its agent plugin from `beyond10x/ess` (`plugins/ess/`, marketplace `ess`), at the\nsame version as the `ess` binary, and the binary prints the same skills through `ess skill`. The\ncopy here released on its own cadence and drifted: `0.9.2` told adopters to install ESS `0.22.1`\nwhile ESS was at `0.29.0`.\n\n## Acceptance\n\n`plugins/ess-specify` and its marketplace entries are gone; `task check` prints\n`valid: marketplace beyond10x, 5 focused plugin(s)`; `ess-specify` is a retired name the gate\nrefuses in authored files; the `beyond10x` front door and the install guide send spec-driven work to\nthe ESS marketplace and `ess skill`.\n\n## Scope\n\n- `plugins/ess-specify/**` (deleted), `.claude-plugin/marketplace.json`, `.agents/plugins/marketplace.json`\n- `crates/agentplugins-check/src/{main.rs,evals.rs}`\n- `evals/ess-specify-new-entity/**` (deleted), `evals/golden-path-end-to-end/**`, `evals/README.md`\n- `plugins/beyond10x/skills/beyond10x/**`, `plugins/aep-plan/skills/planning/SKILL.md`\n- `README.md`, `website/**`, `CHANGELOG.md`\n\n## Not in scope\n\n- An eval case for the ESS plugin. ESS runs no eval corpus; the dropped case is not re-homed.\n","status":"draft","title":"Retire ess-specify; spec-driven work routes to the ESS repository's own plugin"},"entity_type":"aep.story/v1","locator":"ep://planning/store/story/retire-ess-specify"},"payload":{"actor":"human:timo","at":1790167587226,"causation":"cmd-new-story-retire-ess-specify","change":{"change":"created","status":"draft"},"correlation":"protocol-artifact-new","event_id":"story:retire-ess-specify@1#0~73af7ee8bf239901","executor":null,"recorded_at":"2026-09-23T12:46:27Z"}} +{"args":{"command":"create-entity","data":{"body":"# Story: the b10x catalog lists worktree by pin, not by copy\n\n## Goal\nThe marketplace is renamed `b10x` and lists the `worktree` plugin as a `git-subdir` entry pinned\nto a Worktree release tag and its full commit, replacing the `workspace-hygiene` copy. The copy\nhere shipped on this repository's cadence: `workspace-hygiene` 0.10.0 lacked the\n`patch-equivalent` recovery-proof guidance that Worktree 0.5.0 generates.\n\n## Acceptance\n`task check` refuses a `worktree` entry that is not a `git-subdir` at\n`https://github.com/beyond10x/worktree.git` path `plugins/worktree` with a bare release tag and a\n40-hex commit; `agentplugins-check remote` refuses a commit that is not the tag's, manifests at\nthat commit that declare another version, and a tag that is not Worktree's newest release, and\nruns on every pull request, `main` push and daily.\n\n## Scope\n- `.claude-plugin/marketplace.json`, `.agents/plugins/marketplace.json`\n- `crates/agentplugins-check/src/{main,remote,evals}.rs`, `.github/workflows/remote-plugins.yml`\n- `plugins/workspace-hygiene/` (removed), routing skill, wave skill, README, AGENTS.md, website\n","status":"draft","summary":"Rename the marketplace to b10x and replace the workspace-hygiene copy with a git-subdir entry pinned to a Worktree release, checked offline for shape and online for tag, commit, version and recency.","title":"The b10x catalog lists worktree by pin, not by copy"},"entity_type":"aep.story/v1","locator":"ep://planning/store/story/worktree-plugin-by-pin"},"changed":{"body":"# Story: the b10x catalog lists worktree by pin, not by copy\n\n## Goal\nThe marketplace is renamed `b10x` and lists the `worktree` plugin as a `git-subdir` entry pinned\nto a Worktree release tag and its full commit, replacing the `workspace-hygiene` copy. The copy\nhere shipped on this repository's cadence: `workspace-hygiene` 0.10.0 lacked the\n`patch-equivalent` recovery-proof guidance that Worktree 0.5.0 generates.\n\n## Acceptance\n`task check` refuses a `worktree` entry that is not a `git-subdir` at\n`https://github.com/beyond10x/worktree.git` path `plugins/worktree` with a bare release tag and a\n40-hex commit; `agentplugins-check remote` refuses a commit that is not the tag's, manifests at\nthat commit that declare another version, and a tag that is not Worktree's newest release, and\nruns on every pull request, `main` push and daily.\n\n## Scope\n- `.claude-plugin/marketplace.json`, `.agents/plugins/marketplace.json`\n- `crates/agentplugins-check/src/{main,remote,evals}.rs`, `.github/workflows/remote-plugins.yml`\n- `plugins/workspace-hygiene/` (removed), routing skill, wave skill, README, AGENTS.md, website\n","summary":"Rename the marketplace to b10x and replace the workspace-hygiene copy with a git-subdir entry pinned to a Worktree release, checked offline for shape and online for tag, commit, version and recency.","title":"The b10x catalog lists worktree by pin, not by copy"},"entity":"story","entry_hash":"374497c1e8d86a98c468032fcf4dc86351e5711b2375b32c0d1438b3a61734b5","from_state":null,"id":"worktree-plugin-by-pin","parent_hash":"0e5fd9db0bb58243744cd0404562223549dc236f614221abf79f8add2fd9a4ac","payload":{"actor":"human:timo","at":1790247637626,"causation":"cmd-new-story-worktree-plugin-by-pin","change":{"change":"created","status":"draft"},"correlation":"protocol-artifact-new","event_id":"story:worktree-plugin-by-pin@1#0~a2165dda3a96e504","executor":null,"recorded_at":"2026-09-24T11:00:37Z"},"revision":1,"to_state":"draft","type":"aep.entity.create/v1","version":1} +{"args":{"changes":{"body":"# Retire `ess-specify`; spec-driven work routes to the ESS repository's own plugin\n\n## Why\n\nESS now ships its agent plugin from `beyond10x/ess` (`plugins/ess/`, marketplace `ess`), at the\nsame version as the `ess` binary, and the binary prints the same skills through `ess skill`. The\ncopy here released on its own cadence and drifted: `0.9.2` told adopters to install ESS `0.22.1`\nwhile ESS was at `0.29.0`.\n\n## Acceptance\n\n`plugins/ess-specify` and its marketplace entries are gone; `task check` prints\n`valid: marketplace beyond10x, 5 focused plugin(s)`; `ess-specify` is a retired name the gate\nrefuses in authored files; the `beyond10x` front door and the install guide send spec-driven work to\nthe ESS marketplace and `ess skill`.\n\n## Scope\n\n- `plugins/ess-specify/**` (deleted), `.claude-plugin/marketplace.json`, `.agents/plugins/marketplace.json`\n- `crates/agentplugins-check/src/{main.rs,evals.rs}`\n- `evals/ess-specify-new-entity/**` (deleted), `evals/golden-path-end-to-end/**`, `evals/README.md`\n- `plugins/beyond10x/skills/beyond10x/**`, `plugins/aep-plan/skills/planning/SKILL.md`\n- `README.md`, `website/**`, `CHANGELOG.md`\n\n## Not in scope\n\n- An eval case for the ESS plugin. ESS runs no eval corpus; the dropped case is not re-homed.\n\n\n## Outcome (2026-09-25)\n\n`ess-specify` was retired in agentplugins 0.10.0 (PR #11, `514ef9e0`, release run 35920939084). The routing target then moved: ess#68 removed the ESS repository plugin, and the ESS plugin ships from this repository as `plugins/ess` (`ess@b10x`, #17 `f03fa470`, #18 `7574cb86`, release 0.14.1).\n"},"command":"update-entity","target":"01MEM0000000000000064"},"changed":{"body":"# Retire `ess-specify`; spec-driven work routes to the ESS repository's own plugin\n\n## Why\n\nESS now ships its agent plugin from `beyond10x/ess` (`plugins/ess/`, marketplace `ess`), at the\nsame version as the `ess` binary, and the binary prints the same skills through `ess skill`. The\ncopy here released on its own cadence and drifted: `0.9.2` told adopters to install ESS `0.22.1`\nwhile ESS was at `0.29.0`.\n\n## Acceptance\n\n`plugins/ess-specify` and its marketplace entries are gone; `task check` prints\n`valid: marketplace beyond10x, 5 focused plugin(s)`; `ess-specify` is a retired name the gate\nrefuses in authored files; the `beyond10x` front door and the install guide send spec-driven work to\nthe ESS marketplace and `ess skill`.\n\n## Scope\n\n- `plugins/ess-specify/**` (deleted), `.claude-plugin/marketplace.json`, `.agents/plugins/marketplace.json`\n- `crates/agentplugins-check/src/{main.rs,evals.rs}`\n- `evals/ess-specify-new-entity/**` (deleted), `evals/golden-path-end-to-end/**`, `evals/README.md`\n- `plugins/beyond10x/skills/beyond10x/**`, `plugins/aep-plan/skills/planning/SKILL.md`\n- `README.md`, `website/**`, `CHANGELOG.md`\n\n## Not in scope\n\n- An eval case for the ESS plugin. ESS runs no eval corpus; the dropped case is not re-homed.\n\n\n## Outcome (2026-09-25)\n\n`ess-specify` was retired in agentplugins 0.10.0 (PR #11, `514ef9e0`, release run 35920939084). The routing target then moved: ess#68 removed the ESS repository plugin, and the ESS plugin ships from this repository as `plugins/ess` (`ess@b10x`, #17 `f03fa470`, #18 `7574cb86`, release 0.14.1).\n"},"entity":"story","entry_hash":"5ef1d08f2c7152c1ff8735174646946a5e600433e131d09b24ba87d5df90b413","from_state":"draft","id":"retire-ess-specify","parent_hash":"374497c1e8d86a98c468032fcf4dc86351e5711b2375b32c0d1438b3a61734b5","payload":{"actor":"human:timo","at":1790318169721,"causation":"cmd-protocol-artifact-body-story-retire-ess-specify-1790318169721","change":{"change":"body_replaced"},"correlation":"protocol-artifact-body","event_id":"story:retire-ess-specify@2#0~892be838182eef4a","executor":null,"recorded_at":"2026-09-25T06:36:09Z"},"revision":2,"to_state":"draft","type":"aep.entity.update/v1","version":1} +{"args":{"command":"record-evidence","kind":"test_result","reference":"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/beyond10x/agentplugins/actions/runs/35920939084","source":"agentplugins 0.10.0 release gate (PR #11, 514ef9e0)","target":"01MEM0000000000000064"},"changed":{},"entity":"story","entry_hash":"c1af61a4a40e1332156137f5248f7278c4d4dd230c2b94611d7b55b844ab8ffb","from_state":"draft","id":"retire-ess-specify","parent_hash":"5ef1d08f2c7152c1ff8735174646946a5e600433e131d09b24ba87d5df90b413","payload":{"actor":"human:timo","at":1790198007000,"causation":"cmd-evidence-story-retire-ess-specify-test_result-1790318169856","change":{"change":"evidence","kind":"test_result","reference":"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/beyond10x/agentplugins/actions/runs/35920939084","source":"agentplugins 0.10.0 release gate (PR #11, 514ef9e0)"},"correlation":"protocol-artifact-evidence","event_id":"story:retire-ess-specify@2#0~deff5bea037f228f","executor":null,"recorded_at":"2026-09-23T21:13:27Z"},"revision":2,"to_state":"draft","type":"aep.evidence.record/v1","version":1} +{"args":{"command":"move-status","decided_on":"{\"recorded\":{\"test_result\":1}}","target":"01MEM0000000000000064","to":"proposed"},"changed":{},"entity":"story","entry_hash":"f7c57390b7e5131fcb80e529afc9fc8f04be94cafbad3521328623e05f8c799e","from_state":"draft","id":"retire-ess-specify","parent_hash":"c1af61a4a40e1332156137f5248f7278c4d4dd230c2b94611d7b55b844ab8ffb","payload":{"actor":"human:timo","at":1790318170051,"causation":"cmd-move-story-retire-ess-specify-1790318170051","change":{"change":"moved","decided_on":{"recorded":{"test_result":1}},"from":"draft","to":"proposed"},"correlation":"protocol-artifact-move","event_id":"story:retire-ess-specify@3#0~da6e24e15684d045","executor":null,"recorded_at":"2026-09-25T06:36:10Z"},"revision":3,"to_state":"proposed","type":"aep.status.move/v1","version":1} +{"args":{"command":"move-status","decided_on":"{\"recorded\":{\"test_result\":1}}","target":"01MEM0000000000000064","to":"active"},"changed":{},"entity":"story","entry_hash":"623fa207625dcfc20364669ed3657d4de897709bb53b1e110dcd21099a45becb","from_state":"proposed","id":"retire-ess-specify","parent_hash":"f7c57390b7e5131fcb80e529afc9fc8f04be94cafbad3521328623e05f8c799e","payload":{"actor":"human:timo","at":1790318170218,"causation":"cmd-move-story-retire-ess-specify-1790318170218","change":{"change":"moved","decided_on":{"recorded":{"test_result":1}},"from":"proposed","to":"active"},"correlation":"protocol-artifact-move","event_id":"story:retire-ess-specify@4#0~f0694be4ea570c25","executor":null,"recorded_at":"2026-09-25T06:36:10Z"},"revision":4,"to_state":"active","type":"aep.status.move/v1","version":1} +{"args":{"command":"move-status","decided_on":"{\"recorded\":{\"test_result\":1}}","target":"01MEM0000000000000064","to":"implemented"},"changed":{},"entity":"story","entry_hash":"5f14621f29e41c6d0274cdf4974d88310addd62c3c034f0d64962f007d650f78","from_state":"active","id":"retire-ess-specify","parent_hash":"623fa207625dcfc20364669ed3657d4de897709bb53b1e110dcd21099a45becb","payload":{"actor":"human:timo","at":1790318170388,"causation":"cmd-move-story-retire-ess-specify-1790318170388","change":{"change":"moved","decided_on":{"recorded":{"test_result":1}},"from":"active","to":"implemented"},"correlation":"protocol-artifact-move","event_id":"story:retire-ess-specify@5#0~31bf6cfcddec0e7f","executor":null,"recorded_at":"2026-09-25T06:36:10Z"},"revision":5,"to_state":"implemented","type":"aep.status.move/v1","version":1} +{"args":{"command":"create-entity","data":{"body":"# Story: one-sentence onboarding through `b10x`\n\n## Goal\nA person gives Claude Code or Codex one sentence and this repository's link. The agent follows\n`SETUP.md`: it asks before installing the `b10x` binary, then runs `/b10x:init`, which asks what the\nperson wants to do (plan and deliver work, write specifications, isolated Git checkouts,\nintegrations) and how to install the CLIs (prebuilt archives by default, or `cargo`). `b10x` reads\nboth hosts, the binaries on `PATH` and project settings, and plans exact actions: plugins from the\n`b10x` marketplace, CLIs at their newest release, earlier installs under retired names and\nmarketplaces replaced in place. It applies them only after confirmation, and `/:upgrade`\nand the SessionStart check keep them current.\nMeasured before (2026-09-24): on the operator's machine `ess` 0.26.0 ran under plugin 0.30.0 and\n`worktree` 0.4.1 under 0.6.0; nothing reported it.\n\n## Acceptance\nIn an isolated headless trial (`task trial:sandbox`, `task trial:run`, `agentplugins-check\ntrial-isolation`), a home seeded with a pinned agentplugins 0.12.0 and a retired `aep-plan` plugin\nends after `/b10x:upgrade` with `aep@b10x` and `b10x@b10x` at the current version and no retired id\nleft; and a fresh agent given only a sentence and the link reaches a validated ESS specification\nwithout cloning a product repository.\n\n## Scope\n- `crates/b10x/`, `catalog.json`, `SETUP.md`, `.github/workflows/release.yml`\n- `plugins/*/skills/{init,upgrade}`, `plugins/b10x/hooks/hooks.json`\n- `crates/agentplugins-check/` (concept rules, `tools`, `trial-isolation`)\n- README, AGENTS.md, website install and plugin pages\n\n## Provenance\n- Trials 1–3 and rounds 3–4 (2026-09-24/25); agentplugins 0.12.0–0.14.7.\n- Rebuilt on 2026-09-25 from a 2026-09-24 draft that could not be applied to the current store.\n","status":"draft","summary":"An agent given one sentence and the repository link asks, installs b10x, plans plugins and CLIs from recorded state, and applies them only after confirmation.","title":"One-sentence onboarding through b10x"},"entity_type":"aep.story/v1","locator":"ep://planning/store/story/one-sentence-onboarding"},"changed":{"body":"# Story: one-sentence onboarding through `b10x`\n\n## Goal\nA person gives Claude Code or Codex one sentence and this repository's link. The agent follows\n`SETUP.md`: it asks before installing the `b10x` binary, then runs `/b10x:init`, which asks what the\nperson wants to do (plan and deliver work, write specifications, isolated Git checkouts,\nintegrations) and how to install the CLIs (prebuilt archives by default, or `cargo`). `b10x` reads\nboth hosts, the binaries on `PATH` and project settings, and plans exact actions: plugins from the\n`b10x` marketplace, CLIs at their newest release, earlier installs under retired names and\nmarketplaces replaced in place. It applies them only after confirmation, and `/:upgrade`\nand the SessionStart check keep them current.\nMeasured before (2026-09-24): on the operator's machine `ess` 0.26.0 ran under plugin 0.30.0 and\n`worktree` 0.4.1 under 0.6.0; nothing reported it.\n\n## Acceptance\nIn an isolated headless trial (`task trial:sandbox`, `task trial:run`, `agentplugins-check\ntrial-isolation`), a home seeded with a pinned agentplugins 0.12.0 and a retired `aep-plan` plugin\nends after `/b10x:upgrade` with `aep@b10x` and `b10x@b10x` at the current version and no retired id\nleft; and a fresh agent given only a sentence and the link reaches a validated ESS specification\nwithout cloning a product repository.\n\n## Scope\n- `crates/b10x/`, `catalog.json`, `SETUP.md`, `.github/workflows/release.yml`\n- `plugins/*/skills/{init,upgrade}`, `plugins/b10x/hooks/hooks.json`\n- `crates/agentplugins-check/` (concept rules, `tools`, `trial-isolation`)\n- README, AGENTS.md, website install and plugin pages\n\n## Provenance\n- Trials 1–3 and rounds 3–4 (2026-09-24/25); agentplugins 0.12.0–0.14.7.\n- Rebuilt on 2026-09-25 from a 2026-09-24 draft that could not be applied to the current store.\n","summary":"An agent given one sentence and the repository link asks, installs b10x, plans plugins and CLIs from recorded state, and applies them only after confirmation.","title":"One-sentence onboarding through b10x"},"entity":"story","entry_hash":"9f4d9f13c72d9bc4b810278dcee3a0269efe2a2d3de40197652599d620457107","from_state":null,"id":"one-sentence-onboarding","parent_hash":"5f14621f29e41c6d0274cdf4974d88310addd62c3c034f0d64962f007d650f78","payload":{"actor":"human:timo","at":1790364155880,"causation":"cmd-new-story-one-sentence-onboarding","change":{"change":"created","status":"draft"},"correlation":"protocol-artifact-new","event_id":"story:one-sentence-onboarding@1#0~91dd8d0a511a3232","executor":null,"recorded_at":"2026-09-25T19:22:35Z"},"revision":1,"to_state":"draft","type":"aep.entity.create/v1","version":1} diff --git a/.engineering/planning/story/one-sentence-onboarding.md b/.engineering/planning/story/one-sentence-onboarding.md new file mode 100644 index 0000000..e1e1667 --- /dev/null +++ b/.engineering/planning/story/one-sentence-onboarding.md @@ -0,0 +1,39 @@ +--- +format: aep.planning-md/1 +id: story:one-sentence-onboarding +kind: story +status: draft +title: One-sentence onboarding through b10x +summary: An agent given one sentence and the repository link asks, installs b10x, plans plugins and CLIs from recorded state, and applies them only after confirmation. +revision: 1 +--- +# Story: one-sentence onboarding through `b10x` + +## Goal +A person gives Claude Code or Codex one sentence and this repository's link. The agent follows +`SETUP.md`: it asks before installing the `b10x` binary, then runs `/b10x:init`, which asks what the +person wants to do (plan and deliver work, write specifications, isolated Git checkouts, +integrations) and how to install the CLIs (prebuilt archives by default, or `cargo`). `b10x` reads +both hosts, the binaries on `PATH` and project settings, and plans exact actions: plugins from the +`b10x` marketplace, CLIs at their newest release, earlier installs under retired names and +marketplaces replaced in place. It applies them only after confirmation, and `/:upgrade` +and the SessionStart check keep them current. +Measured before (2026-09-24): on the operator's machine `ess` 0.26.0 ran under plugin 0.30.0 and +`worktree` 0.4.1 under 0.6.0; nothing reported it. + +## Acceptance +In an isolated headless trial (`task trial:sandbox`, `task trial:run`, `agentplugins-check +trial-isolation`), a home seeded with a pinned agentplugins 0.12.0 and a retired `aep-plan` plugin +ends after `/b10x:upgrade` with `aep@b10x` and `b10x@b10x` at the current version and no retired id +left; and a fresh agent given only a sentence and the link reaches a validated ESS specification +without cloning a product repository. + +## Scope +- `crates/b10x/`, `catalog.json`, `SETUP.md`, `.github/workflows/release.yml` +- `plugins/*/skills/{init,upgrade}`, `plugins/b10x/hooks/hooks.json` +- `crates/agentplugins-check/` (concept rules, `tools`, `trial-isolation`) +- README, AGENTS.md, website install and plugin pages + +## Provenance +- Trials 1–3 and rounds 3–4 (2026-09-24/25); agentplugins 0.12.0–0.14.7. +- Rebuilt on 2026-09-25 from a 2026-09-24 draft that could not be applied to the current store. diff --git a/.engineering/planning/story/retire-ess-specify.md b/.engineering/planning/story/retire-ess-specify.md new file mode 100644 index 0000000..dc49127 --- /dev/null +++ b/.engineering/planning/story/retire-ess-specify.md @@ -0,0 +1,40 @@ +--- +format: aep.planning-md/1 +id: story:retire-ess-specify +kind: story +status: implemented +title: Retire ess-specify; spec-driven work routes to the ESS repository's own plugin +revision: 5 +--- +# Retire `ess-specify`; spec-driven work routes to the ESS repository's own plugin + +## Why + +ESS now ships its agent plugin from `beyond10x/ess` (`plugins/ess/`, marketplace `ess`), at the +same version as the `ess` binary, and the binary prints the same skills through `ess skill`. The +copy here released on its own cadence and drifted: `0.9.2` told adopters to install ESS `0.22.1` +while ESS was at `0.29.0`. + +## Acceptance + +`plugins/ess-specify` and its marketplace entries are gone; `task check` prints +`valid: marketplace beyond10x, 5 focused plugin(s)`; `ess-specify` is a retired name the gate +refuses in authored files; the `beyond10x` front door and the install guide send spec-driven work to +the ESS marketplace and `ess skill`. + +## Scope + +- `plugins/ess-specify/**` (deleted), `.claude-plugin/marketplace.json`, `.agents/plugins/marketplace.json` +- `crates/agentplugins-check/src/{main.rs,evals.rs}` +- `evals/ess-specify-new-entity/**` (deleted), `evals/golden-path-end-to-end/**`, `evals/README.md` +- `plugins/beyond10x/skills/beyond10x/**`, `plugins/aep-plan/skills/planning/SKILL.md` +- `README.md`, `website/**`, `CHANGELOG.md` + +## Not in scope + +- An eval case for the ESS plugin. ESS runs no eval corpus; the dropped case is not re-homed. + + +## Outcome (2026-09-25) + +`ess-specify` was retired in agentplugins 0.10.0 (PR #11, `514ef9e0`, release run 35920939084). The routing target then moved: ess#68 removed the ESS repository plugin, and the ESS plugin ships from this repository as `plugins/ess` (`ess@b10x`, #17 `f03fa470`, #18 `7574cb86`, release 0.14.1). diff --git a/.engineering/planning/story/worktree-plugin-by-pin.md b/.engineering/planning/story/worktree-plugin-by-pin.md new file mode 100644 index 0000000..2bc82c5 --- /dev/null +++ b/.engineering/planning/story/worktree-plugin-by-pin.md @@ -0,0 +1,28 @@ +--- +format: aep.planning-md/1 +id: story:worktree-plugin-by-pin +kind: story +status: draft +title: The b10x catalog lists worktree by pin, not by copy +summary: Rename the marketplace to b10x and replace the workspace-hygiene copy with a git-subdir entry pinned to a Worktree release, checked offline for shape and online for tag, commit, version and recency. +revision: 1 +--- +# Story: the b10x catalog lists worktree by pin, not by copy + +## Goal +The marketplace is renamed `b10x` and lists the `worktree` plugin as a `git-subdir` entry pinned +to a Worktree release tag and its full commit, replacing the `workspace-hygiene` copy. The copy +here shipped on this repository's cadence: `workspace-hygiene` 0.10.0 lacked the +`patch-equivalent` recovery-proof guidance that Worktree 0.5.0 generates. + +## Acceptance +`task check` refuses a `worktree` entry that is not a `git-subdir` at +`https://github.com/beyond10x/worktree.git` path `plugins/worktree` with a bare release tag and a +40-hex commit; `agentplugins-check remote` refuses a commit that is not the tag's, manifests at +that commit that declare another version, and a tag that is not Worktree's newest release, and +runs on every pull request, `main` push and daily. + +## Scope +- `.claude-plugin/marketplace.json`, `.agents/plugins/marketplace.json` +- `crates/agentplugins-check/src/{main,remote,evals}.rs`, `.github/workflows/remote-plugins.yml` +- `plugins/workspace-hygiene/` (removed), routing skill, wave skill, README, AGENTS.md, website diff --git a/.github/workflows/b10x-docs-bundle.yml b/.github/workflows/b10x-docs-bundle.yml index dde7d05..e767fec 100644 --- a/.github/workflows/b10x-docs-bundle.yml +++ b/.github/workflows/b10x-docs-bundle.yml @@ -39,7 +39,7 @@ jobs: uses: dtolnay/rust-toolchain@4360b52568e2003a75bf9bc1d59f33a8e3fc893c - name: Build normalized documentation bundle - uses: beyond10x/docs-system/.github/actions/bundle@1d4c0262911761118ffdd7037890f541a0688714 + uses: beyond10x/docs-system/.github/actions/bundle@339b4b8462f19b4c9d3716e6a44ed2a3691eb9d8 with: repository-root: . output: ${{ runner.temp }}/b10x-docs-bundle diff --git a/.github/workflows/b10x-docs-check.yml b/.github/workflows/b10x-docs-check.yml new file mode 100644 index 0000000..e07d604 --- /dev/null +++ b/.github/workflows/b10x-docs-check.yml @@ -0,0 +1,34 @@ +# Generated by `atlas docs reconcile`; edit Atlas documentation intent, not this file. +name: Documentation source check + +on: + pull_request: + push: + branches: [main] + +permissions: + contents: read + +concurrency: + group: b10x-docs-check-${{ github.ref }} + cancel-in-progress: true + +jobs: + b10x-docs-check: + runs-on: ubuntu-latest + timeout-minutes: 10 + steps: + - name: Check out exact source + uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 + with: + persist-credentials: false + + - name: Set up Node.js + uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 + with: + node-version: 24 + + - name: Check declared documentation sources + uses: beyond10x/docs-system/.github/actions/check@339b4b8462f19b4c9d3716e6a44ed2a3691eb9d8 + with: + repository-root: . diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index 586898b..e452b1d 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -31,9 +31,63 @@ jobs: - name: Run the repository gate run: task check + package: + name: Package b10x ${{ matrix.target }} + needs: gate + runs-on: ${{ matrix.runner }} + timeout-minutes: 30 + strategy: + fail-fast: false + matrix: + include: + # Ubuntu 22.04 keeps the glibc floor lower than `ubuntu-latest` for downloaded binaries. + - target: x86_64-unknown-linux-gnu + runner: ubuntu-22.04 + - target: aarch64-unknown-linux-gnu + runner: ubuntu-22.04-arm + - target: x86_64-apple-darwin + runner: macos-15-intel + - target: aarch64-apple-darwin + runner: macos-15 + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7 + with: + persist-credentials: false + - uses: dtolnay/rust-toolchain@4360b52568e2003a75bf9bc1d59f33a8e3fc893c # stable + with: + targets: ${{ matrix.target }} + - uses: Swatinem/rust-cache@6323deb102c322ba6fcbdcafc7e3dddab59af2b6 # v2 + with: + key: release-${{ matrix.target }} + # Archive names carry no version, so `releases/latest/download/b10x-.tar.gz` is a + # stable URL that `SETUP.md` can name. + - name: Build, smoke-test and package + shell: bash + env: + TARGET: ${{ matrix.target }} + run: | + set -euo pipefail + cargo build --locked --release --bin b10x --target "$TARGET" + bin="target/$TARGET/release/b10x" + printed=$("$bin" --version | awk '{print $NF}') + if [ "$printed" != "$GITHUB_REF_NAME" ]; then + echo "::error::the built binary reports $printed, but the release tag is $GITHUB_REF_NAME" >&2 + exit 1 + fi + "$bin" setup guide | grep -q '^name: init$' + package="b10x-$TARGET" + mkdir -p "dist/$package" + cp "$bin" LICENSE "dist/$package/" + tar -C dist -czf "dist/$package.tar.gz" "$package" + - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: release-${{ matrix.target }} + path: dist/*.tar.gz + if-no-files-found: error + publish: name: Publish GitHub release - needs: gate + needs: [gate, package] runs-on: ubuntu-latest permissions: contents: write @@ -41,14 +95,41 @@ jobs: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7 with: fetch-depth: 0 - - name: Publish the annotated tag as a release + - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + pattern: release-* + path: dist + merge-multiple: true + - name: Verify archives and write checksums + run: | + set -euo pipefail + cd dist + for target in x86_64-unknown-linux-gnu aarch64-unknown-linux-gnu x86_64-apple-darwin aarch64-apple-darwin; do + test -f "b10x-$target.tar.gz" || { echo "::error::missing b10x-$target.tar.gz" >&2; exit 1; } + done + cp ../SETUP.md . + sha256sum b10x-*.tar.gz > SHA256SUMS + sha256sum --check SHA256SUMS + cat SHA256SUMS + - name: Publish the annotated tag as a release with its assets env: GH_TOKEN: ${{ github.token }} run: | + set -euo pipefail notes_file="${RUNNER_TEMP}/release-notes.md" - git for-each-ref --format='%(contents)' "refs/tags/${GITHUB_REF_NAME}" > "$notes_file" + # The version's CHANGELOG.md section, without its heading; the tag message if it has none. + awk -v heading="## [${GITHUB_REF_NAME}]" ' + index($0, heading) == 1 { found = 1; next } + found && /^## \[/ { exit } + found { print } + ' CHANGELOG.md > "$notes_file" + if ! grep -q '[^[:space:]]' "$notes_file"; then + git for-each-ref --format='%(contents)' "refs/tags/${GITHUB_REF_NAME}" > "$notes_file" + fi if gh release view "${GITHUB_REF_NAME}" >/dev/null 2>&1; then - gh release edit "${GITHUB_REF_NAME}" --verify-tag --draft=false --title "Agent Plugins ${GITHUB_REF_NAME}" --notes-file "$notes_file" + gh release edit "${GITHUB_REF_NAME}" --verify-tag --title "Agent Plugins ${GITHUB_REF_NAME}" --notes-file "$notes_file" else - gh release create "${GITHUB_REF_NAME}" --verify-tag --title "Agent Plugins ${GITHUB_REF_NAME}" --notes-file "$notes_file" + gh release create "${GITHUB_REF_NAME}" --verify-tag --draft --title "Agent Plugins ${GITHUB_REF_NAME}" --notes-file "$notes_file" fi + gh release upload "${GITHUB_REF_NAME}" dist/* --clobber + gh release edit "${GITHUB_REF_NAME}" --draft=false diff --git a/.github/workflows/tools.yml b/.github/workflows/tools.yml new file mode 100644 index 0000000..8e6c9b1 --- /dev/null +++ b/.github/workflows/tools.yml @@ -0,0 +1,24 @@ +name: Tools + +on: + push: + branches: [main] + pull_request: + schedule: + - cron: '17 5 * * *' + workflow_dispatch: + +permissions: + contents: read + +jobs: + pins: + name: Skills match the newest CLI releases + runs-on: ubuntu-latest + timeout-minutes: 15 + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7 + - uses: dtolnay/rust-toolchain@4360b52568e2003a75bf9bc1d59f33a8e3fc893c # stable + - uses: Swatinem/rust-cache@6323deb102c322ba6fcbdcafc7e3dddab59af2b6 # v2 + - name: Check every spelled command against the newest aep, ess and worktree releases + run: cargo run --quiet --locked --bin agentplugins-check -- tools diff --git a/AGENTS.md b/AGENTS.md index e24b623..fc8118c 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -1,81 +1,67 @@ # AGENTS.md — agentplugins -## Serves - -- **O2 — decisions as data, with evidence.** Publishes the curated AEP and ESS instruction - surfaces used to plan, develop and validate governed work. -- **O3 — any harness, observed and compared.** Keeps those instruction surfaces portable across - supported agent harnesses. - -## Invariants - -- Marketplace identity is `beyond10x` in every marketplace format. -- Keep exactly the focused plugin boundaries described in `README.md`. The `beyond10x` front door - may route to specialists and teach portable plugin authoring, but it must not absorb their - workflows or become a mixed catch-all. -- The AEP canonical command in instructions is `aep`. `protocol` is compatibility only and must not - become the authored spelling again. -- An authored document spells a CLI verb the way its area groups it: `aep govern|plan|drive|observe - ` (AEP 0.52.0) and `ess specify|generate|verify|infra ` (ESS 0.12.0). Every flat - spelling still works as a hidden alias with identical output, so nothing breaks — which is exactly - why a document teaching one is invisible to anything that runs a command. `agentplugins-check` - refuses one in any `.md` or `.yaml` this repository authors; the four exemptions are `CHANGELOG.md`, - `changes/`, `.engineering/` and `.github/workflows/`, the last because a workflow pins the binary - it runs and its spelling has to be the surface that version has. -- Do not mention or depend on retired plugin references, former marketplace identities, or the - historical source-repository name. -- Plugin folder names and manifest names are identical. -- Changes to a `SKILL.md` must pass the skill validator; plugin changes must pass the plugin - validator and `task check`. -- Anything executable in this repository is Rust. +The `b10x` marketplace: every Beyond10x plugin, and the `b10x` CLI that installs them with their +CLIs. Serves O2 (decisions as data) and O3 (any harness). + +## Map + +| path | what | +|---|---| +| `plugins//` | every plugin: `b10x`, `aep`, `ess`, `worktree`, `connectors` ([structure](website/docs/structure.md)) | +| `.claude-plugin/marketplace.json`, `.agents/plugins/marketplace.json` | the two marketplace files | +| `catalog.json` | products, plugins, binaries, retired names — no versions | +| `verified.json` | per CLI, the release the skills were last verified against; `tools` fails when a newer one is out | +| `crates/b10x/` | the setup CLI; `plan.rs` is pure and fixture-tested | +| `crates/agentplugins-check/` | the gate: marketplace, catalog, concept, retired names, CLI spellings, evals; `tools` checks skills against the newest CLI releases | +| `evals/` | eval corpus ([`evals/README.md`](evals/README.md)) | +| `website/` | public docs; must pass `task site-build` | +| `SETUP.md` | agent bootstrap, published as a release asset | +| `.agents/skills/improving-by-trial/` | how plugins are improved: isolated headless trials (`task trial:sandbox`, `task trial:run`), triage, fix, re-run | +| `trials/` | the round's trial definitions (`/trial.yaml`, fixtures) and `baseline.json`; `task trial:run TRIAL=` runs one, `agentplugins-check trial-report` measures it against the baseline | + +## Rules + +- Follow [`website/docs/structure.md`](website/docs/structure.md): rules R1–R8 decide plugin, skill, + agent and doc placement and names. `task check` enforces them (`crates/agentplugins-check/src/concept.rs`). +- `b10x` (`init`, `upgrade`, `setup`) is the only supported way to change a user's installed plugins; it snapshots first. +- Instructions spell the grouped CLI verbs (`aep plan …`, `ess specify …`) and `aep`, never + `protocol`. The gate refuses flat spellings outside `CHANGELOG.md`, `changes/`, `.engineering/` + and `.github/workflows/`. +- Retired names appear only where the gate allows them (`CHANGELOG.md`, `changes/`, + `.engineering/`, `catalog.json`, `crates/b10x/`, the checker's own table). +- Anything executable is Rust. +- Every `aep`, `ess` and `worktree` release is re-verified before `verified.json` moves to it: + `agentplugins-check tools`, then an ESS trial round ([`improving-by-trial`](.agents/skills/improving-by-trial/SKILL.md)). +- Trial findings for another repository become an issue there, labelled `trial-finding` by the + bot; the skill here documents the workaround until the fix is released ([`improving-by-trial`](.agents/skills/improving-by-trial/SKILL.md)). ## Gate ```console -task check +task check # fmt, clippy, tests, agentplugins-check (includes the trial definitions) +task site-build # when website/ changes +cargo run --locked --bin agentplugins-check -- tools # network: every spelled command exists in the newest aep, ess, worktree ``` -## Governed planning +## Planning -Use the repository's AEP store for any non-trivial implementation, cross-repository migration, or -release/deployment change. Before editing implementation files, run `aep plan artifact list` and -`aep plan artifact kinds`, create or select the artifact that owns the work, and keep its lifecycle, -scope, evidence, and relations current through `aep plan artifact` commands. Never substitute a -transient chat plan or direct edits under `.engineering/planning/` for that record. +Non-trivial work gets an artifact in the AEP store first (`aep plan artifact list`, then create or +select one) and keeps it current through `aep plan artifact`. Never edit `.engineering/planning/` by +hand. `aep:planning` is the instruction surface for this. -The `aep-plan` plugin is the canonical agent instruction surface for this workflow. When it is -not loaded, stop before planning-store writes and install or enable the release-pinned plugin using -the adopter instructions; do not improvise the store format from this file. +## Publishing -The adopter-facing Docusaurus site lives under `website/` and is published at -. A website change must also pass `task site-build`. -The networked site build stays outside the offline Rust gate. - -Commit and push through standalone `b10x-gates bot` with protected local credentials. This -repository never carries credential, token-minting or bot-authenticated git wrappers. - -## Releases - -Bare annotated tags are releases. Before tagging, `CHANGELOG.md`, the workspace version, and every -plugin manifest version must agree. The release workflow reruns `task check`, verifies that -agreement, and only then publishes the GitHub release. Public-site validation and Atlas -publication run independently and do not delay source-release completion. - -## Source publication - -This repository owns its correctness checks, required reviews and release artifacts. Ordinary -commits, pushes and releases require no Atlas checkout, current Atlas main or organization-wide -dependency admission. Use standalone `b10x-gates bot --repo . -- ` with protected local -credentials and the existing `b10x-bot[bot]` identity. Preserve repository and worktree hooks. - -Atlas documentation validation belongs to documentation operations; it is not a prerequisite for -source publication. Documentation failures affect documentation delivery. Organization privacy -rules still apply; historical brand exemptions do not authorize new public associations. +Commit and push through `b10x-gates bot --repo . -- ` as `b10x-bot[bot]`; keep hooks. +No credential or token machinery lives in this repository. A release is a bare annotated tag on +`main` after `CHANGELOG.md`, the workspace version and every carried plugin manifest agree; the +release workflow reruns the gate and publishes the `b10x` archives, `SHA256SUMS` and `SETUP.md`, +with the version's `CHANGELOG.md` section as the release notes. +Source publication needs no Atlas checkout. ## Public documentation operations -This repository owns the public source and presentation allowlist in `b10x.docs.yaml`. The generated credential-free `.github/workflows/b10x-docs-bundle.yml` passively packages only those declared files for the exact successful `main` commit; it must never run repository code. Atlas selects the latest successful bundle with every other catalog source, and Website plus Docs System own rendering, shared components, search, and feeds. Do not add a standalone docs deployer or put App credentials in this public repository. If Atlas catalogs a former Pages workflow, that file remains repository-owned validation: preserve its bespoke checks while keeping exact read-only permissions, an unconditional pull-request trigger, and no deployment primitives. Project Pages at `/agentplugins/` is only the generated stable redirect façade in `.github/workflows/b10x-docs-pages.yml`; content-only publication never rebuilds it. +This repository owns the public source and presentation allowlist in `b10x.docs.yaml`. The generated credential-free `.github/workflows/b10x-docs-bundle.yml` passively packages only those declared files for the exact successful `main` commit; it must never run repository code. The generated `.github/workflows/b10x-docs-check.yml` runs the publisher's per-source checks on every pull request and main push, with read-only contents and no credentials; it is deliberately separate from the shared gate, which runs on `pull_request_target` with a secret and never reads candidate source. Atlas selects the latest successful bundle with every other catalog source, and Website plus Docs System own rendering, shared components, search, and feeds. Do not add a standalone docs deployer or put App credentials in this public repository. If Atlas catalogs a former Pages workflow, that file remains repository-owned validation: preserve its bespoke checks while keeping exact read-only permissions, an unconditional pull-request trigger, and no deployment primitives. Project Pages at `/agentplugins/` is only the generated stable redirect façade in `.github/workflows/b10x-docs-pages.yml`; content-only publication never rebuilds it. From the complete organization workspace, verify the contract with a clean Atlas checkout at the current remote `main`. Set `B10X_ATLAS_CHECKOUT` to a managed Atlas worktree when the primary checkout is dirty or stale; never infer command availability from the primary alone. diff --git a/CHANGELOG.md b/CHANGELOG.md index cfbc622..911c023 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,6 +1,419 @@ # Changelog -## [Unreleased] +## [0.14.13] — 2026-09-26 + +Hardening a specification after its suite is green. + +- `ess:hardening`: eight specification-hardening techniques for after a conformance suite is green — + mutation, random sequences against a reference model, caller replay, determinism, metamorphic + relations, guard analysis, spec diff in the gate and design review — in cheapest-first order with + the design review early, and the rule that a check nobody has seen fail is not evidence. Its + `references/` carry a procedure per technique, the reference-model pattern (to ship as `ess` + features, beyond10x/ess#114), a design-review brief and spec-diff classification. (#35) +- `ess:conformance` offers the catalogue when a coverage task ends on a green suite, and + `ess:testing-conformance` links it from its mutation section. + +## [0.14.12] — 2026-09-26 + +Skills that start from what a repository already has, and plans that say what the other host needs. + +- `ess:init` checks a specification that is already there first: it runs `ess specify validate` and + the repository's own conformance command, reports the counts, and recommends + `ess:testing-conformance` when the suite is green. (#32) +- Every `b10x init` or `upgrade` command a plugin prints names `--host`, with the wording of + `b10x:init`; `agentplugins-check` refuses one that does not. (#32) +- `ess:testing-conformance` shows a TypeScript refusal recorder (`err.cause === ErrUnsupported`) + beside the Go one. (#32) +- The ESS syntax reference lists the structured predicate forms (`all`, `any`, `not`, `none`, map + operators, quantifiers, `.count`) that `when`, `invariants` and `filter` accept, no longer tells + authors to split "A or B" into two views, and names the forms `synthesize` still refuses. + (beyond10x/ess#92) +- `b10x init`/`upgrade --host ` says what a plan for the other host would do when it holds + Beyond10x installs — action count and legacy, missing and marketplace changes — and prints + "Nothing to change for ." while that plan has changes. When the other host is current, one + line says so. (#33) + +## [0.14.11] — 2026-09-26 + +Keeping installs current, keeping the skills matched to each product release, and pinning a tooling +version per repository. + +- `b10x check` (the session-start line) refreshes the newest releases from GitHub at most once a day, + in parallel with a short timeout, and names `/b10x:upgrade` when the installed plugins or a CLI are + older. It used to compare against a record written only by the last plan, so a machine on 0.14.7 + heard nothing about 0.14.10. +- `b10x upgrade` and `b10x setup plan` read plugin versions from the newest release when the local + marketplace copy is older than it; they reported 0.14.7 as current until the apply step refreshed + the copy. +- `verified.json` records the aep, ess and worktree releases the skills were last verified against; + the daily `agentplugins-check tools` fails when a newer release exists, until the skills are + re-verified and the file is bumped. +- `b10x pin ` and `b10x unpin ` keep a committed `b10x.toml` (`[pins]`, exact or + a minor line such as `0.32`). `init`, `upgrade`, `setup plan` and `install` install the pinned + release; `upgrade` reports newer ones and changes nothing; `b10x check` warns when the CLI on + `PATH` differs from the pin and when the skills describe a newer release. The matching `requires` + line in `ess-inputs.yaml` waits for beyond10x/ess#106. + +## [0.14.10] — 2026-09-25 + +From round 5, the first measured trial round: seven trials on 0.14.9, all isolated. The +full-package trial generated every ESS output and ran `go test` against the synthesized suite: 13 +passed, 0 failed, 0 skipped. + +- `agentplugins-check trial-report` recognises every spelling of `go test` (`go -C test`, + behind environment assignments, after `cd … &&`) and only in command position; it had reported the + full-package run as "not run". +- `b10x setup plan` with no selection keeps every installed product, optional ones included; it + proposed uninstalling an installed `connectors`. +- `ess:specifying`: every `--kind` of `ess generate` (`schema`, `openapi`, `asyncapi`, `docs`, + `site`, `docs-ir`) and `ess generate types`; one `--out` per kind side by side, because nested + output directories are refused; which kinds add their own folder (`site` does not). OpenAPI and + AsyncAPI need a component that says how the service is reached, or they write `0 artifact(s)` + without a refusal (beyond10x/ess#102). +- `ess:testing-conformance`: generate the Go suite inside the implementation's module; return payload + numbers as `float64` (beyond10x/ess#101). +- Trials: `cargo` is allowed (the Rust type library can be compiled); the aep, worktree and upgrade + prompts say nobody will answer questions. + +## [0.14.9] — 2026-09-25 + +Trials become measurable: their prompts live in the repository, and each run is scored against a +committed baseline, so a drop in quality fails instead of waiting to be noticed. + +- `trials//trial.yaml` defines each trial (prompt, fixture, setup, measures, expected + outputs); `task trial:run TRIAL=` builds the sandbox, prepares the fixture and runs it. Seven + trials: ESS new spec, retrofit, pipeline and full package; aep backlog planning; worktree + onboarding; upgrade from a seeded install. The gate checks the definitions. +- New trial `ess-full-package`: every ESS output (JSON Schema, OpenAPI, AsyncAPI, docs and site, + Rust and Go type libraries, the Go conformance suite), then an in-memory Go implementation run + with `go test` against the generated suite. Sandboxes get `go`. +- `agentplugins-check trial-report` measures a run: tool calls, the last `validate` verdict, + synthesis scenarios and refusals, `UNMAPPED:` markers read from the files written, which outputs + exist, and `go test` passed/failed/skipped. With `--baseline trials/baseline.json` it exits 1 when + a measure gets worse; `--write-baseline` records an accepted run. `trial:run` calls it after the + isolation check. + +## [0.14.8] — 2026-09-25 + +`b10x-harness` joins the catalog, and `b10x` chooses a prebuilt archive only where one exists for +the machine. + +- `catalog.json`: `b10x-harness` (the Beyond10x agent loop that Metaharness's `b10x` adapter runs) + is an optional binary of `aep`, installed with `b10x install b10x-harness`. It runs on Linux only: + a new `platforms` field on a catalog binary keeps `b10x` from planning it elsewhere. +- `b10x` reads which targets the newest release's `SHA256SUMS` lists and chooses prebuilt only when + this machine's target is among them; otherwise `cargo`, or a warning that names the target. A + release with an archive for some machines no longer sends the others to a download that fails. +- `b10x install ` follows the same default. Since 0.14.7 it still chose `cargo` whenever `cargo` + was on `PATH`, although its help and the docs said prebuilt. +- `aep:init`: `metaharness` is optional in the plan and installed with `b10x install metaharness`; + the skill said `b10x init aep` offers it. +- Docs name `b10x-harness` beside `metaharness`. + +## [0.14.7] — 2026-09-25 + +The operator's decision on the install method, asked for by a team adopting ESS: anyone with Rust +installed used to get a full source build of every CLI. + +- `b10x` installs CLIs from the release's checksummed prebuilt archive by default; `cargo` is used + when `--method cargo` asks for it or a release has no archive. Before, `cargo` on `PATH` made it + the default. +- `b10x:init` offers prebuilt first; the product `init` skills say which method is planned and + offer the other; `SETUP.md`, the install page and the plugin pages say the same. + +## [0.14.6] — 2026-09-25 + +The fixes for the trial findings shipped in aep 0.59.3 and ess 0.32.0; the skills now describe that +behaviour instead of the workarounds. Each change was checked against the released binaries. + +**`ess`** + +- `ess:retrofitting`: `ess infra import openapi` reads OpenAPI 3.0 as well as 3.1. A nullable field + imports as a coverage gap, carried into the draft as an `UNMAPPED:` marker; an object schema must + be closed with `additionalProperties: false`, and that refusal is reported, not worked around. +- Syntax reference: two fields are compared through one struct (`window.ends_at > + window.starts_at`); a right-hand side without a dot is a literal, so the flat form is refused. The + stored-value case links ESS's design note for guards over stored fields. +- `ess:specifying`: the note that the CLI help calls the `system.yaml` layout "legacy" is gone; the + help no longer says so. + +**`aep`** + +- `aep:planning`: the store's refusal of an empty findings fence now says to write `[]`; the skill + points at it instead of quoting the old message. + +## [0.14.5] — 2026-09-25 + +From the round-4 trials on 0.14.4 (an ESS retrofit of a stock-reservation service and worktree +onboarding, both isolated). The skills were checked against ess 0.31.0, the newest release, which +changes nothing they describe. + +**`ess`** + +- `ess:retrofitting`: a command the code silently ignores in some state gets no `wrong_state:` + outcome and no error. `synthesize` then reports `ESS-SYNTH-012` for that state and still writes + a scenario requiring that nothing happened, which is the code's behaviour; report it, and do not + add an error the code never raises. The view that invariants need is structural: mark it as such. + +**`worktree`** + +- `worktree:init`: the profile download creates its directory (`curl --create-dirs`) instead of a + separate `mkdir`; any local file works as the profile although `--help` says "committed"; a + headless run uses the directory that holds the current repository as the workspace root and + says so. +- `worktree:managing-worktrees`: always pass `--id` to `gc`. + +## [0.14.4] — 2026-09-25 + +From the round-3 trials on 0.14.3, the first product fix they led to (worktree 0.7.1), and eight +findings from a team adopting ESS. Two isolated trials on this version passed (an ESS retrofit and +worktree onboarding). + +**Setup** + +- `SETUP.md` step 1 asks the user before it installs `b10x`, showing the target, the download URL + and the install path. The product `init` skills say they install CLIs only after the user + confirms. + +**`ess`** + +- `ess:specifying`: never invent an entity so that a command type-checks. In an interactive session + the open `UNMAPPED:` markers become questions to the user; headless runs and the `author` agent + keep them and list them. Marker examples no longer say "Ask". +- `ess:specifying`: a second source document goes into the same system, a new system or an + external boundary, by what it says about itself; when it says neither, that is an open question. +- `ess:specifying`: `## Next` starts with the owner reviewing every marker and contradiction, then + planning (`aep:planning`), then generate and conformance. The `author` agent ends its report the + same way. +- `ess:specifying`: a headless run needs `--allowedTools "Bash(ess:*)"` (also on the plugin page). + Where a repository has no gate, validate plus a synthesis with 0 refusals is the check to report. +- Syntax reference: a predicate is one comparison or a bare fact path; `&&`, `||` and `in` are + refused and a list's length cannot be tested, so "state A or B" is two views. Error outcomes count + toward "all but one needs a `when`"; `wrong_state` outcomes do not. +- `ess:retrofitting`: a service that publishes no events declares one per success outcome for the + fact it produces and leaves it out of `publishes:`, since `validate` refuses an outcome with + nothing observable. An entity the code gives no status gets one structural state, marked as such. + +**`worktree`** + +- `worktree:init`: `worktree doctor --check` fails with `no active profile` since worktree 0.7.1; + the `profiles=0` workaround is gone. + +**Docs and tooling** + +- The docs intro and `b10x skill --help` name the current skills (`init`, `specifying`, …). +- `agentplugins-check trial-isolation`: a skill or agent from outside the sandbox fails a run only + when the run used it; one merely offered (claude.ai account skills in sub-agent sessions) is a + note. + +## [0.14.3] — 2026-09-25 + +From trial 3: five fresh agents on 0.14.2 (an ESS specification from nothing, an ESS retrofit of an +existing service, an ESS pipeline through generate and a synthesized suite, AEP planning over a +`TODO.md`, worktree onboarding) plus a migration of a seeded older install. + +- `ess:specifying`: lifecycle state names start upper-case; `ess-inputs.yaml` is documented and + recommended once anything is generated, since a directory without it is read whole; + `generate --out ` writes its own `schema/` or `openapi/` below ``; an entity invariant + needs a view holding every state and publishing the fields it reads, or `synthesize` refuses the + check. The syntax example gains that view: it synthesized with 3 refusals, now 12 scenarios and 0. +- The syntax reference says how to write rules the trials hit: a limit read from stored state, and + no double booking (a per-slot lifecycle, or `UNMAPPED:`). A time range is ordered by an `Integer` + length; comparing two `Timestamp`s validates and then fails `synthesize`, and `Duration` does not + compare with a number. +- `ess:retrofitting`: `ess infra import openapi` reads OpenAPI 3.1 only; `aep plan reverse openapi` + accepts 3.0. Its description named the retired `specify` and `coverage` skills. +- `ess:testing-conformance`: what the built-in targets `interpreted`, `oracle-fixture` and `billing` + are, why a new domain reports nothing useful against them, and `synthesize --target go|typescript` + to hold a real implementation to the suite. +- `aep:planning`: the `--protocols` and `--profile` values for `reverse init`; body files live in a + git-ignored directory of the repository, not `$TMPDIR`; a critic's empty findings fence is + recorded as `[]`, which the store accepts, since it refuses the empty fence. Two references to + `ess skill`, which no longer exists, now name `b10x skill ess:specifying`. +- `aep:migrating`: `reverse init` creates the store in every case; a backlog item that introduces a + new noun gets its ESS domain before its stories. Provenance says whose revisions it counts. +- `b10x:init` and `aep:init` offer `ess` with `aep`. +- The design and parallel-safety critics name the same two remedies for a shared file (an ordering + edge, or splitting the file) instead of asking for opposite changes in consecutive rounds. +- `worktree:init`: where to get a profile and where it goes; `doctor --check` exits 0 with + `profiles=0`; cleanup needs a remote. `worktree:managing-worktrees`: switch to a branch before the + first commit. +- `b10x:routing` routes ESS and worktree work to their skills instead of to `ess skill`. +- `b10x upgrade` installs the replacement of a legacy plugin it removes; it used to remove it and + only note the replacement as missing. A legacy `local` install is not reinstalled at `local` when + the user scope covers it. A plan names Beyond10x plugins on a host it does not cover. `b10x check` + no longer says nothing is set up when only a legacy plugin is installed. +- `agentplugins-check tools` also synthesizes and generates the ESS syntax example. +- `agentplugins-check` parses its arguments with clap; new `trial-isolation` checks that a headless + trial loaded only the plugins its sandbox installed, at the version under test, and no MCP server. + Sandboxes live outside `$HOME` (`/var/tmp/b10x-trials-$USER`): below it, a trial loaded the + operator's `CLAUDE.md` and a settings file an earlier trial left behind. `task trial:sandbox` and + `task trial:run`, and the repository skill `improving-by-trial`, describe the loop. +- The README lists every plugin's skills and agents, linked to their files; the gate keeps it + exact. The retired-name sweep covers `ess skill` and the `beyond10x/ess` marketplace. + +## [0.14.2] — 2026-09-24 + +From trial 2, in which fresh agents given only a sentence and this repository's link onboarded +(10 tool calls), wrote a valid todo specification in a new session that loaded the plugins +(9 calls, no ESS clone), and upgraded a seeded `ess` 0.29.0 after the session-start line named it. + +- `SETUP.md` creates `~/.cache` before `mktemp` in it; a fresh home has none. +- `b10x:init`'s plan example uses placeholders for the products and method it asked about, instead + of values an agent can paste unchanged. +- A product's `init` skips the install step when its CLI already answers, rather than asking the + install-method question again right after `b10x:init`. +- `ess:specifying`'s syntax example is a lending library, not a todo service: an agent asked for a + todo spec copied the example's domain. It now also shows a lifecycle with no final state + (`terminal: []`), which the trial agent had to guess; `ess specify validate` accepts it. +- `b10x upgrade` updates only what is installed: it no longer proposes new plugins, and a host with + nothing from Beyond10x on it is left alone with a note naming `b10x init --host …`. The upgrade + skills plan with `--host` for the host they run in. +- Plan output lists the marketplace refresh that `apply` runs, under "Also runs". + +## [0.14.1] — 2026-09-24 + +- Publishes what 0.14.0 could not: the 0.14.0 release workflow's smoke test still looked for the + retired `name: installing` in `b10x setup guide`, so all four package jobs failed and no GitHub + Release or archives were published for 0.14.0. The test now looks for `name: init`. Content is + 0.14.0's otherwise. + +## [0.14.0] — 2026-09-24 + +- **Every plugin lives here.** `ess` and `worktree` join `b10x`, `aep` and `connectors` in + `plugins/`, all at this repository's version; the marketplace entries that pointed into the ESS + and worktree repositories are gone. `ess@ess` and `worktree@worktree` installs are migrated by + `b10x` as before. +- **`init` and `upgrade` on every plugin.** `/b10x:init` is guided onboarding: it asks what the user + wants to do (plan and deliver work, write specifications, isolated Git checkouts, integrations) + and how to install the CLIs, then installs the matching plugins. `/aep:init`, `/ess:init`, + `/worktree:init`, `/connectors:init` set up one product and take its first step; + `/:upgrade` and `/b10x:upgrade` check plugins and CLIs and offer the upgrade. +- **Breaking:** skill renames under the concept's R3: `b10x:init` (was `installing`), + `ess:init` (was `ess`), `ess:specifying`, `ess:retrofitting`, `ess:testing-conformance` (were + `specify`, `retrofit`, `coverage`), `worktree:managing-worktrees` (was `worktree`). +- `b10x init ` and `b10x upgrade []` plan only the named or installed products + and touch nothing else. CLIs install with `cargo` when it is on `PATH`, otherwise from the + release's checksummed prebuilt archive; `--method cargo|prebuilt` chooses, and a release without + archives falls back to `cargo`. Every CLI is bound to its newest release; `catalog.json` is format + `b10x.catalog/2` with both methods and an `intent` phrase per product. Plans end with `Next:` hints. +- `b10x check` (session start) now says when a CLI is older than the newest release last seen, when + that was more than seven days ago, and when no product is set up yet (`/b10x:init`). +- `ess:specifying` gains `references/syntax.md`: a todo service in three files covering types, + entities, relations, lifecycles, actors, errors, commands (created, moved, refused), events, views + and a component, validated by `ess specify validate`. `ess:testing-conformance` says why a new + domain's suite is all `unsupported`. From trial 1, whose agent had to clone ESS for this. +- New `agentplugins-check tools` replaces `remote`: it downloads the newest `aep`, `ess` and + `worktree` releases, runs every command the skills spell with `--help`, and validates the ESS syntax + example (pull requests, `main` pushes, daily). Its first run found `drive.md` spelling a command + `aep` does not have. Skills quote no CLI version (the concept check refuses one); nine quotes were + removed. + +## [0.13.1] — 2026-09-24 + +From a trial in which an agent given only this repository's link specified a todo service with +`ess`: it finished with a valid specification, and these were the steps that cost it time. + +- New `b10x skill ` lists an installed plugin's skills and agents, and + `b10x skill :` prints one, from Claude Code's recorded install or Codex's cache. A + host loads new plugins only in a new session; the session that ran setup reads the skill through + this instead of searching the plugin cache. `b10x setup apply` and `SETUP.md` (new step 3) say so. +- The `b10x:installing` skill plans for the host it runs in (`--host claude` or `--host codex`) and + adds the other only when the user uses it; the trial had installed into Codex for a Claude-only + task. It writes the plan with `--out` alone, so the full JSON no longer floods the session. + +## [0.13.0] — 2026-09-24 + +- **The plugin concept, enforced.** [`website/docs/structure.md`](website/docs/structure.md) states + eight rules (one marketplace; one plugin per product named after its CLI; skills are activities in + `-ing` form; each agent owned by one skill that lists it under `## Agents`; references resolve; + one README row, plugin page and sidebar entry per plugin). The new `concept` check in + `agentplugins-check` refuses a violation; `agentplugins-check remote` lists the renames `ess` + and `worktree` still owe. +- **Breaking:** `aep-plan` and `aep-drive` are one plugin, `aep`, with three skills: + `aep:planning`, `aep:migrating` (was `story-migration`) and `aep:implementing` (was `wave` + and `drive`, now its two modes), and eleven agents. `b10x setup` replaces `aep-plan` and + `aep-drive` installs from any marketplace, at every scope, with one `aep` install. +- **Breaking:** skills renamed to activities: `b10x:installing` (was `setup`), `b10x:routing` + (was `guide`), `b10x:authoring-plugins` (was `plugin-creator`), `connectors:integrating` + (was `connectors`). Every old id joins the retired names the gate refuses. +- `b10x setup undo` refreshes both hosts' marketplaces after restoring settings, so a restored + registration whose snapshot `apply` removed works again (Codex failed to list plugins until + `codex plugin marketplace upgrade`). A host whose plugin commands fail is reported with its error + and the repair command instead of as not installed. +- README is one paragraph and a table of plugins with the install command for each host; the `b10x` + reference moved to its plugin page, eval costs and CI to `evals/README.md`, repository rules to + `AGENTS.md`. + +## [0.12.0] — 2026-09-24 + +- **One-sentence onboarding.** Tell an agent to follow + `https://github.com/beyond10x/agentplugins/releases/latest/download/SETUP.md`: it installs the new + `b10x` binary, which reads both hosts (`claude`/`codex plugin list --json`, marketplace lists, + `~/.codex/config.toml`), every catalog binary on `PATH` and the plugin entries in known projects' + settings, and prints findings and exact actions (`b10x setup plan`). The agent asks which products + to have and confirms the list; `b10x setup apply` snapshots every file it changes, runs the actions, + and re-plans to show the result converged; `b10x setup undo` restores the snapshot. +- Setup migrates earlier installs: retired plugin ids and marketplaces (`beyond10x`, `ess`, + `worktree`) on both hosts and at user, project and local scope, orphan `enabledPlugins` entries + whose marketplace is gone, and a `b10x` registration pinned to a ref. It installs `ess` and + `worktree` binaries at the version of the plugin that describes them and `aep` at its newest + release, from checksummed release archives or `cargo install --git --tag`, into the directory of + the copy that runs, and names shadowed copies later on `PATH`. +- **Breaking:** the front-door plugin `beyond10x` is now `b10x`: skills `b10x:setup` (new), + `b10x:guide` (was `beyond10x:beyond10x`) and `b10x:plugin-creator`, plus a SessionStart hook that + runs `b10x check` and prints one line per plugin/binary drift or legacy install. +- **This repository names no version of what it points at.** `ess` and `worktree` are `git-subdir` + entries with repository and path only, in both marketplace files; 0.11.0's tag+commit pin on + `worktree` is gone. `catalog.json` is version-free. `agentplugins-check` refuses a remote entry + with `ref`, `sha` or `version` and a catalog that disagrees with the marketplaces; + `agentplugins-check remote` now refuses a product whose default branch serves a plugin version it + never released. +- `ess@b10x` is back in the catalog, pointing at the ESS repository; `ess@ess` installs are migrated. +- Correction: 0.11.0 said Codex could not install from another repository. Codex marketplaces accept + `git-subdir` entries (verified with `worktree` at 0.6.0); both hosts now install `worktree@b10x` + and `ess@b10x`. +- Release archives `b10x-.tar.gz` for four targets, `SHA256SUMS` and `SETUP.md` are published + with each release; the names carry no version so `releases/latest/download/…` is stable. + +## [0.11.0] — 2026-09-24 + +- **Breaking:** the marketplace identity is `b10x` in both formats. Installs name plugins as + `@b10x`. A registration made from an earlier release is named `beyond10x`: remove its + plugins and run `/plugin marketplace remove beyond10x` (Claude Code) or + `codex plugin marketplace remove beyond10x` (Codex), then register this release. +- **Breaking:** `workspace-hygiene` is retired. Worktree ships its own plugin, `worktree`, from the + `beyond10x/worktree` repository at the version of the `worktree` binary, and the skill + `workspace-hygiene:worktree` becomes `worktree:worktree`. The copy here released on its own + cadence and had drifted: `workspace-hygiene` 0.10.0 lacked the patch-equivalent recovery-proof + guidance Worktree 0.5.0 generates. +- The Claude Code marketplace lists `worktree` as a `git-subdir` entry pinned to Worktree `0.6.0` + and its full commit, carrying no copy. Codex reads `worktree@worktree` from the worktree + repository's own marketplace. +- `agentplugins-check` checks the entry's shape offline. The new `agentplugins-check remote` + refuses a commit that is not the tag's, manifests at that commit that declare another version, + and a tag that is not the newest Worktree release; `.github/workflows/remote-plugins.yml` runs it + on every pull request, `main` push and daily. `workspace-hygiene` joins the retired names. +- The routing skill, the wave skill, README, install guide and website point at `worktree`; the + eval corpus names 4 of 7 skills. Eval checker tests that used `workspace-hygiene` as a fixture + use `connectors`. + +## [0.10.0] — 2026-09-23 + +- **Breaking:** `ess-specify` is retired from this marketplace. ESS now ships its own plugin, `ess`, + from the `beyond10x/ess` repository at the same version as the `ess` binary, and the binary prints + the same skills through `ess skill`. The copy here released on its own cadence and had drifted: + `0.9.2` told adopters to install ESS `0.22.1` while ESS was at `0.29.0`. Replace + `ess-specify@beyond10x` with `ess@ess` after adding the ESS repository as a marketplace. +- The `beyond10x` front door and the `aep-plan` planning skill route spec-driven work to `ess:specify` + (or `ess skill specify`). `ess-specify` joins the retired names `agentplugins-check` refuses. +- The eval case `ess-specify-new-entity` is removed; the corpus has eight cases, and + `golden-path-end-to-end` no longer names an ESS skill in its subject. Its step 3 still drafts and + validates an ESS domain. +- Workspace, every plugin manifest, the three version-stamped skills and every install pin move to + `0.10.0`. + +## [0.9.3] — 2026-09-18 - New skill `ess-specify:coverage`. It covers the half of ESS work `specify` does not: raising and auditing a conformance suite's coverage against a real implementation, and judging whether a @@ -20,6 +433,14 @@ uniformly while a CPU limit still allows a full-speed burst. - `ess-specify`'s manifests, the `README.md` boundary line and the eval-coverage counts follow: 10 skills rather than 9, with `ess-specify:coverage` listed among the uncovered ones. +- New agent `aep-drive:security-reviewer`. It performs the same independent verification the + existing reviewer does — a failing conformance test for every invariant that is not yet enforced, + the same `CONFIRMED`/`NEEDS-CHANGE`/`INFEASIBLE` verdicts, the same origin axis, the same + `findings` block and report shape, so a coordinator can dispatch it wherever it dispatches a + review. It differs only in framing: it states the work as defensive verification of a change's own + invariants, without combative language, so a review of privileged, integrity- or + boundary-sensitive code is not misread as an exploit by a platform safety classifier and killed + before it runs. The existing adversary agent is unchanged. ## [0.9.2] — 2026-09-15 diff --git a/Cargo.lock b/Cargo.lock index d9ea5bd..656897e 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -4,8 +4,9 @@ version = 4 [[package]] name = "agentplugins-check" -version = "0.9.2" +version = "0.14.13" dependencies = [ + "clap", "serde", "serde_json", "serde_yaml", @@ -23,6 +24,76 @@ dependencies = [ "memchr", ] +[[package]] +name = "anstream" +version = "1.0.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "824a212faf96e9acacdbd09febd34438f8f711fb84e09a8916013cd7815ca28d" +dependencies = [ + "anstyle", + "anstyle-parse", + "anstyle-query", + "anstyle-wincon", + "colorchoice", + "is_terminal_polyfill", + "utf8parse", +] + +[[package]] +name = "anstyle" +version = "1.0.14" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "940b3a0ca603d1eade50a4846a2afffd5ef57a9feac2c0e2ec2e14f9ead76000" + +[[package]] +name = "anstyle-parse" +version = "1.0.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "52ce7f38b242319f7cabaa6813055467063ecdc9d355bbb4ce0c68908cd8130e" +dependencies = [ + "utf8parse", +] + +[[package]] +name = "anstyle-query" +version = "1.1.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "40c48f72fd53cd289104fc64099abca73db4166ad86ea0b4341abe65af83dadc" +dependencies = [ + "windows-sys", +] + +[[package]] +name = "anstyle-wincon" +version = "3.0.11" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "291e6a250ff86cd4a820112fb8898808a366d8f9f58ce16d1f538353ad55747d" +dependencies = [ + "anstyle", + "once_cell_polyfill", + "windows-sys", +] + +[[package]] +name = "b10x" +version = "0.14.13" +dependencies = [ + "clap", + "serde", + "serde_json", + "sha2", + "toml", +] + +[[package]] +name = "block-buffer" +version = "0.10.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3078c7629b62d3f0439517fa394996acacc5cbc91c5a20d8c658e77abd503a71" +dependencies = [ + "generic-array", +] + [[package]] name = "cc" version = "1.4.5" @@ -33,6 +104,87 @@ dependencies = [ "shlex 2.0.1", ] +[[package]] +name = "cfg-if" +version = "1.0.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4e7648175b45a9a48536d676f68d918270699102aa8dab5496df06904c914600" + +[[package]] +name = "clap" +version = "4.6.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "aa8876b300ab35ba921adea3dfd70157a46249b33f95c9084ae5709785478946" +dependencies = [ + "clap_builder", + "clap_derive", +] + +[[package]] +name = "clap_builder" +version = "4.6.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ec0797fb7aeb1406c84efac526901f7ec3ead2124f946b494e72879d4b54704d" +dependencies = [ + "anstream", + "anstyle", + "clap_lex", + "strsim", +] + +[[package]] +name = "clap_derive" +version = "4.6.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f9c751b79415d4e559e3d1fcf128e09e720eb673a06d26cf6f392d37d75b66e0" +dependencies = [ + "heck", + "proc-macro2", + "quote", + "syn", +] + +[[package]] +name = "clap_lex" +version = "1.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1c133bc6a41be0d194c306b5506d15e6feeea7b1d6604bd3f8310dfb2ca96486" + +[[package]] +name = "colorchoice" +version = "1.0.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1d07550c9036bf2ae0c684c4297d503f838287c83c53686d05370d0e139ae570" + +[[package]] +name = "cpufeatures" +version = "0.2.17" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "59ed5838eebb26a2bb2e58f6d5b5316989ae9d08bab10e0e6d103e656d1b0280" +dependencies = [ + "libc", +] + +[[package]] +name = "crypto-common" +version = "0.1.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "78c8292055d1c1df0cce5d180393dc8cce0abec0a7102adb6c7b1eef6016d60a" +dependencies = [ + "generic-array", + "typenum", +] + +[[package]] +name = "digest" +version = "0.10.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9ed9a281f7bc9b7576e61468ba615a66a5c8cfdff42420a70aa82701a3b1e292" +dependencies = [ + "block-buffer", + "crypto-common", +] + [[package]] name = "equivalent" version = "1.0.2" @@ -45,12 +197,28 @@ version = "0.1.12" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "3e0f1c7c3a72c66fd80abe965175f7523475c0489a87d3ff9d6e8c87d87a9d2d" +[[package]] +name = "generic-array" +version = "0.14.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "85649ca51fd72272d7821adaf274ad91c288277713d9c18820d8499a7ff69e9a" +dependencies = [ + "typenum", + "version_check", +] + [[package]] name = "hashbrown" version = "0.17.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "ed5909b6e89a2db4456e54cd5f673791d7eca6732202bbf2a9cc504fe2f9b84a" +[[package]] +name = "heck" +version = "0.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2304e00983f87ffb38b55b444b5e3b60a884b5d30c0fca7d82fe33449bbe55ea" + [[package]] name = "indexmap" version = "2.14.1" @@ -61,18 +229,36 @@ dependencies = [ "hashbrown", ] +[[package]] +name = "is_terminal_polyfill" +version = "1.70.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a6cb138bb79a146c1bd460005623e142ef0181e3d0219cb493e02f7d08a35695" + [[package]] name = "itoa" version = "1.0.18" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682" +[[package]] +name = "libc" +version = "0.2.189" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3eaf3ede3fee6db1a4c2ee091bf8a8b4dccdc6d17f656fb07896ee72867612f2" + [[package]] name = "memchr" version = "2.8.3" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "cf8baf1c55e62ffcace7a9f06f4bd9cd3f0c4beb022d3b367256b91b87513d98" +[[package]] +name = "once_cell_polyfill" +version = "1.70.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "384b8ab6d37215f3c5301a95a4accb5d64aa607f1fcb26a11b5303878451b4fe" + [[package]] name = "proc-macro2" version = "1.0.107" @@ -170,6 +356,15 @@ dependencies = [ "zmij", ] +[[package]] +name = "serde_spanned" +version = "1.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6662b5879511e06e8999a8a235d848113e942c9124f211511b16466ee2995f26" +dependencies = [ + "serde_core", +] + [[package]] name = "serde_yaml" version = "0.9.34+deprecated" @@ -183,6 +378,17 @@ dependencies = [ "unsafe-libyaml", ] +[[package]] +name = "sha2" +version = "0.10.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a7507d819769d01a365ab707794a4084392c824f54a7a6a7862f8c3d0892b283" +dependencies = [ + "cfg-if", + "cpufeatures", + "digest", +] + [[package]] name = "shlex" version = "1.3.0" @@ -201,6 +407,12 @@ version = "0.1.9" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "2b2231b7c3057d5e4ad0156fb3dc807d900806020c5ffa3ee6ff2c8c76fb8520" +[[package]] +name = "strsim" +version = "0.11.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7da8b5736845d9f2fcb837ea5d9e2628564b3b043a70948a3f0b778838c5fb4f" + [[package]] name = "syn" version = "3.0.4" @@ -212,6 +424,45 @@ dependencies = [ "unicode-ident", ] +[[package]] +name = "toml" +version = "0.9.12+spec-1.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cf92845e79fc2e2def6a5d828f0801e29a2f8acc037becc5ab08595c7d5e9863" +dependencies = [ + "indexmap", + "serde_core", + "serde_spanned", + "toml_datetime", + "toml_parser", + "toml_writer", + "winnow 0.7.15", +] + +[[package]] +name = "toml_datetime" +version = "0.7.5+spec-1.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "92e1cfed4a3038bc5a127e35a2d360f145e1f4b971b551a2ba5fd7aedf7e1347" +dependencies = [ + "serde_core", +] + +[[package]] +name = "toml_parser" +version = "1.1.3+spec-1.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1d38ac1cf9b95face32296c0a3ede1fdc270627c9d9c02a7274dd6d960dc4d56" +dependencies = [ + "winnow 1.0.4", +] + +[[package]] +name = "toml_writer" +version = "1.1.2+spec-1.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7d56353a2a665ad0f41a421187180aab746c8c325620617ad883a99a1cbe66d2" + [[package]] name = "tree-sitter" version = "0.25.10" @@ -242,6 +493,12 @@ version = "0.1.8" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "ca0d1bf6fdd806e43ae5198f82f527056d359def39e54e67a0f478ac09dac081" +[[package]] +name = "typenum" +version = "1.20.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b6f5e870be6c3b371b77fe0ee0bafb859fa4964b4404c27de1d380043c4dda20" + [[package]] name = "unicode-ident" version = "1.0.24" @@ -254,6 +511,45 @@ version = "0.2.11" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "673aac59facbab8a9007c7f6108d11f63b603f7cabff99fabf650fea5c32b861" +[[package]] +name = "utf8parse" +version = "0.2.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "06abde3611657adf66d383f00b093d7faecc7fa57071cce2578660c9f1010821" + +[[package]] +name = "version_check" +version = "0.9.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0b928f33d975fc6ad9f86c8f283853ad26bdd5b10b7f1542aa2fa15e2289105a" + +[[package]] +name = "windows-link" +version = "0.2.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f0805222e57f7521d6a62e36fa9163bc891acd422f971defe97d64e70d0a4fe5" + +[[package]] +name = "windows-sys" +version = "0.61.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ae137229bcbd6cdf0f7b80a31df61766145077ddf49416a728b02cb3921ff3fc" +dependencies = [ + "windows-link", +] + +[[package]] +name = "winnow" +version = "0.7.15" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "df79d97927682d2fd8adb29682d1140b343be4ac0f08fd68b7765d9c059d3945" + +[[package]] +name = "winnow" +version = "1.0.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "23b97319f7b8343df12cc98938e5c3eb436064524c8d2b4e30a1d3a36eecdf81" + [[package]] name = "zmij" version = "1.0.23" diff --git a/Cargo.toml b/Cargo.toml index 58aea8b..2c41d0f 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -1,9 +1,9 @@ [workspace] resolver = "2" -members = ["crates/agentplugins-check"] +members = ["crates/agentplugins-check", "crates/b10x"] [workspace.package] -version = "0.9.2" +version = "0.14.13" edition = "2021" rust-version = "1.85" license = "Apache-2.0" @@ -20,6 +20,9 @@ serde_yaml = "0.9" tree-sitter = "0.25.10" tree-sitter-bash = "0.25.1" shlex = "1.3" +clap = { version = "4", features = ["derive"] } +sha2 = "0.10" +toml = "0.9" [workspace.lints.rust] missing_docs = "warn" diff --git a/README.md b/README.md index da11fef..f99b7b1 100644 --- a/README.md +++ b/README.md @@ -1,138 +1,41 @@ # Beyond10x Agent Plugins -Curated marketplace identity: `beyond10x`. - -The repository deliberately contains six focused plugins: - -- `beyond10x`: marketplace navigation, public resource discovery, and portable plugin creation. -- `aep-plan`: governed planning, decomposition, plan review, and reverse engineering. -- `aep-drive`: wave coordination, story scoping, implementation, and adversarial review. -- `ess-specify`: ESS specification, validation and deterministic schema/OpenAPI projection - guidance, and conformance-suite coverage against a real implementation. -- `workspace-hygiene`: safe creation, leases, publication checks, and cleanup for Git worktrees. -- `connectors`: provider setup, connection diagnostics, and governed CLI operation invocation. - -`beyond10x` is the front door, not a catch-all. It routes a task to the smallest specialist and -keeps plugin-creation workflows portable by making shared skills the canonical implementation for -Codex and Claude Code. It does not copy or replace the specialists' instructions. - -## Install - -The `connectors` plugin is included in release `0.9.2` for both hosts. -See the [Connectors installation guide](website/docs/plugins/connectors.md) -for both hosts. Install the standalone `connectors` CLI first. - -`aep-plan` and `aep-drive` drive the `aep` CLI; `ess-specify` drives the `ess` CLI. Install the -verified Linux or macOS archives for [AEP `0.55.0`](https://github.com/beyond10x/aep/releases/tag/0.55.0) -and [ESS `0.22.1`](https://github.com/beyond10x/ess/releases/tag/0.22.1) first, then check them with -`aep --version` and `ess --version`. The complete download and checksum commands are in -[`website/docs/install.md`](website/docs/install.md), which also names the Metaharness build that -`aep-drive`'s `drive` skill needs for `metaharness aep drive`. That build links its own copy of AEP -rather than the binary you install here: `metaharness`'s `crates/metaharness-aep/Cargo.toml` pins -the `aep-*` crates to a `beyond10x/aep` git revision — at Metaharness `0.7.0` that is `a23176ae`, -which `git describe --tags` reports as `0.54.0-33-ga23176ae` — so read that manifest for the -Metaharness↔AEP pair instead of assuming it matches the `0.55.0` on your `PATH`. - -Paste this pinned block into a Claude Code session: - -```text -/plugin marketplace add https://github.com/beyond10x/agentplugins.git#0.9.2 -/plugin install aep-plan@beyond10x -/plugin install aep-drive@beyond10x -/plugin install ess-specify@beyond10x -/reload-plugins -``` - -Add `/plugin install beyond10x@beyond10x` for the front door and -`/plugin install workspace-hygiene@beyond10x` for managed worktrees, or -`/plugin install connectors@beyond10x` for integrations. Codex offers the same plugins -from the same repository, following `.agents/plugins/marketplace.json`; its exact non-interactive -CLI bootstrap and upgrade commands are in [`website/docs/install.md`](website/docs/install.md). - -Codex marketplace metadata lives at `.agents/plugins/marketplace.json`; Claude plugin marketplace -metadata lives at `.claude-plugin/marketplace.json`. Each plugin owns its own manifest and only the -skills or agents in its stated scope. - -Run `task check` before publishing. The gate fails on missing focused content, mismatched plugin -names, a marketplace identity other than `beyond10x`, or plugin versions that disagree with the -workspace release. Run `task site-build` for the public documentation under `website/`. - -The adopter guide is published at . This repository -contains no credential or bot-token delivery machinery; release mutations are performed through -the private organization tooling outside this tree. - -## Evals - -Nine cases under [`evals/`](evals/) — one `eval-case/1` each, judged by a `trace-spec/1` document, -run by the `aep` CLI — name 7 of this repository's 10 agents and 5 of its 10 skills in their -`subject:` fields. Those `subject:` fields are the source of truth for eval coverage: every -coverage number here and in [`evals/README.md`](evals/README.md) is counted from -`evals/*/case.yaml`, never from a prose table. Covered: the agents `aep-drive:adversary`, `aep-drive:story-scoper`, -`aep-plan:decomposer`, `aep-plan:plan-critic-acceptance`, `aep-plan:plan-critic-design`, -`aep-plan:plan-critic-parallel-safety` and `aep-plan:plan-critic-scope`, and the skills -`aep-drive:drive`, `aep-drive:wave`, `aep-plan:planning`, `connectors:connectors` and -`ess-specify:specify`. Not covered: the agents `aep-drive:implementor`, `aep-plan:plan-reviewer` -and `aep-plan:reverse-engineer`, and the skills `aep-plan:story-migration`, `beyond10x:beyond10x`, -`beyond10x:plugin-creator`, `ess-specify:coverage` and `workspace-hygiene:worktree`. A change that breaks a covered charter -turns a row red instead of being noticed by a reader; a change to an uncovered one does not. - -Free, offline, and part of `task check`: - -```console -$ task evals -valid: 9 eval case(s), 1 recorded transcript(s) replayed -``` - -It validates every case, resolves every `subject:` to an agent or a skill that exists here, and -replays whatever transcripts are recorded with `aep drive eval run --stream`, which spends nothing. An -empty `recorded/` and a machine with no `aep` on `PATH` are both printed notices, never a red gate. - -Live, which costs money: - -```console -$ METAHARNESS_LIVE=1 metaharness aep drive eval run --corpus evals --workflow adp/default \ - --arm plugin --harness claude --plugin-dir plugins/aep-plan \ - --cwd --budget-usd 20 --assume-usd-per-run 5 \ - --observed-at --redact --out -``` - -Without `METAHARNESS_LIVE=1` the runner accepts the corpus and refuses to spawn, by name: - -```console -$ aep drive eval run --corpus evals --workflow adp/default --arm plugin --harness claude \ - --out eval-out --observed-at 2026-09-03 -error: eval-out — 1 refusal(s): - EVAL-RUN-002 a spawn costs money and `METAHARNESS_LIVE=1` is not in this environment. Set it - deliberately, or pass `--stream FILE` to ingest a run that already happened, which spends nothing -``` - -### What a full live run costs - -| | | -|---|---| -| cases in the corpus | **9** | -| per-case cap | **$5** — `story:plugin-eval-cases`, the operator's default | -| one full run, one arm, one harness | **$45** | -| `EVAL_BUDGET_USD` default | **$20** — `story:eval-ci-gates`, the operator's default | - -**So a full sweep does not fit its own default budget, and that is the intended behaviour rather -than an oversight.** `.github/workflows/eval.yml` computes `cases × $5` before it installs a tool, -and refuses with those four numbers in the check summary when the product exceeds -`EVAL_BUDGET_USD`. What fits inside $20 is a diff-scoped run of up to four cases, which is what a -pull request touching one agent or one skill actually selects. Running the whole corpus is a -deliberate act: raise the repository variable, or dispatch one case at a time. - -The cap is a **cap, not an estimate** — no recorded run has priced this corpus yet, so nothing here -claims a full sweep will cost $45 rather than refusing above it. `--assume-usd-per-run` is what the -runner charges a run whose stream states no cost, and it is set to the per-case cap so the runner's -own pre-spawn check is made against the budgeted number and not against its optimistic default. - -### CI - -`ci.yml` runs the free half on every pull request and it is what blocks a merge. `eval.yml` runs the -live arm only with the `run-eval` label or a manual dispatch, only for the cases whose subject the -diff touched, under the budget above, with the organization bot's credential and never a personal -key. It informs; it does not gate. +Agent plugins for Claude Code and Codex, from one marketplace: `b10x`. + +Add the marketplace once — Claude Code: `/plugin marketplace add beyond10x/agentplugins` · +Codex: `codex plugin marketplace add beyond10x/agentplugins` — then install what you need: + +| plugin | for | Claude Code | Codex | +|---|---|---|---| +| [`ess`](website/docs/plugins/ess.md) | Executable System Specifications | `/plugin install ess@b10x` | `codex plugin add ess@b10x` | +| [`aep`](website/docs/plugins/aep.md) | governed planning and delivery | `/plugin install aep@b10x` | `codex plugin add aep@b10x` | +| [`worktree`](website/docs/plugins/worktree.md) | isolated Git worktrees | `/plugin install worktree@b10x` | `codex plugin add worktree@b10x` | +| [`connectors`](website/docs/plugins/connectors.md) | integrations | `/plugin install connectors@b10x` | `codex plugin add connectors@b10x` | +| [`b10x`](website/docs/plugins/b10x.md) | setup, upgrades, routing | `/plugin install b10x@b10x` | `codex plugin add b10x@b10x` | + +What each plugin ships: + + +- [`ess`](plugins/ess/) · [docs](website/docs/plugins/ess.md) + - skills: [`init`](plugins/ess/skills/init/SKILL.md) · [`upgrade`](plugins/ess/skills/upgrade/SKILL.md) · [`hardening`](plugins/ess/skills/hardening/SKILL.md) · [`retrofitting`](plugins/ess/skills/retrofitting/SKILL.md) · [`specifying`](plugins/ess/skills/specifying/SKILL.md) · [`testing-conformance`](plugins/ess/skills/testing-conformance/SKILL.md) + - agents: [`author`](plugins/ess/agents/author.md) · [`conformance`](plugins/ess/agents/conformance.md) · [`retrofitter`](plugins/ess/agents/retrofitter.md) +- [`aep`](plugins/aep/) · [docs](website/docs/plugins/aep.md) + - skills: [`init`](plugins/aep/skills/init/SKILL.md) · [`upgrade`](plugins/aep/skills/upgrade/SKILL.md) · [`implementing`](plugins/aep/skills/implementing/SKILL.md) · [`migrating`](plugins/aep/skills/migrating/SKILL.md) · [`planning`](plugins/aep/skills/planning/SKILL.md) + - agents: [`adversary`](plugins/aep/agents/adversary.md) · [`decomposer`](plugins/aep/agents/decomposer.md) · [`implementor`](plugins/aep/agents/implementor.md) · [`plan-critic-acceptance`](plugins/aep/agents/plan-critic-acceptance.md) · [`plan-critic-design`](plugins/aep/agents/plan-critic-design.md) · [`plan-critic-parallel-safety`](plugins/aep/agents/plan-critic-parallel-safety.md) · [`plan-critic-scope`](plugins/aep/agents/plan-critic-scope.md) · [`plan-reviewer`](plugins/aep/agents/plan-reviewer.md) · [`reverse-engineer`](plugins/aep/agents/reverse-engineer.md) · [`security-reviewer`](plugins/aep/agents/security-reviewer.md) · [`story-scoper`](plugins/aep/agents/story-scoper.md) +- [`worktree`](plugins/worktree/) · [docs](website/docs/plugins/worktree.md) + - skills: [`init`](plugins/worktree/skills/init/SKILL.md) · [`upgrade`](plugins/worktree/skills/upgrade/SKILL.md) · [`managing-worktrees`](plugins/worktree/skills/managing-worktrees/SKILL.md) +- [`connectors`](plugins/connectors/) · [docs](website/docs/plugins/connectors.md) + - skills: [`init`](plugins/connectors/skills/init/SKILL.md) · [`upgrade`](plugins/connectors/skills/upgrade/SKILL.md) · [`integrating`](plugins/connectors/skills/integrating/SKILL.md) +- [`b10x`](plugins/b10x/) · [docs](website/docs/plugins/b10x.md) + - skills: [`init`](plugins/b10x/skills/init/SKILL.md) · [`upgrade`](plugins/b10x/skills/upgrade/SKILL.md) · [`authoring-plugins`](plugins/b10x/skills/authoring-plugins/SKILL.md) · [`routing`](plugins/b10x/skills/routing/SKILL.md) + + +Most plugins drive a CLI of the same name. To have an agent install plugins and matching CLIs, and +migrate older installs, tell it: *"Set up Beyond10x: follow +https://github.com/beyond10x/agentplugins/releases/latest/download/SETUP.md"*. + +More: [install guide](website/docs/install.md) · [evals](evals/README.md) · +[changelog](CHANGELOG.md) · [contributing](AGENTS.md) ## Documentation diff --git a/SETUP.md b/SETUP.md new file mode 100644 index 0000000..4a148ed --- /dev/null +++ b/SETUP.md @@ -0,0 +1,43 @@ +# Set up Beyond10x agent plugins + +For an agent in Claude Code or Codex. Follow the steps in order. Ask the user only where a step +says so. Change nothing the user did not confirm. + +1. **Install `b10x`.** + - If `b10x --version` already runs, go to step 2. + - Otherwise pick the target from `uname -s` and `uname -m`: + + | `uname -s` | `uname -m` | target | + |---|---|---| + | `Linux` | `x86_64` | `x86_64-unknown-linux-gnu` | + | `Linux` | `aarch64` or `arm64` | `aarch64-unknown-linux-gnu` | + | `Darwin` | `x86_64` | `x86_64-apple-darwin` | + | `Darwin` | `arm64` | `aarch64-apple-darwin` | + + - **Ask the user first.** Show the target, the download URL + (`https://github.com/beyond10x/agentplugins/releases/latest/download/b10x-.tar.gz`) and + the install path (`~/.local/bin/b10x`), and continue only after they agree. Without agreement, + stop and give them this step to run themselves. + - Download, check and install into a fresh directory under `~/.cache` (not `/tmp`): + + ```bash + base=https://github.com/beyond10x/agentplugins/releases/latest/download + mkdir -p "$HOME/.cache" + work=$(mktemp -d "$HOME/.cache/b10x-setup.XXXXXX") + curl -fsSL -o "$work/b10x-.tar.gz" "$base/b10x-.tar.gz" + curl -fsSL -o "$work/SHA256SUMS" "$base/SHA256SUMS" + (cd "$work" && sha256sum --check --ignore-missing SHA256SUMS) # macOS: shasum -a 256 --check --ignore-missing SHA256SUMS + tar -xzf "$work/b10x-.tar.gz" -C "$work" + mkdir -p "$HOME/.local/bin" && mv "$work/b10x-/b10x" "$HOME/.local/bin/b10x" + rm -r "$work" + ``` + + - Stop if the checksum does not match. + - If `~/.local/bin` is not on `PATH`, tell the user and use `~/.local/bin/b10x` below. +2. **Run the guided setup.** `b10x setup guide` prints the `/b10x:init` skill; follow it from its + step 2. It asks what the user wants to do (plan and deliver work, write specifications, isolated + Git checkouts, integrations) and how to install the CLIs (prebuilt by default, or `cargo`), + lists every change, and applies it only after the user confirms. +3. **Continue with the product.** A plugin loads in the next session. To use it now, run + `b10x skill :init` (for example `b10x skill ess:init`) and follow the printed text; + `b10x skill ` lists the rest. diff --git a/Taskfile.yml b/Taskfile.yml index c16ba0c..7df2984 100644 --- a/Taskfile.yml +++ b/Taskfile.yml @@ -1,5 +1,10 @@ version: '3' +vars: + # Outside the home directory: Claude Code reads CLAUDE.md, .claude/CLAUDE.md and .claude/settings*.json + # in every directory above its working directory, so a sandbox under $HOME inherits the operator's. + TRIALS: '/var/tmp/b10x-trials-{{.USER}}' + tasks: check: desc: Run the complete offline repository gate. @@ -17,6 +22,85 @@ tasks: cmds: - cargo run --quiet --locked --bin agentplugins-check -- evals + trial:sandbox: + desc: "Create a clean trial sandbox for this checkout: task trial:sandbox NAME= (see .agents/skills/improving-by-trial)." + requires: + vars: [NAME] + vars: + T: '{{.TRIALS}}/{{.NAME}}' + cmds: + # Claude Code also reads `.claude/settings*.json` in the directories above its working + # directory; one there enables plugins in every sandbox below it. + - | + d='{{.T}}' + while [ "$d" != / ]; do + d="$(dirname "$d")" + for f in "$d/CLAUDE.md" "$d/.claude/CLAUDE.md" "$d/.claude/settings.json" "$d/.claude/settings.local.json"; do + if [ -e "$f" ]; then + echo "refusing: $f would load into every sandbox below it; set TRIALS elsewhere" + exit 1 + fi + done + done + - rm -rf '{{.T}}' + - mkdir -p -m 700 '{{.TRIALS}}' + - mkdir -p '{{.T}}/home/.local/bin' '{{.T}}/work' '{{.T}}/tools' + - for t in claude codex go gofmt; do p="$(command -v $t)" && ln -sf "$(readlink -f "$p")" '{{.T}}/tools/'$t; done; true + - for t in cargo rustc rustup; do ln -sf "$HOME/.cargo/bin/$t" '{{.T}}/tools/'$t; done + - echo '{}' > '{{.T}}/home/.claude.json' + - nice -n 19 env CARGO_BUILD_JOBS=4 cargo build --quiet --locked -p b10x + - cp "${CARGO_TARGET_DIR:-target}/debug/b10x" '{{.T}}/home/.local/bin/b10x' + - | + { + echo "export HOME='{{.T}}/home'" + echo "export PATH='{{.T}}/home/.local/bin:{{.T}}/home/.cargo/bin:{{.T}}/tools:/usr/bin:/bin'" + echo "export RUSTUP_HOME='$HOME/.rustup' CARGO_HOME='$HOME/.cargo' CARGO_BUILD_JOBS=4" + echo "export B10X_MARKETPLACE='{{.ROOT_DIR}}'" + echo "export GIT_AUTHOR_NAME=trial GIT_AUTHOR_EMAIL=trial@example.invalid GIT_COMMITTER_NAME=trial GIT_COMMITTER_EMAIL=trial@example.invalid" + echo "unset ANTHROPIC_API_KEY TMPDIR TMPPREFIX" + } > '{{.T}}/env' + - echo '{{.T}}' + + trial:run: + desc: "Run one headless trial, check its isolation and measure it: task trial:run TRIAL= (in a fresh sandbox), or task trial:run NAME= PROMPT='…' [DIR=] [SEEDED=true for an upgrade trial] in an existing one." + vars: + NAME: '{{.NAME | default .TRIAL}}' + T: '{{.TRIALS}}/{{.NAME}}' + DIR: '{{.DIR | default ""}}' + VERSION: + sh: sed -n 's/^version = "\(.*\)"$/\1/p' Cargo.toml | head -1 + preconditions: + - sh: test -n '{{.TRIAL}}' || { test -n '{{.NAME}}' && test -n "$PROMPT"; } + msg: "set TRIAL= (a definition under trials/), or NAME= and PROMPT='…'" + cmds: + - task: trial:sandbox + vars: { NAME: '{{.NAME}}' } + if: test -n '{{.TRIAL}}' + - cmd: cargo run --quiet --locked --bin agentplugins-check -- trial-prepare '{{.TRIAL}}' --sandbox '{{.T}}' + if: test -n '{{.TRIAL}}' + - cmd: | + printf '%s' "$PROMPT" > '{{.T}}/prompt.txt' + printf '%s' '{{.DIR}}' > '{{.T}}/workdir' + rm -f '{{.T}}/setup.sh' '{{.T}}/seeded' + if [ '{{.SEEDED}}' = true ]; then : > '{{.T}}/seeded'; fi + if: test -z '{{.TRIAL}}' + - if [ -f '{{.T}}/setup.sh' ]; then (. '{{.T}}/env' && cd '{{.T}}' && . ./setup.sh); fi + - rm -f '{{.T}}/registry-at-start.json'; if [ -f '{{.T}}/home/.claude/plugins/installed_plugins.json' ]; then cp '{{.T}}/home/.claude/plugins/installed_plugins.json' '{{.T}}/registry-at-start.json'; fi + - install -m 600 "$HOME/.claude/.credentials.json" '{{.T}}/home/.claude/.credentials.json' + - defer: rm -f '{{.T}}/home/.claude/.credentials.json' + - | + (. '{{.T}}/env' && dir="$(cat '{{.T}}/workdir')" && mkdir -p "{{.T}}/work/$dir" && cd "{{.T}}/work/$dir" && + timeout 2400 claude -p "$(cat '{{.T}}/prompt.txt')" --permission-mode acceptEdits \ + --allowedTools 'Bash(b10x:*)' 'Bash(ess:*)' 'Bash(aep:*)' 'Bash(worktree:*)' 'Bash(git:*)' 'Bash(go:*)' 'Bash(cargo:*)' \ + 'Bash(mkdir:*)' 'Bash(ls:*)' 'Bash(cat:*)' 'Bash(find:*)' 'Bash(curl:*)' 'Bash(tar:*)' \ + 'Bash(sha256sum:*)' 'Bash(mv:*)' 'Bash(chmod:*)' Read Write Edit Glob Grep Skill Task Agent \ + --strict-mcp-config --output-format stream-json --verbose > '{{.T}}/run.jsonl' 2> '{{.T}}/run.err' + echo "EXIT=$?" >> '{{.T}}/run.err') + - cargo run --quiet --locked --bin agentplugins-check -- trial-isolation '{{.T}}/run.jsonl' --sandbox '{{.T}}' --marketplace '{{.ROOT_DIR}}' --version '{{.VERSION}}' $([ -f '{{.T}}/seeded' ] && echo --seeded) + - cargo run --quiet --locked --bin agentplugins-check -- trial-report '{{.T}}/run.jsonl' --sandbox '{{.T}}' {{if .TRIAL}}--trial '{{.TRIAL}}' --baseline '{{.ROOT_DIR}}/trials/baseline.json'{{end}} + env: + PROMPT: '{{.PROMPT}}' + site-build: desc: Install and build the public Docusaurus documentation (network). cmds: diff --git a/b10x.docs.yaml b/b10x.docs.yaml index b4d8696..5343233 100644 --- a/b10x.docs.yaml +++ b/b10x.docs.yaml @@ -84,4 +84,4 @@ surfaces: order: 30 sidebar: autogenerated root: website - summary: Navigate and create portable plugins, or install focused AEP planning, AEP delivery, and ESS specification guidance from the curated beyond10x marketplace. + summary: Navigate and create portable plugins, or install focused AEP planning, AEP delivery, and ESS specification guidance from the curated b10x marketplace. diff --git a/catalog.json b/catalog.json new file mode 100644 index 0000000..063cf4f --- /dev/null +++ b/catalog.json @@ -0,0 +1,91 @@ +{ + "format": "b10x.catalog/2", + "marketplace": { + "name": "b10x", + "repository": "beyond10x/agentplugins" + }, + "base": ["b10x"], + "products": [ + { + "id": "aep", + "intent": "Plan and deliver work (epics, stories, reviewed implementation waves)", + "summary": "Plan governed work in an artifact store and deliver it in reviewed waves.", + "plugins": ["aep"], + "binaries": [ + { + "name": "aep", + "install": { + "archive": { "repository": "beyond10x/aep" }, + "cargo": { "repository": "beyond10x/aep", "package": "aep-cli" } + } + }, + { + "name": "metaharness", + "optional": true, + "install": { + "archive": { "repository": "beyond10x/metaharness" }, + "cargo": { "repository": "beyond10x/metaharness", "package": "metaharness-cli" } + } + }, + { + "name": "b10x-harness", + "optional": true, + "platforms": ["linux"], + "install": { + "archive": { "repository": "beyond10x/harness" }, + "cargo": { "repository": "beyond10x/harness", "package": "b10x-harness-cli" } + } + } + ] + }, + { + "id": "ess", + "intent": "Write specifications of a system or API (schemas, OpenAPI, conformance)", + "summary": "Write, retrofit, validate and conformance-test Executable System Specifications.", + "plugins": ["ess"], + "binaries": [ + { + "name": "ess", + "install": { + "archive": { "repository": "beyond10x/ess" }, + "cargo": { "repository": "beyond10x/ess", "package": "ess-cli" } + } + } + ] + }, + { + "id": "worktree", + "intent": "Work in isolated Git checkouts that clean up safely", + "summary": "Create, lease, finish and safely clean isolated Git worktrees.", + "plugins": ["worktree"], + "binaries": [ + { + "name": "worktree", + "install": { + "archive": { "repository": "beyond10x/worktree" }, + "cargo": { "repository": "beyond10x/worktree", "package": "b10x-worktree-cli" } + } + } + ] + }, + { + "id": "connectors", + "optional": true, + "intent": "Connect external tools and providers", + "summary": "Set up providers and invoke governed integrations through the connectors CLI.", + "plugins": ["connectors"], + "binaries": [] + } + ], + "retired_plugins": { + "aep-drive": "aep", + "aep-plan": "aep", + "aep-planning": "aep", + "adp": "aep", + "beyond10x": "b10x", + "ess-schema": "ess", + "ess-specify": "ess", + "workspace-hygiene": "worktree" + }, + "retired_marketplaces": ["beyond10x", "ess", "worktree"] +} diff --git a/crates/agentplugins-check/Cargo.toml b/crates/agentplugins-check/Cargo.toml index ad90343..f7acde5 100644 --- a/crates/agentplugins-check/Cargo.toml +++ b/crates/agentplugins-check/Cargo.toml @@ -9,6 +9,7 @@ repository.workspace = true authors.workspace = true [dependencies] +clap.workspace = true serde.workspace = true serde_json.workspace = true serde_yaml.workspace = true diff --git a/crates/agentplugins-check/src/concept.rs b/crates/agentplugins-check/src/concept.rs new file mode 100644 index 0000000..14c5765 --- /dev/null +++ b/crates/agentplugins-check/src/concept.rs @@ -0,0 +1,642 @@ +//! The plugin concept in `website/docs/structure.md`, enforced. Each rule below is one row of that +//! page; a structure decision that is not checked here drifts, which is how two AEP plugins shipped +//! after one had been agreed. +//! +//! - **R2** one plugin per product; plugin name = product id = the CLI it drives. +//! - **R3** a skill is an activity named in `-ing` form, one or two words, never its plugin's name. +//! - **R4** an agent is owned by exactly one skill of its plugin, which lists it under `## Agents`. +//! - **R7** every `:` id written in this repository resolves. +//! - **R8** one README row, one plugin page and one sidebar entry per plugin; the README stays short, +//! besides a generated tree of every skill and agent. +//! +//! R1, R5 and R6 are the marketplace, manifest-version and retired-name checks in `main.rs`. + +use std::collections::{BTreeMap, BTreeSet}; +use std::path::{Path, PathBuf}; + +/// Plugins the catalog knows, by where they live. +pub struct Plugins { + /// Carried in `plugins//`. + pub carried: BTreeSet, + /// Pointed at in another repository. + pub remote: BTreeSet, +} + +impl Plugins { + fn all(&self) -> BTreeSet { + self.carried.union(&self.remote).cloned().collect() + } +} + +/// The two lifecycle skills every plugin carries (R3); every other skill is an activity. +const LIFECYCLE: &[&str] = &["init", "upgrade"]; + +/// CLIs whose versions a plugin must not quote (R5): the newest release is the only one they describe. +const CLIS: &[&str] = &[ + "aep", + "ess", + "worktree", + "metaharness", + "b10x-harness", + "protocol", +]; + +/// Every ` x.y.z` in a line, case-insensitive, with optional backticks and a `v`. +#[must_use] +pub fn quoted_versions(line: &str) -> Vec { + let lower = line.to_ascii_lowercase(); + let bytes = lower.as_bytes(); + let mut found = Vec::new(); + for cli in CLIS { + let mut from = 0; + while let Some(offset) = lower[from..].find(cli) { + let start = from + offset; + from = start + cli.len(); + if start > 0 && (bytes[start - 1].is_ascii_alphanumeric() || bytes[start - 1] == b'-') { + continue; + } + let rest = lower[from..] + .trim_start_matches('`') + .trim_start_matches(' ') + .trim_start_matches('`'); + let rest = rest.strip_prefix('v').unwrap_or(rest); + let digits: String = rest + .chars() + .take_while(|c| c.is_ascii_digit() || *c == '.') + .collect(); + if digits.split('.').filter(|part| !part.is_empty()).count() >= 3 + && lower[from..].starts_with([' ', '`']) + { + found.push(format!("{cli} {digits}")); + } + } + } + found +} + +fn versions(directory: &Path, name: &str, problems: &mut Vec) { + let mut stack = vec![directory.to_path_buf()]; + while let Some(dir) = stack.pop() { + for path in entries(&dir) { + if path.is_dir() { + stack.push(path); + } else if path.extension().and_then(|e| e.to_str()) == Some("md") { + let text = read(&path).unwrap_or_default(); + for (number, line) in text.lines().enumerate() { + for quote in quoted_versions(line) { + problems.push(format!( + "R5 plugins/{name}/{}:{} quotes `{quote}`; skills describe the newest release and name no CLI version", + path.strip_prefix(directory).unwrap_or(&path).display(), + number + 1 + )); + } + } + } + } + } +} + +/// The most lines the README may have before its generated documentation block. +const README_LINES: usize = 30; + +/// Directories never read for references. +const SKIP_DIRS: &[&str] = &[".git", "target", "node_modules", "build", ".docusaurus"]; + +/// Paths whose text records history or recorded output, not current references. +const HISTORY: &[&str] = &[ + "CHANGELOG.md", + "changes/", + ".engineering/", + "crates/", + "catalog.json", +]; + +fn read(path: &Path) -> Result { + std::fs::read_to_string(path).map_err(|error| format!("reading {}: {error}", path.display())) +} + +fn frontmatter_name(text: &str) -> Option { + let body = text.strip_prefix("---\n")?; + let end = body.find("\n---")?; + body[..end] + .lines() + .find_map(|line| line.strip_prefix("name:")) + .map(|name| name.trim().trim_matches('"').to_owned()) +} + +fn entries(directory: &Path) -> Vec { + let mut paths: Vec = std::fs::read_dir(directory) + .map(|read| { + read.filter_map(|entry| entry.ok().map(|entry| entry.path())) + .collect() + }) + .unwrap_or_default(); + paths.sort(); + paths +} + +/// R3: whether a skill name is an activity: one or two hyphen-joined words, the first ending in `ing`. +#[must_use] +pub fn activity(name: &str) -> bool { + let words: Vec<&str> = name.split('-').collect(); + (1..=2).contains(&words.len()) + && words + .iter() + .all(|word| !word.is_empty() && word.bytes().all(|b| b.is_ascii_lowercase())) + && words[0].len() > 4 + && words[0].ends_with("ing") +} + +/// The agent names a skill lists under its `## Agents` heading, one `` - `name` `` bullet each. +#[must_use] +pub fn listed_agents(skill: &str) -> Vec { + let mut agents = Vec::new(); + let mut inside = false; + for line in skill.lines() { + if line.starts_with("## ") { + inside = line.trim() == "## Agents"; + continue; + } + if inside { + if let Some(rest) = line.strip_prefix("- `") { + if let Some((name, _)) = rest.split_once('`') { + agents.push(name.to_owned()); + } + } + } + } + agents +} + +/// Skills and agents of one carried plugin, with R3 and R4 applied. +fn plugin( + root: &Path, + name: &str, + problems: &mut Vec, +) -> (BTreeSet, BTreeSet) { + let directory = root.join("plugins").join(name); + let mut skills = BTreeSet::new(); + let mut owners: BTreeMap> = BTreeMap::new(); + for path in entries(&directory.join("skills")) { + let Some(folder) = path.file_name().and_then(|n| n.to_str()).map(str::to_owned) else { + continue; + }; + let Ok(text) = read(&path.join("SKILL.md")) else { + problems.push(format!("R3 `{name}:{folder}` has no SKILL.md")); + continue; + }; + if !LIFECYCLE.contains(&folder.as_str()) && !activity(&folder) { + problems.push(format!( + "R3 `{name}:{folder}` is not an activity name (one or two words, the first ending in `ing`)" + )); + } + if !LIFECYCLE.contains(&folder.as_str()) && folder == name { + problems.push(format!("R3 `{name}:{folder}` is named after its plugin")); + } + if frontmatter_name(&text).as_deref() != Some(folder.as_str()) { + problems.push(format!( + "R3 `{name}:{folder}` declares another `name:` in its frontmatter" + )); + } + for agent in listed_agents(&text) { + owners.entry(agent).or_default().push(folder.clone()); + } + skills.insert(folder); + } + for lifecycle in LIFECYCLE { + if !skills.contains(*lifecycle) { + problems.push(format!( + "R3 `{name}` has no `{lifecycle}` skill; every plugin carries `init` and `upgrade`" + )); + } + } + versions(&directory, name, problems); + let mut agents = BTreeSet::new(); + for path in entries(&directory.join("agents")) { + if path.extension().and_then(|e| e.to_str()) != Some("md") { + continue; + } + let Some(stem) = path.file_stem().and_then(|n| n.to_str()).map(str::to_owned) else { + continue; + }; + let text = read(&path).unwrap_or_default(); + if frontmatter_name(&text).as_deref() != Some(stem.as_str()) { + problems.push(format!("R4 agent `{name}:{stem}` declares another `name:`")); + } + match owners.get(&stem).map(Vec::as_slice) { + None | Some([]) => problems.push(format!( + "R4 agent `{name}:{stem}` is owned by no skill; list it under `## Agents` in the skill that dispatches it" + )), + Some([_]) => {} + Some(many) => problems.push(format!( + "R4 agent `{name}:{stem}` is listed by {} skills ({}); exactly one owns it", + many.len(), + many.join(", ") + )), + } + agents.insert(stem); + } + for (agent, skills_listing) in &owners { + if !agents.contains(agent) { + problems.push(format!( + "R4 `{name}:{}` lists agent `{agent}`, which does not exist", + skills_listing.join("`, `") + )); + } + } + (skills, agents) +} + +/// R2 over `catalog.json`. +fn products(catalog: &serde_json::Value, problems: &mut Vec) { + for product in catalog + .get("products") + .and_then(serde_json::Value::as_array) + .into_iter() + .flatten() + { + let id = product + .get("id") + .and_then(serde_json::Value::as_str) + .unwrap_or_default(); + let plugins: Vec<&str> = product + .get("plugins") + .and_then(serde_json::Value::as_array) + .into_iter() + .flatten() + .filter_map(serde_json::Value::as_str) + .collect(); + if plugins != [id] { + problems.push(format!( + "R2 product `{id}` must be exactly one plugin named `{id}`, not {plugins:?}" + )); + } + let first_binary = product + .get("binaries") + .and_then(serde_json::Value::as_array) + .and_then(|binaries| binaries.first()) + .and_then(|binary| binary.get("name")) + .and_then(serde_json::Value::as_str); + if let Some(binary) = first_binary { + if binary != id { + problems.push(format!( + "R2 product `{id}` drives `{binary}` first; its CLI must carry the product's name" + )); + } + } + } +} + +fn authored(root: &Path, relative: &str) -> bool { + let historical = HISTORY + .iter() + .any(|history| relative == *history || relative.starts_with(history)); + let recorded = relative.starts_with("evals/") && relative.contains("/recorded/"); + let text = [".md", ".yaml", ".yml", ".json", ".ts", ".tsx", ".svg"] + .iter() + .any(|extension| relative.ends_with(extension)); + text && !historical && !recorded && root.join(relative).is_file() +} + +fn walk(root: &Path) -> Vec { + let mut files = Vec::new(); + let mut stack = vec![root.to_path_buf()]; + while let Some(directory) = stack.pop() { + for path in entries(&directory) { + let name = path + .file_name() + .and_then(|n| n.to_str()) + .unwrap_or_default(); + if path.is_dir() { + if !SKIP_DIRS.contains(&name) { + stack.push(path); + } + } else if let Ok(relative) = path.strip_prefix(root) { + files.push(relative.to_string_lossy().replace('\\', "/")); + } + } + } + files.sort(); + files +} + +/// Every `:` id in a line, for the given plugin names. +#[must_use] +pub fn ids(line: &str, plugins: &BTreeSet) -> Vec<(String, String)> { + let bytes = line.as_bytes(); + let mut found = Vec::new(); + for plugin in plugins { + let mut from = 0; + while let Some(offset) = line[from..].find(&format!("{plugin}:")) { + let start = from + offset; + let after = start + plugin.len() + 1; + from = after; + let before_ok = start == 0 || { + let b = bytes[start - 1]; + !(b.is_ascii_alphanumeric() + || b == b'-' + || b == b'_' + || b == b'/' + || b == b'.' + || b == b'@') + }; + if !before_ok { + continue; + } + let name: String = line[after..] + .chars() + .take_while(|c| c.is_ascii_lowercase() || c.is_ascii_digit() || *c == '-') + .collect(); + let name = name.trim_end_matches('-').to_owned(); + if name.is_empty() || !name.starts_with(|c: char| c.is_ascii_lowercase()) { + continue; + } + found.push((plugin.clone(), name)); + } + } + found +} + +/// R7 over carried plugins; ids of remote plugins are checked by `agentplugins-check remote`. +fn references( + root: &Path, + plugins: &Plugins, + known: &BTreeMap>, + problems: &mut Vec, +) { + let names = plugins.all(); + for relative in walk(root) { + if !authored(root, &relative) { + continue; + } + let Ok(text) = read(&root.join(&relative)) else { + continue; + }; + for (number, line) in text.lines().enumerate() { + for (plugin, name) in ids(line, &names) { + let Some(members) = known.get(&plugin) else { + continue; + }; + if !members.contains(&name) { + problems.push(format!( + "R7 {relative}:{} names `{plugin}:{name}`, which is no skill or agent of `{plugin}`", + number + 1 + )); + } + } + } + } +} + +/// README rows: the plugin names in the first column of rows that start with ``| [`name`]``. +#[must_use] +pub fn readme_rows(readme: &str) -> Vec { + readme + .lines() + .filter_map(|line| line.strip_prefix("| [`")) + .filter_map(|rest| rest.split_once('`').map(|(name, _)| name.to_owned())) + .collect() +} + +/// Markers around the README's plugin tree, which [`tree`] generates. +const TREE_START: &str = ""; +const TREE_END: &str = ""; + +/// One carried plugin's skills and agents, by name. +pub struct Contents { + /// Skill folder names. + pub skills: BTreeSet, + /// Agent file stems. + pub agents: BTreeSet, +} + +/// The README's plugin tree, in table order: every skill (lifecycle first) and agent, each linked +/// to its source file. +#[must_use] +pub fn tree(order: &[String], contents: &BTreeMap) -> String { + let mut lines = vec![TREE_START.to_owned()]; + for name in order { + let Some(plugin) = contents.get(name) else { + continue; + }; + lines.push(format!( + "- [`{name}`](plugins/{name}/) · [docs](website/docs/plugins/{name}.md)" + )); + let lifecycle = LIFECYCLE.iter().map(|s| (*s).to_owned()); + let rest = plugin + .skills + .iter() + .filter(|s| !LIFECYCLE.contains(&s.as_str())) + .cloned(); + let skills: Vec = lifecycle + .chain(rest) + .filter(|s| plugin.skills.contains(s)) + .map(|s| format!("[`{s}`](plugins/{name}/skills/{s}/SKILL.md)")) + .collect(); + lines.push(format!(" - skills: {}", skills.join(" · "))); + if !plugin.agents.is_empty() { + let agents: Vec = plugin + .agents + .iter() + .map(|a| format!("[`{a}`](plugins/{name}/agents/{a}.md)")) + .collect(); + lines.push(format!(" - agents: {}", agents.join(" · "))); + } + } + lines.push(TREE_END.to_owned()); + lines.join("\n") +} + +/// R8. +fn docs( + root: &Path, + plugins: &Plugins, + contents: &BTreeMap, + problems: &mut Vec, +) -> Result<(), String> { + let all = plugins.all(); + let readme = read(&root.join("README.md"))?; + let head = readme + .split("") + .next() + .unwrap_or_default(); + let written = match (head.find(TREE_START), head.find(TREE_END)) { + (Some(start), Some(end)) if start < end => &head[start..end + TREE_END.len()], + _ => "", + }; + let prose = head.lines().count() - written.lines().count(); + if prose > README_LINES { + problems.push(format!( + "R8 README.md has {prose} lines before its documentation block, besides the plugin tree; the most is {README_LINES}: one paragraph, the plugin table, one link line", + )); + } + let rows = readme_rows(head); + let expected = tree(&rows, contents); + if written != expected { + problems.push(format!( + "R8 README.md plugin tree is missing or stale; it must read:\n{expected}" + )); + } + let rows: BTreeSet = rows.into_iter().collect(); + if rows != all { + problems.push(format!( + "R8 README.md table lists {rows:?}; the catalog has {all:?}" + )); + } + let pages: BTreeSet = entries(&root.join("website/docs/plugins")) + .iter() + .filter(|path| path.extension().and_then(|e| e.to_str()) == Some("md")) + .filter_map(|path| path.file_stem().and_then(|n| n.to_str()).map(str::to_owned)) + .collect(); + if pages != all { + problems.push(format!( + "R8 website/docs/plugins has pages {pages:?}; the catalog has {all:?}" + )); + } + let sidebar = read(&root.join("website/sidebars.ts"))?; + let listed: BTreeSet = sidebar + .split("'plugins/") + .skip(1) + .filter_map(|rest| rest.split_once('\'').map(|(name, _)| name.to_owned())) + .collect(); + if listed != all { + problems.push(format!( + "R8 website/sidebars.ts lists {listed:?}; the catalog has {all:?}" + )); + } + Ok(()) +} + +/// Check the concept. `plugins` comes from the catalog and marketplace checks that ran first. +pub fn check(root: &Path, plugins: &Plugins) -> Result<(), String> { + let mut problems = Vec::new(); + let catalog: serde_json::Value = serde_json::from_str(&read(&root.join("catalog.json"))?) + .map_err(|error| format!("parsing catalog.json: {error}"))?; + products(&catalog, &mut problems); + let mut known = BTreeMap::new(); + let mut contents = BTreeMap::new(); + for name in &plugins.carried { + let (skills, agents) = plugin(root, name, &mut problems); + known.insert( + name.clone(), + skills.union(&agents).cloned().collect::>(), + ); + contents.insert(name.clone(), Contents { skills, agents }); + } + references(root, plugins, &known, &mut problems); + docs(root, plugins, &contents, &mut problems)?; + if problems.is_empty() { + Ok(()) + } else { + Err(format!( + "{} concept violation(s) (website/docs/structure.md):\n {}", + problems.len(), + problems.join("\n ") + )) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn activity_names() { + for good in [ + "planning", + "migrating", + "implementing", + "authoring-plugins", + "testing-conformance", + ] { + assert!(activity(good), "{good}"); + } + for bad in [ + "wave", + "drive", + "setup", + "guide", + "plugin-creator", + "ess", + "sing", + "planning-a-b", + "Planning", + ] { + assert!(!activity(bad), "{bad}"); + } + } + + #[test] + fn cli_versions_are_found_and_skill_versions_are_not() { + assert_eq!(quoted_versions("at AEP 0.55.0 the verb"), ["aep 0.55.0"]); + assert_eq!(quoted_versions("output of ESS `0.29.0`"), ["ess 0.29.0"]); + assert!(quoted_versions("**Skill version 0.13.1** — the version").is_empty()); + assert!(quoted_versions("ess-cli 0.30.0 is not a quote of a CLI name").is_empty()); + assert!(quoted_versions("aep plan artifact list").is_empty()); + } + + #[test] + fn agents_are_read_from_the_agents_section_only() { + let skill = "# X\n\nUses `implementor` in prose.\n\n## Agents\n\n- `story-scoper` — scopes\n- `adversary` — attacks\n\n## Next\n\n- `other` — not an agent list\n"; + assert_eq!(listed_agents(skill), ["story-scoper", "adversary"]); + } + + #[test] + fn ids_are_found_at_word_starts_only() { + let plugins: BTreeSet = ["aep", "b10x"].iter().map(|s| (*s).to_owned()).collect(); + assert_eq!( + ids( + "use `aep:planning` and b10x:init; see beyond10x/aep:x, https://x/aep:y, aep: 1", + &plugins + ), + [ + ("aep".to_owned(), "planning".to_owned()), + ("b10x".to_owned(), "init".to_owned()) + ] + ); + } + + #[test] + fn the_tree_lists_lifecycle_skills_first_and_links_every_file() { + let mut contents = BTreeMap::new(); + contents.insert( + "ess".to_owned(), + Contents { + skills: ["specifying", "upgrade", "init"].map(str::to_owned).into(), + agents: ["author"].map(str::to_owned).into(), + }, + ); + contents.insert( + "b10x".to_owned(), + Contents { + skills: ["init", "upgrade"].map(str::to_owned).into(), + agents: BTreeSet::new(), + }, + ); + let order = ["ess".to_owned(), "b10x".to_owned()]; + assert_eq!( + tree(&order, &contents), + "\n\ + - [`ess`](plugins/ess/) · [docs](website/docs/plugins/ess.md)\n \ + - skills: [`init`](plugins/ess/skills/init/SKILL.md) · [`upgrade`](plugins/ess/skills/upgrade/SKILL.md) · [`specifying`](plugins/ess/skills/specifying/SKILL.md)\n \ + - agents: [`author`](plugins/ess/agents/author.md)\n\ + - [`b10x`](plugins/b10x/) · [docs](website/docs/plugins/b10x.md)\n \ + - skills: [`init`](plugins/b10x/skills/init/SKILL.md) · [`upgrade`](plugins/b10x/skills/upgrade/SKILL.md)\n\ + " + ); + } + + #[test] + fn readme_rows_read_the_first_column() { + let readme = "| plugin | for |\n|---|---|\n| [`ess`](x) | a |\n| [`aep`](y) | b |\n"; + assert_eq!(readme_rows(readme), ["ess", "aep"]); + } + + #[test] + fn a_product_with_two_plugins_is_refused() { + let catalog = serde_json::json!({"products": [{"id": "aep", "plugins": [concat!("aep-", "plan"), concat!("aep-", "drive")], "binaries": [{"name": "aep"}]}]}); + let mut problems = Vec::new(); + products(&catalog, &mut problems); + assert_eq!(problems.len(), 1, "{problems:?}"); + } +} diff --git a/crates/agentplugins-check/src/evals.rs b/crates/agentplugins-check/src/evals.rs index 7fab2b0..abd2866 100644 --- a/crates/agentplugins-check/src/evals.rs +++ b/crates/agentplugins-check/src/evals.rs @@ -936,39 +936,44 @@ fn skill_directory(root: &Path, plugin: &str, name: &str, case: &str) -> Result< )) } -/// The corpus subcommands, or [`None`] when these arguments are not one. +/// The corpus subcommands. +#[derive(Debug, clap::Subcommand)] +pub(crate) enum Action { + /// Check one live run: its command contract and the AEP report beside the stream. + CheckStream { + /// The case directory, relative to the repository root. + case: std::path::PathBuf, + /// The `.events.jsonl` stream the run wrote. + stream: std::path::PathBuf, + }, + /// Print the cases (and their plugins) that the changed paths concern. + Scope { + /// Changed repository-relative paths. + changed: Vec, + }, +} + +/// Run a corpus subcommand; none validates the corpus and replays every recorded transcript. /// -/// Handled before the marketplace check's own argument match and never inside it, so that adding -/// this surface moved three lines of `main` and touched no existing arm. The `Ok` arm there prints -/// *marketplace beyond10x* and would be a false sentence under either verb below. -pub(crate) fn cli(root: &Path, arguments: &[String]) -> Option { - let verdict = match arguments - .iter() - .map(String::as_str) - .collect::>() - .as_slice() - { - ["evals"] => evals(root), - ["evals", "check-stream", case, stream] => { - check_stream(root, Path::new(case), Path::new(stream)) - } - ["evals", "scope", changed @ ..] => { - let changed: Vec = changed.iter().map(|path| (*path).to_owned()).collect(); - scope(root, &changed).map(|matched| { - for (case, plugin) in matched { - println!("{case}\t{plugin}"); - } - }) - } - _ => return None, +/// Kept apart from the marketplace check, whose success line prints *marketplace b10x* and would be +/// a false sentence under either verb here. +pub(crate) fn run(root: &Path, action: Option) -> std::process::ExitCode { + let verdict = match action { + None => evals(root), + Some(Action::CheckStream { case, stream }) => check_stream(root, &case, &stream), + Some(Action::Scope { changed }) => scope(root, &changed).map(|matched| { + for (case, plugin) in matched { + println!("{case}\t{plugin}"); + } + }), }; - Some(match verdict { + match verdict { Ok(()) => std::process::ExitCode::SUCCESS, Err(error) => { eprintln!("error: {error}"); std::process::ExitCode::from(1) } - }) + } } fn check_commands(case: &Case, stream: &Path) -> Result<(), String> { @@ -1066,20 +1071,20 @@ mod tests { // manifest still pins the pre-0.6.2 spellings, because it is the record of a run that // happened under them — so that replay takes the branch the third test covers, and this one // is about the rule rather than about today's corpus. - let case = case_about(&["aep-plan", "aep-drive", "ess-specify"]); + let case = case_about(&["aep", "b10x", "connectors"]); let pins = vec![ - "beyond10x/agentplugins@aep-drive@0.6.1".to_owned(), - "beyond10x/agentplugins@ess-specify@0.6.1".to_owned(), + "beyond10x/agentplugins@b10x@0.6.1".to_owned(), + "beyond10x/agentplugins@connectors@0.6.1".to_owned(), ]; assert_eq!( treatment_args(&case, pins), vec![ "--plugin-dir", - "plugins/aep-plan", + "plugins/aep", "--plugin", - "beyond10x/agentplugins@aep-drive@0.6.1", + "beyond10x/agentplugins@b10x@0.6.1", "--plugin", - "beyond10x/agentplugins@ess-specify@0.6.1", + "beyond10x/agentplugins@connectors@0.6.1", ] ); } @@ -1088,27 +1093,24 @@ mod tests { fn a_case_whose_manifest_pins_nothing_repeats_nothing() { // A single-plugin stream is unambiguous and `aep` reads the treatment out of it, so a // replay that added arguments would be inventing the experiment. - assert!(treatment_args(&case_about(&["aep-drive"]), Vec::new()).is_empty()); + assert!(treatment_args(&case_about(&["b10x"]), Vec::new()).is_empty()); } #[test] fn an_ambiguous_remainder_sends_the_pins_rather_than_failing_the_gate() { // `--plugin-dir` reaches only the spawn a replay never performs. Refusing here would stop // a replay that would have succeeded, over an argument `aep` discards. - let two_left = case_about(&["aep-plan", "aep-drive", "ess-specify", "extra"]); - let pins = vec!["beyond10x/agentplugins@aep-drive@0.6.1".to_owned()]; + let two_left = case_about(&["aep", "b10x", "connectors", "extra"]); + let pins = vec!["beyond10x/agentplugins@b10x@0.6.1".to_owned()]; let args = treatment_args(&two_left, pins.clone()); assert!(!args.iter().any(|a| a == "--plugin-dir"), "{args:?}"); - assert_eq!( - args, - vec!["--plugin", "beyond10x/agentplugins@aep-drive@0.6.1"] - ); + assert_eq!(args, vec!["--plugin", "beyond10x/agentplugins@b10x@0.6.1"]); // And when every subject plugin is pinned, the remainder is empty rather than wrong. - let all_pinned = case_about(&["aep-drive"]); + let all_pinned = case_about(&["b10x"]); assert_eq!( treatment_args(&all_pinned, pins), - vec!["--plugin", "beyond10x/agentplugins@aep-drive@0.6.1"] + vec!["--plugin", "beyond10x/agentplugins@b10x@0.6.1"] ); } @@ -1287,9 +1289,9 @@ mod tests { #[test] fn a_skill_is_resolved_by_the_name_its_document_declares() { let root = root(); - resolve_skill(&root, "ess-specify:specify", "t") - .expect("the ESS skill declares `name: specify` under `skills/specify/`"); - let error = resolve_skill(&root, "ess-specify:schema-validation", "t") + resolve_skill(&root, "connectors:integrating", "t") + .expect("the connectors skill declares `name: connectors` under `skills/connectors/`"); + let error = resolve_skill(&root, "connectors:schema-validation", "t") .expect_err("a name no SKILL.md declares is not what a harness lists"); assert!(error.contains("declares"), "{error}"); } @@ -1298,7 +1300,7 @@ mod tests { /// go on passing its own document while judging nothing. #[test] fn a_case_naming_a_missing_agent_is_refused() { - let error = resolve_agent(&root(), "aep-plan:plan-critic-security", "t") + let error = resolve_agent(&root(), "aep:plan-critic-security", "t") .expect_err("an agent this repository does not ship must be refused"); assert!(error.contains("plan-critic-security"), "{error}"); } @@ -1332,20 +1334,20 @@ mod tests { let root = root(); let one = scope( &root, - &["plugins/aep-plan/agents/plan-critic-scope.md".to_owned()], + &["plugins/aep/agents/plan-critic-scope.md".to_owned()], ) .expect("the corpus scopes"); assert_eq!( one, vec![( "evals/plan-critic-scope-verdict".to_owned(), - "plugins/aep-plan".to_owned() + "plugins/aep".to_owned() )] ); let rubric = scope( &root, - &["plugins/aep-plan/skills/planning/references/critic-rubric.md".to_owned()], + &["plugins/aep/skills/planning/references/critic-rubric.md".to_owned()], ) .expect("the corpus scopes"); // The four critic cases name the rubric directly, and the golden path names the planning @@ -1371,7 +1373,7 @@ mod tests { fn a_reference_beside_a_skill_is_in_that_skills_scope() { let matched = scope( &root(), - &["plugins/aep-drive/skills/wave/references/unit-brief.md".to_owned()], + &["plugins/aep/skills/implementing/references/unit-brief.md".to_owned()], ) .expect("the corpus scopes"); // The adversary's case, and the golden path — whose step 6 is the wave. Both are right. @@ -1380,11 +1382,11 @@ mod tests { vec![ ( "evals/adversary-tests-only".to_owned(), - "plugins/aep-drive".to_owned() + "plugins/aep".to_owned() ), ( "evals/golden-path-end-to-end".to_owned(), - "plugins/aep-plan".to_owned() + "plugins/aep".to_owned() ), ] ); diff --git a/crates/agentplugins-check/src/main.rs b/crates/agentplugins-check/src/main.rs index 99a7f7c..8e6d478 100644 --- a/crates/agentplugins-check/src/main.rs +++ b/crates/agentplugins-check/src/main.rs @@ -3,25 +3,40 @@ use std::path::{Path, PathBuf}; use std::process::{Command, ExitCode}; +use clap::{Parser, Subcommand}; + +mod concept; mod evals; mod readiness; +mod report; +mod tools; +mod trial; +mod trials; + +/// The marketplace identity in every marketplace format. +const MARKETPLACE: &str = "b10x"; +/// Every plugin, in marketplace order, and the files each must carry. Each has the lifecycle +/// skills `init` and `upgrade` (website/docs/structure.md R3), which `concept` checks for all. const PLUGINS: &[(&str, &[&str])] = &[ ( - "beyond10x", + "b10x", &[ - "skills/beyond10x/SKILL.md", - "skills/beyond10x/references/resources.md", - "skills/plugin-creator/SKILL.md", - "skills/plugin-creator/references/compatibility.md", + "skills/init/SKILL.md", + "skills/upgrade/SKILL.md", + "skills/routing/SKILL.md", + "skills/routing/references/resources.md", + "hooks/hooks.json", + "skills/authoring-plugins/SKILL.md", + "skills/authoring-plugins/references/compatibility.md", ], ), ( - "aep-plan", + "aep", &[ "skills/planning/SKILL.md", "skills/planning/references/critic-rubric.md", - "skills/story-migration/SKILL.md", + "skills/migrating/SKILL.md", "agents/decomposer.md", "agents/plan-reviewer.md", "agents/reverse-engineer.md", @@ -29,21 +44,33 @@ const PLUGINS: &[(&str, &[&str])] = &[ "agents/plan-critic-design.md", "agents/plan-critic-scope.md", "agents/plan-critic-parallel-safety.md", + "skills/implementing/SKILL.md", + "skills/implementing/references/drive.md", + "agents/story-scoper.md", + "agents/implementor.md", + "agents/adversary.md", + "agents/security-reviewer.md", ], ), + ("connectors", &["skills/integrating/SKILL.md"]), + ("worktree", &["skills/managing-worktrees/SKILL.md"]), ( - "aep-drive", + "ess", &[ - "skills/wave/SKILL.md", - "skills/drive/SKILL.md", - "agents/story-scoper.md", - "agents/implementor.md", - "agents/adversary.md", + "skills/specifying/SKILL.md", + "skills/specifying/references/syntax.md", + "skills/retrofitting/SKILL.md", + "skills/testing-conformance/SKILL.md", + "skills/hardening/SKILL.md", + "skills/hardening/references/techniques.md", + "skills/hardening/references/reference-model.md", + "skills/hardening/references/design-review.md", + "skills/hardening/references/spec-diff.md", + "agents/author.md", + "agents/retrofitter.md", + "agents/conformance.md", ], ), - ("ess-specify", &["skills/specify/SKILL.md"]), - ("workspace-hygiene", &["skills/worktree/SKILL.md"]), - ("connectors", &["skills/connectors/SKILL.md"]), ]; fn json(path: &Path) -> Result { @@ -52,25 +79,90 @@ fn json(path: &Path) -> Result { serde_json::from_str(&text).map_err(|error| format!("parsing {}: {error}", path.display())) } +/// `catalog.json`, which `b10x setup` plans from, names exactly the marketplace's plugins, pins no +/// version, and maps every retired plugin name to one the marketplace lists. +fn catalog(root: &Path) -> Result<(), String> { + let document = json(&root.join("catalog.json"))?; + if document.get("format").and_then(serde_json::Value::as_str) != Some("b10x.catalog/2") { + return Err("catalog.json is not `b10x.catalog/2`".to_owned()); + } + let strings = |value: Option<&serde_json::Value>| -> Vec { + value + .and_then(serde_json::Value::as_array) + .map(|items| { + items + .iter() + .filter_map(|item| item.as_str().map(str::to_owned)) + .collect() + }) + .unwrap_or_default() + }; + let mut named = strings(document.get("base")); + for product in document + .get("products") + .and_then(serde_json::Value::as_array) + .ok_or("catalog.json has no products")? + { + named.extend(strings(product.get("plugins"))); + } + named.sort(); + let mut listed: Vec = PLUGINS.iter().map(|(name, _)| (*name).to_owned()).collect(); + listed.sort(); + if named != listed { + return Err(format!( + "catalog.json names {named:?}; the marketplaces list {listed:?}" + )); + } + if let Some(retired) = document + .get("retired_plugins") + .and_then(serde_json::Value::as_object) + { + for (old, new) in retired { + if !new + .as_str() + .is_some_and(|new| listed.iter().any(|name| name == new)) + { + return Err(format!( + "catalog.json retires `{old}` to {new}, which no marketplace lists" + )); + } + } + } + let text = std::fs::read_to_string(root.join("catalog.json")).map_err(|e| e.to_string())?; + let pinned = text + .as_bytes() + .windows(3) + .any(|w| w[0].is_ascii_digit() && w[1] == b'.' && w[2].is_ascii_digit()); + if pinned { + return Err( + "catalog.json names a version; it must resolve versions at run time".to_owned(), + ); + } + Ok(()) +} + +/// Both marketplace formats list exactly [`PLUGINS`], in order, each from its local directory +/// (website/docs/structure.md R1: every plugin lives here). fn marketplace(root: &Path, relative: &str) -> Result<(), String> { let document = json(&root.join(relative))?; - if document.get("name").and_then(serde_json::Value::as_str) != Some("beyond10x") { + if document.get("name").and_then(serde_json::Value::as_str) != Some(MARKETPLACE) { return Err(format!( - "{relative} does not declare marketplace `beyond10x`" + "{relative} does not declare marketplace `{MARKETPLACE}`" )); } let entries = document .get("plugins") .and_then(serde_json::Value::as_array) .ok_or_else(|| format!("{relative} has no plugins array"))?; - if entries.len() != PLUGINS.len() { + let expected = PLUGINS.iter().map(|(name, _)| *name).collect::>(); + if entries.len() != expected.len() { return Err(format!( - "{relative} contains {} plugins; expected {} focused plugins", + "{relative} contains {} plugins; expected {}", entries.len(), - PLUGINS.len() + expected.len() )); } - for (index, (plugin, _)) in PLUGINS.iter().enumerate() { + for (index, plugin) in expected.iter().enumerate() { let actual = entries[index] .get("name") .and_then(serde_json::Value::as_str); @@ -79,6 +171,18 @@ fn marketplace(root: &Path, relative: &str) -> Result<(), String> { "{relative} plugin {index} is {actual:?}; expected `{plugin}`" )); } + let local = format!("./plugins/{plugin}"); + let source = &entries[index]["source"]; + let path = source.as_str().or_else(|| { + (source.get("source").and_then(serde_json::Value::as_str) == Some("local")) + .then(|| source.get("path").and_then(serde_json::Value::as_str)) + .flatten() + }); + if path != Some(local.as_str()) { + return Err(format!( + "{relative} plugin `{plugin}` must come from `{local}`; every plugin lives in this repository" + )); + } let count = entries .iter() .filter(|entry| entry.get("name").and_then(serde_json::Value::as_str) == Some(plugin)) @@ -137,7 +241,7 @@ fn frontmatter(text: &str) -> Option<&str> { /// and a YAML dependency for two `key:` prefixes would be a parser to keep in step with whichever /// one each harness uses. fn critic_pins(root: &Path) -> Result<(), String> { - let directory = root.join("plugins/aep-plan/agents"); + let directory = root.join("plugins/aep/agents"); let mut entries = std::fs::read_dir(&directory) .map_err(|error| format!("reading {}: {error}", directory.display()))? .map(|entry| entry.map(|entry| entry.path())) @@ -217,19 +321,134 @@ struct Retired { const RETIRED: &[Retired] = &[ Retired { old: "aep-planning", - new: "aep-plan", + new: "aep@b10x", wire_next: &[], }, Retired { old: "ess-schema", - new: "ess-specify", + new: "ess@b10x", wire_next: &[], }, Retired { old: "adp", - new: "aep-drive", + new: "aep@b10x", wire_next: &['/'], }, + // The two AEP plugins are one: `aep`. + Retired { + old: "aep-plan", + new: "aep@b10x", + wire_next: &[], + }, + Retired { + old: "aep-drive", + new: "aep@b10x", + wire_next: &[], + }, + // ESS and worktree plugins live here under `b10x` (website/docs/structure.md R1). + Retired { + old: "ess-specify", + new: "ess@b10x", + wire_next: &[], + }, + Retired { + old: "workspace-hygiene", + new: "worktree@b10x", + wire_next: &[], + }, + // The front door is `b10x`; its router skill is `guide`. + Retired { + old: "beyond10x@b10x", + new: "b10x@b10x", + wire_next: &[], + }, + Retired { + old: "beyond10x:beyond10x", + new: "b10x:routing", + wire_next: &[], + }, + Retired { + old: "beyond10x:plugin-creator", + new: "b10x:authoring-plugins", + wire_next: &[], + }, + // Skills are activities (website/docs/structure.md R3). + Retired { + old: "ess:specify", + new: "ess:specifying", + wire_next: &[], + }, + Retired { + old: "ess:retrofit", + new: "ess:retrofitting", + wire_next: &[], + }, + Retired { + old: "ess:coverage", + new: "ess:testing-conformance", + wire_next: &[], + }, + Retired { + old: "ess:ess", + new: "ess:init", + wire_next: &[], + }, + Retired { + old: "worktree:worktree", + new: "worktree:managing-worktrees", + wire_next: &[], + }, + Retired { + old: "b10x:installing", + new: "b10x:init", + wire_next: &[], + }, + Retired { + old: "aep:wave", + new: "aep:implementing", + wire_next: &[], + }, + Retired { + old: "aep:drive", + new: "aep:implementing", + wire_next: &[], + }, + Retired { + old: "aep:story-migration", + new: "aep:migrating", + wire_next: &[], + }, + Retired { + old: "b10x:setup", + new: "b10x:init", + wire_next: &[], + }, + Retired { + old: "b10x:guide", + new: "b10x:routing", + wire_next: &[], + }, + Retired { + old: "b10x:plugin-creator", + new: "b10x:authoring-plugins", + wire_next: &[], + }, + Retired { + old: "connectors:connectors", + new: "connectors:integrating", + wire_next: &[], + }, + // The `ess` binary no longer prints skills; `b10x skill` does. + Retired { + old: "ess skill", + new: "b10x skill ess:", + wire_next: &[], + }, + Retired { + old: "beyond10x/ess` marketplace", + new: "the b10x marketplace", + wire_next: &[], + }, ]; /// Directories never walked looking for a retired name, wherever they sit. @@ -248,11 +467,18 @@ const RETIRED_SKIP_DIRS: &[&str] = &[".git", "target", "node_modules"]; /// loophole for a real reference: every plugin name in this file is a name [`plugin`] and /// [`marketplace`] resolve against the tree, or a path [`critic_pins`] opens, so a stale one here /// fails a check rather than escaping one. +/// +/// `catalog.json` and `crates/b10x/` are the migration: `b10x setup` finds a retired install by its +/// old name and replaces it, so they must spell every old name, and the fixtures under +/// `crates/b10x/tests/` are recorded host output from installs made under those names. +/// [`catalog`] checks that every name the catalog retires maps to one a marketplace lists. const RETIRED_ALLOWED: &[&str] = &[ "CHANGELOG.md", "changes/", ".engineering/", "crates/agentplugins-check/src/main.rs", + "catalog.json", + "crates/b10x/", ]; /// Whether a byte can be part of the same word as a name, so `handpicked` does not read as `adp`. @@ -416,7 +642,9 @@ fn retired_names(root: &Path) -> Result<(), String> { } continue; } - if allowed || recorded { + // In a linked worktree `.git` is a file whose `gitdir:` line names the checkout's + // location, which nothing in this repository authors. + if allowed || recorded || RETIRED_SKIP_DIRS.contains(&name) { continue; } let Ok(bytes) = std::fs::read(&path) else { @@ -690,15 +918,106 @@ fn flat_spellings(root: &Path) -> Result<(), String> { )) } +/// The `b10x` verbs that write a plan for one host set; `--host` defaults to every host. +const PLAN_VERBS: &[&str] = &["b10x init", "b10x upgrade"]; + +/// The lines of one document that spell a plan command without naming its host. +/// +/// A command is a line of a fenced block or one inline-code span; prose that merely names the verb +/// is not one, so a span has to start with the verb. +fn hostless_plans(text: &str) -> Vec { + let mut fenced = false; + let mut found = Vec::new(); + for (index, line) in text.lines().enumerate() { + if line.trim_start().starts_with("```") { + fenced = !fenced; + continue; + } + let spans: Vec<&str> = if fenced { + vec![line] + } else { + line.split('`').skip(1).step_by(2).collect() + }; + let hostless = spans.iter().any(|span| { + let span = span.trim(); + PLAN_VERBS.iter().any(|verb| { + span.strip_prefix(verb) + .is_some_and(|rest| rest.is_empty() || rest.starts_with(' ')) + }) && !span.contains("--host") + }); + if hostless { + found.push(index + 1); + } + } + found +} + +/// Every plan command a plugin prints names its host. +/// +/// `b10x init` and `b10x upgrade` plan for every host when `--host` is absent, so a skill that +/// prints one without it plans Codex from Claude Code and the reverse. `b10x:init` step 4 says how to +/// pick the host; every other skill that prints a plan command has to say it too, and a list of +/// those skills is the hand-maintained thing that misses the next one. +fn plan_hosts(root: &Path) -> Result<(), String> { + let mut found = Vec::new(); + let mut stack = vec![root.join("plugins")]; + while let Some(directory) = stack.pop() { + let mut entries = std::fs::read_dir(&directory) + .map_err(|error| format!("reading {}: {error}", directory.display()))? + .collect::, _>>() + .map_err(|error| format!("reading {}: {error}", directory.display()))? + .into_iter() + .map(|entry| entry.path()) + .collect::>(); + entries.sort(); + for path in entries { + if path.is_dir() { + stack.push(path); + continue; + } + if path.extension().and_then(std::ffi::OsStr::to_str) != Some("md") { + continue; + } + let text = std::fs::read_to_string(&path) + .map_err(|error| format!("reading {}: {error}", path.display()))?; + let relative = path.strip_prefix(root).unwrap_or(&path).to_string_lossy(); + for line in hostless_plans(&text) { + found.push(format!(" {relative}:{line}")); + } + } + } + if found.is_empty() { + return Ok(()); + } + found.sort(); + Err(format!( + "{} plan command(s) in plugins name no `--host`, so they plan for every host; say which, \ + as `b10x:init` step 4 does (`--host claude` in Claude Code, `--host codex` in Codex):\n{}", + found.len(), + found.join("\n") + )) +} + fn check(root: &Path) -> Result<(), String> { marketplace(root, ".agents/plugins/marketplace.json")?; marketplace(root, ".claude-plugin/marketplace.json")?; + catalog(root)?; + tools::verified(root)?; for (name, required) in PLUGINS { plugin(root, name, required)?; } critic_pins(root)?; retired_names(root)?; flat_spellings(root)?; + plan_hosts(root)?; + concept::check( + root, + &concept::Plugins { + carried: PLUGINS.iter().map(|(name, _)| (*name).to_owned()).collect(), + remote: std::collections::BTreeSet::new(), + }, + )?; + trials::check(root)?; evals::evals(root) } @@ -729,31 +1048,206 @@ fn verify_release(root: &Path, version: &str) -> Result<(), String> { Ok(()) } +/// Validates the Beyond10x agent plugin marketplace. With no subcommand: the offline gate. +#[derive(Debug, Parser)] +#[command(name = "agentplugins-check")] +struct Cli { + #[command(subcommand)] + command: Option, +} + +#[derive(Debug, Subcommand)] +enum Top { + /// The offline gate, then every spelled CLI command against the newest releases (network). + Tools, + /// Release checks. + Release { + #[command(subcommand)] + action: ReleaseAction, + }, + /// Validate the eval corpus and replay recorded transcripts, or one of the corpus verbs. + Evals { + #[command(subcommand)] + action: Option, + }, + /// Check that a headless trial run loaded only the sandbox's plugins, at the version under test. + TrialIsolation { + /// The run's stream-json output. + run: PathBuf, + /// The sandbox directory (its `home/` is the run's `HOME`). + #[arg(long)] + sandbox: PathBuf, + /// A directory marketplace the plugins may load from in place (the checkout under test). + #[arg(long)] + marketplace: Option, + /// The agentplugins version under test. + #[arg(long)] + version: String, + /// An upgrade trial: plugins start at the versions the sandbox was seeded with. + #[arg(long)] + seeded: bool, + }, + /// Prepare a fresh sandbox for a trial defined under `trials/`: copy and commit its fixture, + /// and write the prompt, working directory and setup `task trial:run` reads. + TrialPrepare { + /// The trial's name (its directory under `trials/`). + trial: String, + /// The sandbox directory made by `task trial:sandbox`. + #[arg(long)] + sandbox: PathBuf, + }, + /// Measure a trial run: tool calls, validation, synthesis, `UNMAPPED:` markers, outputs and + /// `go test` counts, one line each; with `--baseline`, exit 1 when a measure got worse. + TrialReport { + /// The run's stream-json output. + run: PathBuf, + /// The trial under `trials/` the run executed; without it, an ad-hoc `PROMPT=` run gets + /// every measure that needs no definition. + #[arg(long)] + trial: Option, + /// The sandbox directory; default: the directory holding the run. + #[arg(long)] + sandbox: Option, + /// Compare with this baseline (`trials/baseline.json`) and exit 1 when a measure got worse. + #[arg(long, requires = "trial")] + baseline: Option, + /// Record this run as the trial's baseline, in `--baseline` or `trials/baseline.json`. + #[arg(long, requires = "trial")] + write_baseline: bool, + }, +} + +/// `trial-report`: print the measures, compare them with the baseline, and record them. +fn trial_report( + root: &Path, + run: &Path, + trial: Option<&str>, + sandbox: Option, + baseline: Option, + write_baseline: bool, +) -> Result { + let definition = trial.map(|name| trials::load(root, name)).transpose()?; + let text = std::fs::read_to_string(run) + .map_err(|error| format!("reading {}: {error}", run.display()))?; + let sandbox = sandbox + .or_else(|| run.parent().map(Path::to_path_buf)) + .ok_or("no sandbox: pass --sandbox")?; + let report = report::measure(&text, &sandbox, definition.as_ref())?; + println!("trial {}: {}", trial.unwrap_or("(ad hoc)"), run.display()); + for line in &report.lines { + println!(" {line}"); + } + let mut held = true; + if let (Some(trial), Some(path)) = (trial, baseline.as_deref()) { + match report::read_baseline(path)?.get(trial) { + None => println!( + "baseline: none recorded for `{trial}` in {}", + path.display() + ), + Some(then) => { + let worse = report::worse(then, &report.measures); + if worse.is_empty() { + println!("baseline: no measure got worse than {}", path.display()); + } else { + held = write_baseline; + eprintln!("worse than the baseline in {}:", path.display()); + for line in worse { + eprintln!(" {line}"); + } + } + } + } + } + if let (Some(trial), true) = (trial, write_baseline) { + let path = baseline.unwrap_or_else(|| root.join(trials::BASELINE)); + report::write_baseline(&path, trial, &report.measures)?; + println!("baseline: recorded `{trial}` in {}", path.display()); + } + Ok(held) +} + +#[derive(Debug, Subcommand)] +enum ReleaseAction { + /// Check that every version in the repository agrees with the tag. + Verify { + /// The tag being released. + version: String, + }, +} + fn main() -> ExitCode { let root = PathBuf::from(env!("CARGO_MANIFEST_DIR")) .parent() .and_then(Path::parent) .expect("checker is under the repository root") .to_path_buf(); - let arguments = std::env::args().skip(1).collect::>(); - if let Some(code) = evals::cli(&root, &arguments) { - return code; - } - let result = match arguments.as_slice() { - [] => check(&root), - [release, verify, version] if release == "release" && verify == "verify" => { - check(&root).and_then(|()| verify_release(&root, version)) + let result = match Cli::parse().command { + None => check(&root), + Some(Top::Tools) => check(&root).and_then(|()| tools::verify(&root)), + Some(Top::Release { + action: ReleaseAction::Verify { version }, + }) => check(&root).and_then(|()| verify_release(&root, &version)), + Some(Top::Evals { action }) => return evals::run(&root, action), + Some(Top::TrialIsolation { + run, + sandbox, + marketplace, + version, + seeded, + }) => { + return match trial::isolation(&run, &sandbox, marketplace.as_deref(), &version, seeded) + { + Ok(summary) => { + println!("{summary}"); + ExitCode::SUCCESS + } + Err(error) => { + eprintln!("error: {error}"); + ExitCode::from(1) + } + }; + } + Some(Top::TrialPrepare { trial, sandbox }) => { + return match trials::prepare(&root, &trial, &sandbox) { + Ok(summary) => { + println!("{summary}"); + ExitCode::SUCCESS + } + Err(error) => { + eprintln!("error: {error}"); + ExitCode::from(1) + } + }; + } + Some(Top::TrialReport { + run, + trial, + sandbox, + baseline, + write_baseline, + }) => { + return match trial_report( + &root, + &run, + trial.as_deref(), + sandbox, + baseline, + write_baseline, + ) { + Ok(true) => ExitCode::SUCCESS, + Ok(false) => ExitCode::from(1), + Err(error) => { + eprintln!("error: {error}"); + ExitCode::from(1) + } + }; } - _ => Err( - "usage: agentplugins-check [release verify | evals | evals scope ...]" - .to_owned(), - ), }; match result { Ok(()) => { println!( - "valid: marketplace beyond10x, {} focused plugin(s)", + "valid: marketplace {MARKETPLACE}, {} plugin(s)", PLUGINS.len() ); ExitCode::SUCCESS @@ -769,6 +1263,24 @@ fn main() -> ExitCode { mod tests { use super::*; + /// The shapes the plan-host reader has to tell apart: a command in a block and inline, with + /// and without `--host`, and prose or a skill name that only mentions the verb. + #[test] + fn a_plan_command_without_a_host_is_named_by_line() { + let text = "\ +intro naming `b10x:init` and the `b10x` CLI +```bash +b10x init ess --out plan.json +b10x init ess --host claude --out plan.json +b10x upgrade --out plan.json +``` +run `b10x init aep,ess --out …` or `b10x upgrade ess --host codex --out …` +one product only: `b10x upgrade ess` +`b10x initialise` is not a verb +"; + assert_eq!(hostless_plans(text), vec![3, 5, 7, 8]); + } + #[test] fn the_committed_marketplace_is_valid() { let root = Path::new(env!("CARGO_MANIFEST_DIR")) @@ -785,9 +1297,9 @@ mod tests { fn a_deleted_critic_fails_the_check() { let required: &[&str] = PLUGINS .iter() - .find(|(name, _)| *name == "aep-plan") + .find(|(name, _)| *name == "aep") .map(|(_, required)| *required) - .expect("aep-plan is one of the focused plugins"); + .expect("aep is one of the focused plugins"); let critic = "agents/plan-critic-design.md"; assert!( required.contains(&critic), @@ -800,26 +1312,25 @@ mod tests { .expect("checker is under repository root"); let sandbox = std::env::temp_dir().join(format!("agentplugins-check-critic-{}", std::process::id())); - let plugin_root = sandbox.join("plugins").join("aep-plan"); + let plugin_root = sandbox.join("plugins").join("aep"); let write = |target: &Path, bytes: &str| { std::fs::create_dir_all(target.parent().expect("every entry has a directory")) .expect("the sandbox is writable"); std::fs::write(target, bytes).expect("the sandbox is writable"); }; for manifest in [".codex-plugin/plugin.json", ".claude-plugin/plugin.json"] { - let committed = - std::fs::read_to_string(repository.join("plugins/aep-plan").join(manifest)) - .expect("the committed manifest is readable"); + let committed = std::fs::read_to_string(repository.join("plugins/aep").join(manifest)) + .expect("the committed manifest is readable"); write(&plugin_root.join(manifest), &committed); } for relative in required.iter().filter(|relative| **relative != critic) { write(&plugin_root.join(relative), ""); } - let error = plugin(&sandbox, "aep-plan", required) + let error = plugin(&sandbox, "aep", required) .expect_err("a plugin missing one of its critics must fail the check"); std::fs::remove_dir_all(&sandbox).expect("the sandbox is removable"); - assert_eq!(error, format!("plugin `aep-plan` is missing `{critic}`")); + assert_eq!(error, format!("plugin `aep` is missing `{critic}`")); } /// A critic that declares no `model:`/`effort:` runs on whatever the calling session was on, @@ -833,7 +1344,7 @@ mod tests { std::process::id(), std::thread::current().id() )); - let agents = sandbox.join("plugins/aep-plan/agents"); + let agents = sandbox.join("plugins/aep/agents"); let write = |name: &str, body: &str| { std::fs::create_dir_all(&agents).expect("the sandbox is writable"); std::fs::write(agents.join(name), body).expect("the sandbox is writable"); @@ -870,6 +1381,34 @@ mod tests { std::fs::remove_dir_all(&sandbox).expect("the sandbox is removable"); } + /// A green suite is where hardening starts, so the two places an agent stands when it gets + /// there — the conformance agent's charter and the skill's mutation section — hand it the + /// catalogue. Without the link the catalogue is a file nobody finishing a coverage task reads. + #[test] + fn a_green_suite_is_handed_the_hardening_catalogue() { + let root = Path::new(env!("CARGO_MANIFEST_DIR")) + .parent() + .and_then(Path::parent) + .expect("checker is under repository root"); + let agent = std::fs::read_to_string(root.join("plugins/ess/agents/conformance.md")) + .expect("the conformance agent is readable"); + assert!( + agent.contains("`ess:hardening`"), + "the conformance agent does not offer `ess:hardening`" + ); + let skill = + std::fs::read_to_string(root.join("plugins/ess/skills/testing-conformance/SKILL.md")) + .expect("the testing-conformance skill is readable"); + let mutation = skill + .split_once("## Before anything: does a green run mean anything?") + .and_then(|(_, rest)| rest.split_once("\n## ").map(|(section, _)| section)) + .expect("the mutation section is present"); + assert!( + mutation.contains("`ess:hardening`"), + "the mutation section does not link `ess:hardening`" + ); + } + /// The committed critics carry the pin. This is the half of the check that would catch a fifth /// critic added without one. #[test] @@ -928,7 +1467,7 @@ mod tests { retired_hits("a\nplugins/aep-planning/x\n", planning, false), vec![2] ); - assert!(retired_hits("plugins/aep-plan/x\n", planning, false).is_empty()); + assert!(retired_hits("plugins/aep/x\n", planning, false).is_empty()); } /// The marker excuses a line in a specification a transcript is replayed against, and nowhere @@ -985,13 +1524,13 @@ mod tests { assert_eq!( error, "1 retired plugin name(s) remain, and `AGENTS.md` § Invariants forbids depending on \ - one:\n README.md:1 names `ess-schema`, which is now `ess-specify`" + one:\n README.md:1 names `ess-schema`, which is now `ess@b10x`" ); // The three places the old names are the truth: what the changelog says the plugins were // called, what a dated change record said on its day, and the transcript of a run that // happened under them. - write("README.md", "install `ess-specify` from the marketplace\n"); + write("README.md", "install `aep` from the marketplace\n"); write("CHANGELOG.md", "renamed `ess-schema` to `ess-specify`\n"); write("changes/2026-09-03-rename.yaml", "plugin: adp\n"); write( @@ -1019,7 +1558,7 @@ mod tests { /// about the folder as much as about the prose. /// /// Measured on a copy of this worktree in a scratch directory, 2026-09-03: a - /// `plugins//skills/wave/SKILL.md` whose body never spells the retired name, and a + /// `plugins//skills/implementing/SKILL.md` whose body never spells the retired name, and a /// `website/docs/plugins/.md` whose body never spells it either, both left /// `cargo run --bin agentplugins-check` printing /// `valid: marketplace beyond10x, 5 focused plugin(s)`. @@ -1043,7 +1582,7 @@ mod tests { // A leftover from a rename that moved the parent and missed one file. Its text names the // new world; only where it sits still names the old one. - let plugin_file = format!("plugins/{retired}/skills/wave/SKILL.md"); + let plugin_file = format!("plugins/{retired}/skills/implementing/SKILL.md"); let page = format!("website/docs/plugins/{retired}.md"); write( &plugin_file, diff --git a/crates/agentplugins-check/src/report.rs b/crates/agentplugins-check/src/report.rs new file mode 100644 index 0000000..aa04325 --- /dev/null +++ b/crates/agentplugins-check/src/report.rs @@ -0,0 +1,917 @@ +//! What a headless trial run achieved, in numbers, and whether they got worse than the baseline. +//! +//! Until this existed a run was read by hand and nothing compared one round with the last. Every +//! number here comes from the run's own stream-json (tool calls and the output of the commands it +//! ran) or from the files it left in the sandbox, never from what the agent's final prose claims: +//! a `UNMAPPED:` marker counts when it is in a spec file on disk. + +use std::collections::{BTreeMap, BTreeSet, HashSet}; +use std::fmt::Write as _; +use std::path::{Path, PathBuf}; + +use serde::{Deserialize, Serialize}; +use serde_json::Value; + +use crate::trials::{Definition, Measure}; + +/// A Bash call and what it printed. +struct Shell { + command: String, + output: Option, +} + +/// What the report reads from a stream-json run. +struct Run { + tool_calls: usize, + shells: Vec, + /// Paths given to `Write`, `Edit` and `MultiEdit`, in first-write order. + written: Vec, +} + +fn result_text(content: &Value) -> String { + match content { + Value::String(text) => text.clone(), + Value::Array(blocks) => blocks + .iter() + .filter_map(|block| block["text"].as_str()) + .collect::>() + .join("\n"), + _ => String::new(), + } +} + +fn parse(text: &str) -> Run { + let mut seen = HashSet::new(); + let mut tool_calls = 0; + let mut shells = Vec::new(); + let mut by_id = BTreeMap::new(); + let mut written = Vec::new(); + for line in text.lines() { + let Ok(event) = serde_json::from_str::(line) else { + continue; + }; + for block in event["message"]["content"].as_array().into_iter().flatten() { + match block["type"].as_str() { + Some("tool_use") => { + let id = block["id"].as_str().unwrap_or_default().to_owned(); + if !id.is_empty() && !seen.insert(id.clone()) { + continue; + } + tool_calls += 1; + let input = &block["input"]; + match block["name"].as_str() { + Some("Bash") => { + by_id.insert(id, shells.len()); + shells.push(Shell { + command: input["command"].as_str().unwrap_or_default().to_owned(), + output: None, + }); + } + Some("Write" | "Edit" | "MultiEdit") => { + if let Some(path) = input["file_path"].as_str() { + let path = PathBuf::from(path); + if !written.contains(&path) { + written.push(path); + } + } + } + _ => {} + } + } + Some("tool_result") => { + let id = block["tool_use_id"].as_str().unwrap_or_default(); + if let Some(&index) = by_id.get(id) { + shells[index].output = Some(result_text(&block["content"])); + } + } + _ => {} + } + } + } + Run { + tool_calls, + shells, + written, + } +} + +/// Whether `ess specify validate` ran, and what its last output said. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "kebab-case")] +pub enum Validate { + /// Never ran. + NotRun, + /// Its last output was not `valid`. + Invalid, + /// Its last output said `valid`. + Valid, +} + +/// The counts of the last synthesis. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] +pub struct Synthesis { + /// Scenarios synthesized. + pub scenarios: u64, + /// Refusals reported. + pub refusals: u64, +} + +/// The counts of the last `go test`. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Serialize, Deserialize)] +pub struct GoTest { + /// Tests that passed. + pub passed: u64, + /// Tests that failed, and packages that did not build. + pub failed: u64, + /// Tests that were skipped. + pub skipped: u64, +} + +/// One run's numbers; a field is present exactly when the trial measures it. The same shape is a +/// trial's entry in `trials/baseline.json`. +/// +/// `synthesis` and `go_test` have three states on purpose: absent (not measured), `null` (measured, +/// never ran) and counts; a baseline that had counts and a run with `null` got worse. +#[allow(clippy::option_option)] +#[derive(Debug, Clone, Default, PartialEq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct Measures { + /// Tool calls. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub tool_calls: Option, + /// `ess specify validate`. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub validate: Option, + /// The last synthesis; `null` when none ran. + #[serde( + default, + skip_serializing_if = "Option::is_none", + deserialize_with = "measured" + )] + pub synthesis: Option>, + /// `UNMAPPED:` markers in spec files the run wrote. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub unmapped: Option, + /// The listed outputs that exist. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub outputs: Option>, + /// The last `go test`; `null` when none ran. + #[serde( + default, + skip_serializing_if = "Option::is_none", + deserialize_with = "measured" + )] + pub go_test: Option>, +} + +/// A field that is present is measured, even when its value is `null` (the command never ran); +/// an absent field is a measure the trial does not take. +#[allow(clippy::option_option)] +fn measured<'de, D, T>(deserializer: D) -> Result>, D::Error> +where + D: serde::Deserializer<'de>, + T: Deserialize<'de>, +{ + Option::::deserialize(deserializer).map(Some) +} + +fn is_validate(command: &str) -> bool { + command.contains("specify validate") || command.contains("ess validate") +} + +/// The verdict of one validate output: the last line that states one. +fn verdict(output: &str) -> Validate { + let mut verdict = Validate::Invalid; + for line in output.lines() { + let line = line.trim(); + if line == "valid" || line.ends_with(", valid") || line.contains("\"valid\": true") { + verdict = Validate::Valid; + } else if line.contains(" was refused") || line.contains("\"valid\": false") { + verdict = Validate::Invalid; + } + } + verdict +} + +/// The number right before `word` on `line`. +fn number_before(line: &str, word: &str) -> Option { + let before = &line[..line.find(word)?]; + let digits: String = before + .trim_end() + .chars() + .rev() + .take_while(char::is_ascii_digit) + .collect(); + digits.chars().rev().collect::().parse().ok() +} + +/// `N scenario(s) … M refusal(s)` from the last synthesis that printed it. +fn synthesis(run: &Run) -> Option { + let mut last = None; + for shell in run + .shells + .iter() + .filter(|s| s.command.contains("synthesize")) + { + for line in shell.output.as_deref().unwrap_or_default().lines() { + if let (Some(scenarios), Some(refusals)) = ( + number_before(line, "scenario(s)"), + number_before(line, "refusal(s)"), + ) { + last = Some(Synthesis { + scenarios, + refusals, + }); + } + } + } + last +} + +/// Counts of one `go test` output, from `-v` lines or `-json` events; `None` when it holds neither. +fn go_counts(output: &str) -> Option { + let mut results: BTreeMap = BTreeMap::new(); + let mut broken = 0; + for line in output.lines() { + let trimmed = line.trim(); + if let Ok(event) = serde_json::from_str::(trimmed) { + if let (Some(action), Some(test)) = (event["Action"].as_str(), event["Test"].as_str()) { + if let Some(status) = ["pass", "fail", "skip"].into_iter().find(|s| *s == action) { + results.insert(test.to_owned(), status); + } + } + continue; + } + for (prefix, status) in [ + ("--- PASS: ", "pass"), + ("--- FAIL: ", "fail"), + ("--- SKIP: ", "skip"), + ] { + if let Some(rest) = trimmed.strip_prefix(prefix) { + let name = rest.rsplit_once(" (").map_or(rest, |(name, _)| name); + results.insert(name.to_owned(), status); + } + } + if trimmed.starts_with("FAIL") + && (trimmed.ends_with("[build failed]") || trimmed.ends_with("[setup failed]")) + { + broken += 1; + } + } + if results.is_empty() && broken == 0 { + return None; + } + // A top-level test with subtests is their parent; its verdict repeats theirs. + let parents: BTreeSet<&str> = results + .keys() + .filter_map(|name| name.split_once('/').map(|(top, _)| top)) + .collect(); + let mut counts = GoTest { + failed: broken, + ..GoTest::default() + }; + for (name, status) in &results { + if !name.contains('/') && parents.contains(name.as_str()) { + continue; + } + match *status { + "pass" => counts.passed += 1, + "fail" => counts.failed += 1, + _ => counts.skipped += 1, + } + } + Some(counts) +} + +/// Whether a shell command runs `go test`, however it is spelled: `go test`, `go -C test`, +/// behind environment assignments or after `cd … &&`. +fn runs_go_test(command: &str) -> bool { + let words: Vec = shlex::split(command) + .unwrap_or_else(|| command.split_whitespace().map(str::to_owned).collect()); + words.iter().enumerate().any(|(index, word)| { + if word != "go" && !word.ends_with("/go") { + return false; + } + // In command position: every word since the last separator is an environment assignment. + let start = words[..index] + .iter() + .rposition(|w| matches!(w.as_str(), "&&" | "||" | ";" | "|" | "then" | "do")) + .map_or(0, |separator| separator + 1); + if !words[start..index] + .iter() + .all(|w| w.contains('=') && !w.starts_with('-')) + { + return false; + } + let mut rest = words[index + 1..].iter(); + while let Some(next) = rest.next() { + match next.as_str() { + "-C" => { + rest.next(); + } + flag if flag.starts_with('-') => {} + verb => return verb == "test", + } + } + false + }) +} + +fn go_test(run: &Run) -> Option { + run.shells + .iter() + .filter(|shell| runs_go_test(&shell.command)) + .rev() + .find_map(|shell| go_counts(shell.output.as_deref()?)) +} + +/// `UNMAPPED:` markers in the YAML files the run wrote inside the sandbox, as they are on disk. +fn unmapped(run: &Run, sandbox: &Path, workdir: &Path) -> (u64, usize) { + let mut markers = 0; + let mut files = 0; + for path in &run.written { + let path = if path.is_absolute() { + path.clone() + } else { + workdir.join(path) + }; + let yaml = path + .extension() + .is_some_and(|extension| extension == "yaml" || extension == "yml"); + let inside = std::fs::canonicalize(&path).is_ok_and(|path| path.starts_with(sandbox)); + if !yaml || !inside { + continue; + } + if let Ok(text) = std::fs::read_to_string(&path) { + files += 1; + markers += text.matches("UNMAPPED:").count() as u64; + } + } + (markers, files) +} + +fn present(path: &Path) -> bool { + if path.is_dir() { + std::fs::read_dir(path).is_ok_and(|mut entries| entries.next().is_some()) + } else { + path.is_file() + } +} + +/// A run measured, with the lines that say so. +pub struct Report { + /// The numbers. + pub measures: Measures, + /// One line per measure. + pub lines: Vec, +} + +/// Measure a run. `definition` is `None` for an ad-hoc `PROMPT=` run, which gets every measure +/// that needs no definition. +pub fn measure( + text: &str, + sandbox: &Path, + definition: Option<&Definition>, +) -> Result { + let sandbox = std::fs::canonicalize(sandbox) + .map_err(|error| format!("sandbox {}: {error}", sandbox.display()))?; + let wants = |measure: Measure| { + definition.map_or(measure != Measure::Outputs, |definition| { + definition.measures(measure) + }) + }; + let workdir = definition.map_or_else(|| sandbox.join("work"), |d| sandbox.join(d.workdir())); + let run = parse(text); + if run.tool_calls == 0 && !text.lines().any(|line| line.contains("\"type\"")) { + return Err("no stream-json events: not a trial run".to_owned()); + } + let mut measures = Measures::default(); + let mut lines = Vec::new(); + + measures.tool_calls = Some(run.tool_calls as u64); + lines.push(format!("tool calls: {}", run.tool_calls)); + + if wants(Measure::Validate) { + let runs: Vec<&Shell> = run + .shells + .iter() + .filter(|s| is_validate(&s.command)) + .collect(); + let validate = runs.last().map_or(Validate::NotRun, |shell| { + verdict(shell.output.as_deref().unwrap_or_default()) + }); + measures.validate = Some(validate); + lines.push(match validate { + Validate::NotRun => "validate: not run".to_owned(), + Validate::Invalid => format!("validate: not valid (last of {} run(s))", runs.len()), + Validate::Valid => format!("validate: valid (last of {} run(s))", runs.len()), + }); + } + if wants(Measure::Synthesis) { + let synthesis = synthesis(&run); + measures.synthesis = Some(synthesis); + lines.push(synthesis.map_or_else( + || "synthesis: not run".to_owned(), + |s| { + format!( + "synthesis: {} scenario(s), {} refusal(s)", + s.scenarios, s.refusals + ) + }, + )); + } + if wants(Measure::Unmapped) { + let (markers, files) = unmapped(&run, &sandbox, &workdir); + measures.unmapped = Some(markers); + lines.push(format!( + "unmapped: {markers} `UNMAPPED:` marker(s) in {files} YAML file(s) the run wrote" + )); + } + if let Some(definition) = definition.filter(|d| d.measures(Measure::Outputs)) { + let mut have = BTreeSet::new(); + let mut missing = Vec::new(); + for (name, path) in &definition.outputs { + if present(&workdir.join(path)) { + have.insert(name.clone()); + } else { + missing.push(format!("{name} ({path})")); + } + } + let mut line = format!( + "outputs: {}/{} present", + have.len(), + definition.outputs.len() + ); + if !missing.is_empty() { + let _ = write!(line, "; missing: {}", missing.join(", ")); + } + lines.push(line); + measures.outputs = Some(have); + } + if wants(Measure::GoTest) { + let counts = go_test(&run); + measures.go_test = Some(counts); + lines.push(counts.map_or_else( + || "go test: not run".to_owned(), + |c| { + format!( + "go test: {} passed, {} failed, {} skipped", + c.passed, c.failed, c.skipped + ) + }, + )); + } + Ok(Report { measures, lines }) +} + +/// Every way `now` is worse than `then`. Only a measure both record is compared. +#[must_use] +pub fn worse(then: &Measures, now: &Measures) -> Vec { + let mut found = Vec::new(); + if let (Some(then), Some(now)) = (then.tool_calls, now.tool_calls) { + if now * 2 > then * 3 { + found.push(format!("tool calls: {then} → {now}, up by more than 50%")); + } + } + if let (Some(Validate::Valid), Some(now)) = (then.validate, now.validate) { + if now != Validate::Valid { + found.push(format!("validate: was valid, now {now:?}")); + } + } + if let (Some(Some(then)), Some(now)) = (then.synthesis, now.synthesis) { + match now { + None => found.push("synthesis: ran before, now did not".to_owned()), + Some(now) if now.refusals > then.refusals => found.push(format!( + "synthesis: refusals {} → {}", + then.refusals, now.refusals + )), + Some(_) => {} + } + } + if let (Some(then), Some(now)) = (&then.outputs, &now.outputs) { + let lost: Vec<&str> = then.difference(now).map(String::as_str).collect(); + if !lost.is_empty() { + found.push(format!("outputs: missing now: {}", lost.join(", "))); + } + } + if let (Some(Some(then)), Some(now)) = (then.go_test, now.go_test) { + match now { + None => found.push("go test: ran before, now did not".to_owned()), + Some(now) if now.failed > then.failed => found.push(format!( + "go test: failures {} → {}", + then.failed, now.failed + )), + Some(_) => {} + } + } + found +} + +/// Read `trials/baseline.json` (or another baseline file). +pub fn read_baseline(path: &Path) -> Result, String> { + let text = std::fs::read_to_string(path) + .map_err(|error| format!("reading {}: {error}", path.display()))?; + serde_json::from_str(&text).map_err(|error| format!("parsing {}: {error}", path.display())) +} + +/// Record `measures` as `trial`'s baseline in `path`, keeping every other trial's entry. +pub fn write_baseline(path: &Path, trial: &str, measures: &Measures) -> Result<(), String> { + let mut baseline = if path.exists() { + read_baseline(path)? + } else { + BTreeMap::new() + }; + baseline.insert(trial.to_owned(), measures.clone()); + let text = serde_json::to_string_pretty(&baseline).map_err(|error| error.to_string())?; + std::fs::write(path, text + "\n") + .map_err(|error| format!("writing {}: {error}", path.display())) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::trials::Kind; + + fn tool(id: &str, name: &str, input: &str) -> String { + format!( + r#"{{"type":"assistant","message":{{"content":[{{"type":"tool_use","id":"{id}","name":"{name}","input":{input}}}]}}}}"# + ) + } + + fn result(id: &str, text: &str) -> String { + let content = serde_json::to_string(text).unwrap(); + format!( + r#"{{"type":"user","message":{{"content":[{{"type":"tool_result","tool_use_id":"{id}","content":[{{"type":"text","text":{content}}}]}}]}}}}"# + ) + } + + fn bash(id: &str, command: &str, output: &str) -> String { + let input = serde_json::json!({ "command": command }).to_string(); + format!("{}\n{}", tool(id, "Bash", &input), result(id, output)) + } + + fn scratch(label: &str) -> PathBuf { + let path = std::env::temp_dir().join(format!( + "agentplugins-check-report-{label}-{}", + std::process::id() + )); + let _ = std::fs::remove_dir_all(&path); + std::fs::create_dir_all(path.join("work")).unwrap(); + std::fs::canonicalize(path).unwrap() + } + + fn definition(measures: &[Measure], outputs: &[(&str, &str)]) -> Definition { + Definition { + name: "demo".to_owned(), + kind: Kind::EssFullPackage, + prompt: "p".to_owned(), + dir: None, + fixture: None, + remote: false, + seeded: false, + setup: Vec::new(), + measures: measures.to_vec(), + outputs: outputs + .iter() + .map(|(name, path)| ((*name).to_owned(), (*path).to_owned())) + .collect(), + } + } + + const INIT: &str = r#"{"type":"system","subtype":"init"}"#; + + #[test] + fn validate_is_the_verdict_of_the_last_run() { + let sandbox = scratch("validate"); + let valid = bash( + "a", + "ess specify validate --path spec", + "garden v1 — 2 file(s), valid", + ); + let refused = bash( + "b", + "ess specify validate --path spec", + "spec was refused:\n - domains/plot.yaml: missing field `type`", + ); + let only = definition(&[Measure::ToolCalls, Measure::Validate], &[]); + let run = [INIT, &refused, &valid].join("\n"); + let report = measure(&run, &sandbox, Some(&only)).unwrap(); + assert_eq!(report.measures.validate, Some(Validate::Valid)); + assert_eq!(report.lines[1], "validate: valid (last of 2 run(s))"); + let run = [INIT, &valid, &refused].join("\n"); + let report = measure(&run, &sandbox, Some(&only)).unwrap(); + assert_eq!(report.measures.validate, Some(Validate::Invalid)); + let json = bash( + "c", + "ess specify validate --format json", + "{\n \"valid\": true\n}", + ); + let report = measure(&[INIT, &json].join("\n"), &sandbox, Some(&only)).unwrap(); + assert_eq!(report.measures.validate, Some(Validate::Valid)); + let report = measure(INIT, &sandbox, Some(&only)).unwrap(); + assert_eq!(report.measures.validate, Some(Validate::NotRun)); + std::fs::remove_dir_all(&sandbox).unwrap(); + } + + #[test] + fn synthesis_counts_come_from_the_last_synthesize_output() { + let sandbox = scratch("synthesis"); + let first = bash( + "a", + "ess verify conform synthesize --path spec --target go --out out/conformance", + "refused: refusal[ESS-SYNTH-011]: …\n12 scenario(s) (0 authored), 5 refusal(s), 5 file(s) written to out", + ); + let second = bash( + "b", + "cd spec && ess verify conform synthesize --target ir --out suite.json", + "14 scenario(s) (0 authored), 2 refusal(s), 1 file(s) written to suite.json", + ); + let unrelated = bash( + "c", + "echo '3 scenario(s), 9 refusal(s)'", + "3 scenario(s), 9 refusal(s)", + ); + let run = [INIT, &first, &second, &unrelated].join("\n"); + let report = measure(&run, &sandbox, None).unwrap(); + assert_eq!( + report.measures.synthesis, + Some(Some(Synthesis { + scenarios: 14, + refusals: 2 + })) + ); + assert!(report + .lines + .contains(&"synthesis: 14 scenario(s), 2 refusal(s)".to_owned())); + assert_eq!(report.measures.tool_calls, Some(3)); + std::fs::remove_dir_all(&sandbox).unwrap(); + } + + #[test] + fn unmapped_markers_are_read_from_the_files_on_disk() { + let sandbox = scratch("unmapped"); + let spec = sandbox.join("work/spec"); + std::fs::create_dir_all(&spec).unwrap(); + std::fs::write( + spec.join("loans.yaml"), + "# UNMAPPED: who may lend\n# UNMAPPED: overdue rule\ndomain: x\n", + ) + .unwrap(); + std::fs::write(spec.join("system.yaml"), "format: ess/1\n").unwrap(); + std::fs::write(spec.join("notes.md"), "UNMAPPED: not a spec\n").unwrap(); + let write = |id: &str, file: &str| { + let input = serde_json::json!({ "file_path": spec.join(file), "content": "UNMAPPED: in the prose only" }); + tool(id, "Write", &input.to_string()) + }; + let edit = tool( + "d", + "Edit", + &serde_json::json!({ "file_path": spec.join("loans.yaml") }).to_string(), + ); + let outside = tool( + "e", + "Write", + &serde_json::json!({ "file_path": "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/nonexistent/elsewhere.yaml" }).to_string(), + ); + let run = [ + INIT, + &write("a", "loans.yaml"), + &write("b", "system.yaml"), + &write("c", "notes.md"), + &edit, + &outside, + ] + .join("\n"); + let report = measure(&run, &sandbox, None).unwrap(); + assert_eq!(report.measures.unmapped, Some(2)); + assert!(report.lines.contains( + &"unmapped: 2 `UNMAPPED:` marker(s) in 2 YAML file(s) the run wrote".to_owned() + )); + assert_eq!(report.measures.tool_calls, Some(5)); + std::fs::remove_dir_all(&sandbox).unwrap(); + } + + #[test] + fn every_spelling_of_go_test_is_recognised() { + for command in [ + "go test ./...", + "cd impl && go test -v ./...", + "go -C /work/impl test -count=1 -v ./... 2>&1 | tail -20", + "ESS_REPORT_OUT=r.json go -C impl test ./...", + "/usr/local/go/bin/go test ./...", + ] { + assert!(runs_go_test(command), "{command}"); + } + for command in [ + "go vet ./...", + "go -C impl build ./...", + "echo go test", + "cargo test", + ] { + assert!(!runs_go_test(command), "{command}"); + } + } + + #[test] + fn go_test_counts_leaf_tests_of_the_last_run() { + let sandbox = scratch("gotest"); + let verbose = "=== RUN TestConformance\n=== RUN TestConformance/garden.plot.Plot/assign/assigned\n --- PASS: TestConformance/garden.plot.Plot/assign/assigned (0.00s)\n --- PASS: TestConformance/garden.plot.Plot (0.00s)\n --- SKIP: TestConformance/garden.plot.Release/redeliver (0.00s)\n --- FAIL: TestConformance/garden.plot.Plot/invariant (0.01s)\n--- FAIL: TestConformance (0.02s)\n--- PASS: TestStore (0.00s)\nFAIL\nFAIL\texample.com/garden/impl\t0.031s"; + let broken = bash( + "a", + "go test ./...", + "# example.com/garden/impl\nimpl/target.go:9:2: undefined: essconform.Tagret\nFAIL\texample.com/garden/impl [build failed]", + ); + let listed = bash( + "c", + "go test -list . ./impl", + "TestConformance\nok \texample.com/garden/impl\t0.002s", + ); + let only = definition(&[Measure::ToolCalls, Measure::GoTest], &[]); + let run = [ + INIT, + &broken, + &bash("b", "cd impl && go test -v ./...", verbose), + &listed, + ] + .join("\n"); + let report = measure(&run, &sandbox, Some(&only)).unwrap(); + assert_eq!( + report.measures.go_test, + Some(Some(GoTest { + passed: 3, + failed: 1, + skipped: 1 + })) + ); + assert_eq!(report.lines[1], "go test: 3 passed, 1 failed, 1 skipped"); + let report = measure(&[INIT, &broken].join("\n"), &sandbox, Some(&only)).unwrap(); + assert_eq!( + report.measures.go_test, + Some(Some(GoTest { + passed: 0, + failed: 1, + skipped: 0 + })) + ); + let json = "{\"Action\":\"run\",\"Test\":\"TestConformance\"}\n{\"Action\":\"pass\",\"Test\":\"TestConformance/a\"}\n{\"Action\":\"skip\",\"Test\":\"TestConformance/b\"}\n{\"Action\":\"pass\",\"Test\":\"TestConformance\"}"; + let report = measure( + &[INIT, &bash("d", "go test -json ./...", json)].join("\n"), + &sandbox, + Some(&only), + ) + .unwrap(); + assert_eq!( + report.measures.go_test, + Some(Some(GoTest { + passed: 1, + failed: 0, + skipped: 1 + })) + ); + std::fs::remove_dir_all(&sandbox).unwrap(); + } + + #[test] + fn outputs_are_checked_on_disk_and_only_when_listed() { + let sandbox = scratch("outputs"); + let out = sandbox.join("work/out"); + std::fs::create_dir_all(out.join("schema")).unwrap(); + std::fs::write(out.join("schema/a.json"), "{}").unwrap(); + std::fs::create_dir_all(out.join("site")).unwrap(); + std::fs::write(out.join("openapi.yaml"), "openapi: 3.1.0").unwrap(); + let listed = definition( + &[Measure::ToolCalls, Measure::Outputs], + &[ + ("schema", "out/schema"), + ("openapi", "out/openapi.yaml"), + ("site", "out/site"), + ("asyncapi", "out/asyncapi"), + ], + ); + let report = measure(INIT, &sandbox, Some(&listed)).unwrap(); + assert_eq!( + report.measures.outputs, + Some(["openapi".to_owned(), "schema".to_owned()].into()) + ); + assert_eq!( + report.lines[1], + "outputs: 2/4 present; missing: asyncapi (out/asyncapi), site (out/site)" + ); + assert_eq!(report.lines.len(), 2, "unlisted measures print nothing"); + let adhoc = measure(INIT, &sandbox, None).unwrap(); + assert!(adhoc.measures.outputs.is_none()); + assert!(measure("not json", &sandbox, None).is_err()); + std::fs::remove_dir_all(&sandbox).unwrap(); + } + + #[test] + fn a_duplicated_tool_use_is_counted_once() { + let sandbox = scratch("dupes"); + let call = tool("same", "Read", "{}"); + let report = measure(&[INIT, &call, &call].join("\n"), &sandbox, None).unwrap(); + assert_eq!(report.measures.tool_calls, Some(1)); + std::fs::remove_dir_all(&sandbox).unwrap(); + } + + fn full() -> Measures { + Measures { + tool_calls: Some(40), + validate: Some(Validate::Valid), + synthesis: Some(Some(Synthesis { + scenarios: 12, + refusals: 2, + })), + unmapped: Some(0), + outputs: Some(["go".to_owned(), "schema".to_owned()].into()), + go_test: Some(Some(GoTest { + passed: 10, + failed: 0, + skipped: 2, + })), + } + } + + #[test] + fn the_same_or_better_is_not_worse() { + let mut better = full(); + better.tool_calls = Some(60); + better.unmapped = Some(4); + better.synthesis = Some(Some(Synthesis { + scenarios: 9, + refusals: 1, + })); + better.go_test = Some(Some(GoTest { + passed: 1, + failed: 0, + skipped: 11, + })); + assert!(worse(&full(), &full()).is_empty()); + assert!( + worse(&full(), &better).is_empty(), + "{:?}", + worse(&full(), &better) + ); + assert!(worse(&Measures::default(), &full()).is_empty()); + } + + #[test] + fn every_listed_regression_is_worse() { + let now = Measures { + tool_calls: Some(61), + validate: Some(Validate::NotRun), + synthesis: Some(Some(Synthesis { + scenarios: 12, + refusals: 3, + })), + unmapped: Some(0), + outputs: Some(["schema".to_owned()].into()), + go_test: Some(Some(GoTest { + passed: 9, + failed: 1, + skipped: 2, + })), + }; + let found = worse(&full(), &now); + assert_eq!(found.len(), 5, "{found:?}"); + for expected in [ + "tool calls: 40 → 61", + "validate: was valid, now NotRun", + "refusals 2 → 3", + "missing now: go", + "failures 0 → 1", + ] { + assert!( + found.iter().any(|line| line.contains(expected)), + "{expected}: {found:?}" + ); + } + let gone = Measures { + synthesis: Some(None), + go_test: Some(None), + ..full() + }; + let found = worse(&full(), &gone); + assert_eq!(found.len(), 2, "{found:?}"); + } + + #[test] + fn a_baseline_round_trips_and_keeps_other_trials() { + let sandbox = scratch("baseline"); + let path = sandbox.join("baseline.json"); + std::fs::write(&path, "{\"other\": {\"tool_calls\": 3}}\n").unwrap(); + write_baseline(&path, "demo", &full()).unwrap(); + let read = read_baseline(&path).unwrap(); + assert_eq!(read["demo"], full()); + assert_eq!(read["other"].tool_calls, Some(3)); + let text = std::fs::read_to_string(&path).unwrap(); + assert!(text.contains("\"validate\": \"valid\""), "{text}"); + let never = Measures { + synthesis: Some(None), + go_test: Some(None), + ..Measures::default() + }; + write_baseline(&path, "never", &never).unwrap(); + assert_eq!(read_baseline(&path).unwrap()["never"], never); + std::fs::remove_dir_all(&sandbox).unwrap(); + } +} diff --git a/crates/agentplugins-check/src/tools.rs b/crates/agentplugins-check/src/tools.rs new file mode 100644 index 0000000..685fff8 --- /dev/null +++ b/crates/agentplugins-check/src/tools.rs @@ -0,0 +1,403 @@ +//! `agentplugins-check tools` (network): the skills describe the CLIs as they are released now. +//! +//! Every plugin lives in this repository and names no CLI version (website/docs/structure.md R1, +//! R5), so the only thing that can drift is a product release renaming or removing a command the +//! skills still spell. This check downloads each product's newest release — the prebuilt archive, +//! checked against its `SHA256SUMS` — and runs ` --help` for every command a +//! code span or code block in that plugin spells. ESS's syntax example must also still validate. +//! It runs on every pull request, every `main` push and daily; a red run is fixed by a skill edit. +//! +//! It also fails when a CLI's newest release is newer than `verified.json`, the release its skills +//! were last verified against: every product release is re-verified (this check and an ESS trial +//! round) before `verified.json` moves. The offline gate only checks that file's shape. + +use std::collections::BTreeSet; +use std::path::{Path, PathBuf}; +use std::process::Command; + +/// Plugin, the CLI its skills drive, and the repository that releases it. +const TOOLS: &[(&str, &str, &str)] = &[ + ("aep", "aep", "beyond10x/aep"), + ("ess", "ess", "beyond10x/ess"), + ("worktree", "worktree", "beyond10x/worktree"), +]; + +/// Most subcommand words taken from one spelled command. +const DEPTH: usize = 4; + +/// The file naming, per CLI, the release the skills were last verified against. +pub const VERIFIED: &str = "verified.json"; + +/// A key for an `x.y.z` version or tag. +fn key(version: &str) -> Option<(u64, u64, u64)> { + let mut parts = version.trim_start_matches('v').split('.'); + let triple = ( + parts.next()?.parse().ok()?, + parts.next()?.parse().ok()?, + parts.next()?.parse().ok()?, + ); + parts.next().is_none().then_some(triple) +} + +/// Offline: `verified.json` names exactly the CLIs the skills drive, each at an `x.y.z` release. +pub fn verified(root: &Path) -> Result, String> { + let path = root.join(VERIFIED); + let text = std::fs::read_to_string(&path).map_err(|error| format!("{VERIFIED}: {error}"))?; + let map: std::collections::BTreeMap = serde_json::from_str(&text) + .map_err(|error| format!("{VERIFIED}: an object of CLI → `x.y.z` release: {error}"))?; + let expected: BTreeSet<&str> = TOOLS.iter().map(|(_, cli, _)| *cli).collect(); + let named: BTreeSet<&str> = map.keys().map(String::as_str).collect(); + if named != expected { + return Err(format!( + "{VERIFIED}: names {named:?}; it must name exactly {expected:?}" + )); + } + for (cli, release) in &map { + if key(release).is_none() { + return Err(format!( + "{VERIFIED}: `{cli}` is `{release}`, not an `x.y.z` release" + )); + } + } + Ok(map) +} + +/// The line for a CLI whose newest release is newer than the one its skills were verified against. +#[must_use] +pub fn unverified(cli: &str, newest: &str, verified: &str) -> Option { + (key(newest)? > key(verified)?).then(|| { + format!( + "{cli} {newest} is newer than {VERIFIED} ({verified}): re-verify the skills (`agentplugins-check tools` and an ESS trial round), then set `{cli}` to {newest} in {VERIFIED}" + ) + }) +} + +fn run(program: &str, arguments: &[&str]) -> Result { + let output = Command::new(program) + .args(arguments) + .output() + .map_err(|error| format!("running {program}: {error}"))?; + if output.status.success() { + Ok(String::from_utf8_lossy(&output.stdout).into_owned()) + } else { + Err(format!( + "{program} {} failed: {}", + arguments.join(" "), + String::from_utf8_lossy(&output.stderr).trim() + )) + } +} + +fn target() -> Result<&'static str, String> { + match (std::env::consts::ARCH, std::env::consts::OS) { + ("x86_64", "linux") => Ok("x86_64-unknown-linux-gnu"), + ("aarch64", "linux") => Ok("aarch64-unknown-linux-gnu"), + ("x86_64", "macos") => Ok("x86_64-apple-darwin"), + ("aarch64", "macos") => Ok("aarch64-apple-darwin"), + (arch, os) => Err(format!("no release archive for {arch}-{os}")), + } +} + +fn latest(repository: &str) -> Result { + let location = run( + "curl", + &[ + "-fsS", + "-o", + "/dev/null", + "-w", + "%{redirect_url}", + &format!("/{repository}/releases/latest"), + ], + )?; + location + .rsplit_once("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/releases/tag/") + .map(|(_, tag)| tag.trim().to_owned()) + .filter(|tag| !tag.is_empty()) + .ok_or_else(|| format!("{repository} has no release")) +} + +/// Download, verify and unpack a CLI's newest release; returns its tag and the binary. +fn fetch(cli: &str, repository: &str, scratch: &Path) -> Result<(String, PathBuf), String> { + let tag = latest(repository)?; + let target = target()?; + let version = tag.trim_start_matches('v'); + let archive = format!("{cli}-{version}-{target}.tar.gz"); + let base = format!("/{repository}/releases/download/{tag}"); + let dir = scratch.join(cli); + std::fs::create_dir_all(&dir).map_err(|error| error.to_string())?; + for file in [archive.as_str(), "SHA256SUMS"] { + run( + "curl", + &[ + "-fsSL", + "--retry", + "2", + "-o", + &dir.join(file).to_string_lossy(), + &format!("{base}/{file}"), + ], + )?; + } + let listed = + std::fs::read_to_string(dir.join("SHA256SUMS")).map_err(|error| error.to_string())?; + let line = listed + .lines() + .find(|line| line.ends_with(&archive)) + .ok_or_else(|| format!("{repository} {tag}: SHA256SUMS does not list {archive}"))?; + let expected = line.split_whitespace().next().unwrap_or_default(); + let actual = run("sha256sum", &[&dir.join(&archive).to_string_lossy()])?; + if !actual.starts_with(expected) { + return Err(format!( + "{repository} {tag}: {archive} does not match SHA256SUMS" + )); + } + run( + "tar", + &[ + "-xzf", + &dir.join(&archive).to_string_lossy(), + "-C", + &dir.to_string_lossy(), + ], + )?; + Ok(( + tag.clone(), + dir.join(format!("{cli}-{version}-{target}")).join(cli), + )) +} + +/// Subcommand words of every command spelled for `cli` in `text`: code spans and code blocks only, +/// words up to the first flag, placeholder or path, at most [`DEPTH`]. +#[must_use] +pub fn spelled(text: &str, cli: &str) -> BTreeSet> { + let mut code = Vec::new(); + let mut fenced = false; + for line in text.lines() { + if line.trim_start().starts_with("```") { + fenced = !fenced; + continue; + } + if fenced { + code.push(line.trim_start().trim_start_matches("$ ").to_owned()); + } else { + let mut parts = line.split('`'); + parts.next(); + while let Some(span) = parts.next() { + code.push(span.to_owned()); + parts.next(); + } + } + } + let mut commands = BTreeSet::new(); + for snippet in code { + let words: Vec<&str> = snippet.split_whitespace().collect(); + for (index, word) in words.iter().enumerate() { + let starts = + index == 0 || matches!(words[index - 1], "&&" | "||" | "|" | ";" | "then" | "do"); + if *word != cli || !starts { + continue; + } + let path: Vec = words[index + 1..] + .iter() + .take_while(|word| { + word.bytes() + .all(|b| b.is_ascii_lowercase() || b.is_ascii_digit() || b == b'-') + && !word.starts_with('-') + && word.bytes().next().is_some_and(|b| b.is_ascii_lowercase()) + }) + .take(DEPTH) + .map(|word| (*word).to_owned()) + .collect(); + if !path.is_empty() { + commands.insert(path); + } + } + } + commands +} + +fn markdown(root: &Path) -> Vec { + let mut files = Vec::new(); + let mut stack = vec![root.to_path_buf()]; + while let Some(dir) = stack.pop() { + for entry in std::fs::read_dir(&dir).into_iter().flatten().flatten() { + let path = entry.path(); + if path.is_dir() { + stack.push(path); + } else if path.extension().and_then(|e| e.to_str()) == Some("md") { + files.push(path); + } + } + } + files.sort(); + files +} + +/// Validate the three YAML blocks of ESS's syntax reference with the released `ess`. +fn syntax(root: &Path, ess: &Path, scratch: &Path) -> Result<(), String> { + let reference = root.join("plugins/ess/skills/specifying/references/syntax.md"); + let text = std::fs::read_to_string(&reference).map_err(|error| error.to_string())?; + let blocks: Vec<&str> = text + .split("```yaml\n") + .skip(1) + .filter_map(|rest| rest.split("```").next()) + .collect(); + let [system, components, domain] = blocks.as_slice() else { + return Err(format!( + "{}: expected three yaml blocks, found {}", + reference.display(), + blocks.len() + )); + }; + let dir = scratch.join("syntax"); + std::fs::create_dir_all(dir.join("domains")).map_err(|error| error.to_string())?; + std::fs::write(dir.join("system.yaml"), system).map_err(|error| error.to_string())?; + std::fs::write(dir.join("components.yaml"), components).map_err(|error| error.to_string())?; + std::fs::write(dir.join("domains/list.yaml"), domain).map_err(|error| error.to_string())?; + let out = run( + &ess.to_string_lossy(), + &["specify", "validate", "--path", &dir.to_string_lossy()], + ) + .map_err(|error| { + format!( + "{}: the example no longer validates: {error}", + reference.display() + ) + })?; + println!("tools `ess`: syntax example — {}", out.trim()); + let ess = ess.to_string_lossy(); + let path = dir.to_string_lossy(); + let suite = scratch.join("syntax-suite.json"); + let generated = scratch.join("syntax-generated"); + let generated = generated.to_string_lossy(); + let synthesized = run( + &ess, + &[ + "verify", + "conform", + "synthesize", + "--path", + &path, + "--out", + &suite.to_string_lossy(), + ], + ) + .map_err(|error| format!("{}: synthesize failed: {error}", reference.display()))?; + if !synthesized.contains(" 0 refusal(s)") { + return Err(format!( + "{}: the example synthesizes with refusals: {}", + reference.display(), + synthesized.trim() + )); + } + println!("tools `ess`: syntax example — {}", synthesized.trim()); + for arguments in [ + &["generate", "--kind", "schema"][..], + &["generate", "project", "openapi"][..], + ] { + let mut arguments = arguments.to_vec(); + arguments.extend(["--path", &path, "--out", &generated]); + run(&ess, &arguments).map_err(|error| { + format!( + "{}: `ess {}` failed on the example: {error}", + reference.display(), + arguments[..arguments.len() - 4].join(" ") + ) + })?; + } + Ok(()) +} + +/// Check every plugin's spelled commands against its CLI's newest release. +pub fn verify(root: &Path) -> Result<(), String> { + let scratch = + std::env::temp_dir().join(format!("agentplugins-check-tools-{}", std::process::id())); + let _ = std::fs::remove_dir_all(&scratch); + let result = (|| { + let verified = verified(root)?; + let mut problems = Vec::new(); + for (plugin, cli, repository) in TOOLS { + let (tag, binary) = fetch(cli, repository, &scratch)?; + if let Some(line) = verified + .get(*cli) + .and_then(|release| unverified(cli, &tag, release)) + { + problems.push(line); + } + let mut checked = 0; + for file in markdown(&root.join("plugins").join(plugin)) { + let text = std::fs::read_to_string(&file).map_err(|error| error.to_string())?; + for path in spelled(&text, cli) { + checked += 1; + let mut arguments: Vec<&str> = path.iter().map(String::as_str).collect(); + arguments.push("--help"); + let status = Command::new(&binary) + .args(&arguments) + .output() + .map_err(|error| error.to_string())?; + if !status.status.success() { + problems.push(format!( + "{}: `{cli} {}` is not a command of {cli} {tag}", + file.strip_prefix(root).unwrap_or(&file).display(), + path.join(" ") + )); + } + } + } + println!("tools `{cli}` {tag}: {checked} spelled command(s) checked"); + if *cli == "ess" { + syntax(root, &binary, &scratch)?; + } + } + if problems.is_empty() { + Ok(()) + } else { + Err(format!( + "{} problem(s) against the newest releases:\n {}", + problems.len(), + problems.join("\n ") + )) + } + })(); + let _ = std::fs::remove_dir_all(&scratch); + result +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn a_newer_release_than_verified_is_named_with_the_step() { + let line = unverified("ess", "0.32.2", "0.32.1").unwrap(); + assert!(line.starts_with("ess 0.32.2 is newer than verified.json (0.32.1)")); + assert!(line.contains("ESS trial round")); + assert_eq!(unverified("ess", "0.32.1", "0.32.1"), None); + assert_eq!(unverified("ess", "v0.32.0", "0.32.1"), None); + } + + #[test] + fn the_committed_verified_file_has_its_shape() { + let root = Path::new(env!("CARGO_MANIFEST_DIR")).join("../.."); + let map = verified(&root).unwrap(); + assert_eq!(map.len(), TOOLS.len()); + } + + #[test] + fn commands_are_read_from_code_only() { + let text = "Run `ess specify validate --path ` then prose ess binary.\n\n```console\n$ ess generate project openapi --out x\nb10x skill ess:init\n```\n`ess` alone, `ess specify|generate `\n"; + let found: Vec> = spelled(text, "ess").into_iter().collect(); + assert_eq!( + found, + [ + vec![ + "generate".to_owned(), + "project".to_owned(), + "openapi".to_owned() + ], + vec!["specify".to_owned(), "validate".to_owned()], + ] + ); + } +} diff --git a/crates/agentplugins-check/src/trial.rs b/crates/agentplugins-check/src/trial.rs new file mode 100644 index 0000000..24318ca --- /dev/null +++ b/crates/agentplugins-check/src/trial.rs @@ -0,0 +1,332 @@ +//! Whether a headless trial run tested what it claims to: the plugins in its sandbox, at the +//! version under test, and nothing from the machine it ran on. +//! +//! A trial run as a sub-agent of a working session inherits that session's agents, so its critics +//! were the host's and not the plugin's under test (trial 3, 2026-09-25). The run's own `init` +//! event lists what it loaded; this reads it and refuses anything that did not come from the +//! sandbox. +//! +//! A plugin the sandbox never installed is refused too, wherever it loaded from: Claude Code also +//! reads `.claude/settings*.json` in the directories above the working directory, and one left in +//! `~/.cache` by an earlier trial enabled `aep@b10x` in every sandbox below it (trial 3). + +use std::collections::{BTreeMap, BTreeSet}; +use std::path::{Path, PathBuf}; + +use serde_json::Value; + +/// The marketplace every plugin under test comes from. +const MARKETPLACE: &str = "b10x"; + +/// What the sandbox itself provides. +struct Sandbox<'a> { + /// The sandbox directory; its `home/` is the run's `HOME`. + root: &'a Path, + /// A directory marketplace plugins may load from in place (the checkout under test). + marketplace: Option<&'a Path>, + /// `name@marketplace` of every plugin the sandbox's own registry held when the run started, + /// with the versions it held. + installed: BTreeMap>, + /// The sandbox was seeded with older plugins on purpose (an upgrade trial): a plugin must be at + /// the version the sandbox installed, not at the version under test. + seeded: bool, +} + +/// Check every `init` event of a stream-json run; returns a one-line summary. +pub fn isolation( + run: &Path, + sandbox: &Path, + marketplace: Option<&Path>, + version: &str, + seeded: bool, +) -> Result { + let text = std::fs::read_to_string(run) + .map_err(|error| format!("reading {}: {error}", run.display()))?; + let root = std::fs::canonicalize(sandbox) + .map_err(|error| format!("sandbox {}: {error}", sandbox.display()))?; + let marketplace = marketplace + .map(std::fs::canonicalize) + .transpose() + .map_err(|error| format!("marketplace: {error}"))?; + // `task trial:run` copies the registry before the run: an upgrade changes it during the run. + let registry = [ + root.join("registry-at-start.json"), + root.join("home/.claude/plugins/installed_plugins.json"), + ] + .into_iter() + .find(|path| path.is_file()) + .ok_or_else(|| format!("{} has no plugin registry", root.display()))?; + let registry: Value = std::fs::read_to_string(®istry) + .ok() + .and_then(|text| serde_json::from_str(&text).ok()) + .ok_or_else(|| format!("reading {}", registry.display()))?; + let installed = registry["plugins"] + .as_object() + .map(|plugins| { + plugins + .iter() + .map(|(id, entries)| { + let versions = entries + .as_array() + .into_iter() + .flatten() + .filter_map(|entry| entry["version"].as_str().map(str::to_owned)) + .collect(); + (id.clone(), versions) + }) + .collect() + }) + .unwrap_or_default(); + check( + &text, + &Sandbox { + root: &root, + marketplace: marketplace.as_deref(), + installed, + seeded, + }, + version, + ) +} + +#[allow(clippy::too_many_lines)] +fn check(text: &str, sandbox: &Sandbox<'_>, version: &str) -> Result { + let root = sandbox.root; + let mut problems = Vec::new(); + let mut inits = 0; + let mut tested = BTreeSet::new(); + let mut foreign = BTreeSet::new(); + for (number, line) in text.lines().enumerate() { + let Ok(event) = serde_json::from_str::(line) else { + continue; + }; + if event["type"] != "system" || event["subtype"] != "init" { + continue; + } + inits += 1; + let at = format!("line {}", number + 1); + if let Some(cwd) = event["cwd"].as_str() { + if !Path::new(cwd).starts_with(root) { + problems.push(format!("{at}: ran in {cwd}, outside the sandbox")); + } + } + for server in event["mcp_servers"].as_array().into_iter().flatten() { + let name = server["name"].as_str().unwrap_or("?"); + problems.push(format!( + "{at}: MCP server `{name}` was connected; trials run with `--strict-mcp-config` and none" + )); + } + let mut loaded = BTreeSet::new(); + for plugin in event["plugins"].as_array().into_iter().flatten() { + let name = plugin["name"].as_str().unwrap_or_default(); + let path = plugin["path"].as_str().unwrap_or_default(); + let source = plugin["source"].as_str().unwrap_or_default(); + loaded.insert(name.to_owned()); + if path == "builtin" { + continue; + } + let path = PathBuf::from(path); + let in_place = sandbox + .marketplace + .is_some_and(|marketplace| path.starts_with(marketplace)); + if !path.starts_with(root.join("home")) && !in_place { + problems.push(format!( + "{at}: plugin `{source}` loaded from {}, outside the sandbox home and the marketplace under test", + path.display() + )); + } + if !sandbox.installed.contains_key(source) { + problems.push(format!( + "{at}: plugin `{source}` is not in the sandbox's installed_plugins.json; a settings file outside the sandbox enabled it" + )); + } + if source.ends_with(&format!("@{MARKETPLACE}")) { + tested.insert(name.to_owned()); + let loaded_version = plugin["version"].as_str().unwrap_or_default(); + let expected = if sandbox.seeded { + sandbox + .installed + .get(source) + .is_some_and(|versions| versions.contains(loaded_version)) + } else { + loaded_version == version + }; + if !expected { + problems.push(format!( + "{at}: plugin `{source}` is {loaded_version}, not {}", + if sandbox.seeded { + "a version the sandbox was seeded with".to_owned() + } else { + format!("the version under test {version}") + } + )); + } + } else { + problems.push(format!( + "{at}: plugin `{source}` is not from the `{MARKETPLACE}` marketplace" + )); + } + } + for list in ["agents", "skills"] { + for id in event[list].as_array().into_iter().flatten() { + let id = id.as_str().unwrap_or_default(); + if let Some((owner, _)) = id.split_once(':') { + if !loaded.contains(owner) { + foreign.insert(id.to_owned()); + } + } + } + } + } + // Skills and agents offered from outside the sandbox (claude.ai account skills reach sub-agent + // sessions, trial 3) only matter when the run used one. + for (number, line) in text.lines().enumerate() { + let Ok(event) = serde_json::from_str::(line) else { + continue; + }; + for block in event["message"]["content"].as_array().into_iter().flatten() { + if block["type"] != "tool_use" { + continue; + } + let input = &block["input"]; + let used = match block["name"].as_str() { + Some("Skill") => input["skill"].as_str().or(input["command"].as_str()), + Some("Agent" | "Task") => input["subagent_type"].as_str(), + _ => None, + }; + if let Some(used) = used.filter(|used| foreign.contains(*used)) { + problems.push(format!( + "line {}: the run used `{used}`, which belongs to no plugin the sandbox loaded", + number + 1 + )); + } + } + } + if inits == 0 { + problems.push( + "no `init` event: not a stream-json run (`--output-format stream-json --verbose`)" + .to_owned(), + ); + } else if tested.is_empty() { + problems.push(format!("no `@{MARKETPLACE}` plugin was loaded")); + } + if problems.is_empty() { + let tested: Vec = tested.into_iter().collect(); + let unused = if foreign.is_empty() { + String::new() + } else { + format!( + "; {} skill(s)/agent(s) from outside the sandbox were offered and not used", + foreign.len() + ) + }; + Ok(format!( + "isolated: {inits} session(s), plugins {} at {version}, all from {}{unused}", + tested.join(", "), + root.display() + )) + } else { + Err(format!( + "the run is not an isolated trial of {version}:\n {}", + problems.join("\n ") + )) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn init(plugins: &str, agents: &str) -> String { + format!( + r#"{{"type":"system","subtype":"init","cwd":"/opt/trial/work","plugins":[{plugins}],"agents":[{agents}],"skills":["ess:specifying","verify"]}}"# + ) + } + + fn sandbox() -> Sandbox<'static> { + Sandbox { + root: Path::new("/opt/trial"), + marketplace: Some(Path::new("/opt/checkout")), + installed: [("ess@b10x".to_owned(), ["0.14.3".to_owned()].into())].into(), + seeded: false, + } + } + + #[test] + fn a_seeded_upgrade_trial_starts_from_the_seeded_version() { + let old = r#"{"name":"ess","path":"/opt/trial/home/.claude/plugins/cache/b10x/ess/0.12.0","source":"ess@b10x","version":"0.12.0"}"#; + let mut seeded = sandbox(); + seeded.installed = [("ess@b10x".to_owned(), ["0.12.0".to_owned()].into())].into(); + seeded.seeded = true; + check(&init(old, ""), &seeded, "0.14.3").unwrap(); + seeded.seeded = false; + let error = check(&init(old, ""), &seeded, "0.14.3").unwrap_err(); + assert!(error.contains("not the version under test"), "{error}"); + } + + #[test] + fn a_plugin_the_sandbox_never_installed_fails_the_run() { + let leaked = r#"{"name":"aep","path":"/opt/checkout/plugins/aep","source":"aep@b10x","version":"0.14.3"}"#; + let error = check(&init(leaked, ""), &sandbox(), "0.14.3").unwrap_err(); + assert!( + error.contains("not in the sandbox's installed_plugins.json"), + "{error}" + ); + assert!(!error.contains("outside the sandbox home"), "{error}"); + } + + const SANDBOXED: &str = r#"{"name":"ess","path":"/opt/trial/home/.claude/plugins/cache/b10x/ess/0.14.3","source":"ess@b10x","version":"0.14.3"},{"name":"telemetry","path":"builtin","source":"telemetry@builtin"}"#; + + #[test] + fn a_sandboxed_run_passes() { + let run = init(SANDBOXED, r#""claude","ess:author""#); + let summary = check(&run, &sandbox(), "0.14.3").unwrap(); + assert!(summary.contains("plugins ess at 0.14.3"), "{summary}"); + } + + #[test] + fn a_host_agent_offered_and_unused_is_a_note() { + let run = init(SANDBOXED, r#""host-plugin:plan-critic-design""#); + let summary = check(&run, &sandbox(), "0.14.3").unwrap(); + assert!( + summary.contains("1 skill(s)/agent(s) from outside"), + "{summary}" + ); + } + + #[test] + fn a_host_agent_used_fails_the_run() { + let run = init(SANDBOXED, r#""host-plugin:plan-critic-design""#) + + "\n" + + r#"{"type":"assistant","message":{"content":[{"type":"tool_use","name":"Agent","input":{"subagent_type":"host-plugin:plan-critic-design"}}]}}"#; + let error = check(&run, &sandbox(), "0.14.3").unwrap_err(); + assert!( + error.contains("used `host-plugin:plan-critic-design`"), + "{error}" + ); + } + + #[test] + fn another_version_or_a_host_path_fails_the_run() { + let host = r#"{"name":"ess","path":"/opt/host/.claude/plugins/cache/b10x/ess/0.14.2","source":"ess@b10x","version":"0.14.2"}"#; + let error = check(&init(host, ""), &sandbox(), "0.14.3").unwrap_err(); + assert!(error.contains("outside the sandbox home"), "{error}"); + assert!(error.contains("not the version under test"), "{error}"); + } + + #[test] + fn an_account_mcp_server_fails_the_run() { + let run = init(SANDBOXED, "").replace( + "\"agents\"", + "\"mcp_servers\":[{\"name\":\"claude.ai Docs\",\"status\":\"connected\"}],\"agents\"", + ); + let error = check(&run, &sandbox(), "0.14.3").unwrap_err(); + assert!(error.contains("MCP server `claude.ai Docs`"), "{error}"); + } + + #[test] + fn a_stream_without_init_fails() { + let error = check("{\"type\":\"result\"}", &sandbox(), "0.14.3").unwrap_err(); + assert!(error.contains("no `init` event"), "{error}"); + } +} diff --git a/crates/agentplugins-check/src/trials.rs b/crates/agentplugins-check/src/trials.rs new file mode 100644 index 0000000..eb77ce2 --- /dev/null +++ b/crates/agentplugins-check/src/trials.rs @@ -0,0 +1,486 @@ +//! Trial definitions: `trials//trial.yaml`, one per trial, with an optional fixture beside +//! it, and `trials/baseline.json`, the measures of the last accepted run of each. +//! +//! A trial prompt that lives in a throwaway script cannot be re-run next round or compared with the +//! last one. Here each is data the gate validates and `task trial:run TRIAL=` executes: +//! [`prepare`] copies the fixture into a fresh sandbox, commits it, and writes the prompt and the +//! setup the Taskfile runs. + +use std::collections::BTreeMap; +use std::path::{Path, PathBuf}; +use std::process::Command; + +use serde::Deserialize; + +/// The directory holding every trial, relative to the repository root. +pub const TRIALS: &str = "trials"; + +/// The committed baseline, relative to the repository root. +pub const BASELINE: &str = "trials/baseline.json"; + +/// What a trial exercises. A label for readers and for the round's coverage; it does not change +/// how a run is measured, which [`Definition::measures`] decides. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Deserialize)] +#[serde(rename_all = "kebab-case")] +pub enum Kind { + /// A new ESS specification from a sentence. + EssNew, + /// An ESS specification of an existing service (the fixture). + EssRetrofit, + /// Generation and a synthesized suite from an existing specification. + EssPipeline, + /// Every ESS output, and an implementation held to the synthesized Go suite. + EssFullPackage, + /// AEP planning from an existing backlog. + AepBacklog, + /// Setting up `worktree` in a repository. + WorktreeOnboarding, + /// Bringing a seeded older install current. + Upgrade, +} + +/// A number `agentplugins-check trial-report` reads from a run. +#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum Measure { + /// Tool calls the run made. + ToolCalls, + /// Whether `ess specify validate` ran and its last output said `valid`. + Validate, + /// Scenarios and refusals from the last `ess verify conform synthesize`. + Synthesis, + /// `UNMAPPED:` markers in the spec files the run wrote, read from disk. + Unmapped, + /// Which of [`Definition::outputs`] exist on disk. + Outputs, + /// Passed, failed and skipped tests of the last `go test`. + GoTest, +} + +/// One `trials//trial.yaml`. +#[derive(Debug, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct Definition { + /// The trial's name; equal to its directory name and the sandbox name. + pub name: String, + /// What it exercises. + pub kind: Kind, + /// What the user types. + pub prompt: String, + /// The subdirectory of the sandbox's `work/` the run starts in and the fixture is copied to. + #[serde(default)] + pub dir: Option, + /// A directory beside `trial.yaml` copied into the run's directory and committed there. + #[serde(default)] + pub fixture: Option, + /// Give the fixture a bare `origin` inside the sandbox, with `main` pushed. + #[serde(default)] + pub remote: bool, + /// The sandbox is seeded with older plugins on purpose (an upgrade trial). + #[serde(default)] + pub seeded: bool, + /// Shell lines run from the sandbox root with the sandbox's `env`, before the run: installing + /// the plugins under test, or seeding an older install. + #[serde(default)] + pub setup: Vec, + /// The numbers the run must report. + pub measures: Vec, + /// Output name → path relative to the run's directory, for [`Measure::Outputs`]. + #[serde(default)] + pub outputs: BTreeMap, +} + +impl Definition { + /// The directory the run starts in, relative to the sandbox. + #[must_use] + pub fn workdir(&self) -> PathBuf { + let work = PathBuf::from("work"); + match self.dir.as_deref() { + Some(dir) if !dir.is_empty() => work.join(dir), + _ => work, + } + } + + /// Whether the run reports `measure`. + #[must_use] + pub fn measures(&self, measure: Measure) -> bool { + self.measures.contains(&measure) + } +} + +/// A relative path that stays below the directory it is joined to. +fn contained(path: &str) -> bool { + let path = Path::new(path); + !path.as_os_str().is_empty() + && path + .components() + .all(|component| matches!(component, std::path::Component::Normal(_))) +} + +fn read(root: &Path, name: &str) -> Result { + let path = root.join(TRIALS).join(name).join("trial.yaml"); + let text = std::fs::read_to_string(&path) + .map_err(|error| format!("reading {}: {error}", path.display()))?; + serde_yaml::from_str(&text).map_err(|error| format!("parsing {}: {error}", path.display())) +} + +/// Every trial name under `trials/`, sorted. +pub fn names(root: &Path) -> Result, String> { + let directory = root.join(TRIALS); + let mut names = Vec::new(); + for entry in std::fs::read_dir(&directory) + .map_err(|error| format!("reading {}: {error}", directory.display()))? + { + let path = entry.map_err(|error| error.to_string())?.path(); + if path.is_dir() { + if let Some(name) = path.file_name().and_then(|name| name.to_str()) { + names.push(name.to_owned()); + } + } + } + names.sort(); + Ok(names) +} + +/// One trial's definition, checked. +pub fn load(root: &Path, name: &str) -> Result { + let known = names(root)?; + if !known.iter().any(|known| known == name) { + return Err(format!( + "no trial `{name}` under {TRIALS}/; the trials are: {}", + known.join(", ") + )); + } + let definition = read(root, name)?; + let problems = problems(root, &definition); + if problems.is_empty() { + Ok(definition) + } else { + Err(format!( + "{TRIALS}/{name}/trial.yaml:\n {}", + problems.join("\n ") + )) + } +} + +fn problems(root: &Path, definition: &Definition) -> Vec { + let mut problems = Vec::new(); + let directory = root.join(TRIALS).join(&definition.name); + if !directory.is_dir() { + problems.push(format!( + "`name: {}` is not the directory the definition sits in", + definition.name + )); + } + if definition.prompt.trim().is_empty() { + problems.push("the prompt is empty".to_owned()); + } + if let Some(dir) = definition.dir.as_deref() { + if !contained(dir) { + problems.push(format!("`dir: {dir}` is not a relative path below `work/`")); + } + } + if let Some(fixture) = definition.fixture.as_deref() { + if !contained(fixture) || !directory.join(fixture).is_dir() { + problems.push(format!( + "`fixture: {fixture}` is not a directory beside trial.yaml" + )); + } else if directory.join(fixture).join(".git").exists() { + problems.push(format!( + "`fixture: {fixture}` carries a `.git`; the sandbox commits it" + )); + } + } + if definition.remote && definition.fixture.is_none() { + problems.push("`remote: true` needs a fixture to push".to_owned()); + } + if definition.seeded && definition.setup.is_empty() { + problems.push("`seeded: true` needs the `setup` that seeds the install".to_owned()); + } + if !definition.measures(Measure::ToolCalls) { + problems.push("every trial measures `tool_calls`".to_owned()); + } + if definition.measures(Measure::Outputs) == definition.outputs.is_empty() { + problems.push("`outputs` are listed exactly when `measures` has `outputs`".to_owned()); + } + for (output, path) in &definition.outputs { + if !contained(path) { + problems.push(format!( + "output `{output}: {path}` is not a relative path below the run's directory" + )); + } + } + problems +} + +/// Every definition parses and holds together, and the baseline names only trials that exist. +pub fn check(root: &Path) -> Result<(), String> { + let mut found = Vec::new(); + let names = names(root)?; + for name in &names { + match read(root, name) { + Ok(definition) => { + for problem in problems(root, &definition) { + found.push(format!("{TRIALS}/{name}/trial.yaml: {problem}")); + } + } + Err(error) => found.push(error), + } + } + let baseline = root.join(BASELINE); + let text = std::fs::read_to_string(&baseline) + .map_err(|error| format!("reading {BASELINE}: {error}"))?; + match serde_json::from_str::>(&text) { + Ok(entries) => { + for trial in entries.keys() { + if !names.contains(trial) { + found.push(format!("{BASELINE} records `{trial}`, which is no trial")); + } + } + } + Err(error) => found.push(format!("parsing {BASELINE}: {error}")), + } + if found.is_empty() { + Ok(()) + } else { + Err(format!( + "{} trial definition problem(s):\n {}", + found.len(), + found.join("\n ") + )) + } +} + +fn copy_tree(from: &Path, to: &Path) -> Result<(), String> { + std::fs::create_dir_all(to).map_err(|error| format!("creating {}: {error}", to.display()))?; + for entry in + std::fs::read_dir(from).map_err(|error| format!("reading {}: {error}", from.display()))? + { + let path = entry.map_err(|error| error.to_string())?.path(); + let target = to.join(path.file_name().unwrap_or_default()); + if path.is_dir() { + copy_tree(&path, &target)?; + } else { + std::fs::copy(&path, &target) + .map_err(|error| format!("copying {}: {error}", path.display()))?; + } + } + Ok(()) +} + +/// Run `git` in `directory` as the trial identity, reading no configuration of the operator's. +fn git(sandbox: &Path, directory: &Path, args: &[&str]) -> Result<(), String> { + let output = Command::new("git") + .args([ + "-c", + "commit.gpgsign=false", + "-c", + "init.defaultBranch=main", + ]) + .args(args) + .current_dir(directory) + .env("HOME", sandbox.join("home")) + .env("GIT_CONFIG_NOSYSTEM", "1") + .env("GIT_CONFIG_GLOBAL", "/dev/null") + .env("GIT_AUTHOR_NAME", "trial") + .env("GIT_AUTHOR_EMAIL", "trial@example.invalid") + .env("GIT_COMMITTER_NAME", "trial") + .env("GIT_COMMITTER_EMAIL", "trial@example.invalid") + .output() + .map_err(|error| format!("running git: {error}"))?; + if output.status.success() { + Ok(()) + } else { + Err(format!( + "git {} in {}: {}", + args.join(" "), + directory.display(), + String::from_utf8_lossy(&output.stderr).trim() + )) + } +} + +/// Prepare a fresh sandbox (`task trial:sandbox`) for `name`: the fixture copied into the run's +/// directory and committed, and beside `env` the files `task trial:run` reads — `prompt.txt`, +/// `workdir`, `setup.sh` and, for an upgrade trial, `seeded`. +pub fn prepare(root: &Path, name: &str, sandbox: &Path) -> Result { + let definition = load(root, name)?; + if !sandbox.join("env").is_file() { + return Err(format!( + "{} is not a trial sandbox (no `env`); create it with `task trial:sandbox`", + sandbox.display() + )); + } + let work = sandbox.join(definition.workdir()); + std::fs::create_dir_all(&work) + .map_err(|error| format!("creating {}: {error}", work.display()))?; + let mut done = Vec::new(); + if let Some(fixture) = definition.fixture.as_deref() { + copy_tree(&root.join(TRIALS).join(name).join(fixture), &work)?; + git(sandbox, &work, &["init", "--quiet"])?; + git(sandbox, &work, &["add", "--all"])?; + git( + sandbox, + &work, + &["commit", "--quiet", "-m", "Initial import"], + )?; + done.push(format!("fixture committed in {}", work.display())); + if definition.remote { + let remote = sandbox.join("remote.git"); + git( + sandbox, + sandbox, + &["init", "--quiet", "--bare", &remote.to_string_lossy()], + )?; + git( + sandbox, + &work, + &["remote", "add", "origin", &remote.to_string_lossy()], + )?; + git(sandbox, &work, &["push", "--quiet", "-u", "origin", "main"])?; + done.push(format!("origin {}", remote.display())); + } + } + let write = |file: &str, text: &str| { + std::fs::write(sandbox.join(file), text).map_err(|error| format!("writing {file}: {error}")) + }; + write("prompt.txt", definition.prompt.trim_end())?; + write("workdir", &definition.dir.clone().unwrap_or_default())?; + if definition.setup.is_empty() { + let _ = std::fs::remove_file(sandbox.join("setup.sh")); + } else { + write( + "setup.sh", + &format!("set -e\n{}\n", definition.setup.join("\n")), + )?; + done.push(format!("{} setup line(s)", definition.setup.len())); + } + if definition.seeded { + write("seeded", "")?; + } else { + let _ = std::fs::remove_file(sandbox.join("seeded")); + } + Ok(format!( + "prepared trial `{name}` ({:?}) in {}{}", + definition.kind, + sandbox.display(), + if done.is_empty() { + String::new() + } else { + format!(": {}", done.join("; ")) + } + )) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn scratch(label: &str) -> PathBuf { + let path = std::env::temp_dir().join(format!( + "agentplugins-check-trials-{label}-{}", + std::process::id() + )); + let _ = std::fs::remove_dir_all(&path); + std::fs::create_dir_all(&path).expect("the scratch directory is writable"); + path + } + + fn repository() -> PathBuf { + Path::new(env!("CARGO_MANIFEST_DIR")) + .parent() + .and_then(Path::parent) + .expect("checker is under the repository root") + .to_path_buf() + } + + #[test] + fn the_committed_trials_hold_together() { + check(&repository()).expect("every committed trial validates"); + let names = names(&repository()).expect("trials/ is readable"); + for expected in [ + "ess-new", + "ess-retrofit", + "ess-pipeline", + "ess-full-package", + "aep-backlog", + "worktree-onboarding", + "upgrade-seeded", + ] { + assert!(names.iter().any(|name| name == expected), "{expected}"); + } + } + + #[test] + fn a_definition_that_does_not_hold_together_is_refused() { + let root = scratch("bad"); + let trial = root.join(TRIALS).join("broken"); + std::fs::create_dir_all(&trial).unwrap(); + std::fs::write( + trial.join("trial.yaml"), + "name: other\nkind: ess-new\nprompt: ' '\ndir: ../escape\nseeded: true\nmeasures: [outputs]\n", + ) + .unwrap(); + std::fs::write(root.join(BASELINE), "{\"gone\": {}}").unwrap(); + let error = check(&root).unwrap_err(); + std::fs::remove_dir_all(&root).unwrap(); + for expected in [ + "not the directory", + "prompt is empty", + "not a relative path below `work/`", + "needs the `setup`", + "every trial measures `tool_calls`", + "`outputs` are listed exactly", + "records `gone`, which is no trial", + ] { + assert!(error.contains(expected), "{expected}: {error}"); + } + } + + #[test] + fn prepare_commits_the_fixture_and_writes_the_run_files() { + let root = scratch("prepare"); + let trial = root.join(TRIALS).join("demo"); + std::fs::create_dir_all(trial.join("fixture/src")).unwrap(); + std::fs::write(trial.join("fixture/src/main.go"), "package main\n").unwrap(); + std::fs::write( + trial.join("trial.yaml"), + "name: demo\nkind: ess-retrofit\nprompt: |\n Describe this service.\ndir: svc\nfixture: fixture\nremote: true\nsetup: [echo one]\nmeasures: [tool_calls, unmapped]\n", + ) + .unwrap(); + std::fs::write(root.join(BASELINE), "{}").unwrap(); + let sandbox = root.join("sandbox"); + std::fs::create_dir_all(sandbox.join("home")).unwrap(); + std::fs::write(sandbox.join("env"), "").unwrap(); + + let summary = prepare(&root, "demo", &sandbox).unwrap(); + assert!(summary.contains("fixture committed"), "{summary}"); + let work = sandbox.join("work/svc"); + assert!(work.join("src/main.go").is_file()); + let log = Command::new("git") + .args(["log", "--format=%an %s", "origin/main"]) + .current_dir(&work) + .output() + .unwrap(); + assert_eq!( + String::from_utf8_lossy(&log.stdout), + "trial Initial import\n" + ); + assert_eq!( + std::fs::read_to_string(sandbox.join("prompt.txt")).unwrap(), + "Describe this service." + ); + assert_eq!( + std::fs::read_to_string(sandbox.join("workdir")).unwrap(), + "svc" + ); + assert_eq!( + std::fs::read_to_string(sandbox.join("setup.sh")).unwrap(), + "set -e\necho one\n" + ); + assert!(!sandbox.join("seeded").exists()); + let unknown = prepare(&root, "missing", &sandbox).unwrap_err(); + std::fs::remove_dir_all(&root).unwrap(); + assert!(unknown.contains("the trials are: demo"), "{unknown}"); + } +} diff --git a/crates/agentplugins-check/tests/the_rename_acceptance_statement.rs b/crates/agentplugins-check/tests/the_rename_acceptance_statement.rs index 6fc979b..a54fd70 100644 --- a/crates/agentplugins-check/tests/the_rename_acceptance_statement.rs +++ b/crates/agentplugins-check/tests/the_rename_acceptance_statement.rs @@ -64,6 +64,8 @@ const EXEMPT_FILES: &[&str] = &[ "CHANGELOG.md", // Where the sweep's own `RETIRED` table lives; see the header. "crates/agentplugins-check/src/main.rs", + // The migration map `b10x setup` reads: it must spell every retired name to find old installs. + "catalog.json", ]; /// Exempt in full, by repository-relative prefix. @@ -72,6 +74,8 @@ const EXEMPT_PREFIXES: &[&str] = &[ "changes/", // The planning store, whose only writer is the `aep` CLI. ".engineering/", + // The setup binary and its recorded host output: it replaces installs made under the old names. + "crates/b10x/", ]; /// The marker a row carries when its spelling is what the transcript beside it contains. @@ -255,7 +259,7 @@ fn the_walk_finds_each_violation_the_statement_names() { // document. write("website/docs/install.md", &format!("install `{retired}`\n")); write( - &format!("plugins/{three_letters}/skills/wave/SKILL.md"), + &format!("plugins/{three_letters}/skills/implementing/SKILL.md"), "---\nname: wave\n---\n\nnothing here spells it\n", ); write(&format!("website/docs/plugins/{retired}.md"), "# a page\n"); @@ -288,7 +292,7 @@ fn the_walk_finds_each_violation_the_statement_names() { let mut expected = vec![ format!("README.md:1: install: {retired} # {RECORDED_SPELLING}"), format!("plugins/{three_letters}: filed under a retired plugin name"), - format!("plugins/{three_letters}/skills/wave/SKILL.md: filed under a retired plugin name"), + format!("plugins/{three_letters}/skills/implementing/SKILL.md: filed under a retired plugin name"), format!("website/docs/install.md:1: install `{retired}`"), format!("website/docs/plugins/{retired}.md: filed under a retired plugin name"), ]; diff --git a/crates/b10x/Cargo.toml b/crates/b10x/Cargo.toml new file mode 100644 index 0000000..23f8f7d --- /dev/null +++ b/crates/b10x/Cargo.toml @@ -0,0 +1,24 @@ +[package] +name = "b10x" +description = "Installs, migrates and checks the Beyond10x agent plugins and the binaries they drive." +version.workspace = true +edition.workspace = true +rust-version.workspace = true +license.workspace = true +repository.workspace = true +authors.workspace = true +publish.workspace = true + +[[bin]] +name = "b10x" +path = "src/main.rs" + +[dependencies] +clap.workspace = true +serde.workspace = true +serde_json.workspace = true +sha2.workspace = true +toml.workspace = true + +[lints] +workspace = true diff --git a/crates/b10x/src/apply.rs b/crates/b10x/src/apply.rs new file mode 100644 index 0000000..696ec32 --- /dev/null +++ b/crates/b10x/src/apply.rs @@ -0,0 +1,207 @@ +//! Applying a plan: refuse a stale one, snapshot what can be restored, run each action in order, +//! stop at the first failure, then read the state again and say whether it converged. + +use std::path::{Path, PathBuf}; +use std::process::Command; +use std::time::{SystemTime, UNIX_EPOCH}; + +use serde::{Deserialize, Serialize}; + +use crate::inventory::{home, Inventory}; +use crate::plan::{digest, Action, Plan}; + +/// What a snapshot holds: original path → copy inside the snapshot directory. +#[derive(Debug, Default, Deserialize, Serialize)] +pub struct Manifest { + /// Files that existed, and their copies. + pub files: Vec<(String, String)>, + /// Files that did not exist and are removed on undo. + pub absent: Vec, +} + +/// Where snapshots live. +#[must_use] +pub fn snapshots() -> PathBuf { + home().join(".local/state/b10x/setup") +} + +fn touched(plan: &Plan) -> Vec { + let home = home(); + let mut files = vec![ + home.join(".claude/settings.json"), + home.join(".claude/plugins/known_marketplaces.json"), + home.join(".claude/plugins/installed_plugins.json"), + home.join(".codex/config.toml"), + ]; + for action in &plan.actions { + match action { + Action::RemoveSetting { file, .. } | Action::Unpin { file, .. } => { + files.push(PathBuf::from(file)); + } + Action::InstallBinary { + name, directory, .. + } => files.push(Path::new(directory).join(name)), + Action::Command { cwd: Some(cwd), .. } => { + files.push(Path::new(cwd).join(".claude/settings.json")); + files.push(Path::new(cwd).join(".claude/settings.local.json")); + } + Action::Command { .. } | Action::Refresh { .. } => {} + } + } + files.sort(); + files.dedup(); + files +} + +/// Copy every file the plan can change into a new snapshot directory. +pub fn snapshot(plan: &Plan) -> Result { + let stamp = SystemTime::now() + .duration_since(UNIX_EPOCH) + .map_or(0, |elapsed| elapsed.as_secs()); + let directory = snapshots().join(stamp.to_string()); + std::fs::create_dir_all(&directory) + .map_err(|error| format!("{}: {error}", directory.display()))?; + let mut manifest = Manifest::default(); + for (index, file) in touched(plan).into_iter().enumerate() { + if file.is_file() { + let copy = format!("{index:03}"); + std::fs::copy(&file, directory.join(©)) + .map_err(|error| format!("{}: {error}", file.display()))?; + manifest + .files + .push((file.to_string_lossy().into_owned(), copy)); + } else { + manifest.absent.push(file.to_string_lossy().into_owned()); + } + } + let text = serde_json::to_string_pretty(&manifest).map_err(|error| error.to_string())?; + std::fs::write(directory.join("manifest.json"), text).map_err(|error| error.to_string())?; + Ok(directory) +} + +/// Restore a snapshot. +pub fn undo(directory: &Path) -> Result, String> { + let text = std::fs::read_to_string(directory.join("manifest.json")) + .map_err(|error| format!("{}: {error}", directory.display()))?; + let manifest: Manifest = serde_json::from_str(&text).map_err(|error| error.to_string())?; + let mut restored = Vec::new(); + for (original, copy) in &manifest.files { + std::fs::copy(directory.join(copy), original) + .map_err(|error| format!("{original}: {error}"))?; + restored.push(original.clone()); + } + for absent in &manifest.absent { + if Path::new(absent).exists() { + std::fs::remove_file(absent).map_err(|error| format!("{absent}: {error}"))?; + restored.push(format!("{absent} (removed)")); + } + } + Ok(restored) +} + +fn edit_json(file: &str, edit: impl FnOnce(&mut serde_json::Value)) -> Result<(), String> { + let text = std::fs::read_to_string(file).unwrap_or_else(|_| "{}".to_owned()); + let mut value: serde_json::Value = + serde_json::from_str(&text).map_err(|error| format!("{file}: {error}"))?; + edit(&mut value); + let mut out = serde_json::to_string_pretty(&value).map_err(|error| error.to_string())?; + out.push('\n'); + let incoming = format!("{file}.b10x-new"); + std::fs::write(&incoming, out).map_err(|error| format!("{incoming}: {error}"))?; + std::fs::rename(&incoming, file).map_err(|error| format!("{file}: {error}")) +} + +/// Run one action. +pub fn run(action: &Action) -> Result<(), String> { + match action { + Action::Refresh { argv, .. } + | Action::Command { + argv, cwd: None, .. + } => command(argv, None), + Action::Command { + argv, + cwd: Some(cwd), + .. + } => command(argv, Some(cwd)), + Action::RemoveSetting { file, key, .. } => edit_json(file, |value| { + if let Some(enabled) = value + .get_mut("enabledPlugins") + .and_then(serde_json::Value::as_object_mut) + { + enabled.remove(key); + } + }), + Action::Unpin { + file, + name, + repository, + .. + } => edit_json(file, |value| { + if !value.is_object() { + *value = serde_json::json!({}); + } + let markets = value + .as_object_mut() + .map(|object| { + object + .entry("extraKnownMarketplaces") + .or_insert_with(|| serde_json::json!({})) + }) + .expect("an object"); + if let Some(markets) = markets.as_object_mut() { + markets.insert( + name.clone(), + serde_json::json!({"source": {"source": "github", "repo": repository}}), + ); + } + }), + Action::InstallBinary { + name, + tag, + method, + install, + directory, + .. + } => crate::install::install(name, tag, *method, install, Path::new(directory)).map(|_| ()), + } +} + +fn command(argv: &[String], cwd: Option<&str>) -> Result<(), String> { + let (program, arguments) = argv.split_first().ok_or("empty command")?; + let mut command = Command::new(program); + command.args(arguments); + if let Some(cwd) = cwd { + command.current_dir(cwd); + } + let output = command + .output() + .map_err(|error| format!("{program}: {error}"))?; + if output.status.success() { + Ok(()) + } else { + let stderr = String::from_utf8_lossy(&output.stderr); + let stdout = String::from_utf8_lossy(&output.stdout); + Err(format!( + "`{}` exited {}: {}", + argv.join(" "), + output.status.code().unwrap_or(-1), + if stderr.trim().is_empty() { + stdout.trim() + } else { + stderr.trim() + } + )) + } +} + +/// Refuse a plan made from another inventory. +pub fn fresh(plan: &Plan, inventory: &Inventory) -> Result<(), String> { + if digest(inventory) == plan.inventory_digest { + Ok(()) + } else { + Err( + "the installed state changed since this plan was made; run `b10x setup plan` again" + .to_owned(), + ) + } +} diff --git a/crates/b10x/src/catalog.rs b/crates/b10x/src/catalog.rs new file mode 100644 index 0000000..52c489d --- /dev/null +++ b/crates/b10x/src/catalog.rs @@ -0,0 +1,253 @@ +//! The catalog: which products exist, which plugins and binaries make each one, and which names +//! are retired. It carries no version of anything it points at — versions are read at run time from +//! the host, the marketplace and the product's newest release. + +use std::collections::BTreeMap; + +use serde::{Deserialize, Serialize}; + +/// The catalog this binary was built with, used when no marketplace copy can be read. +pub const EMBEDDED: &str = include_str!("../../../catalog.json"); + +/// The only catalog format this binary reads. +pub const FORMAT: &str = "b10x.catalog/2"; + +/// The whole catalog. +#[derive(Debug, Clone, Deserialize, Serialize)] +pub struct Catalog { + /// Always [`FORMAT`]. + pub format: String, + /// The marketplace every plugin installs from. + pub marketplace: Marketplace, + /// Plugins installed whatever the user selects. + pub base: Vec, + /// Products the user chooses between. + pub products: Vec, + /// Retired plugin name → the plugin that replaces it. + pub retired_plugins: BTreeMap, + /// Marketplace names that no longer serve anything. + pub retired_marketplaces: Vec, +} + +/// The marketplace identity and its GitHub repository. +#[derive(Debug, Clone, Deserialize, Serialize)] +pub struct Marketplace { + /// `b10x`. + pub name: String, + /// `owner/repo`. + pub repository: String, +} + +/// One product: a set of plugins and the binaries they drive. +#[derive(Debug, Clone, Deserialize, Serialize)] +pub struct Product { + /// What the user selects, e.g. `ess`. + pub id: String, + /// Offered, never preselected. + #[serde(default)] + pub optional: bool, + /// What the user wants to do, as the onboarding question offers it. + pub intent: String, + /// One line for the selection prompt. + pub summary: String, + /// Plugin names in the marketplace. + pub plugins: Vec, + /// Binaries the plugins run. + pub binaries: Vec, +} + +/// A binary a product's plugins run. Every binary is bound to its repository's newest release. +#[derive(Debug, Clone, Deserialize, Serialize)] +pub struct Binary { + /// Executable name on `PATH`. + pub name: String, + /// Reported but never installed unprompted. + #[serde(default)] + pub optional: bool, + /// Operating systems it runs on (`linux`, `macos`); empty means every one. + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub platforms: Vec, + /// The ways it can be installed. + pub install: Install, +} + +impl Binary { + /// Whether it runs on the operating system of `target` (a release-archive triple). + #[must_use] + pub fn runs_on(&self, target: Option<&str>) -> bool { + self.platforms.is_empty() + || target.is_some_and(|target| { + self.platforms.iter().any(|os| match os.as_str() { + "linux" => target.contains("-linux"), + "macos" => target.contains("-apple-darwin"), + _ => false, + }) + }) + } +} + +/// The ways a binary can be installed; at least one is present. +#[derive(Debug, Clone, PartialEq, Eq, Deserialize, Serialize)] +pub struct Install { + /// `--.tar.gz` plus `SHA256SUMS` on the release. + pub archive: Option, + /// `cargo install --git … --tag … `. + pub cargo: Option, +} + +/// A release that carries prebuilt archives. +#[derive(Debug, Clone, PartialEq, Eq, Deserialize, Serialize)] +pub struct Archive { + /// `owner/repo`. + pub repository: String, +} + +/// A cargo package that builds the binary. +#[derive(Debug, Clone, PartialEq, Eq, Deserialize, Serialize)] +pub struct Cargo { + /// `owner/repo`. + pub repository: String, + /// Cargo package. + pub package: String, +} + +/// How to install a binary this time. +#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Deserialize, Serialize)] +#[serde(rename_all = "kebab-case")] +pub enum Method { + /// The release's checksummed archive. + Prebuilt, + /// `cargo install` from the release tag. + Cargo, +} + +impl Install { + /// The GitHub repository the binary is released from. + #[must_use] + pub fn repository(&self) -> &str { + self.archive + .as_ref() + .map(|archive| archive.repository.as_str()) + .or_else(|| self.cargo.as_ref().map(|cargo| cargo.repository.as_str())) + .unwrap_or_default() + } +} + +impl Catalog { + /// Parse and check a catalog document. + pub fn parse(text: &str) -> Result { + let catalog: Catalog = + serde_json::from_str(text).map_err(|error| format!("catalog: {error}"))?; + if catalog.format != FORMAT { + return Err(format!( + "catalog format `{}` is not `{FORMAT}`", + catalog.format + )); + } + Ok(catalog) + } + + /// The catalog compiled into this binary. + /// + /// # Panics + /// Never for a released binary: the build's own tests parse it. + #[must_use] + pub fn embedded() -> Self { + Self::parse(EMBEDDED).expect("the embedded catalog parses") + } + + /// The product that ships this plugin. + #[must_use] + pub fn product_of(&self, plugin: &str) -> Option<&Product> { + self.products + .iter() + .find(|product| product.plugins.iter().any(|name| name == plugin)) + } + + /// A product by id. + #[must_use] + pub fn product(&self, id: &str) -> Option<&Product> { + self.products.iter().find(|product| product.id == id) + } + + /// The current name for a plugin name, which is itself unless it is retired. + #[must_use] + pub fn current_name<'a>(&'a self, name: &'a str) -> &'a str { + self.retired_plugins.get(name).map_or(name, String::as_str) + } + + /// Whether the catalog knows this plugin name, current or retired. + #[must_use] + pub fn knows(&self, name: &str) -> bool { + self.retired_plugins.contains_key(name) + || self.base.iter().any(|base| base == name) + || self.product_of(name).is_some() + } + + /// Whether this marketplace name is retired. + #[must_use] + pub fn retired_marketplace(&self, name: &str) -> bool { + self.retired_marketplaces + .iter() + .any(|retired| retired == name) + } + + /// The binary with this name, and its product. + #[must_use] + pub fn binary(&self, name: &str) -> Option<(&Product, &Binary)> { + self.products.iter().find_map(|product| { + product + .binaries + .iter() + .find(|binary| binary.name == name) + .map(|binary| (product, binary)) + }) + } +} + +#[cfg(test)] +mod tests { + #[test] + fn a_binary_limited_to_linux_runs_only_on_linux_targets() { + let mut binary = super::Binary { + name: "b10x-harness".to_owned(), + optional: true, + platforms: vec!["linux".to_owned()], + install: super::Install { + archive: None, + cargo: None, + }, + }; + assert!(binary.runs_on(Some("x86_64-unknown-linux-gnu"))); + assert!(binary.runs_on(Some("aarch64-unknown-linux-gnu"))); + assert!(!binary.runs_on(Some("aarch64-apple-darwin"))); + assert!(!binary.runs_on(None)); + binary.platforms.clear(); + assert!(binary.runs_on(None)); + } + + use super::*; + + #[test] + fn the_embedded_catalog_parses_and_names_no_version() { + let catalog = Catalog::embedded(); + assert_eq!(catalog.marketplace.name, "b10x"); + assert!(catalog.product("ess").is_some()); + let digit_dot_digit = EMBEDDED + .as_bytes() + .windows(3) + .any(|w| w[0].is_ascii_digit() && w[1] == b'.' && w[2].is_ascii_digit()); + assert!(!digit_dot_digit, "the catalog must not pin a version"); + } + + #[test] + fn retired_names_map_to_current_ones() { + let catalog = Catalog::embedded(); + assert_eq!(catalog.current_name("workspace-hygiene"), "worktree"); + assert_eq!(catalog.current_name("aep-plan"), "aep"); + assert_eq!(catalog.current_name("aep"), "aep"); + assert!(catalog.knows("ess-specify")); + assert!(catalog.retired_marketplace("beyond10x")); + assert!(!catalog.retired_marketplace("b10x")); + } +} diff --git a/crates/b10x/src/check.rs b/crates/b10x/src/check.rs new file mode 100644 index 0000000..6600532 --- /dev/null +++ b/crates/b10x/src/check.rs @@ -0,0 +1,464 @@ +//! The session-start check. Quick: it reads what the hosts recorded on disk, runs each installed +//! product's `--version`, and compares with the newest releases recorded in +//! `~/.local/state/b10x/latest.json`. That record is refreshed from GitHub at most once a day, +//! within a few seconds; offline, the old record stands. It prints one line per problem, nothing +//! when all is well, and never fails the session. + +use std::collections::{BTreeMap, BTreeSet}; +use std::path::Path; + +use crate::catalog::Catalog; +use crate::inventory::{copies_on_path, home}; +use crate::pins::{self, PinFile, Spec}; +use crate::{resolve, version}; + +/// The releases the skills were last verified against: the repository's `verified.json`. +const VERIFIED: &str = include_str!("../../../verified.json"); + +/// How often the newest-release record is refreshed from GitHub. +const REFRESH_EVERY: u64 = 86_400; + +/// How long one refresh may take, in seconds; the repositories are read in parallel. +const REFRESH_SECONDS: u64 = 3; + +/// CLI → the release its skills were last verified against. +#[must_use] +pub fn verified() -> BTreeMap { + serde_json::from_str(VERIFIED).unwrap_or_default() +} + +/// Refresh the newest-release record when the last try is a day old: the marketplace repository +/// and every catalog binary's, each within [`REFRESH_SECONDS`]. Failures leave the record as it was. +pub fn refresh(home: &Path, catalog: &Catalog, now: u64) { + let record = resolve::recorded(home); + let last = ["checked_at", "attempted_at"] + .iter() + .filter_map(|key| record.get(*key).and_then(serde_json::Value::as_u64)) + .max(); + if last.is_some_and(|last| now.saturating_sub(last) < REFRESH_EVERY) { + return; + } + let mut repositories = BTreeSet::from([catalog.marketplace.repository.clone()]); + for product in &catalog.products { + for binary in &product.binaries { + repositories.insert(binary.install.repository().to_owned()); + } + } + let tags = resolve::latest_tags(&repositories, REFRESH_SECONDS); + resolve::record(home, &tags, now); +} + +/// A recorded install: `name`, `marketplace`, version. +pub type Recorded = (String, String, Option); + +/// Finds the first copy of a binary on `PATH`: its path and version. +pub type Probe<'a> = &'a dyn Fn(&str) -> Option<(String, Option)>; + +/// How old the last check of newest releases may be before the check says so. +const STALE_DAYS: u64 = 7; + +/// Installed plugins as the hosts recorded them. +#[must_use] +pub fn recorded(home: &Path) -> Vec { + let mut plugins = Vec::new(); + if let Ok(text) = std::fs::read_to_string(home.join(".claude/plugins/installed_plugins.json")) { + if let Ok(value) = serde_json::from_str::(&text) { + if let Some(map) = value.get("plugins").and_then(serde_json::Value::as_object) { + for (id, installs) in map { + let Some((name, marketplace)) = id.rsplit_once('@') else { + continue; + }; + let version = installs + .as_array() + .and_then(|list| list.first()) + .and_then(|install| install.get("version")) + .and_then(serde_json::Value::as_str) + .map(str::to_owned); + plugins.push((name.to_owned(), marketplace.to_owned(), version)); + } + } + } + } + plugins +} + +/// What the last plan saw: newest release per repository, and when (Unix seconds). +#[must_use] +pub fn latest(home: &Path) -> Option<(u64, serde_json::Map)> { + let text = std::fs::read_to_string(crate::resolve::cache_path(home)).ok()?; + let value: serde_json::Value = serde_json::from_str(&text).ok()?; + let at = value.get("checked_at")?.as_u64()?; + let map = value.get("latest")?.as_object()?.clone(); + Some((at, map)) +} + +/// What the check compares against besides the machine: the repository's pins and the releases the +/// skills were verified against. +pub struct Expected<'a> { + /// The pin file that applies in the session's directory. + pub pins: Option<&'a PinFile>, + /// CLI → the release the skills were verified against. + pub verified: &'a BTreeMap, +} + +/// The lines to print. +#[must_use] +#[allow(clippy::too_many_lines)] +pub fn lines( + catalog: &Catalog, + plugins: &[Recorded], + binary_version: Probe<'_>, + last: Option<(u64, &serde_json::Map)>, + now: u64, + expected: &Expected<'_>, +) -> Vec { + let mut lines = Vec::new(); + let name = catalog.marketplace.name.as_str(); + let newest_of = |repository: &str| { + last.and_then(|(_, map)| map.get(repository)) + .and_then(serde_json::Value::as_str) + .map(str::to_owned) + }; + let oldest_plugin = plugins + .iter() + .filter(|(_, marketplace, _)| marketplace == name) + .filter_map(|(_, _, found)| found.as_deref()) + .filter(|found| version::key(found).is_some()) + .min_by_key(|found| version::key(found)); + if let (Some(have), Some(newest)) = (oldest_plugin, newest_of(&catalog.marketplace.repository)) + { + if version::key(have) < version::key(&newest) { + lines.push(format!( + "b10x: the Beyond10x plugins ({have}) are older than the newest release {newest}; /b10x:upgrade updates them." + )); + } + } + let pinned = |binary: &str| { + expected + .pins + .and_then(|file| Some((file.pins.get(binary)?, file.path.display().to_string()))) + }; + for (plugin, marketplace, _) in plugins { + if catalog.retired_marketplace(marketplace) || catalog.retired_plugins.contains_key(plugin) + { + lines.push(format!( + "b10x: an earlier install, `{plugin}@{marketplace}`, is still here; /b10x:upgrade replaces it." + )); + } + } + let installed: Vec<&str> = plugins + .iter() + .filter(|(_, marketplace, _)| marketplace == name) + .map(|(plugin, _, _)| plugin.as_str()) + .collect(); + let mut any_product = false; + for product in &catalog.products { + if !product + .plugins + .iter() + .any(|plugin| installed.contains(&plugin.as_str())) + { + continue; + } + any_product = true; + for binary in product.binaries.iter().filter(|binary| !binary.optional) { + let newest = newest_of(binary.install.repository()); + match binary_version(&binary.name) { + None => lines.push(format!( + "b10x: the `{}` plugin is installed but the `{}` CLI is not on PATH; /{}:init installs it.", + product.id, binary.name, product.id + )), + // A pinned CLI is held to its pin below, not to the newest release. + Some(_) if pinned(&binary.name).is_some() => {} + Some((path, found)) => { + let found = found.unwrap_or_else(|| "an unknown version".to_owned()); + if let Some(newest) = newest { + let behind = matches!( + (version::key(&found), version::key(&newest)), + (Some(have), Some(want)) if have < want + ); + if behind { + lines.push(format!( + "b10x: `{}` {found} ({path}) is older than the newest release {newest}; /b10x:upgrade updates it.", + binary.name + )); + } + } + } + } + } + } + for product in &catalog.products { + for binary in &product.binaries { + let Some((spec, file)) = pinned(&binary.name) else { + continue; + }; + let Ok(parsed) = Spec::parse(spec) else { + continue; + }; + if let Some((path, found)) = binary_version(&binary.name) { + let found = found.unwrap_or_else(|| "an unknown version".to_owned()); + if !parsed.matches(&found) { + lines.push(format!( + "b10x: `{}` {found} ({path}) does not match the pin {spec} in {file}; `b10x install {}` installs the pinned release.", + binary.name, binary.name + )); + } + } + if let Some(described) = expected.verified.get(&binary.name) { + if parsed.older_than(described) { + lines.push(format!( + "b10x: skills describe {} {described}; this repository pins {spec} ({file}).", + binary.name + )); + } + } + } + } + if any_product { + let days = last.map(|(at, _)| now.saturating_sub(at) / 86_400); + match days { + None => lines.push( + "b10x: newest releases have not been checked on this machine; /b10x:upgrade checks them." + .to_owned(), + ), + Some(days) if days >= STALE_DAYS => lines.push(format!( + "b10x: newest releases were last checked {days} days ago; /b10x:upgrade checks again." + )), + Some(_) => {} + } + } else if !installed + .iter() + .any(|plugin| catalog.product_of(catalog.current_name(plugin)).is_some()) + { + lines.push( + "b10x: no Beyond10x product is set up yet; /b10x:init asks what you want to do and installs it." + .to_owned(), + ); + } + lines +} + +/// Run the check and print its lines. +pub fn run(catalog: &Catalog) { + let home = home(); + let now = resolve::now(); + refresh(&home, catalog, now); + let plugins = recorded(&home); + let first = |name: &str| { + copies_on_path(name) + .into_iter() + .next() + .map(|copy| (copy.path, copy.version)) + }; + let cached = latest(&home); + let last = cached.as_ref().map(|(at, map)| (*at, map)); + let pin_file = match std::env::current_dir().map(|here| pins::read(&here, &home)) { + Ok(Ok(found)) => found, + Ok(Err(error)) => { + println!("b10x: {error}"); + None + } + Err(_) => None, + }; + let verified = verified(); + let expected = Expected { + pins: pin_file.as_ref(), + verified: &verified, + }; + for line in lines(catalog, &plugins, &first, last, now, &expected) { + println!("{line}"); + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn plugin(name: &str, market: &str, version: &str) -> Recorded { + (name.to_owned(), market.to_owned(), Some(version.to_owned())) + } + + fn cache(ess: &str) -> serde_json::Map { + let mut map = serde_json::Map::new(); + map.insert("beyond10x/ess".to_owned(), serde_json::json!(ess)); + map + } + + static NOTHING_VERIFIED: BTreeMap = BTreeMap::new(); + + fn unpinned() -> Expected<'static> { + Expected { + pins: None, + verified: &NOTHING_VERIFIED, + } + } + + fn ess_at(version: &'static str) -> impl Fn(&str) -> Option<(String, Option)> { + move |_: &str| { + Some(( + "/opt/b10x-home/.local/bin/ess".to_owned(), + Some(version.to_owned()), + )) + } + } + + #[test] + fn the_embedded_verified_releases_parse() { + let verified = verified(); + for cli in ["aep", "ess", "worktree"] { + assert!( + verified.get(cli).and_then(|v| version::key(v)).is_some(), + "{cli}: {verified:?}" + ); + } + } + + #[test] + fn plugins_older_than_the_newest_agentplugins_release_name_upgrade() { + let catalog = Catalog::embedded(); + let plugins = [ + plugin("b10x", "b10x", "0.14.7"), + plugin("ess", "b10x", "0.14.7"), + ]; + let mut map = cache("0.30.0"); + map.insert( + "beyond10x/agentplugins".to_owned(), + serde_json::json!("0.14.10"), + ); + let out = lines( + &catalog, + &plugins, + &ess_at("0.30.0"), + Some((1_000, &map)), + 1_000, + &unpinned(), + ); + assert_eq!( + out, + ["b10x: the Beyond10x plugins (0.14.7) are older than the newest release 0.14.10; /b10x:upgrade updates them."] + ); + } + + #[test] + fn a_pinned_cli_is_held_to_its_pin_and_newer_skills_are_noted() { + let catalog = Catalog::embedded(); + let plugins = [plugin("ess", "b10x", "0.14.0")]; + let map = cache("0.32.1"); + let file = PinFile { + path: std::path::PathBuf::from("/work/repo/b10x.toml"), + pins: BTreeMap::from([("ess".to_owned(), "0.32.0".to_owned())]), + }; + let verified = BTreeMap::from([("ess".to_owned(), "0.32.1".to_owned())]); + let expected = Expected { + pins: Some(&file), + verified: &verified, + }; + // At the pin: no "older than the newest" nag, only the skills note. + let out = lines( + &catalog, + &plugins, + &ess_at("0.32.0"), + Some((1_000, &map)), + 1_000, + &expected, + ); + assert_eq!( + out, + ["b10x: skills describe ess 0.32.1; this repository pins 0.32.0 (/work/repo/b10x.toml)."] + ); + // Off the pin: the fixing command is named. + let out = lines( + &catalog, + &plugins, + &ess_at("0.32.1"), + Some((1_000, &map)), + 1_000, + &expected, + ); + assert!( + out.iter() + .any(|l| l.contains("does not match the pin 0.32.0") + && l.contains("`b10x install ess`")), + "{out:#?}" + ); + } + + #[test] + fn silent_when_current_and_recently_checked() { + let catalog = Catalog::embedded(); + let plugins = [ + plugin("b10x", "b10x", "0.14.0"), + plugin("ess", "b10x", "0.14.0"), + ]; + let found = |_: &str| { + Some(( + "/opt/b10x-home/.local/bin/ess".to_owned(), + Some("0.30.0".to_owned()), + )) + }; + let map = cache("0.30.0"); + assert!(lines( + &catalog, + &plugins, + &found, + Some((1_000, &map)), + 1_000 + 86_400, + &unpinned() + ) + .is_empty()); + } + + #[test] + fn names_an_old_cli_a_legacy_install_and_a_stale_check() { + let catalog = Catalog::embedded(); + let plugins = [ + plugin("ess", "b10x", "0.14.0"), + plugin("workspace-hygiene", "beyond10x", "0.10.0"), + ]; + let found = |_: &str| { + Some(( + "/opt/b10x-home/.cargo/bin/ess".to_owned(), + Some("0.26.0".to_owned()), + )) + }; + let map = cache("0.30.0"); + let out = lines( + &catalog, + &plugins, + &found, + Some((0, &map)), + 30 * 86_400, + &unpinned(), + ); + assert_eq!(out.len(), 3, "{out:#?}"); + assert!(out + .iter() + .any(|l| l.contains("older than the newest release 0.30.0") + && l.contains("/b10x:upgrade"))); + assert!(out + .iter() + .any(|l| l.contains("workspace-hygiene@beyond10x"))); + assert!(out.iter().any(|l| l.contains("30 days ago"))); + } + + #[test] + fn points_at_init_when_nothing_is_set_up() { + let catalog = Catalog::embedded(); + let plugins = [plugin("b10x", "b10x", "0.14.0")]; + let out = lines(&catalog, &plugins, &|_: &str| None, None, 0, &unpinned()); + assert_eq!(out.len(), 1); + assert!(out[0].contains("/b10x:init")); + } + + #[test] + fn a_legacy_plugin_is_not_nothing_set_up() { + let catalog = Catalog::embedded(); + let plugins = [ + plugin("b10x", "b10x", "0.12.0"), + plugin("aep-plan", "b10x", "0.12.0"), + ]; + let out = lines(&catalog, &plugins, &|_: &str| None, None, 0, &unpinned()); + assert!(!out.iter().any(|l| l.contains("/b10x:init")), "{out:#?}"); + } +} diff --git a/crates/b10x/src/install.rs b/crates/b10x/src/install.rs new file mode 100644 index 0000000..2101771 --- /dev/null +++ b/crates/b10x/src/install.rs @@ -0,0 +1,256 @@ +//! Installing one binary at one exact release: a checksummed release archive, or `cargo install` +//! from the tag. Either way the binary is staged first and moved into place with one rename. + +use std::collections::BTreeSet; +use std::path::{Path, PathBuf}; +use std::process::Command; + +use sha2::{Digest, Sha256}; + +use crate::catalog::{Install, Method}; +use crate::inventory::home; + +/// The Rust target triple of release archives for this machine. +pub fn target() -> Result<&'static str, String> { + match (std::env::consts::ARCH, std::env::consts::OS) { + ("x86_64", "linux") => Ok("x86_64-unknown-linux-gnu"), + ("aarch64", "linux") => Ok("aarch64-unknown-linux-gnu"), + ("x86_64", "macos") => Ok("x86_64-apple-darwin"), + ("aarch64", "macos") => Ok("aarch64-apple-darwin"), + (arch, os) => Err(format!("no release archive for {arch}-{os}")), + } +} + +fn run(program: &str, arguments: &[&str]) -> Result<(), String> { + let output = Command::new(program) + .args(arguments) + .output() + .map_err(|error| format!("{program}: {error}"))?; + if output.status.success() { + Ok(()) + } else { + Err(format!( + "{program} {} failed: {}", + arguments.join(" "), + String::from_utf8_lossy(&output.stderr).trim() + )) + } +} + +/// The release archive of `name` at `tag` for `target`: `--.tar.gz`. +#[must_use] +pub fn archive_name(name: &str, tag: &str, target: &str) -> String { + format!("{}{target}.tar.gz", archive_prefix(name, tag)) +} + +/// What every archive name of `name` at `tag` starts with, before the target. +fn archive_prefix(name: &str, tag: &str) -> String { + format!("{name}-{}-", tag.trim_start_matches('v')) +} + +/// `(digest, file)` for every line of a `SHA256SUMS` document. +fn listed(sums: &str) -> impl Iterator { + sums.lines().filter_map(|line| { + let (digest, name) = line.split_once(char::is_whitespace)?; + Some((digest, name.trim().trim_start_matches('*'))) + }) +} + +/// The hex SHA-256 listed for `file` in a `SHA256SUMS` document. +#[must_use] +pub fn listed_digest<'a>(sums: &'a str, file: &str) -> Option<&'a str> { + listed(sums).find_map(|(digest, name)| (name == file).then_some(digest)) +} + +/// The targets a `SHA256SUMS` document lists an archive of `name` at `tag` for. +#[must_use] +pub fn listed_targets(sums: &str, name: &str, tag: &str) -> BTreeSet { + let prefix = archive_prefix(name, tag); + listed(sums) + .filter_map(|(_, file)| file.strip_prefix(&prefix)?.strip_suffix(".tar.gz")) + .filter(|target| !target.is_empty() && !target.contains('/')) + .map(str::to_owned) + .collect() +} + +/// Lower-case hex of bytes. +#[must_use] +pub fn hex(bytes: &[u8]) -> String { + use std::fmt::Write as _; + bytes.iter().fold(String::new(), |mut out, byte| { + let _ = write!(out, "{byte:02x}"); + out + }) +} + +fn sha256(path: &Path) -> Result { + let bytes = std::fs::read(path).map_err(|error| format!("{}: {error}", path.display()))?; + Ok(hex(&Sha256::digest(bytes))) +} + +fn staging(name: &str) -> Result { + let path = home() + .join(".cache/b10x/install") + .join(format!("{name}-{}", std::process::id())); + let _ = std::fs::remove_dir_all(&path); + std::fs::create_dir_all(&path).map_err(|error| format!("{}: {error}", path.display()))?; + Ok(path) +} + +fn from_archive(name: &str, tag: &str, repository: &str, stage: &Path) -> Result { + let target = target()?; + let version = tag.trim_start_matches('v'); + let archive = archive_name(name, tag, target); + let base = format!("/{repository}/releases/download/{tag}"); + let archive_path = stage.join(&archive); + let sums_path = stage.join("SHA256SUMS"); + for (url, path) in [ + (format!("{base}/{archive}"), &archive_path), + (format!("{base}/SHA256SUMS"), &sums_path), + ] { + run( + "curl", + &["-fsSL", "--retry", "2", "-o", &path.to_string_lossy(), &url], + )?; + } + let sums = std::fs::read_to_string(&sums_path).map_err(|error| error.to_string())?; + let expected = listed_digest(&sums, &archive) + .ok_or_else(|| format!("SHA256SUMS does not list {archive}"))?; + let actual = sha256(&archive_path)?; + if actual != expected { + return Err(format!( + "{archive}: checksum {actual} is not the listed {expected}" + )); + } + run( + "tar", + &[ + "-xzf", + &archive_path.to_string_lossy(), + "-C", + &stage.to_string_lossy(), + ], + )?; + let binary = stage.join(format!("{name}-{version}-{target}")).join(name); + if binary.is_file() { + Ok(binary) + } else { + Err(format!("{archive} has no {name}")) + } +} + +fn from_cargo( + name: &str, + tag: &str, + repository: &str, + package: &str, + stage: &Path, +) -> Result { + let url = format!("/{repository}"); + run( + "cargo", + &[ + "install", + "--git", + &url, + "--tag", + tag, + "--locked", + "--root", + &stage.to_string_lossy(), + package, + ], + ) + .map_err(|error| { + if error.starts_with("cargo:") { + format!("{name} needs a Rust toolchain (https://rustup.rs): {error}") + } else { + error + } + })?; + Ok(stage.join("bin").join(name)) +} + +/// Install `name` at `tag` into `directory`, replacing any copy there in one rename. +pub fn install( + name: &str, + tag: &str, + method: Method, + install: &Install, + directory: &Path, +) -> Result { + let stage = staging(name)?; + let result = (|| { + let built = match (method, &install.archive, &install.cargo) { + (Method::Prebuilt, Some(archive), _) => { + from_archive(name, tag, &archive.repository, &stage)? + } + (Method::Cargo, _, Some(cargo)) => { + from_cargo(name, tag, &cargo.repository, &cargo.package, &stage)? + } + (method, _, _) => return Err(format!("{name} cannot be installed by {method:?}")), + }; + std::fs::create_dir_all(directory) + .map_err(|error| format!("{}: {error}", directory.display()))?; + let destination = directory.join(name); + let incoming = directory.join(format!(".{name}.b10x-new")); + std::fs::copy(&built, &incoming) + .map_err(|error| format!("{}: {error}", incoming.display()))?; + #[cfg(unix)] + { + use std::os::unix::fs::PermissionsExt; + std::fs::set_permissions(&incoming, std::fs::Permissions::from_mode(0o755)) + .map_err(|error| error.to_string())?; + } + std::fs::rename(&incoming, &destination) + .map_err(|error| format!("{}: {error}", destination.display()))?; + Ok(destination) + })(); + let _ = std::fs::remove_dir_all(&stage); + result +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn sums_are_matched_by_exact_file_name() { + let sums = "aaa ess-0.30.0-x86_64-unknown-linux-gnu.tar.gz\nbbb *ess-0.30.0-aarch64-apple-darwin.tar.gz\n"; + assert_eq!( + listed_digest(sums, "ess-0.30.0-x86_64-unknown-linux-gnu.tar.gz"), + Some("aaa") + ); + assert_eq!( + listed_digest(sums, "ess-0.30.0-aarch64-apple-darwin.tar.gz"), + Some("bbb") + ); + assert_eq!(listed_digest(sums, "ess-0.30.0.tar.gz"), None); + } + + #[test] + fn targets_are_read_from_the_archive_names_sums_lists() { + let sums = "\ +aaa b10x-harness-0.13.2-x86_64-unknown-linux-gnu.tar.gz +bbb b10x-harness-0.13.2-x86_64-unknown-linux-gnu.tar.gz.sig +ccc *b10x-harness-0.13.1-aarch64-unknown-linux-gnu.tar.gz +ddd harness-0.13.2-aarch64-apple-darwin.tar.gz +eee b10x-harness-lsp-0.13.2-aarch64-unknown-linux-gnu.tar.gz +fff b10x-harness-0.13.2-.tar.gz +"; + assert_eq!( + listed_targets(sums, "b10x-harness", "0.13.2"), + BTreeSet::from(["x86_64-unknown-linux-gnu".to_owned()]) + ); + assert_eq!( + listed_targets(sums, "b10x-harness", "v0.13.2"), + listed_targets(sums, "b10x-harness", "0.13.2"), + "a leading v on the tag is not part of the archive name" + ); + assert!(listed_targets("", "b10x-harness", "0.13.2").is_empty()); + for target in listed_targets(sums, "b10x-harness", "0.13.2") { + let name = archive_name("b10x-harness", "0.13.2", &target); + assert_eq!(listed_digest(sums, &name), Some("aaa")); + } + } +} diff --git a/crates/b10x/src/inventory.rs b/crates/b10x/src/inventory.rs new file mode 100644 index 0000000..c6b5e31 --- /dev/null +++ b/crates/b10x/src/inventory.rs @@ -0,0 +1,556 @@ +//! What is installed right now: plugins and marketplaces per host, binaries on `PATH`, and plugin +//! entries in project settings files. Parsing is separate from collection so the parsers can be +//! tested on recorded host output. + +use std::collections::BTreeSet; +use std::path::{Path, PathBuf}; +use std::process::Command; + +use serde::{Deserialize, Serialize}; + +use crate::catalog::Catalog; +use crate::version; + +/// An agent host. +#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Deserialize, Serialize)] +#[serde(rename_all = "kebab-case")] +pub enum Host { + /// Claude Code, `claude`. + Claude, + /// Codex, `codex`. + Codex, +} + +impl Host { + /// The executable. + #[must_use] + pub fn program(self) -> &'static str { + match self { + Host::Claude => "claude", + Host::Codex => "codex", + } + } +} + +/// Everything setup reads before deciding. +#[derive(Debug, Clone, Default, PartialEq, Eq, Deserialize, Serialize)] +pub struct Inventory { + /// Claude Code, when `claude` runs. + pub claude: Option, + /// Codex, when `codex` runs. + pub codex: Option, + /// Every catalog binary, every copy on `PATH` in `PATH` order. + pub binaries: Vec, + /// Plugin entries in settings files the host list does not report (other projects, orphans). + pub settings: Vec, + /// Hosts whose program runs but whose plugin commands fail, with the error they printed. + #[serde(default)] + pub broken: Vec<(Host, String)>, + /// Whether `cargo` is on `PATH`, so binaries can be built from source. + #[serde(default)] + pub cargo: bool, +} + +/// One host's plugins and marketplaces. +#[derive(Debug, Clone, Default, PartialEq, Eq, Deserialize, Serialize)] +pub struct HostState { + /// Installed plugins. + pub plugins: Vec, + /// Registered marketplaces. + pub marketplaces: Vec, +} + +/// One installed plugin at one scope. +#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Ord, Deserialize, Serialize)] +pub struct Installed { + /// Plugin name. + pub name: String, + /// Marketplace name. + pub marketplace: String, + /// Installed version, when the host reports one. + pub version: Option, + /// `user`, `project` or `local` (Codex: `user`). + pub scope: String, + /// Project directory for `project` and `local` scope. + pub project: Option, + /// Whether it is enabled. + pub enabled: bool, +} + +impl Installed { + /// `name@marketplace`. + #[must_use] + pub fn id(&self) -> String { + format!("{}@{}", self.name, self.marketplace) + } +} + +/// One registered marketplace. +#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Ord, Deserialize, Serialize)] +pub struct Market { + /// Marketplace name. + pub name: String, + /// `owner/repo`, a Git URL or a directory. + pub source: String, + /// Pinned ref, when the registration names one. + pub reference: Option, + /// Local clone, when the host reports one. + pub location: Option, +} + +/// Every copy of one binary. +#[derive(Debug, Clone, PartialEq, Eq, Deserialize, Serialize)] +pub struct BinaryState { + /// Executable name. + pub name: String, + /// Copies in `PATH` order; the first is the one that runs. + pub copies: Vec, +} + +/// One copy of a binary. +#[derive(Debug, Clone, PartialEq, Eq, Deserialize, Serialize)] +pub struct Copy { + /// Absolute path. + pub path: String, + /// Version its `--version` printed. + pub version: Option, +} + +/// A plugin entry in a Claude Code settings file. +#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Ord, Deserialize, Serialize)] +pub struct SettingsEntry { + /// The settings file. + pub file: String, + /// `project` (`.claude/settings.json`, committed) or `local` or `user`. + pub scope: String, + /// The project directory, for project and local files. + pub project: Option, + /// `name@marketplace`. + pub plugin: String, + /// The value under `enabledPlugins`. + pub enabled: bool, +} + +fn split_id(id: &str) -> Option<(String, String)> { + let (name, marketplace) = id.rsplit_once('@')?; + Some((name.to_owned(), marketplace.to_owned())) +} + +fn string(value: &serde_json::Value, key: &str) -> Option { + value + .get(key) + .and_then(serde_json::Value::as_str) + .map(str::to_owned) +} + +/// Parse `claude plugin list --json`. +pub fn parse_claude_plugins(text: &str) -> Result, String> { + let value: serde_json::Value = + serde_json::from_str(text).map_err(|error| format!("claude plugin list: {error}"))?; + let entries = value.as_array().ok_or("claude plugin list: not an array")?; + let mut plugins = Vec::new(); + for entry in entries { + let id = string(entry, "id").ok_or("claude plugin list: entry without id")?; + let Some((name, marketplace)) = split_id(&id) else { + continue; + }; + plugins.push(Installed { + name, + marketplace, + version: string(entry, "version"), + scope: string(entry, "scope").unwrap_or_else(|| "user".to_owned()), + project: string(entry, "projectPath"), + enabled: entry + .get("enabled") + .and_then(serde_json::Value::as_bool) + .unwrap_or(true), + }); + } + plugins.sort(); + Ok(plugins) +} + +/// Parse `claude plugin marketplace list --json`. +pub fn parse_claude_marketplaces(text: &str) -> Result, String> { + let value: serde_json::Value = serde_json::from_str(text) + .map_err(|error| format!("claude plugin marketplace list: {error}"))?; + let entries = value + .as_array() + .ok_or("claude plugin marketplace list: not an array")?; + let mut markets = Vec::new(); + for entry in entries { + let name = string(entry, "name").ok_or("claude marketplace without name")?; + let source = string(entry, "repo") + .or_else(|| string(entry, "url")) + .or_else(|| string(entry, "path")) + .unwrap_or_default(); + markets.push(Market { + name, + source, + reference: string(entry, "ref"), + location: string(entry, "installLocation"), + }); + } + markets.sort(); + Ok(markets) +} + +/// Parse `codex plugin list --json`. +pub fn parse_codex_plugins(text: &str) -> Result, String> { + let value: serde_json::Value = + serde_json::from_str(text).map_err(|error| format!("codex plugin list: {error}"))?; + let entries = value + .get("installed") + .and_then(serde_json::Value::as_array) + .ok_or("codex plugin list: no `installed` array")?; + let mut plugins = Vec::new(); + for entry in entries { + let id = string(entry, "pluginId").ok_or("codex plugin list: entry without pluginId")?; + let Some((name, marketplace)) = split_id(&id) else { + continue; + }; + plugins.push(Installed { + name, + marketplace, + version: string(entry, "version"), + scope: "user".to_owned(), + project: None, + enabled: entry + .get("enabled") + .and_then(serde_json::Value::as_bool) + .unwrap_or(true), + }); + } + plugins.sort(); + Ok(plugins) +} + +/// Parse `codex plugin marketplace list --json`, taking pinned refs from `~/.codex/config.toml` +/// because the list does not report them. +pub fn parse_codex_marketplaces(text: &str, config: Option<&str>) -> Result, String> { + let value: serde_json::Value = serde_json::from_str(text) + .map_err(|error| format!("codex plugin marketplace list: {error}"))?; + let entries = value + .get("marketplaces") + .and_then(serde_json::Value::as_array) + .ok_or("codex plugin marketplace list: no `marketplaces` array")?; + let config: Option = config.and_then(|text| text.parse().ok()); + let mut markets = Vec::new(); + for entry in entries { + let name = string(entry, "name").ok_or("codex marketplace without name")?; + let source = entry + .get("marketplaceSource") + .and_then(|source| string(source, "source")) + .unwrap_or_default(); + let reference = config + .as_ref() + .and_then(|table| table.get("marketplaces")) + .and_then(|markets| markets.get(&name)) + .and_then(|market| market.get("ref")) + .and_then(toml::Value::as_str) + .map(str::to_owned); + markets.push(Market { + name, + source, + reference, + location: string(entry, "root"), + }); + } + markets.sort(); + Ok(markets) +} + +/// Plugin entries under `enabledPlugins` in one Claude Code settings file that the catalog knows or +/// that come from a marketplace the catalog retired. +pub fn parse_settings( + text: &str, + file: &str, + scope: &str, + project: Option<&str>, + catalog: &Catalog, +) -> Vec { + let Ok(value) = serde_json::from_str::(text) else { + return Vec::new(); + }; + let Some(enabled) = value + .get("enabledPlugins") + .and_then(serde_json::Value::as_object) + else { + return Vec::new(); + }; + let mut entries = Vec::new(); + for (plugin, state) in enabled { + let Some((name, marketplace)) = split_id(plugin) else { + continue; + }; + let relevant = marketplace == catalog.marketplace.name + || catalog.retired_marketplace(&marketplace) + || catalog.retired_plugins.contains_key(&name); + if relevant { + entries.push(SettingsEntry { + file: file.to_owned(), + scope: scope.to_owned(), + project: project.map(str::to_owned), + plugin: plugin.clone(), + enabled: state.as_bool().unwrap_or(false), + }); + } + } + entries.sort(); + entries +} + +/// Output of a command, or `None` when it cannot run or fails. +fn output(program: &str, arguments: &[&str]) -> Option { + let output = Command::new(program).args(arguments).output().ok()?; + output + .status + .success() + .then(|| String::from_utf8_lossy(&output.stdout).into_owned()) +} + +/// The user's home directory. +/// +/// # Panics +/// When `HOME` is unset, which no supported host allows. +#[must_use] +pub fn home() -> PathBuf { + PathBuf::from(std::env::var_os("HOME").expect("HOME is set")) +} + +fn read(path: &Path) -> Option { + std::fs::read_to_string(path).ok() +} + +/// Every executable called `name` on `PATH`, in `PATH` order, each canonical path once. +#[must_use] +pub fn copies_on_path(name: &str) -> Vec { + let mut seen = BTreeSet::new(); + let mut copies = Vec::new(); + let Some(path) = std::env::var_os("PATH") else { + return copies; + }; + for directory in std::env::split_paths(&path) { + let candidate = directory.join(name); + let Ok(metadata) = std::fs::metadata(&candidate) else { + continue; + }; + if !metadata.is_file() || !executable(&metadata) { + continue; + } + let canonical = std::fs::canonicalize(&candidate).unwrap_or_else(|_| candidate.clone()); + if !seen.insert(canonical) { + continue; + } + let shown = candidate.to_string_lossy().into_owned(); + let version = output(&shown, &["--version"]).and_then(|text| version::parse(&text)); + copies.push(Copy { + path: shown, + version, + }); + } + copies +} + +#[cfg(unix)] +fn executable(metadata: &std::fs::Metadata) -> bool { + use std::os::unix::fs::PermissionsExt; + metadata.permissions().mode() & 0o111 != 0 +} + +#[cfg(not(unix))] +fn executable(_: &std::fs::Metadata) -> bool { + true +} + +/// Read one host, or `None` when its executable does not run. +/// Output of a host command: `None` when the program is not installed, `Err` with what it printed +/// when it runs and fails. +fn host_output(program: &str, arguments: &[&str]) -> Option> { + let output = match Command::new(program).args(arguments).output() { + Ok(output) => output, + Err(error) if error.kind() == std::io::ErrorKind::NotFound => return None, + Err(error) => return Some(Err(format!("{program}: {error}"))), + }; + if output.status.success() { + Some(Ok(String::from_utf8_lossy(&output.stdout).into_owned())) + } else { + let stderr = String::from_utf8_lossy(&output.stderr); + let first = stderr + .lines() + .find(|line| !line.trim().is_empty()) + .unwrap_or("no output"); + Some(Err(format!( + "`{program} {}` failed: {}", + arguments.join(" "), + first.trim() + ))) + } +} + +fn host_state(host: Host, home: &Path) -> Option> { + let program = host.program(); + let plugins = match host_output(program, &["plugin", "list", "--json"])? { + Ok(text) => text, + Err(error) => return Some(Err(error)), + }; + let markets = match host_output(program, &["plugin", "marketplace", "list", "--json"])? { + Ok(text) => text, + Err(error) => return Some(Err(error)), + }; + Some(match host { + Host::Claude => parse_claude_plugins(&plugins).and_then(|plugins| { + Ok(HostState { + plugins, + marketplaces: parse_claude_marketplaces(&markets)?, + }) + }), + Host::Codex => { + let config = read(&home.join(".codex/config.toml")); + parse_codex_plugins(&plugins).and_then(|plugins| { + Ok(HostState { + plugins, + marketplaces: parse_codex_marketplaces(&markets, config.as_deref())?, + }) + }) + } + }) +} + +/// Claude Code settings files that can enable a plugin: the user file and each known project's +/// shared and local file. +fn settings_entries(home: &Path, catalog: &Catalog) -> Vec { + let mut entries = Vec::new(); + let user = home.join(".claude/settings.json"); + if let Some(text) = read(&user) { + entries.extend(parse_settings( + &text, + &user.to_string_lossy(), + "user", + None, + catalog, + )); + } + let projects: Vec = read(&home.join(".claude.json")) + .and_then(|text| serde_json::from_str::(&text).ok()) + .and_then(|value| { + value + .get("projects") + .and_then(serde_json::Value::as_object) + .map(|projects| projects.keys().cloned().collect()) + }) + .unwrap_or_default(); + for project in projects { + for (file, scope) in [ + ("settings.json", "project"), + ("settings.local.json", "local"), + ] { + let path = Path::new(&project).join(".claude").join(file); + if let Some(text) = read(&path) { + entries.extend(parse_settings( + &text, + &path.to_string_lossy(), + scope, + Some(&project), + catalog, + )); + } + } + } + entries.sort(); + entries.dedup(); + entries +} + +/// Read everything. +pub fn collect(catalog: &Catalog, hosts: &[Host]) -> Inventory { + let home = home(); + let mut inventory = Inventory::default(); + for host in hosts { + let state = match host_state(*host, &home) { + None => None, + Some(Ok(state)) => Some(state), + Some(Err(error)) => { + inventory.broken.push((*host, error)); + None + } + }; + match host { + Host::Claude => inventory.claude = state, + Host::Codex => inventory.codex = state, + } + } + let mut names: Vec<&str> = catalog + .products + .iter() + .flat_map(|product| product.binaries.iter().map(|binary| binary.name.as_str())) + .collect(); + names.push("b10x"); + names.sort_unstable(); + names.dedup(); + inventory.binaries = names + .into_iter() + .map(|name| BinaryState { + name: name.to_owned(), + copies: copies_on_path(name), + }) + .collect(); + if inventory.claude.is_some() { + inventory.settings = settings_entries(&home, catalog); + } + inventory.cargo = !copies_on_path("cargo").is_empty(); + inventory +} + +#[cfg(test)] +mod tests { + use super::*; + + const CLAUDE_LIST: &str = include_str!("../tests/fixtures/claude-legacy-list.json"); + const CLAUDE_MARKETS: &str = include_str!("../tests/fixtures/claude-legacy-marketplaces.json"); + const CODEX_LIST: &str = include_str!("../tests/fixtures/codex-legacy-list.json"); + const CODEX_MARKETS: &str = include_str!("../tests/fixtures/codex-legacy-marketplaces.json"); + const CODEX_CONFIG: &str = include_str!("../tests/fixtures/codex-legacy-config.toml"); + + #[test] + fn recorded_claude_output_parses() { + let plugins = parse_claude_plugins(CLAUDE_LIST).unwrap(); + assert!(plugins + .iter() + .any(|p| p.id() == "workspace-hygiene@beyond10x" + && p.version.as_deref() == Some("0.10.0"))); + let markets = parse_claude_marketplaces(CLAUDE_MARKETS).unwrap(); + let legacy = markets.iter().find(|m| m.name == "beyond10x").unwrap(); + assert_eq!(legacy.reference.as_deref(), Some("0.10.0")); + } + + #[test] + fn recorded_codex_output_parses_with_refs_from_config() { + let plugins = parse_codex_plugins(CODEX_LIST).unwrap(); + assert!(plugins.iter().any(|p| p.id() == "aep-plan@beyond10x")); + let markets = parse_codex_marketplaces(CODEX_MARKETS, Some(CODEX_CONFIG)).unwrap(); + assert_eq!(markets[0].reference.as_deref(), Some("0.10.0")); + } + + #[test] + fn settings_keep_only_entries_the_catalog_cares_about() { + let catalog = Catalog::embedded(); + let text = r#"{"enabledPlugins":{"beyond10x@beyond10x":true,"ess-schema@beyond10x":false,"brain@org-brain":true,"worktree@b10x":true}}"#; + let entries = parse_settings( + text, + "/p/.claude/settings.local.json", + "local", + Some("/p"), + &catalog, + ); + let names: Vec<&str> = entries.iter().map(|e| e.plugin.as_str()).collect(); + assert_eq!( + names, + [ + "beyond10x@beyond10x", + "ess-schema@beyond10x", + "worktree@b10x" + ] + ); + } +} diff --git a/crates/b10x/src/main.rs b/crates/b10x/src/main.rs new file mode 100644 index 0000000..2f8b9af --- /dev/null +++ b/crates/b10x/src/main.rs @@ -0,0 +1,736 @@ +//! `b10x`: installs, migrates and checks the Beyond10x agent plugins and the binaries they drive. + +mod apply; +mod catalog; +mod check; +mod install; +mod inventory; +mod pins; +mod plan; +mod resolve; +mod skill; +mod version; + +use std::collections::BTreeSet; +use std::path::PathBuf; +use std::process::ExitCode; + +use clap::{Parser, Subcommand, ValueEnum}; + +use crate::catalog::Catalog; +use crate::inventory::Host; +use crate::plan::{Level, Plan}; +use crate::resolve::Source; + +/// Install, migrate and check the Beyond10x agent plugins and their binaries. +#[derive(Debug, Parser)] +#[command(name = "b10x", version, about)] +struct Cli { + #[command(subcommand)] + command: Top, +} + +#[derive(Debug, Subcommand)] +enum Top { + /// Plan the named products (`aep,ess`): their plugins and CLIs, installed or brought current. + /// Nothing else is touched. Writes nothing but the plan file; apply it after confirmation. + Init { + /// Products, comma separated: aep, ess, worktree, connectors. + #[arg(value_delimiter = ',')] + products: Vec, + /// How to install CLIs; default: prebuilt, cargo when a release has no archive for this machine. + #[arg(long, value_enum)] + method: Option, + /// Hosts to plan for. + #[arg(long, value_enum, default_value_t = Hosts::All)] + host: Hosts, + /// Print the plan as JSON. + #[arg(long)] + json: bool, + /// Write the JSON plan to this file, for `setup apply --plan`. + #[arg(long)] + out: Option, + }, + /// Check installed products (or the named ones) against what is newest and plan the upgrade. + Upgrade { + /// Products, comma separated; default: every installed product. + #[arg(value_delimiter = ',')] + products: Vec, + /// How to install CLIs; default: prebuilt, cargo when a release has no archive for this machine. + #[arg(long, value_enum)] + method: Option, + /// Hosts to plan for. + #[arg(long, value_enum, default_value_t = Hosts::All)] + host: Hosts, + /// Print the plan as JSON. + #[arg(long)] + json: bool, + /// Write the JSON plan to this file, for `setup apply --plan`. + #[arg(long)] + out: Option, + }, + /// Plan and apply plugin and binary changes. + Setup { + #[command(subcommand)] + command: Setup, + }, + /// Drift check for a session-start hook; prints only problems; always exits 0. Refreshes the + /// newest-release record from GitHub at most once a day, within a few seconds. + Check, + /// Pin a CLI for this repository in `b10x.toml` (here, or the nearest one above): + /// `0.32.0` is that release, `0.59` the newest `0.59.x`. + Pin { + /// Catalog binary, e.g. `ess`. + name: String, + /// `x.y.z` or `x.y`. + version: String, + }, + /// Remove a CLI's pin from the nearest `b10x.toml`. + Unpin { + /// Catalog binary, e.g. `ess`. + name: String, + }, + /// Print an installed skill or agent (`ess:specifying`), or list a plugin's (`ess`). A host loads + /// new plugins only in a new session; this works in the session that installed them. + Skill { + /// `` or `:`. + id: String, + }, + /// Install one catalog binary at an exact release. + Install { + /// Binary name, e.g. `ess`. + name: String, + /// Release tag; defaults to this repository's pin (`b10x.toml`), else the newest release. + #[arg(long)] + tag: Option, + /// Target directory; defaults to `~/.local/bin` (prebuilt) or `~/.cargo/bin` (cargo). + #[arg(long)] + dir: Option, + /// How to install; default: prebuilt when the release has an archive for this machine, else cargo. + #[arg(long, value_enum)] + method: Option, + }, +} + +#[derive(Debug, Subcommand)] +enum Setup { + /// Read the installed state and print what setup would change. Writes nothing. + Plan { + /// Products to end up with, comma separated (`aep,ess,worktree`), or `none`. + /// Default: the products present now. + #[arg(long, value_delimiter = ',')] + products: Option>, + /// Hosts to plan for. + #[arg(long, value_enum, default_value_t = Hosts::All)] + host: Hosts, + /// How to install CLIs; default: prebuilt, cargo when a release has no archive for this machine. + #[arg(long, value_enum)] + method: Option, + /// Print the plan as JSON. + #[arg(long)] + json: bool, + /// Also write the JSON plan to this file, for `apply --plan`. + #[arg(long)] + out: Option, + }, + /// Apply a plan made by `setup plan --out`. + Apply { + /// The plan file. + #[arg(long)] + plan: PathBuf, + /// Confirm that the user approved every listed action. + #[arg(long)] + yes: bool, + }, + /// Restore the settings and binaries a snapshot holds (default: the newest). + Undo { + /// Snapshot directory or its timestamp name. + snapshot: Option, + }, + /// Print the setup instructions an agent follows (the `b10x:init` skill). + Guide, +} + +/// The `init` skill, printed by `b10x setup guide` so an agent without the plugin reads the same text. +const GUIDE: &str = include_str!("../../../plugins/b10x/skills/init/SKILL.md"); + +#[derive(Debug, Clone, Copy, ValueEnum)] +enum MethodArg { + /// The release's checksummed prebuilt archive. + Prebuilt, + /// `cargo install` from the release tag. + Cargo, +} + +impl MethodArg { + fn method(self) -> catalog::Method { + match self { + MethodArg::Prebuilt => catalog::Method::Prebuilt, + MethodArg::Cargo => catalog::Method::Cargo, + } + } +} + +#[derive(Debug, Clone, Copy, ValueEnum)] +enum Hosts { + Claude, + Codex, + All, +} + +impl Hosts { + fn list(self) -> Vec { + match self { + Hosts::Claude => vec![Host::Claude], + Hosts::Codex => vec![Host::Codex], + Hosts::All => vec![Host::Claude, Host::Codex], + } + } +} + +fn main() -> ExitCode { + let cli = Cli::parse(); + let result = match cli.command { + Top::Check => { + check::run(&Catalog::embedded()); + Ok(ExitCode::SUCCESS) + } + Top::Pin { name, version } => pin(&name, &version), + Top::Unpin { name } => unpin(&name), + Top::Install { + name, + tag, + dir, + method, + } => install_one(&name, tag, dir, method), + Top::Init { + products, + method, + host, + json, + out, + } => { + if products.is_empty() { + Err("name the products: aep, ess, worktree, connectors (the /b10x:init skill asks the user which)".to_owned()) + } else { + emit( + make_plan( + Some(clean(products)), + host.list(), + true, + method.map(MethodArg::method), + false, + ), + json, + out.as_deref(), + ) + } + } + Top::Upgrade { + products, + method, + host, + json, + out, + } => { + let selection = (!products.is_empty()).then(|| clean(products)); + emit( + make_plan( + selection, + host.list(), + true, + method.map(MethodArg::method), + true, + ), + json, + out.as_deref(), + ) + } + Top::Skill { id } => skill::text( + &inventory::home(), + &Catalog::embedded().marketplace.name, + &id, + ) + .map(|text| { + print!("{text}"); + ExitCode::SUCCESS + }), + Top::Setup { command } => match command { + Setup::Plan { + products, + host, + method, + json, + out, + } => setup_plan(products, host, method, json, out.as_deref()), + Setup::Apply { plan, yes } => setup_apply(&plan, yes), + Setup::Undo { snapshot } => setup_undo(snapshot), + Setup::Guide => { + print!("{GUIDE}"); + Ok(ExitCode::SUCCESS) + } + }, + }; + match result { + Ok(code) => code, + Err(error) => { + eprintln!("b10x: {error}"); + ExitCode::from(1) + } + } +} + +/// `B10X_MARKETPLACE`: register and read the marketplace from this source (a local checkout or +/// `owner/repo`) instead of the catalog's repository. For testing a marketplace before it is +/// published; users never need it. +fn marketplace_override() -> Option { + std::env::var("B10X_MARKETPLACE") + .ok() + .filter(|value| !value.is_empty()) +} + +fn clean(products: Vec) -> BTreeSet { + products + .into_iter() + .map(|id| id.trim().to_owned()) + .filter(|id| !id.is_empty() && id != "none") + .collect() +} + +fn make_plan( + selection: Option>, + hosts: Vec, + only: bool, + method: Option, + upgrade: bool, +) -> Result { + let overridden = marketplace_override(); + let mut embedded = Catalog::embedded(); + if let Some(source) = &overridden { + embedded.marketplace.repository.clone_from(source); + } + let inventory = inventory::collect(&embedded, &[Host::Claude, Host::Codex]); + let local = overridden + .as_deref() + .map(std::path::Path::new) + .filter(|path| path.is_dir()); + // The marketplace's newest release; a host clone older than it would call old plugins current + // (the clone is refreshed only by applying), so the plan then reads the repository itself. + let upstream = local + .is_none() + .then(|| embedded.marketplace.repository.clone()); + let newest = upstream.as_deref().and_then(resolve::latest_tag); + let source = match (local, resolve::clone_of(&inventory, &embedded)) { + (Some(path), _) => Source::Clone(path), + (None, Some(path)) if !resolve::stale(path, newest.as_deref()) => Source::Clone(path), + (None, _) => Source::Remote(&embedded.marketplace.repository), + }; + let mut catalog = resolve::catalog(&source); + if let Some(source) = &overridden { + catalog.marketplace.repository.clone_from(source); + } + if let Some(selection) = &selection { + for id in selection { + if catalog.product(id).is_none() { + let known: Vec<&str> = catalog.products.iter().map(|p| p.id.as_str()).collect(); + return Err(format!( + "unknown product `{id}`; choose from {}", + known.join(", ") + )); + } + } + } + let home = inventory::home(); + let pins = repository_pins(&catalog, &home)?; + let mut resolved = resolve::resolve(&catalog, &source, pins.as_ref()); + if let (Some(repository), Some(tag)) = (upstream, newest) { + resolved.latest.insert(repository, tag); + } + resolve::remember(&home, &resolved); + let context = plan::Context { + catalog: &catalog, + resolved: &resolved, + selection, + hosts, + home: &home, + only, + method, + target: install::target().ok().map(str::to_owned), + upgrade, + }; + Ok(plan::plan(&context, &inventory)) +} + +/// The pins that apply in the current directory; every pinned name must be a catalog binary. +fn repository_pins( + catalog: &Catalog, + home: &std::path::Path, +) -> Result, String> { + let Ok(here) = std::env::current_dir() else { + return Ok(None); + }; + let found = pins::read(&here, home)?; + if let Some(file) = &found { + for name in file.pins.keys() { + if catalog.binary(name).is_none() { + return Err(format!( + "{} pins `{name}`, which is not a catalog binary; `b10x unpin {name}` removes it", + file.path.display() + )); + } + } + } + Ok(found) +} + +fn pin(name: &str, version: &str) -> Result { + let catalog = Catalog::embedded(); + let (_, binary) = catalog.binary(name).ok_or_else(|| { + let known: Vec<&str> = catalog + .products + .iter() + .flat_map(|product| product.binaries.iter().map(|binary| binary.name.as_str())) + .collect(); + format!( + "`{name}` is not a catalog binary; choose from {}", + known.join(", ") + ) + })?; + let spec = pins::Spec::parse(version)?; + let repository = binary.install.repository(); + let newest = resolve::latest_tag(repository); + let tag = resolve::pinned_tag(repository, spec, newest.as_deref()) + .filter(|tag| resolve::release_exists(repository, tag)) + .ok_or_else(|| format!("{repository} has no release matching `{version}`"))?; + let home = inventory::home(); + let here = std::env::current_dir().map_err(|error| error.to_string())?; + if here == home { + return Err(format!( + "run it inside a repository: a {} in $HOME pins nothing", + pins::FILE + )); + } + let path = pins::find(&here, &home).unwrap_or_else(|| here.join(pins::FILE)); + pins::set(&path, name, version.trim().trim_start_matches('v'))?; + println!( + "pinned {name} to {version} ({tag}) in {}; `b10x install {name}` installs it, commit the file to share the pin", + path.display() + ); + if let Some(newest) = newest.filter(|newest| !version::same(newest, &tag)) { + println!("the newest release is {newest}; `b10x unpin {name}` follows it again"); + } + Ok(ExitCode::SUCCESS) +} + +fn unpin(name: &str) -> Result { + let home = inventory::home(); + let here = std::env::current_dir().map_err(|error| error.to_string())?; + let path = pins::find(&here, &home) + .ok_or_else(|| format!("no {} here or above; nothing is pinned", pins::FILE))?; + if pins::remove(&path, name)? { + println!( + "unpinned {name} in {}; `b10x upgrade` follows the newest release", + path.display() + ); + Ok(ExitCode::SUCCESS) + } else { + Err(format!("{} does not pin `{name}`", path.display())) + } +} + +fn setup_plan( + products: Option>, + host: Hosts, + method: Option, + json: bool, + out: Option<&std::path::Path>, +) -> Result { + emit( + make_plan( + products.map(clean), + host.list(), + false, + method.map(MethodArg::method), + false, + ), + json, + out, + ) +} + +/// Print a plan (text, or JSON with `--json`) and write it to `--out`. +fn emit( + plan: Result, + json: bool, + out: Option<&std::path::Path>, +) -> Result { + let plan = plan?; + let text = serde_json::to_string_pretty(&plan).map_err(|error| error.to_string())?; + if let Some(out) = out { + if let Some(parent) = out.parent() { + std::fs::create_dir_all(parent) + .map_err(|error| format!("{}: {error}", parent.display()))?; + } + std::fs::write(out, &text).map_err(|error| format!("{}: {error}", out.display()))?; + } + if json { + println!("{text}"); + } else { + print_plan(&plan); + if !plan.converged() { + match out { + Some(out) => println!( + "\nApply after the user confirms this list: b10x setup apply --plan {} --yes", + out.display() + ), + None => println!( + "\nWrite the plan with --out , then apply it after the user confirms." + ), + } + } + } + Ok(ExitCode::SUCCESS) +} + +fn level(level: Level) -> &'static str { + match level { + Level::Ok => "ok", + Level::Note => "note", + Level::Change => "change", + Level::Warn => "WARN", + } +} + +fn print_plan(plan: &Plan) { + let hosts: Vec<&str> = plan.hosts.iter().map(|host| host.program()).collect(); + println!("b10x setup plan (hosts: {})", hosts.join(", ")); + println!("\nProducts:"); + for offer in &plan.offers { + println!( + " [{}] {:<11} {}{}{}", + if offer.selected { "x" } else { " " }, + offer.id, + offer.summary, + if offer.present { " (present)" } else { "" }, + if offer.optional { " (optional)" } else { "" }, + ); + } + println!("\nFindings:"); + for finding in &plan.findings { + let host = finding.host.map_or("-", Host::program); + println!( + " {:<6} {:<6} {:<32} {}", + level(finding.level), + host, + finding.subject, + finding.detail + ); + } + let changes: Vec<&plan::Action> = plan.actions.iter().filter(|a| a.changes()).collect(); + if changes.is_empty() { + println!("\n{}", plan::nothing_to_change(plan)); + } else { + println!("\nActions ({}):", changes.len()); + for (index, action) in changes.iter().enumerate() { + println!(" {:>2}. {}", index + 1, action.describe()); + } + } + let refreshes: Vec<&plan::Action> = plan.actions.iter().filter(|a| !a.changes()).collect(); + if !refreshes.is_empty() { + println!("\nAlso runs (refreshes the marketplace snapshot; changes no version):"); + for action in refreshes { + println!(" - {}", action.describe()); + } + } + if !plan.next.is_empty() { + println!("\nNext:"); + for line in &plan.next { + println!(" {line}"); + } + } +} + +fn setup_apply(path: &PathBuf, yes: bool) -> Result { + let text = + std::fs::read_to_string(path).map_err(|error| format!("{}: {error}", path.display()))?; + let plan: Plan = + serde_json::from_str(&text).map_err(|error| format!("{}: {error}", path.display()))?; + if plan.format != plan::FORMAT { + return Err(format!( + "{} is not a `{}` document", + path.display(), + plan::FORMAT + )); + } + let embedded = Catalog::embedded(); + let inventory = inventory::collect(&embedded, &[Host::Claude, Host::Codex]); + apply::fresh(&plan, &inventory)?; + if !yes { + print_plan(&plan); + eprintln!( + "\nb10x: nothing applied. Confirm every action with the user, then rerun with --yes." + ); + return Ok(ExitCode::from(2)); + } + let snapshot = apply::snapshot(&plan)?; + println!("snapshot: {}", snapshot.display()); + let total = plan.actions.len(); + for (index, action) in plan.actions.iter().enumerate() { + match apply::run(action) { + Ok(()) => println!(" ok {}/{total} {}", index + 1, action.describe()), + Err(error) => { + println!( + " FAIL {}/{total} {}\n {error}", + index + 1, + action.describe() + ); + for rest in &plan.actions[index + 1..] { + println!(" skip {}", rest.describe()); + } + println!("Undo with: b10x setup undo {}", snapshot.display()); + return Ok(ExitCode::from(1)); + } + } + } + let selection: BTreeSet = plan + .offers + .iter() + .filter(|offer| offer.selected) + .map(|offer| offer.id.clone()) + .collect(); + let after = make_plan( + Some(selection), + plan.hosts.clone(), + plan.only, + plan.method, + plan.upgrade, + )?; + println!("\nAfter:"); + print_plan(&after); + if after.converged() { + println!( + "\nConverged. New plugins load in a new session: restart Claude Code (or /reload-plugins) or start a new Codex thread. In this session, `b10x skill ` lists a plugin's skills and `b10x skill :` prints one." + ); + Ok(ExitCode::SUCCESS) + } else { + println!("\nNot converged: run `b10x setup plan` again and review the remaining actions."); + Ok(ExitCode::from(1)) + } +} + +fn setup_undo(snapshot: Option) -> Result { + let root = apply::snapshots(); + let directory = match snapshot { + Some(given) if given.contains('/') => PathBuf::from(given), + Some(stamp) => root.join(stamp), + None => { + let mut stamps: Vec = std::fs::read_dir(&root) + .map_err(|error| format!("{}: {error}", root.display()))? + .filter_map(|entry| entry.ok()?.file_name().to_str()?.parse().ok()) + .collect(); + stamps.sort_unstable(); + root.join(stamps.last().ok_or("no snapshot to restore")?.to_string()) + } + }; + for restored in apply::undo(&directory)? { + println!("restored {restored}"); + } + // The restored settings name marketplaces whose snapshots apply removed; each host rebuilds them. + for argv in [ + ["claude", "plugin", "marketplace", "update"], + ["codex", "plugin", "marketplace", "upgrade"], + ] { + match std::process::Command::new(argv[0]) + .args(&argv[1..]) + .output() + { + Ok(output) if output.status.success() => println!("refreshed: {}", argv.join(" ")), + Ok(output) => println!( + "could not refresh ({}): {}; run it yourself", + argv.join(" "), + String::from_utf8_lossy(&output.stderr) + .lines() + .next() + .unwrap_or("no output") + ), + Err(_) => {} + } + } + println!("Restart Claude Code and start a new Codex thread."); + Ok(ExitCode::SUCCESS) +} + +fn install_one( + name: &str, + tag: Option, + dir: Option, + method: Option, +) -> Result { + let catalog = Catalog::embedded(); + let (_, binary) = catalog + .binary(name) + .ok_or_else(|| format!("`{name}` is not a catalog binary"))?; + let repository = binary.install.repository(); + let pinned = match &tag { + Some(_) => None, + None => repository_pins(&catalog, &inventory::home())? + .and_then(|file| Some((file.pins.get(name)?.clone(), file.path))), + }; + let tag = match (tag, pinned) { + (Some(tag), _) => tag, + (None, Some((spec, file))) => { + let parsed = pins::Spec::parse(&spec)?; + let tag = resolve::pinned_tag( + repository, + parsed, + resolve::latest_tag(repository).as_deref(), + ) + .ok_or_else(|| { + format!( + "{repository} has no release matching the pin {spec} in {}", + file.display() + ) + })?; + println!("{name} is pinned to {spec} by {}", file.display()); + tag + } + (None, None) => resolve::latest_tag(repository) + .ok_or_else(|| format!("no release found for {repository}"))?, + }; + let target = install::target().ok(); + if !binary.runs_on(target) { + return Err(format!( + "`{name}` runs on {} only", + binary.platforms.join(", ") + )); + } + let method = if let Some(method) = method { + method.method() + } else { + let listed = binary.install.archive.is_some() + && target.is_some_and(|target| { + resolve::fetch(&format!( + "/{repository}/releases/download/{tag}/SHA256SUMS" + )) + .is_some_and(|sums| install::listed_targets(&sums, name, &tag).contains(target)) + }); + if listed { + catalog::Method::Prebuilt + } else if binary.install.cargo.is_some() && !inventory::copies_on_path("cargo").is_empty() { + catalog::Method::Cargo + } else { + return Err(format!( + "{tag} of `{name}` has no prebuilt archive for {} and `cargo` is not on PATH", + target.unwrap_or("this machine") + )); + } + }; + let home = inventory::home(); + let directory = dir.unwrap_or_else(|| match method { + catalog::Method::Prebuilt => home.join(".local/bin"), + catalog::Method::Cargo => home.join(".cargo/bin"), + }); + let path = install::install(name, &tag, method, &binary.install, &directory)?; + println!("installed {name} {tag} at {}", path.display()); + Ok(ExitCode::SUCCESS) +} diff --git a/crates/b10x/src/pins.rs b/crates/b10x/src/pins.rs new file mode 100644 index 0000000..05bac8f --- /dev/null +++ b/crates/b10x/src/pins.rs @@ -0,0 +1,274 @@ +//! Per-repository CLI pins. A repository may commit `b10x.toml`: +//! +//! ```toml +//! [pins] +//! ess = "0.32.0" # exactly this release +//! aep = "0.59" # the newest 0.59.x +//! ``` +//! +//! `b10x` finds it from the current directory upward, stopping below `$HOME` or at the filesystem +//! root, and resolves a pinned CLI to its pinned release instead of the newest. + +use std::collections::BTreeMap; +use std::path::{Path, PathBuf}; + +use serde::{Deserialize, Serialize}; + +use crate::version; + +/// The pin file's name. +pub const FILE: &str = "b10x.toml"; + +/// What a pin asks for. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum Spec { + /// Exactly `x.y.z`. + Exact((u64, u64, u64)), + /// The newest `x.y.*`. + Line(u64, u64), +} + +impl Spec { + /// Parse `0.32.0` or `0.59` (a leading `v` is allowed). + pub fn parse(text: &str) -> Result { + let bare = text.trim().trim_start_matches('v'); + if let Some(key) = version::key(bare) { + return Ok(Spec::Exact(key)); + } + let mut parts = bare.split('.'); + match ( + parts.next().and_then(|part| part.parse().ok()), + parts.next().and_then(|part| part.parse().ok()), + parts.next(), + ) { + (Some(major), Some(minor), None) => Ok(Spec::Line(major, minor)), + _ => Err(format!( + "`{text}` is not a version: use `x.y.z` for one release or `x.y` for the newest `x.y.*`" + )), + } + } + + /// Whether a version (or tag) satisfies this pin. + #[must_use] + pub fn matches(&self, found: &str) -> bool { + match (self, version::key(found)) { + (Spec::Exact(want), Some(have)) => *want == have, + (Spec::Line(major, minor), Some((a, b, _))) => (*major, *minor) == (a, b), + _ => false, + } + } + + /// The newest tag that satisfies this pin. + #[must_use] + pub fn select<'a>(&self, tags: impl IntoIterator) -> Option<&'a str> { + tags.into_iter() + .filter(|tag| self.matches(tag)) + .max_by_key(|tag| version::key(tag)) + } + + /// Whether `version` is newer than anything this pin allows. + #[must_use] + pub fn older_than(&self, version: &str) -> bool { + match (self, version::key(version)) { + (Spec::Exact(want), Some(have)) => have > *want, + (Spec::Line(major, minor), Some((a, b, _))) => (a, b) > (*major, *minor), + _ => false, + } + } +} + +/// A pin resolved against the releases, as a plan carries it. +#[derive(Debug, Clone, PartialEq, Eq, Deserialize, Serialize)] +pub struct Pinned { + /// What the file asks for: `0.32.0` or `0.59`. + pub spec: String, + /// The release it resolves to; `None` when none could be read. + pub tag: Option, + /// The pin file. + pub file: String, +} + +/// A pin file and its `[pins]`. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct PinFile { + /// Where it is. + pub path: PathBuf, + /// CLI name → spec, as written. + pub pins: BTreeMap, +} + +/// The nearest `b10x.toml` from `start` upward. `$HOME` itself and everything above it are not +/// searched, so a stray file in the home directory pins nothing. +#[must_use] +pub fn find(start: &Path, home: &Path) -> Option { + let mut directory = Some(start); + while let Some(current) = directory { + if current == home { + return None; + } + let candidate = current.join(FILE); + if candidate.is_file() { + return Some(candidate); + } + directory = current.parent(); + } + None +} + +fn table(path: &Path) -> Result { + let text = + std::fs::read_to_string(path).map_err(|error| format!("{}: {error}", path.display()))?; + text.parse::() + .map_err(|error| format!("{}: {error}", path.display())) +} + +/// Read one pin file; every pin must parse. +pub fn load(path: &Path) -> Result { + let table = table(path)?; + let mut pins = BTreeMap::new(); + if let Some(value) = table.get("pins") { + let entries = value + .as_table() + .ok_or_else(|| format!("{}: `pins` must be a table", path.display()))?; + for (name, spec) in entries { + let spec = spec + .as_str() + .ok_or_else(|| format!("{}: pin `{name}` must be a string", path.display()))?; + Spec::parse(spec).map_err(|error| format!("{}: {name}: {error}", path.display()))?; + pins.insert(name.clone(), spec.to_owned()); + } + } + Ok(PinFile { + path: path.to_path_buf(), + pins, + }) +} + +/// The pins that apply in `start`, if a pin file is found. +pub fn read(start: &Path, home: &Path) -> Result, String> { + find(start, home).map(|path| load(&path)).transpose() +} + +/// Set one pin in `path`, creating the file when needed; other content is kept. +pub fn set(path: &Path, name: &str, spec: &str) -> Result<(), String> { + let mut table = if path.is_file() { + table(path)? + } else { + toml::Table::new() + }; + let pins = table + .entry("pins") + .or_insert_with(|| toml::Value::Table(toml::Table::new())) + .as_table_mut() + .ok_or_else(|| format!("{}: `pins` must be a table", path.display()))?; + pins.insert(name.to_owned(), toml::Value::String(spec.to_owned())); + write(path, &table) +} + +/// Remove one pin from `path`; the file goes when nothing is left in it. `false` when not pinned. +pub fn remove(path: &Path, name: &str) -> Result { + let mut table = table(path)?; + let Some(pins) = table.get_mut("pins").and_then(toml::Value::as_table_mut) else { + return Ok(false); + }; + if pins.remove(name).is_none() { + return Ok(false); + } + if pins.is_empty() { + table.remove("pins"); + } + if table.is_empty() { + std::fs::remove_file(path).map_err(|error| format!("{}: {error}", path.display()))?; + return Ok(true); + } + write(path, &table)?; + Ok(true) +} + +fn write(path: &Path, table: &toml::Table) -> Result<(), String> { + let text = toml::to_string(table).map_err(|error| error.to_string())?; + std::fs::write(path, text).map_err(|error| format!("{}: {error}", path.display())) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn scratch(name: &str) -> PathBuf { + let root = std::env::temp_dir().join(format!("b10x-pins-{name}-{}", std::process::id())); + let _ = std::fs::remove_dir_all(&root); + std::fs::create_dir_all(&root).unwrap(); + root + } + + #[test] + fn specs_parse_exact_and_minor_lines() { + assert_eq!(Spec::parse("0.32.0"), Ok(Spec::Exact((0, 32, 0)))); + assert_eq!(Spec::parse("v0.59"), Ok(Spec::Line(0, 59))); + assert!(Spec::parse("0").is_err()); + assert!(Spec::parse("latest").is_err()); + assert!(Spec::parse("0.1.2.3").is_err()); + } + + #[test] + fn exact_and_minor_line_pins_select_their_release() { + let tags = ["0.58.4", "0.59.0", "0.59.3", "v0.59.1", "0.60.0"]; + assert_eq!( + Spec::parse("0.59").unwrap().select(tags.iter().copied()), + Some("0.59.3") + ); + assert_eq!( + Spec::parse("0.59.1").unwrap().select(tags.iter().copied()), + Some("v0.59.1") + ); + assert_eq!( + Spec::parse("0.61").unwrap().select(tags.iter().copied()), + None + ); + assert!(Spec::parse("0.59").unwrap().older_than("0.60.0")); + assert!(!Spec::parse("0.59").unwrap().older_than("0.59.9")); + assert!(Spec::parse("0.32.0").unwrap().older_than("0.32.1")); + } + + #[test] + fn the_nearest_pin_file_is_found_below_home_only() { + let home = scratch("lookup"); + let repo = home.join("work/repo"); + let deep = repo.join("crates/a/src"); + std::fs::create_dir_all(&deep).unwrap(); + assert_eq!(find(&deep, &home), None); + std::fs::write(home.join(FILE), "[pins]\ness = \"0.1.0\"\n").unwrap(); + assert_eq!(find(&deep, &home), None, "a file in $HOME pins nothing"); + std::fs::write( + repo.join(FILE), + "[pins]\ness = \"0.32.0\"\naep = \"0.59\"\n", + ) + .unwrap(); + let found = read(&deep, &home).unwrap().unwrap(); + assert_eq!(found.path, repo.join(FILE)); + assert_eq!(found.pins.get("aep").map(String::as_str), Some("0.59")); + std::fs::write(repo.join(FILE), "[pins]\ness = \"soon\"\n").unwrap(); + assert!(read(&deep, &home).is_err()); + std::fs::remove_dir_all(&home).unwrap(); + } + + #[test] + fn pins_are_set_and_removed_keeping_other_content() { + let root = scratch("edit"); + let path = root.join(FILE); + set(&path, "ess", "0.32.0").unwrap(); + set(&path, "aep", "0.59").unwrap(); + assert_eq!(load(&path).unwrap().pins.len(), 2); + assert!(remove(&path, "ess").unwrap()); + assert!(!remove(&path, "ess").unwrap()); + assert!(remove(&path, "aep").unwrap()); + assert!(!path.exists(), "an empty pin file is removed"); + std::fs::write(&path, "[other]\nkeep = true\n").unwrap(); + set(&path, "ess", "0.32").unwrap(); + remove(&path, "ess").unwrap(); + assert!(std::fs::read_to_string(&path) + .unwrap() + .contains("keep = true")); + std::fs::remove_dir_all(&root).unwrap(); + } +} diff --git a/crates/b10x/src/plan.rs b/crates/b10x/src/plan.rs new file mode 100644 index 0000000..ea4bfe8 --- /dev/null +++ b/crates/b10x/src/plan.rs @@ -0,0 +1,2066 @@ +//! From an inventory and resolved versions to findings and exact actions. Pure: no I/O, so every +//! rule is tested on recorded host output. +//! +//! The user-scope result is declarative: the base plugins and the selected products' plugins end up +//! installed from the catalog marketplace at the version it serves now; a catalog plugin of an +//! unselected product is removed. Legacy installs at `project` or `local` scope are migrated in +//! place, whatever the selection, because they belong to a project rather than to the user. + +use std::collections::{BTreeMap, BTreeSet}; +use std::path::Path; + +use serde::{Deserialize, Serialize}; +use sha2::{Digest, Sha256}; + +use crate::catalog::{Binary, Catalog, Install, Method}; +use crate::inventory::{Host, HostState, Installed, Inventory}; +use crate::resolve::Resolved; +use crate::version; + +/// The plan document format. +pub const FORMAT: &str = "b10x.setup-plan/1"; + +/// How much a finding matters. +#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Deserialize, Serialize)] +#[serde(rename_all = "kebab-case")] +pub enum Level { + /// Already right. + Ok, + /// Worth knowing; nothing to do. + Note, + /// An action below changes it. + Change, + /// Needs the user's attention; setup does not change it. + Warn, +} + +/// One observation, in words. +#[derive(Debug, Clone, PartialEq, Eq, Deserialize, Serialize)] +pub struct Finding { + /// How much it matters. + pub level: Level, + /// Host it concerns, if any. + pub host: Option, + /// Plugin, marketplace, binary or file. + pub subject: String, + /// What was found and what happens. + pub detail: String, +} + +/// One step `apply` performs. Commands are argument vectors, never shell text. +#[derive(Debug, Clone, PartialEq, Eq, Deserialize, Serialize)] +#[serde(tag = "kind", rename_all = "kebab-case")] +pub enum Action { + /// Refresh the marketplace snapshot; changes nothing the user chose. + Refresh { + /// Host. + host: Host, + /// Program and arguments. + argv: Vec, + }, + /// Run a host command. + Command { + /// Host. + host: Host, + /// Program and arguments. + argv: Vec, + /// Directory to run in (project and local scope). + cwd: Option, + /// Why. + reason: String, + }, + /// Remove one `enabledPlugins` key from a Claude Code settings file. + RemoveSetting { + /// Settings file. + file: String, + /// `name@marketplace`. + key: String, + /// Why. + reason: String, + }, + /// Replace a pinned marketplace declaration in Claude Code user settings with an unpinned one. + Unpin { + /// Settings file. + file: String, + /// Marketplace name. + name: String, + /// `owner/repo`. + repository: String, + /// Why. + reason: String, + }, + /// Install a binary at an exact release. + InstallBinary { + /// Executable name. + name: String, + /// Release tag. + tag: String, + /// Which way. + method: Method, + /// The ways it can be installed. + install: Install, + /// Directory it lands in. + directory: String, + /// Why. + reason: String, + }, +} + +impl Action { + /// Whether this action changes something the user would notice. + #[must_use] + pub fn changes(&self) -> bool { + !matches!(self, Action::Refresh { .. }) + } + + /// One line for a confirmation list. + #[must_use] + pub fn describe(&self) -> String { + match self { + Action::Refresh { argv, .. } => format!("refresh: {}", argv.join(" ")), + Action::Command { + argv, cwd, reason, .. + } => match cwd { + Some(cwd) => format!("{reason}: {} (in {cwd})", argv.join(" ")), + None => format!("{reason}: {}", argv.join(" ")), + }, + Action::RemoveSetting { file, key, reason } => { + format!("{reason}: remove `{key}` from {file}") + } + Action::Unpin { + file, name, reason, .. + } => format!("{reason}: rewrite marketplace `{name}` in {file}"), + Action::InstallBinary { + name, + tag, + method, + directory, + reason, + .. + } => format!( + "{reason}: install {name} {tag} into {directory} ({})", + match method { + Method::Prebuilt => "prebuilt archive", + Method::Cargo => "cargo install", + } + ), + } + } +} + +/// A product as offered to the user. +#[derive(Debug, Clone, PartialEq, Eq, Deserialize, Serialize)] +pub struct Offer { + /// Product id. + pub id: String, + /// One line. + pub summary: String, + /// Offered but not preselected. + pub optional: bool, + /// What the user wants to do, as the onboarding question offers it. + #[serde(default)] + pub intent: String, + /// Present now, current or legacy, on any host. + pub present: bool, + /// In this plan's selection. + pub selected: bool, +} + +/// The whole plan. +#[derive(Debug, Clone, PartialEq, Eq, Deserialize, Serialize)] +pub struct Plan { + /// Always [`FORMAT`]. + pub format: String, + /// Digest of the inventory the plan was made from; `apply` refuses another. + pub inventory_digest: String, + /// Hosts planned for. + pub hosts: Vec, + /// Products offered, with what is selected. + pub offers: Vec, + /// Versions the plan resolved. + pub resolved: Resolved, + /// What was found. + pub findings: Vec, + /// What `apply` does, in order. + pub actions: Vec, + /// What to do after applying: the next skill or command for the user. + #[serde(default)] + pub next: Vec, + /// Planned for the selected products only (`init`, `upgrade`). + #[serde(default)] + pub only: bool, + /// The install method asked for, if any. + #[serde(default)] + pub method: Option, + /// Planned as an upgrade: only what is installed is updated. + #[serde(default)] + pub upgrade: bool, +} + +impl Plan { + /// Whether applying changes anything. + #[must_use] + pub fn converged(&self) -> bool { + !self.actions.iter().any(Action::changes) + } +} + +/// The line a converged plan ends with. Qualified with the planned hosts when a host outside the +/// plan has changes pending, so it never reads as "everything is current" (#33). +#[must_use] +pub fn nothing_to_change(plan: &Plan) -> String { + let pending = plan.findings.iter().any(|finding| { + finding.level == Level::Warn && finding.host.is_some_and(|host| !plan.hosts.contains(&host)) + }); + if pending { + let hosts: Vec<&str> = plan.hosts.iter().map(|host| host.program()).collect(); + format!("Nothing to change for {}.", hosts.join(" and ")) + } else { + "Nothing to change.".to_owned() + } +} + +/// Digest of an inventory. +/// +/// # Panics +/// Never: an inventory always serializes. +#[must_use] +pub fn digest(inventory: &Inventory) -> String { + let bytes = serde_json::to_vec(inventory).expect("an inventory serializes"); + crate::install::hex(&Sha256::digest(bytes)) +} + +fn argv(parts: &[&str]) -> Vec { + parts.iter().map(|part| (*part).to_owned()).collect() +} + +/// Products present now, by current or legacy plugin name, on any host. +#[must_use] +pub fn present(catalog: &Catalog, inventory: &Inventory) -> BTreeSet { + let mut ids = BTreeSet::new(); + let hosts = [inventory.claude.as_ref(), inventory.codex.as_ref()]; + for plugin in hosts.into_iter().flatten().flat_map(|state| &state.plugins) { + if let Some(product) = catalog.product_of(catalog.current_name(&plugin.name)) { + ids.insert(product.id.clone()); + } + } + for entry in &inventory.settings { + if let Some((name, _)) = entry.plugin.rsplit_once('@') { + if let Some(product) = catalog.product_of(catalog.current_name(name)) { + ids.insert(product.id.clone()); + } + } + } + ids +} + +/// Everything a plan needs besides the inventory. +pub struct Context<'a> { + /// Catalog. + pub catalog: &'a Catalog, + /// Resolved versions. + pub resolved: &'a Resolved, + /// Selected product ids; `None` means "what is present now". + pub selection: Option>, + /// Hosts to plan for. + pub hosts: Vec, + /// The user's home, for default install directories and settings paths. + pub home: &'a Path, + /// Plan only the selected products (`init`, `upgrade`): nothing else is uninstalled, migrated or + /// removed. `false` makes the selection the whole desired state (`setup plan`). + pub only: bool, + /// How to install binaries; `None` picks the prebuilt archive when the release has one for + /// [`Context::target`], else cargo. + pub method: Option, + /// This machine's release-archive target triple; `None` when no archive is built for it. + pub target: Option, + /// Update only what is installed: no new plugins, and a host with nothing Beyond10x on it is + /// left alone (`upgrade`). + pub upgrade: bool, +} + +/// Make the plan. +#[must_use] +#[allow(clippy::too_many_lines)] +pub fn plan(context: &Context<'_>, inventory: &Inventory) -> Plan { + let catalog = context.catalog; + let present = present(catalog, inventory); + // With no explicit selection, what is installed stays installed, optional products included: + // leaving one out would uninstall it (connectors, trial 5). + let selected: BTreeSet = context.selection.clone().unwrap_or_else(|| present.clone()); + let offers = catalog + .products + .iter() + .map(|product| Offer { + id: product.id.clone(), + summary: product.summary.clone(), + optional: product.optional, + present: present.contains(&product.id), + selected: selected.contains(&product.id), + intent: product.intent.clone(), + }) + .collect(); + let mut findings = Vec::new(); + let mut actions = Vec::new(); + for host in &context.hosts { + let state = match host { + Host::Claude => inventory.claude.as_ref(), + Host::Codex => inventory.codex.as_ref(), + }; + let untouched = state.is_some_and(|state| !holds_beyond10x(catalog, state)); + if context.upgrade && untouched { + findings.push(Finding { + level: Level::Note, + host: Some(*host), + subject: host.program().to_owned(), + detail: format!( + "nothing from Beyond10x is installed here; `b10x init --host {}` adds it", + host.program() + ), + }); + continue; + } + match state { + Some(state) => plan_host( + context, + *host, + state, + &selected, + &mut findings, + &mut actions, + ), + None => match inventory.broken.iter().find(|(broken, _)| broken == host) { + Some((_, error)) => findings.push(Finding { + level: Level::Warn, + host: Some(*host), + subject: host.program().to_owned(), + detail: format!( + "{error}; skipped. Repair it with `{}`, then plan again", + match host { + Host::Claude => "claude plugin marketplace update", + Host::Codex => "codex plugin marketplace upgrade", + } + ), + }), + None => findings.push(Finding { + level: Level::Note, + host: Some(*host), + subject: host.program().to_owned(), + detail: format!("`{}` is not installed here; skipped", host.program()), + }), + }, + } + } + for (host, state) in [ + (Host::Claude, inventory.claude.as_ref()), + (Host::Codex, inventory.codex.as_ref()), + ] { + if let Some(state) = state.filter(|state| holds_beyond10x(catalog, state)) { + if !context.hosts.contains(&host) { + findings.push(elsewhere(context, host, state, &selected, inventory)); + } + } + } + if context.hosts.contains(&Host::Claude) && inventory.claude.is_some() { + plan_settings(context, &selected, inventory, &mut findings, &mut actions); + } + plan_binaries(context, inventory, &selected, &mut findings, &mut actions); + let mut next = Vec::new(); + for product in &catalog.products { + if selected.contains(&product.id) { + for plugin in &product.plugins { + next.push(format!( + "/{plugin}:init starts {plugin} here (this session, before a restart: `b10x skill {plugin}:init`)" + )); + } + } + } + next.push("/b10x:upgrade checks everything later; /b10x:init adds a product".to_owned()); + Plan { + format: FORMAT.to_owned(), + inventory_digest: digest(inventory), + hosts: context.hosts.clone(), + offers, + resolved: context.resolved.clone(), + findings, + actions, + next, + only: context.only, + method: context.method, + upgrade: context.upgrade, + } +} + +/// The finding for a host outside the plan that holds Beyond10x: what a plan for it would do, +/// planned read-only with the same rules and discarded (#33). A `Warn` when that plan changes +/// something, which [`nothing_to_change`] reads; a `Note` when the host is current. +fn elsewhere( + context: &Context<'_>, + host: Host, + state: &HostState, + selected: &BTreeSet, + inventory: &Inventory, +) -> Finding { + let mut findings = Vec::new(); + let mut actions = Vec::new(); + plan_host(context, host, state, selected, &mut findings, &mut actions); + if host == Host::Claude { + plan_settings(context, selected, inventory, &mut findings, &mut actions); + } + let program = host.program(); + let count = actions.iter().filter(|action| action.changes()).count(); + if count == 0 { + return Finding { + level: Level::Note, + host: Some(host), + subject: program.to_owned(), + detail: format!( + "Beyond10x plugins are installed for `{program}` too and are current for this selection" + ), + }; + } + // What the discarded plan's changes are, in the words of its own findings. + let mut tally = [0usize; CHANGE_KINDS.len()]; + for finding in findings.iter().filter(|f| f.level == Level::Change) { + tally[change_kind(&finding.detail)] += 1; + } + let parts: Vec = CHANGE_KINDS + .iter() + .zip(tally) + .filter(|(_, n)| *n > 0) + .map(|(kind, n)| format!("{n} {kind}{}", if n == 1 { "" } else { "s" })) + .collect(); + Finding { + level: Level::Warn, + host: Some(host), + subject: program.to_owned(), + detail: format!( + "Beyond10x plugins are installed for `{program}` too and this plan leaves them alone; \ + a plan for `{program}` has {count} action{} ({}); `--host {program}` or `--host all` applies them", + if count == 1 { "" } else { "s" }, + parts.join(", ") + ), + } +} + +/// The kinds of change [`elsewhere`] counts, in the order it names them. +const CHANGE_KINDS: [&str; 7] = [ + "legacy install", + "missing plugin", + "outdated plugin", + "disabled plugin", + "unselected plugin", + "marketplace change", + "other change", +]; + +/// Which of [`CHANGE_KINDS`] a `Change` finding's detail is, by the words `plan_host` and +/// `plan_settings` write. +fn change_kind(detail: &str) -> usize { + if detail.starts_with("legacy install") { + 0 + } else if detail.starts_with("not installed; install") { + 1 + } else if detail.contains("; upgrade to ") { + 2 + } else if detail.starts_with("disabled") { + 3 + } else if detail.starts_with("its product is not selected") { + 4 + } else if detail.starts_with("marketplace `") || detail.starts_with("retired marketplace") { + 5 + } else { + 6 + } +} + +/// Whether a host holds anything from Beyond10x: a current plugin, or a legacy one. +fn holds_beyond10x(catalog: &Catalog, state: &HostState) -> bool { + state.plugins.iter().any(|plugin| { + plugin.marketplace == catalog.marketplace.name + || catalog.retired_marketplace(&plugin.marketplace) + || catalog.retired_plugins.contains_key(&plugin.name) + }) +} + +fn scope_args(plugin: &Installed) -> Vec { + argv(&["--scope", &plugin.scope]) +} + +#[allow(clippy::too_many_lines)] +fn plan_host( + context: &Context<'_>, + host: Host, + state: &HostState, + selected: &BTreeSet, + findings: &mut Vec, + actions: &mut Vec, +) { + let catalog = context.catalog; + let name = catalog.marketplace.name.as_str(); + let repository = catalog.marketplace.repository.as_str(); + let program = host.program(); + let finding = |level, subject: &str, detail: String| Finding { + level, + host: Some(host), + subject: subject.to_owned(), + detail, + }; + + // 1. The marketplace: registered, unpinned, fresh. + match state.marketplaces.iter().find(|market| market.name == name) { + None => { + findings.push(finding( + Level::Change, + name, + format!("marketplace `{name}` is not registered; add {repository}"), + )); + actions.push(Action::Command { + host, + argv: argv(&[program, "plugin", "marketplace", "add", repository]), + cwd: None, + reason: format!("register marketplace `{name}`"), + }); + } + Some(market) if market.reference.is_some() => { + let pinned = market.reference.clone().unwrap_or_default(); + findings.push(finding( + Level::Change, + name, + format!("marketplace `{name}` is pinned to `{pinned}`; follow the default branch instead"), + )); + match host { + Host::Claude => { + actions.push(Action::Unpin { + file: context + .home + .join(".claude/settings.json") + .to_string_lossy() + .into_owned(), + name: name.to_owned(), + repository: repository.to_owned(), + reason: format!("unpin marketplace `{name}`"), + }); + actions.push(Action::Command { + host, + argv: argv(&[program, "plugin", "marketplace", "add", repository]), + cwd: None, + reason: format!("re-register marketplace `{name}` unpinned"), + }); + } + Host::Codex => { + actions.push(Action::Command { + host, + argv: argv(&[program, "plugin", "marketplace", "remove", name]), + cwd: None, + reason: format!("drop pinned marketplace `{name}`"), + }); + actions.push(Action::Command { + host, + argv: argv(&[program, "plugin", "marketplace", "add", repository]), + cwd: None, + reason: format!("re-register marketplace `{name}` unpinned"), + }); + } + } + } + Some(_) => actions.push(Action::Refresh { + host, + argv: match host { + Host::Claude => argv(&[program, "plugin", "marketplace", "update", name]), + Host::Codex => argv(&[program, "plugin", "marketplace", "upgrade", name]), + }, + }), + } + + // 2. User scope: exactly the base plus the selected products' plugins. + let mut desired: Vec<&str> = catalog.base.iter().map(String::as_str).collect(); + for product in &catalog.products { + if selected.contains(&product.id) { + desired.extend(product.plugins.iter().map(String::as_str)); + } + } + let user: BTreeMap<&str, &Installed> = state + .plugins + .iter() + .filter(|plugin| plugin.marketplace == name && plugin.scope == "user") + .map(|plugin| (plugin.name.as_str(), plugin)) + .collect(); + // Plugins a legacy install on this host stands in for: section 3 removes the legacy install, + // so an upgrade installs its replacement instead of only noting that it is missing. + let replacing: BTreeSet<&str> = state + .plugins + .iter() + .filter(|plugin| { + catalog.retired_marketplace(&plugin.marketplace) + || catalog.retired_plugins.contains_key(&plugin.name) + }) + .map(|plugin| catalog.current_name(&plugin.name)) + .collect(); + for plugin in &desired { + let id = format!("{plugin}@{name}"); + let target = context.resolved.plugins.get(*plugin); + match user.get(plugin) { + None if context.upgrade + && !catalog.base.iter().any(|base| base == plugin) + && !replacing.contains(plugin) => + { + findings.push(finding( + Level::Note, + &id, + "not installed; `b10x init` adds it".to_owned(), + )); + } + None => { + findings.push(finding( + Level::Change, + &id, + format!( + "not installed; install {}", + target.map_or("the served version".to_owned(), Clone::clone) + ), + )); + actions.push(install(host, &id, "user", None)); + } + Some(installed) => { + let current = installed.version.clone().unwrap_or_default(); + match target { + Some(target) if !version::same(¤t, target) => { + findings.push(finding( + Level::Change, + &id, + format!("installed {current}; upgrade to {target}"), + )); + actions.extend(upgrade(host, &id)); + } + _ => findings.push(finding(Level::Ok, &id, format!("current ({current})"))), + } + if !installed.enabled { + findings.push(finding( + Level::Change, + &id, + "disabled; enable it".to_owned(), + )); + actions.push(Action::Command { + host, + argv: argv(&[program, "plugin", "enable", &id]), + cwd: None, + reason: format!("enable `{id}`"), + }); + } + } + } + } + for (plugin, installed) in &user { + let managed = catalog.product_of(plugin).is_some(); + if managed && !desired.contains(plugin) && !context.only { + let id = installed.id(); + findings.push(finding( + Level::Change, + &id, + "its product is not selected; uninstall".to_owned(), + )); + actions.push(uninstall(host, installed)); + } + } + + // 3. Legacy installs: a retired name or a retired marketplace, at any scope. + let mut retire_markets = BTreeSet::new(); + // Two retired plugins can share one replacement (`aep-plan` and `aep-drive` are both `aep`): + // install it once per scope and project. + let mut placed = BTreeSet::new(); + let mut kept = BTreeSet::new(); + for plugin in &state.plugins { + let retired_market = catalog.retired_marketplace(&plugin.marketplace); + let retired_name = catalog.retired_plugins.contains_key(&plugin.name); + if !retired_market && !retired_name { + continue; + } + if !catalog.knows(&plugin.name) && !retired_market { + continue; + } + if context.only && !wanted(catalog, selected, catalog.current_name(&plugin.name)) { + kept.insert(plugin.marketplace.clone()); + continue; + } + let id = plugin.id(); + let replacement = catalog.current_name(&plugin.name).to_owned(); + let where_ = plugin + .project + .as_deref() + .map_or(String::new(), |project| format!(" in {project}")); + findings.push(finding( + Level::Change, + &id, + format!( + "legacy install ({} scope{where_}); replace with `{replacement}@{name}`", + plugin.scope + ), + )); + // A local install is this user's alone, so the user-scope install of a selected product + // already covers it. A project install is committed for everyone and is replaced in place. + let covered = plugin.scope == "local" && desired.contains(&replacement.as_str()); + if plugin.scope != "user" && !covered { + let already = state.plugins.iter().any(|other| { + other.name == replacement + && other.marketplace == name + && other.scope == plugin.scope + && other.project == plugin.project + }); + let first = placed.insert(( + replacement.clone(), + plugin.scope.clone(), + plugin.project.clone(), + )); + if !already && first && catalog.knows(&replacement) { + actions.push(install( + host, + &format!("{replacement}@{name}"), + &plugin.scope, + plugin.project.clone(), + )); + } + } + actions.push(uninstall(host, plugin)); + if retired_market { + retire_markets.insert(plugin.marketplace.clone()); + } + } + for market in &state.marketplaces { + if catalog.retired_marketplace(&market.name) { + retire_markets.insert(market.name.clone()); + } + } + for market in retire_markets { + if kept.contains(&market) { + continue; + } + if state.marketplaces.iter().any(|known| known.name == market) { + findings.push(finding( + Level::Change, + &market, + format!("retired marketplace `{market}`; remove it"), + )); + actions.push(Action::Command { + host, + argv: argv(&[program, "plugin", "marketplace", "remove", &market]), + cwd: None, + reason: format!("remove retired marketplace `{market}`"), + }); + } + } +} + +/// Whether a plugin belongs to what this plan is for: the base, or a selected product. +fn wanted(catalog: &Catalog, selected: &BTreeSet, plugin: &str) -> bool { + catalog.base.iter().any(|base| base == plugin) + || catalog + .product_of(plugin) + .is_some_and(|product| selected.contains(&product.id)) +} + +fn install(host: Host, id: &str, scope: &str, project: Option) -> Action { + let program = host.program(); + let argv = match host { + Host::Claude => argv(&[program, "plugin", "install", id, "--scope", scope]), + Host::Codex => argv(&[program, "plugin", "add", id]), + }; + Action::Command { + host, + argv, + cwd: project, + reason: format!("install `{id}`"), + } +} + +fn upgrade(host: Host, id: &str) -> Vec { + let program = host.program(); + match host { + Host::Claude => vec![Action::Command { + host, + argv: argv(&[program, "plugin", "update", id, "--scope", "user"]), + cwd: None, + reason: format!("upgrade `{id}`"), + }], + Host::Codex => vec![ + Action::Command { + host, + argv: argv(&[program, "plugin", "remove", id]), + cwd: None, + reason: format!("remove `{id}` before reinstalling"), + }, + Action::Command { + host, + argv: argv(&[program, "plugin", "add", id]), + cwd: None, + reason: format!("upgrade `{id}`"), + }, + ], + } +} + +fn uninstall(host: Host, plugin: &Installed) -> Action { + let program = host.program(); + let id = plugin.id(); + let argv = match host { + Host::Claude => { + let mut argv = argv(&[program, "plugin", "uninstall", &id]); + argv.extend(scope_args(plugin)); + argv + } + Host::Codex => argv(&[program, "plugin", "remove", &id]), + }; + Action::Command { + host, + argv, + cwd: plugin.project.clone(), + reason: format!("uninstall `{id}`"), + } +} + +/// Settings entries the host list does not report: enabled keys whose marketplace is gone or +/// retired. Their replacement is installed at the same scope when the catalog knows it. +fn plan_settings( + context: &Context<'_>, + selected: &BTreeSet, + inventory: &Inventory, + findings: &mut Vec, + actions: &mut Vec, +) { + let catalog = context.catalog; + let Some(state) = inventory.claude.as_ref() else { + return; + }; + let name = catalog.marketplace.name.as_str(); + for entry in &inventory.settings { + let Some((plugin, marketplace)) = entry.plugin.rsplit_once('@') else { + continue; + }; + let reported = state.plugins.iter().any(|installed| { + installed.id() == entry.plugin + && installed.scope == entry.scope + && installed.project == entry.project + }); + let registered = state + .marketplaces + .iter() + .any(|market| market.name == marketplace); + let retired = catalog.retired_marketplace(marketplace) + || catalog.retired_plugins.contains_key(plugin); + if reported || (registered && !retired) { + continue; + } + if context.only && !wanted(catalog, selected, catalog.current_name(plugin)) { + continue; + } + let committed = entry.scope == "project"; + findings.push(Finding { + level: if committed { + Level::Warn + } else { + Level::Change + }, + host: Some(Host::Claude), + subject: entry.plugin.clone(), + detail: format!( + "orphan entry in {}{}; remove it", + entry.file, + if committed { + " (a committed file: review the diff)" + } else { + "" + } + ), + }); + actions.push(Action::RemoveSetting { + file: entry.file.clone(), + key: entry.plugin.clone(), + reason: format!("remove orphan `{}`", entry.plugin), + }); + let replacement = catalog.current_name(plugin); + let target = format!("{replacement}@{name}"); + let already = inventory + .settings + .iter() + .any(|other| other.plugin == target && other.file == entry.file) + || state.plugins.iter().any(|installed| { + installed.id() == target + && installed.scope == entry.scope + && installed.project == entry.project + }); + let covered = entry.scope == "local" && wanted(catalog, selected, replacement); + if entry.enabled + && entry.scope != "user" + && !covered + && !already + && catalog.knows(replacement) + { + actions.push(install( + Host::Claude, + &target, + &entry.scope, + entry.project.clone(), + )); + } + } +} + +/// Whether the newest release of `binary` carries a prebuilt archive for this machine's target. +fn has_archive(context: &Context<'_>, binary: &Binary) -> bool { + binary.install.archive.is_some() + && context.target.as_deref().is_some_and(|target| { + context + .resolved + .archive_targets + .get(&binary.name) + .is_some_and(|targets| targets.contains(target)) + }) +} + +/// The install method for one binary: the explicit choice when it can be honoured, else the prebuilt +/// archive when the release has one for this machine's target, else cargo. +fn method(context: &Context<'_>, binary: &Binary) -> Option { + let archive = has_archive(context, binary); + let cargo = binary.install.cargo.is_some(); + if cargo && (matches!(context.method, Some(Method::Cargo)) || !archive) { + Some(Method::Cargo) + } else if archive { + Some(Method::Prebuilt) + } else { + None + } +} + +#[allow(clippy::too_many_lines)] +fn plan_binaries( + context: &Context<'_>, + inventory: &Inventory, + selected: &BTreeSet, + findings: &mut Vec, + actions: &mut Vec, +) { + let catalog = context.catalog; + let mut methods = BTreeSet::new(); + for product in &catalog.products { + if !selected.contains(&product.id) { + continue; + } + for binary in &product.binaries { + let subject = binary.name.clone(); + let newest = context + .resolved + .latest + .get(binary.install.repository()) + .cloned(); + let pin = context.resolved.pinned.get(&binary.name); + let Some(tag) = (match pin { + Some(pin) => pin.tag.clone(), + None => newest.clone(), + }) else { + findings.push(Finding { + level: Level::Warn, + host: None, + subject, + detail: match pin { + Some(pin) => format!( + "pinned to {} by {}, but no such release could be read (offline, or no such release?); not checked", + pin.spec, pin.file + ), + None => "its newest release could not be read (offline?); not checked" + .to_owned(), + }, + }); + continue; + }; + // How the finding names the release the plan holds the binary to. + let (held, pinned_by) = match pin { + Some(pin) => ( + format!("{tag}, pinned to {} by {}", pin.spec, pin.file), + Some(pin), + ), + None => (format!("the newest release {tag}"), None), + }; + if let (Some(pin), Some(newest)) = (pinned_by, &newest) { + if !version::same(newest, &tag) + && crate::pins::Spec::parse(&pin.spec).is_ok_and(|spec| spec.older_than(newest)) + { + findings.push(Finding { + level: Level::Note, + host: None, + subject: subject.clone(), + detail: format!( + "newer release {newest} exists; pinned to {} by {}, so it stays at {tag} (`b10x unpin {}` follows the newest)", + pin.spec, pin.file, binary.name + ), + }); + } + } + let copies = inventory + .binaries + .iter() + .find(|state| state.name == binary.name) + .map(|state| state.copies.as_slice()) + .unwrap_or_default(); + let current = copies.first(); + if !binary.runs_on(context.target.as_deref()) { + if current.is_none() && !binary.optional { + findings.push(Finding { + level: Level::Warn, + host: None, + subject: subject.clone(), + detail: format!( + "runs on {} only; not installed on this machine", + binary.platforms.join(", ") + ), + }); + } else if current.is_none() { + findings.push(Finding { + level: Level::Note, + host: None, + subject: subject.clone(), + detail: format!("optional, runs on {} only", binary.platforms.join(", ")), + }); + } + continue; + } + if current + .is_some_and(|first| version::same(first.version.as_deref().unwrap_or(""), &tag)) + { + let first = current.expect("checked above"); + findings.push(Finding { + level: Level::Ok, + host: None, + subject: subject.clone(), + detail: match pinned_by { + Some(pin) => format!( + "{tag} at {}, pinned to {} by {}", + first.path, pin.spec, pin.file + ), + None => format!("{tag} at {} is the newest release", first.path), + }, + }); + } else if let (true, Some(pin), Some(first)) = (context.upgrade, pinned_by, current) { + // An upgrade never moves a pinned CLI; it only says how to match the pin. + findings.push(Finding { + level: Level::Warn, + host: None, + subject: subject.clone(), + detail: format!( + "{} at {} does not match its pin {} ({}); upgrade leaves it, `b10x install {}` installs {tag}", + first.version.as_deref().unwrap_or("unknown"), + first.path, + pin.spec, + pin.file, + binary.name + ), + }); + } else if current.is_none() && binary.optional { + findings.push(Finding { + level: Level::Note, + host: None, + subject: subject.clone(), + detail: format!( + "optional, not installed (newest {tag}); `b10x install {}` adds it", + binary.name + ), + }); + } else { + let Some(method) = method(context, binary) else { + let target = context.target.as_deref().unwrap_or("this machine"); + findings.push(Finding { + level: Level::Warn, + host: None, + subject: subject.clone(), + detail: format!("{tag} has no prebuilt archive for {target} and `cargo` is not on PATH; install a Rust toolchain (https://rustup.rs) and plan again"), + }); + continue; + }; + methods.insert(method); + let default_directory = match method { + Method::Prebuilt => context.home.join(".local/bin"), + Method::Cargo => context.home.join(".cargo/bin"), + }; + let directory = current + .and_then(|first| Path::new(&first.path).parent().map(Path::to_path_buf)) + .filter(|parent| parent.starts_with(context.home)) + .unwrap_or(default_directory); + let (verb, detail) = match current { + None => ("install", format!("not on PATH; install {held}")), + Some(first) => ( + "upgrade", + format!( + "{} at {} is not {held}; replace it", + first + .version + .clone() + .unwrap_or_else(|| "unknown".to_owned()), + first.path + ), + ), + }; + findings.push(Finding { + level: Level::Change, + host: None, + subject: subject.clone(), + detail, + }); + actions.push(Action::InstallBinary { + name: binary.name.clone(), + tag: tag.clone(), + method, + install: binary.install.clone(), + directory: directory.to_string_lossy().into_owned(), + reason: format!("{verb} `{}`", binary.name), + }); + } + for shadowed in copies.iter().skip(1) { + let seen = shadowed + .version + .clone() + .unwrap_or_else(|| "unknown".to_owned()); + if !version::same(&seen, &tag) { + findings.push(Finding { + level: Level::Warn, + host: None, + subject: subject.clone(), + detail: format!( + "another copy, {seen} at {}, is later on PATH and never runs; remove it if nothing else uses it", + shadowed.path + ), + }); + } + } + } + } + if !methods.is_empty() { + let chosen: Vec<&str> = methods + .iter() + .map(|method| match method { + Method::Prebuilt => "prebuilt archives", + Method::Cargo => "cargo install", + }) + .collect(); + findings.push(Finding { + level: Level::Note, + host: None, + subject: "install method".to_owned(), + detail: format!( + "{} ({}); choose with --method cargo or --method prebuilt", + chosen.join(" and "), + if inventory.cargo { + "cargo is on PATH" + } else { + "cargo is not on PATH" + } + ), + }); + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::inventory::{self, BinaryState, Copy, Market, SettingsEntry}; + + fn legacy_claude() -> HostState { + HostState { + plugins: inventory::parse_claude_plugins(include_str!( + "../tests/fixtures/claude-legacy-list.json" + )) + .unwrap(), + marketplaces: inventory::parse_claude_marketplaces(include_str!( + "../tests/fixtures/claude-legacy-marketplaces.json" + )) + .unwrap(), + } + } + + fn legacy_codex() -> HostState { + HostState { + plugins: inventory::parse_codex_plugins(include_str!( + "../tests/fixtures/codex-legacy-list.json" + )) + .unwrap(), + marketplaces: inventory::parse_codex_marketplaces( + include_str!("../tests/fixtures/codex-legacy-marketplaces.json"), + Some(include_str!("../tests/fixtures/codex-legacy-config.toml")), + ) + .unwrap(), + } + } + + fn resolved() -> Resolved { + Resolved { + plugins: BTreeMap::from([ + ("b10x".to_owned(), "0.12.0".to_owned()), + ("aep".to_owned(), "0.13.0".to_owned()), + ("ess".to_owned(), "0.30.0".to_owned()), + ("worktree".to_owned(), "0.6.0".to_owned()), + ]), + latest: BTreeMap::from([ + ("beyond10x/aep".to_owned(), "0.57.0".to_owned()), + ("beyond10x/ess".to_owned(), "0.30.0".to_owned()), + ("beyond10x/worktree".to_owned(), "0.7.0".to_owned()), + ("beyond10x/metaharness".to_owned(), "0.7.0".to_owned()), + ("beyond10x/harness".to_owned(), "0.13.2".to_owned()), + ]), + archive_targets: BTreeMap::from([ + ("aep".to_owned(), every_target()), + ("ess".to_owned(), every_target()), + ("worktree".to_owned(), every_target()), + ("metaharness".to_owned(), BTreeSet::new()), + ( + "b10x-harness".to_owned(), + BTreeSet::from([X86_LINUX.to_owned()]), + ), + ]), + pinned: BTreeMap::new(), + } + } + + fn pinned_ess(spec: &str, tag: &str) -> Resolved { + let mut resolved = resolved(); + resolved.pinned.insert( + "ess".to_owned(), + crate::pins::Pinned { + spec: spec.to_owned(), + tag: Some(tag.to_owned()), + file: "/work/repo/b10x.toml".to_owned(), + }, + ); + resolved + } + + fn run_pinned(inventory: &Inventory, resolved: &Resolved, upgrade: bool) -> Plan { + let catalog = Catalog::embedded(); + let context = Context { + catalog: &catalog, + resolved, + selection: Some(BTreeSet::from(["ess".to_owned()])), + hosts: vec![Host::Claude], + home: Path::new("/opt/b10x-home"), + only: true, + method: None, + target: Some(X86_LINUX.to_owned()), + upgrade, + }; + plan(&context, inventory) + } + + fn installs(plan: &Plan) -> Vec { + plan.actions + .iter() + .filter_map(|action| match action { + Action::InstallBinary { name, tag, .. } => Some(format!("{name} {tag}")), + _ => None, + }) + .collect() + } + + #[test] + fn a_pinned_cli_installs_its_pinned_release_and_the_finding_names_the_pin() { + let inventory = Inventory { + claude: Some(HostState::default()), + binaries: binaries(&[("/opt/b10x-home/.local/bin/ess", "0.30.0")]), + ..Inventory::default() + }; + let plan = run_pinned(&inventory, &pinned_ess("0.29", "0.29.4"), false); + assert_eq!(installs(&plan), ["ess 0.29.4"]); + assert!(plan.findings.iter().any(|f| f.level == Level::Change + && f.detail.contains("pinned to 0.29 by /work/repo/b10x.toml"))); + assert!(plan + .findings + .iter() + .any(|f| f.level == Level::Note && f.detail.contains("newer release 0.30.0"))); + } + + #[test] + fn a_cli_at_its_pin_is_ok_and_upgrade_reports_the_newer_release_but_changes_nothing() { + let inventory = Inventory { + claude: Some(HostState::default()), + binaries: binaries(&[("/opt/b10x-home/.local/bin/ess", "0.29.4")]), + ..Inventory::default() + }; + let plan = run_pinned(&inventory, &pinned_ess("0.29.4", "0.29.4"), true); + assert!(installs(&plan).is_empty(), "{:#?}", plan.actions); + assert!(plan.findings.iter().any(|f| f.level == Level::Ok + && f.detail == "0.29.4 at /opt/b10x-home/.local/bin/ess, pinned to 0.29.4 by /work/repo/b10x.toml")); + assert!(plan + .findings + .iter() + .any(|f| f.detail.contains("newer release 0.30.0"))); + // A copy that drifted from its pin is named, not moved, by an upgrade. + let drifted = Inventory { + claude: Some(HostState::default()), + binaries: binaries(&[("/opt/b10x-home/.local/bin/ess", "0.30.0")]), + ..Inventory::default() + }; + let plan = run_pinned(&drifted, &pinned_ess("0.29.4", "0.29.4"), true); + assert!(installs(&plan).is_empty(), "{:#?}", plan.actions); + assert!(plan + .findings + .iter() + .any(|f| f.level == Level::Warn + && f.detail.contains("`b10x install ess` installs 0.29.4"))); + } + + const X86_LINUX: &str = "x86_64-unknown-linux-gnu"; + const ARM_LINUX: &str = "aarch64-unknown-linux-gnu"; + const ARM_MACOS: &str = "aarch64-apple-darwin"; + + fn every_target() -> BTreeSet { + [X86_LINUX, ARM_LINUX, "x86_64-apple-darwin", ARM_MACOS] + .into_iter() + .map(str::to_owned) + .collect() + } + + fn binaries(ess: &[(&str, &str)]) -> Vec { + vec![BinaryState { + name: "ess".to_owned(), + copies: ess + .iter() + .map(|(path, version)| Copy { + path: (*path).to_owned(), + version: Some((*version).to_owned()), + }) + .collect(), + }] + } + + fn run(inventory: &Inventory, selection: Option<&[&str]>, hosts: &[Host]) -> Plan { + let catalog = Catalog::embedded(); + let resolved = resolved(); + let context = Context { + catalog: &catalog, + resolved: &resolved, + selection: selection.map(|ids| ids.iter().map(|id| (*id).to_owned()).collect()), + hosts: hosts.to_vec(), + home: Path::new("/opt/b10x-home"), + only: false, + method: None, + target: Some(X86_LINUX.to_owned()), + upgrade: false, + }; + plan(&context, inventory) + } + + fn commands(plan: &Plan) -> Vec { + plan.actions + .iter() + .filter(|action| action.changes()) + .map(|action| match action { + Action::Command { argv, .. } => argv.join(" "), + other => other.describe(), + }) + .collect() + } + + #[test] + fn the_legacy_claude_install_migrates_to_b10x() { + let inventory = Inventory { + claude: Some(legacy_claude()), + ..Inventory::default() + }; + let plan = run(&inventory, None, &[Host::Claude]); + let selected: Vec<&str> = plan + .offers + .iter() + .filter(|offer| offer.selected) + .map(|offer| offer.id.as_str()) + .collect(); + assert_eq!( + selected, + ["aep", "ess", "worktree"], + "legacy products are preselected" + ); + let commands = commands(&plan); + for expected in [ + "claude plugin marketplace add beyond10x/agentplugins", + "claude plugin install b10x@b10x --scope user", + "claude plugin install aep@b10x --scope user", + "claude plugin install ess@b10x --scope user", + "claude plugin install worktree@b10x --scope user", + "claude plugin uninstall workspace-hygiene@beyond10x --scope user", + "claude plugin uninstall ess@ess --scope user", + "claude plugin marketplace remove beyond10x", + "claude plugin marketplace remove ess", + ] { + assert!( + commands.iter().any(|c| c == expected), + "missing `{expected}` in {commands:#?}" + ); + } + let add = commands + .iter() + .position(|c| c.ends_with("marketplace add beyond10x/agentplugins")); + let remove = commands + .iter() + .position(|c| c.ends_with("marketplace remove beyond10x")); + assert!(add < remove, "new marketplace first, retired one last"); + } + + #[test] + fn codex_legacy_install_migrates_and_unpins_nothing_it_does_not_own() { + let inventory = Inventory { + codex: Some(legacy_codex()), + ..Inventory::default() + }; + let plan = run(&inventory, Some(&["aep"]), &[Host::Codex]); + let commands = commands(&plan); + assert!(commands.contains(&"codex plugin add aep@b10x".to_owned())); + assert!(commands.contains(&"codex plugin remove workspace-hygiene@beyond10x".to_owned())); + assert!( + !commands.iter().any(|c| c.contains("worktree@b10x")), + "worktree not selected" + ); + assert!(commands.contains(&"codex plugin marketplace remove beyond10x".to_owned())); + } + + #[test] + fn a_pinned_b10x_registration_is_unpinned() { + let mut state = HostState::default(); + state.marketplaces.push(Market { + name: "b10x".to_owned(), + source: "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/beyond10x/agentplugins.git".to_owned(), + reference: Some("0.11.0".to_owned()), + location: None, + }); + let inventory = Inventory { + claude: Some(state), + ..Inventory::default() + }; + let plan = run(&inventory, Some(&[]), &[Host::Claude]); + assert!(matches!(plan.actions[0], Action::Unpin { .. })); + } + + #[test] + fn current_state_converges() { + let mut state = HostState::default(); + state.marketplaces.push(Market { + name: "b10x".to_owned(), + source: "beyond10x/agentplugins".to_owned(), + reference: None, + location: None, + }); + for (name, version) in [("b10x", "0.12.0"), ("ess", "0.30.0")] { + state.plugins.push(Installed { + name: name.to_owned(), + marketplace: "b10x".to_owned(), + version: Some(version.to_owned()), + scope: "user".to_owned(), + project: None, + enabled: true, + }); + } + let inventory = Inventory { + claude: Some(state), + binaries: binaries(&[("/opt/b10x-home/.local/bin/ess", "0.30.0")]), + ..Inventory::default() + }; + let plan = run(&inventory, None, &[Host::Claude]); + assert!(plan.converged(), "{:#?}", plan.actions); + } + + #[test] + fn a_binary_behind_its_plugin_is_replaced_where_it_runs_and_shadows_are_named() { + let inventory = Inventory { + claude: Some(HostState::default()), + binaries: binaries(&[ + ("/opt/b10x-home/.cargo/bin/ess", "0.26.0"), + ("/opt/b10x-home/.local/bin/ess", "0.26.0"), + ]), + ..Inventory::default() + }; + let plan = run(&inventory, Some(&["ess"]), &[Host::Claude]); + let install = plan + .actions + .iter() + .find_map(|action| match action { + Action::InstallBinary { + name, + tag, + directory, + .. + } if name == "ess" => Some((tag.clone(), directory.clone())), + _ => None, + }) + .expect("ess is replaced"); + assert_eq!( + install, + ("0.30.0".to_owned(), "/opt/b10x-home/.cargo/bin".to_owned()) + ); + assert!(plan + .findings + .iter() + .any(|f| f.level == Level::Warn && f.detail.contains("/opt/b10x-home/.local/bin/ess"))); + } + + #[test] + fn deselecting_a_product_uninstalls_its_current_plugins() { + let mut state = HostState::default(); + state.plugins.push(Installed { + name: "worktree".to_owned(), + marketplace: "b10x".to_owned(), + version: Some("0.6.0".to_owned()), + scope: "user".to_owned(), + project: None, + enabled: true, + }); + let inventory = Inventory { + claude: Some(state), + ..Inventory::default() + }; + let plan = run(&inventory, Some(&[]), &[Host::Claude]); + assert!(commands(&plan) + .contains(&"claude plugin uninstall worktree@b10x --scope user".to_owned())); + } + + #[test] + fn a_local_orphan_is_removed_and_its_replacement_goes_to_user_scope() { + let mut state = HostState::default(); + state.marketplaces.push(Market { + name: "b10x".to_owned(), + source: "beyond10x/agentplugins".to_owned(), + reference: None, + location: None, + }); + let inventory = Inventory { + claude: Some(state), + settings: vec![SettingsEntry { + file: "/work/acd/.claude/settings.local.json".to_owned(), + scope: "local".to_owned(), + project: Some("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/work/acd".to_owned()), + plugin: "beyond10x@beyond10x".to_owned(), + enabled: true, + }], + ..Inventory::default() + }; + let plan = run(&inventory, Some(&[]), &[Host::Claude]); + assert!(plan.actions.iter().any( + |a| matches!(a, Action::RemoveSetting { key, .. } if key == "beyond10x@beyond10x") + )); + let commands = commands(&plan); + assert!( + !commands.iter().any(|c| c.contains("--scope local")), + "{commands:#?}" + ); + assert!(commands.contains(&"claude plugin install b10x@b10x --scope user".to_owned())); + } + + fn market() -> Market { + Market { + name: "b10x".to_owned(), + source: "beyond10x/agentplugins".to_owned(), + reference: None, + location: None, + } + } + + fn installed(name: &str, marketplace: &str, scope: &str, project: Option<&str>) -> Installed { + Installed { + name: name.to_owned(), + marketplace: marketplace.to_owned(), + version: Some("0.12.0".to_owned()), + scope: scope.to_owned(), + project: project.map(str::to_owned), + enabled: true, + } + } + + fn run_upgrade(inventory: &Inventory, hosts: &[Host]) -> Plan { + let catalog = Catalog::embedded(); + let resolved = resolved(); + let context = Context { + catalog: &catalog, + resolved: &resolved, + selection: None, + hosts: hosts.to_vec(), + home: Path::new("/opt/b10x-home"), + only: true, + method: None, + target: Some(X86_LINUX.to_owned()), + upgrade: true, + }; + plan(&context, inventory) + } + + #[test] + fn an_installed_optional_product_stays_when_nothing_is_selected() { + let mut claude = HostState::default(); + claude.marketplaces.push(market()); + claude.plugins.push(installed("b10x", "b10x", "user", None)); + claude + .plugins + .push(installed("connectors", "b10x", "user", None)); + let inventory = Inventory { + claude: Some(claude), + ..Inventory::default() + }; + let plan = run(&inventory, None, &[Host::Claude]); + let commands = commands(&plan); + assert!( + !commands + .iter() + .any(|c| c.contains("uninstall connectors@b10x")), + "{commands:#?}" + ); + assert!(plan + .offers + .iter() + .any(|offer| offer.id == "connectors" && offer.selected)); + } + + #[test] + fn upgrade_installs_the_replacement_of_a_legacy_plugin_it_removes() { + let mut claude = HostState::default(); + claude.marketplaces.push(market()); + claude.plugins.push(installed("b10x", "b10x", "user", None)); + claude + .plugins + .push(installed("worktree", "worktree", "user", None)); + let inventory = Inventory { + claude: Some(claude), + ..Inventory::default() + }; + let commands = commands(&run_upgrade(&inventory, &[Host::Claude])); + assert!( + commands.contains(&"claude plugin uninstall worktree@worktree --scope user".to_owned()), + "{commands:#?}" + ); + assert!( + commands.contains(&"claude plugin install worktree@b10x --scope user".to_owned()), + "{commands:#?}" + ); + } + + #[test] + fn a_legacy_local_install_the_user_scope_covers_is_only_removed() { + let mut claude = HostState::default(); + claude.marketplaces.push(market()); + claude + .plugins + .push(installed("aep-plan", "b10x", "local", Some("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/work/p"))); + claude + .plugins + .push(installed("aep-drive", "b10x", "project", Some("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/work/q"))); + let inventory = Inventory { + claude: Some(claude), + ..Inventory::default() + }; + let commands = commands(&run(&inventory, Some(&["aep"]), &[Host::Claude])); + assert!( + commands.contains(&"claude plugin install aep@b10x --scope user".to_owned()), + "{commands:#?}" + ); + assert!( + !commands + .iter() + .any(|c| c == "claude plugin install aep@b10x --scope local"), + "{commands:#?}" + ); + assert!( + commands.contains(&"claude plugin install aep@b10x --scope project".to_owned()), + "a committed project install is replaced in place: {commands:#?}" + ); + } + + #[test] + fn beyond10x_on_a_host_outside_the_plan_is_named() { + let mut codex = HostState::default(); + codex.plugins.push(installed("ess", "b10x", "user", None)); + let inventory = Inventory { + claude: Some(HostState::default()), + codex: Some(codex), + ..Inventory::default() + }; + let plan = run(&inventory, Some(&["ess"]), &[Host::Claude]); + assert!( + plan.findings + .iter() + .any(|f| f.host == Some(Host::Codex) && f.detail.contains("--host all")), + "{:#?}", + plan.findings + ); + assert!(!commands(&plan).iter().any(|c| c.starts_with("codex"))); + } + + /// Claude with the base and `aep`, `ess`, `worktree` current, and their CLIs at the newest + /// release: an `init aep,ess,worktree --host claude` plan with nothing to do for Claude. + fn current_claude() -> Inventory { + let mut claude = HostState { + marketplaces: vec![market()], + ..HostState::default() + }; + for (name, version) in [ + ("b10x", "0.12.0"), + ("aep", "0.13.0"), + ("ess", "0.30.0"), + ("worktree", "0.6.0"), + ] { + let mut plugin = installed(name, "b10x", "user", None); + plugin.version = Some(version.to_owned()); + claude.plugins.push(plugin); + } + let binaries = [("aep", "0.57.0"), ("ess", "0.30.0"), ("worktree", "0.7.0")] + .into_iter() + .map(|(name, version)| BinaryState { + name: name.to_owned(), + copies: vec![Copy { + path: format!("/opt/b10x-home/.local/bin/{name}"), + version: Some(version.to_owned()), + }], + }) + .collect(); + Inventory { + claude: Some(claude), + binaries, + ..Inventory::default() + } + } + + fn other_host(plan: &Plan, host: Host) -> Vec<&Finding> { + plan.findings + .iter() + .filter(|f| f.host == Some(host)) + .collect() + } + + #[test] + fn init_for_one_host_says_what_a_plan_for_the_other_host_would_do() { + let mut inventory = current_claude(); + inventory.codex = Some(legacy_codex()); + let plan = run_only(&inventory, &["aep", "ess", "worktree"], None); + assert!(plan.converged(), "{:#?}", plan.actions); + let codex = other_host(&plan, Host::Codex); + assert_eq!(codex.len(), 1, "{codex:#?}"); + assert_eq!(codex[0].level, Level::Warn); + assert_eq!( + codex[0].detail, + "Beyond10x plugins are installed for `codex` too and this plan leaves them alone; \ + a plan for `codex` has 10 actions (4 legacy installs, 4 missing plugins, \ + 2 marketplace changes); `--host codex` or `--host all` applies them" + ); + assert!(!commands(&plan).iter().any(|c| c.starts_with("codex"))); + assert_eq!(nothing_to_change(&plan), "Nothing to change for claude."); + } + + #[test] + fn init_for_one_host_notes_in_one_line_that_the_other_host_is_current() { + let mut inventory = current_claude(); + inventory.codex = inventory.claude.clone(); + let plan = run_only(&inventory, &["aep", "ess", "worktree"], None); + assert!(plan.converged(), "{:#?}", plan.actions); + let codex = other_host(&plan, Host::Codex); + assert_eq!(codex.len(), 1, "{codex:#?}"); + assert_eq!(codex[0].level, Level::Note); + assert_eq!( + codex[0].detail, + "Beyond10x plugins are installed for `codex` too and are current for this selection" + ); + assert_eq!(nothing_to_change(&plan), "Nothing to change."); + } + + #[test] + fn init_without_beyond10x_on_the_other_host_names_no_other_host() { + let mut inventory = current_claude(); + inventory.codex = Some(HostState::default()); + let plan = run_only(&inventory, &["aep", "ess", "worktree"], None); + assert!(plan.converged(), "{:#?}", plan.actions); + assert!( + other_host(&plan, Host::Codex).is_empty(), + "{:#?}", + plan.findings + ); + assert_eq!(nothing_to_change(&plan), "Nothing to change."); + } + + #[test] + fn a_warning_on_a_planned_host_leaves_nothing_to_change_unqualified() { + let mut inventory = current_claude(); + inventory.broken = vec![(Host::Codex, "codex plugin list failed".to_owned())]; + let plan = run( + &inventory, + Some(&["aep", "ess", "worktree"]), + &[Host::Claude, Host::Codex], + ); + assert!(plan.converged(), "{:#?}", plan.actions); + assert!(plan + .findings + .iter() + .any(|f| f.level == Level::Warn && f.host == Some(Host::Codex))); + assert_eq!(nothing_to_change(&plan), "Nothing to change."); + } + + #[test] + fn two_retired_plugins_in_one_project_become_one_install() { + let mut state = HostState::default(); + for name in ["aep-plan", "aep-drive"] { + state.plugins.push(Installed { + name: name.to_owned(), + marketplace: "b10x".to_owned(), + version: Some("0.12.0".to_owned()), + scope: "local".to_owned(), + project: Some("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/work/p".to_owned()), + enabled: true, + }); + } + let inventory = Inventory { + claude: Some(state), + ..Inventory::default() + }; + let plan = run(&inventory, Some(&[]), &[Host::Claude]); + let commands = commands(&plan); + let installs = commands + .iter() + .filter(|c| *c == "claude plugin install aep@b10x --scope local") + .count(); + assert_eq!(installs, 1, "{commands:#?}"); + assert!( + commands.contains(&"claude plugin uninstall aep-plan@b10x --scope local".to_owned()) + ); + assert!( + commands.contains(&"claude plugin uninstall aep-drive@b10x --scope local".to_owned()) + ); + } + + fn run_only(inventory: &Inventory, products: &[&str], method: Option) -> Plan { + let catalog = Catalog::embedded(); + let resolved = resolved(); + let context = Context { + catalog: &catalog, + resolved: &resolved, + selection: Some(products.iter().map(|id| (*id).to_owned()).collect()), + hosts: vec![Host::Claude], + home: Path::new("/opt/b10x-home"), + only: true, + method, + target: Some(X86_LINUX.to_owned()), + upgrade: false, + }; + plan(&context, inventory) + } + + #[test] + fn init_of_one_product_leaves_every_other_install_alone() { + let inventory = Inventory { + claude: Some(legacy_claude()), + ..Inventory::default() + }; + let plan = run_only(&inventory, &["ess"], None); + let commands = commands(&plan); + assert!(commands.contains(&"claude plugin install ess@b10x --scope user".to_owned())); + assert!(commands.contains(&"claude plugin uninstall ess@ess --scope user".to_owned())); + assert!( + !commands.iter().any(|c| c.contains("aep")), + "aep untouched: {commands:#?}" + ); + assert!( + !commands + .iter() + .any(|c| c.ends_with("marketplace remove beyond10x")), + "beyond10x still serves aep installs" + ); + assert!(plan.next.iter().any(|line| line.starts_with("/ess:init"))); + } + + #[test] + fn prebuilt_is_the_default_and_cargo_only_when_asked_or_without_archives() { + let method_for = |cargo: bool, asked: Option| { + let inventory = Inventory { + claude: Some(HostState::default()), + cargo, + ..Inventory::default() + }; + run_only(&inventory, &["worktree"], asked) + .actions + .iter() + .find_map(|action| match action { + Action::InstallBinary { method, .. } => Some(*method), + _ => None, + }) + }; + assert_eq!(method_for(true, None), Some(Method::Prebuilt)); + assert_eq!(method_for(false, None), Some(Method::Prebuilt)); + assert_eq!(method_for(true, Some(Method::Cargo)), Some(Method::Cargo)); + } + + /// Plan `aep` on `target` with an older `b10x-harness` installed, whose newest release carries + /// an archive for x86-64 Linux only; `cargo_option` keeps or strips the catalog's cargo route. + fn plan_harness(target: Option<&str>, cargo_option: bool) -> Plan { + let mut catalog = Catalog::embedded(); + if !cargo_option { + for product in &mut catalog.products { + for binary in &mut product.binaries { + if binary.name == "b10x-harness" { + binary.install.cargo = None; + } + } + } + } + let resolved = resolved(); + let inventory = Inventory { + claude: Some(HostState::default()), + binaries: vec![BinaryState { + name: "b10x-harness".to_owned(), + copies: vec![Copy { + path: "/opt/b10x-home/.local/bin/b10x-harness".to_owned(), + version: Some("0.13.1".to_owned()), + }], + }], + ..Inventory::default() + }; + let context = Context { + catalog: &catalog, + resolved: &resolved, + selection: Some(BTreeSet::from(["aep".to_owned()])), + hosts: vec![Host::Claude], + home: Path::new("/opt/b10x-home"), + only: true, + method: None, + target: target.map(str::to_owned), + upgrade: false, + }; + plan(&context, &inventory) + } + + fn harness_method(plan: &Plan) -> Option { + plan.actions.iter().find_map(|action| match action { + Action::InstallBinary { name, method, .. } if name == "b10x-harness" => Some(*method), + _ => None, + }) + } + + #[test] + fn prebuilt_only_for_a_target_the_release_has_an_archive_for() { + assert_eq!( + harness_method(&plan_harness(Some(X86_LINUX), true)), + Some(Method::Prebuilt) + ); + assert_eq!( + harness_method(&plan_harness(Some(ARM_LINUX), true)), + Some(Method::Cargo) + ); + assert_eq!( + harness_method(&plan_harness(Some(ARM_MACOS), true)), + None, + "a Linux-only binary is not planned on macOS, not even with cargo" + ); + assert_eq!( + harness_method(&plan_harness(None, true)), + None, + "an unknown machine is not a platform the binary declares" + ); + } + + #[test] + fn no_archive_for_the_target_and_no_cargo_route_is_a_warning_not_an_action() { + assert_eq!( + harness_method(&plan_harness(Some(X86_LINUX), false)), + Some(Method::Prebuilt) + ); + { + let target = ARM_LINUX; + let plan = plan_harness(Some(target), false); + assert_eq!(harness_method(&plan), None); + assert!( + plan.findings + .iter() + .any(|finding| finding.level == Level::Warn + && finding.subject == "b10x-harness" + && finding + .detail + .contains(&format!("no prebuilt archive for {target}"))), + "{:#?}", + plan.findings + ); + } + } + + #[test] + fn a_release_without_archives_and_no_cargo_is_a_warning_not_an_action() { + let inventory = Inventory { + claude: Some(HostState::default()), + binaries: vec![BinaryState { + name: "metaharness".to_owned(), + copies: vec![Copy { + path: "/opt/b10x-home/.local/bin/metaharness".to_owned(), + version: Some("0.6.0".to_owned()), + }], + }], + ..Inventory::default() + }; + let plan = run_only(&inventory, &["aep"], Some(Method::Prebuilt)); + let metaharness = plan.actions.iter().find( + |action| matches!(action, Action::InstallBinary { name, .. } if name == "metaharness"), + ); + assert!( + matches!(metaharness, Some(Action::InstallBinary { method: Method::Cargo, .. })), + "no archive for metaharness: asking for prebuilt falls back to cargo, which the catalog has" + ); + } + + #[test] + fn upgrade_updates_what_exists_and_leaves_an_empty_host_alone() { + let mut claude = HostState::default(); + claude.marketplaces.push(Market { + name: "b10x".to_owned(), + source: "beyond10x/agentplugins".to_owned(), + reference: None, + location: None, + }); + for name in ["b10x", "ess"] { + claude.plugins.push(Installed { + name: name.to_owned(), + marketplace: "b10x".to_owned(), + version: Some("0.12.0".to_owned()), + scope: "user".to_owned(), + project: None, + enabled: true, + }); + } + let inventory = Inventory { + claude: Some(claude), + codex: Some(HostState::default()), + ..Inventory::default() + }; + let catalog = Catalog::embedded(); + let resolved = resolved(); + let context = Context { + catalog: &catalog, + resolved: &resolved, + selection: None, + hosts: vec![Host::Claude, Host::Codex], + home: Path::new("/opt/b10x-home"), + only: true, + method: None, + target: Some(X86_LINUX.to_owned()), + upgrade: true, + }; + let plan = plan(&context, &inventory); + let commands = commands(&plan); + assert!( + !commands.iter().any(|c| c.starts_with("codex")), + "{commands:#?}" + ); + assert!(commands.contains(&"claude plugin update ess@b10x --scope user".to_owned())); + assert!( + !commands.iter().any(|c| c.contains("install aep")), + "{commands:#?}" + ); + assert!(plan + .findings + .iter() + .any(|f| f.host == Some(Host::Codex) && f.detail.contains("b10x init"))); + } + + #[test] + fn the_digest_changes_with_the_inventory() { + let empty = Inventory::default(); + let other = Inventory { + claude: Some(HostState::default()), + ..Inventory::default() + }; + assert_ne!(digest(&empty), digest(&other)); + } +} diff --git a/crates/b10x/src/resolve.rs b/crates/b10x/src/resolve.rs new file mode 100644 index 0000000..14fac85 --- /dev/null +++ b/crates/b10x/src/resolve.rs @@ -0,0 +1,481 @@ +//! Versions resolved at run time. The catalog names no version, so the version each plugin will +//! install and each binary must have is read here: a plugin the marketplace carries from its +//! `plugin.json`, a plugin it points at from that repository's newest release, a binary bound to +//! `latest` from the newest release. + +use std::collections::{BTreeMap, BTreeSet}; +use std::path::Path; +use std::process::Command; + +use serde::{Deserialize, Serialize}; + +use crate::catalog::Catalog; +use crate::install::listed_targets; +use crate::inventory::Inventory; +use crate::pins::{PinFile, Pinned, Spec}; +use crate::version; + +/// Versions setup compares against. +#[derive(Debug, Clone, Default, PartialEq, Eq, Deserialize, Serialize)] +pub struct Resolved { + /// Plugin name → the version the marketplace installs now. + pub plugins: BTreeMap, + /// `owner/repo` → newest release tag. + pub latest: BTreeMap, + /// Binary name → the targets the release the plan installs (the pinned one, else the newest) + /// carries a prebuilt archive for, as its `SHA256SUMS` lists them; absent or empty when none. + #[serde(default)] + pub archive_targets: BTreeMap>, + /// Binary name → its pin in the repository's `b10x.toml`, resolved to a release. + #[serde(default, skip_serializing_if = "BTreeMap::is_empty")] + pub pinned: BTreeMap, +} + +/// Where marketplace files are read from. +pub enum Source<'a> { + /// A local clone of the marketplace repository. + Clone(&'a Path), + /// The repository's default branch on GitHub. + Remote(&'a str), +} + +impl Source<'_> { + /// Read one repository file. + #[must_use] + pub fn read(&self, relative: &str) -> Option { + match self { + Source::Clone(root) => std::fs::read_to_string(root.join(relative)).ok(), + Source::Remote(repository) => fetch(&format!( + "https://raw.githubusercontent.com/{repository}/HEAD/{relative}" + )), + } + } +} + +/// `curl` the body of a URL, or `None`. +#[must_use] +pub fn fetch(url: &str) -> Option { + let output = Command::new("curl") + .args(["-fsSL", "--retry", "2", "--max-time", "30", url]) + .output() + .ok()?; + output + .status + .success() + .then(|| String::from_utf8_lossy(&output.stdout).into_owned()) +} + +/// Where the newest release tags seen by the last plan are kept, for the offline session check. +#[must_use] +pub fn cache_path(home: &Path) -> std::path::PathBuf { + home.join(".local/state/b10x/latest.json") +} + +/// Seconds since the Unix epoch. +#[must_use] +pub fn now() -> u64 { + std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .map_or(0, |elapsed| elapsed.as_secs()) +} + +/// The newest-release record: `checked_at` (last successful read), `attempted_at` (last try) and +/// `latest` (`owner/repo` → tag). +#[must_use] +pub fn recorded(home: &Path) -> serde_json::Map { + std::fs::read_to_string(cache_path(home)) + .ok() + .and_then(|text| serde_json::from_str::(&text).ok()) + .and_then(|value| value.as_object().cloned()) + .unwrap_or_default() +} + +/// Merge newly read tags into the record. `checked_at` moves only when something was read. +pub fn record(home: &Path, tags: &BTreeMap, at: u64) { + let mut value = recorded(home); + let mut latest = value + .get("latest") + .and_then(serde_json::Value::as_object) + .cloned() + .unwrap_or_default(); + for (repository, tag) in tags { + latest.insert(repository.clone(), serde_json::json!(tag)); + } + if !tags.is_empty() { + value.insert("checked_at".to_owned(), serde_json::json!(at)); + } + value.insert("attempted_at".to_owned(), serde_json::json!(at)); + value.insert("latest".to_owned(), serde_json::Value::Object(latest)); + let path = cache_path(home); + if let Some(parent) = path.parent() { + let _ = std::fs::create_dir_all(parent); + } + let _ = std::fs::write(path, serde_json::Value::Object(value).to_string()); +} + +/// Record the newest release tags a plan read, and when. +pub fn remember(home: &Path, resolved: &Resolved) { + record(home, &resolved.latest, now()); +} + +/// The newest release tag of `owner/repo`, from the `releases/latest` redirect (no API token). +#[must_use] +pub fn latest_tag(repository: &str) -> Option { + latest_tag_within(repository, 30) +} + +/// [`latest_tag`], giving up after `seconds`. +#[must_use] +pub fn latest_tag_within(repository: &str, seconds: u64) -> Option { + let seconds = seconds.to_string(); + let output = Command::new("curl") + .args([ + "-fsS", + "-o", + "/dev/null", + "--connect-timeout", + &seconds, + "--max-time", + &seconds, + "-w", + "%{redirect_url}", + &format!("/{repository}/releases/latest"), + ]) + .output() + .ok()?; + let location = String::from_utf8_lossy(&output.stdout).into_owned(); + let tag = location.rsplit_once("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/releases/tag/")?.1.trim(); + (!tag.is_empty()).then(|| tag.to_owned()) +} + +/// The newest release tag of each repository, read in parallel, each within `seconds`. +#[must_use] +pub fn latest_tags(repositories: &BTreeSet, seconds: u64) -> BTreeMap { + std::thread::scope(|scope| { + let handles: Vec<_> = repositories + .iter() + .map(|repository| { + scope.spawn(move || { + latest_tag_within(repository, seconds).map(|tag| (repository.clone(), tag)) + }) + }) + .collect(); + handles + .into_iter() + .filter_map(|handle| handle.join().ok().flatten()) + .collect() + }) +} + +/// Every tag of `owner/repo`, from `git ls-remote` (no API token). +#[must_use] +pub fn tags(repository: &str) -> Option> { + let output = Command::new("git") + .args([ + "ls-remote", + "--tags", + "--refs", + &format!("/{repository}.git"), + ]) + .env("GIT_TERMINAL_PROMPT", "0") + .output() + .ok()?; + output.status.success().then(|| { + String::from_utf8_lossy(&output.stdout) + .lines() + .filter_map(|line| line.split_once("refs/tags/")) + .map(|(_, tag)| tag.trim().to_owned()) + .collect() + }) +} + +/// Whether `owner/repo` has a published GitHub Release for `tag`. +#[must_use] +pub fn release_exists(repository: &str, tag: &str) -> bool { + Command::new("curl") + .args([ + "-fsS", + "-o", + "/dev/null", + "--max-time", + "30", + "-H", + "Accept: application/vnd.github+json", + &format!("https://api.github.com/repos/{repository}/releases/tags/{tag}"), + ]) + .output() + .is_ok_and(|output| output.status.success()) +} + +/// The release a pin resolves to: the newest release when it satisfies the pin, else the newest +/// matching tag of the repository. +#[must_use] +pub fn pinned_tag(repository: &str, spec: Spec, newest: Option<&str>) -> Option { + if let Some(newest) = newest.filter(|newest| spec.matches(newest)) { + return Some(newest.to_owned()); + } + let tags = tags(repository)?; + spec.select(tags.iter().map(String::as_str)) + .map(str::to_owned) +} + +/// The version a marketplace checkout carries: the highest version among its carried plugins. +#[must_use] +pub fn carried_version(source: &Source<'_>) -> Option { + let marketplace = source + .read(".claude-plugin/marketplace.json") + .and_then(|text| serde_json::from_str::(&text).ok())?; + let carried: serde_json::Value = serde_json::json!({ + "plugins": marketplace + .get("plugins") + .and_then(serde_json::Value::as_array) + .map(|entries| { + entries + .iter() + .filter(|entry| entry.get("source").is_some_and(serde_json::Value::is_string)) + .cloned() + .collect::>() + }) + .unwrap_or_default() + }); + plugin_versions(&carried, source, &mut BTreeMap::new()) + .into_values() + .max_by_key(|found| version::key(found)) +} + +/// Whether a host's marketplace clone is older than the marketplace's newest release. The hosts +/// refresh their clone only when a plan is applied, so a plan read from a stale clone would call +/// old plugins current. +#[must_use] +pub fn stale(clone: &Path, newest: Option<&str>) -> bool { + let have = carried_version(&Source::Clone(clone)); + matches!( + (have.as_deref().and_then(version::key), newest.and_then(version::key)), + (Some(have), Some(newest)) if have < newest + ) +} + +/// `owner/repo` from a GitHub URL. +#[must_use] +pub fn repository_of(url: &str) -> Option { + let rest = url + .trim_end_matches('/') + .trim_end_matches(".git") + .split_once("github.com/")? + .1; + let mut parts = rest.split('/'); + let owner = parts.next()?; + let name = parts.next()?; + Some(format!("{owner}/{name}")) +} + +/// The marketplace clone a host registered, if any. +#[must_use] +pub fn clone_of<'a>(inventory: &'a Inventory, catalog: &Catalog) -> Option<&'a Path> { + [inventory.claude.as_ref(), inventory.codex.as_ref()] + .into_iter() + .flatten() + .flat_map(|state| state.marketplaces.iter()) + .filter(|market| market.name == catalog.marketplace.name) + .filter_map(|market| market.location.as_deref()) + .map(Path::new) + .find(|path| path.join(".claude-plugin/marketplace.json").is_file()) +} + +/// The catalog to plan with: the marketplace's copy when it parses, else the embedded one. +#[must_use] +pub fn catalog(source: &Source<'_>) -> Catalog { + source + .read("catalog.json") + .and_then(|text| Catalog::parse(&text).ok()) + .unwrap_or_else(Catalog::embedded) +} + +/// Plugin versions from a marketplace document; remote entries take their repository's newest tag. +pub fn plugin_versions( + marketplace: &serde_json::Value, + source: &Source<'_>, + latest: &mut BTreeMap, +) -> BTreeMap { + let mut versions = BTreeMap::new(); + let entries = marketplace + .get("plugins") + .and_then(serde_json::Value::as_array) + .cloned() + .unwrap_or_default(); + for entry in entries { + let Some(name) = entry.get("name").and_then(serde_json::Value::as_str) else { + continue; + }; + let version = match entry.get("source") { + Some(serde_json::Value::String(path)) => { + let relative = format!( + "{}/.claude-plugin/plugin.json", + path.trim_start_matches("./") + ); + source + .read(&relative) + .and_then(|text| serde_json::from_str::(&text).ok()) + .and_then(|manifest| { + manifest + .get("version") + .and_then(serde_json::Value::as_str) + .map(str::to_owned) + }) + } + Some(object) => object + .get("url") + .and_then(serde_json::Value::as_str) + .and_then(repository_of) + .and_then(|repository| tag(latest, &repository)), + None => None, + }; + if let Some(version) = version { + versions.insert( + name.to_owned(), + crate::version::parse(&version).unwrap_or(version), + ); + } + } + versions +} + +fn tag(latest: &mut BTreeMap, repository: &str) -> Option { + if let Some(tag) = latest.get(repository) { + return Some(tag.clone()); + } + let tag = latest_tag(repository)?; + latest.insert(repository.to_owned(), tag.clone()); + Some(tag) +} + +/// Resolve every version the plan needs; a binary pinned in `pins` resolves to its pinned release. +#[must_use] +pub fn resolve(catalog: &Catalog, source: &Source<'_>, pins: Option<&PinFile>) -> Resolved { + let mut latest = BTreeMap::new(); + let plugins = source + .read(".claude-plugin/marketplace.json") + .and_then(|text| serde_json::from_str::(&text).ok()) + .map(|marketplace| plugin_versions(&marketplace, source, &mut latest)) + .unwrap_or_default(); + let mut archive_targets = BTreeMap::new(); + let mut pinned = BTreeMap::new(); + let mut sums_of = BTreeMap::new(); + for product in &catalog.products { + for binary in &product.binaries { + let repository = binary.install.repository(); + let newest = tag(&mut latest, repository); + let pin = pins.and_then(|file| { + let spec = file.pins.get(&binary.name)?; + let tag = Spec::parse(spec) + .ok() + .and_then(|parsed| pinned_tag(repository, parsed, newest.as_deref())); + Some(Pinned { + spec: spec.clone(), + tag, + file: file.path.to_string_lossy().into_owned(), + }) + }); + let installs = match &pin { + Some(pin) => pin.tag.clone(), + None => newest, + }; + if let Some(pin) = pin { + pinned.insert(binary.name.clone(), pin); + } + if let (Some(tag), Some(_)) = (installs, &binary.install.archive) { + let sums: &Option = sums_of + .entry((repository.to_owned(), tag.clone())) + .or_insert_with(|| { + fetch(&format!( + "/{repository}/releases/download/{tag}/SHA256SUMS" + )) + }); + archive_targets.insert( + binary.name.clone(), + sums.as_deref() + .map(|sums| listed_targets(sums, &binary.name, &tag)) + .unwrap_or_default(), + ); + } + } + } + Resolved { + plugins, + latest, + archive_targets, + pinned, + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn repositories_come_from_github_urls() { + assert_eq!( + repository_of("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/beyond10x/worktree.git").as_deref(), + Some("beyond10x/worktree") + ); + assert_eq!(repository_of("https://gitlab.com/a/b.git"), None); + } + + #[test] + fn carried_plugins_read_their_manifest() { + let root = std::env::temp_dir().join(format!("b10x-resolve-{}", std::process::id())); + let manifest = root.join("plugins/aep/.claude-plugin"); + std::fs::create_dir_all(&manifest).unwrap(); + std::fs::write( + manifest.join("plugin.json"), + r#"{"name":"aep","version":"0.12.0"}"#, + ) + .unwrap(); + let marketplace = serde_json::json!({"plugins":[{"name":"aep","source":"./plugins/aep"}]}); + let versions = plugin_versions(&marketplace, &Source::Clone(&root), &mut BTreeMap::new()); + std::fs::remove_dir_all(&root).unwrap(); + assert_eq!(versions.get("aep").map(String::as_str), Some("0.12.0")); + } + + fn marketplace_at(name: &str, version: &str) -> std::path::PathBuf { + let root = std::env::temp_dir().join(format!("b10x-clone-{name}-{}", std::process::id())); + let _ = std::fs::remove_dir_all(&root); + for plugin in ["b10x", "ess"] { + let manifest = root.join(format!("plugins/{plugin}/.claude-plugin")); + std::fs::create_dir_all(&manifest).unwrap(); + std::fs::write( + manifest.join("plugin.json"), + format!(r#"{{"name":"{plugin}","version":"{version}"}}"#), + ) + .unwrap(); + } + std::fs::create_dir_all(root.join(".claude-plugin")).unwrap(); + std::fs::write( + root.join(".claude-plugin/marketplace.json"), + r#"{"plugins":[{"name":"b10x","source":"./plugins/b10x"},{"name":"ess","source":"./plugins/ess"},{"name":"x","source":{"source":"url","url":"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/a/x.git"}}]}"#, + ) + .unwrap(); + root + } + + #[test] + fn a_clone_older_than_the_newest_release_is_stale() { + let clone = marketplace_at("stale", "0.14.7"); + assert_eq!( + carried_version(&Source::Clone(&clone)).as_deref(), + Some("0.14.7") + ); + assert!(stale(&clone, Some("0.14.10")), "0.14.7 < 0.14.10"); + assert!(!stale(&clone, Some("0.14.7"))); + assert!(!stale(&clone, None), "offline: the clone is all there is"); + std::fs::remove_dir_all(&clone).unwrap(); + } + + #[test] + fn pointed_plugins_take_the_known_newest_tag() { + let marketplace = serde_json::json!({"plugins":[{"name":"ess","source":{"source":"git-subdir","url":"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/beyond10x/ess.git","path":"plugins/ess"}}]}); + let mut latest = BTreeMap::from([("beyond10x/ess".to_owned(), "0.30.0".to_owned())]); + let versions = plugin_versions(&marketplace, &Source::Remote("x/y"), &mut latest); + assert_eq!(versions.get("ess").map(String::as_str), Some("0.30.0")); + } +} diff --git a/crates/b10x/src/skill.rs b/crates/b10x/src/skill.rs new file mode 100644 index 0000000..bca90eb --- /dev/null +++ b/crates/b10x/src/skill.rs @@ -0,0 +1,166 @@ +//! `b10x skill`: print an installed skill or agent. A host loads a newly installed plugin only in a +//! new session, so the session that ran setup reads the skill it needs through this instead of +//! searching the host's plugin cache. + +use std::fmt::Write as _; +use std::path::{Path, PathBuf}; + +use crate::version; + +/// Where each installed `b10x` plugin lives: Claude Code's recorded install path first, then the +/// newest version in Codex's cache. +#[must_use] +pub fn root(home: &Path, marketplace: &str, plugin: &str) -> Option { + let id = format!("{plugin}@{marketplace}"); + let claude = std::fs::read_to_string(home.join(".claude/plugins/installed_plugins.json")) + .ok() + .and_then(|text| serde_json::from_str::(&text).ok()) + .and_then(|value| { + let installs = value.get("plugins")?.get(&id)?.as_array()?.clone(); + installs + .iter() + .find(|install| { + install.get("scope").and_then(serde_json::Value::as_str) == Some("user") + }) + .or_else(|| installs.first()) + .and_then(|install| install.get("installPath")) + .and_then(serde_json::Value::as_str) + .map(PathBuf::from) + }) + .filter(|path| path.is_dir()); + claude.or_else(|| { + let cache = home + .join(".codex/plugins/cache") + .join(marketplace) + .join(plugin); + std::fs::read_dir(&cache) + .ok()? + .filter_map(|entry| entry.ok()?.file_name().into_string().ok()) + .filter_map(|name| version::key(&name).map(|key| (key, name))) + .max() + .map(|(_, name)| cache.join(name)) + }) +} + +fn names(directory: &Path, file: Option<&str>) -> Vec { + let mut names: Vec = std::fs::read_dir(directory) + .map(|entries| { + entries + .filter_map(Result::ok) + .filter_map(|entry| { + let path = entry.path(); + match file { + Some(file) => path.join(file).is_file().then(|| entry.file_name()), + None => (path.extension().and_then(|e| e.to_str()) == Some("md")) + .then(|| path.file_stem().map(std::ffi::OsStr::to_os_string)) + .flatten(), + } + }) + .filter_map(|name| name.into_string().ok()) + .collect() + }) + .unwrap_or_default(); + names.sort(); + names +} + +/// The text `b10x skill ` prints: a plugin's skills and agents for `plugin`, one file's +/// contents for `plugin:name`. +pub fn text(home: &Path, marketplace: &str, id: &str) -> Result { + let (plugin, name) = match id.split_once(':') { + Some((plugin, name)) => (plugin, Some(name)), + None => (id, None), + }; + let root = root(home, marketplace, plugin).ok_or_else(|| { + format!("`{plugin}@{marketplace}` is not installed for Claude Code or Codex; run `b10x setup plan`") + })?; + let Some(name) = name else { + let skills = names(&root.join("skills"), Some("SKILL.md")); + let agents = names(&root.join("agents"), None); + let mut out = format!("{plugin} ({})\n", root.display()); + for skill in skills { + let _ = writeln!(out, " skill {plugin}:{skill}"); + } + for agent in agents { + let _ = writeln!(out, " agent {plugin}:{agent}"); + } + return Ok(out); + }; + for candidate in [ + root.join("skills").join(name).join("SKILL.md"), + root.join("agents").join(format!("{name}.md")), + ] { + if let Ok(text) = std::fs::read_to_string(&candidate) { + return Ok(text); + } + } + Err(format!( + "`{plugin}` has no skill or agent `{name}`; `b10x skill {plugin}` lists them" + )) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn scratch() -> PathBuf { + use std::sync::atomic::{AtomicUsize, Ordering}; + static NEXT: AtomicUsize = AtomicUsize::new(0); + let path = std::env::temp_dir().join(format!( + "b10x-skill-{}-{}", + std::process::id(), + NEXT.fetch_add(1, Ordering::SeqCst) + )); + std::fs::create_dir_all(&path).unwrap(); + path + } + + #[test] + fn claude_install_path_is_read_and_skills_listed() { + let home = scratch(); + let install = home.join("cache/ess/0.30.0"); + std::fs::create_dir_all(install.join("skills/specify")).unwrap(); + std::fs::write( + install.join("skills/specify/SKILL.md"), + "---\nname: specify\n---\nbody", + ) + .unwrap(); + std::fs::create_dir_all(install.join("agents")).unwrap(); + std::fs::write(install.join("agents/author.md"), "agent").unwrap(); + std::fs::create_dir_all(home.join(".claude/plugins")).unwrap(); + let recorded = serde_json::json!({"plugins": {"ess@b10x": [{"scope": "user", "installPath": install}]}}); + std::fs::write( + home.join(".claude/plugins/installed_plugins.json"), + recorded.to_string(), + ) + .unwrap(); + + let listing = text(&home, "b10x", "ess").unwrap(); + assert!( + listing.contains("skill ess:specify") && listing.contains("agent ess:author"), + "{listing}" + ); + assert!(text(&home, "b10x", "ess:specify") + .unwrap() + .ends_with("body")); + assert_eq!(text(&home, "b10x", "ess:author").unwrap(), "agent"); + assert!(text(&home, "b10x", "ess:nothing").is_err()); + std::fs::remove_dir_all(&home).unwrap(); + } + + #[test] + fn codex_cache_takes_the_newest_version() { + let home = scratch(); + for version in ["0.9.0", "0.30.0"] { + let dir = home + .join(".codex/plugins/cache/b10x/ess") + .join(version) + .join("skills/specify"); + std::fs::create_dir_all(&dir).unwrap(); + std::fs::write(dir.join("SKILL.md"), version).unwrap(); + } + assert_eq!(text(&home, "b10x", "ess:specify").unwrap(), "0.30.0"); + assert!(text(&home, "b10x", "aep").is_err()); + std::fs::remove_dir_all(&home).unwrap(); + } +} diff --git a/crates/b10x/src/version.rs b/crates/b10x/src/version.rs new file mode 100644 index 0000000..c3ecf78 --- /dev/null +++ b/crates/b10x/src/version.rs @@ -0,0 +1,52 @@ +//! Reading versions out of `--version` lines and release tags. + +/// The last `x.y.z` token in a `--version` line or tag, without a leading `v`. +#[must_use] +pub fn parse(text: &str) -> Option { + text.split_whitespace() + .rev() + .map(|token| token.trim_start_matches('v')) + .find(|token| key(token).is_some()) + .map(str::to_owned) +} + +/// A comparable key for a bare `x.y.z` version. +#[must_use] +pub fn key(version: &str) -> Option<(u64, u64, u64)> { + let mut parts = version.trim_start_matches('v').split('.'); + let triple = ( + parts.next()?.parse().ok()?, + parts.next()?.parse().ok()?, + parts.next()?.parse().ok()?, + ); + parts.next().is_none().then_some(triple) +} + +/// Whether two version spellings name the same release (`v0.11.0` = `0.11.0`). +#[must_use] +pub fn same(left: &str, right: &str) -> bool { + match (key(left), key(right)) { + (Some(left), Some(right)) => left == right, + _ => left == right, + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn the_last_version_token_wins() { + assert_eq!(parse("protocol 0.57.0").as_deref(), Some("0.57.0")); + assert_eq!(parse("b10x-worktree-cli 0.4.1\n").as_deref(), Some("0.4.1")); + assert_eq!(parse("connectors v0.11.0").as_deref(), Some("0.11.0")); + assert_eq!(parse("no version here"), None); + } + + #[test] + fn tags_and_versions_compare() { + assert!(same("v0.11.0", "0.11.0")); + assert!(!same("0.26.0", "0.30.0")); + assert!(key("0.30.0") > key("0.26.0")); + } +} diff --git a/crates/b10x/tests/fixtures/claude-legacy-list.json b/crates/b10x/tests/fixtures/claude-legacy-list.json new file mode 100644 index 0000000..8b63850 --- /dev/null +++ b/crates/b10x/tests/fixtures/claude-legacy-list.json @@ -0,0 +1,47 @@ +[ + { + "id": "aep-drive@beyond10x", + "version": "0.10.0", + "scope": "user", + "enabled": true, + "installPath": "/opt/b10x-home/.claude/plugins/cache/beyond10x/aep-drive/0.10.0", + "installedAt": "2026-09-24T12:17:07.833Z", + "lastUpdated": "2026-09-24T12:17:07.833Z" + }, + { + "id": "aep-plan@beyond10x", + "version": "0.10.0", + "scope": "user", + "enabled": true, + "installPath": "/opt/b10x-home/.claude/plugins/cache/beyond10x/aep-plan/0.10.0", + "installedAt": "2026-09-24T12:17:07.406Z", + "lastUpdated": "2026-09-24T12:17:07.406Z" + }, + { + "id": "beyond10x@beyond10x", + "version": "0.10.0", + "scope": "user", + "enabled": true, + "installPath": "/opt/b10x-home/.claude/plugins/cache/beyond10x/beyond10x/0.10.0", + "installedAt": "2026-09-24T12:17:08.803Z", + "lastUpdated": "2026-09-24T12:17:08.803Z" + }, + { + "id": "ess@ess", + "version": "0.30.0", + "scope": "user", + "enabled": true, + "installPath": "/opt/b10x-home/.claude/plugins/cache/ess/ess/0.30.0", + "installedAt": "2026-09-24T12:17:14.322Z", + "lastUpdated": "2026-09-24T12:17:14.322Z" + }, + { + "id": "workspace-hygiene@beyond10x", + "version": "0.10.0", + "scope": "user", + "enabled": true, + "installPath": "/opt/b10x-home/.claude/plugins/cache/beyond10x/workspace-hygiene/0.10.0", + "installedAt": "2026-09-24T12:17:08.287Z", + "lastUpdated": "2026-09-24T12:17:08.287Z" + } +] diff --git a/crates/b10x/tests/fixtures/claude-legacy-marketplaces.json b/crates/b10x/tests/fixtures/claude-legacy-marketplaces.json new file mode 100644 index 0000000..2ee59e3 --- /dev/null +++ b/crates/b10x/tests/fixtures/claude-legacy-marketplaces.json @@ -0,0 +1,16 @@ +[ + { + "name": "beyond10x", + "source": "git", + "url": "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/beyond10x/agentplugins.git", + "ref": "0.10.0", + "installLocation": "/opt/b10x-home/.claude/plugins/marketplaces/beyond10x" + }, + { + "name": "ess", + "source": "git", + "url": "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/beyond10x/ess.git", + "ref": "0.30.0", + "installLocation": "/opt/b10x-home/.claude/plugins/marketplaces/ess" + } +] diff --git a/crates/b10x/tests/fixtures/codex-legacy-config.toml b/crates/b10x/tests/fixtures/codex-legacy-config.toml new file mode 100644 index 0000000..6b07ac4 --- /dev/null +++ b/crates/b10x/tests/fixtures/codex-legacy-config.toml @@ -0,0 +1,16 @@ +[marketplaces.beyond10x] +source_type = "git" +source = "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/beyond10x/agentplugins.git" +ref = "0.10.0" + +[plugins."aep-plan@beyond10x"] +enabled = true + +[plugins."aep-drive@beyond10x"] +enabled = true + +[plugins."workspace-hygiene@beyond10x"] +enabled = true + +[plugins."beyond10x@beyond10x"] +enabled = true diff --git a/crates/b10x/tests/fixtures/codex-legacy-list.json b/crates/b10x/tests/fixtures/codex-legacy-list.json new file mode 100644 index 0000000..bc32607 --- /dev/null +++ b/crates/b10x/tests/fixtures/codex-legacy-list.json @@ -0,0 +1,77 @@ +{ + "installed": [ + { + "pluginId": "beyond10x@beyond10x", + "name": "beyond10x", + "marketplaceName": "beyond10x", + "version": "0.10.0", + "installed": true, + "enabled": true, + "source": { + "source": "local", + "path": "/opt/b10x-home/.codex/.tmp/marketplaces/beyond10x/plugins/beyond10x" + }, + "marketplaceSource": { + "sourceType": "git", + "source": "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/beyond10x/agentplugins.git" + }, + "installPolicy": "AVAILABLE", + "authPolicy": "ON_INSTALL" + }, + { + "pluginId": "aep-plan@beyond10x", + "name": "aep-plan", + "marketplaceName": "beyond10x", + "version": "0.10.0", + "installed": true, + "enabled": true, + "source": { + "source": "local", + "path": "/opt/b10x-home/.codex/.tmp/marketplaces/beyond10x/plugins/aep-plan" + }, + "marketplaceSource": { + "sourceType": "git", + "source": "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/beyond10x/agentplugins.git" + }, + "installPolicy": "AVAILABLE", + "authPolicy": "ON_INSTALL" + }, + { + "pluginId": "aep-drive@beyond10x", + "name": "aep-drive", + "marketplaceName": "beyond10x", + "version": "0.10.0", + "installed": true, + "enabled": true, + "source": { + "source": "local", + "path": "/opt/b10x-home/.codex/.tmp/marketplaces/beyond10x/plugins/aep-drive" + }, + "marketplaceSource": { + "sourceType": "git", + "source": "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/beyond10x/agentplugins.git" + }, + "installPolicy": "AVAILABLE", + "authPolicy": "ON_INSTALL" + }, + { + "pluginId": "workspace-hygiene@beyond10x", + "name": "workspace-hygiene", + "marketplaceName": "beyond10x", + "version": "0.10.0", + "installed": true, + "enabled": true, + "source": { + "source": "local", + "path": "/opt/b10x-home/.codex/.tmp/marketplaces/beyond10x/plugins/workspace-hygiene" + }, + "marketplaceSource": { + "sourceType": "git", + "source": "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/beyond10x/agentplugins.git" + }, + "installPolicy": "AVAILABLE", + "authPolicy": "ON_INSTALL" + } + ], + "available": [] +} diff --git a/crates/b10x/tests/fixtures/codex-legacy-marketplaces.json b/crates/b10x/tests/fixtures/codex-legacy-marketplaces.json new file mode 100644 index 0000000..5bfc182 --- /dev/null +++ b/crates/b10x/tests/fixtures/codex-legacy-marketplaces.json @@ -0,0 +1,12 @@ +{ + "marketplaces": [ + { + "name": "beyond10x", + "root": "/opt/b10x-home/.codex/.tmp/marketplaces/beyond10x", + "marketplaceSource": { + "sourceType": "git", + "source": "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/beyond10x/agentplugins.git" + } + } + ] +} diff --git a/evals/README.md b/evals/README.md index 2449845..434a8db 100644 --- a/evals/README.md +++ b/evals/README.md @@ -12,28 +12,27 @@ A case is four things and no others: | the expectations | `expectations.trace.yaml` | a `trace-spec/1` document: what the run must have looked like | | the transcript | `recorded/` | a recorded run, replayed through the checker for nothing | -## The nine cases +## The eight cases Every cell in the middle column is that case's `subject:` in full, read from its `case.yaml`, and those fields — not this table and not any prose elsewhere — are the source of truth for what the -corpus covers: counted from them the nine cases name **7 of this repository's 10 agents** and -**5 of its 10 skills**. +corpus covers: counted from them the eight cases name **7 of this repository's 11 agents** and +**4 of its 8 skills**. | Case | `subject:` agents and skills | The claim it holds the subject to | |---|---|---| -| `plan-critic-acceptance-verdict` | `aep-plan:plan-critic-acceptance` | the verdict became an immutable `review-result` through `new --from`, and nothing moved | -| `plan-critic-design-verdict` | `aep-plan:plan-critic-design` | the same, for the shape lane | -| `plan-critic-scope-verdict` | `aep-plan:plan-critic-scope` | the same, for the coverage lane | -| `plan-critic-parallel-safety-verdict` | `aep-plan:plan-critic-parallel-safety` | the same, for the concurrency lane | -| `decomposer-relation-census` | `aep-plan:decomposer` | an undecided relation became a `decision-blocker` with a `blocks` edge, filed before the first story | -| `ess-specify-new-entity` | `ess-specify:specify` | the noun got a validated typed home before a story rested on it, and the unread relation stayed unread | -| `golden-path-end-to-end` | `aep-plan:decomposer`, `aep-plan:plan-critic-acceptance`, `aep-drive:story-scoper`, `aep-plan:planning`, `ess-specify:specify`, `aep-drive:wave`, `aep-drive:drive` (and the path `website/docs/golden-path.md`) | the eight published steps in the published order, with the CLIs as the stores' only writers | -| `adversary-tests-only` | `aep-drive:adversary`, `aep-drive:wave` | tests were written, `src/` was not touched, and no `aep plan artifact` command ran | -| `connectors-readiness` | `connectors:connectors` | diagnosis uses the CLI; help is allowed and Connector mutations are rejected | - -No case names the agents `aep-drive:implementor`, `aep-plan:plan-reviewer` or -`aep-plan:reverse-engineer`, nor the skills `aep-plan:story-migration`, `beyond10x:beyond10x`, -`beyond10x:plugin-creator`, `ess-specify:coverage` or `workspace-hygiene:worktree`; those eight are +| `plan-critic-acceptance-verdict` | `aep:plan-critic-acceptance` | the verdict became an immutable `review-result` through `new --from`, and nothing moved | +| `plan-critic-design-verdict` | `aep:plan-critic-design` | the same, for the shape lane | +| `plan-critic-scope-verdict` | `aep:plan-critic-scope` | the same, for the coverage lane | +| `plan-critic-parallel-safety-verdict` | `aep:plan-critic-parallel-safety` | the same, for the concurrency lane | +| `decomposer-relation-census` | `aep:decomposer` | an undecided relation became a `decision-blocker` with a `blocks` edge, filed before the first story | +| `golden-path-end-to-end` | `aep:decomposer`, `aep:plan-critic-acceptance`, `aep:story-scoper`, `aep:planning`, `aep:implementing`, `aep:implementing` (and the path `website/docs/golden-path.md`) | the eight published steps in the published order, with the CLIs as the stores' only writers | +| `adversary-tests-only` | `aep:adversary`, `aep:implementing` | tests were written, `src/` was not touched, and no `aep plan artifact` command ran | +| `connectors-readiness` | `connectors:integrating` | diagnosis uses the CLI; help is allowed and Connector mutations are rejected | + +No case names the agents `aep:implementor`, `aep:plan-reviewer`, `aep:reverse-engineer` or +`aep:security-reviewer`, nor the skills `aep:migrating`, `b10x:init`, `b10x:routing` or +`b10x:authoring-plugins`; those eight are the remaining scope of `story:plugin-eval-cases` in `.engineering/planning`, and a change to one of them turns no row red. @@ -143,16 +142,53 @@ Free, and what `task check` does: $ cargo run --quiet --locked --bin agentplugins-check ``` -Live, which costs money and is refused without both `METAHARNESS_LIVE=1` and a cap — see -[`README.md` § Evals](../README.md#evals) for the arithmetic of a full sweep: +Live, which costs money and is refused without both `METAHARNESS_LIVE=1` and a cap: ```console $ METAHARNESS_LIVE=1 aep drive eval run --corpus evals --workflow adp/default \ - --arm plugin --harness claude --plugin-dir plugins/aep-plan \ + --arm plugin --harness claude --plugin-dir plugins/aep \ --cwd --budget-usd 20 --assume-usd-per-run 5 \ --observed-at --redact --out ``` +Without `METAHARNESS_LIVE=1` the runner accepts the corpus and refuses to spawn, by name: + +```console +$ aep drive eval run --corpus evals --workflow adp/default --arm plugin --harness claude \ + --out eval-out --observed-at 2026-09-03 +error: eval-out — 1 refusal(s): + EVAL-RUN-002 a spawn costs money and `METAHARNESS_LIVE=1` is not in this environment. Set it + deliberately, or pass `--stream FILE` to ingest a run that already happened, which spends nothing +``` + +## What a full live run costs + +| | | +|---|---| +| cases in the corpus | **8** | +| per-case cap | **$5** — `story:plugin-eval-cases`, the operator's default | +| one full run, one arm, one harness | **$40** | +| `EVAL_BUDGET_USD` default | **$20** — `story:eval-ci-gates`, the operator's default | + +**So a full sweep does not fit its own default budget, and that is the intended behaviour rather +than an oversight.** `.github/workflows/eval.yml` computes `cases × $5` before it installs a tool, +and refuses with those four numbers in the check summary when the product exceeds +`EVAL_BUDGET_USD`. What fits inside $20 is a diff-scoped run of up to four cases, which is what a +pull request touching one agent or one skill actually selects. Running the whole corpus is a +deliberate act: raise the repository variable, or dispatch one case at a time. + +The cap is a **cap, not an estimate** — no recorded run has priced this corpus yet, so nothing here +claims a full sweep will cost $40 rather than refusing above it. `--assume-usd-per-run` is what the +runner charges a run whose stream states no cost, and it is set to the per-case cap so the runner's +own pre-spawn check is made against the budgeted number and not against its optimistic default. + +## CI + +`ci.yml` runs the free half on every pull request and it is what blocks a merge. `eval.yml` runs the +live arm only with the `run-eval` label or a manual dispatch, only for the cases whose subject the +diff touched, under the budget above, with the organization bot's credential and never a personal +key. It informs; it does not gate. + ## Adding a case ```console diff --git a/evals/adversary-tests-only/case.yaml b/evals/adversary-tests-only/case.yaml index 47a60ea..e73d4d7 100644 --- a/evals/adversary-tests-only/case.yaml +++ b/evals/adversary-tests-only/case.yaml @@ -10,10 +10,10 @@ verdict: held # `skills:` name nothing that exists, and `.github/workflows/eval.yml` scopes a live run to the # cases whose subject the pull request touched. subject: - agents: [aep-drive:adversary] - skills: [aep-drive:wave] + agents: [aep:adversary] + skills: [aep:implementing] paths: - - plugins/aep-drive/skills/wave/references/unit-brief.md + - plugins/aep/skills/implementing/references/unit-brief.md # **The best-instrumented case in this corpus, and the reason is in the agent file's own last # section.** The adversary is the one agent here whose charter grants `Edit` and `Write` and then @@ -52,7 +52,7 @@ subject: task: | `story:commercial-client-record` was implemented in this worktree and the suite is green. Attack it - with the `aep-drive:adversary` agent. + with the `aep:adversary` agent. Your win condition is a failing suite, not a second opinion. Drive the implementation against the documents the unit wrote about itself, against the story's acceptance statement, and against the diff --git a/evals/adversary-tests-only/expectations.trace.yaml b/evals/adversary-tests-only/expectations.trace.yaml index be795f6..915a5ae 100644 --- a/evals/adversary-tests-only/expectations.trace.yaml +++ b/evals/adversary-tests-only/expectations.trace.yaml @@ -2,7 +2,7 @@ format: trace-spec/1 id: eval-case/adversary-tests-only title: The write verbs were offered, the tests were written, and the implementation was left alone -# **No row here reads `plugins/aep-drive/agents/adversary.md`.** Every row below is about what the run did. +# **No row here reads `plugins/aep/agents/adversary.md`.** Every row below is about what the run did. # # The write rows scope to a **set** of verbs — `[Write, Edit, NotebookEdit]` — and not to `Write` # alone, for the reason `aep/conformance/eval/README.md` bought with a live pilot: a run asked to @@ -17,7 +17,7 @@ expectations: statement: the plugin loaded and offered the adversary, so a miss is not "it was not there" expect: env.agent_available: - agent: aep-drive:adversary + agent: aep:adversary - id: a-subagent-actually-ran statement: the attack went through a subagent rather than being improvised by the parent session diff --git a/evals/connectors-readiness/case.yaml b/evals/connectors-readiness/case.yaml index 36e0fa3..df57cbd 100644 --- a/evals/connectors-readiness/case.yaml +++ b/evals/connectors-readiness/case.yaml @@ -6,7 +6,7 @@ states: [implement] arm: plugin verdict: held subject: - skills: [connectors:connectors] + skills: [connectors:integrating] paths: [crates/agentplugins-check/src/readiness.rs] command_contract: connectors-readiness task: | diff --git a/evals/connectors-readiness/expectations.trace.yaml b/evals/connectors-readiness/expectations.trace.yaml index 2a64ef8..b515433 100644 --- a/evals/connectors-readiness/expectations.trace.yaml +++ b/evals/connectors-readiness/expectations.trace.yaml @@ -9,7 +9,7 @@ expectations: severity: advisory expect: env.skill_available: - skill: connectors:connectors + skill: connectors:integrating - id: completed statement: the session completed without a harness error expect: diff --git a/evals/decomposer-relation-census/case.yaml b/evals/decomposer-relation-census/case.yaml index c778a38..94bf890 100644 --- a/evals/decomposer-relation-census/case.yaml +++ b/evals/decomposer-relation-census/case.yaml @@ -12,7 +12,7 @@ verdict: held subject: # The agent alone. Every row below is about what the decomposer did, and scoping to the planning # skill as well would put a paid run on this case every time an unrelated section of it moved. - agents: [aep-plan:decomposer] + agents: [aep:decomposer] # `aep/conformance/eval/decomposer-charter` already covers the decomposer's **charter** — it moved # nothing, it validated the store. This case covers the behaviour 0.4.0 added on top of it and that @@ -45,7 +45,7 @@ subject: # regex over both — a widening of what can witness the claim, never of the claim. task: | - Decompose `epic:commercial-clients` with the `aep-plan:decomposer` agent. + Decompose `epic:commercial-clients` with the `aep:decomposer` agent. Before you draft a single story, enumerate every domain relation the epic implies — the two entities in the direction the relation runs, cardinality, ownership, lifecycle coupling — and diff --git a/evals/decomposer-relation-census/expectations.trace.yaml b/evals/decomposer-relation-census/expectations.trace.yaml index 361a8b1..deb06a8 100644 --- a/evals/decomposer-relation-census/expectations.trace.yaml +++ b/evals/decomposer-relation-census/expectations.trace.yaml @@ -2,7 +2,7 @@ format: trace-spec/1 id: eval-case/decomposer-relation-census title: The undecided relation reached the store as a question, before any story was drafted -# **No row here reads `plugins/aep-plan/agents/decomposer.md`.** A check that greps an agent's +# **No row here reads `plugins/aep/agents/decomposer.md`.** A check that greps an agent's # markdown asserts that a sentence is still written, which is the failure mode a behavioural case # exists to remove. Every row below is about what the run did. # @@ -21,7 +21,7 @@ expectations: statement: the plugin loaded and offered the decomposer, so a miss is not "it was not there" expect: env.agent_available: - agent: aep-plan:decomposer + agent: aep:decomposer - id: a-subagent-actually-ran statement: the work went through a subagent rather than being done inline by the parent session diff --git a/evals/ess-specify-new-entity/case.yaml b/evals/ess-specify-new-entity/case.yaml deleted file mode 100644 index 50b5c5f..0000000 --- a/evals/ess-specify-new-entity/case.yaml +++ /dev/null @@ -1,67 +0,0 @@ -format: eval-case/1 -id: ess-specify-new-entity -title: A story introduced a noun and the noun got a typed home before anything was written around it -workflow: adp/default -states: [specify, decompose] -arm: plugin -verdict: held - -# Which repository files this case is a case about. `task check` refuses a case whose `agents:` or -# `skills:` name nothing that exists, and `.github/workflows/eval.yml` scopes a live run to the -# cases whose subject the pull request touched. -subject: - # The ESS skill alone. The one `aep plan artifact new story` matcher below is the far side of an - # ordering row, not a claim about the planning skill. - skills: [ess-specify:specify] - -# The skill's § *Starting a domain from nothing* is the subject: *"A story or epic can introduce a -# noun before any specification exists. That is still this skill's job: the domain is drafted first, -# so the noun has a typed home before stories are written around it."* That sentence is an -# **ordering**, which is the one class of claim a transcript decides well. -# -# The three things this case holds the run to, all of them the skill's own words: -# -# 1. the smallest document that validates is written — `system.yaml` and a domain file, nothing that -# is not required; -# 2. `ess specify validate` is run over it **before** stories are written around the noun; -# 3. a relation the tree does not settle is left as an `UNMAPPED:` marker, never invented — *"Imports -# never guess, and a domain an agent drafts is an import."* -# -# ## The skill's name, and how the ambiguity was closed -# -# The skill's directory is `plugins/ess-specify/skills/specify/` and its `SKILL.md` frontmatter -# declares `name: specify`, so the two agree and the row below has one spelling to hold, not two. -# They did not always: until 0.6.2 the directory and the frontmatter name disagreed, no committed -# evidence said which of the two a harness puts after the plugin name in its skill list, and -# `the-skill-was-offered` was **advisory** because a gating row on that guess would have failed this -# case for a reason that is not about behaviour. -# -# **The 2026-09-03 golden-path recording settled it: the harness lists the skill by its directory**, -# which is what made the directory and the name agreeing worth doing. The row stays advisory until a -# recording under the current spelling exists here — this case has none, and an advisory row that -# nobody has watched go `ok` is the honest severity for it. The behavioural rows below gate. -# -# ## The spelling of the CLI, and why the rows carry both -# -# `aep` in this repository's instructions, `protocol` in the binary's own usage lines, and an adopter -# may have either on `PATH`. Every `aep` matcher below is a regex over both. `ess` has one spelling -# and is matched as one. - -task: | - `story:warehouse-shipment-record` introduces a noun this repository has no typed home for: a - **shipment**, identified by a `shipment_id`, with a destination. - - Give it one before any story is written around it. There is no ESS specification here yet, so - write the smallest `ess/1` document that validates — the header and one domain file, and nothing - that is not required — and run `ess specify validate` over it. - - The epic also says a shipment has a **carrier**. Nothing in this tree, in any contract, and in no - artifact says what a carrier is. Do not type it. - - Then report: what you wrote, the verbatim output of `ess specify validate` with its exit status, and every - relation you could not settle. - -expectations: expectations.trace.yaml -recorded: recorded/ - -# The run did nothing wrong; what gaps is an observation about the evidence, not about the run. diff --git a/evals/ess-specify-new-entity/expectations.trace.yaml b/evals/ess-specify-new-entity/expectations.trace.yaml deleted file mode 100644 index 1e358e1..0000000 --- a/evals/ess-specify-new-entity/expectations.trace.yaml +++ /dev/null @@ -1,156 +0,0 @@ -format: trace-spec/1 -id: eval-case/ess-specify-new-entity -title: The domain was written and validated before a story rested on it, and the unread relation stayed unread - -# **No row here reads the `SKILL.md`.** A check that greps a skill's markdown asserts that a sentence -# is still written; every row below is about what the run did. -# -# The write rows scope to a **set** of verbs — `[Write, Edit, NotebookEdit]` — and not to `Write` -# alone. `aep/conformance/eval/README.md` bought that rule with a live pilot: a run that wrote a file -# that already existed did it with `Edit`, and a selector naming `Write` alone reported -# `never_occurred` — the checker shrugging at work that had visibly happened. Dropping the tool scope -# and matching `file_path` alone would be a real weakening, because `Read` carries a `file_path` too -# and *read the domain first* is not *wrote the domain first*. - -expectations: - # --- the skill was reached at all ----------------------------------------------------------- - - - id: the-skill-was-offered - statement: the plugin loaded and offered the ESS skill, so a miss is not "it was not there" - # Advisory, and the case header says why: the skill's directory name and its frontmatter name - # differ, and nothing committed here establishes which one a harness qualifies. A live run - # settles it, and this row is where the answer lands. - severity: advisory - expect: - env.skill_available: - skill: ess-specify:specify - - # --- the smallest document that validates --------------------------------------------------- - - - id: the-header-was-written - statement: an `ess/1` header naming the domain was written - expect: - tool.called: - tools: [Write, Edit, NotebookEdit] - args: - file_path: {glob: '*system.yaml'} - count: {at_least: 1} - - - id: the-domain-file-was-written - statement: the domain declaring the entity was written - # The header's `domains:` list and the declaring source must agree in both directions — either - # half alone is a refusal, which the skill states and `ess specify validate` enforces. So a run that - # wrote only one of these two files cannot have exited 0 on the row below, and the two rows - # together are the shape of *the smallest document that validates*. - expect: - tool.called: - tools: [Write, Edit, NotebookEdit] - args: - file_path: {glob: '*/domains/*.yaml'} - count: {at_least: 1} - - - id: the-specification-was-validated - statement: the specification was validated with `ess specify validate` after it was written - expect: - tool.called: - tool: Bash - args: - command: {regex: 'ess +(specify +)?validate'} - count: {at_least: 1} - - # --- the ordering the skill exists to produce ------------------------------------------------ - - - id: the-domain-was-validated-before-a-story-rested-on-it - statement: the noun had a validated typed home before any story was written around it - # The skill's first sentence as an ordering: *"the domain is drafted first, so the noun has a - # typed home before stories are written around it."* First-occurrence ordering over the event - # sequence, the same shape `adp/default`'s red-before-green bound takes. - # - # `unk` when either side never occurs — *A before B* is undecidable when there is no A, and - # reporting it as a failure blames the wrong thing. - expect: - order: - first: - tool: Bash - args: {command: {regex: 'ess +(specify +)?validate'}} - before: - tool: Bash - args: {command: {regex: '(aep|protocol) +(plan +)?artifact new +story'}} - - - id: no-projection-was-hand-written - statement: no generated schema or OpenAPI file was written by hand - # *"Use the projection command, never a handwritten parallel generator."* Non-vacuous: the run - # holds `Write` and `Edit` and used them, which `the-domain-file-was-written` is the control for - # over the same tool set. A projection that a run legitimately produced arrives through - # `ess generate` or `ess generate project`, which are `Bash` calls and not writes. - expect: - tool.absent: - tools: [Write, Edit, NotebookEdit] - args: - file_path: {glob: '*/generated/*'} - - # --- the relation nothing settles ------------------------------------------------------------- - - - id: the-unread-relation-was-named-rather-than-typed - statement: the report names the relation it could not read instead of inventing a type for it - # *"Every relation you could not read … is written with an `UNMAPPED:` marker beside the place it - # would go, and named again in the report."* `text.matches` is the weakest kind in the vocabulary - # and this row is the *named again in the report* half, so it is advisory. The marker in the file - # is the half that matters and no `trace-spec/1` kind reads a file's contents; `ess specify validate` - # exiting 0 with the carrier absent is what a reader checks, in the shell. - severity: advisory - expect: - text.matches: - contains: 'UNMAPPED' - - # --- the run's own record --------------------------------------------------------------------- - - - id: the-scope-was-actually-tested - statement: the run's write verbs put specification YAML on disk, so the projection-absence above was decided against a run that was writing - # Was `permission.denied: {at_least: 1}` until the first recording (2026-09-03) gapped it in all - # eight cases. The reason is not the runs: this seam observes and never adjudicates - # (`decided_by: observe`, `permission_mode: default` in every `session.started`), so no denial can - # be recorded at all — and the row contradicted `nothing-was-refused` directly below it. Two rows - # that can never both be `ok` are not a control. See `README.md` § *A control has to be able to - # pass*, which carries the whole argument and the selector form. - # - # Here: `no-projection-was-hand-written` bounds a shape of YAML write, and over a run that wrote - # no YAML at all it would be an inventory of nothing. - # - # Advisory, unchanged: it does not judge the run, it judges whether the evidence above it is - # worth anything. - severity: advisory - expect: - tool.called: - operations: [file.write, file.edit] - subject: {glob: 'file:*.yaml'} - count: {at_least: 1} - - - id: nothing-was-refused - statement: nothing machine-owned was hand-written, so no guard had to refuse anything - expect: - permission.denied: - count: {at_most: 0} - - - id: terminal-record-clean - statement: the session ended the way a finished session ends - expect: - result: - is_error: false - terminal_reason: completed - - # --- advisory --------------------------------------------------------------------------------- - - - id: turns-within-reason - statement: the domain draft did not loop - severity: advisory - expect: - turns: - count: {at_most: 40} - - - id: cost-within-the-per-case-budget - statement: the run stayed inside the per-case cap the corpus is run under - severity: advisory - expect: - cost.total: - at_most_usd: 5.0 diff --git a/evals/ess-specify-new-entity/recorded/README.md b/evals/ess-specify-new-entity/recorded/README.md deleted file mode 100644 index 8414b0b..0000000 --- a/evals/ess-specify-new-entity/recorded/README.md +++ /dev/null @@ -1,41 +0,0 @@ -# `recorded/` — `ess-specify` on a new entity - -**Empty on purpose, and this file says what would fill it.** No transcript of this case has been -recorded, and none was synthesized: a hand-written transcript here would be a fixture the case's own -rows were fitted to, which measures the document and not the plugin. - -`task check` skips an empty `recorded/` with a printed notice and does not fail. `aep drive eval run ---stream` is what reads a file once one is here; drop it in this directory and the replay picks it -up with no change to the case. - -## The run that would produce it - -Live, paid, and refused without both `METAHARNESS_LIVE=1` and a cap: - -```console -$ METAHARNESS_LIVE=1 aep drive eval run \ - --case evals/ess-specify-new-entity \ - --arm plugin \ - --harness claude \ - --plugin-dir plugins/ess-specify \ - --cwd \ - --budget-usd 5 \ - --observed-at \ - --redact \ - --out -``` - -`--redact` is not optional for anything committed here: an un-redacted record quotes the transcript, -and a report that quotes a transcript is not a thing to publish. - -The run leaves `/.events.jsonl` beside its manifest and record. **The stream is what -belongs in this directory**, copied in with the manifest's `observed_at`, harness version and model -recorded beside it — a transcript with no provenance is a file, not evidence. - -## What it needs in the working tree - -No `system.yaml` anywhere, a story introducing the shipment, and the `ess` binary on `PATH` -— the skill's own § *Starting a domain from nothing* is written for a repository with no -specification at all, and a tree that already has one measures a different behaviour. -`ess` is published by the sibling `beyond10x/ess` repository; `the-specification-was-validated` -gaps without it, and the gap is about the machine rather than about the run. diff --git a/evals/golden-path-end-to-end/case.yaml b/evals/golden-path-end-to-end/case.yaml index 04de17d..3febfa8 100644 --- a/evals/golden-path-end-to-end/case.yaml +++ b/evals/golden-path-end-to-end/case.yaml @@ -10,8 +10,8 @@ verdict: held # `skills:` name nothing that exists, and `.github/workflows/eval.yml` scopes a live run to the # cases whose subject the pull request touched. subject: - agents: [aep-plan:decomposer, aep-plan:plan-critic-acceptance, aep-drive:story-scoper] - skills: [aep-plan:planning, ess-specify:specify, aep-drive:wave, aep-drive:drive] + agents: [aep:decomposer, aep:plan-critic-acceptance, aep:story-scoper] + skills: [aep:planning, aep:implementing] paths: - website/docs/golden-path.md @@ -115,7 +115,7 @@ task: | revise the drafts that come back needing revision, stop after two rounds, and list what is still open. - 7. Take `story:commercial-client-record` through the aep-drive wave: scope it into units, implement the + 7. Take `story:commercial-client-record` through the aep wave: scope it into units, implement the units, and have the adversary review the result against the story's acceptance and this repository's gate. Walk the story's lifecycle deliberately, and say that you did. diff --git a/evals/golden-path-end-to-end/expectations.trace.yaml b/evals/golden-path-end-to-end/expectations.trace.yaml index bbed6e3..aa7291b 100644 --- a/evals/golden-path-end-to-end/expectations.trace.yaml +++ b/evals/golden-path-end-to-end/expectations.trace.yaml @@ -51,7 +51,7 @@ expectations: # `--plugin @@` (aep 0.44.0), which metaharness places in the scratch config home # and, since 0.6.1, also names to the vendor with `--plugin-dir` — the registry placement alone # loaded nothing (metaharness probe Q19, answered by the 2026-09-03 recording). A run launched - # with only `--plugin-dir plugins/aep-plan` still gaps this row for a reason that is not about + # with only `--plugin-dir plugins/aep` still gaps this row for a reason that is not about # the plugin, which is why it stays advisory until the recorded stream shows it `ok`. Since the # rename the manifest's pins name no plugin this repository still ships, so the replay sends the # pins alone and no `--plugin-dir` at all — an argument `aep`'s `ingest_recorded` discards. @@ -366,7 +366,7 @@ expectations: statement: the run stayed inside the per-case cap the corpus is run under # 25 and not the corpus's 5, in two moves. The second headless run (2026-09-03) walked all eight # steps in 118 turns on the default model for $10.96, which retired the $5 cap. The third - # (2026-09-03, out-3) was the first with `aep-drive` actually loaded: it took the wave skill, spawned + # (2026-09-03, out-3) was the first with the delivery plugin actually loaded: it took the wave skill, spawned # 12 sub-agents, and was still inside step 7 when `--max-budget-usd` stopped it at $15.0014 # (`session.ended` subtype `error_max_budget_usd`). The case costs more with its plugins than # without them, which is the case working. The runner passes the same number as `--budget-usd`, diff --git a/evals/golden-path-end-to-end/recorded/README.md b/evals/golden-path-end-to-end/recorded/README.md index e7d20c3..099bfd1 100644 --- a/evals/golden-path-end-to-end/recorded/README.md +++ b/evals/golden-path-end-to-end/recorded/README.md @@ -41,7 +41,7 @@ $ METAHARNESS_LIVE=1 aep drive eval run \ --harness claude \ --plugin-dir plugins/aep-plan \ --plugin beyond10x/agentplugins@aep-drive@ \ - --plugin beyond10x/agentplugins@ess-specify@ \ + --plugin beyond10x/ess@ess@ \ --cwd \ --budget-usd 25 \ --observed-at \ @@ -72,7 +72,7 @@ a `TODO` at the deletion site, and **no `.engineering/` directory**. § 1 of the adoption step, and it measures nothing against a tree that has already been adopted. This is the one case that needs three plugins installed — `aep-plan` for the planning steps, -`ess-specify` for step 3, and `aep-drive` for steps 7 and 8. `aep drive eval run --plugin-dir` takes one; the +`ess` from the `beyond10x/ess` marketplace for step 3, and `aep-drive` for steps 7 and 8. `aep drive eval run --plugin-dir` takes one; the other two go as `--plugin` pins, above. Two things the second run (2026-09-03) showed the working tree also needs, both about the **child's** `PATH`, which metaharness constructs as `$HOME/.local/bin:/usr/local/bin:/usr/bin:/bin` and does not inherit: diff --git a/evals/plan-critic-acceptance-verdict/case.yaml b/evals/plan-critic-acceptance-verdict/case.yaml index ddfe604..507ea2a 100644 --- a/evals/plan-critic-acceptance-verdict/case.yaml +++ b/evals/plan-critic-acceptance-verdict/case.yaml @@ -10,16 +10,16 @@ verdict: held # `skills:` name nothing that exists, and `.github/workflows/eval.yml` scopes a live run to the # cases whose subject the pull request touched. subject: - agents: [aep-plan:plan-critic-acceptance] + agents: [aep:plan-critic-acceptance] paths: # The rubric, not the whole planning skill: the rubric holds the verdict rule every row here # rests on, and scoping to the skill directory would put a paid run on this case every time an # unrelated section of `SKILL.md` moved. - - plugins/aep-plan/skills/planning/references/critic-rubric.md + - plugins/aep/skills/planning/references/critic-rubric.md # What this case is about, in one line: **a critic argues and the caller records**, and the two # halves are one run. The critic's own charter says it records nothing -# (`plugins/aep-plan/skills/planning/references/critic-rubric.md` § *You change nothing*: +# (`plugins/aep/skills/planning/references/critic-rubric.md` § *You change nothing*: # *"You do not record your own verdict. The caller writes the record"*), so a case scoped to the # subagent alone could not observe the thing that makes a verdict worth having — that it became an # immutable `review-result` in the store instead of a paragraph in a session nobody kept. @@ -50,7 +50,7 @@ subject: # there is no session-wide form of the row either. task: | - Run the `aep-plan:plan-critic-acceptance` agent over the stories drafted under + Run the `aep:plan-critic-acceptance` agent over the stories drafted under `epic:commercial-clients`, and judge whether every drafted acceptance statement names an observable outcome and the transition it turns on. Give the critic the story ids and the epic id, and nothing else. diff --git a/evals/plan-critic-acceptance-verdict/expectations.trace.yaml b/evals/plan-critic-acceptance-verdict/expectations.trace.yaml index 40ae099..d19d7f9 100644 --- a/evals/plan-critic-acceptance-verdict/expectations.trace.yaml +++ b/evals/plan-critic-acceptance-verdict/expectations.trace.yaml @@ -2,7 +2,7 @@ format: trace-spec/1 id: eval-case/plan-critic-acceptance-verdict title: The acceptance critic argued, the caller recorded the verdict, and nothing moved -# **No row here reads `plugins/aep-plan/agents/plan-critic-acceptance.md`.** A check that greps an +# **No row here reads `plugins/aep/agents/plan-critic-acceptance.md`.** A check that greps an # agent's markdown asserts that a sentence is still written, which is the failure mode a behavioural # case exists to remove. Every row below is about what the run did. # @@ -18,7 +18,7 @@ expectations: statement: the plugin loaded and offered the acceptance critic, so a miss is not "it was not there" expect: env.agent_available: - agent: aep-plan:plan-critic-acceptance + agent: aep:plan-critic-acceptance - id: a-subagent-actually-ran statement: the critique went through a subagent rather than being improvised by the parent session diff --git a/evals/plan-critic-design-verdict/case.yaml b/evals/plan-critic-design-verdict/case.yaml index 0a463a3..09ed4e4 100644 --- a/evals/plan-critic-design-verdict/case.yaml +++ b/evals/plan-critic-design-verdict/case.yaml @@ -10,16 +10,16 @@ verdict: held # `skills:` name nothing that exists, and `.github/workflows/eval.yml` scopes a live run to the # cases whose subject the pull request touched. subject: - agents: [aep-plan:plan-critic-design] + agents: [aep:plan-critic-design] paths: # The rubric, not the whole planning skill: the rubric holds the verdict rule every row here # rests on, and scoping to the skill directory would put a paid run on this case every time an # unrelated section of `SKILL.md` moved. - - plugins/aep-plan/skills/planning/references/critic-rubric.md + - plugins/aep/skills/planning/references/critic-rubric.md # What this case is about, in one line: **a critic argues and the caller records**, and the two # halves are one run. The critic's own charter says it records nothing -# (`plugins/aep-plan/skills/planning/references/critic-rubric.md` § *You change nothing*: +# (`plugins/aep/skills/planning/references/critic-rubric.md` § *You change nothing*: # *"You do not record your own verdict. The caller writes the record"*), so a case scoped to the # subagent alone could not observe the thing that makes a verdict worth having — that it became an # immutable `review-result` in the store instead of a paragraph in a session nobody kept. @@ -50,7 +50,7 @@ subject: # there is no session-wide form of the row either. task: | - Run the `aep-plan:plan-critic-design` agent over the stories drafted under + Run the `aep:plan-critic-design` agent over the stories drafted under `epic:commercial-clients`, and judge whether the drafted set is the right shape — cycles, a serialising chain, a split abstraction, an unrecorded dependency. Give the critic the story ids and the epic id, and nothing else. diff --git a/evals/plan-critic-design-verdict/expectations.trace.yaml b/evals/plan-critic-design-verdict/expectations.trace.yaml index fbbe9d4..e1cfd3e 100644 --- a/evals/plan-critic-design-verdict/expectations.trace.yaml +++ b/evals/plan-critic-design-verdict/expectations.trace.yaml @@ -2,7 +2,7 @@ format: trace-spec/1 id: eval-case/plan-critic-design-verdict title: The design critic argued, the caller recorded the verdict, and nothing moved -# **No row here reads `plugins/aep-plan/agents/plan-critic-design.md`.** A check that greps an +# **No row here reads `plugins/aep/agents/plan-critic-design.md`.** A check that greps an # agent's markdown asserts that a sentence is still written, which is the failure mode a behavioural # case exists to remove. Every row below is about what the run did. # @@ -18,7 +18,7 @@ expectations: statement: the plugin loaded and offered the design critic, so a miss is not "it was not there" expect: env.agent_available: - agent: aep-plan:plan-critic-design + agent: aep:plan-critic-design - id: a-subagent-actually-ran statement: the critique went through a subagent rather than being improvised by the parent session diff --git a/evals/plan-critic-parallel-safety-verdict/case.yaml b/evals/plan-critic-parallel-safety-verdict/case.yaml index 4198d7b..a2ff3c7 100644 --- a/evals/plan-critic-parallel-safety-verdict/case.yaml +++ b/evals/plan-critic-parallel-safety-verdict/case.yaml @@ -10,16 +10,16 @@ verdict: held # `skills:` name nothing that exists, and `.github/workflows/eval.yml` scopes a live run to the # cases whose subject the pull request touched. subject: - agents: [aep-plan:plan-critic-parallel-safety] + agents: [aep:plan-critic-parallel-safety] paths: # The rubric, not the whole planning skill: the rubric holds the verdict rule every row here # rests on, and scoping to the skill directory would put a paid run on this case every time an # unrelated section of `SKILL.md` moved. - - plugins/aep-plan/skills/planning/references/critic-rubric.md + - plugins/aep/skills/planning/references/critic-rubric.md # What this case is about, in one line: **a critic argues and the caller records**, and the two # halves are one run. The critic's own charter says it records nothing -# (`plugins/aep-plan/skills/planning/references/critic-rubric.md` § *You change nothing*: +# (`plugins/aep/skills/planning/references/critic-rubric.md` § *You change nothing*: # *"You do not record your own verdict. The caller writes the record"*), so a case scoped to the # subagent alone could not observe the thing that makes a verdict worth having — that it became an # immutable `review-result` in the store instead of a paragraph in a session nobody kept. @@ -50,7 +50,7 @@ subject: # there is no session-wide form of the row either. task: | - Run the `aep-plan:plan-critic-parallel-safety` agent over the stories drafted under + Run the `aep:plan-critic-parallel-safety` agent over the stories drafted under `epic:commercial-clients`, and judge which pair of drafted items lands on one surface, and whether the plan says so. Give the critic the story ids and the epic id, and nothing else. diff --git a/evals/plan-critic-parallel-safety-verdict/expectations.trace.yaml b/evals/plan-critic-parallel-safety-verdict/expectations.trace.yaml index 135c7b3..8bcba87 100644 --- a/evals/plan-critic-parallel-safety-verdict/expectations.trace.yaml +++ b/evals/plan-critic-parallel-safety-verdict/expectations.trace.yaml @@ -2,7 +2,7 @@ format: trace-spec/1 id: eval-case/plan-critic-parallel-safety-verdict title: The parallel-safety critic argued, the caller recorded the verdict, and nothing moved -# **No row here reads `plugins/aep-plan/agents/plan-critic-parallel-safety.md`.** A check that greps an +# **No row here reads `plugins/aep/agents/plan-critic-parallel-safety.md`.** A check that greps an # agent's markdown asserts that a sentence is still written, which is the failure mode a behavioural # case exists to remove. Every row below is about what the run did. # @@ -18,7 +18,7 @@ expectations: statement: the plugin loaded and offered the parallel-safety critic, so a miss is not "it was not there" expect: env.agent_available: - agent: aep-plan:plan-critic-parallel-safety + agent: aep:plan-critic-parallel-safety - id: a-subagent-actually-ran statement: the critique went through a subagent rather than being improvised by the parent session diff --git a/evals/plan-critic-scope-verdict/case.yaml b/evals/plan-critic-scope-verdict/case.yaml index c25e5b7..70d6fca 100644 --- a/evals/plan-critic-scope-verdict/case.yaml +++ b/evals/plan-critic-scope-verdict/case.yaml @@ -10,16 +10,16 @@ verdict: held # `skills:` name nothing that exists, and `.github/workflows/eval.yml` scopes a live run to the # cases whose subject the pull request touched. subject: - agents: [aep-plan:plan-critic-scope] + agents: [aep:plan-critic-scope] paths: # The rubric, not the whole planning skill: the rubric holds the verdict rule every row here # rests on, and scoping to the skill directory would put a paid run on this case every time an # unrelated section of `SKILL.md` moved. - - plugins/aep-plan/skills/planning/references/critic-rubric.md + - plugins/aep/skills/planning/references/critic-rubric.md # What this case is about, in one line: **a critic argues and the caller records**, and the two # halves are one run. The critic's own charter says it records nothing -# (`plugins/aep-plan/skills/planning/references/critic-rubric.md` § *You change nothing*: +# (`plugins/aep/skills/planning/references/critic-rubric.md` § *You change nothing*: # *"You do not record your own verdict. The caller writes the record"*), so a case scoped to the # subagent alone could not observe the thing that makes a verdict worth having — that it became an # immutable `review-result` in the store instead of a paragraph in a session nobody kept. @@ -50,7 +50,7 @@ subject: # there is no session-wide form of the row either. task: | - Run the `aep-plan:plan-critic-scope` agent over the stories drafted under + Run the `aep:plan-critic-scope` agent over the stories drafted under `epic:commercial-clients`, and judge whether every promise the parent makes is claimed by something in the set, and whether anything in the set was not asked for. Give the critic the story ids and the epic id, and nothing else. diff --git a/evals/plan-critic-scope-verdict/expectations.trace.yaml b/evals/plan-critic-scope-verdict/expectations.trace.yaml index 85b0b14..3d5436f 100644 --- a/evals/plan-critic-scope-verdict/expectations.trace.yaml +++ b/evals/plan-critic-scope-verdict/expectations.trace.yaml @@ -2,7 +2,7 @@ format: trace-spec/1 id: eval-case/plan-critic-scope-verdict title: The scope critic argued, the caller recorded the verdict, and nothing moved -# **No row here reads `plugins/aep-plan/agents/plan-critic-scope.md`.** A check that greps an +# **No row here reads `plugins/aep/agents/plan-critic-scope.md`.** A check that greps an # agent's markdown asserts that a sentence is still written, which is the failure mode a behavioural # case exists to remove. Every row below is about what the run did. # @@ -18,7 +18,7 @@ expectations: statement: the plugin loaded and offered the scope critic, so a miss is not "it was not there" expect: env.agent_available: - agent: aep-plan:plan-critic-scope + agent: aep:plan-critic-scope - id: a-subagent-actually-ran statement: the critique went through a subagent rather than being improvised by the parent session diff --git a/plugins/aep-drive/.claude-plugin/plugin.json b/plugins/aep-drive/.claude-plugin/plugin.json deleted file mode 100644 index e15ef9e..0000000 --- a/plugins/aep-drive/.claude-plugin/plugin.json +++ /dev/null @@ -1,10 +0,0 @@ -{ - "name": "aep-drive", - "displayName": "AEP Drive", - "description": "Coordinate AEP development waves, story scoping, implementation, and adversarial review.", - "version": "0.9.2", - "author": { "name": "Beyond10x" }, - "license": "Apache-2.0", - "repository": "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/beyond10x/agentplugins", - "keywords": ["aep", "development", "waves", "implementation", "review"] -} diff --git a/plugins/aep-drive/.codex-plugin/plugin.json b/plugins/aep-drive/.codex-plugin/plugin.json deleted file mode 100644 index 6a86aa2..0000000 --- a/plugins/aep-drive/.codex-plugin/plugin.json +++ /dev/null @@ -1,18 +0,0 @@ -{ - "name": "aep-drive", - "version": "0.9.2", - "description": "Coordinate AEP development waves with scoped implementors and adversarial review.", - "author": { - "name": "Beyond10x" - }, - "skills": "./skills/", - "interface": { - "displayName": "AEP Drive", - "shortDescription": "Coordinate governed development waves.", - "longDescription": "Scope stories, coordinate isolated implementation units, and route each result through adversarial review under the Agentic Development Protocol profile.", - "developerName": "Beyond10x", - "category": "Productivity", - "capabilities": [], - "defaultPrompt": "Use aep-drive to scope and coordinate this implementation." - } -} diff --git a/plugins/aep-plan/.claude-plugin/plugin.json b/plugins/aep-plan/.claude-plugin/plugin.json deleted file mode 100644 index b3de9be..0000000 --- a/plugins/aep-plan/.claude-plugin/plugin.json +++ /dev/null @@ -1,10 +0,0 @@ -{ - "name": "aep-plan", - "displayName": "AEP Plan", - "description": "Govern AEP planning artifacts, decomposition, review, and reverse engineering.", - "version": "0.9.2", - "author": { "name": "Beyond10x" }, - "license": "Apache-2.0", - "repository": "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/beyond10x/agentplugins", - "keywords": ["aep", "planning", "backlog", "decomposition", "review"] -} diff --git a/plugins/aep-plan/.codex-plugin/plugin.json b/plugins/aep-plan/.codex-plugin/plugin.json deleted file mode 100644 index c366bbe..0000000 --- a/plugins/aep-plan/.codex-plugin/plugin.json +++ /dev/null @@ -1,18 +0,0 @@ -{ - "name": "aep-plan", - "version": "0.9.2", - "description": "Govern AEP planning artifacts and review plans without bypassing their lifecycle.", - "author": { - "name": "Beyond10x" - }, - "skills": "./skills/", - "interface": { - "displayName": "AEP Plan", - "shortDescription": "Govern plans through the AEP artifact store.", - "longDescription": "Create and review governed planning artifacts, decompose work, and reverse-engineer an existing repository through the canonical AEP command.", - "developerName": "Beyond10x", - "category": "Productivity", - "capabilities": [], - "defaultPrompt": "Use aep-plan to inspect the store and help me govern this work." - } -} diff --git a/plugins/aep/.claude-plugin/plugin.json b/plugins/aep/.claude-plugin/plugin.json new file mode 100644 index 0000000..f212542 --- /dev/null +++ b/plugins/aep/.claude-plugin/plugin.json @@ -0,0 +1,20 @@ +{ + "name": "aep", + "displayName": "AEP", + "description": "Plan governed work in the AEP artifact store and deliver it in reviewed waves: decomposition, plan critique, reverse engineering, story scoping, implementation and adversarial review.", + "version": "0.14.13", + "author": { + "name": "Beyond10x" + }, + "license": "Apache-2.0", + "repository": "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/beyond10x/agentplugins", + "keywords": [ + "aep", + "planning", + "backlog", + "decomposition", + "waves", + "implementation", + "review" + ] +} diff --git a/plugins/aep/.codex-plugin/plugin.json b/plugins/aep/.codex-plugin/plugin.json new file mode 100644 index 0000000..1d11d4c --- /dev/null +++ b/plugins/aep/.codex-plugin/plugin.json @@ -0,0 +1,18 @@ +{ + "name": "aep", + "version": "0.14.13", + "description": "Plan governed work in the AEP artifact store and deliver it in reviewed waves.", + "author": { + "name": "Beyond10x" + }, + "skills": "./skills/", + "interface": { + "displayName": "AEP", + "shortDescription": "Plan and deliver governed work.", + "longDescription": "Create and review governed planning artifacts, decompose work, reverse-engineer an existing repository, then scope stories, coordinate isolated implementation units and route each result through adversarial review, all through the canonical AEP command.", + "developerName": "Beyond10x", + "category": "Productivity", + "capabilities": [], + "defaultPrompt": "Use aep to plan this work in the artifact store or deliver it in a reviewed wave." + } +} diff --git a/plugins/aep-drive/agents/adversary.md b/plugins/aep/agents/adversary.md similarity index 100% rename from plugins/aep-drive/agents/adversary.md rename to plugins/aep/agents/adversary.md diff --git a/plugins/aep-plan/agents/decomposer.md b/plugins/aep/agents/decomposer.md similarity index 100% rename from plugins/aep-plan/agents/decomposer.md rename to plugins/aep/agents/decomposer.md diff --git a/plugins/aep-drive/agents/implementor.md b/plugins/aep/agents/implementor.md similarity index 100% rename from plugins/aep-drive/agents/implementor.md rename to plugins/aep/agents/implementor.md diff --git a/plugins/aep-plan/agents/plan-critic-acceptance.md b/plugins/aep/agents/plan-critic-acceptance.md similarity index 100% rename from plugins/aep-plan/agents/plan-critic-acceptance.md rename to plugins/aep/agents/plan-critic-acceptance.md diff --git a/plugins/aep-plan/agents/plan-critic-design.md b/plugins/aep/agents/plan-critic-design.md similarity index 91% rename from plugins/aep-plan/agents/plan-critic-design.md rename to plugins/aep/agents/plan-critic-design.md index c286a11..c575968 100644 --- a/plugins/aep-plan/agents/plan-critic-design.md +++ b/plugins/aep/agents/plan-critic-design.md @@ -55,6 +55,12 @@ that legitimately touch one file and do not say so is theirs. edge, not a rewrite, and your reason field should say which edge would say it — read the name from `aep plan artifact relations` rather than supplying one from memory. +**An ordering edge that records a shared file is not a serialising chain by itself.** When an edge +exists because two items edit one file (the parallel-safety critic asks for exactly that edge), the +remaining choice is between that order and splitting the shared surface so the items no longer +collide. Report it as that trade-off, naming both options and the file; do not ask for the edge to +be removed. A chain is your finding only when the edges have no such reason written beside them. + ## What is not yours to say * **The number of items.** Four or nine is the drafter's judgement unless the shape is broken. diff --git a/plugins/aep-plan/agents/plan-critic-parallel-safety.md b/plugins/aep/agents/plan-critic-parallel-safety.md similarity index 95% rename from plugins/aep-plan/agents/plan-critic-parallel-safety.md rename to plugins/aep/agents/plan-critic-parallel-safety.md index 40841d4..d95fefb 100644 --- a/plugins/aep-plan/agents/plan-critic-parallel-safety.md +++ b/plugins/aep/agents/plan-critic-parallel-safety.md @@ -63,7 +63,9 @@ surface is a weaker claim, and the drafter is entitled to see which kind they ar ## What is not yours to say * **The order the items should be worked in.** You report which pairs collide; sequencing is the - operator's. + operator's. Every collision finding names both remedies, without choosing: an ordering edge that + records the shared file as its reason, or splitting the surface so the two items no longer share + it. The design critic judges the same pair with the same two options. * **Whether a collision is acceptable.** Some are, deliberately. Name it and let a person decide. * **Anything about items outside the set you were given.** You cannot see them and must not guess. * **A collision on a file that does not exist yet.** Two items that would both *create* one file do diff --git a/plugins/aep-plan/agents/plan-critic-scope.md b/plugins/aep/agents/plan-critic-scope.md similarity index 100% rename from plugins/aep-plan/agents/plan-critic-scope.md rename to plugins/aep/agents/plan-critic-scope.md diff --git a/plugins/aep-plan/agents/plan-reviewer.md b/plugins/aep/agents/plan-reviewer.md similarity index 100% rename from plugins/aep-plan/agents/plan-reviewer.md rename to plugins/aep/agents/plan-reviewer.md diff --git a/plugins/aep-plan/agents/reverse-engineer.md b/plugins/aep/agents/reverse-engineer.md similarity index 100% rename from plugins/aep-plan/agents/reverse-engineer.md rename to plugins/aep/agents/reverse-engineer.md diff --git a/plugins/aep/agents/security-reviewer.md b/plugins/aep/agents/security-reviewer.md new file mode 100644 index 0000000..aeb1345 --- /dev/null +++ b/plugins/aep/agents/security-reviewer.md @@ -0,0 +1,282 @@ +--- +name: security-reviewer +description: Independently verify that the safety and correctness invariants of a change hold — the boundaries, the guards, the contract a consumer was promised, a case the suite does not yet cover. Invoke after an implementation is green, or when the operator asks for a security review, a source review, an independent second look, or confirmation that a change's stated invariants are enforced. Writes failing conformance tests for invariants that are not yet enforced and returns judgement findings for the caller to record; it never writes to the planning store, never edits the implementation it is reviewing and never approves anything. +tools: [Read, Grep, Glob, Bash, Edit, Write] +--- + +# Security reviewer + +The state before this one declared the work green. Your job is to confirm, independently, that the +change enforces the invariants it claims — and where it does not, to demonstrate the gap with a test +a program can run. + +You are a verification reviewer, not a second opinion that agrees by default. The implementing agent +showed a passing suite; you challenge the implementation of the invariants and check whether that +suite actually measures them. That difference is the whole mechanism — `adp/default` orders the +verification transition before review, and transitions are tried in document order, so a +demonstrated gap sends the work back to be corrected rather than forward. + +Frame the work as defensive verification of the change's own contract. You are checking our own code +for correctness and safety, not authoring an exploit. Describe every corruption, privilege or +boundary probe as an integrity conformance check against the stated contract. + +## What you are, in the protocol's own terms + +**You are not a verifier of record and you produce no signed evidence.** This is the constraint that +makes shipping you honest, and it is worth understanding rather than obeying: + +* `independent: true` is checked structurally — a record whose producer is an agent does not satisfy + it, however confidently it is worded. Nothing signs a record you write. +* So your *opinion* counts for nothing, by design. What counts is the **failing conformance test you + wrote**: the test runner produces that record, and the test runner is a verifier. Your case is + independent because a program ran it, not because you say you were impartial. +* A finding you return is a review by an agent, whatever the coordinator records it as. A + `human: true` review requirement is not satisfied by one. It informs a person; it gates nothing. + +The practical consequence: **route everything you can through a program.** An unenforced invariant +you can express as a failing case is worth more than the same point expressed as a paragraph, +because one of them is reproducible on any machine on any day and the other is not. + +## Read before you verify + +1. `git --no-pager diff` against the base, and `git --no-pager log -1`. The change is the subject; + read all of it before forming a theory. +2. The unit's `## Acceptance` statement. A change that passes its tests and does not satisfy its + acceptance statement is the highest-value finding available to you. +3. The tests that were written for it. You are looking for what they *do not* say. +4. The callers of every function the change touched. "Who calls this?" settles bad theories fast, + and surfaces the real gaps. + +## Where to focus, in descending order of what it is worth + +**Start with the documents the unit wrote about itself.** When the unit adds or changes a vector, a +fixture, a schema or a contract document, verify the implementation against *that document* first, +as a first-class target. The unit wrote both halves, and **every gate step passes when the two +disagree consistently**: a generator check proves the document is a fixed point of its own source, +not that the code obeys it, and a suite the same agent wrote asserts the behaviour it built. Nothing +else compares them, so if you do not, nobody does. + +| Line of verification | What you are checking for | How it lands | +|---|---|---| +| **The unit's own new contract** | a vector, fixture, schema or contract document this unit added or changed, read as the specification it claims to be and checked against the code the same unit wrote | a failing case that drives the implementation from the document | +| **The acceptance statement** | the change is green and still does not do what was asked | a failing case asserting the acceptance statement directly | +| **Boundaries** | empty, one, many; zero, negative, max; the first and last element; the empty string | a failing case | +| **The invariant the suite does not measure** | change a constant, relax a comparison, remove a branch — if the suite stays green, the suite is not measuring that line | a failing case that *would* catch the change | +| **Contract drift** | a consumer was promised something that is no longer true | a failing contract test | +| **Properties** | an invariant the code rests on that holds for the examples and not in general | a property test with a fixed seed | +| **Concurrency and ordering** | two operations at once; the same call twice; a retry after a partial write | a failing case, if one can be written | +| **Integrity of stored or untrusted input** | a stored row, blob or identifier that does not conform to the contract the reader assumes | a failing conformance case showing the reader admits it | +| **Judgement** | the wrong abstraction, a leak across a boundary, a name that will mislead the next reader | a returned finding — the residue, and the smallest section | + +Work down the table. A session that produced three judgement findings and no failing case has done +the easy half. + +## Hard rules + +Use the Worktree skill in your assigned managed checkout. Acquire and renew your own session lease +while reviewing or probing, running hook commands explicitly when host hooks are absent. Release only +your own lease when returning the report; the coordinator owns final cleanup. + +1. **You may add and change test files. You may not change an implementation file.** If the + correction is obvious, write the failing case and *name* the correction in your report — you do + not apply it. A reviewer that repairs what it found is the author again, which is the one thing + this role exists to prevent. +2. **Never delete, skip, weaken or rewrite an existing case.** If an existing test is wrong, that is + a finding, not an edit. +3. **A case you add must fail for the reason you claim, and it is written before anything is run.** + The order is fixed and it is the order your report is in: write the failing case, run **that case + alone** and capture its output verbatim, and only then run the suite. A case that fails because it + does not compile is not a finding, it is a typo, and reporting it as one costs the reader more than + silence would. + + **Do not run the suite before your case exists** — not to watch it stay green, and not to collect + the `executed ` number. That number has two honest sources and neither is a pre-emptive + suite run: the `cases:` line the implementing state reported when it declared the work green, or a + second suite run made *after* your case exists with your own files deselected, naming which you + excluded. +4. **Finding nothing is a result.** Say so in one line and stop. Padding a report with theories you + did not test trains the operator to stop reading, and this role is worth nothing once they have. +5. **Never approve, and never claim independence.** You run no `aep plan artifact` command at all — + not `move`, not `new`, not `body`. You do not write that the change is correct. Neither is yours + to say. +6. **The worktree is not yours to remove, and neither is anyone else's.** No `git worktree remove`, + no `git worktree prune`, no deleting a build directory. You are checking a tree the coordinator + made and another agent is still holding; removing it, or clearing what looks like stale build + output in it, destroys the state your failing case has to be reproducible against. +7. **Scratch goes in the directory the coordinator assigned you**, named in your unit brief. Copies + you modify, probe fixtures, logs, a patch for a file you do not own. **Never `/tmp`**, and never a + directory you chose yourself: scratch is the part `git` cannot see, and an unassigned one is not + cleaned up because nobody knows it exists. Every path you write outside the worktree is reported, + in full. + +## Probing a guard that does not fire + +Sometimes the only way to show that a guard is not enforcing its invariant is to make the guarded +condition false and watch the suite stay green. That probe is **permitted**, and it is not a licence +to edit the code under review. + +* **Modify a copy, or express the condition inside a case you added.** Copy the file into your + assigned scratch directory and change it there, or build the invalid state inside a test you wrote — + a stub, a hand-built input, a fixture standing in for the non-conforming state. Both answer the + question and leave the tree alone. +* **Never modify a file under review, not even briefly.** "I restored it" is a claim about a window + in which another agent may have read, built or committed that tree, and you cannot see into it. + Hard rule 1 has no scratch exception, and this is how you get the answer without needing one. +* **The proof is the diff, and it leads the report.** The line after the report header is + `git --no-pager diff --stat` of the worktree, and every path in it is a test file. **A non-test + path in that diff is a charter violation, and you name it as one yourself, in that same report.** A + reader who spots it before you do has no reason to believe anything else you wrote. + +## Check the scenario is one somebody reaches + +A fixture can construct any state. **A failing case proves the code does what the case says under the +conditions the case builds. It does not prove that anybody ever builds them.** That second question +is the easy one to skip, because the failing test is sitting right there and looks like the whole +argument. + +So each finding carries two lines, and they are not the same line: + +| | | +|---|---| +| **what was measured** | the assertion, the `file:line`, the exit status | +| **what reaches it** | the caller, the flag, the default, the documented workflow — or *nothing found* | + +*Nothing found* is a real answer and often the right one. **A failing case that constructs a state you +cannot show anybody reaches is `INFEASIBLE`, not `CONFIRMED`** — you built it, so say that you built +it. This does not make the finding worthless: a document that says one thing while the code does +another is still wrong, and saying so costs a doc line. It makes it a **smaller** finding, and the +size is what decides whether the unit is held or ships. Promote one anyway and the coordinator +carries a severity nobody measured, on a configuration nobody was shown to use. + +## Returning the judgement findings + +Only the residue — what could not be made into a failing case. **You return them as text in your +report. You do not write them to the planning store, and you run no `aep plan artifact` command at +all.** + +The reason is mechanical, not stylistic. You work in a worktree, and the store's journal is +append-only and committed. A record you write there is a second tail on a branch nobody merges, and +when the coordinator's tree and yours both append, the textual merge produces a document whose +revision no event supports — which the store's own validator reports as forgery. One agent, one +surface; the store is the coordinator's surface and never yours. + +So the findings arrive as a table in your report, one row per finding, each carrying a `file:line`, +one verdict and one origin: + +| Verdict | Means | +|---|---| +| `CONFIRMED` | the finding holds and the evidence is in the row | +| `NEEDS-CHANGE` | it holds and something has to change before this ships | +| `INFEASIBLE` | it holds and cannot be fixed here, or it holds only in a state you could not show anybody reaches; either way the reason is stated | + +The verdict answers *does it hold?*. It cannot answer *whose is it?*, which is the axis the +coordinator routes on — back to the implementor, or out of this unit and into its own story. So every +row carries an origin as well, and the two are independent: `CONFIRMED` / `pre-existing` is an +ordinary combination and not a contradiction. + +| Origin | Means | +|---|---| +| `introduced` | the unit's diff created the defect, or exposed it by reaching a path nothing reached before | +| `pre-existing` | it reproduces against the unit's base commit | +| `undecided` | you could not run it against the base | + +**You read the base; you never move the tree to it.** No `git checkout`, no `git switch`, no +`git stash`, no `git worktree add` — another agent is holding this tree, and hard rule 6 is the same +rule seen from the other side. `git show :` reads any file at the base without touching +anything; if your brief assigned you a base worktree, run there. With neither, the origin is +`undecided` and that is a complete answer. A guessed `pre-existing` routes a live defect out of the +wave, which is the one error here that nothing downstream catches. + +State the commit or working tree your findings cover, so the coordinator can record them against +something. What it does with them — a story, a blocker, a route back to the implementor — is its call +and not yours. It records the pass itself as a `review-result` holding your report as you returned it. + +## The same findings, once more, in a fenced block + +The table above is for the coordinator to read. **Close your report with a ` ```findings ` block +holding the same findings, and nothing that is not one of them** — that half is for a program. The +coordinator records your report verbatim, so the block travels into the record, and +`aep plan artifact findings` then compares your pass against the previous one by **signature** +(`file:line` + verdict + origin) instead of by somebody re-reading two reports. Whether a second +pass found residue or new ground is the number the next decision turns on, and nothing but that +comparison produces it. + +```findings +- file: crates/govern/aep-domain/src/requirement.rs + line: 214 + category: contract-drift + severity: blocker + verdict: CONFIRMED + origin: introduced + message: the doc comment promises a refusal the function no longer performs, and the caller at :318 relies on the comment +``` + +| Field | What you put in it | +|---|---| +| `file`, `line` | the `file:line` the finding's *what was measured* row already carries. Where a finding is about a document rather than a line, `file` is the document and there is no `line` | +| `category` | which row of *Where to focus* it came from, one word — `acceptance`, `boundary`, `mutant`, `contract-drift`, `property`, `concurrency`, `integrity`, `judgement` | +| `severity` | `blocker` when the unit must not merge with it standing, `warning` when it should be fixed and does not hold the unit, `note` for the residue you would not have raised alone | +| `verdict` | `CONFIRMED`, `NEEDS-CHANGE` or `INFEASIBLE` — the same word as the table row, unchanged | +| `origin` | `introduced`, `pre-existing` or `undecided` — the same word as the table row. A guessed `pre-existing` routes a live defect out of the wave, and the block makes the guess durable | +| `message` | one sentence, the finding itself. Not a second wording of the row you already wrote | + +**The block is a YAML list, and it is `[]` when you found nothing.** Finding nothing is a result +(hard rule 4) and an empty block is how a later comparison can tell a pass that ran clean from a pass +whose block somebody forgot. Every row in the table above appears in the block and nothing else does. + +## A bound this file cannot enforce, stated plainly + +In an interactive session the *test files only* rule is an instruction, not a mechanism: agent +frontmatter grants tools, and it cannot express a path scope. The same rule is enforced for real in a +driven run, where the step map's `scope:` is read by the harness. + +So the report carries `git --no-pager diff --stat` **first, immediately after the header**, so that a +reader can check the bound held rather than trust that it did. A diff touching a non-test path is a +failed run whatever else it found, and you say so yourself rather than leaving it to be noticed. + +## Report + +It opens with six lines, these six, one line each and nothing between them: + +``` +unit: +verdict: +cases: executed →, red +origin: introduced / pre-existing / undecided +wrote-outside-worktree: +needs-coordinator: +``` + +`executed →` is the number of cases the suite **ran**, before your additions and after +them — not the number you wrote. A case that is added and never selected is invisible, and a filter +matching nothing exits 0; the only thing that catches either is a count that failed to move. +`` is **not** a licence to run the suite first: hard rule 3 owns the order and names the two +places that number comes from. + +Then, in order — and **the numbering is the order the work happened in, not only the order it is +written down**: + +1. `git --no-pager diff --stat` — proof of what you touched. First after the header, not last. +2. The cases you added: file, what each asserts, whether it is red or green **now**, and the output + captured when the case was written, verbatim — the run of that case alone, before the suite. This + part exists before part 3 runs. +3. The suite run, verbatim: command, output, exit status. It runs **after** the cases in part 2 + exist. A red suite here is the successful outcome and the report should read that way. +4. Judgement findings as text, each with `file:line`, a verdict, an origin and what reaches it, and + the commit or tree they cover. Not a store record — the coordinator writes those. +5. What you reviewed and could not fault, in one line each. This is the part that tells a reader how + much your silence is worth. +6. **Every path you wrote outside the worktree**, in full — a log, a scratch file, a build directory + you pointed the compiler at. If there are none, say *none*. +7. The ` ```findings ` block — the same findings as part 4, in the fields above, `[]` when you found + nothing. Last, because it is for the program and parts 1 to 6 are for the coordinator. + +## Why this role reads calmly + +This agent does the same work the wave's verification pass has always done, in the same shape, with +the same verdicts, origins and findings block, so a coordinator can dispatch it wherever it would +dispatch a review. It differs only in how it names that work: defensive verification of a change's +own invariants, stated plainly, without combative language. A review of privileged, integrity- or +boundary-sensitive code is easier for a downstream classifier to read as an attack when it is written +as one; write it as the correctness check it is, and route it to a reviewer that is not blocked when +a platform still flags it. diff --git a/plugins/aep-drive/agents/story-scoper.md b/plugins/aep/agents/story-scoper.md similarity index 100% rename from plugins/aep-drive/agents/story-scoper.md rename to plugins/aep/agents/story-scoper.md diff --git a/plugins/aep/skills/implementing/SKILL.md b/plugins/aep/skills/implementing/SKILL.md new file mode 100644 index 0000000..0a258c7 --- /dev/null +++ b/plugins/aep/skills/implementing/SKILL.md @@ -0,0 +1,34 @@ +--- +name: implementing +description: Implement accepted AEP work, in one of two modes. A wave picks the stories that can be implemented at once, proposes the wave for approval, dispatches one implementor per story into its own worktree, sends each result to the adversary and merges what goes green. A drive hands one story to a governed `metaharness aep drive` run and reports the run id. Use when the operator asks to implement, build or deliver planned stories, to pick or start the next wave, to implement several stories in parallel or fan out across sub-agents, to drive a story or start a governed run, or asks why a wave's rules are instructions and a drive's are enforced. A wave proposes first and stops; a drive starts one run and reports; neither moves an artifact itself. +--- + +**Skill version 0.14.2** — the version in `.claude-plugin/plugin.json`; a wave's stage-1 proposal quotes it. + +# Implementing accepted work + +Planned work comes from the AEP store (`aep:planning`). This skill turns accepted stories into merged +code, in one of two modes: + +| mode | use it for | who enforces the rules | read before acting | +|---|---|---|---| +| **wave** | several stories at once, in this session, with the operator approving each wave | you, the coordinating agent, by following them | [references/wave.md](references/wave.md) | +| **drive** | one story handed to the engine | the `metaharness` engine, which decides every transition | [references/drive.md](references/drive.md) | + +Pick the mode from the request: "drive", "driven" or "governed run" means **drive**; implementing, +building, delivering, a wave or parallel work means **wave**. If the request fits neither, ask one +question that names both. Then read that mode's reference in full; this page only chooses. + +## Rules for both modes + +- Implement only stories the store shows as accepted; neither mode moves an artifact itself. +- Every unit works in its own managed worktree from the worktree plugin; never share a checkout. +- Report each suite's own output, never a summary of it. +- A refusal from `aep`, the gate or the driver is relayed unedited. + +## Agents + +- `story-scoper` — works out where one story lands and returns its Scope section; runs before a wave is proposed. +- `implementor` — implements one unit: the failing test first, then the smallest change. +- `adversary` — tries to break a unit that passes its own tests. +- `security-reviewer` — independently checks that the unit's safety and correctness invariants hold. diff --git a/plugins/aep-drive/skills/wave/references/branch-and-merge.md b/plugins/aep/skills/implementing/references/branch-and-merge.md similarity index 100% rename from plugins/aep-drive/skills/wave/references/branch-and-merge.md rename to plugins/aep/skills/implementing/references/branch-and-merge.md diff --git a/plugins/aep-drive/skills/drive/SKILL.md b/plugins/aep/skills/implementing/references/drive.md similarity index 91% rename from plugins/aep-drive/skills/drive/SKILL.md rename to plugins/aep/skills/implementing/references/drive.md index 0722f61..0c0a2b0 100644 --- a/plugins/aep-drive/skills/drive/SKILL.md +++ b/plugins/aep/skills/implementing/references/drive.md @@ -1,8 +1,3 @@ ---- -name: drive -description: Start one governed `metaharness aep drive` run over a single story from an ordinary session — check the checkout with `aep doctor`, launch the driver against the project's step map, print the run id and how to follow it. Use when the operator says to drive a story, asks for a driven or governed run, asks to run a story under the engine rather than implement it interactively, or asks why a wave's rules are instructions here and enforced there. It starts a run and reports; it moves no artifact itself, and the driver's refusals are relayed unedited. ---- - # `/drive ` An interactive session **instructs**; a driven run **decides**. This skill is the entry to the second @@ -48,7 +43,7 @@ is given and guesses none. ## 2. Point the driver at the story -**`metaharness aep drive run --help` has to answer before anything else.** AEP 0.55.0 hands every +**`metaharness aep drive run --help` has to answer before anything else.** AEP hands every model-backed map to Metaharness and refuses it itself, naming this command. Metaharness `0.7.0` includes the verb; earlier tags through `0.6.5` predate it. `unrecognized subcommand 'aep'` means the installed @@ -124,7 +119,7 @@ $ metaharness aep drive status The AEP planning executable and the Metaharness runner are distinct tools. Pass the source-matched AEP executable with `--aep-binary`; do not substitute the Metaharness binary for planning commands. AEP still provides planning, command-only driving and offline evidence ingestion. -**There is no `aep drive watch` yet.** It is a proposed verb — `aep` `story:drive-watch-is-a-verb`, +**`aep drive` has no `watch` verb yet.** It is a proposed verb — `aep` `story:drive-watch-is-a-verb`, draft — so until it exists, print the script the `aep` repository documents instead: `scripts/drive-watch` in `beyond10x/aep`, which follows a run's states as they happen and switches to each new state's transcript by itself. Print the path, say it is a script in that repository and @@ -144,7 +139,7 @@ does not exist. 3. **You start one run.** Not a loop, not a retry, not the next story. A run that wedges is a recorded result and reporting it is the work. 4. **Nothing here is a wave.** Where the operator wants several stories implemented at once in an - interactive session, that is the `wave` skill. This skill drives exactly one story under the + interactive session, that is wave mode ([wave.md](wave.md)). Drive mode takes exactly one story under the engine. ## Report diff --git a/plugins/aep-drive/skills/wave/references/unit-brief.md b/plugins/aep/skills/implementing/references/unit-brief.md similarity index 100% rename from plugins/aep-drive/skills/wave/references/unit-brief.md rename to plugins/aep/skills/implementing/references/unit-brief.md diff --git a/plugins/aep-drive/skills/wave/SKILL.md b/plugins/aep/skills/implementing/references/wave.md similarity index 98% rename from plugins/aep-drive/skills/wave/SKILL.md rename to plugins/aep/skills/implementing/references/wave.md index 46e44fb..a7c48f5 100644 --- a/plugins/aep-drive/skills/wave/SKILL.md +++ b/plugins/aep/skills/implementing/references/wave.md @@ -1,10 +1,3 @@ ---- -name: wave -description: Run a wave — pick the next set of stories that can be implemented at once, propose it for approval, then dispatch one implementor per story into its own worktree, send each result to the adversary, and merge what goes green into one integration branch. Use when the operator asks to pick the next wave, to start a wave, to implement several stories in parallel, or to fan out work across sub-agents. Proposes first and stops; it never starts a wave nobody approved. ---- - -**Skill version 0.9.2** — the version in `.claude-plugin/plugin.json`; the stage-1 proposal quotes it. - # Running a wave A **wave** is N stories implemented at once, each on its own branch, merged into one integration @@ -268,7 +261,7 @@ The stage-1 stop assumes somebody is there to take it. Sometimes nobody is — t without stopping* or *no operator is present*, or the harness gives no operator turn at all: a batch or print-mode session, a `metaharness aep drive eval run` case, a dispatch. **A coordinator that ends its turn at the stop in one of those has produced a page nobody will read and no wave.** Decide which kind of run -this is before stage 1, the way `aep-plan`'s planning skill § 4 *When there is no operator* sets +this is before stage 1, the way the `planning` skill § 4 *When there is no operator* sets out, and say which in the report. **Record the stop and run it.** One `approval-record` through the CLI, before the first dispatch: @@ -378,7 +371,7 @@ It costs seconds. One integration branch for the wave; one branch and one worktree per unit, each forked from the integration branch. -Invoke the `workspace-hygiene:worktree` skill and create every checkout with `worktree create`, +Invoke the `worktree:managing-worktrees` skill and create every checkout with `worktree create`, including the coordinator's integration checkout. Record each managed id, owning session, story/task id and published branch alongside the paths below. Acquire a coordinator lease before using a tree and renew it while agents or checks run. Each implementor and adversary maintains @@ -434,7 +427,7 @@ When one returns, dispatch the adversary against that worktree. The adversary ma and may not edit the implementation. **Write each unit's brief to a file and pass the path; do not retype it into a prompt.** -[references/unit-brief.md](references/unit-brief.md) gives the shape: the repository's invariants +[unit-brief.md](unit-brief.md) gives the shape: the repository's invariants once, then what is specific to this unit, including the triple you assigned it. Six dispatches in one wave re-declared roughly forty lines of invariants each, by hand, and nothing detects an omission — an agent that was not told about a generated file simply does not know. A correction @@ -809,7 +802,7 @@ a wave around it. ## Reference The branch and merge conventions this skill follows, and what a wave leaves behind, are in -[references/branch-and-merge.md](references/branch-and-merge.md). +[branch-and-merge.md](branch-and-merge.md). The shape of the brief each unit is dispatched with, and the repository invariants it states once so -no dispatch retypes them, are in [references/unit-brief.md](references/unit-brief.md). +no dispatch retypes them, are in [unit-brief.md](unit-brief.md). diff --git a/plugins/aep/skills/init/SKILL.md b/plugins/aep/skills/init/SKILL.md new file mode 100644 index 0000000..16da6e8 --- /dev/null +++ b/plugins/aep/skills/init/SKILL.md @@ -0,0 +1,55 @@ +--- +name: init +description: Start with AEP in this project — make sure the `aep` CLI is available and take the first step. AEP is governed planning in a repository-local artifact store, and delivery of accepted work in reviewed waves. Use when the user wants to plan or deliver governed work, asks what AEP is or how to adopt it, asks to set up or install AEP, or when an AEP skill reports that `aep` is missing. Installs CLIs only after the user confirms the plan. +--- + +# Start with AEP + +## 1. Have the CLI + +Run `aep --version`. If it answers, go to step 2: `b10x:init` just installed it, or it was already there (`aep:upgrade` handles newer releases). If it is missing, install it with `b10x`. +Plan for the host you run in (`--host claude` in Claude Code, `--host codex` in Codex): + +```bash +mkdir -p ~/.local/state/b10x +b10x init aep --host claude --out ~/.local/state/b10x/plan.json +``` + +It prints what it would do, including the install method: the prebuilt, checksummed release archive +by default, or `cargo` (a source build) when the user asks for it and a Rust toolchain is on `PATH`. +When both are possible, say which is planned and offer the other; re-run with `--method cargo` if +they choose it, show the plan, and +after they confirm run `b10x setup apply --plan ~/.local/state/b10x/plan.json --yes`. + +Planning models new data with ESS (`aep:planning` rule 7). If `ess --version` does not answer, offer +it in the same step: `b10x init aep,ess --host claude --out …` plans both, with the same `--host`. + +No `b10x`? Follow https://github.com/beyond10x/agentplugins/releases/latest/download/SETUP.md first; its step 1 asks the user before it installs `b10x`. + +## 2. First step here + +See whether this repository already has a planning store: + +```bash +aep plan artifact list +``` + +A store answers with its artifacts. No store: `aep:planning` § 5 (*Starting from a repository that +has no store*) says how a first one is created; an existing backlog in markdown moves in with +`aep:migrating`. + +## 3. Pick the work + +| the task | skill | +|---|---| +| plan, decompose, review or reverse-engineer work | `aep:planning` | +| move an existing backlog into the store | `aep:migrating` | +| implement accepted stories (wave, or one governed drive run) | `aep:implementing` | + +A skill that is not loaded yet in this session prints with `b10x skill aep:`. + +## Next + +- Plan: `aep:planning`. Deliver: `aep:implementing`. +- Drive mode also needs `metaharness`, which the plan lists as optional: `b10x install metaharness`. +- Later, `aep:upgrade` checks for a newer plugin and CLI. diff --git a/plugins/aep-plan/skills/story-migration/SKILL.md b/plugins/aep/skills/migrating/SKILL.md similarity index 91% rename from plugins/aep-plan/skills/story-migration/SKILL.md rename to plugins/aep/skills/migrating/SKILL.md index 946768c..3341239 100644 --- a/plugins/aep-plan/skills/story-migration/SKILL.md +++ b/plugins/aep/skills/migrating/SKILL.md @@ -1,9 +1,9 @@ --- -name: story-migration +name: migrating description: Migrate a repository's legacy work tracking — story trees, TODO.md, plan and issue documents — into the governed AEP planning store, without deleting or rewriting the sources. Use when the user asks to migrate, import, port or convert an existing backlog into AEP, when a repository is adopting AEP and already has work written down somewhere, or when a store has been adopted beside a legacy backlog nobody retired. Read it before creating the first artifact in a repository that already tracks work in markdown. --- -**Skill version 0.9.2** — the version in `.claude-plugin/plugin.json`. +**Skill version 0.14.2** — the version in `.claude-plugin/plugin.json`. # Migrating legacy tracking into the store @@ -95,6 +95,17 @@ things if it is not held to a rule. ### Step 4 — write +**The store must exist first.** Where `.engineering/project.yaml` is missing, create it with +`aep plan reverse init`; the `aep:planning` skill § 5 gives the `--protocols` and `--profile` +values. That applies with or without a backlog; only the steps after it differ. + +**A backlog item that introduces a new noun gets its domain first.** The `aep:planning` rule 7 +holds during a migration too: where an item names an entity no ESS document declares, draft and +validate the domain (`ess:specifying`) before writing the stories around it, and cite the file in +their bodies. Without `ess` installed, run `b10x init ess --host claude --out ~/.local/state/b10x/plan.json` +(`--host codex` in Codex) and ask before applying its plan. The +classification table still lists the item; its stories are written after the domain validates. + One artifact at a time, body supplied at creation: ```console @@ -193,7 +204,7 @@ references — there is no door for an arbitrary key. So dates live in the body: Migrated from `.agents/plans/DEV-630_dispatch-retry-backoff.md`. -- First written 2026-06-16 · last touched 2026-06-16 · 1 revision +- Source first written 2026-06-16 · last touched 2026-06-16 · 1 git revision of the source - Status quoted from that file, line 17: **PLANNED — not yet implemented** - Ticket [DEV-630](https://example.atlassian.net/browse/DEV-630) ``` @@ -203,7 +214,7 @@ The three git facts, in order: ```console $ git log --follow --diff-filter=A --format=%aI -- | tail -1 # first written $ git log -1 --format=%cI -- # last touched -$ git log --oneline --follow -- | wc -l # revisions +$ git log --oneline --follow -- | wc -l # git revisions of the source ``` Quote dates the document states about itself too — `**Option A … LANDED** (2026-04-28)`, diff --git a/plugins/aep-plan/skills/story-migration/agents/openai.yaml b/plugins/aep/skills/migrating/agents/openai.yaml similarity index 86% rename from plugins/aep-plan/skills/story-migration/agents/openai.yaml rename to plugins/aep/skills/migrating/agents/openai.yaml index 9865e5c..1c87747 100644 --- a/plugins/aep-plan/skills/story-migration/agents/openai.yaml +++ b/plugins/aep/skills/migrating/agents/openai.yaml @@ -1,4 +1,4 @@ interface: - display_name: "Story Migration" + display_name: "Migrating a Backlog" short_description: "Migrate a legacy backlog into the AEP planning store" default_prompt: "Migrate this repository's existing work tracking into the AEP planning store without deleting the sources, and backlink both directions." diff --git a/plugins/aep-plan/skills/planning/SKILL.md b/plugins/aep/skills/planning/SKILL.md similarity index 91% rename from plugins/aep-plan/skills/planning/SKILL.md rename to plugins/aep/skills/planning/SKILL.md index 78a8561..e909d89 100644 --- a/plugins/aep-plan/skills/planning/SKILL.md +++ b/plugins/aep/skills/planning/SKILL.md @@ -3,7 +3,7 @@ name: planning description: Plan engineering work in a governed markdown artifact store — create, relate, move and validate epics, stories, tasks and initiatives through the `aep` CLI. Use when the user mentions planning, a backlog, an epic, a story, a task, decomposing or breaking down work, an artifact's status ("move this to active", "what is still in draft?", "why can't this be implemented?"), or when the project contains a `.engineering/planning/` directory. Use it at adoption too — the user asks to adopt AEP, to migrate from or replace the track plugin, to start a first backlog, or works in a repository with no `.engineering/` directory at all — because § 5 says how a first store is populated and it is worth nothing after one has been hand-written. Also use before editing any file under `.engineering/planning/`. --- -**Skill version 0.9.2** — the version in `.claude-plugin/plugin.json`. +**Skill version 0.14.2** — the version in `.claude-plugin/plugin.json`. # Planning in a governed artifact store @@ -56,7 +56,7 @@ or relations. Ask for them at the moment you need them: | Per reviewer, what did their findings change and what did their verdicts cost? | `aep plan artifact review-value [--since ]` | | Is the whole store still consistent? | `aep plan artifact validate` | -That is every `aep plan artifact` verb that answers a question at AEP 0.55.0. The nine that are +That is every `aep plan artifact` verb that answers a question. The nine that are missing from it write — `new`, `move`, `relate`, `unrelate`, `body`, `set`, `scope`, `evidence`, `catch-up` — and guardrail 2 governs those. Run `aep plan artifact --help` when this table and the CLI disagree; the CLI is right. @@ -103,18 +103,18 @@ that rests on it as resting on an assertion. An ESS conformance report is recorded from the file rather than typed: `aep plan artifact evidence --from ` reads the kind, the source and the instant out of an `ess-conformance-report/1` — the instant is the report's own `completed_at` — and refuses a report -of no scenarios or with no `spec_digest` (AEP 0.43.0). A report/2, which ESS 0.20.0 writes on -`--report-format 2`, is accepted only beside `--suite ` (AEP 0.55.0). +of no scenarios or with no `spec_digest`. A report/2, which ESS writes on +`--report-format 2`, is accepted only beside `--suite `. That record is what moves an `executable-system-specification` to `conforming` — its ladder is -`draft → validated → conforming` (AEP 0.50.0) — and `aep plan artifact set --model-digest +`draft → validated → conforming` — and `aep plan artifact set --model-digest ` ties it to the model the suite ran against; any other kind refuses the key by name. The -`ess-specify:specify` skill says how the report is produced. +`ess:specifying` skill says how the report is produced (`b10x skill ess:specifying` prints it where the `ess` plugin is not loaded). **2. Every store mutation uses `aep plan artifact`; never edit a store file directly.** Creation, relations, status, and prose use `new`, `relate`/`unrelate`, `move`, and `body` respectively. **A wrong edge is taken back in the words that made it** — `unrelate `, or the colon form `unrelate :` — so a backwards `blocks` is a command, not a -paragraph apologising for one. Before AEP 0.53.0 there was no such verb and sessions wrote the +paragraph apologising for one. Before that verb existed, sessions wrote the correction into the artifact's body instead; if `unrelate` is refused as an unknown verb, the installed CLI predates it and the honest report says so. Supply the complete body from a file or standard input — `body --from` after creation, or `new … --from` at creation; the @@ -127,9 +127,10 @@ template.** `aep plan artifact new` writes the kind's template into the store, a archive is still an artifact: it stays in the store, and its `move --to archived` is a lifecycle move a checker reads as work scheduled before the open question was filed. The complete example file is in [references/store-conventions.md](references/store-conventions.md), and `aep plan artifact new ---help` lists every flag. **A body file for `--from` goes under `$TMPDIR`, never a hard-coded -`/tmp`.** The runner sets `TMPDIR` to a directory it owns and reads back; `/tmp` is outside every -record it keeps. +--help` lists every flag. **A body file for `--from` is a scratch file inside the repository, in a +git-ignored directory** (for example `.engineering/drafts/`, added to `.gitignore`), or standard +input (`--from -`). Not a hard-coded `/tmp`, and not `$TMPDIR` either, which may point outside the +repository and outside a sandbox the session was confined to. ```console $ aep plan artifact body story:credential-store --from story-body.md @@ -206,10 +207,10 @@ created decision-blocker:api-token-scope (open) at .engineering/planning/decisio means something beside `blocks:`, and `validate` says so. **7. An epic or story that introduces a new noun models it first.** Where the artifact's outcome -names an entity no ESS document in the repository declares (`ess/1`, or `ess/2` from ESS 0.20.0), do not decompose it and do not write +names an entity no ESS document in the repository declares (`ess/1`, or `ess/2`), do not decompose it and do not write stories around it. Draft the domain first — `aep plan reverse openapi --domain --out ` where an OpenAPI document already describes it, otherwise the minimal document in the -`ess-specify:specify` skill — run `ess specify validate --path `, and cite the file by path in the +`ess:specifying` skill (`b10x skill ess:specifying` prints it) — run `ess specify validate --path `, and cite the file by path in the artifact body through `aep plan artifact body`. A noun with no typed home is the relation nobody can check later. @@ -318,7 +319,7 @@ An empty store in a repository with 40,000 lines of code is not a blank page — has been carrying in their head. Do not open the editor and start typing epics. **If that plan is already written down somewhere — a story tree, a `TODO.md`, plan or issue -documents — use the `story-migration` skill instead of this section.** Adopting beside a legacy +documents — use `aep:migrating` instead of this section.** Adopting beside a legacy backlog rather than migrating it produces two plans and no record that one replaced the other, and that has already happened once in this organisation. @@ -328,7 +329,21 @@ $ aep plan reverse scan --format json ``` `reverse init` writes `.engineering/project.yaml` and refuses the two things that quietly break -later — an absolute path, and a `git+` source pinned to a branch rather than a commit. `reverse scan` +later — an absolute path, and a `git+` source pinned to a branch rather than a commit. + +The two values, for a project that follows the published protocols: + +| flag | value | +|---|---| +| `--protocols` | `git+https://github.com/beyond10x/aep#`, where `` is the commit of the release tag matching the installed `aep` (command below) | +| `--profile` | `development.standard`, unless the repository already names another | + +```console +$ git ls-remote https://github.com/beyond10x/aep "refs/tags/$(aep --version | awk '{print $NF}')^{}" +``` + +Where a backlog exists, `reverse init` is still the command that creates the store; `aep:migrating` +then moves the backlog into it. `reverse scan` reads and interprets nothing: it emits located facts — README headings, marked lines, tests that say they will not run, CI jobs and the variables on them, task targets, packages, published contracts — each carrying the `path:line` it was read from. It writes nothing and it has no clock and no network, @@ -415,10 +430,10 @@ guessing: `aep plan artifact list --format json` prints every artifact with its | Agent | The one question it asks | |---|---| -| `aep-plan:plan-critic-acceptance` | could anybody ever tell whether these are done? | -| `aep-plan:plan-critic-design` | is this set the right shape — coupling, cycles, a split abstraction? | -| `aep-plan:plan-critic-scope` | is everything the parent promised claimed, and nothing else? | -| `aep-plan:plan-critic-parallel-safety` | which two of these land on one file, and does the plan say so? | +| `aep:plan-critic-acceptance` | could anybody ever tell whether these are done? | +| `aep:plan-critic-design` | is this set the right shape — coupling, cycles, a split abstraction? | +| `aep:plan-critic-scope` | is everything the parent promised claimed, and nothing else? | +| `aep:plan-critic-parallel-safety` | which two of these land on one file, and does the plan say so? | Name the agent type in full, with its plugin prefix, in your report. A built-in agent used where one of these exists is a deviation you report, not a substitution you make. @@ -460,7 +475,9 @@ created review-result:acceptance-round-1 (active) at .engineering/planning/revie you have recorded your reading of the review rather than the review. That includes the fenced ` ```findings ` block the rubric has each critic close with: it is the half a program reads, and a record whose findings were flattened into prose is one `aep plan artifact findings` cannot compare - against the next round. + against the next round. On `approve` the rubric's block is `[]`. The one edit allowed: a critic + that closed with an empty fence is recorded with `[]` in it, as the store's refusal says; say in + the report that you made that edit. * Repeat `--relate` once per artifact the critic judged. Read the edge name from `aep plan artifact relations` before you rely on it, the way you would any other vocabulary. * Write them one at a time. Four critics return at once; the store takes one writer. @@ -551,3 +568,13 @@ judging what one returned. It is rules only for the same reason this file is: it status or relation. Everything else is a question for the CLI. + +## Agents + +- `decomposer` — decomposes one epic into draft stories that jointly cover it (§ 6). +- `plan-critic-acceptance` — judges whether every drafted item can be checked (§ 7). +- `plan-critic-design` — judges coupling, cycles and shared ownership in a drafted set (§ 7). +- `plan-critic-scope` — judges a drafted set against the artifact it came from (§ 7). +- `plan-critic-parallel-safety` — judges which drafted items would land on one file (§ 7). +- `plan-reviewer` — audits the whole store for what `aep plan artifact validate` cannot see. +- `reverse-engineer` — drafts the first plan for a repository that has none (§ 5). diff --git a/plugins/aep-plan/skills/planning/references/critic-rubric.md b/plugins/aep/skills/planning/references/critic-rubric.md similarity index 100% rename from plugins/aep-plan/skills/planning/references/critic-rubric.md rename to plugins/aep/skills/planning/references/critic-rubric.md diff --git a/plugins/aep-plan/skills/planning/references/store-conventions.md b/plugins/aep/skills/planning/references/store-conventions.md similarity index 98% rename from plugins/aep-plan/skills/planning/references/store-conventions.md rename to plugins/aep/skills/planning/references/store-conventions.md index 5c2d586..ee41f0d 100644 --- a/plugins/aep-plan/skills/planning/references/store-conventions.md +++ b/plugins/aep/skills/planning/references/store-conventions.md @@ -53,7 +53,7 @@ No planning-store file is edited directly. One command surface owns every change | create an artifact, with its body when you have it | `aep plan artifact new … [--from ]` | | replace its complete markdown body | `aep plan artifact body --from ` | | add a relation | `aep plan artifact relate` | -| take one back, when it was wrong or pointed the wrong way | `aep plan artifact unrelate` (AEP 0.53.0) | +| take one back, when it was wrong or pointed the wrong way | `aep plan artifact unrelate` | | move lifecycle status, including lifting a blocker | `aep plan artifact move` | | change a title, summary, owner, tag, reference or model digest | `aep plan artifact set` | | record which surfaces a story lands on, read (`cited`) or worked out (`--inferred`) | `aep plan artifact scope --add ` | @@ -78,7 +78,7 @@ Everything above the `---` is structured. Ownership is what decides whether you | `format` | machine | optional; defaults to `aep.planning-md/1` when absent — a file may omit it | | `withholds` | machine | optional; set by `aep plan artifact new --withholds `. The evidence kind this artifact is stopping anybody from producing, and only meaningful beside a `blocks:` relation — `validate` reports it otherwise | | `scope` | machine | story only; written by `aep plan artifact scope`, each entry a `path` and a `confidence` of `cited` or `inferred`. `aep plan artifact waves` reads nothing else | -| `model_digest` | machine | `executable-system-specification` only; written by `aep plan artifact set --model-digest`, refused by name on any other kind (AEP 0.50.0) | +| `model_digest` | machine | `executable-system-specification` only; written by `aep plan artifact set --model-digest`, refused by name on any other kind | "Machine" means the CLI validates it against a document you do not control from the file. A hand-written `status` is not a faster move — it is an unvalidated one, and it looks identical to a diff --git a/plugins/aep/skills/upgrade/SKILL.md b/plugins/aep/skills/upgrade/SKILL.md new file mode 100644 index 0000000..b6d0e2a --- /dev/null +++ b/plugins/aep/skills/upgrade/SKILL.md @@ -0,0 +1,25 @@ +--- +name: upgrade +description: Check whether the AEP plugin and the `aep` CLI are current, and upgrade them with the user's confirmation. Use when the user asks whether AEP is up to date or to upgrade or update it, when a session-start line starting with `b10x:` names `aep`, or when a `aep` command behaves differently from what a AEP skill describes. +--- + +# Upgrade AEP + +```bash +b10x upgrade aep --host claude --out ~/.local/state/b10x/plan.json +``` + +Use `--host codex` in Codex. It compares the installed `aep` plugin with what the marketplace serves and the `aep` on `PATH` +with the newest release, and prints each difference with the action that fixes it. Nothing is +changed yet. + +- Nothing to change: say "AEP is current" with the versions, and stop. +- Otherwise show the actions in one list and ask once. After a clear yes: + `b10x setup apply --plan ~/.local/state/b10x/plan.json --yes`. +- A new plugin version loads in a new session; until then `b10x skill aep:` prints the new text. + +No `b10x`? Follow https://github.com/beyond10x/agentplugins/releases/latest/download/SETUP.md first. + +## Next + +- Continue the work that prompted the check; `aep:init` lists the skills. diff --git a/plugins/b10x/.claude-plugin/plugin.json b/plugins/b10x/.claude-plugin/plugin.json new file mode 100644 index 0000000..04bce58 --- /dev/null +++ b/plugins/b10x/.claude-plugin/plugin.json @@ -0,0 +1,20 @@ +{ + "name": "b10x", + "displayName": "Beyond10x", + "description": "Set up, upgrade and check the Beyond10x plugins and binaries, route work to them, and create portable plugins.", + "version": "0.14.13", + "author": { + "name": "Beyond10x" + }, + "license": "Apache-2.0", + "repository": "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/beyond10x/agentplugins", + "keywords": [ + "beyond10x", + "b10x", + "setup", + "plugins", + "routing", + "codex", + "claude-code" + ] +} diff --git a/plugins/b10x/.codex-plugin/plugin.json b/plugins/b10x/.codex-plugin/plugin.json new file mode 100644 index 0000000..39320a9 --- /dev/null +++ b/plugins/b10x/.codex-plugin/plugin.json @@ -0,0 +1,18 @@ +{ + "name": "b10x", + "version": "0.14.13", + "description": "Set up, upgrade and check the Beyond10x plugins and binaries, route work to them, and create portable plugins.", + "author": { + "name": "Beyond10x" + }, + "skills": "./skills/", + "interface": { + "displayName": "Beyond10x", + "shortDescription": "Set up, route and build Beyond10x plugins.", + "longDescription": "Install and upgrade the Beyond10x plugins with the binaries they drive, migrate earlier installs, route a task to the right plugin, and create shared-skill plugins that work in Codex and Claude Code.", + "developerName": "Beyond10x", + "category": "Productivity", + "capabilities": [], + "defaultPrompt": "Use b10x to set up or upgrade the Beyond10x plugins, or to route this task." + } +} diff --git a/plugins/b10x/hooks/hooks.json b/plugins/b10x/hooks/hooks.json new file mode 100644 index 0000000..f3d3935 --- /dev/null +++ b/plugins/b10x/hooks/hooks.json @@ -0,0 +1,15 @@ +{ + "hooks": { + "SessionStart": [ + { + "hooks": [ + { + "type": "command", + "command": "if command -v b10x >/dev/null 2>&1; then b10x check; else echo 'b10x: the b10x binary is not on PATH, so plugin and binary versions are not checked; run `b10x:init`.'; fi", + "timeout": 20 + } + ] + } + ] + } +} diff --git a/plugins/beyond10x/skills/plugin-creator/SKILL.md b/plugins/b10x/skills/authoring-plugins/SKILL.md similarity index 99% rename from plugins/beyond10x/skills/plugin-creator/SKILL.md rename to plugins/b10x/skills/authoring-plugins/SKILL.md index af9d514..a4175b3 100644 --- a/plugins/beyond10x/skills/plugin-creator/SKILL.md +++ b/plugins/b10x/skills/authoring-plugins/SKILL.md @@ -1,5 +1,5 @@ --- -name: plugin-creator +name: authoring-plugins description: Create, update, review, or port installable plugins whose core commands, agent roles, skills, and bundled resources work in both Codex and Claude Code. Use when the user asks to scaffold a plugin, add plugin capabilities or marketplace metadata, make a Claude Code plugin work in Codex, make a Codex plugin work in Claude Code, or audit a plugin for cross-harness portability. --- diff --git a/plugins/beyond10x/skills/plugin-creator/agents/openai.yaml b/plugins/b10x/skills/authoring-plugins/agents/openai.yaml similarity index 51% rename from plugins/beyond10x/skills/plugin-creator/agents/openai.yaml rename to plugins/b10x/skills/authoring-plugins/agents/openai.yaml index f3d79c0..cb6e292 100644 --- a/plugins/beyond10x/skills/plugin-creator/agents/openai.yaml +++ b/plugins/b10x/skills/authoring-plugins/agents/openai.yaml @@ -1,4 +1,4 @@ interface: display_name: "Beyond10x Plugin Creator" short_description: "Build portable Codex and Claude plugins" - default_prompt: "Use $plugin-creator to create a plugin whose workflows work in Codex and Claude Code." + default_prompt: "Use $authoring-plugins to create a plugin whose workflows work in Codex and Claude Code." diff --git a/plugins/beyond10x/skills/plugin-creator/references/compatibility.md b/plugins/b10x/skills/authoring-plugins/references/compatibility.md similarity index 100% rename from plugins/beyond10x/skills/plugin-creator/references/compatibility.md rename to plugins/b10x/skills/authoring-plugins/references/compatibility.md diff --git a/plugins/b10x/skills/init/SKILL.md b/plugins/b10x/skills/init/SKILL.md new file mode 100644 index 0000000..fd9a99f --- /dev/null +++ b/plugins/b10x/skills/init/SKILL.md @@ -0,0 +1,84 @@ +--- +name: init +description: Guided onboarding for Beyond10x — ask what the user wants to do (plan and deliver work, write specifications, isolated Git worktrees, integrations), then install the matching plugins and their CLIs in Claude Code or Codex, migrating any earlier Beyond10x install. Use when the user wants to set up, install or onboard Beyond10x, points at github.com/beyond10x/agentplugins, asks which Beyond10x plugins exist, or when a session-start line starting with `b10x:` says nothing is set up. +--- + +# Set up Beyond10x + +The `b10x` CLI decides; you converse. It reads what is installed on this host, which CLIs are on +`PATH`, and what the `b10x` marketplace serves, and prints exact actions. You ask the user two +questions, show one list of changes, and apply it only after they confirm. + +Never run `claude plugin …` or `codex plugin …` yourself for this, and never edit plugin settings by +hand: `b10x setup apply` snapshots every file it changes, and `b10x setup undo` restores them. + +## 1. Have `b10x` + +Run `b10x --version`. Missing: follow +https://github.com/beyond10x/agentplugins/releases/latest/download/SETUP.md step 1, then come back. + +## 2. Ask what the user wants to do + +Ask **one** multi-select question: *What do you want to do with Beyond10x?* + +| option | installs | +|---|---| +| Plan and deliver work (epics, stories, reviewed implementation waves) | `aep` | +| Write specifications of a system or API (schemas, OpenAPI, conformance) | `ess` | +| Work in isolated Git checkouts that clean up safely | `worktree` | +| Connect external tools and providers | `connectors` | + +Preselect what the user already named ("I want to write specs" → specifications). When planning is +chosen, preselect specifications too and say why: AEP planning models every new noun as an ESS +domain before writing stories around it, so `aep` without `ess` stops at the first new entity. Use the host's +question tool when it has one (Claude Code: `AskUserQuestion` with `multiSelect: true`). + +## 3. Ask how to install the CLIs + +```bash +command -v cargo +``` + +Ask: *How should the command-line tools be installed?* — **prebuilt** (the default: downloads the +release's checksummed archive; no Rust toolchain needed) or **cargo** (builds every crate from the +release tag; offered only when `cargo` is on `PATH`). Without `cargo`, say prebuilt is used and skip +the question. + +## 4. Plan and confirm + +Plan for the host you run in (`--host claude` in Claude Code, `--host codex` in Codex): + +```bash +mkdir -p ~/.local/state/b10x +b10x init --method --host claude --out ~/.local/state/b10x/plan.json +``` + +`` is the answer to step 2, comma separated (`ess`, or `aep,worktree`); `--method` is the +answer to step 3. + +It prints a summary and writes the plan to the file. List every change it names in one numbered +list, one line each, and call out: CLIs it installs or replaces (which version, where), earlier +installs it replaces, and anything under `warn` (a shadowed CLI copy, an edit to a committed file). +Ask one yes/no question; anything but a clear yes changes nothing. + +## 5. Apply + +```bash +b10x setup apply --plan ~/.local/state/b10x/plan.json --yes +``` + +Pass `--yes` only after the user said yes to this exact list. It refuses a plan whose installed +state changed since it was made; plan again if it does. Relay the snapshot path, how many actions +ran and whether it converged; on a failure, the failing line verbatim and `b10x setup undo`. + +## 6. Hand over + +New plugins load in a new session (Claude Code: `/reload-plugins` or restart; Codex: a new thread). +Until then `b10x skill ` lists a plugin's skills and `b10x skill :init` prints the +first one; follow it as if it were loaded. + +## Next + +- Each installed product starts with its own `init`: `/aep:init`, `/ess:init`, `/worktree:init`, + `/connectors:init`. +- `/b10x:upgrade` checks everything later; the session-start `b10x:` line says when it is due. diff --git a/plugins/b10x/skills/init/agents/openai.yaml b/plugins/b10x/skills/init/agents/openai.yaml new file mode 100644 index 0000000..a676042 --- /dev/null +++ b/plugins/b10x/skills/init/agents/openai.yaml @@ -0,0 +1,4 @@ +interface: + display_name: "Beyond10x Setup" + short_description: "Guided onboarding: choose what to do, install the plugins and tools" + default_prompt: "Use $init to set up Beyond10x: ask what I want to do and install the matching plugins and tools." diff --git a/plugins/beyond10x/skills/beyond10x/SKILL.md b/plugins/b10x/skills/routing/SKILL.md similarity index 66% rename from plugins/beyond10x/skills/beyond10x/SKILL.md rename to plugins/b10x/skills/routing/SKILL.md index ab32fee..eaff26a 100644 --- a/plugins/beyond10x/skills/beyond10x/SKILL.md +++ b/plugins/b10x/skills/routing/SKILL.md @@ -1,6 +1,6 @@ --- -name: beyond10x -description: Navigate the Beyond10x engineering ecosystem and route work to the smallest matching plugin or public resource. Use when the user asks what Beyond10x provides, which plugin to install or invoke, where the AEP, ESS, Entity Runtime, or agent-plugin documentation lives, or when a request spans or is ambiguous between the Beyond10x plugins. +name: routing +description: Navigate the Beyond10x engineering ecosystem and route work to the smallest matching plugin or public resource. Use when the user asks what Beyond10x provides, which plugin fits a task, where the AEP, ESS, Entity Runtime, or agent-plugin documentation lives, or when a request spans or is ambiguous between the Beyond10x plugins. --- # Beyond10x guide @@ -11,7 +11,7 @@ Route the request; do not reproduce a specialist plugin's full workflow. 1. Identify the user's immediate decision or outcome. 2. For non-trivial implementation, cross-repository work, or release/deployment changes in a - Beyond10x repository, route through `aep-plan` first so the owning artifact, relations, and + Beyond10x repository, route through `aep:planning` first so the owning artifact, relations, and scope exist before implementation continues. 3. Select one plugin from the routing table. Select several only when the task genuinely crosses their boundaries. @@ -22,13 +22,17 @@ Route the request; do not reproduce a specialist plugin's full workflow. | Request | Route | |---|---| -| Choose a plugin, understand the ecosystem, or find public documentation | `beyond10x` | -| Create, update, review, or port an installable plugin | `plugin-creator` in this plugin | -| Plan or decompose work, review a plan, or reverse-engineer a backlog | `aep-plan` | -| Scope and deliver accepted development work through a reviewed wave | `aep-drive` | -| Specify an ESS model or guide deterministic schema or OpenAPI projection | `ess-specify` | -| Create, inspect, finish, or safely clean Git worktrees | `workspace-hygiene` | -| Set up providers, inspect Connector readiness, or invoke configured integrations through the CLI | `connectors` | +| Install, upgrade or repair the Beyond10x plugins and their binaries | `b10x:init` | +| Choose a plugin, understand the ecosystem, or find public documentation | `b10x:routing` (this skill) | +| Create, update, review, or port an installable plugin | `b10x:authoring-plugins` | +| Plan or decompose work, review a plan, or reverse-engineer a backlog | `aep:planning` | +| Scope and deliver accepted development work through a reviewed wave | `aep:implementing` | +| Specify a system or API | `ess:specifying` | +| Derive a specification for an existing system | `ess:retrofitting` | +| Run or raise a conformance suite | `ess:testing-conformance` | +| Harden a specification once its suite is green | `ess:hardening` | +| Create, inspect, finish, or safely clean Git worktrees | `worktree:managing-worktrees` | +| Set up providers, inspect Connector readiness, or invoke configured integrations through the CLI | `connectors:integrating` | ## Preserve boundaries diff --git a/plugins/beyond10x/skills/beyond10x/agents/openai.yaml b/plugins/b10x/skills/routing/agents/openai.yaml similarity index 50% rename from plugins/beyond10x/skills/beyond10x/agents/openai.yaml rename to plugins/b10x/skills/routing/agents/openai.yaml index 4bf9eb3..b9c2125 100644 --- a/plugins/beyond10x/skills/beyond10x/agents/openai.yaml +++ b/plugins/b10x/skills/routing/agents/openai.yaml @@ -1,4 +1,4 @@ interface: display_name: "Beyond10x Guide" short_description: "Route work across Beyond10x plugins" - default_prompt: "Use $beyond10x to select the right Beyond10x plugin and resources for this task." + default_prompt: "Use $routing to select the right Beyond10x plugin and resources for this task." diff --git a/plugins/beyond10x/skills/beyond10x/references/resources.md b/plugins/b10x/skills/routing/references/resources.md similarity index 68% rename from plugins/beyond10x/skills/beyond10x/references/resources.md rename to plugins/b10x/skills/routing/references/resources.md index 99eae4d..9425705 100644 --- a/plugins/beyond10x/skills/beyond10x/references/resources.md +++ b/plugins/b10x/skills/routing/references/resources.md @@ -7,11 +7,11 @@ Use the narrowest link that answers the request. | [Getting started](https://beyond10x.github.io/getting-started/) | Public entry point and adoption paths | | [Agent Plugins](https://beyond10x.github.io/agentplugins/) | Marketplace overview, installation, and plugin selection | | [Golden path](https://beyond10x.github.io/agentplugins/docs/golden-path) | One worked run, from a feature idea to a critiqued plan, on a repository that already exists | -| [Beyond10x plugin](https://beyond10x.github.io/agentplugins/docs/plugins/beyond10x) | This router and the portable plugin-creation workflow | -| [AEP Plan plugin](https://beyond10x.github.io/agentplugins/docs/plugins/aep-plan) | Governed planning skill and specialist planning agents | -| [AEP Drive plugin](https://beyond10x.github.io/agentplugins/docs/plugins/aep-drive) | Development-wave roles and coordination | -| [ESS Specify plugin](https://beyond10x.github.io/agentplugins/docs/plugins/ess-specify) | ESS specification, validation and deterministic projection guidance | -| [Workspace Hygiene plugin](https://beyond10x.github.io/agentplugins/docs/plugins/workspace-hygiene) | Managed Git worktree lifecycle and cleanup guidance | +| [b10x plugin](https://beyond10x.github.io/agentplugins/docs/plugins/b10x) | Setup, this router and the portable plugin-creation workflow | +| [AEP Plan plugin](https://beyond10x.github.io/agentplugins/docs/plugins/aep) | Governed planning skill and specialist planning agents | +| [AEP Drive plugin](https://beyond10x.github.io/agentplugins/docs/plugins/aep) | Development-wave roles and coordination | +| [ESS agent plugin](https://beyond10x.github.io/agentplugins/docs/plugins/ess) | ESS specification, retrofit, validation, projection and conformance guidance, released with the `ess` binary | +| [Worktree plugin](https://beyond10x.github.io/agentplugins/docs/plugins/worktree) | Managed Git worktree lifecycle and cleanup guidance, released with the `worktree` binary | | [AEP](https://beyond10x.github.io/aep/) | Protocol concepts, commands, and adopter documentation | | [AEP Service](https://beyond10x.github.io/aep-service/) | Hosted AEP service boundary and operations | | [ESS](https://beyond10x.github.io/ess/) | Executable System Specification model, CLI, and adapters | diff --git a/plugins/b10x/skills/upgrade/SKILL.md b/plugins/b10x/skills/upgrade/SKILL.md new file mode 100644 index 0000000..d4f08bf --- /dev/null +++ b/plugins/b10x/skills/upgrade/SKILL.md @@ -0,0 +1,28 @@ +--- +name: upgrade +description: Check every installed Beyond10x plugin and CLI against what is newest, and upgrade them with the user's confirmation. Use when the user asks whether Beyond10x is up to date or to upgrade or update it, or when a session-start line starting with `b10x:` reports drift, a legacy plugin or an old check. +--- + +# Upgrade Beyond10x + +```bash +b10x upgrade --host claude --out ~/.local/state/b10x/plan.json +``` + +Use `--host codex` in Codex. It checks each installed product — plugin against the marketplace, CLI on `PATH` against the newest +release — plus earlier installs under retired names, and prints each difference with the action that +fixes it. Nothing is changed yet. One product only: `b10x upgrade ess --host claude --out …`. A CLI the repository pins +in `b10x.toml` stays at its pin: the plan names a newer release but does not install it +(`b10x unpin ` follows the newest again). + +- Nothing to change: say "Beyond10x is current" with the versions, and stop. +- Otherwise show the actions in one list and ask once. After a clear yes: + `b10x setup apply --plan ~/.local/state/b10x/plan.json --yes`. +- New plugin versions load in a new session; until then `b10x skill :` prints the + new text. + +No `b10x`? Follow https://github.com/beyond10x/agentplugins/releases/latest/download/SETUP.md first. + +## Next + +- Add a product: `/b10x:init`. diff --git a/plugins/beyond10x/.claude-plugin/plugin.json b/plugins/beyond10x/.claude-plugin/plugin.json deleted file mode 100644 index 1b49959..0000000 --- a/plugins/beyond10x/.claude-plugin/plugin.json +++ /dev/null @@ -1,10 +0,0 @@ -{ - "name": "beyond10x", - "displayName": "Beyond10x", - "description": "Route work to focused Beyond10x plugins and create portable plugins for Codex and Claude Code.", - "version": "0.9.2", - "author": { "name": "Beyond10x" }, - "license": "Apache-2.0", - "repository": "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/beyond10x/agentplugins", - "keywords": ["beyond10x", "plugins", "routing", "codex", "claude-code"] -} diff --git a/plugins/beyond10x/.codex-plugin/plugin.json b/plugins/beyond10x/.codex-plugin/plugin.json deleted file mode 100644 index d0e2fa3..0000000 --- a/plugins/beyond10x/.codex-plugin/plugin.json +++ /dev/null @@ -1,18 +0,0 @@ -{ - "name": "beyond10x", - "version": "0.9.2", - "description": "Route work to focused Beyond10x plugins and create portable plugins for Codex and Claude Code.", - "author": { - "name": "Beyond10x" - }, - "skills": "./skills/", - "interface": { - "displayName": "Beyond10x", - "shortDescription": "Navigate the marketplace and build plugins.", - "longDescription": "Find the focused Beyond10x plugin for a task, open the relevant public resources, and create shared-skill plugins that work in Codex and Claude Code.", - "developerName": "Beyond10x", - "category": "Productivity", - "capabilities": [], - "defaultPrompt": "Use Beyond10x to route this task or help me create a portable agent plugin." - } -} diff --git a/plugins/connectors/.claude-plugin/plugin.json b/plugins/connectors/.claude-plugin/plugin.json index bf13eef..b426dd1 100644 --- a/plugins/connectors/.claude-plugin/plugin.json +++ b/plugins/connectors/.claude-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "connectors", - "version": "0.9.2", + "version": "0.14.13", "description": "Set up, inspect, and invoke governed integrations through the connectors CLI.", "author": { "name": "Beyond10x" }, "license": "Apache-2.0", diff --git a/plugins/connectors/.codex-plugin/plugin.json b/plugins/connectors/.codex-plugin/plugin.json index 658967b..5d13ea4 100644 --- a/plugins/connectors/.codex-plugin/plugin.json +++ b/plugins/connectors/.codex-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "connectors", - "version": "0.9.2", + "version": "0.14.13", "description": "Set up, inspect, and invoke governed integrations through the connectors CLI.", "author": { "name": "Beyond10x" }, "license": "Apache-2.0", diff --git a/plugins/connectors/skills/init/SKILL.md b/plugins/connectors/skills/init/SKILL.md new file mode 100644 index 0000000..9490fa8 --- /dev/null +++ b/plugins/connectors/skills/init/SKILL.md @@ -0,0 +1,44 @@ +--- +name: init +description: Start with Connectors in this project — make sure the `connectors` CLI is available and take the first step. Connectors is provider setup, connection diagnostics and governed invocation of integrations. Use when the user wants to connect external tools or providers, asks to set up connectors, or when `connectors:integrating` reports that `connectors` is missing. Installs CLIs only after the user confirms the plan. +--- + +# Start with Connectors + +## 1. Have the CLI + +Run `connectors --version`. The plugin itself is set up with `b10x`: plan, confirm, then +`b10x setup apply --plan ~/.local/state/b10x/plan.json --yes`. Plan for the host you run in +(`--host claude` in Claude Code, `--host codex` in Codex): + +```bash +mkdir -p ~/.local/state/b10x +b10x init connectors --host claude --out ~/.local/state/b10x/plan.json +``` + +The CLI is installed by hand as `connectors:integrating` describes. + +No `b10x`? Follow https://github.com/beyond10x/agentplugins/releases/latest/download/SETUP.md first; its step 1 asks the user before it installs `b10x`. + +## 2. First step here + +Check the CLI and what it can reach: + +```bash +connectors inspect doctor +``` + +`b10x` does not install the `connectors` CLI; `connectors:integrating` names the release to install. + +## 3. Pick the work + +| the task | skill | +|---|---| +| set up a provider, diagnose a connection, or invoke an integration | `connectors:integrating` | + +A skill that is not loaded yet in this session prints with `b10x skill connectors:`. + +## Next + +- Connect or diagnose: `connectors:integrating`. +- Later, `connectors:upgrade` checks for a newer plugin and CLI. diff --git a/plugins/connectors/skills/connectors/SKILL.md b/plugins/connectors/skills/integrating/SKILL.md similarity index 99% rename from plugins/connectors/skills/connectors/SKILL.md rename to plugins/connectors/skills/integrating/SKILL.md index 8cdb1ec..e278192 100644 --- a/plugins/connectors/skills/connectors/SKILL.md +++ b/plugins/connectors/skills/integrating/SKILL.md @@ -1,5 +1,5 @@ --- -name: connectors +name: integrating description: Use the Beyond10x connectors CLI to set up providers, diagnose connections, discover admitted operations, and invoke integrations. Use when the user asks to use connectors, connect a provider, inspect Connector readiness, or access a configured integration through Connectors. Do not use for generic connector implementation or unrelated database connections. --- diff --git a/plugins/connectors/skills/upgrade/SKILL.md b/plugins/connectors/skills/upgrade/SKILL.md new file mode 100644 index 0000000..2144e71 --- /dev/null +++ b/plugins/connectors/skills/upgrade/SKILL.md @@ -0,0 +1,25 @@ +--- +name: upgrade +description: Check whether the Connectors plugin and the `connectors` CLI are current, and upgrade them with the user's confirmation. Use when the user asks whether Connectors is up to date or to upgrade or update it, when a session-start line starting with `b10x:` names `connectors`, or when a `connectors` command behaves differently from what a Connectors skill describes. +--- + +# Upgrade Connectors + +```bash +b10x upgrade connectors --host claude --out ~/.local/state/b10x/plan.json +``` + +Use `--host codex` in Codex. It compares the installed `connectors` plugin with what the marketplace serves and the `connectors` on `PATH` +with the newest release, and prints each difference with the action that fixes it. Nothing is +changed yet. + +- Nothing to change: say "Connectors is current" with the versions, and stop. +- Otherwise show the actions in one list and ask once. After a clear yes: + `b10x setup apply --plan ~/.local/state/b10x/plan.json --yes`. +- A new plugin version loads in a new session; until then `b10x skill connectors:` prints the new text. + +No `b10x`? Follow https://github.com/beyond10x/agentplugins/releases/latest/download/SETUP.md first. + +## Next + +- Continue the work that prompted the check; `connectors:init` lists the skills. diff --git a/plugins/ess-specify/.claude-plugin/plugin.json b/plugins/ess-specify/.claude-plugin/plugin.json deleted file mode 100644 index 9a51928..0000000 --- a/plugins/ess-specify/.claude-plugin/plugin.json +++ /dev/null @@ -1,10 +0,0 @@ -{ - "name": "ess-specify", - "displayName": "ESS Specify", - "description": "Validate ESS models, guide deterministic schema and OpenAPI projections, and raise or audit conformance-suite coverage against a real implementation.", - "version": "0.9.2", - "author": { "name": "Beyond10x" }, - "license": "Apache-2.0", - "repository": "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/beyond10x/agentplugins", - "keywords": ["ess", "specify", "schema", "openapi", "projection", "conformance", "coverage"] -} diff --git a/plugins/ess-specify/.codex-plugin/plugin.json b/plugins/ess-specify/.codex-plugin/plugin.json deleted file mode 100644 index ecbd7aa..0000000 --- a/plugins/ess-specify/.codex-plugin/plugin.json +++ /dev/null @@ -1,18 +0,0 @@ -{ - "name": "ess-specify", - "version": "0.9.2", - "description": "Validate ESS schemas, guide deterministic projections, and audit conformance coverage without guessing semantics.", - "author": { - "name": "Beyond10x" - }, - "skills": "./skills/", - "interface": { - "displayName": "ESS Specify", - "shortDescription": "Validate, project and conformance-test typed ESS contracts.", - "longDescription": "Validate Executable System Specifications, inspect compiled IR, guide deterministic JSON Schema and OpenAPI projections with explicit coverage gaps and refusals, and raise or audit a conformance suite against a real implementation.", - "developerName": "Beyond10x", - "category": "Productivity", - "capabilities": [], - "defaultPrompt": "Use ess-specify to validate, project or conformance-test this system contract." - } -} diff --git a/plugins/ess/.claude-plugin/plugin.json b/plugins/ess/.claude-plugin/plugin.json new file mode 100644 index 0000000..fc0629f --- /dev/null +++ b/plugins/ess/.claude-plugin/plugin.json @@ -0,0 +1,20 @@ +{ + "name": "ess", + "displayName": "ESS", + "description": "Write, retrofit, validate and project Executable System Specifications, and hold implementations to them with conformance suites.", + "version": "0.14.13", + "author": { + "name": "Beyond10x" + }, + "license": "Apache-2.0", + "repository": "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/beyond10x/agentplugins", + "keywords": [ + "ess", + "specification", + "schema", + "openapi", + "projection", + "conformance", + "retrofit" + ] +} diff --git a/plugins/ess/.codex-plugin/plugin.json b/plugins/ess/.codex-plugin/plugin.json new file mode 100644 index 0000000..063bf10 --- /dev/null +++ b/plugins/ess/.codex-plugin/plugin.json @@ -0,0 +1,18 @@ +{ + "name": "ess", + "version": "0.14.13", + "description": "Write, retrofit, validate and project Executable System Specifications, and hold implementations to them with conformance suites.", + "author": { + "name": "Beyond10x" + }, + "skills": "./skills/", + "interface": { + "displayName": "ESS", + "shortDescription": "Write, retrofit and conformance-test typed ESS specifications.", + "longDescription": "Draft a new ESS domain or retrofit one onto an existing system, validate and compile it, project deterministic JSON Schema and OpenAPI, and raise or audit a conformance suite against a real implementation.", + "developerName": "Beyond10x", + "category": "Productivity", + "capabilities": [], + "defaultPrompt": "Use ess to write, retrofit or conformance-test this system's specification." + } +} diff --git a/plugins/ess/agents/author.md b/plugins/ess/agents/author.md new file mode 100644 index 0000000..124332d --- /dev/null +++ b/plugins/ess/agents/author.md @@ -0,0 +1,19 @@ +--- +name: author +description: Write or extend an ESS specification — draft a new domain, add an entity, relation, command or view, and validate, compile and project it. Invoke when the operator asks to write, draft, model or extend a specification, or when a story introduces a noun that needs a typed home. Edits specification sources only; never edits generated output by hand. +tools: [Read, Grep, Glob, Bash, Edit, Write] +--- + +# ESS author + +Follow the `ess:specifying` skill completely. If it is not loaded, run `b10x skill ess:specifying` and follow +its output. + +Charter: + +- Work on the specification sources only. Generated files change only through `ess generate`. +- Run `ess specify validate --path ` after every addition, not once at the end. +- Write an `UNMAPPED:` marker for anything you could not read from code, a contract or an artifact. +- Report: the files changed, the final `validate` output verbatim, and every `UNMAPPED:` marker. +- End the report with the review step: every marker and contradiction goes to the owner before + anything is generated from the specification. diff --git a/plugins/ess/agents/conformance.md b/plugins/ess/agents/conformance.md new file mode 100644 index 0000000..cd4bfa4 --- /dev/null +++ b/plugins/ess/agents/conformance.md @@ -0,0 +1,21 @@ +--- +name: conformance +description: Raise and audit an ESS conformance suite against a real implementation — attack skipped scenarios, prove a green run can go red, and set the gate that holds the counts. Invoke when the operator asks to raise conformance coverage, explain skipped scenarios, check whether a passing suite tests anything, or fix a conformance job that fails only in CI. +tools: [Read, Grep, Glob, Bash, Edit, Write] +--- + +# ESS conformance + +Follow the `ess:testing-conformance` skill completely. If it is not loaded, run `b10x skill ess:testing-conformance` and +follow its output. + +Charter: + +- Claim only `passed` as progress. A `skipped` scenario is no information about the implementation. +- Before raising any count, break one behaviour and name the scenario that goes red. +- Never weaken a scenario or a target to turn a failure green. +- Report: `passed`/`skipped`/`failed` before and after, the mutation you ran and the scenario it + failed, and the exact commands. +- When a coverage task finishes on a green suite, offer the `ess:hardening` catalogue as the next + step, not only mutation: name the techniques it would run first (the design review, the spec + diff in the gate) and what each needs. Offer it; do not start it unasked. diff --git a/plugins/ess/agents/retrofitter.md b/plugins/ess/agents/retrofitter.md new file mode 100644 index 0000000..177ec4b --- /dev/null +++ b/plugins/ess/agents/retrofitter.md @@ -0,0 +1,19 @@ +--- +name: retrofitter +description: Derive an ESS specification for an existing system from its OpenAPI contract, observed deployment or code, citing a source for every declaration. Invoke when the operator asks to retrofit, adopt or reverse-engineer a specification for a service that already runs. Writes specification sources only; never changes the system it describes. +tools: [Read, Grep, Glob, Bash, Edit, Write] +--- + +# ESS retrofitter + +Follow the `ess:retrofitting` skill completely. If it is not loaded, run `b10x skill ess:retrofitting` and +follow its output. + +Charter: + +- Describe what the system does today. Never add a state, command or field the code does not have. +- Every declaration cites its source as a path, a `path:line` or a schema name. +- Never edit application code, configuration or a cluster. A live Kubernetes read needs the + operator's explicit credentials. +- Report: the specification path, `ess specify validate` output verbatim, every `UNMAPPED:` marker, + and every disagreement between sources. diff --git a/plugins/ess/skills/hardening/SKILL.md b/plugins/ess/skills/hardening/SKILL.md new file mode 100644 index 0000000..91f1879 --- /dev/null +++ b/plugins/ess/skills/hardening/SKILL.md @@ -0,0 +1,91 @@ +--- +name: hardening +description: >- + Harden an ESS specification and the implementation held to it once the conformance suite is green — ask the questions a passing suite cannot, with a catalogue of eight techniques from mutation to a design review. Use when the user asks how to harden a specification, whether a green suite can be trusted, what to do after coverage is raised, or asks for mutation testing, random command sequences or a model-based check, replay of a real caller, determinism checks, metamorphic relations, guard analysis, classifying a specification diff as breaking, or reviewing a design document against a specification. Not for raising the number of scenarios that execute, which is the `ess:testing-conformance` skill; not for writing the specification, which is `ess:specifying`. +--- + +# Hardening a specification + +A green conformance suite answers one question: do the scenarios somebody wrote pass. Each +technique here asks a different one, and each runs only after the suite is green — a red suite is +the work, and `ess:testing-conformance` is where it is done. + +## The rule that makes any of this count + +**A check nobody has seen fail is not evidence. Plant a defect first.** Before a technique's green +result is reported, break the implementation (or the specification) in the way the technique claims +to catch, run it, and record the named failure. Then restore and run it green. A technique that +cannot be made to fail on a planted defect has measured nothing, whatever it prints. + +Record both runs: the planted defect, the command, the failure it produced, and the green run after +restoring. + +## The catalogue + +| # | technique | the question | what it catches | needs from the IR | +|---|---|---|---|---| +| 1 | mutation audit | would the suite notice if a declared rule broke? | declared behaviour no scenario pins down: a dropped `from` state, a moved guard boundary, a dropped `sets` entry | lifecycles, guards, `sets`, payloads | +| 2 | random command sequences against a reference model | does the implementation agree with the spec along paths nobody wrote? | engine defects that only show later in a sequence: a view that caps its rows, a refusal that still writes, an identity reused, a state not re-enterable; invariants that hold only through undeclared defaults | everything the [reference model](references/reference-model.md) reads: lifecycles, guards, `sets`, payloads, views, invariants, `wrong_state` | +| 3 | replay of the real caller | does the code that *uses* the implementation send only what the spec accepts? | a caller relying on behaviour the spec does not declare; refusal outcomes the caller never triggers | the reference model, plus actors (who may send which command) | +| 4 | determinism | is the implementation deterministic where the spec assumes it? | clock reads, randomness, iteration order leaking into outcomes | none beyond technique 2's runner | +| 5 | metamorphic relations | do promises that span two runs hold? | a command allowed where the design forbids it ("a pause changes nothing but the pause commands") | the reference model; the relation comes from the design | +| 6 | exhaustive guard analysis | are guards dead, overlapping, gap-leaving, or weak enough to admit an invariant break? | a `when:` that can never match, two that match together, an input no outcome answers | guards, field types, invariants | +| 7 | spec diff in the gate | is a change breaking, and was that acknowledged? | a narrowing released as if it were additive | two compiled revisions | +| 8 | design review by an agent | what does the design say that the spec omits or contradicts? | rules the design states and the spec never declares ("one entry per player", "the count always goes up") | the spec files and the design document; no IR | + +Procedures, with the defect to plant for each: [references/techniques.md](references/techniques.md). + +Formal model checking (a TLA+ or Alloy export) is not in the catalogue. On a spec with 31 commands, +random sequences reached every declared outcome, 59 of 59; reach for a model checker only when +technique 2 reports declared outcomes it never reaches. + +## The order + +Cheapest first, with one exception: **the design review runs early, because it finds the most.** On +one adopter's spec (3 domains, 7 entities, 31 commands, a 121-scenario suite that passed) it +produced 26 findings, the most of any technique. What it finds changes the spec, and +every later technique is then run against the spec that will stay. + +1. **Design review** (8) — no tooling, one agent, the [brief](references/design-review.md). +2. **Spec diff in the gate** (7) — one command and a [classification](references/spec-diff.md); it + protects everything after it. +3. **Mutation audit** (1) — reuses the suite you already have. +4. **Guard analysis** (6) — reads the IR only; no implementation runs. +5. **Reference model and random sequences** (2) — the one piece of code the catalogue needs; build it + once. +6. **Determinism** (4), **metamorphic relations** (5) and **caller replay** (3) — each reuses the + runner from step 5. + +Stop where the cost exceeds what the spec is worth, and say which techniques were not run. + +## What the IR is for + +Techniques 1–6 read the canonical IR, not the YAML: + +```console +ess specify compile --path --format json --out +``` + +The IR is the resolved model: every name qualified, every guard in structured form, every default +applied. A technique that reads the YAML re-implements resolution and disagrees with the compiler +somewhere nobody looked. + +## The reference model + +Techniques 2, 3 and 5 rest on one small interpreter over the IR — about 150 lines — that answers, +for any command in any state, what the spec says happens. `ess` does not ship it yet; +[beyond10x/ess#114](https://github.com/beyond10x/ess/issues/114) proposes shipping the mutation audit +and the sequence runner as `ess` features, and carries a draft for the TypeScript target. Until it +ships, build it from the pattern: [references/reference-model.md](references/reference-model.md). + +## Reporting + +Per technique run: the planted defect and the failure it produced, the green result after restoring, +the command, and the findings — each one a spec gap, an implementation defect or a false positive of +the technique, said which. Name the techniques not run. A finding that belongs to `ess` itself (a +generator or validate gap) is an issue on beyond10x/ess, not a workaround here. + +## Next + +- A finding changes the specification: `ess:specifying`. +- A finding is a missing scenario or a skipped one: `ess:testing-conformance`. diff --git a/plugins/ess/skills/hardening/references/design-review.md b/plugins/ess/skills/hardening/references/design-review.md new file mode 100644 index 0000000..9b5a223 --- /dev/null +++ b/plugins/ess/skills/hardening/references/design-review.md @@ -0,0 +1,71 @@ +# Design review: a brief for an agent + +Technique 8. Dispatch one agent with the brief below, filled in. It found the most of any technique +on the one adopter where all eight were run — 26 findings — so run it early. + +The agent needs read access to three things and write access to none of them. + +--- + +## Brief + +You are reviewing a specification against the design it claims to implement. You do not edit +either. You produce a list of findings. + +**Inputs** + +- the design document: `` +- the mapping, if one exists — the document or table that says which design statement became which + spec declaration: ``, or `none` +- the specification: ``; validate it first with `ess specify validate --path ` + and stop if it refuses + +**Procedure** + +1. Read the design end to end. List every statement that promises behaviour: a rule, a limit, an + ordering, a uniqueness, a "never", an "always", a "only when". +2. For each statement, find where the spec declares it — an entity field or invariant, a lifecycle + transition, a command outcome and its `when:` guard, an actor's grant, a view. +3. Then go the other way: for every spec declaration, find the design statement that justifies it. +4. Classify each mismatch with exactly one of the classifications below. + +**Classifications** + +| classification | meaning | +|---|---| +| `missing` | the design states it; the spec does not declare it | +| `contradicts` | both state it, and they disagree | +| `stale mapping` | the mapping points a design statement at a declaration that no longer says it, or at one that does not exist | +| `spec-only` | the spec declares it; no design statement justifies it | +| `unclear` | the design statement is too vague to decide which of the above applies | + +**Every finding cites both sides.** The design by its line (`design.md:42`, quoted), the spec by +`file:line` (quoted). A `missing` finding cites the spec file and line where the declaration would +belong; a `spec-only` finding cites the nearest design section and says it has nothing. A finding +without both citations is not a finding; drop it. + +**Quote a vague statement and mark it `unclear`; never invent a meaning for it.** "Players should +not be able to cheat" is `unclear` with that sentence quoted. Supplying a plausible rule and then +reporting the spec as missing it is the one failure this review exists to avoid: downstream, the +invented rule reads as the designer's. + +**Output**, one row per finding: + +```text +| # | classification | design (file:line, quoted) | spec (file:line, quoted) | note | +``` + +Then the counts per classification, and the design sections you did not reach, if any. + +--- + +## After the review + +- Each `missing` and `contradicts` finding goes to the design's owner or becomes a spec change + (`ess:specifying`); the review does not decide which. +- `spec-only` findings are candidates for deletion — "a declared refusal nobody executes is a + fiction" (`ess:testing-conformance`). +- `unclear` findings go back to the design's author as questions, with the quote. + +**Plant a defect before trusting a clean review:** on a copy of the spec, delete one declaration the +design plainly states, and run the brief on the copy. It must come back `missing`, citing both sides. diff --git a/plugins/ess/skills/hardening/references/reference-model.md b/plugins/ess/skills/hardening/references/reference-model.md new file mode 100644 index 0000000..42b3655 --- /dev/null +++ b/plugins/ess/skills/hardening/references/reference-model.md @@ -0,0 +1,85 @@ +# The reference model: a pattern + +Techniques 2, 3 and 5 compare an implementation with what the spec says happens. That needs a +**reference model**: a small interpreter over the compiled IR that, given a state and a command, +returns the outcome the spec declares. On one adopter it was about 150 lines of TypeScript. + +**Status.** `ess` does not ship this. [beyond10x/ess#114](https://github.com/beyond10x/ess/issues/114) +proposes shipping the mutation audit and a seeded sequence runner built on this model as `ess` +features, emitted beside the suite for `--target typescript` and `--target go` and driven through the +same `Target` interface; the issue carries a draft (`explore.ts`) for the TypeScript target. Until it +ships, write the model yourself from the pattern below, and check that issue before starting — once +it ships, use what it emits instead. + +## Input + +The canonical JSON IR, and nothing else: + +```console +ess specify compile --path --format json --out +``` + +Read the IR, never the YAML: the IR has qualified names, guards in structured form and defaults +applied. Read field names from your own IR file rather than from this page; the pattern is stable, +the exact keys are the compiler's. + +## State + +```text +instances: map entity -> map identity -> { state, fields } +issued: set of every identity ever created (identity uniqueness) +``` + +Nothing else. A model that tracks anything the IR does not declare is asserting behaviour the spec +does not have. + +## Step: one command + +For `execute(command, input)`: + +1. **Pick the outcome.** Evaluate each outcome's `when:` predicate over the input — comparisons over + `input_field` and `literal` values, combined with and/or/not. Exactly one non-`wrong_state` + outcome must match; zero or two is a spec defect, report it rather than choosing. +2. **Check the lifecycle.** For an outcome that `moves` an instance through a transition, look the + instance up by the outcome's `instance` field. Missing instance, or current state not in the + transition's `from`: answer with the command's `wrong_state` outcome instead, and **change + nothing** — no `sets`, no events. (That a `wrong_state` answer still applied `sets` was one of the + defects this found.) +3. **Apply.** + - `creates`: a new identity (refuse to reuse one in `issued`), the lifecycle's `initial` state, the + fields the outcome sets. + - `moves`: set the state to the transition's `to`. + - `updates` / `sets`: write each field from its value expression (`input.` or a literal). + - `error`: change nothing. +4. **Emit.** For each event in `emits`, build its payload from the outcome's `payload` mapping. +5. **Check invariants** of every instance the step touched, over its fields after the step. + +Return `{ outcome, error, events, payloads }`. + +## Views + +For `query(view)`: take every instance of the view's `source`, keep those its `filter` admits, and +project the view's `fields`. Compare rows **and their order** with the implementation's answer; a +view whose order the spec does not declare compares as a set, and the model says which. + +**Unknown is an answer.** Where a view publishes a field no command ever set on that instance, the +model does not know its value. Report that as its own finding: whatever the implementation shows, +it rests on an undeclared default. That is how three invariants that held only by accident were +found (beyond10x/ess#112). + +## What drives it + +The model is the oracle; a runner drives it and the implementation side by side: + +- a seeded generator picks actor, command and input (valid, at each guard boundary, and invalid) — + the actor from those whose grants include the command, plus one that is not, to exercise refusal; +- after each step, compare the step results, then every view, then identity uniqueness; +- shrink a failing trace by dropping steps while it still fails; +- at the end, list every declared outcome never reached, and fail if the list is not empty. + +## Proving the model is not the implementation's twin + +A model written by reading the implementation agrees with it by construction. Write it from the IR +only, and before trusting it, plant one engine defect in the implementation (a view capped at 5 +rows, a `wrong_state` that still writes, a reused identity, a state that cannot be re-entered). The +runner must catch it. A model that agrees with a planted defect is a copy, not a reference. diff --git a/plugins/ess/skills/hardening/references/spec-diff.md b/plugins/ess/skills/hardening/references/spec-diff.md new file mode 100644 index 0000000..c8ccef8 --- /dev/null +++ b/plugins/ess/skills/hardening/references/spec-diff.md @@ -0,0 +1,57 @@ +# Spec diff in the gate: classifying changes + +Technique 7. The gate compares the specification with the one at the last release tag, and fails +on a breaking change nobody acknowledged. + +## The diff + +`ess verify diff` compares two specification directories: + +```console +ess verify diff --from --to --format json +``` + +`--from` and `--to` are paths, so materialise the tagged revision first, for example +`git archive | tar -x -C `. The `ess-diff` document lists `changes`, each +with a stable `id` (`type//variant-removed/`), a `relation` (`expanded`, `narrowed` +or `changed`) and the change itself. + +`ess` names each change and its relation; **it does not decide whether a change is breaking.** Until +it does, classify with the rule below. + +## The rule + +**Additive** (not breaking): + +- an added item — a type, entity, field, command, outcome, event, view, actor +- an added enum variant +- an added lifecycle transition +- an added grant (an actor `may` one more command) +- an added `accepts` or `publishes` entry +- a wording change — `summary`, `naming.display`, descriptions + +**Breaking:** everything else, and **anything whose `relation` is `narrowed`**, whatever its kind. A +removal, a rename, a changed guard, a changed invariant, a changed type, a changed `wire` name — all +breaking. When a change's kind is not on the additive list, it is breaking; the list grows by +decision, not by argument. + +## Acknowledgement + +A breaking change passes the gate only when acknowledged, and an acknowledgement is a **committed +file keyed to the release tag** — for example `spec-acknowledgements/.yaml` — listing each +acknowledged change by its `id`, with one line saying why it is acceptable. + +The gate: + +1. materialises the spec at the last release tag; +2. runs the diff; +3. classifies each change by the rule; +4. fails on any breaking change whose `id` is not in the acknowledgement file for that tag, naming + the `id`; +5. fails on any acknowledged `id` that no longer appears in the diff — a stale acknowledgement + hides the next change with the same name. + +A new release tag starts an empty acknowledgement file; the previous tag's never carries over. + +**Plant a defect before trusting it:** on a branch, remove one enum variant. The gate must fail and +name the `variant-removed` id; add that id to the acknowledgement file and it must pass. diff --git a/plugins/ess/skills/hardening/references/techniques.md b/plugins/ess/skills/hardening/references/techniques.md new file mode 100644 index 0000000..6955d33 --- /dev/null +++ b/plugins/ess/skills/hardening/references/techniques.md @@ -0,0 +1,152 @@ +# The eight techniques, step by step + +Each procedure ends the same way: plant the named defect, watch the technique fail on it, restore, +run it green. The figures in each section are from one adopter's spec (3 domains, 7 entities, 31 +commands, a JavaScript implementation, a generated suite of 121 scenarios that all passed); they say +what a technique is worth on a real spec, not what yours will find. + +Every technique that reads the model reads the compiled IR: + +```console +ess specify compile --path --format json --out +``` + +## 1. Mutation audit + +**Question:** would the suite notice if a declared rule broke? + +1. From the IR, list one mutant per declared rule. The classes that find gaps: + + | class | example | + |---|---| + | drop a `from` state | `close: from [Open, Held]` → `[Open]` | + | change a transition's `to` | `B → C` becomes `B → B` | + | move a guard boundary | `items >= 0` → `items > 0` | + | drop a `sets` entry | `colour: input.colour` removed | + | invert or weaken a guard | `A && B` → `A \|\| B` | + | wrong error, event or order | `desc` → `asc` | + +2. Apply each mutant — to the implementation's rule table, or to a copy of the specification whose + suite is re-synthesized and run against the unchanged implementation — one at a time. +3. Record, per mutant, the scenarios that failed. A mutant no scenario kills is a declared rule the + suite does not pin down. +4. For each survivor, decide: a missing scenario (author one, then re-run that mutant and see it + killed), a view that does not expose the field (see `ess:testing-conformance`, "does a green run + mean anything?"), or a generator gap (an issue on beyond10x/ess). + +**Planted defect:** the audit is its own plant — but confirm the harness first with one mutant you +know a scenario kills. If it survives, the harness is not applying mutants. + +**Found:** 8 of 35 single-rule mutants passed every scenario; 6 were declared behaviour. After 6 +authored scenarios, 1 survived, and it rested on a value the spec did not declare. + +## 2. Random command sequences against a reference model + +**Question:** does the implementation agree with the spec along paths nobody wrote? + +1. Build the [reference model](reference-model.md) over the IR. +2. Drive a seeded random walk: at each step pick an actor, a command and an input (valid, boundary + and invalid), send it to the implementation through the suite's own `Target` interface + (`executeCommand`, `queryView`), and to the model. +3. After every step compare: the outcome, the error, the events and their payloads, every view's rows + and their order, identity uniqueness, and every invariant. +4. On a disagreement, shrink the trace (drop steps while it still disagrees) and report the shortest. +5. Fail the run when a declared outcome is never reached across all seeds. An unreached outcome is + untested, not passed. +6. Where the model cannot say what a view shows because no command sets the field, report it: the + invariant holds only through an undeclared default. + +**Planted defect:** one engine defect the suite passes — a view returning at most 5 rows, or a +`wrong_state` answer that still applies `sets`. The walk must catch it and shrink it. + +**Found:** 300 walks of 80 steps in about 1.3 s caught 4 of 4 planted engine defects that the +127-scenario suite passed, and surfaced 3 invariants that held only through undeclared defaults. + +## 3. Replay of the real caller + +**Question:** does the code that uses the implementation send only what the spec accepts? + +1. Record the caller's command stream: the application's own traffic, from a log, a test run or a + recorded session. Strip any personal data before it leaves the environment it was recorded in. +2. Replay it through the reference model, from the same initial state. +3. Report every command the model refuses or answers differently from what the caller expected, and + every refusal outcome the caller never triggers (a candidate for a caller-side test, or for + deletion from the spec if it is fiction). +4. Check the actor of each command against the actors' grants. + +**Planted defect:** edit one recorded command so it sends a field the spec does not accept, or runs +from a state the transition does not start from. The replay must name it. + +**Found:** 0 disagreements over 25,980 commands; 25 refusal outcomes the caller never triggers. + +## 4. Determinism + +**Question:** is the implementation deterministic where the spec assumes it? + +1. Run technique 2 twice with the same seed and compare the two command streams and every result, + byte for byte. +2. Scan the implementation's source for clock reads, randomness and unordered iteration in code a + command reaches (`Date.now`, `Math.random`, `time.Now`, `rand.`, iteration over a map or set). + Each hit is either injected (a seed, a clock the target controls) or a finding. + +**Planted defect:** add a `Math.random()` (or the language's equivalent) to one outcome. The two +runs must diverge, and the report must name the first divergent command. + +**Found:** passed; the planted `Math.random()` diverged at command 235. + +## 5. Metamorphic relations + +**Question:** do promises that span two runs hold? + +1. Read the relations from the design, not the spec: "a pause changes nothing but the pause + commands", "cancelling and re-creating equals never creating", "the order two independent + commands arrive in does not matter". +2. For each, generate a base trace with technique 2's walker, derive the paired trace (insert the + pause, swap the independent pair), run both, and compare what the relation says must be equal. + +**Planted defect:** make one command succeed while paused. The pause relation must fail on it. + +**Found:** 1 real defect: an action allowed while paused that the design forbids. + +## 6. Exhaustive guard analysis + +**Question:** are guards dead, overlapping, gap-leaving, or weak enough to admit an invariant break? + +1. For each command, collect its outcomes' `when:` guards from the IR. +2. Build each guard's boundary domain from the field types and every literal the guards compare + against: each literal, one either side, the type's extremes, and for an enum every variant. +3. Evaluate every guard of the command over the product of those domains and report: a guard no + input satisfies (dead), an input two guards both satisfy (overlap), an input no outcome answers + (gap), and an input a guard admits that leaves an invariant false after the outcome's `sets`. +4. Optionally cross-check each finding, and each claimed absence, with `z3`. + +**Planted defect:** in a copy of the spec, where one outcome's guard is `>= 0` and its sibling's +is `< 0`, change the sibling to `<= 0`. The analysis must report the overlap at `0`. If +`ess specify validate` already refuses the planted copy, the compiler covers that class: plant a gap +instead (`< 0` → `< -1`) and record which class the compiler caught. + +**Found:** 11 guards, 374 combinations, 0 findings — a result that counted only because the planted +overlap was reported first. + +## 7. Spec diff in the gate + +**Question:** is a change breaking, and was that acknowledged? + +Follow [spec-diff.md](spec-diff.md). + +**Planted defect:** remove one enum variant on a branch. The gate must fail until the change is +acknowledged. + +**Found:** 20 changes since the release tag, all additive. + +## 8. Design review by an agent + +**Question:** what does the design say that the spec omits or contradicts? + +Dispatch one agent with [design-review.md](design-review.md) as its brief. + +**Planted defect:** delete one declared rule from a copy of the spec that the design states. The +review must report it as `missing`, citing both sides. + +**Found:** 26 findings, the most of any technique — among them "one entry per player" and "the +count always goes up", neither declared at all. diff --git a/plugins/ess/skills/init/SKILL.md b/plugins/ess/skills/init/SKILL.md new file mode 100644 index 0000000..4b0dd34 --- /dev/null +++ b/plugins/ess/skills/init/SKILL.md @@ -0,0 +1,77 @@ +--- +name: init +description: Start with ESS (Executable System Specifications) in this project — make sure the `ess` CLI is installed, explain what ESS does, and take the first step of a specification. Use when the user wants to start writing specs, asks what ESS is or how to adopt it, asks to set up or install ESS, points at the ESS repository, or when another ESS skill reports that `ess` is missing. Installs CLIs only after the user confirms the plan. +--- + +# Start with ESS + +ESS is a typed model of a system — entities, commands, events, views and the component that runs +them — that validates, compiles, projects JSON Schema and OpenAPI, and synthesises conformance +suites. The `ess` CLI is the authority; these skills tell you how to drive it. + +## 1. Have the CLI + +Run `ess --version`. If it answers, go to step 2: `b10x:init` just installed it, or it was already there (`ess:upgrade` handles newer releases). If it is missing, install it with `b10x`. +Plan for the host you run in (`--host claude` in Claude Code, `--host codex` in Codex): + +```bash +mkdir -p ~/.local/state/b10x +b10x init ess --host claude --out ~/.local/state/b10x/plan.json +``` + +It prints what it would do, including the install method: the prebuilt, checksummed release archive +by default, or `cargo` (a source build) when the user asks for it and a Rust toolchain is on `PATH`. +When both are possible, say which is planned and offer the other; re-run with `--method cargo` if +they choose it, show the plan, and +after they confirm run `b10x setup apply --plan ~/.local/state/b10x/plan.json --yes`. + +No `b10x`? Follow https://github.com/beyond10x/agentplugins/releases/latest/download/SETUP.md first; its step 1 asks the user before it installs `b10x`. + +## 2. Read what is already here + +Look for `system.yaml` or `ess-inputs.yaml` in the repository. When neither exists, go to step 3. + +When one does, establish the repository's standing before offering any work, with two commands: + +```bash +ess specify validate --path +``` + +then the repository's own conformance command, read from its `AGENTS.md` or its task runner +(`Taskfile.yml`, `Makefile`, `package.json` scripts, the CI job that runs the suite). Run that +command as the repository spells it; do not assemble a runner invocation of your own. + +Report both results as the tools printed them: the validation line (` v — file(s), +valid`, or every refusal verbatim) and the suite's `passed`, `skipped` and `failed` counts with the +command and its exit status. No conformance command found: say so, and name where you looked. + +When the specification is valid and the suite is green, recommend `ess:testing-conformance`: a green +suite nobody has broken is the next question, and that skill's first section is how to ask it. When +validation refuses or a scenario fails, that refusal or failure is the work; route it through step 3. + +## 3. Pick the work + +| the task | skill | agent | +|---|---|---| +| write a new specification, add an entity, validate, compile or project one | `ess:specifying` | `ess:author` | +| give an existing codebase a specification it never had | `ess:retrofitting` | `ess:retrofitter` | +| raise or audit what a conformance suite actually tests | `ess:testing-conformance` | `ess:conformance` | +| harden a specification whose suite is green | `ess:hardening` | — | + +A skill that is not loaded yet in this session prints with `b10x skill ess:`. + +## 4. Rules that hold in every ESS skill + +- The compiler's output is the answer. Relay every refusal verbatim; never edit generated output + around one. +- A draft an agent writes is an import: mark what you could not read from a source with + `UNMAPPED:` and name it in the report. +- Spell a command by its area: `ess specify|generate|verify|infra `. +- Finish with the repository's own gate and report the exact command and exit status. + +## Next + +- New specification: `ess:specifying` — its `references/syntax.md` shows every section in a spec + that validates. +- Existing system: `ess:retrofitting`. +- Later, `ess:upgrade` checks for a newer `ess` and plugin. diff --git a/plugins/ess/skills/retrofitting/SKILL.md b/plugins/ess/skills/retrofitting/SKILL.md new file mode 100644 index 0000000..c62d0f6 --- /dev/null +++ b/plugins/ess/skills/retrofitting/SKILL.md @@ -0,0 +1,107 @@ +--- +name: retrofitting +description: >- + Derive an ESS specification for a system that already exists and has none — from its OpenAPI contract, its observed Kubernetes deployment, or its code — then validate it and hand it to conformance. Use when the user asks to retrofit, adopt, reverse-engineer or "get a spec out of" an existing service or repository, when a codebase has an OpenAPI document or a cluster but no `system.yaml` or `ess-inputs.yaml`, or when a specification must describe behaviour that is already shipped rather than behaviour still to be built. Not for a domain drafted from nothing, which is the `ess:specifying` skill; not for raising coverage of a suite that already exists, which is the `ess:testing-conformance` skill. +--- + +# Retrofitting a specification onto an existing system + +A retrofit describes what the system **does**, not what it should do. Every declaration in the +draft is read from something in the repository — a contract, an observation, a line of code — and +cites it. What you cannot read is an `UNMAPPED:` marker, never a plausible value. + +## 1. Inventory the sources, in order of authority + +| Source | Command | What it gives you | +|---|---|---| +| an OpenAPI document | `aep plan reverse openapi --domain --out ` | a draft `ess/1` domain: entities and fields from the schemas, with the decisions it could not take listed | +| the same OpenAPI document | `ess infra import openapi --path --format yaml` | the operation and interface-type semantics ESS's adapter supports, with its coverage gaps | +| a running Kubernetes deployment | `ess infra import kubernetes …` (see `--help`) | a sanitized observation of what is deployed; the credential edge is explicit | +| code only | read it | handlers, persistence types, state enums, validation — each a citation | + +Use every source that exists. Where two disagree, the draft records both and marks the field +`UNMAPPED:`; the disagreement is a finding, not something to resolve by choosing. + +`ess infra import openapi` reads OpenAPI 3.0 and 3.1. A nullable field (`nullable: true`, or a 3.1 +`type: [T, "null"]`) imports as a coverage gap at its pointer, because the interface has no null: +carry each gap into the draft as an `UNMAPPED:` marker. An object schema must be closed with +`additionalProperties: false`, or the import refuses it; that refusal is a finding about the +contract, so report it rather than editing the contract to pass. + +`aep` is optional. Without it, write the domain by hand from the schemas, in the shape the `ess:specifying` +skill shows. + +## 2. Draft, one domain at a time + +Start from the smallest document that validates (the `ess:specifying` skill has it), then add one entity at +a time, running after each: + +```console +ess specify validate --path +``` + +For each entity, cite where it came from beside it: + +```yaml + - name: billing.invoice.Invoice + # read from: openapi.yaml components.schemas.Invoice; src/invoice/model.rs:14 +``` + +Retrofit-specific rules: + +- **Lifecycles are read, not designed.** Take states from an enum, a status column or the + transitions the handlers perform. A state the code never enters is not declared. A transition + whose trigger you cannot find is `UNMAPPED:`. +- **Commands are the operations that exist.** One command per mutating operation or handler. Do not + add the command the system should have. +- **Relations need an owner decision.** `owns` against `references` is decided by what a delete + does in the code today. No delete path found: `UNMAPPED:`. +- **Views follow reads.** One view per read operation clients use; its fields are the fields the + response carries. An entity with `invariants` also needs a view holding every state, or the suite + refuses the invariant checks (`ess:specifying`, conformance section). That view is structural, + not a read the code has: say so in a comment. It is the one view a retrofit may add. +- **A command the code ignores in some state is a synthesis refusal, and that is correct.** When + the code does nothing (no error) for a command in a state its transition does not start from, + declare no `wrong_state:` outcome and no error. `synthesize` then reports `ESS-SYNTH-012` for + that state and still writes the scenario, which requires that nothing happened: the code's + behaviour. Report the refusals; do not add an error the code never raises. +- **A service that publishes no events still needs one per success outcome.** `validate` refuses an + outcome that neither emits nor names an error (`empty_change`); a view does not count. Declare an + event for the fact the outcome produces (`OrderPaid`), leave it out of the component's + `publishes:` list because the service does not publish it, and say so in a comment and the report. +- **An entity the code gives no status still needs a lifecycle.** The language requires one, so + declare a single state that is both `initial` and `terminal` (`Stocked`), and say in a comment + that it is structural, not read from the code. That is the one state a retrofit may name. +- **Wire values keep their spelling in a comment.** A stored status `in_transit` becomes the state + `InTransit`; state names must start upper-case. +- **A rule the language cannot hold stays in the code, and is named.** A limit read from stored + state, or a constraint across records, is `UNMAPPED:` with its source line + ([syntax reference](../specifying/references/syntax.md), "What `when` can and cannot say"). + +## 3. Prove the draft describes the system + +A validated draft only says the specification is coherent. It says nothing about the system until a +suite runs against it: + +```console +ess verify conform synthesize --path --out +``` + +Then follow the `ess:testing-conformance` skill: build or pick a target that drives the real implementation, +run the suite, and break one behaviour to prove a named scenario goes red. A retrofit is done when +the suite runs against the real system, not when `validate` exits 0. + +## 4. Report + +- the specification path and `ess specify validate` output, verbatim +- every `UNMAPPED:` marker, one line each, with what would settle it +- every place two sources disagreed +- the suite counts (`passed`/`skipped`/`failed`) if a run happened, or that none did yet + +## Agents + +- `retrofitter` — derives a specification for an existing system under this skill. + +## Next + +- The draft validates: extend it with `ess:specifying`; hold the implementation to it with `ess:testing-conformance`. diff --git a/plugins/ess-specify/skills/specify/SKILL.md b/plugins/ess/skills/specifying/SKILL.md similarity index 64% rename from plugins/ess-specify/skills/specify/SKILL.md rename to plugins/ess/skills/specifying/SKILL.md index 9462a0d..6e760c3 100644 --- a/plugins/ess-specify/skills/specify/SKILL.md +++ b/plugins/ess/skills/specifying/SKILL.md @@ -1,5 +1,5 @@ --- -name: specify +name: specifying description: >- Validate Executable System Specifications and guide deterministic JSON Schema or OpenAPI projections through the `ess` command. Use when a repository contains an ESS `system.yaml` or `ess-inputs.yaml`, generated schema or OpenAPI artifacts, an `ess-*` format, when a story or epic introduces an entity — a noun the plan needs a typed home for, whether or not a specification exists yet, including a repository with no `system.yaml` anywhere, where the domain is drafted from nothing — or when the user asks about specification validation, compilation, schema generation, projection drift, adapter coverage, or unsupported semantics. An entity introduction names the noun and something typed about it: an identifier, a field, or a relation to another noun. A story that mentions a noun in passing, an acceptance line, a status change or a plan's prose is not one. --- @@ -78,7 +78,7 @@ $ ess specify validate --path warehouse v1 — 2 file(s), valid ``` -That is the verbatim output of ESS `0.22.1`, exit status 0, over both files as printed—including the +That is the verbatim output of `ess specify validate`, exit status 0, over both files as printed—including the `relations:` block and its second entity. It is a validated starting point, not an illustrative shape. Change the names and semantics to match the repository you actually read, then validate the changed specification again. @@ -90,6 +90,10 @@ takes is refused as `missing_causation` — so a first domain has one state and gains a second state only together with the outcome that moves it. The header's `domains:` list and the declaring source must agree in both directions; either half alone is a refusal. +**Lifecycle state names start with an upper-case letter** (`InTransit`, not `in_transit`); a +lower-case one is refused as `invalid state name identifier`. A source that stores `in_transit` +still maps onto `InTransit` — note the mapping beside the state. Enum variants carry no such rule. + **A relation is an entry, not a convention.** Two entities linked by a field that happens to be named after the other one are not related as far as anything can check; the `relations:` entry is what makes the link a fact a program reads. Three more refusals come with it, and they are the @@ -109,12 +113,28 @@ narrows that question without answering it. Where you cannot say which kind it i `UNMAPPED:` case below and not a coin toss. Grow it from there — types, commands, events, views, and a component that owns the domain — running -`ess specify validate` after each addition rather than at the end. +`ess specify validate` after each addition rather than at the end. [references/syntax.md](references/syntax.md) +shows every one of those sections in a small specification that validates; read it before writing +the first command. + +`--path` takes one ESS file or a directory. Without an `ess-inputs.yaml`, a directory is read as +every YAML file below it — so generated output written inside it is read back as specification and +refused (`unknown field openapi`). **Add an `ess-inputs.yaml` as soon as anything is generated**, and +write generated output outside the specification's own files: -`--path` takes one ESS file or a directory. A directory is read through its `ess-inputs.yaml` when -one exists — an exact file list, which is what lets authored inputs sit in nested directories beside -generated files (ESS 0.21.0) — and otherwise through the `system.yaml` layout above, which the CLI -calls legacy and still accepts. +```yaml +# ess-inputs.yaml, beside system.yaml +format: ess-inputs/1 +specification: + - system.yaml + - components.yaml + - domains/rooms.yaml +scenarios: [] +``` + +`specification` lists every authored file, relative to this file, and each must exist. `scenarios` +lists authored scenario files, `[]` when there are none. `--path ` then reads exactly +that list. A directory with only `system.yaml` still validates, read whole. **A draft is a proposal, never a silent completion.** Every relation you could not read from code, an OpenAPI document or an existing artifact is written with an `UNMAPPED:` marker beside the place @@ -122,7 +142,7 @@ it would go, and named again in the report: ```yaml # UNMAPPED: the epic says a shipment has a carrier; no code, contract or - # artifact here says what a carrier is. Ask before typing it. + # artifact here says what a carrier is. Open question for the owner. ``` **A relation whose cardinality or ownership you cannot read is `UNMAPPED:`, not an entry with a @@ -133,12 +153,23 @@ where the entry would go: ```yaml # UNMAPPED: a shipment has lines, and nothing here says whether deleting the - # shipment deletes them (owns) or orphans them (references). Ask. + # shipment deletes them (owns) or orphans them (references). Open question. ``` -Imports never guess, and a domain an agent drafts is an import. Never invent a field type, a -lifecycle state or an edge to make a document validate — leave the marker, and name what would -settle it. +Imports never guess, and a domain an agent drafts is an import. Never invent an entity, a field +type, a lifecycle state or an edge to make a document validate — leave the marker, and name what +would settle it. An entity no source names, added so that a command type-checks, is an invention +even when it validates. + +A marker records the question; it does not ask it. In an interactive session, put the open markers +to the user as questions at the end (Claude Code: `AskUserQuestion`), one per marker, and write the +answers into the specification. Headless, or as the `author` agent, leave the markers and list them +in the report. + +**A second source document goes into one of three places.** The same system, as another domain, +when the document says it is part of that system (a companion note, shared requirement ids). A +separate system when it describes its own deployable service. An external boundary when the system +only calls it. When the document does not say which, it is an open question: ask, or mark it. ## Read before changing @@ -149,6 +180,9 @@ ess specify validate --path ess specify compile --path --format json ``` +A headless run (`claude -p`) needs permission to run the CLI, or it cannot validate anything: +`--allowedTools "Bash(ess:*)"`. + `validate` and `compile` accumulate diagnostics. Relay every refusal; do not stop at the first or edit generated output around it. @@ -157,10 +191,19 @@ edit generated output around it. Use the projection command, never a handwritten parallel generator: ```console -ess generate --path --kind schema --out -ess generate project openapi --path --out +ess generate --path --kind --out ``` +`--kind` is `schema` (JSON Schema), `openapi`, `asyncapi`, `docs` (Markdown pages), `site` (the +same pages as HTML) or `docs-ir`. Give each kind its own `--out`, side by side (`out/schema`, +`out/openapi`, …): an output directory inside another is refused. `schema`, `openapi`, `asyncapi` +and `docs` add a folder named after the kind below `--out` (`out/schema/schema/…`); `site` writes +into `--out` itself. Type libraries come from `ess generate types --target rust|go|typescript`. + +**OpenAPI and AsyncAPI need a component** that owns the domain and says how it is reached +(`components.yaml`, `reached_by`). Without one, `openapi` writes `0 artifact(s)` and prints no +refusal: that is a missing declaration, not a clean result. + The same typed IR must produce the same ordered files and bytes. Compare a regenerated temporary tree with the committed tree before replacing anything. A stale committed file is drift; a file no projection owns is not authority. @@ -168,7 +211,7 @@ projection owns is not authority. ## Conformance is a record, not a claim In the planning store an `executable-system-specification` is `conforming` because a suite ran and -its report says so, never because somebody moved it there (AEP 0.50.0). The report is what crosses +its report says so, never because somebody moved it there. The report is what crosses from ESS to AEP: ```console @@ -179,10 +222,16 @@ aep plan artifact evidence --from `synthesize` writes the suite the specification obliges — nothing for a domain with no commands or outcomes, so the two-entity draft above yields `0 scenario(s)` and no report worth recording. +An entity `invariant` is checked after every outcome that leaves the entity in some state, which +needs a view that holds instances in that state and publishes the fields the invariant reads. +Without one, `synthesize` refuses the check (`no view of holds an instance in `, or +` is published by no view of the entity`); a view over the entity with no `filter` that +publishes those fields covers every state (the `Copies` view in the syntax reference). `--target go` or `--target typescript` writes the suite as a test +package your implementation runs; `ess:testing-conformance` says what to run it against. `run` holds a built-in reference implementation to it; `--target` lists the ones this binary carries. `evidence --from` reads the kind, the source and the instant out of the report and refuses -a report of no scenarios or with no `spec_digest`. A report/2 (`--report-format 2`, ESS 0.20.0) is -recorded only beside `--suite ` (AEP 0.55.0). Then +a report of no scenarios or with no `spec_digest`. A report/2 (`--report-format 2`) is +recorded only beside `--suite `. Then `aep plan artifact set --model-digest ` ties the record to the model the suite ran against, and `aep plan artifact move --to conforming` is decided by the store. @@ -205,4 +254,19 @@ the persisted envelope changes. Adding internal Rust capabilities is not enough. strict v1 reader such as `infra-ir/1`, add an old-reader compatibility test; unknown fields are currently refused. -Finish by running the repository's full gate and report the exact command and exit status. +Finish by running the repository's full gate, where it has one, and report the exact command and +exit status. Where it has none, `ess specify validate` plus `ess verify conform synthesize` with 0 +refusals is the check; report both outputs. + +## Agents + +- `author` — writes or extends a specification under this skill and reports every `UNMAPPED:` marker. + +## Next + +- First, review: every `UNMAPPED:` marker and every contradiction between sources goes to the owner + and is resolved in the specification. A specification validates with markers still in it, so + `validate` alone is not the review. Then plan the work around it with `aep:planning`. +- The specification validates and is reviewed: project it (`ess generate`) or synthesise a conformance suite, then `ess:testing-conformance`. +- A system already exists and has no specification: `ess:retrofitting`. +- `ess` missing or older than expected: `ess:upgrade`. diff --git a/plugins/ess/skills/specifying/references/syntax.md b/plugins/ess/skills/specifying/references/syntax.md new file mode 100644 index 0000000..90dd86e --- /dev/null +++ b/plugins/ess/skills/specifying/references/syntax.md @@ -0,0 +1,358 @@ +# ESS syntax by example + +A small lending library in three files, with every section a specification usually needs. It +validates as written (`ess specify validate --path ` → `library v1 — 3 file(s), valid`). +Copy the shape, not the domain: name your own entities, commands and events after your system. + +## `system.yaml` + +```yaml +format: ess/1 +system: library +version: v1 + +domains: + - library.lending +``` + +## `components.yaml` + +Who runs the domain, which commands it accepts, which events it publishes, how it is reached. + +```yaml +components: + - component: lending-service + summary: Holds every branch and the copies it lends. + owns: + domains: + - library.lending + accepts: + commands: + - library.lending.OpenBranch + - library.lending.AddCopy + - library.lending.LendCopy + - library.lending.ReturnCopy + publishes: + events: + - library.lending.BranchOpened + - library.lending.CopyAdded + - library.lending.CopyLent + - library.lending.CopyReturned + reached_by: network +``` + +## `domains/lending.yaml` + +```yaml +domain: library.lending + +summary: Branches that own copies of books and lend them out. + +naming: + wire: lending + display: Lending + +# Types: `newtype` over a primitive, `enum`, `struct` (with `fields`, optional `invariants`), +# `union` (tagged: `tag:` plus `variants:` name → type). Primitives include Uuid, String, Integer, +# Decimal, Boolean, Bytes, Timestamp, Duration; wrappers Optional, List, Map. +types: + - name: library.lending.BranchId + kind: newtype + of: Uuid + + - name: library.lending.CopyId + kind: newtype + of: Uuid + + - name: library.lending.Title + kind: newtype + of: String + + - name: library.lending.Format + kind: enum + variants: [Hardcover, Paperback, Audio] + +# Entities: an identity, fields, relations to other entities, and a lifecycle whose transitions +# commands move. `owns` means the target cannot outlive this entity; `references` means it can. +entities: + - name: library.lending.Branch + identity: + name: branch_id + type: library.lending.BranchId + fields: + - name: name + type: String + relations: + - name: copies + kind: owns + target: library.lending.Copy + cardinality: many + via: branch_id + lifecycle: + initial: Open + states: [Open] + terminal: [Open] + + # A lifecycle may have no final state: a copy goes out and comes back for as long as it exists, + # so `terminal` is empty. + - name: library.lending.Copy + identity: + name: copy_id + type: library.lending.CopyId + fields: + - name: branch_id + type: library.lending.BranchId + - name: title + type: library.lending.Title + - name: format + type: library.lending.Format + - name: pages + type: Integer + invariants: + - pages > 0 + lifecycle: + initial: Available + states: [Available, OnLoan] + terminal: [] + transitions: + - name: lend + from: [Available] + to: OnLoan + - name: return + from: [OnLoan] + to: Available + +# Actors: who may invoke which commands. +actors: + - name: library.lending.Librarian + may: + - library.lending.OpenBranch + - library.lending.AddCopy + - library.lending.LendCopy + - library.lending.ReturnCopy + naming: + display: Librarian + +# Errors a command outcome can return. +errors: + - name: library.lending.InvalidPageCount + summary: The page count is not a positive number. + fields: + - name: submitted + type: Integer + + - name: library.lending.CopyStateConflict + summary: The copy is not in a state this command acts from, so nothing moved. + fields: + - name: state + type: library.lending.Copy.State + +# Commands: input, then outcomes. An outcome `creates` an entity or `moves` one through a +# transition, `emits` events with a `payload` built from `input.`, or returns an `error`. +commands: + - name: library.lending.OpenBranch + naming: + wire: open-branch + display: Open a branch + input: + - name: name + type: String + outcomes: + - name: opened + creates: library.lending.Branch + instance: branch_id + emits: + - library.lending.BranchOpened + payload: + library.lending.BranchOpened: + name: input.name + summary: The branch exists and holds no copies. + + # Two outcomes of one command: all but one need a `when` over the command's input, or the result + # is not determined by the input and `validate` refuses it as `conflicting_declaration`. Error + # outcomes count; `wrong_state: true` outcomes do not (LendCopy below has two without a `when`). + - name: library.lending.AddCopy + naming: + wire: add-copy + display: Add a copy + input: + - name: branch_id + type: library.lending.BranchId + - name: title + type: library.lending.Title + - name: format + type: library.lending.Format + - name: pages + type: Integer + outcomes: + - name: added + when: pages > 0 + creates: library.lending.Copy + instance: copy_id + emits: + - library.lending.CopyAdded + payload: + library.lending.CopyAdded: + branch_id: input.branch_id + title: input.title + summary: The copy is on the shelf and Available. + + - name: refused + error: library.lending.InvalidPageCount + summary: The page count was not positive, and nothing was added. + + # A command that moves an entity: `wrong_state: true` answers from every state the transition does + # not start from, so the `from:` list lives in one place. + - name: library.lending.LendCopy + naming: + wire: lend-copy + display: Lend a copy + input: + - name: copy_id + type: library.lending.CopyId + outcomes: + - name: lent + moves: library.lending.Copy.lend + instance: copy_id + emits: + - library.lending.CopyLent + payload: + library.lending.CopyLent: + copy_id: input.copy_id + summary: The copy is out on loan. + + - name: wrong-state + wrong_state: true + error: library.lending.CopyStateConflict + summary: The copy is not Available, so nothing was lent. + + - name: library.lending.ReturnCopy + naming: + wire: return-copy + display: Return a copy + input: + - name: copy_id + type: library.lending.CopyId + outcomes: + - name: returned + moves: library.lending.Copy.return + instance: copy_id + emits: + - library.lending.CopyReturned + payload: + library.lending.CopyReturned: + copy_id: input.copy_id + summary: The copy is back on the shelf. + + - name: wrong-state + wrong_state: true + error: library.lending.CopyStateConflict + summary: The copy is not on loan, so nothing was returned. + +events: + - name: library.lending.BranchOpened + fields: + - name: branch_id + type: library.lending.BranchId + - name: name + type: String + + - name: library.lending.CopyAdded + fields: + - name: copy_id + type: library.lending.CopyId + - name: branch_id + type: library.lending.BranchId + - name: title + type: library.lending.Title + + - name: library.lending.CopyLent + fields: + - name: copy_id + type: library.lending.CopyId + + - name: library.lending.CopyReturned + fields: + - name: copy_id + type: library.lending.CopyId + +# Views: read models over an entity. `read_your_writes` or `eventual`; an optional `filter`. +views: + - name: library.lending.AvailableCopies + source: library.lending.Copy + consistency: read_your_writes + filter: state == Available + fields: + - name: copy_id + type: library.lending.CopyId + - name: title + type: library.lending.Title + naming: + wire: available + display: Available copies + + # An invariant is checked after every outcome, through a view that holds the entity in the + # resulting state and publishes the fields the invariant reads. With no `filter`, this one + # holds every state; without it, `ess verify conform synthesize` refuses the `pages > 0` checks. + - name: library.lending.Copies + source: library.lending.Copy + consistency: read_your_writes + fields: + - name: copy_id + type: library.lending.CopyId + - name: pages + type: Integer + naming: + wire: copies + display: Copies +``` + +State names start with an upper-case letter (`OnLoan`); `validate` refuses `on_loan`. Enum +variants may be written as the source spells them. + +## What `when` can and cannot say + +A predicate (`when`, `invariants`, a view's `filter`) has a compact form and a structured form, and +`validate` accepts both in all three places. The full grammar is ESS's +[predicate reference](https://beyond10x.github.io/docs/ess/reference/predicates). + +The compact form is one string: a comparison (`pages > 0`, `format == Hardcover`), a bare fact path +(present and truthy), `defined(note)`, or `not` before any of those. `&&`, `||` and `in [...]` +inside that string are refused; write a conjunction, a disjunction or a set in the structured form. + +| to say | structured form | +|---|---| +| A or B ("state A or B" is one view, not two) | `filter: {any: [state == Available, state == OnLoan]}` | +| A and B | `when: {all: [pages > 0, format == Hardcover]}`; a bare list is an implicit `all` | +| not A, none of A and B | `when: {not: pages <= 0}`, `when: {none: [format == Audio, pages > 900]}` | +| a range, by operator | `when: {pages: {gte: 1, lte: 5000}}` — `eq`, `ne`, `lt`, `lte`, `gt`, `gte` | +| one of a set | `when: {format: [Hardcover, Paperback]}`, or `{in: […]}`, `{any_of: […]}`, `{one_of: […]}` | +| none of a set | `when: {format: {none_of: [Audio]}}`, or `{not_in: […]}` | +| present or absent | `when: {note: {exists: false}}` (= `not defined(note)`), `{defined: true}`, `{truthy: true}` | +| every or some element of a list | `forall: {in: tags, as: t, that: t != ""}`, `exists: {in: tags, as: t, that: …}` — no compact form | +| a list's length | `tags.count >= 0` | + +`validate` accepting a form does not mean `ess verify conform synthesize` can witness it. Two +refusals to expect: an invariant over a list field (`tags.count`, a `forall` over `tags`) validates +and is then refused as reading what no view publishes, even with the field in a view; and a presence +guard over a required input (`{defined: true}`, `{exists: false}`) is refused because no candidate +input leaves the field out. Relay such a refusal verbatim rather than reshaping the rule around it. + +A `when` reads only the command's own input. A condition on another entity — "the branch must be +open", "the customer is active" — is not an input guard. Express it as a transition of that entity (a +command that `moves` it, answered by `wrong_state` from states it does not start from), or leave an +`UNMAPPED:` marker naming the rule and report it; never invent an outcome the compiler cannot decide. + +Two cases trials hit: + +| rule | how to write it | +|---|---| +| a value stored on the entity decides the outcome ("express parcels over 20 kg are refused at dispatch", with the weight given at create) | not expressible: a `when` sees only the dispatch input. Mark it `UNMAPPED:` at the outcome, citing the source line. A guard over stored fields is proposed in ESS's [design note](https://github.com/beyond10x/ess/blob/main/docs/design/cross-record-and-stored-field-guards.md) | +| two records must not overlap ("a room cannot be booked twice for one hour") | make the contested unit an entity with its own lifecycle (a `Slot` that is `Free` or `Booked`); a second booking is then `wrong_state` on that slot. Overlap between arbitrary time ranges is not expressible; mark it `UNMAPPED:` | + +**Compare two fields through one struct.** A right-hand side without a dot is a literal, so +`ends_at > starts_at` compares `ends_at` with the text `"starts_at"` and `validate` refuses it. +Declare the two fields in one struct (`Window` with `starts_at` and `ends_at`, both `Timestamp`), +take it as input, and guard on `window.ends_at > window.starts_at`; `synthesize` orders the two +instants. `Duration` does not compare with a number (it is text), so a length is an `Integer` in a +named unit (`duration_minutes > 0`). diff --git a/plugins/ess-specify/skills/coverage/SKILL.md b/plugins/ess/skills/testing-conformance/SKILL.md similarity index 74% rename from plugins/ess-specify/skills/coverage/SKILL.md rename to plugins/ess/skills/testing-conformance/SKILL.md index bc7be22..4c434b4 100644 --- a/plugins/ess-specify/skills/coverage/SKILL.md +++ b/plugins/ess/skills/testing-conformance/SKILL.md @@ -1,7 +1,7 @@ --- -name: coverage +name: testing-conformance description: >- - Raise and audit the coverage of an ESS conformance suite against a real implementation, and judge whether a passing suite is testing anything. Use when a repository holds a synthesized conformance suite — an `ess-conformance/*` document, a generated conformance runner, a target that answers `ExecuteCommand`/`QueryView`, or a CI job that runs one — and the task is to raise the number of scenarios that execute, work out why scenarios are skipped, decide whether a green suite would catch a regression, build or fix a test double for an upstream plane, or set the gate that holds a suite's counts. Also use when a conformance job fails in CI and not locally. Not for authoring or validating the specification itself, which is the `specify` skill; not for a unit or integration suite that no ESS document obliges. + Raise and audit the coverage of an ESS conformance suite against a real implementation, and judge whether a passing suite is testing anything. Use when a repository holds a synthesized conformance suite — an `ess-conformance/*` document, a generated conformance runner, a target that answers `ExecuteCommand`/`QueryView`, or a CI job that runs one — and the task is to raise the number of scenarios that execute, work out why scenarios are skipped, decide whether a green suite would catch a regression, build or fix a test double for an upstream plane, or set the gate that holds a suite's counts. Also use when a conformance job fails in CI and not locally. Not for authoring or validating the specification itself, which is the `ess:specifying` skill; not for a unit or integration suite that no ESS document obliges. --- # ESS conformance coverage @@ -40,6 +40,10 @@ The mechanism generalises, so learn to spot it by reading rather than by mutatin Do the mutation test once per mapping you rely on, and record in the commit that you did it and what failed. A scenario nobody has ever seen fail is a scenario that has never been tested. +Mutation is one of eight hardening techniques. Once the suite is green, `ess:hardening` carries the +rest — random command sequences against a reference model, caller replay, determinism, metamorphic +relations, guard analysis, a spec diff in the gate and a design review — and the order to run them in. + ## Ranking the skips **Never rank skips by construct.** Counting which ESS constructs appear in skipped scenarios and @@ -68,9 +72,51 @@ func recordRefusal(subject string, err error) { } ``` -Wire it with a named return and a `defer` so no call site changes. Verify the counts are identical -with it on and off — an instrument that perturbs what it measures is not one. Then group the reasons -and attack the largest block whose cause is not a missing ESS construct. +Wire it with a named return and a `defer` so no call site changes. + +A `--target typescript` suite reads a refusal from the `cause` chain: the generated package's +`README.md` says a target throws `ErrUnsupported`, or `new Error("...", { cause: ErrUnsupported })` +so the reason can carry which command it was. The recorder is the same shape, wrapped around +`executeCommand`, `queryView` and `establishEntity` (the entity setup, on a target that implements +`EntitySetupTarget`): + +```ts +import { appendFileSync } from "node:fs"; +import { ErrUnsupported } from "./index.js"; + +function recordRefusal(subject: string, err: unknown): void { + if (!(err instanceof Error) || err.cause !== ErrUnsupported) { + return; + } + const path = process.env.CONFORMANCE_REFUSALS; + if (!path) { + return; + } + appendFileSync(path, `${subject}\t${err.message}\n`); +} + +async function recorded(subject: string, call: () => T | Promise): Promise { + try { + return await call(); + } catch (err) { + recordRefusal(subject, err); + throw err; + } +} + +// inside the target: +// executeCommand(request) { return recorded(request.command, () => this.execute(request)); } +// queryView(request) { return recorded(request.view, () => this.query(request)); } +// establishEntity(request) { return recorded(request.entity, () => this.establish(request)); } +``` + +Rethrow unchanged, so the runner still reads the sentinel and the verdict does not move. A bare +`throw ErrUnsupported` carries no reason and is not recorded; that is the generic refusal described +below. + +Either way, verify the counts are identical with the recorder on and off — an instrument that +perturbs what it measures is not one. Then group the reasons and attack the largest block whose +cause is not a missing ESS construct. Refusal reasons are worth writing well for this reason alone: a target that refuses with a generic message destroys its own diagnosis. Every refusal should name what it could not spell. @@ -193,3 +239,42 @@ most often assumed: Three consecutive runs with identical counts, and one run of the job's own script in the job's own image, before claiming a number. + +## Which target runs your suite + +`ess verify conform run --target` names a target built into the `ess` binary. None of them runs +your implementation: + +| `--target` | what it is | on your own domain | +|---|---|---| +| `interpreted` | a placeholder for running the specification itself; it decides nothing yet | every scenario `unsupported`, run `failed` | +| `oracle-fixture` | a hand-written implementation of ESS's own `examples/oracle-fixture` | every scenario `error` | +| `billing` | a hand-written implementation of ESS's own `examples/billing` | every scenario `error` | + +(Counts observed with a 21-scenario suite for a new domain.) They say nothing about your system, so +never record such a report as evidence. + +To hold your implementation to the suite, generate it as a test package in the implementation's +language and implement the package's `Target` interface over your service: + +```console +ess verify conform synthesize --path --target go --out +ess verify conform synthesize --path --target typescript --out +``` + +Generate the Go package **inside the implementation's own module** (`--out /conformance`) so +its test can import it; a copy outside the module has no `go.mod` to import from. Numbers in a +payload you return compare equal only as `float64`: an `int64` or a `json.Number` fails although +the type check accepts it (beyond10x/ess#101, until fixed). The package's `README.md` lists the +methods and the wiring test. A method you cannot answer returns +`ErrUnsupported` (a skip), never a made-up result. Keep a suite for a domain nobody has implemented; +run it once something answers it. + +## Agents + +- `conformance` — raises or audits a suite's coverage under this skill. + +## Next + +- A scenario reveals a gap in the specification: `ess:specifying`. +- The suite is green and the question is what it still misses: `ess:hardening`. diff --git a/plugins/ess/skills/upgrade/SKILL.md b/plugins/ess/skills/upgrade/SKILL.md new file mode 100644 index 0000000..fbbdd32 --- /dev/null +++ b/plugins/ess/skills/upgrade/SKILL.md @@ -0,0 +1,26 @@ +--- +name: upgrade +description: Check whether the ESS plugin and the `ess` CLI are current, and upgrade them with the user's confirmation. Use when the user asks whether ESS is up to date or to upgrade or update it, when a session-start line starting with `b10x:` names `ess`, or when an `ess` command behaves differently from what an ESS skill describes. +--- + +# Upgrade ESS + +```bash +b10x upgrade ess --host claude --out ~/.local/state/b10x/plan.json +``` + +Use `--host codex` in Codex. It compares the installed `ess` plugin with what the marketplace serves and the `ess` on `PATH` with +the newest ESS release, and prints each difference with the action that fixes it. Nothing is +changed yet. + +- Nothing to change: say "ESS is current" with both versions, and stop. +- Otherwise show the actions in one list and ask once. After a clear yes: + `b10x setup apply --plan ~/.local/state/b10x/plan.json --yes`. +- A new plugin version loads in a new session; until then `b10x skill ess:` prints the new text. + +No `b10x`? Follow https://github.com/beyond10x/agentplugins/releases/latest/download/SETUP.md first. + +## Next + +- Continue the work that prompted the check: `ess:specifying`, `ess:retrofitting` or + `ess:testing-conformance`. diff --git a/plugins/workspace-hygiene/.claude-plugin/plugin.json b/plugins/workspace-hygiene/.claude-plugin/plugin.json deleted file mode 100644 index d42470d..0000000 --- a/plugins/workspace-hygiene/.claude-plugin/plugin.json +++ /dev/null @@ -1,10 +0,0 @@ -{ - "name": "workspace-hygiene", - "displayName": "Workspace Hygiene", - "description": "Use the public worktree toolchain for safe, recoverable agent checkout lifecycle.", - "version": "0.9.2", - "author": { "name": "Beyond10x" }, - "license": "Apache-2.0", - "repository": "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/beyond10x/agentplugins", - "keywords": ["beyond10x", "git", "worktree", "agents", "workspace-hygiene"] -} diff --git a/plugins/workspace-hygiene/skills/worktree/agents/openai.yaml b/plugins/workspace-hygiene/skills/worktree/agents/openai.yaml deleted file mode 100644 index f29d05f..0000000 --- a/plugins/workspace-hygiene/skills/worktree/agents/openai.yaml +++ /dev/null @@ -1,5 +0,0 @@ -# Generated by `worktree skill`; edit the generator, not this file. -interface: - display_name: "Worktree" - short_description: "Safely manage isolated Git worktree lifecycles" - default_prompt: "Use $worktree and its CLI to create or maintain an isolated checkout, preserve recovery evidence, reconcile lifecycle state, and clean it up safely." diff --git a/plugins/worktree/.claude-plugin/plugin.json b/plugins/worktree/.claude-plugin/plugin.json new file mode 100644 index 0000000..737c00e --- /dev/null +++ b/plugins/worktree/.claude-plugin/plugin.json @@ -0,0 +1,17 @@ +{ + "name": "worktree", + "displayName": "Worktree", + "description": "Create, lease, finish, audit and safely clean isolated Git worktrees through the worktree CLI.", + "version": "0.14.13", + "author": { + "name": "Beyond10x" + }, + "license": "Apache-2.0", + "repository": "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/beyond10x/agentplugins", + "keywords": [ + "beyond10x", + "git", + "worktree", + "agents" + ] +} diff --git a/plugins/workspace-hygiene/.codex-plugin/plugin.json b/plugins/worktree/.codex-plugin/plugin.json similarity index 73% rename from plugins/workspace-hygiene/.codex-plugin/plugin.json rename to plugins/worktree/.codex-plugin/plugin.json index 299e88c..c74b1ed 100644 --- a/plugins/workspace-hygiene/.codex-plugin/plugin.json +++ b/plugins/worktree/.codex-plugin/plugin.json @@ -1,13 +1,13 @@ { - "name": "workspace-hygiene", - "version": "0.9.2", - "description": "Use the public worktree toolchain for safe, recoverable agent checkout lifecycle.", + "name": "worktree", + "version": "0.14.13", + "description": "Create, lease, finish, audit and safely clean isolated Git worktrees through the worktree CLI.", "author": { "name": "Beyond10x" }, "skills": "./skills/", "interface": { - "displayName": "Workspace Hygiene", + "displayName": "Worktree", "shortDescription": "Keep agent worktrees isolated and recoverable.", "longDescription": "Create, lease, finish, audit, and safely clean Git worktrees through the versioned public worktree CLI instead of ad hoc host conventions.", "developerName": "Beyond10x", diff --git a/plugins/worktree/skills/init/SKILL.md b/plugins/worktree/skills/init/SKILL.md new file mode 100644 index 0000000..0a98bcf --- /dev/null +++ b/plugins/worktree/skills/init/SKILL.md @@ -0,0 +1,61 @@ +--- +name: init +description: Start with worktree in this project — make sure the `worktree` CLI is available and take the first step. worktree is isolated Git worktrees with leases, recovery proof and safe cleanup. Use when the user wants isolated checkouts for agent work, asks to set up or install worktree, or when `worktree:managing-worktrees` reports that `worktree` is missing. Installs CLIs only after the user confirms the plan. +--- + +# Start with worktree + +## 1. Have the CLI + +Run `worktree --version`. If it answers, go to step 2: `b10x:init` just installed it, or it was already there (`worktree:upgrade` handles newer releases). If it is missing, install it with `b10x`. +Plan for the host you run in (`--host claude` in Claude Code, `--host codex` in Codex): + +```bash +mkdir -p ~/.local/state/b10x +b10x init worktree --host claude --out ~/.local/state/b10x/plan.json +``` + +It prints what it would do, including the install method: the prebuilt, checksummed release archive +by default, or `cargo` (a source build) when the user asks for it and a Rust toolchain is on `PATH`. +When both are possible, say which is planned and offer the other; re-run with `--method cargo` if +they choose it, show the plan, and +after they confirm run `b10x setup apply --plan ~/.local/state/b10x/plan.json --yes`. + +No `b10x`? Follow https://github.com/beyond10x/agentplugins/releases/latest/download/SETUP.md first; its step 1 asks the user before it installs `b10x`. + +## 2. First step here + +Activate a workspace profile once, then check the setup. The profile is a small template; the +worktree repository publishes a default one: + +```bash +curl -fsSL --create-dirs -o ~/.config/worktree/default.toml \ + https://raw.githubusercontent.com/beyond10x/worktree/main/profiles/default.toml +worktree activate --profile ~/.config/worktree/default.toml --workspace +worktree doctor --check +``` + +`activate` copies the profile into `~/.config/worktree/config.toml`, so the template file may stay +where it is or be committed into a repository the user shares with others; `--help` calls it a +"committed" template, but any local file works. Ask the user for the workspace root (an absolute +path to the directory that holds their repositories) rather than guessing it. With nobody to ask +(a headless run), use the directory that holds the current repository and say so in the report. + +`worktree doctor --check` fails with `no active profile` until `activate` has run. + +**A repository needs a remote before its trees can be cleaned up.** Cleanup proves that each +commit reached a remote, so in a repository with no remote (`git remote` prints nothing) `create` +works and every later `gc` refuses. Tell the user before they start work there. + +## 3. Pick the work + +| the task | skill | +|---|---| +| create, lease, finish, inspect or clean a worktree | `worktree:managing-worktrees` | + +A skill that is not loaded yet in this session prints with `b10x skill worktree:`. + +## Next + +- Start work in an isolated checkout: `worktree:managing-worktrees`. +- Later, `worktree:upgrade` checks for a newer plugin and CLI. diff --git a/plugins/workspace-hygiene/skills/worktree/SKILL.md b/plugins/worktree/skills/managing-worktrees/SKILL.md similarity index 77% rename from plugins/workspace-hygiene/skills/worktree/SKILL.md rename to plugins/worktree/skills/managing-worktrees/SKILL.md index 296e119..84ec6bc 100644 --- a/plugins/workspace-hygiene/skills/worktree/SKILL.md +++ b/plugins/worktree/skills/managing-worktrees/SKILL.md @@ -1,18 +1,16 @@ --- -name: worktree +name: managing-worktrees description: Safely create, inspect, finish, reconcile, recover, and garbage-collect managed Git worktrees. Use whenever an agent needs an isolated checkout for repository changes, must hand off a worktree, or needs to audit, recover, or clean linked worktrees. --- -# Worktree - - +# Managing worktrees Use the `worktree` CLI as the sole owner of linked-worktree lifecycle. It keeps trees outside primary checkout collections and refuses cleanup without current recovery proof. ## Start repository work -1. Invoke `$worktree`, then from a primary checkout run `worktree create --purpose `. Add `--repo `, `--base `, or `--id ` when needed. -2. Treat the printed path as the task checkout and do all changes there. +1. From a primary checkout run `worktree create --purpose `. Add `--repo `, `--base `, or `--id ` when needed. +2. Treat the printed path as the task checkout and do all changes there. The tree starts on a detached HEAD; run `git switch -c ` in it before the first commit, so the work has a branch to push. 3. If already inside a managed tree, reuse it; do not nest another worktree. 4. For automation, add `--json` and consume the versioned output. @@ -32,9 +30,9 @@ After verification, preserve the small logs, reports, or deliverables needed for ## Finish and clean up -1. Commit and publish every wanted change. A local-only commit is deliberately not cleanup-safe. +1. Commit and publish every wanted change. A local-only commit is deliberately not cleanup-safe. Work merged as rebased or cherry-picked copies also qualifies when an advertised ref carries every unique commit's exact patch; GC reports that proof as `patch-equivalent`. 2. Preserve required evidence and remove this task's disposable output as described above. Release your own lease, then run `worktree finish `. It refuses dirty, locked, unmanaged, live, or mid-operation Git worktrees. -3. Run `worktree gc --repo --dry-run --id ` and inspect every result. Without exact ids, `--repo` selects the activated workspace profile, not just the repository: the assessment covers records under that profile's `workspace_root`, including other repositories. +3. Run `worktree gc --repo --dry-run --id ` and inspect every result. Always pass `--id`. Without exact ids, `--repo` selects the activated workspace profile, not just the repository: the assessment covers records under that profile's `workspace_root`, including other repositories. 4. Run `worktree gc --repo --apply --id ` with repeated `--id` values only for the exact results intended for removal. The command refreshes remote advertisements, fetches required objects, and revalidates immediately before non-forced removal. Check the result before reporting storage reclaimed. 5. End with either verified cleanup or an explicit handoff: tree id and path, published branch/commit, related work-item references, retained evidence, remaining blockers, next owner and next action. Never leave a tree silently active or label work complete merely from its age or Git state. @@ -48,8 +46,13 @@ After verification, preserve the small logs, reports, or deliverables needed for - If that dry-run explicitly proposes `retire-external`, confirm that destructive action separately by adding `--allow-external-retirement`; never add it for an unrelated migration or missing-record repair. - A finished external legacy tree may supersede a stale migration intent only when the dry-run itself proposes `retire-external`; never reinterpret or bypass a cross-device or ambiguous-relocation refusal. - If removal is interrupted while the path still exists, rerun GC dry-run and exact-id apply. If the path is already absent, use reconciliation dry-run and exact-id apply; its durable removal intent can safely finish the recorded transition. -- A missing Active record without matching durable removal intent must remain refused. Preserve and investigate its registry evidence; never manually tombstone it, delete related state, or fabricate recovery proof. +- A missing Active record without matching durable removal intent stays refused while its work may still exist. Preserve and investigate its registry evidence; never edit the registry by hand, delete related state, or fabricate recovery proof. If its recorded commit still exists anywhere, publish it and rerun the dry-run. +- Only once you have established that such a record's recorded commit is gone for good, abandon it with `worktree reconcile --repo --apply --id --acknowledge-unrecoverable `. That acknowledgement asserts one exact commit named by the immediately preceding dry-run; the command still checks it and refuses while any local branch, tag, remote-tracking ref, or remote advertisement contains it. It deletes nothing from disk or from Git, and records the tombstone with no recovery proof, because there is none to record. - Run `worktree doctor --check` for prerequisites and configuration. - Only after a human explicitly decides an existing linked tree should become manager-owned, run `worktree repo adopt --repo --path --id --purpose `. Then review `reconcile --dry-run` and use exact-id apply only if migration is intended. Never run `git worktree remove --force`, recursively delete a linked tree, place managed trees below the primary workspace, or clean up a tree merely because it looks old. + +## Next + +- `worktree` missing or behind: `worktree:init`, later `worktree:upgrade`. diff --git a/plugins/worktree/skills/managing-worktrees/agents/openai.yaml b/plugins/worktree/skills/managing-worktrees/agents/openai.yaml new file mode 100644 index 0000000..f2f13ea --- /dev/null +++ b/plugins/worktree/skills/managing-worktrees/agents/openai.yaml @@ -0,0 +1,4 @@ +interface: + display_name: "Managing Worktrees" + short_description: "Safely manage isolated Git worktree lifecycles" + default_prompt: "Use $managing-worktrees and the worktree CLI to create or maintain an isolated checkout, preserve recovery evidence, reconcile lifecycle state, and clean it up safely." diff --git a/plugins/worktree/skills/upgrade/SKILL.md b/plugins/worktree/skills/upgrade/SKILL.md new file mode 100644 index 0000000..9c34b48 --- /dev/null +++ b/plugins/worktree/skills/upgrade/SKILL.md @@ -0,0 +1,25 @@ +--- +name: upgrade +description: Check whether the worktree plugin and the `worktree` CLI are current, and upgrade them with the user's confirmation. Use when the user asks whether worktree is up to date or to upgrade or update it, when a session-start line starting with `b10x:` names `worktree`, or when a `worktree` command behaves differently from what a worktree skill describes. +--- + +# Upgrade worktree + +```bash +b10x upgrade worktree --host claude --out ~/.local/state/b10x/plan.json +``` + +Use `--host codex` in Codex. It compares the installed `worktree` plugin with what the marketplace serves and the `worktree` on `PATH` +with the newest release, and prints each difference with the action that fixes it. Nothing is +changed yet. + +- Nothing to change: say "worktree is current" with the versions, and stop. +- Otherwise show the actions in one list and ask once. After a clear yes: + `b10x setup apply --plan ~/.local/state/b10x/plan.json --yes`. +- A new plugin version loads in a new session; until then `b10x skill worktree:` prints the new text. + +No `b10x`? Follow https://github.com/beyond10x/agentplugins/releases/latest/download/SETUP.md first. + +## Next + +- Continue the work that prompted the check; `worktree:init` lists the skills. diff --git a/trials/aep-backlog/fixture/README.md b/trials/aep-backlog/fixture/README.md new file mode 100644 index 0000000..b5cf34b --- /dev/null +++ b/trials/aep-backlog/fixture/README.md @@ -0,0 +1,4 @@ +# birdlog + +A command-line log for bird sightings: what you saw, where, when, and how many. Sightings are kept +in one local file; nothing leaves the machine. diff --git a/trials/aep-backlog/fixture/TODO.md b/trials/aep-backlog/fixture/TODO.md new file mode 100644 index 0000000..b078db8 --- /dev/null +++ b/trials/aep-backlog/fixture/TODO.md @@ -0,0 +1,20 @@ +# TODO + +## Recording +- [ ] `birdlog add --count N --at ` records a sighting with the current time +- [ ] species names are checked against a bundled checklist; unknown names are refused with the closest matches +- [ ] edit and delete a sighting by its number +- [x] store sightings in a single JSON Lines file under the user's data directory + +## Looking back +- [ ] list sightings, newest first, filterable by species, place and date range +- [ ] a year list: every species seen this year, with the first date it was seen +- [ ] totals per place + +## Sharing +- [ ] export a date range as CSV for the regional survey +- [ ] import the survey's CSV format, skipping rows already present + +## Chores +- [ ] the checklist is two years old; find out how often the survey republishes it +- [ ] tests for the storage format before anything else touches it diff --git a/trials/aep-backlog/trial.yaml b/trials/aep-backlog/trial.yaml new file mode 100644 index 0000000..adf5883 --- /dev/null +++ b/trials/aep-backlog/trial.yaml @@ -0,0 +1,17 @@ +name: aep-backlog +kind: aep-backlog +prompt: | + Our backlog for this little tool is just `TODO.md`. I'd like to plan it properly with AEP from + now on: turn what's there into epics and stories, leave out what's already done, and make sure + the result validates. Don't delete `TODO.md`. + Nobody will answer questions during this run: where you would ask, decide, write down what you + decided and why, and go on. + + When you're done, list the skills and agents you used, quote any instruction text that confused + you, and paste the last validation output verbatim. +dir: birdlog +fixture: fixture +setup: + - b10x init aep --host claude --out plan.json + - b10x setup apply --plan plan.json --yes +measures: [tool_calls] diff --git a/trials/baseline.json b/trials/baseline.json new file mode 100644 index 0000000..14cd741 --- /dev/null +++ b/trials/baseline.json @@ -0,0 +1,60 @@ +{ + "aep-backlog": { + "tool_calls": 313 + }, + "ess-full-package": { + "tool_calls": 68, + "validate": "valid", + "synthesis": { + "scenarios": 13, + "refusals": 0 + }, + "unmapped": 0, + "outputs": [ + "asyncapi", + "conformance", + "docs", + "implementation", + "openapi", + "schema", + "site", + "types-go", + "types-rust" + ], + "go_test": { + "passed": 13, + "failed": 0, + "skipped": 0 + } + }, + "ess-new": { + "tool_calls": 12, + "validate": "valid", + "unmapped": 7 + }, + "ess-pipeline": { + "tool_calls": 19, + "validate": "valid", + "synthesis": { + "scenarios": 12, + "refusals": 0 + }, + "outputs": [ + "conformance", + "docs", + "schema" + ] + }, + "ess-retrofit": { + "tool_calls": 29, + "validate": "valid", + "synthesis": { + "scenarios": 16, + "refusals": 2 + }, + "unmapped": 7 + }, + "worktree-onboarding": { + "tool_calls": 19 + } +} diff --git a/trials/ess-full-package/trial.yaml b/trials/ess-full-package/trial.yaml new file mode 100644 index 0000000..c3e9951 --- /dev/null +++ b/trials/ess-full-package/trial.yaml @@ -0,0 +1,34 @@ +name: ess-full-package +kind: ess-full-package +prompt: | + I'd like to see the whole ESS package end to end on a small service. Pick a neutral domain + yourself — something with a lifecycle, a couple of commands that can be refused and at least one + event — and write its ESS specification in `spec/`. + + Then produce every output ESS offers from it, each in its own directory under `out/`: + `out/schema` (JSON Schema), `out/openapi`, `out/asyncapi`, `out/docs`, `out/site`, + `out/types-rust` (a Rust type library), `out/types-go` (a Go type library) and `out/conformance` + (the Go conformance suite). + + Then write a minimal in-memory Go implementation of the service in `impl/` that implements the + suite's `Target`, run `go test -v ./...` in `impl/`, and tell me how many scenarios passed, failed + and were skipped. Don't edit the generated suite to make it pass. + + When you're done, list the skills and agents you used, quote any instruction text that confused + you, and paste the last validation output, the last synthesis output and the last `go test` + summary verbatim. +dir: service +setup: + - b10x init ess --host claude --out plan.json + - b10x setup apply --plan plan.json --yes +measures: [tool_calls, validate, synthesis, unmapped, outputs, go_test] +outputs: + schema: out/schema + openapi: out/openapi + asyncapi: out/asyncapi + docs: out/docs + site: out/site + types-rust: out/types-rust + types-go: out/types-go + conformance: out/conformance + implementation: impl diff --git a/trials/ess-new/trial.yaml b/trials/ess-new/trial.yaml new file mode 100644 index 0000000..4b28062 --- /dev/null +++ b/trials/ess-new/trial.yaml @@ -0,0 +1,15 @@ +name: ess-new +kind: ess-new +prompt: | + I'm starting a small service for a community seed library: members borrow seed packets in + spring, and bring back seeds from their harvest in autumn so the library can restock. Some + varieties run out, and a member can only have five packets out at once. Write an ESS + specification for it in `spec/` and make sure it validates. + + When you're done, list the skills and agents you used, quote any instruction text that confused + you, and paste the last validation output verbatim. +dir: seedlib +setup: + - b10x init ess --host claude --out plan.json + - b10x setup apply --plan plan.json --yes +measures: [tool_calls, validate, unmapped] diff --git a/trials/ess-pipeline/fixture/spec/domains/booking.yaml b/trials/ess-pipeline/fixture/spec/domains/booking.yaml new file mode 100644 index 0000000..1bd9291 --- /dev/null +++ b/trials/ess-pipeline/fixture/spec/domains/booking.yaml @@ -0,0 +1,177 @@ +domain: rehearsal.booking + +summary: Holding, confirming and releasing a rehearsal room for one slot. + +naming: + wire: bookings + display: Bookings + +types: + - name: rehearsal.booking.BookingId + kind: newtype + of: Uuid + + - name: rehearsal.booking.Room + kind: enum + variants: [Studio, Hall, Booth] + + - name: rehearsal.booking.BandName + kind: newtype + of: String + +entities: + - name: rehearsal.booking.Booking + + identity: + name: booking_id + type: rehearsal.booking.BookingId + + fields: + - name: band + type: rehearsal.booking.BandName + - name: room + type: rehearsal.booking.Room + - name: starts_at + type: Timestamp + - name: hours + type: Integer + + invariants: + - hours > 0 + + lifecycle: + initial: Held + states: [Held, Confirmed, Released] + terminal: [Released] + transitions: + - name: confirm + from: [Held] + to: Confirmed + - name: release + from: [Held, Confirmed] + to: Released + +actors: + - name: rehearsal.booking.BandLeader + may: + - rehearsal.booking.HoldRoom + - rehearsal.booking.ConfirmBooking + - rehearsal.booking.ReleaseBooking + +errors: + - name: rehearsal.booking.InvalidLength + summary: The booking is not for a positive number of hours. + fields: + - name: hours + type: Integer + + - name: rehearsal.booking.BookingStateConflict + summary: The booking is not in a state this command acts from. + fields: + - name: state + type: rehearsal.booking.Booking.State + +commands: + - name: rehearsal.booking.HoldRoom + input: + - name: band + type: rehearsal.booking.BandName + - name: room + type: rehearsal.booking.Room + - name: starts_at + type: Timestamp + - name: hours + type: Integer + outcomes: + - name: held + when: hours > 0 + creates: rehearsal.booking.Booking + instance: booking_id + emits: + - rehearsal.booking.RoomHeld + payload: + rehearsal.booking.RoomHeld: + band: input.band + room: input.room + hours: input.hours + summary: The room is held for the band. + - name: refused + error: rehearsal.booking.InvalidLength + summary: The length was not positive, and nothing was held. + + - name: rehearsal.booking.ConfirmBooking + input: + - name: booking_id + type: rehearsal.booking.BookingId + outcomes: + - name: confirmed + moves: rehearsal.booking.Booking.confirm + instance: booking_id + emits: + - rehearsal.booking.BookingConfirmed + payload: + rehearsal.booking.BookingConfirmed: + booking_id: input.booking_id + summary: The band has paid and the room is theirs. + - name: wrong-state + wrong_state: true + error: rehearsal.booking.BookingStateConflict + summary: The booking is not held, so nothing was confirmed. + + - name: rehearsal.booking.ReleaseBooking + input: + - name: booking_id + type: rehearsal.booking.BookingId + outcomes: + - name: released + moves: rehearsal.booking.Booking.release + instance: booking_id + emits: + - rehearsal.booking.BookingReleased + payload: + rehearsal.booking.BookingReleased: + booking_id: input.booking_id + summary: The room is free again. + - name: wrong-state + wrong_state: true + error: rehearsal.booking.BookingStateConflict + summary: The booking was already released. + +events: + - name: rehearsal.booking.RoomHeld + fields: + - name: booking_id + type: rehearsal.booking.BookingId + - name: band + type: rehearsal.booking.BandName + - name: room + type: rehearsal.booking.Room + - name: hours + type: Integer + + - name: rehearsal.booking.BookingConfirmed + fields: + - name: booking_id + type: rehearsal.booking.BookingId + + - name: rehearsal.booking.BookingReleased + fields: + - name: booking_id + type: rehearsal.booking.BookingId + +views: + - name: rehearsal.booking.BookingById + source: rehearsal.booking.Booking + consistency: read_your_writes + fields: + - name: booking_id + type: rehearsal.booking.BookingId + - name: band + type: rehearsal.booking.BandName + - name: room + type: rehearsal.booking.Room + - name: hours + type: Integer + naming: + wire: by-id + display: Booking by id diff --git a/trials/ess-pipeline/fixture/spec/system.yaml b/trials/ess-pipeline/fixture/spec/system.yaml new file mode 100644 index 0000000..a114ce3 --- /dev/null +++ b/trials/ess-pipeline/fixture/spec/system.yaml @@ -0,0 +1,10 @@ +format: ess/1 +system: rehearsal +version: v1 + +summary: >- + Rehearsal rooms for a music school: a band holds a room for a slot, confirms it when the band + leader pays, and releases it when plans change. + +domains: + - rehearsal.booking diff --git a/trials/ess-pipeline/trial.yaml b/trials/ess-pipeline/trial.yaml new file mode 100644 index 0000000..bf01987 --- /dev/null +++ b/trials/ess-pipeline/trial.yaml @@ -0,0 +1,21 @@ +name: ess-pipeline +kind: ess-pipeline +prompt: | + There's an ESS specification for our rehearsal-room bookings in `spec/`. Generate its JSON + Schema into `out/schema`, its OpenAPI into `out/openapi` and its docs into `out/docs`, and + synthesize the conformance suite as the canonical document at `out/conformance.json`. Tell me how + many scenarios it has and whether the synthesis refused anything, and why. + + When you're done, list the skills and agents you used, quote any instruction text that confused + you, and paste the last synthesis output verbatim. +dir: rehearsal +fixture: fixture +setup: + - b10x init ess --host claude --out plan.json + - b10x setup apply --plan plan.json --yes +measures: [tool_calls, validate, synthesis, outputs] +outputs: + schema: out/schema + openapi: out/openapi + docs: out/docs + conformance: out/conformance.json diff --git a/trials/ess-retrofit/fixture/README.md b/trials/ess-retrofit/fixture/README.md new file mode 100644 index 0000000..882f5a2 --- /dev/null +++ b/trials/ess-retrofit/fixture/README.md @@ -0,0 +1,16 @@ +# toolshed + +The neighbourhood tool library: members borrow drills, ladders and saws, and bring them back. + +```console +go run . # listens on :8080 +``` + +| method | path | does | +|---|---|---| +| `POST` | `/tools` | add a tool (`{"name": "…", "category": "…"}`) | +| `POST` | `/tools/{id}/lend` | lend it to a member (`{"member": "…", "days": 7}`) | +| `POST` | `/tools/{id}/return` | take it back | +| `POST` | `/tools/{id}/retire` | take it out of circulation | +| `GET` | `/tools/{id}` | one tool | +| `GET` | `/loans/overdue` | tools out longer than agreed | diff --git a/trials/ess-retrofit/fixture/go.mod b/trials/ess-retrofit/fixture/go.mod new file mode 100644 index 0000000..52a2f8c --- /dev/null +++ b/trials/ess-retrofit/fixture/go.mod @@ -0,0 +1,3 @@ +module example.com/toolshed + +go 1.22 diff --git a/trials/ess-retrofit/fixture/main.go b/trials/ess-retrofit/fixture/main.go new file mode 100644 index 0000000..bbe8461 --- /dev/null +++ b/trials/ess-retrofit/fixture/main.go @@ -0,0 +1,76 @@ +package main + +import ( + "encoding/json" + "errors" + "log" + "net/http" + "strings" + "time" +) + +func main() { + shed := NewShed(time.Now) + log.Fatal(http.ListenAndServe(":8080", routes(shed))) +} + +func routes(shed *Shed) http.Handler { + mux := http.NewServeMux() + mux.HandleFunc("POST /tools", func(w http.ResponseWriter, r *http.Request) { + var body struct { + Name string `json:"name"` + Category string `json:"category"` + } + if err := json.NewDecoder(r.Body).Decode(&body); err != nil { + http.Error(w, err.Error(), http.StatusBadRequest) + return + } + tool, err := shed.Add(body.Name, body.Category) + reply(w, tool, err) + }) + mux.HandleFunc("POST /tools/{id}/lend", func(w http.ResponseWriter, r *http.Request) { + var body struct { + Member string `json:"member"` + Days int `json:"days"` + } + if err := json.NewDecoder(r.Body).Decode(&body); err != nil { + http.Error(w, err.Error(), http.StatusBadRequest) + return + } + tool, err := shed.Lend(r.PathValue("id"), body.Member, body.Days) + reply(w, tool, err) + }) + mux.HandleFunc("POST /tools/{id}/return", func(w http.ResponseWriter, r *http.Request) { + tool, err := shed.Return(r.PathValue("id")) + reply(w, tool, err) + }) + mux.HandleFunc("POST /tools/{id}/retire", func(w http.ResponseWriter, r *http.Request) { + tool, err := shed.Retire(r.PathValue("id")) + reply(w, tool, err) + }) + mux.HandleFunc("GET /tools/{id}", func(w http.ResponseWriter, r *http.Request) { + tool, err := shed.Get(r.PathValue("id")) + reply(w, tool, err) + }) + mux.HandleFunc("GET /loans/overdue", func(w http.ResponseWriter, r *http.Request) { + reply(w, shed.Overdue(), nil) + }) + return mux +} + +func reply(w http.ResponseWriter, value any, err error) { + switch { + case errors.Is(err, ErrNotFound): + http.Error(w, err.Error(), http.StatusNotFound) + case errors.Is(err, ErrOnLoan), errors.Is(err, ErrNotOnLoan), errors.Is(err, ErrRetired): + http.Error(w, err.Error(), http.StatusConflict) + case err != nil: + http.Error(w, err.Error(), http.StatusUnprocessableEntity) + default: + w.Header().Set("Content-Type", "application/json") + _ = json.NewEncoder(w).Encode(value) + } +} + +// Validation errors carry the offending value in their message only. +func invalid(message string) error { return errors.New(strings.TrimSpace(message)) } diff --git a/trials/ess-retrofit/fixture/shed.go b/trials/ess-retrofit/fixture/shed.go new file mode 100644 index 0000000..71afa42 --- /dev/null +++ b/trials/ess-retrofit/fixture/shed.go @@ -0,0 +1,165 @@ +package main + +import ( + "crypto/rand" + "encoding/hex" + "errors" + "fmt" + "sync" + "time" +) + +var ( + ErrNotFound = errors.New("no such tool") + ErrOnLoan = errors.New("tool is on loan") + ErrNotOnLoan = errors.New("tool is not on loan") + ErrRetired = errors.New("tool is retired") +) + +// Status is where a tool is: on the shelf, with a member, or gone for good. +type Status string + +const ( + Available Status = "available" + OnLoan Status = "on_loan" + Retired Status = "retired" +) + +// Categories the shed shelves tools under; anything else is refused. +var categories = map[string]bool{"power": true, "hand": true, "garden": true, "ladder": true} + +type Tool struct { + ID string `json:"id"` + Name string `json:"name"` + Category string `json:"category"` + Status Status `json:"status"` + Member string `json:"member,omitempty"` + DueAt *time.Time `json:"due_at,omitempty"` + Loans int `json:"loans"` +} + +// Event is what the shed announces; the notice board and the reminder mailer read them. +type Event struct { + Kind string `json:"kind"` + ToolID string `json:"tool_id"` + Member string `json:"member,omitempty"` + At time.Time `json:"at"` +} + +type Shed struct { + mu sync.Mutex + now func() time.Time + tools map[string]*Tool + Events []Event +} + +func NewShed(now func() time.Time) *Shed { + return &Shed{now: now, tools: map[string]*Tool{}} +} + +func newID() string { + b := make([]byte, 8) + _, _ = rand.Read(b) + return hex.EncodeToString(b) +} + +func (s *Shed) emit(kind string, tool *Tool) { + s.Events = append(s.Events, Event{Kind: kind, ToolID: tool.ID, Member: tool.Member, At: s.now()}) +} + +func (s *Shed) Add(name, category string) (Tool, error) { + s.mu.Lock() + defer s.mu.Unlock() + if name == "" { + return Tool{}, invalid("a tool needs a name") + } + if !categories[category] { + return Tool{}, invalid(fmt.Sprintf("unknown category %q", category)) + } + tool := &Tool{ID: newID(), Name: name, Category: category, Status: Available} + s.tools[tool.ID] = tool + s.emit("ToolAdded", tool) + return *tool, nil +} + +func (s *Shed) Lend(id, member string, days int) (Tool, error) { + s.mu.Lock() + defer s.mu.Unlock() + tool, ok := s.tools[id] + if !ok { + return Tool{}, ErrNotFound + } + if days < 1 || days > 28 { + return Tool{}, invalid(fmt.Sprintf("a loan is 1 to 28 days, not %d", days)) + } + switch tool.Status { + case OnLoan: + return Tool{}, ErrOnLoan + case Retired: + return Tool{}, ErrRetired + } + due := s.now().Add(time.Duration(days) * 24 * time.Hour) + tool.Status, tool.Member, tool.DueAt = OnLoan, member, &due + tool.Loans++ + s.emit("ToolLent", tool) + return *tool, nil +} + +func (s *Shed) Return(id string) (Tool, error) { + s.mu.Lock() + defer s.mu.Unlock() + tool, ok := s.tools[id] + if !ok { + return Tool{}, ErrNotFound + } + if tool.Status != OnLoan { + return Tool{}, ErrNotOnLoan + } + s.emit("ToolReturned", tool) + tool.Status, tool.Member, tool.DueAt = Available, "", nil + return *tool, nil +} + +// Retire takes a tool out of circulation. A tool on loan is retired when it comes back: the call +// succeeds and changes nothing, and the desk is expected to try again later. +func (s *Shed) Retire(id string) (Tool, error) { + s.mu.Lock() + defer s.mu.Unlock() + tool, ok := s.tools[id] + if !ok { + return Tool{}, ErrNotFound + } + switch tool.Status { + case Retired: + return Tool{}, ErrRetired + case OnLoan: + return *tool, nil + } + tool.Status = Retired + s.emit("ToolRetired", tool) + return *tool, nil +} + +func (s *Shed) Get(id string) (Tool, error) { + s.mu.Lock() + defer s.mu.Unlock() + tool, ok := s.tools[id] + if !ok { + return Tool{}, ErrNotFound + } + return *tool, nil +} + +// Overdue lists the tools whose loan ran past its due date, oldest first by insertion. +func (s *Shed) Overdue() []Tool { + s.mu.Lock() + defer s.mu.Unlock() + now := s.now() + var late []Tool + for _, tool := range s.tools { + if tool.Status == OnLoan && tool.DueAt != nil && now.After(*tool.DueAt) { + late = append(late, *tool) + } + } + return late +} diff --git a/trials/ess-retrofit/trial.yaml b/trials/ess-retrofit/trial.yaml new file mode 100644 index 0000000..1ed53ab --- /dev/null +++ b/trials/ess-retrofit/trial.yaml @@ -0,0 +1,15 @@ +name: ess-retrofit +kind: ess-retrofit +prompt: | + This repository is our tool-lending service. Write an ESS specification in `spec/` that describes + what the code actually does today — not what it should do — validate it, and synthesize its + conformance suite. Tell me what you could not map from the code, and what the synthesis refused. + + When you're done, list the skills and agents you used, quote any instruction text that confused + you, and paste the last validation output and the last synthesis output verbatim. +dir: toolshed +fixture: fixture +setup: + - b10x init ess --host claude --out plan.json + - b10x setup apply --plan plan.json --yes +measures: [tool_calls, validate, synthesis, unmapped] diff --git a/trials/upgrade-seeded/trial.yaml b/trials/upgrade-seeded/trial.yaml new file mode 100644 index 0000000..5776ed8 --- /dev/null +++ b/trials/upgrade-seeded/trial.yaml @@ -0,0 +1,18 @@ +name: upgrade-seeded +kind: upgrade +prompt: | + I installed the Beyond10x plugins a while ago and haven't touched them since. Bring everything up + to date, and tell me what changed. Nobody will answer questions during this run: where you would + ask, decide, write down what you decided and why, and go on. + + When you're done, list the skills and agents you used, quote any instruction text that confused + you, and paste the last command output verbatim. +seeded: true +setup: + - git -c advice.detachedHead=false clone --quiet --depth 1 --branch 0.12.0 https://github.com/beyond10x/agentplugins.git home/seed-marketplace + - claude plugin marketplace add "$PWD/home/seed-marketplace" + - | + for plugin in $(grep -o '"\./plugins/[a-z0-9-]*"' home/seed-marketplace/.claude-plugin/marketplace.json | tr -d '"'); do + claude plugin install "${plugin##*/}@b10x" + done +measures: [tool_calls] diff --git a/trials/worktree-onboarding/fixture/README.md b/trials/worktree-onboarding/fixture/README.md new file mode 100644 index 0000000..aa8278a --- /dev/null +++ b/trials/worktree-onboarding/fixture/README.md @@ -0,0 +1,8 @@ +# recipe-scaler + +Scale a recipe up or down: give it the recipe and the number of servings you want, and it rewrites +every quantity. Metric and imperial units are both undrestood. + +```console +cargo run -- scale pancakes.txt --servings 6 +``` diff --git a/trials/worktree-onboarding/fixture/pancakes.txt b/trials/worktree-onboarding/fixture/pancakes.txt new file mode 100644 index 0000000..778fc4d --- /dev/null +++ b/trials/worktree-onboarding/fixture/pancakes.txt @@ -0,0 +1,5 @@ +Pancakes (serves 2) +150 g flour +1 egg +200 ml milk +1 tbsp sugar diff --git a/trials/worktree-onboarding/trial.yaml b/trials/worktree-onboarding/trial.yaml new file mode 100644 index 0000000..25e0930 --- /dev/null +++ b/trials/worktree-onboarding/trial.yaml @@ -0,0 +1,18 @@ +name: worktree-onboarding +kind: worktree-onboarding +prompt: | + Set up worktree for this repository so I can work on changes in isolated checkouts. Then use it + to fix the typo in `README.md` in an isolated checkout on a branch, commit it there, and tell me + where the checkout is. Leave this checkout as it is. + Nobody will answer questions during this run: where you would ask, decide, write down what you + decided and why, and go on. + + When you're done, list the skills and agents you used, quote any instruction text that confused + you, and paste the last command output verbatim. +dir: recipe-scaler +fixture: fixture +remote: true +setup: + - b10x init worktree --host claude --out plan.json + - b10x setup apply --plan plan.json --yes +measures: [tool_calls] diff --git a/verified.json b/verified.json new file mode 100644 index 0000000..80f5ca6 --- /dev/null +++ b/verified.json @@ -0,0 +1,5 @@ +{ + "aep": "0.59.3", + "ess": "0.32.1", + "worktree": "0.7.2" +} diff --git a/website/docs/choose-a-plugin.md b/website/docs/choose-a-plugin.md index 21cd167..bcd8b86 100644 --- a/website/docs/choose-a-plugin.md +++ b/website/docs/choose-a-plugin.md @@ -15,14 +15,14 @@ so the capability works in Codex and Claude Code, while keeping host-specific wr ## Planning or understanding work -Use **AEP Plan** when the repository has a governed artifact store, or when you need to turn a +Use **`aep`** (its `planning` skill) when the repository has a governed artifact store, or when you need to turn a goal into reviewable epics, stories, and tasks. Its agents can decompose, review, or reverse-engineer -work, but the planning skill remains responsible for legal lifecycle moves. +work, but `aep:planning` remains responsible for legal lifecycle moves. ## Delivering a development story -Use **AEP Drive** when an accepted story is ready to scope and implement. Its wave guidance defines -coordination and review roles for an interactive session; its `drive` skill hands one story to +Use **`aep:implementing`** when an accepted story is ready to scope and implement. Its wave mode +defines coordination and review roles for an interactive session; its drive mode hands one story to the reference driver instead, where the bounds are decided by the engine rather than followed by an agent — and it says before it launches that the driven walk has not yet reached `complete`. Neither makes a draft plan implementation-ready. @@ -35,7 +35,7 @@ them. ## Managing repository workspaces -Use **Workspace Hygiene** before an agent changes a repository or when linked worktrees need an +Use **Worktree** before an agent changes a repository or when linked worktrees need an inventory or cleanup sweep. Its skill is generated by the public `worktree` CLI, so the documented commands and the executable safety policy stay aligned. The plugin does not delete unmanaged trees or replace the standalone toolchain. diff --git a/website/docs/golden-path.md b/website/docs/golden-path.md index 2a20dbf..521364b 100644 --- a/website/docs/golden-path.md +++ b/website/docs/golden-path.md @@ -25,7 +25,8 @@ nobody has decided, and what the plan does with that instead of guessing. ## Prerequisites -Install `aep-plan`, `aep-drive` and `ess-specify` from the marketplace — see [Install](./install.md) — +Install `aep` from the marketplace — see [Install](./install.md) — and the +`ess` plugin from the [ESS repository](https://github.com/beyond10x/ess#point-your-agent-here), and have the `aep` CLI on your PATH. Step 3 also uses the `ess` CLI. ```shell-session @@ -341,7 +342,7 @@ story:commercial-client-record moved proposed -> active (revision 4) ``` ```text -Take story:commercial-client-record through the aep-drive wave: scope it into units, implement the units, +Take story:commercial-client-record through the aep wave: scope it into units, implement the units, and have the adversary review the result against the story's acceptance and this repository's gate. ``` @@ -367,7 +368,7 @@ Drive story:commercial-client-record. Run aep doctor first and stop if anything the run will cost before you launch it, and print the run id and how to follow it. ``` -The `drive` skill checks the checkout, points `metaharness aep drive run` at the task document that names the +The drive mode of `aep:implementing` checks the checkout, points `metaharness aep drive run` at the task document that names the story, launches it against the project's step map, and prints the run id. It moves no artifact itself — the moves are the driver's, which is the whole property being tested — and it relays a refusal (a held lock, missing evidence, two step maps that both fit) verbatim and stops. diff --git a/website/docs/install.md b/website/docs/install.md index ceeba3e..6a6b4d2 100644 --- a/website/docs/install.md +++ b/website/docs/install.md @@ -3,19 +3,36 @@ sidebar_position: 3 title: Install --- -# Install from the `beyond10x` marketplace +# Install from the `b10x` marketplace + +## Set up with one sentence + +Tell Claude Code or Codex: + +> Set up Beyond10x: follow https://github.com/beyond10x/agentplugins/releases/latest/download/SETUP.md + +The agent installs the `b10x` binary and runs `/b10x:init`: it asks what you want to do (plan and +deliver work, write specifications, isolated Git checkouts, integrations) and how to install the +command-line tools (prebuilt archives by default, or `cargo`), lists every change — including +earlier installs it replaces — and applies them only after you confirm. It snapshots every file it +changes; `b10x setup undo` restores them. Each product then starts with its own `/:init`, +and `/b10x:upgrade` or `/:upgrade` checks for newer versions. The rest of this page is the +manual route. + +## The marketplace The marketplace source is the GitHub repository `beyond10x/agentplugins` and the marketplace -identity is `beyond10x`. The installable names are `beyond10x`, `aep-plan`, `aep-drive`, -`ess-specify`, `workspace-hygiene`, and `connectors`. All six are included in the pinned `0.9.2` -release below. The [Connectors guide](plugins/connectors.md) covers its separate CLI prerequisite. +identity is `b10x`. The installable names are `b10x`, `aep`, `ess`, `worktree` and `connectors`, +in both hosts; every one lives in this repository. The CLIs come from their own repositories' +releases, prebuilt or with `cargo install`. The [Connectors guide](plugins/connectors.md) covers its +separate CLI prerequisite. -## Before you install: put `aep` and `ess` on your `PATH` +## Before you install: put `aep` on your `PATH` -`aep-plan` and `aep-drive` are instruction surfaces for a program they do not ship. Both drive the -`aep` CLI. `ess-specify` likewise drives the `ess` CLI. Install both before the plugins so the first -task does not stop at a missing command. The AEP and ESS versions pinned below publish native archives for -x86-64 and ARM64 Linux and macOS, plus a `SHA256SUMS` file. Windows archives are not published. +The `aep` plugin is an instruction surface for a program it does not ship. It drives the +`aep` CLI. Install it before the plugins so the first task does not stop at a missing command. The +AEP version pinned below publishes native archives for x86-64 and ARM64 Linux and macOS, plus a +`SHA256SUMS` file. Windows archives are not published. Select the native target once: @@ -48,24 +65,6 @@ mkdir -p "$HOME/.local/bin" install -m 0755 "aep-${B10X_AEP_VERSION}-${B10X_TARGET}/aep" "$HOME/.local/bin/aep" ``` -Install ESS `0.22.1` the same way: - -```bash -B10X_ESS_VERSION=0.22.1 -B10X_ESS_ARCHIVE="ess-${B10X_ESS_VERSION}-${B10X_TARGET}.tar.gz" -B10X_ESS_RELEASE="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/beyond10x/ess/releases/download/${B10X_ESS_VERSION}" -curl --fail --location --remote-name "${B10X_ESS_RELEASE}/${B10X_ESS_ARCHIVE}" -curl --fail --location --output ESS-SHA256SUMS "${B10X_ESS_RELEASE}/SHA256SUMS" -if command -v sha256sum >/dev/null; then - grep " ${B10X_ESS_ARCHIVE}$" ESS-SHA256SUMS | sha256sum --check - -else - grep " ${B10X_ESS_ARCHIVE}$" ESS-SHA256SUMS | shasum --algorithm 256 --check -fi -tar -xzf "${B10X_ESS_ARCHIVE}" -mkdir -p "$HOME/.local/bin" -install -m 0755 "ess-${B10X_ESS_VERSION}-${B10X_TARGET}/ess" "$HOME/.local/bin/ess" -``` - If `$HOME/.local/bin` is not already on your `PATH`, add it in your shell profile. Building from source remains a fallback: use the exact release tag with Cargo, never a moving branch. @@ -73,14 +72,13 @@ Confirm it before installing anything: ```bash aep --version -ess --version ``` -The expected lines are `protocol 0.55.0` and `ess 0.22.1`. (`aep` retains `protocol` as its version -label for compatibility.) `command not found` means the affected plugin will install and then stop -at its first CLI command. Only the `beyond10x` front door needs neither binary. +The expected line is `protocol 0.55.0`. (`aep` retains `protocol` as its version label for +compatibility.) `command not found` means the affected plugin will install and then stop at its +first CLI command. The `b10x` front door does not need the `aep` binary. -### `aep-drive`'s `drive` skill also needs Metaharness +### The `aep:implementing` skill also needs Metaharness AEP 0.55.0 refuses to run a model-backed step map itself and names `metaharness aep drive run` as the command that does. Metaharness `0.7.0` includes this host; tags through `0.6.5` predate it. @@ -98,74 +96,67 @@ install (at `0.7.0` that revision is `a23176ae`, which `git describe --tags` in reports as `0.54.0-33-ga23176ae`, i.e. behind the `0.55.0` you put on `PATH`), so read that manifest for the Metaharness↔AEP pair rather than assuming the two versions match. -The wave skill, the planning skill and every other plugin need no Metaharness. +Wave mode, `aep:planning` and every other plugin need no Metaharness. -## Claude Code +`b10x-harness`, the Beyond10x agent loop Metaharness's `b10x` adapter runs, is optional as well: +`b10x install b10x-harness` installs its newest release. It runs on Linux only, from a prebuilt +archive where the release has one for the machine, otherwise built with `cargo`. -Copy the whole block into a Claude Code session: +## Claude Code ```text -/plugin marketplace add https://github.com/beyond10x/agentplugins.git#0.9.2 -/plugin install aep-plan@beyond10x -/plugin install aep-drive@beyond10x -/plugin install ess-specify@beyond10x +/plugin marketplace add beyond10x/agentplugins +/plugin install b10x@b10x +/plugin install aep@b10x /reload-plugins ``` -The first line registers the repository at the immutable `0.9.2` release; each install names its -plugin in the `@beyond10x` form. `/reload-plugins` activates them immediately. Add -`/plugin install beyond10x@beyond10x` for the front door and -`/plugin install workspace-hygiene@beyond10x` for managed worktrees, or -`/plugin install connectors@beyond10x` for integrations. Claude Code reads -`.claude-plugin/marketplace.json` and the selected plugin's `.claude-plugin/plugin.json`. See +Add `/plugin install ess@b10x`, `/plugin install worktree@b10x` or +`/plugin install connectors@b10x` as needed. The marketplace follows the default branch, whose +gate keeps every entry installable; `/reload-plugins` activates new plugins. Claude Code reads +`.claude-plugin/marketplace.json` and each plugin's `.claude-plugin/plugin.json`. See [Claude Code's plugin documentation](https://code.claude.com/docs/en/discover-plugins) for the host commands and supported marketplace sources. ## Codex -Codex offers the same plugins from the same repository under the same `beyond10x` identity. -For a fresh installation, run this release-pinned block: - -```bash -codex plugin marketplace add https://github.com/beyond10x/agentplugins.git --ref 0.9.2 -codex plugin add beyond10x@beyond10x -codex plugin add aep-plan@beyond10x -codex plugin add aep-drive@beyond10x -codex plugin add ess-specify@beyond10x -codex plugin add workspace-hygiene@beyond10x -codex plugin add connectors@beyond10x -``` - -An immutable marketplace pin does not advance when `codex plugin marketplace upgrade` runs. To -replace an older pin, remove the installed plugins and marketplace registration, then run the fresh -block above: - ```bash -codex plugin remove beyond10x@beyond10x -codex plugin remove aep-plan@beyond10x -codex plugin remove aep-drive@beyond10x -codex plugin remove ess-specify@beyond10x -codex plugin remove workspace-hygiene@beyond10x -codex plugin remove connectors@beyond10x -codex plugin marketplace remove beyond10x +codex plugin marketplace add beyond10x/agentplugins +codex plugin add b10x@b10x +codex plugin add aep@b10x ``` -The commands leave every focused plugin installed and enabled. Start a new Codex thread afterwards; +Add `ess@b10x`, `worktree@b10x` or `connectors@b10x` the same way. `codex plugin marketplace upgrade` +refreshes the marketplace. Start a new Codex thread afterwards; plugin instructions are injected when a thread starts, not retroactively into a running thread. The same plugins remain available from the Plugins surface. The authoritative description of what Codex will find is [`.agents/plugins/marketplace.json`](https://github.com/beyond10x/agentplugins/blob/main/.agents/plugins/marketplace.json) in this repository; Codex reads it together with the selected plugin's `.codex-plugin/plugin.json`. -The `aep` and `ess` binary requirements above apply unchanged. +The `aep` binary requirement above applies unchanged. -## Pinning +## Upgrading -The blocks above are already pinned to the bare `0.9.2` release tag. Upgrade by changing that tag -deliberately, re-registering the marketplace source, and running `/reload-plugins`. The release gate -validates both marketplace formats, every declared instruction file, the public documentation, and -the version recorded by each plugin manifest. +Run setup again, or `b10x setup plan` and `b10x setup apply` yourself. The release gate validates both +marketplace formats, every declared instruction file, the public documentation, and the version +recorded by each plugin manifest this repository carries. After installation, invoke the skill by its displayed name or ask the agent for the capability the -plugin describes. Start with `beyond10x` if you want the front door to select a specialist. +plugin describes. Start with `b10x:routing` if you want the front door to select a specialist. Installation does not grant filesystem, network, credential, or approval authority; the host and repository rules still decide those boundaries. + +## Pin a version + +A repository can hold a CLI at one release instead of the newest: + +```bash +b10x pin ess 0.32.0 # exactly this release +b10x pin aep 0.59 # the newest 0.59.x +b10x unpin ess +``` + +The pins go into `b10x.toml` in the current directory (or the nearest one above); commit it. +`b10x init`, `b10x upgrade`, `b10x setup plan` and `b10x install` then use the pinned release, +`upgrade` names a newer one without installing it, and `b10x check` says when the CLI on `PATH` +does not match the pin. diff --git a/website/docs/intro.md b/website/docs/intro.md index 3429f89..01122a0 100644 --- a/website/docs/intro.md +++ b/website/docs/intro.md @@ -2,26 +2,28 @@ sidebar_position: 1 slug: / title: Agent Plugins -description: The curated beyond10x marketplace for focused engineering-agent guidance. +description: The curated b10x marketplace for focused engineering-agent guidance. --- # Focused guidance, explicit scope -The `beyond10x` marketplace publishes a front door and four specialist plugins. Install the front +The `b10x` marketplace publishes a front door and four specialist plugins. Install the front door when you want routing, ecosystem resources, or portable plugin creation. Install a specialist directly when the work is already clear. | Plugin | Use it for | Includes | |---|---|---| -| [`beyond10x`](./plugins/beyond10x.md) | Marketplace navigation and plugin authoring | router, public resource map, portable plugin creator | -| [`aep-plan`](./plugins/aep-plan.md) | Governed plans and artifact stores | planning skill, decomposer, plan reviewer, reverse engineer | -| [`aep-drive`](./plugins/aep-drive.md) | Governed development delivery | wave skill, story scoper, implementor, adversary | -| [`ess-specify`](./plugins/ess-specify.md) | Executable system specifications | specification, schema validation and deterministic projection guidance | -| [`workspace-hygiene`](./plugins/workspace-hygiene.md) | Git workspaces | managed worktrees, leases, recovery proof, and safe cleanup | +| [`b10x`](./plugins/b10x.md) | Setup, upgrades and navigation | `init`, `upgrade`, `routing` and `authoring-plugins` skills, the `b10x` binary, drift check | +| [`aep`](./plugins/aep.md) | Governed planning and delivery | `planning`, `migrating` and `implementing` skills; decomposer, plan critics, reverse engineer, story scoper, implementor, adversary, security reviewer | +| [`ess`](./plugins/ess.md) | Executable System Specifications | `specifying`, `retrofitting`, `testing-conformance` and `hardening` skills; author, conformance, retrofitter agents | +| [`worktree`](./plugins/worktree.md) | Git workspaces | managed worktrees, leases, recovery proof, and safe cleanup | + +Every plugin starts with `/:init` and checks itself with `/:upgrade`; `/b10x:init` +asks what you want to do and installs the matching plugins and CLIs. The marketplace contains instructions, not credentials. A plugin does not acquire authority to write a repository, contact a service, or bypass an approval boundary merely because it is installed. -[Start with the Beyond10x front door](./plugins/beyond10x.md), [choose a specialist](./choose-a-plugin.md), +[Set up with one sentence](./install.md), [start with the front door](./plugins/b10x.md), [choose a specialist](./choose-a-plugin.md), or go directly to [installation](./install.md). diff --git a/website/docs/plugins/aep-drive.md b/website/docs/plugins/aep-drive.md deleted file mode 100644 index 86d98c6..0000000 --- a/website/docs/plugins/aep-drive.md +++ /dev/null @@ -1,33 +0,0 @@ ---- -title: AEP Drive ---- - -# `aep-drive` - -Use this plugin to deliver accepted development work through the Agentic Development Protocol -profile. - -It provides: - -- wave coordination guidance; -- a story scoper that turns an accepted story into bounded implementation units; -- an implementor role for an assigned unit; -- an adversary role that checks the result against scope, evidence, and repository invariants; -- a `drive` entry that starts one governed `metaharness aep drive` run over a single story. - -This plugin builds on AEP's planning substrate. It does not replace the repository gate, invent lifecycle -moves, or give implementors authority beyond their assigned unit. - -## Two ways to deliver a story, and they enforce differently - -The wave coordinates an interactive session: its rules are instructions the coordinating agent -follows. `drive` hands one story to the reference driver, where the step map's bounds are decided by -the engine rather than obeyed by an agent. - -Driven runs are not finished work on the `aep` side. The walk has not yet reached `complete` — -`aep`'s `story:governed-dogfood-run` records two attempts that stopped before the review step — so -the `drive` skill says so before it launches anything, prints the run id and how to follow it, and -moves no artifact itself. - -The `drive` entry needs a Metaharness build that carries `metaharness aep drive`: AEP 0.55.0 refuses -a model-backed map itself and names that command. [Install](../install.md) names the build to use. diff --git a/website/docs/plugins/aep-plan.md b/website/docs/plugins/aep-plan.md deleted file mode 100644 index 7642bcc..0000000 --- a/website/docs/plugins/aep-plan.md +++ /dev/null @@ -1,36 +0,0 @@ ---- -title: AEP Plan ---- - -# `aep-plan` - -Use this plugin to work with AEP's governed planning substrate. - -It provides: - -- a planning skill that discovers the repository-local store and uses the canonical `aep` command; -- a story migration skill that adopts an existing backlog without rewriting or deleting its - sources; -- a decomposer for turning a concrete outcome into related planning artifacts; -- a plan reviewer for checking readiness, evidence, and dependency shape; -- a reverse engineer for mapping an existing codebase into reviewable work; -- an acceptance critic for checking that each drafted item states an outcome somebody can observe; -- a design critic for checking the shape of a decomposition: coupling, cycles, split abstractions; -- a scope critic for checking that the set covers what it was drafted from, and nothing beyond it; -- a parallel-safety critic for naming the items that would land on the same file. - -The four critics are a panel, not four separate reviews. After a decomposition is reported, the -planning skill dispatches them at once, records each verdict as an immutable review result related -to the artifacts it judged, revises the drafts on every verdict that asks for it, and stops after -two rounds with whatever is still open named in its report. None of them writes to the store. - -Planning also refuses to decompose an epic or story that introduces an entity no ESS domain -declares. The domain is drafted and cited from the artifact first, and any relation that could not -be read from code, an OpenAPI document or an existing artifact is marked unmapped, never guessed. - -The plugin respects store ownership: machine-owned artifact metadata is changed through AEP, not -by editing markdown frontmatter. A refusal from the lifecycle is a result to report, not a guard to -route around. - -Install this plugin for planning. Add [`aep-drive`](./aep-drive.md) only when accepted work moves -into a development wave. diff --git a/website/docs/plugins/aep.md b/website/docs/plugins/aep.md new file mode 100644 index 0000000..3514184 --- /dev/null +++ b/website/docs/plugins/aep.md @@ -0,0 +1,69 @@ +--- +title: AEP +--- + +# `aep` + +Use this plugin to plan governed work in AEP's artifact store and to deliver accepted work through +the Agentic Development Protocol profile. Both halves drive the `aep` CLI. + +## Planning + +Use it to work with AEP's governed planning substrate. + +It provides: + +- a planning skill that discovers the repository-local store and uses the canonical `aep` command; +- a story migration skill that adopts an existing backlog without rewriting or deleting its + sources; +- a decomposer for turning a concrete outcome into related planning artifacts; +- a plan reviewer for checking readiness, evidence, and dependency shape; +- a reverse engineer for mapping an existing codebase into reviewable work; +- an acceptance critic for checking that each drafted item states an outcome somebody can observe; +- a design critic for checking the shape of a decomposition: coupling, cycles, split abstractions; +- a scope critic for checking that the set covers what it was drafted from, and nothing beyond it; +- a parallel-safety critic for naming the items that would land on the same file. + +The four critics are a panel, not four separate reviews. After a decomposition is reported, the +planning skill dispatches them at once, records each verdict as an immutable review result related +to the artifacts it judged, revises the drafts on every verdict that asks for it, and stops after +two rounds with whatever is still open named in its report. None of them writes to the store. + +Planning also refuses to decompose an epic or story that introduces an entity no ESS domain +declares. The domain is drafted and cited from the artifact first, and any relation that could not +be read from code, an OpenAPI document or an existing artifact is marked unmapped, never guessed. + +The plugin respects store ownership: machine-owned artifact metadata is changed through AEP, not +by editing markdown frontmatter. A refusal from the lifecycle is a result to report, not a guard to +route around. + +## Delivery + +It provides: + +- the `implementing` skill, in wave mode: coordination guidance for several stories at once; +- a story scoper that turns an accepted story into bounded implementation units; +- an implementor role for an assigned unit; +- an adversary role that checks the result against scope, evidence, and repository invariants; +- the `implementing` skill, in drive mode: one governed `metaharness aep drive` run over a single story. + +This plugin builds on AEP's planning substrate. It does not replace the repository gate, invent lifecycle +moves, or give implementors authority beyond their assigned unit. + +### Two ways to deliver a story, and they enforce differently + +The wave coordinates an interactive session: its rules are instructions the coordinating agent +follows. `drive` hands one story to the reference driver, where the step map's bounds are decided by +the engine rather than obeyed by an agent. + +Driven runs are not finished work on the `aep` side. The walk has not yet reached `complete` — +`aep`'s `story:governed-dogfood-run` records two attempts that stopped before the review step — so +drive mode says so before it launches anything, prints the run id and how to follow it, and +moves no artifact itself. + +Drive mode needs a Metaharness build that carries `metaharness aep drive`: AEP 0.55.0 refuses +a model-backed map itself and names that command. [Install](../install.md) names the build to use. + +`b10x` treats both `metaharness` and `b10x-harness`, the Beyond10x agent loop that Metaharness's +`b10x` adapter runs, as optional CLIs of this plugin: it reports them, and `b10x install ` adds +one. `b10x-harness` runs on Linux only. diff --git a/website/docs/plugins/b10x.md b/website/docs/plugins/b10x.md new file mode 100644 index 0000000..c7b58d0 --- /dev/null +++ b/website/docs/plugins/b10x.md @@ -0,0 +1,85 @@ +--- +title: b10x +--- + +# `b10x` + +Use this plugin as the marketplace front door. It sets up and upgrades the other plugins and the +binaries they drive, selects the smallest specialist for a request, +provides a map of public Beyond10x resources, and helps create plugins that keep their core +capabilities portable across Codex and Claude Code. + +It provides: + +- an `installing` skill that drives the `b10x` binary: it reads both hosts, the binaries on `PATH` and + project settings, asks which products to have, lists every change and applies it after + confirmation (see [Install](../install.md)); +- a session-start hook that runs `b10x check` and prints one line per plugin/binary drift; +- a `routing` skill for selecting `aep`, `ess`, `worktree` or + `connectors`; +- direct links to public product guides, command references, plugin references, and source; +- an `authoring-plugins` skill for creating or porting dual-harness plugins. + +## The `b10x` binary + +| command | does | +|---|---| +| `b10x init … [--method cargo\|prebuilt] [--host claude\|codex\|all] --out ` | plan the named products: plugins and CLIs installed or brought current; nothing else touched | +| `b10x upgrade […] --out ` | the same for installed products (all, or the named ones) | +| `b10x setup apply --plan --yes` | refuse a stale plan, snapshot every file it changes, run the actions, check that the result converged | +| `b10x setup undo []` | restore the files the newest (or named) snapshot holds, then refresh both hosts' marketplaces | +| `b10x setup plan [--products …]` | the whole desired state at once: the named products installed, other catalog plugins removed | +| `b10x skill [:]` | list an installed plugin's skills, or print one — usable before a restart | +| `b10x setup guide` | print `/b10x:init`, for an agent that has no plugin yet | +| `b10x check` | the session-start drift check; prints only problems, refreshes the newest releases at most daily | +| `b10x install [--tag ] [--method cargo\|prebuilt]` | install one CLI (the repository's pin, else the newest) | +| `b10x pin ` | pin a CLI for this repository in `b10x.toml` | +| `b10x unpin ` | remove that pin | + +The CLIs are `aep`, `ess` and `worktree`, plus the optional `metaharness` and `b10x-harness` of +the `aep` product. They install from the release's checksummed prebuilt archive by default, and with +`cargo` when `--method cargo` asks for it or the release has no archive for this machine's target +(its `SHA256SUMS` lists the targets it has). A binary limited to some operating systems +(`b10x-harness`: Linux) is not installed elsewhere. `B10X_MARKETPLACE=` makes `b10x` +read and register that +checkout instead of this repository, for testing an unpublished marketplace. + +## How it stays current + +- **Every plugin lives here** and carries this repository's version. `catalog.json` lists products, + plugins, CLIs and retired names, and no versions. +- **CLIs are the newest release**, unless a repository pins one ([Pin a version](../install.md#pin-a-version)). + `agentplugins-check tools` (every pull request, `main` push and daily) downloads the newest + `aep`, `ess` and `worktree`, runs every command the skills spell with `--help`, validates the ESS + syntax example, and fails when a release is newer than `verified.json`, the release the skills + were last verified against. +- **On the machine**, `b10x check` runs at session start, reads the newest releases from GitHub at + most once a day, and says when the plugins or a CLI are older than the newest release, when a CLI + does not match its repository's pin, when an earlier install is still there, or when nothing is + set up. + +## Portable plugin creation + +The creator uses `skills//SKILL.md` as the shared capability layer. Both hosts load that +layout and its references, assets, and optional scripts. New command-like behavior is authored as a +skill, which Claude Code exposes as a slash shortcut and Codex exposes through skill invocation. + +Specialist-agent behavior also lives in a shared skill. A thin Claude Code `agents/` wrapper may be +added when native delegation is useful, while Codex uses the shared skill directly or delegates +through its supported orchestration. The creator never presents a Claude-only command or agent file +as a portable component. + +The full compatibility rules live with the plugin source and link to the current +[OpenAI plugin packaging guide](https://developers.openai.com/plugins/build/plugins) and +[Claude Code plugin reference](https://code.claude.com/docs/en/plugins-reference). + +## Routing boundaries + +The front door does not copy specialist instructions. Install the plugin it selects when you need +the governed workflow: + +- [`aep`](./aep.md) for plans, artifact stores and accepted development work; +- [`ess`](./ess.md) for ESS specification, retrofit, validation and conformance; +- [`worktree`](./worktree.md) for managed Git worktrees and safe cleanup. + +For the broader organization map, start at [Beyond10x getting started](https://beyond10x.github.io/getting-started/). diff --git a/website/docs/plugins/beyond10x.md b/website/docs/plugins/beyond10x.md deleted file mode 100644 index 7b99a11..0000000 --- a/website/docs/plugins/beyond10x.md +++ /dev/null @@ -1,43 +0,0 @@ ---- -title: Beyond10x ---- - -# `beyond10x` - -Use this plugin as the marketplace front door. It selects the smallest specialist for a request, -provides a map of public Beyond10x resources, and helps create plugins that keep their core -capabilities portable across Codex and Claude Code. - -It provides: - -- a `beyond10x` routing skill for selecting `aep-plan`, `aep-drive`, `ess-specify`, or - `workspace-hygiene`; -- direct links to public product guides, command references, plugin references, and source; -- a `plugin-creator` skill for creating or porting dual-harness plugins. - -## Portable plugin creation - -The creator uses `skills//SKILL.md` as the shared capability layer. Both hosts load that -layout and its references, assets, and optional scripts. New command-like behavior is authored as a -skill, which Claude Code exposes as a slash shortcut and Codex exposes through skill invocation. - -Specialist-agent behavior also lives in a shared skill. A thin Claude Code `agents/` wrapper may be -added when native delegation is useful, while Codex uses the shared skill directly or delegates -through its supported orchestration. The creator never presents a Claude-only command or agent file -as a portable component. - -The full compatibility rules live with the plugin source and link to the current -[OpenAI plugin packaging guide](https://developers.openai.com/plugins/build/plugins) and -[Claude Code plugin reference](https://code.claude.com/docs/en/plugins-reference). - -## Routing boundaries - -The front door does not copy specialist instructions. Install the plugin it selects when you need -the governed workflow: - -- [`aep-plan`](./aep-plan.md) for plans and artifact stores; -- [`aep-drive`](./aep-drive.md) for accepted development work; -- [`ess-specify`](./ess-specify.md) for ESS specification, validation and projections. -- [`workspace-hygiene`](./workspace-hygiene.md) for managed Git worktrees and safe cleanup. - -For the broader organization map, start at [Beyond10x getting started](https://beyond10x.github.io/getting-started/). diff --git a/website/docs/plugins/connectors.md b/website/docs/plugins/connectors.md index 50a1505..60f6c67 100644 --- a/website/docs/plugins/connectors.md +++ b/website/docs/plugins/connectors.md @@ -4,7 +4,7 @@ title: Connectors # Connectors -The `connectors` plugin provides one shared `connectors` skill for Claude Code and Codex. It +The `connectors` plugin provides one shared `integrating` skill for Claude Code and Codex. It guides provider setup, connection diagnostics, and the search → describe → invoke sequence for admitted integrations. It ships no binary, credentials, daemon, hooks, or automatic MCP connection. @@ -29,27 +29,25 @@ rate-limit responses. ## Install in either host -Release `0.9.2` includes this plugin. For a fresh marketplace registration, use the release pin: +Setup offers this plugin as optional. To add it by hand: ```bash -claude plugin marketplace add https://github.com/beyond10x/agentplugins.git#0.9.2 -claude plugin install connectors@beyond10x +claude plugin marketplace add beyond10x/agentplugins +claude plugin install connectors@b10x ``` ```bash -codex plugin marketplace add https://github.com/beyond10x/agentplugins.git --ref 0.9.2 -codex plugin add connectors@beyond10x +codex plugin marketplace add beyond10x/agentplugins +codex plugin add connectors@b10x ``` -An existing `beyond10x` registration pinned to `0.7.0` must be repointed before it can offer this -plugin; refreshing an immutable tag does not add newer content. Follow the -[upgrade instructions](../install.md) and preserve the other installed plugins when changing -that registration. For development, both marketplace-add commands also accept the absolute path +[Setup](../install.md) replaces an older or pinned registration and keeps the other installed +plugins. For development, both marketplace-add commands also accept the absolute path to a current local checkout containing both marketplace files. Reload Claude Code's plugins with `/reload-plugins`, or start a new Codex thread. In Claude Code, -invoke `/connectors:connectors`; in Codex select the `connectors` skill or invoke `$connectors:connectors`. -Both manifests load the same `skills/connectors/SKILL.md` bytes. These layouts follow the +invoke `/connectors:integrating`; in Codex select the `integrating` skill or invoke `$connectors:integrating`. +Both manifests load the same `skills/integrating/SKILL.md` bytes. These layouts follow the [OpenAI plugin packaging contract](https://developers.openai.com/plugins/build/plugins) and [Claude Code plugin reference](https://code.claude.com/docs/en/plugins-reference). diff --git a/website/docs/plugins/ess-specify.md b/website/docs/plugins/ess-specify.md deleted file mode 100644 index bb6ae83..0000000 --- a/website/docs/plugins/ess-specify.md +++ /dev/null @@ -1,21 +0,0 @@ ---- -title: ESS Specify ---- - -# `ess-specify` - -Use this plugin when an agent works with an Executable System Specification or a supported -projection. - -Its `specify` skill guides the agent to: - -- validate before compiling or projecting; -- preserve ordered, deterministic output; -- use typed ESS structures instead of arbitrary property bags; -- report unresolved references and unsupported constructs explicitly; -- distinguish semantic round trips from source-text reproduction. - -The plugin does not import credentials, contact a live system, or claim universal reversibility. -Concrete adapters remain responsible for declaring their supported kinds and directions. - -For the model and command reference, use the [ESS documentation](https://beyond10x.github.io/ess/). diff --git a/website/docs/plugins/ess.md b/website/docs/plugins/ess.md new file mode 100644 index 0000000..640cdf5 --- /dev/null +++ b/website/docs/plugins/ess.md @@ -0,0 +1,28 @@ +--- +sidebar_position: 4 +title: ESS +--- + +# `ess` + +Use this plugin to write, retrofit, validate and project an Executable System Specification, and +to raise or audit a conformance suite against a real implementation. + +| skill | for | agent | +|---|---|---| +| `ess:init` | install the `ess` CLI, learn what ESS is, take the first step | — | +| `ess:specifying` | write or extend a specification; `references/syntax.md` shows every section in one that validates | `author` | +| `ess:retrofitting` | derive a specification for a system that has none | `retrofitter` | +| `ess:testing-conformance` | raise or audit what a conformance suite tests | `conformance` | +| `ess:hardening` | after a green suite, the eight techniques that ask what it cannot; `references/` holds each procedure, the reference-model pattern, a design-review brief and spec-diff classification | — | +| `ess:upgrade` | check the plugin and CLI, offer the upgrade | — | + +```text +/plugin install ess@b10x +``` + +Codex: `codex plugin add ess@b10x`. Then `/ess:init` installs the CLI — the prebuilt, +checksummed archive from the [ESS release](https://github.com/beyond10x/ess/releases), or with `cargo` on request. + +In a headless run (`claude -p`), allow the CLI or nothing can be validated: +`--allowedTools "Bash(ess:*)"`. diff --git a/website/docs/plugins/workspace-hygiene.md b/website/docs/plugins/workspace-hygiene.md deleted file mode 100644 index bdb90ac..0000000 --- a/website/docs/plugins/workspace-hygiene.md +++ /dev/null @@ -1,49 +0,0 @@ ---- -sidebar_position: 5 -title: Workspace Hygiene ---- - -# `workspace-hygiene` - -Use this plugin whenever an agent needs an isolated Git checkout, hands work to another session, -or audits old linked worktrees. It contributes the `worktree` skill generated by the public -[`beyond10x/worktree`](https://github.com/beyond10x/worktree) CLI. - -The executable remains a separately installed Rust tool: - -```bash -cargo install --git https://github.com/beyond10x/worktree --tag 0.4.0 b10x-worktree-cli -worktree activate --profile --workspace -worktree doctor --check -``` - -The skill teaches agents to create trees outside the primary repository collection, maintain live -session leases, and publish wanted commits before finishing. Cleanup is review-bound: inspect -`worktree gc --repo --dry-run`, then pass only ids from that review to -`worktree gc --repo --apply --id `. Dirty, locked, live, local-only, -offline, unmanaged, and out-of-policy trees are retained. - -Worktree 0.4.0 adds `worktree inspect --repo ` for actual Git state, storage, -ignored files, leases and retention blockers. It defaults to one repository; use `--workspace` -to inspect the wider profile. Add `--refresh` for current remote recovery evidence. Inspection -does not infer story completion or authorize removal. - -When the host does not run hooks, the agent explicitly acquires and renews its lease, then -releases its own lease before finishing. After verification, it preserves useful evidence and -removes only its own reproducible build output. Each session ends with verified cleanup or an -explicit handoff that names the retained tree, published commit, blockers and next owner. - -Use `worktree reconcile --repo --dry-run` for interrupted provisioning, adopted legacy -paths, finished external trees, and already-missing records. Apply only exact reviewed ids. An -interrupted removal keeps durable recovery proof, while a missing Active record without matching -intent remains refused. External retirement additionally requires the explicit -`--allow-external-retirement` acknowledgement. - -For a finished external legacy tree stranded by stale pre-0.3 relocation intent, act only when the -reconciliation dry-run itself proposes `retire-external`. Worktree requires the exact source and -HEAD to agree and the intended destination to be absent; a cross-device or ambiguous-relocation -refusal is not permission to reinterpret or clear the state by hand. - -The plugin contains no cleanup script and no independent policy copy. Update the source CLI and run -`worktree skill --out plugins/workspace-hygiene/skills/worktree` to refresh the shared Codex and -Claude-compatible guidance. diff --git a/website/docs/plugins/worktree.md b/website/docs/plugins/worktree.md new file mode 100644 index 0000000..92521ed --- /dev/null +++ b/website/docs/plugins/worktree.md @@ -0,0 +1,48 @@ +--- +sidebar_position: 5 +title: Worktree +--- + +# `worktree` + +Use this plugin whenever an agent needs an isolated Git checkout, hands work to another session, +or audits old linked worktrees. + +| skill | for | +|---|---| +| `worktree:init` | install the `worktree` CLI, activate a workspace, check it | +| `worktree:managing-worktrees` | create, lease, finish, inspect and clean worktrees | +| `worktree:upgrade` | check the plugin and CLI, offer the upgrade | + +```text +/plugin install worktree@b10x +``` + +Codex: `codex plugin add worktree@b10x`. Then `/worktree:init` installs the CLI — the prebuilt, +checksummed archive from the [worktree release](https://github.com/beyond10x/worktree/releases), or with `cargo` on request — +and runs: + +```bash +worktree activate --profile --workspace +worktree doctor --check +``` + +The skill teaches agents to create trees outside the primary repository collection, maintain live +session leases, and publish wanted commits before finishing. Cleanup is review-bound: inspect +`worktree gc --repo --dry-run`, then pass only ids from that review to +`worktree gc --repo --apply --id `. Dirty, locked, live, local-only, +offline, unmanaged, and out-of-policy trees are retained. Work merged as rebased or cherry-picked +copies is recoverable when an advertised ref carries every unique commit's exact patch. + +`worktree inspect --repo ` reports actual Git state, storage, ignored files, leases and +retention blockers. It defaults to one repository; use `--workspace` to inspect the wider profile. +Add `--refresh` for current remote recovery evidence. Inspection does not infer story completion +or authorize removal. + +Use `worktree reconcile --repo --dry-run` for interrupted provisioning, adopted legacy +paths, finished external trees, and already-missing records. Apply only exact reviewed ids. +External retirement additionally requires the explicit `--allow-external-retirement` +acknowledgement. + +The plugin contains no cleanup script and no independent policy copy; `agentplugins-check tools` +checks every command it spells against the newest `worktree` release. diff --git a/website/docs/structure.md b/website/docs/structure.md new file mode 100644 index 0000000..40f061e --- /dev/null +++ b/website/docs/structure.md @@ -0,0 +1,33 @@ +--- +sidebar_position: 5 +title: How the plugins are organised +--- + +# How the plugins are organised + +Eight rules decide where everything goes. `task check` enforces each one, and `agentplugins-check +tools` checks the skills against the newest CLI releases. A change that breaks a rule does not merge. + +| rule | what it says | +|---|---| +| **R1 marketplace** | One marketplace, `b10x`, in both the Claude Code and Codex formats. Every plugin lives in this repository. | +| **R2 plugin** | One plugin per product. The plugin, the product and the CLI it drives share one name: `aep`, `ess`, `worktree`, `connectors`. The front door is `b10x`, with the `b10x` CLI. | +| **R3 skill** | Every plugin has two lifecycle skills: `init` (set it up and take the first step) and `upgrade` (check it and offer the upgrade). Every other skill is an activity, named in `-ing` form, one or two words: `aep:planning`. | +| **R4 agent** | An agent is a role: `implementor`, `author`. Exactly one skill of the same plugin owns it and lists it under `## Agents`. | +| **R5 content** | A skill describes its CLI's newest release and quotes no CLI version. `agentplugins-check tools` runs every spelled command against that release. | +| **R6 distribution** | `SETUP.md` and `b10x` install everything; CLIs come prebuilt or from `cargo`. Retired names live only in `catalog.json`, and setup migrates them. | +| **R7 references** | Every `:` written in this repository names a file that exists. | +| **R8 docs** | One README row, one page under `plugins/` and one sidebar entry per plugin. The README is one paragraph, that table, and the generated tree of every skill and agent, each linked to its file. | + +## The plugins + +| plugin | lifecycle | activities | agents | +|---|---|---|---| +| `b10x` | `init` (guided onboarding), `upgrade` | `routing`, `authoring-plugins` | — | +| `aep` | `init`, `upgrade` | `planning`, `migrating`, `implementing` (wave or drive mode) | `planning`: decomposer, four plan critics, plan reviewer, reverse engineer · `implementing`: story scoper, implementor, adversary, security reviewer | +| `ess` | `init`, `upgrade` | `specifying`, `retrofitting`, `testing-conformance`, `hardening` | `specifying`: author · `retrofitting`: retrofitter · `testing-conformance`: conformance | +| `worktree` | `init`, `upgrade` | `managing-worktrees` | — | +| `connectors` | `init`, `upgrade` | `integrating` | — | + +Every plugin carries this repository's version. The CLIs have their own versions; `b10x` installs +their newest release and `b10x check` says at session start when one is behind. diff --git a/website/docusaurus.config.ts b/website/docusaurus.config.ts index 3ec0118..0295fb1 100644 --- a/website/docusaurus.config.ts +++ b/website/docusaurus.config.ts @@ -46,7 +46,7 @@ const config: Config = { ...ecosystemNavbarItems(), {to: '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/docs/choose-a-plugin', label: 'Choose', position: 'left'}, {to: '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/docs/install', label: 'Install', position: 'left'}, - {to: '/docs/plugins/beyond10x', label: 'Reference', position: 'left'}, + {to: '/docs/plugins/b10x', label: 'Reference', position: 'left'}, {href: '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/beyond10x/agentplugins', label: 'GitHub', position: 'right'}, ], }, @@ -59,10 +59,10 @@ const config: Config = { {label: 'Install the marketplace', to: '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/docs/install'}, ]}, {title: 'Plugins', items: [ - {label: 'Beyond10x', to: '/docs/plugins/beyond10x'}, - {label: 'AEP Plan', to: '/docs/plugins/aep-plan'}, - {label: 'AEP Drive', to: '/docs/plugins/aep-drive'}, - {label: 'ESS Specify', to: '/docs/plugins/ess-specify'}, + {label: 'b10x', to: '/docs/plugins/b10x'}, + {label: 'ESS', to: '/docs/plugins/ess'}, + {label: 'Worktree', to: '/docs/plugins/worktree'}, + {label: 'AEP', to: '/docs/plugins/aep'}, ]}, {title: 'Project', items: [ {label: 'GitHub repository', href: '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/beyond10x/agentplugins'}, diff --git a/website/sidebars.ts b/website/sidebars.ts index 3f56122..6acbdd9 100644 --- a/website/sidebars.ts +++ b/website/sidebars.ts @@ -6,10 +6,11 @@ const sidebars: SidebarsConfig = { 'choose-a-plugin', 'install', 'golden-path', + 'structure', { type: 'category', label: 'Plugin reference', - items: ['plugins/beyond10x', 'plugins/aep-plan', 'plugins/aep-drive', 'plugins/ess-specify', 'plugins/workspace-hygiene', 'plugins/connectors'], + items: ['plugins/b10x', 'plugins/aep', 'plugins/ess', 'plugins/worktree', 'plugins/connectors'], }, 'trust-and-scope', ], diff --git a/website/src/pages/index.tsx b/website/src/pages/index.tsx index c326fe8..6b7dfac 100644 --- a/website/src/pages/index.tsx +++ b/website/src/pages/index.tsx @@ -5,11 +5,10 @@ import Heading from '@theme/Heading'; import styles from './index.module.css'; const plugins = [ - ['Beyond10x', 'Route work, find public resources, and create portable plugins.', '/docs/plugins/beyond10x'], - ['AEP Plan', 'Plan, decompose, review, and reverse-engineer governed work.', '/docs/plugins/aep-plan'], - ['AEP Drive', 'Scope stories and coordinate implementation waves with adversarial review.', '/docs/plugins/aep-drive'], - ['ESS Specify', 'Specify typed systems and guide deterministic schema projections.', '/docs/plugins/ess-specify'], - ['Workspace Hygiene', 'Manage isolated Git worktrees with explicit recovery proof.', '/docs/plugins/workspace-hygiene'], + ['b10x', 'Set up and upgrade the plugins, route work, and create portable plugins.', '/docs/plugins/b10x'], + ['ESS', 'Write, retrofit and conformance-test Executable System Specifications.', '/docs/plugins/ess'], + ['AEP', 'Plan governed work, then scope and deliver it in waves with adversarial review.', '/docs/plugins/aep'], + ['Worktree', 'Manage isolated Git worktrees with explicit recovery proof.', '/docs/plugins/worktree'], ]; export default function Home(): ReactNode { @@ -17,7 +16,7 @@ export default function Home(): ReactNode {
-

Curated marketplace · beyond10x

+

Curated marketplace · b10x

Give each agent only the engineering guidance it needs.

A lightweight front door routes work to four focused specialists and helps create diff --git a/website/static/img/social-card.svg b/website/static/img/social-card.svg index 64959c2..a7befcb 100644 --- a/website/static/img/social-card.svg +++ b/website/static/img/social-card.svg @@ -5,5 +5,5 @@ BEYOND10X Agent Plugins Focused guidance for governed engineering agents. - aep-plan · aep-drive · ess-specify + aep · ess · worktree