From 1eac2fd8bfc1489769d7ffd0560d345fc009be57 Mon Sep 17 00:00:00 2001 From: Relayflow Lead Date: Tue, 22 Sep 2026 22:08:20 -0700 Subject: [PATCH 1/6] =?UTF-8?q?feat(examples):=20prompt-lab=20=E2=80=94=20?= =?UTF-8?q?the=20Prompt=20Lab=20product=20brief=20as=20one=20relayflow?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Job 1 (new agency), Job 2 (detect and fix one question) and the test-patient creator from the brief, each job file its diagram line by line: deterministic steps for System boxes, f.human gates asked of input.reviewer for You boxes, lab-store writes for Outcome boxes. Apricot's Bank is a local JSON lab written only by an idempotent store CLI; every read and write is a journaled f.run. prove.sh drives all three jobs end to end locally with real Claude calls and captures every command's output under evidence/run. Runtime defects found on the way are captured under evidence/runtime-findings, with workarounds commented where they live. Co-Authored-By: Claude Opus 5.5 (1M context) --- examples/prompt-lab/README.md | 150 ++ .../prompt-lab/evidence/run/01-job1-run.txt | 84 + .../evidence/run/02-reviewer-edit.diff | 13 + .../prompt-lab/evidence/run/03-answer.txt | 4 + .../prompt-lab/evidence/run/04-resume.txt | 60 + .../evidence/run/05-reviewer-edit.diff | 10 + .../prompt-lab/evidence/run/06-answer.txt | 4 + .../prompt-lab/evidence/run/07-resume.txt | 59 + .../evidence/run/08-patient-run.txt | 28 + .../prompt-lab/evidence/run/09-answer.txt | 4 + .../prompt-lab/evidence/run/10-resume.txt | 41 + .../prompt-lab/evidence/run/11-job2-run.txt | 34 + .../prompt-lab/evidence/run/12-answer.txt | 4 + .../prompt-lab/evidence/run/13-resume.txt | 80 + .../prompt-lab/evidence/run/14-answer.txt | 4 + .../prompt-lab/evidence/run/15-resume.txt | 49 + .../prompt-lab/evidence/run/16-lab-state.txt | 16 + .../prompt-lab/evidence/run/lab/agencies.json | 30 + .../prompt-lab/evidence/run/lab/bank.json | 34 + .../prompt-lab/evidence/run/lab/gold.json | 22 + .../prompt-lab/evidence/run/lab/guidelines.md | 9 + .../evidence/run/lab/queue/issues.json | 20 + .../run/lab/queue/patient-briefs.json | 9 + .../evidence/run/lab/shelf/jordan.json | 6 + .../evidence/run/lab/shelf/marguerite.json | 10 + .../evidence/run/lab/shelf/pat.json | 6 + .../evidence/run/lab/shelf/riley.json | 6 + .../prompt-lab/evidence/run/lab/targets.json | 47 + .../patients/gap-ostomy-supplies/plan.json | 4 + .../work/q-wound-status/d61687f9/grid.json | 72 + .../work/q-wound-status/d61687f9/score.json | 54 + .../lab/work/sunrise-soc/62c71a0f/commit.json | 6 + .../lab/work/sunrise-soc/62c71a0f/grid.json | 168 ++ .../00-job1-attempt1-fenced-json.txt | 57 + ...0-job1-attempt2-parallel-lease-expired.txt | 56 + ...-job1-attempt3-capacity1-lease-expired.txt | 56 + ...ttempt4-predicate-gate-resume-conflict.txt | 27 + .../runtime-findings/00-job1-attempt4-run.txt | 73 + .../00-patient-attempt1-fenced-json.txt | 29 + .../00-patient-attempt1-run.txt | 19 + ...-attempt1-lease-conflict-after-success.txt | 25 + .../00-sonnet-run-pat-healed-high.txt | 7 + .../runtime-parallel-llm-repro.flow.ts | 8 + .../runtime-parallel-llm-repro.txt | 13 + examples/prompt-lab/fixtures/agencies.json | 30 + examples/prompt-lab/fixtures/bank.json | 12 + examples/prompt-lab/fixtures/guidelines.md | 9 + .../prompt-lab/fixtures/shelf/jordan.json | 6 + examples/prompt-lab/fixtures/shelf/pat.json | 6 + examples/prompt-lab/fixtures/shelf/riley.json | 6 + examples/prompt-lab/flows.json | 1 + examples/prompt-lab/jobs/fix.ts | 104 + examples/prompt-lab/jobs/job.ts | 5 + examples/prompt-lab/jobs/new-agency.ts | 144 ++ examples/prompt-lab/jobs/patient.ts | 58 + examples/prompt-lab/jobs/shared.ts | 98 + examples/prompt-lab/lib/commit.ts | 17 + examples/prompt-lab/lib/grid.ts | 50 + examples/prompt-lab/lib/hash.ts | 4 + examples/prompt-lab/lib/lab.ts | 58 + examples/prompt-lab/lib/phi.ts | 13 + examples/prompt-lab/lib/piles.ts | 39 + examples/prompt-lab/lib/reply.ts | 54 + examples/prompt-lab/lib/score.ts | 25 + examples/prompt-lab/lib/types.ts | 57 + examples/prompt-lab/package-lock.json | 2257 +++++++++++++++++ examples/prompt-lab/package.json | 13 + examples/prompt-lab/prompt-lab.flow.ts | 40 + examples/prompt-lab/prompts.ts | 170 ++ examples/prompt-lab/prove.sh | 84 + examples/prompt-lab/store.ts | 157 ++ examples/prompt-lab/tests/lib.test.ts | 113 + examples/prompt-lab/tests/store.test.ts | 71 + examples/prompt-lab/tsconfig.json | 9 + 74 files changed, 5227 insertions(+) create mode 100644 examples/prompt-lab/README.md create mode 100644 examples/prompt-lab/evidence/run/01-job1-run.txt create mode 100644 examples/prompt-lab/evidence/run/02-reviewer-edit.diff create mode 100644 examples/prompt-lab/evidence/run/03-answer.txt create mode 100644 examples/prompt-lab/evidence/run/04-resume.txt create mode 100644 examples/prompt-lab/evidence/run/05-reviewer-edit.diff create mode 100644 examples/prompt-lab/evidence/run/06-answer.txt create mode 100644 examples/prompt-lab/evidence/run/07-resume.txt create mode 100644 examples/prompt-lab/evidence/run/08-patient-run.txt create mode 100644 examples/prompt-lab/evidence/run/09-answer.txt create mode 100644 examples/prompt-lab/evidence/run/10-resume.txt create mode 100644 examples/prompt-lab/evidence/run/11-job2-run.txt create mode 100644 examples/prompt-lab/evidence/run/12-answer.txt create mode 100644 examples/prompt-lab/evidence/run/13-resume.txt create mode 100644 examples/prompt-lab/evidence/run/14-answer.txt create mode 100644 examples/prompt-lab/evidence/run/15-resume.txt create mode 100644 examples/prompt-lab/evidence/run/16-lab-state.txt create mode 100644 examples/prompt-lab/evidence/run/lab/agencies.json create mode 100644 examples/prompt-lab/evidence/run/lab/bank.json create mode 100644 examples/prompt-lab/evidence/run/lab/gold.json create mode 100644 examples/prompt-lab/evidence/run/lab/guidelines.md create mode 100644 examples/prompt-lab/evidence/run/lab/queue/issues.json create mode 100644 examples/prompt-lab/evidence/run/lab/queue/patient-briefs.json create mode 100644 examples/prompt-lab/evidence/run/lab/shelf/jordan.json create mode 100644 examples/prompt-lab/evidence/run/lab/shelf/marguerite.json create mode 100644 examples/prompt-lab/evidence/run/lab/shelf/pat.json create mode 100644 examples/prompt-lab/evidence/run/lab/shelf/riley.json create mode 100644 examples/prompt-lab/evidence/run/lab/targets.json create mode 100644 examples/prompt-lab/evidence/run/lab/work/patients/gap-ostomy-supplies/plan.json create mode 100644 examples/prompt-lab/evidence/run/lab/work/q-wound-status/d61687f9/grid.json create mode 100644 examples/prompt-lab/evidence/run/lab/work/q-wound-status/d61687f9/score.json create mode 100644 examples/prompt-lab/evidence/run/lab/work/sunrise-soc/62c71a0f/commit.json create mode 100644 examples/prompt-lab/evidence/run/lab/work/sunrise-soc/62c71a0f/grid.json create mode 100644 examples/prompt-lab/evidence/runtime-findings/00-job1-attempt1-fenced-json.txt create mode 100644 examples/prompt-lab/evidence/runtime-findings/00-job1-attempt2-parallel-lease-expired.txt create mode 100644 examples/prompt-lab/evidence/runtime-findings/00-job1-attempt3-capacity1-lease-expired.txt create mode 100644 examples/prompt-lab/evidence/runtime-findings/00-job1-attempt4-predicate-gate-resume-conflict.txt create mode 100644 examples/prompt-lab/evidence/runtime-findings/00-job1-attempt4-run.txt create mode 100644 examples/prompt-lab/evidence/runtime-findings/00-patient-attempt1-fenced-json.txt create mode 100644 examples/prompt-lab/evidence/runtime-findings/00-patient-attempt1-run.txt create mode 100644 examples/prompt-lab/evidence/runtime-findings/00-prove-attempt1-lease-conflict-after-success.txt create mode 100644 examples/prompt-lab/evidence/runtime-findings/00-sonnet-run-pat-healed-high.txt create mode 100644 examples/prompt-lab/evidence/runtime-findings/runtime-parallel-llm-repro.flow.ts create mode 100644 examples/prompt-lab/evidence/runtime-findings/runtime-parallel-llm-repro.txt create mode 100644 examples/prompt-lab/fixtures/agencies.json create mode 100644 examples/prompt-lab/fixtures/bank.json create mode 100644 examples/prompt-lab/fixtures/guidelines.md create mode 100644 examples/prompt-lab/fixtures/shelf/jordan.json create mode 100644 examples/prompt-lab/fixtures/shelf/pat.json create mode 100644 examples/prompt-lab/fixtures/shelf/riley.json create mode 100644 examples/prompt-lab/flows.json create mode 100644 examples/prompt-lab/jobs/fix.ts create mode 100644 examples/prompt-lab/jobs/job.ts create mode 100644 examples/prompt-lab/jobs/new-agency.ts create mode 100644 examples/prompt-lab/jobs/patient.ts create mode 100644 examples/prompt-lab/jobs/shared.ts create mode 100644 examples/prompt-lab/lib/commit.ts create mode 100644 examples/prompt-lab/lib/grid.ts create mode 100644 examples/prompt-lab/lib/hash.ts create mode 100644 examples/prompt-lab/lib/lab.ts create mode 100644 examples/prompt-lab/lib/phi.ts create mode 100644 examples/prompt-lab/lib/piles.ts create mode 100644 examples/prompt-lab/lib/reply.ts create mode 100644 examples/prompt-lab/lib/score.ts create mode 100644 examples/prompt-lab/lib/types.ts create mode 100644 examples/prompt-lab/package-lock.json create mode 100644 examples/prompt-lab/package.json create mode 100644 examples/prompt-lab/prompt-lab.flow.ts create mode 100644 examples/prompt-lab/prompts.ts create mode 100755 examples/prompt-lab/prove.sh create mode 100644 examples/prompt-lab/store.ts create mode 100644 examples/prompt-lab/tests/lib.test.ts create mode 100644 examples/prompt-lab/tests/store.test.ts create mode 100644 examples/prompt-lab/tsconfig.json diff --git a/examples/prompt-lab/README.md b/examples/prompt-lab/README.md new file mode 100644 index 00000000..2a46e73a --- /dev/null +++ b/examples/prompt-lab/README.md @@ -0,0 +1,150 @@ +# prompt-lab + +The Prompt Lab product brief as one relayflow. Prompt Lab is the workbench for +writing and fixing the prompts that draft home-health charts. The brief defines +two jobs and a shelf of fake patients. This flow covers all three: + +| `job` | Brief section | What it does | +| --- | --- | --- | +| `new-agency` | Job 1 · Config-level / new agency | Sorts the agency's questions into sharing piles, writes first-pass prompts for agency-specific questions, runs them on shelf patients, and puts the output in an edit grid. Your first pass is saved as targets. Iteration runs on changed rows. Agency-specific prompts are committed (all / only / all except). Shared rows go to the question manager. | +| `fix` | Job 2 · Question-level refinement | Takes one question, from an issue or picked directly. Runs it on shelf patients, and your review of the grid is saved as gold. The iterator rewrites the prompt, Prompt QA checks it, and it re-runs and scores against gold (% worked, per patient). Marking it done makes it live. | +| `patient` | Set up test patient | Takes a gap brief from the manager queue. An agent writes a patient plan and you kick generate. Patient QA loops until it passes, then locks the patient onto the shelf. | + +Each job file is its brief diagram, line by line: + +- **System** boxes are either deterministic TypeScript over journaled reads, or model calls. +- **You** boxes are `f.human` gates, asked of `input.reviewer`. +- **Outcome** boxes are writes to the lab store. + +## What maps to what + +| Brief | Here | +| --- | --- | +| Apricot's Bank (`livePromptId`, global prompts) | `bank.json` in the lab directory, written only by [`store.ts`](store.ts) | +| The chart-filling engine nurses use | an `llm` step given the live prompt and the patient. Its answer must be on the agency's menu (the schema's `enum`) | +| Sharing piles: a deterministic lookup, never an agent | [`lib/piles.ts`](lib/piles.ts) | +| Computer-highlighted rows | [`lib/grid.ts`](lib/grid.ts) `highlights`: confidence below High, a mismatch pile, or a never-reviewed first-pass prompt | +| First pass persists as targets / gold | `store.ts record`, fed only from the grid you reviewed. AI output never writes gold directly | +| Iterator: rewrite-only | [`prompts.ts`](prompts.ts) `iterate`. Input is the changeset, patients, brief and existing prompt; output is new prompt text | +| Prompt QA: the brief plus shared guidelines | `promptQa`, looping with the iterator. Capped at 3 tries, then the run parks as `needs_human` | +| Done = live | `store.ts publish` sets `livePromptId`. There is no promote step. A shared question warns which agencies it will change | +| Shared frozen at config level | a changed shared or mismatch row becomes a `config-send` issue that carries its targets and a proposed rewrite | +| Test planner: never invents a patient | picks shelf ids (checked deterministically); each hole becomes a gap brief | +| You do not approve the chart | the only patient gate is *kick generate*. Patient QA, plus a deterministic identifier check ([`lib/phi.ts`](lib/phi.ts)), locks it | + +Every read and write of the lab is a journaled `f.run` step. Every store verb +is idempotent, so a retried step lands the lab in the same state. `write-new` +never overwrites an edit you made. + +**Not built:** anything the brief lists under "Not at the start". Also not +built: + +- Drafting a question brief with an agent. The flow reads a brief when one + exists in `briefs/.md`. +- The Apricot patient-brief generator. It runs on live patients, so it belongs + in Apricot. +- The UI screens. + +The flow is the job graph that sits under those screens. + +## Run it + +```sh +npm install +npm test # 16 unit tests over the deterministic parts +node --experimental-strip-types store.ts ./my-lab seed fixtures +npx flows run prompt-lab.flow.ts --local-agent --input \ + '{"job":"new-agency","reviewer":"","lab":"./my-lab","agency":"sunrise","visitType":"soc"}' +``` + +`reviewer` and `lab` are required, and there is no default person. The run +parks at each gate and prints the file to edit plus the `flows answer` / +`flows resume` commands. Edit the file, answer `yes`, resume. Answering `no` +stops the job as `declined` and keeps what was saved. + +The fixtures are invented and reproduce the brief's own examples. The new +agency `sunrise` asks four questions: + +- **wound-status**: shared with harbor and maple, with an identical menu. +- **mood**: a shared prompt, but sunrise adds "Agitated" to the menu, so it's a mismatch. +- **living-situation**: agency-specific, with no prompt yet. +- **ostomy-supplies**: agency-specific, with no shelf patient. + +The shelf holds Pat, Jordan and Riley. The shared wound prompt contains a +deliberate flaw: it lets the referral overrule today's visit notes. + +## Proof + +[`prove.sh`](prove.sh) seeds a fresh lab and drives all three jobs through the +real kernel with real Claude calls (`--local-agent`). At each gate it acts as +the reviewer and applies the edit described in its comments, captured as a +diff. It captures every command with its output and exit code in +[`evidence/run/`](evidence/run/), and the final lab lands in +`evidence/run/lab/`. + +```sh +./prove.sh evidence/run +``` + +The captured run (Claude Code 2.1.280, the adapter's default model): + +| Run | Result | What happened | +| --- | --- | --- | +| Job 1 · `sunrise` / `soc` ([01](evidence/run/01-job1-run.txt), [04](evidence/run/04-resume.txt), [07](evidence/run/07-resume.txt)) | 29 steps, `success` | Piles came out as shared / mismatch / agency-specific. Both agency-specific questions got first-pass prompts that passed Prompt QA. The planner covered 3 questions and queued `gap-ostomy-supplies`. Gate 1: the reviewer rewrote Pat's wound explanation and added a note ([02](evidence/run/02-reviewer-edit.diff)). That shared row went to the question manager with its target. Gate 2: committed all except `ostomy-supplies`, which no patient has exercised yet ([05](evidence/run/05-reviewer-edit.diff)). `living-situation` went live. | +| Patient · `gap-ostomy-supplies` ([08](evidence/run/08-patient-run.txt), [10](evidence/run/10-resume.txt)) | 9 steps, `success` | Plan, then kick generate. The chart passed Patient QA on its first try and `marguerite` locked onto the shelf. The brief is marked `locked`. | +| Job 2 · the config-send issue ([11](evidence/run/11-job2-run.txt), [13](evidence/run/13-resume.txt), [15](evidence/run/15-resume.txt)) | 24 steps, `success` | Four shelf patients, including the new one. Gold came prefilled, with Pat's config target carried over. The iterator's rewrite passed Prompt QA. Re-run scored 4 of 4 golded patients worked (100%), and the gate warned it would change harbor, maple and sunrise. Done made the new prompt live and closed the issue. | + +Final state: [16-lab-state.txt](evidence/run/16-lab-state.txt), with the whole lab in `evidence/run/lab/`. + +**What this run does not show:** an answer flipping from wrong to right. Here +the model answered Pat "Ongoing" even under the flawed wound prompt, so Job 2 +iterated on the explanation and on source priority, not on the answer. The +brief's exact failure, "Healed / High" for Pat, did occur in an earlier +`claude-sonnet-5` run of the same prompt. Its engine step's journal is in +[00-sonnet-run-pat-healed-high.txt](evidence/runtime-findings/00-sonnet-run-pat-healed-high.txt). +The reviewer at every gate is `prove.sh`, applying fixed edits. It is not a +clinician's judgment. + +## Runtime findings (relayflows 2.0.29) + +Four things in the runtime shaped this flow or its proof. Each workaround is commented +where it lives, and each has captured evidence in +[`evidence/runtime-findings/`](evidence/runtime-findings/): + +1. **More than a few concurrent `f.llm` calls lose the run.** The queued + calls' 30 s leases expire before the worker takes them. The late completion + of a dead attempt is then refused ("Agent lease is already expired"), and + the CLI turns that refusal into a fatal `protocol_error`. Five parallel + calls passed and nine failed, with `--agent-capacity` 4 or 1. + [`runtime-parallel-llm-repro.flow.ts`](evidence/runtime-findings/runtime-parallel-llm-repro.flow.ts) + reproduces it with no Prompt Lab code. **Workaround:** model calls run + sequentially. +2. **A predicate `.gate(fn)` before an `f.human` can't be resumed.** The + verdict is read back from the `predicate-gates` stream with its keys + re-ordered. The lowered `.gate` command no longer matches, and the + resume is refused as `run_admission_conflict`. The fix is one line in + `packages/sdk/src/authored-flow-executor.ts` `applyPredicateGate`: build the + literal from fixed fields, not from `JSON.stringify(record)`. + **Workaround:** the checks run in the body and fail through a journaled + failing step (`failStep`). +3. **`f.llm(prompt, { output })` fails when the reply is fenced JSON.** The + worker validates the raw reply. Sonnet sometimes wraps valid JSON in + ```` ```json ```` anyway, and a failed run can't be resumed. + **Workaround:** text-form `f.llm`, then + [`lib/reply.ts`](lib/reply.ts) strips one fence and validates the schema, + with one bounded re-ask. The text form takes no `model`, so calls use the + Claude adapter's default model. + +4. **A lease renewal that races a completion kills the run.** A step whose + child run journaled `success` was reported as + `lease_conflict: attempt has no active worker lease`. The CLI made that a + fatal `protocol_error` + ([00-prove-attempt1-lease-conflict-after-success.txt](evidence/runtime-findings/00-prove-attempt1-lease-conflict-after-success.txt)). + It's intermittent: it happened once in about 50 sequential calls. There's + no workaround in the flow, so rerun. + +Findings 1 and 4 are the same class of problem: late lease traffic becomes +fatal to the whole run instead of being ignored. + +Local only for now: Cloud receives a single authored source, and this flow +imports sibling modules. diff --git a/examples/prompt-lab/evidence/run/01-job1-run.txt b/examples/prompt-lab/evidence/run/01-job1-run.txt new file mode 100644 index 00000000..a5bd9e9f --- /dev/null +++ b/examples/prompt-lab/evidence/run/01-job1-run.txt @@ -0,0 +1,84 @@ +$ npx flows run --no-observer-link --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data --local-agent prompt-lab.flow.ts --input {"job":"new-agency","reviewer":"prompt-lab-reviewer","lab":"evidence/run/lab","agency":"sunrise","visitType":"soc"} +○ run-1 (deterministic) 0.00s +✓ run-1 (deterministic) 0.06s completionReason: success +○ llm-2 (llm) 0.00s +WAITING [worker_lease] Run "01M36695SX9Z217ZK23C7R4XC4" step "llm-2" (llm) is running under a worker lease until 1790135569521. +↻ llm-2 (llm) 3.67s +WAITING [worker_lease] Run "01M36695SX9Z217ZK23C7R4XC4" step "llm-2" (llm) is running under a worker lease until 1790135579526. +↻ llm-2 (llm) 13.68s +✓ llm-2 (llm) 21.24s completionReason: success +○ llm-3 (llm) 0.00s +WAITING [worker_lease] Run "01M3669TF3YG6DC3N7WXK899YW" step "llm-3" (llm) is running under a worker lease until 1790135590679. +↻ llm-3 (llm) 3.59s +WAITING [worker_lease] Run "01M3669TF3YG6DC3N7WXK899YW" step "llm-3" (llm) is running under a worker lease until 1790135600685. +↻ llm-3 (llm) 13.60s +✓ llm-3 (llm) 17.58s completionReason: success +○ run-4 (deterministic) 0.00s +✓ run-4 (deterministic) 0.06s completionReason: success +○ llm-5 (llm) 0.00s +WAITING [worker_lease] Run "01M366ABZWVZVKACSBTK4V0S0S" step "llm-5" (llm) is running under a worker lease until 1790135608622. +↻ llm-5 (llm) 3.89s +WAITING [worker_lease] Run "01M366ABZWVZVKACSBTK4V0S0S" step "llm-5" (llm) is running under a worker lease until 1790135618625. +↻ llm-5 (llm) 13.90s +WAITING [worker_lease] Run "01M366ABZWVZVKACSBTK4V0S0S" step "llm-5" (llm) is running under a worker lease until 1790135628635. +↻ llm-5 (llm) 23.92s +✓ llm-5 (llm) 28.89s completionReason: success +○ llm-6 (llm) 0.00s +WAITING [worker_lease] Run "01M366B8HTX6D89C26FARJH7S7" step "llm-6" (llm) is running under a worker lease until 1790135637870. +↻ llm-6 (llm) 4.24s +✓ llm-6 (llm) 9.91s completionReason: success +○ run-7 (deterministic) 0.00s +✓ run-7 (deterministic) 0.06s completionReason: success +○ llm-8 (llm) 0.00s +WAITING [worker_lease] Run "01M366BM1FQN24G3D8RFC30WBE" step "llm-8" (llm) is running under a worker lease until 1790135649636. +↻ llm-8 (llm) 6.04s +WAITING [worker_lease] Run "01M366BM1FQN24G3D8RFC30WBE" step "llm-8" (llm) is running under a worker lease until 1790135659640. +↻ llm-8 (llm) 16.07s +✓ llm-8 (llm) 24.61s completionReason: success +○ run-9 (deterministic) 0.00s +✓ run-9 (deterministic) 0.06s completionReason: success +○ llm-10 (llm) 0.00s +WAITING [worker_lease] Run "01M366CAWMTF1ZYN956WYE5PT6" step "llm-10" (llm) is running under a worker lease until 1790135673031. +↻ llm-10 (llm) 4.76s +✓ llm-10 (llm) 10.78s completionReason: success +○ llm-11 (llm) 0.00s +WAITING [worker_lease] Run "01M366CMM9N0C5674SRVQVGNDP" step "llm-11" (llm) is running under a worker lease until 1790135683006. +↻ llm-11 (llm) 3.95s +✓ llm-11 (llm) 8.75s completionReason: success +○ llm-12 (llm) 0.00s +WAITING [worker_lease] Run "01M366CYN97BBESM8SDRM3Y7G8" step "llm-12" (llm) is running under a worker lease until 1790135693276. +↻ llm-12 (llm) 5.48s +✓ llm-12 (llm) 9.54s completionReason: success +○ llm-13 (llm) 0.00s +WAITING [worker_lease] Run "01M366D6Y4HVGJNBN1F45HHWJJ" step "llm-13" (llm) is running under a worker lease until 1790135701751. +↻ llm-13 (llm) 4.41s +✓ llm-13 (llm) 8.46s completionReason: success +○ llm-14 (llm) 0.00s +WAITING [worker_lease] Run "01M366DGCFA2F4V97Y920YHH2M" step "llm-14" (llm) is running under a worker lease until 1790135711427. +↻ llm-14 (llm) 5.63s +✓ llm-14 (llm) 9.38s completionReason: success +○ llm-15 (llm) 0.00s +WAITING [worker_lease] Run "01M366DQN7RCZ9R3QP9RP5WJTS" step "llm-15" (llm) is running under a worker lease until 1790135718875. +↻ llm-15 (llm) 3.70s +✓ llm-15 (llm) 7.88s completionReason: success +○ llm-16 (llm) 0.00s +WAITING [worker_lease] Run "01M366DZTJBB1V1G5K7G1RWTWT" step "llm-16" (llm) is running under a worker lease until 1790135727237. +↻ llm-16 (llm) 4.18s +✓ llm-16 (llm) 10.67s completionReason: success +○ llm-17 (llm) 0.00s +WAITING [worker_lease] Run "01M366EBC622QRBM6NAV93Q65K" step "llm-17" (llm) is running under a worker lease until 1790135739066. +↻ llm-17 (llm) 5.34s +✓ llm-17 (llm) 10.03s completionReason: success +○ llm-18 (llm) 0.00s +WAITING [worker_lease] Run "01M366EKED0RJASW9AM0Q2EY28" step "llm-18" (llm) is running under a worker lease until 1790135747331. +↻ llm-18 (llm) 3.57s +✓ llm-18 (llm) 9.69s completionReason: success +○ run-19 (deterministic) 0.00s +✓ run-19 (deterministic) 0.07s completionReason: success +○ human-20 (deterministic) 0.00s +⏸ human-20 (human) 0.00s +PARKED [run_parked] Run "01M3669253ED8QPNX7N7RE7PXF" is waiting for prompt-lab-reviewer to answer human-20: "Config workbench · sunrise soc: 9 rows on 3 questions (6 highlighted, listed first); 1 gap brief(s) queued for the test patient manager.\nEdit target answer / confidence / explanation / notes in evidence/run/lab/work/sunrise-soc/62c71a0f/grid.json. Your first pass persists as the target for each question × patient.\nyes = persist targets and run iteration on every changed row; no = stop without persisting." +Answer with: flows answer --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data 01M3669253ED8QPNX7N7RE7PXF human-20 yes|no +Then continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data --local-agent 01M3669253ED8QPNX7N7RE7PXF +RUN 01M3669253ED8QPNX7N7RE7PXF parked +exit=3 diff --git a/examples/prompt-lab/evidence/run/02-reviewer-edit.diff b/examples/prompt-lab/evidence/run/02-reviewer-edit.diff new file mode 100644 index 00000000..d9ca47c6 --- /dev/null +++ b/examples/prompt-lab/evidence/run/02-reviewer-edit.diff @@ -0,0 +1,13 @@ +# reviewer edit: evidence/run/lab/work/sunrise-soc/62c71a0f/grid.json +@@ -144,9 +144,9 @@ + "target": { + "answer": "Ongoing", + "confidence": "High", +- "explanation": "Today's SOC visit documents an open left heel wound measuring 2.0 x 1.5 cm with moderate serous drainage and yellow slough, dressed with foam per protocol, so the discharge summary's \"wound closed\" note is outdated and the current assessment governs." ++ "explanation": "Today's visit notes describe an open 2.0 x 1.5 cm left heel wound with serous drainage and slough; the discharge summary's closed status is outdated." + }, +- "notes": "" ++ "notes": "The prompt makes the referral win over what the nurse saw today. Today's notes must win (shared guideline 1)." + }, + { + "questionId": "wound-status", diff --git a/examples/prompt-lab/evidence/run/03-answer.txt b/examples/prompt-lab/evidence/run/03-answer.txt new file mode 100644 index 00000000..183ec4f4 --- /dev/null +++ b/examples/prompt-lab/evidence/run/03-answer.txt @@ -0,0 +1,4 @@ +$ npx flows answer --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data 01M3669253ED8QPNX7N7RE7PXF human-20 yes --by prompt-lab-reviewer +ANSWERED 01M3669253ED8QPNX7N7RE7PXF human-20 yes +Continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data --local-agent 01M3669253ED8QPNX7N7RE7PXF +exit=0 diff --git a/examples/prompt-lab/evidence/run/04-resume.txt b/examples/prompt-lab/evidence/run/04-resume.txt new file mode 100644 index 00000000..06a9a034 --- /dev/null +++ b/examples/prompt-lab/evidence/run/04-resume.txt @@ -0,0 +1,60 @@ +$ npx flows resume --no-observer-link --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data --local-agent 01M3669253ED8QPNX7N7RE7PXF +○ run-1 (deterministic) 0.00s +✓ run-1 (deterministic) 0.01s completionReason: success +○ llm-2 (llm) 0.00s +✓ llm-2 (llm) 3.84s completionReason: success +○ llm-3 (llm) 0.00s +✓ llm-3 (llm) 3.87s completionReason: success +○ run-4 (deterministic) 0.00s +✓ run-4 (deterministic) 0.01s completionReason: success +○ llm-5 (llm) 0.00s +✓ llm-5 (llm) 3.80s completionReason: success +○ llm-6 (llm) 0.00s +✓ llm-6 (llm) 4.16s completionReason: success +○ run-7 (deterministic) 0.00s +✓ run-7 (deterministic) 0.01s completionReason: success +○ llm-8 (llm) 0.00s +✓ llm-8 (llm) 3.87s completionReason: success +○ run-9 (deterministic) 0.00s +✓ run-9 (deterministic) 0.01s completionReason: success +○ llm-10 (llm) 0.00s +✓ llm-10 (llm) 3.66s completionReason: success +○ llm-11 (llm) 0.00s +✓ llm-11 (llm) 3.85s completionReason: success +○ llm-12 (llm) 0.00s +✓ llm-12 (llm) 3.67s completionReason: success +○ llm-13 (llm) 0.00s +✓ llm-13 (llm) 3.68s completionReason: success +○ llm-14 (llm) 0.00s +✓ llm-14 (llm) 4.33s completionReason: success +○ llm-15 (llm) 0.00s +✓ llm-15 (llm) 4.92s completionReason: success +○ llm-16 (llm) 0.00s +✓ llm-16 (llm) 4.21s completionReason: success +○ llm-17 (llm) 0.00s +✓ llm-17 (llm) 3.86s completionReason: success +○ llm-18 (llm) 0.00s +✓ llm-18 (llm) 3.77s completionReason: success +○ run-19 (deterministic) 0.00s +✓ run-19 (deterministic) 0.01s completionReason: success +○ human-20 (deterministic) 0.00s +✓ human-20 (deterministic) 0.02s completionReason: success +○ run-21 (deterministic) 0.00s +✓ run-21 (deterministic) 0.06s completionReason: success +○ run-22 (deterministic) 0.00s +✓ run-22 (deterministic) 0.06s completionReason: success +○ llm-23 (llm) 0.00s +WAITING [worker_lease] Run "01M366GMPW32EQPWHW07KEQPC2" step "llm-23" (llm) is running under a worker lease until 1790135814160. +↻ llm-23 (llm) 3.64s +✓ llm-23 (llm) 13.20s completionReason: success +○ run-24 (deterministic) 0.00s +✓ run-24 (deterministic) 0.06s completionReason: success +○ run-25 (deterministic) 0.00s +✓ run-25 (deterministic) 0.06s completionReason: success +○ human-26 (deterministic) 0.00s +⏸ human-26 (human) 0.00s +PARKED [run_parked] Run "01M3669253ED8QPNX7N7RE7PXF" is waiting for prompt-lab-reviewer to answer human-26: "Commit output · sunrise: agency-specific prompts ready to go live in Apricot: living-situation, ostomy-supplies.\nSent to the question manager (frozen here): wound-status → config-sunrise-wound-status-ce89aea7.\nTo commit only some, edit evidence/run/lab/work/sunrise-soc/62c71a0f/commit.json: {\"mode\":\"only\"|\"except\",\"questions\":[...]}.\nyes = commit (Done = live, no promote step); no = leave them as drafts." +Answer with: flows answer --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data 01M3669253ED8QPNX7N7RE7PXF human-26 yes|no +Then continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data --local-agent 01M3669253ED8QPNX7N7RE7PXF +RUN 01M3669253ED8QPNX7N7RE7PXF parked +exit=3 diff --git a/examples/prompt-lab/evidence/run/05-reviewer-edit.diff b/examples/prompt-lab/evidence/run/05-reviewer-edit.diff new file mode 100644 index 00000000..3b99d02a --- /dev/null +++ b/examples/prompt-lab/evidence/run/05-reviewer-edit.diff @@ -0,0 +1,10 @@ +# reviewer edit: evidence/run/lab/work/sunrise-soc/62c71a0f/commit.json +@@ -1,4 +1,6 @@ + { +- "mode": "all", +- "questions": [] ++ "mode": "except", ++ "questions": [ ++ "ostomy-supplies" ++ ] + } diff --git a/examples/prompt-lab/evidence/run/06-answer.txt b/examples/prompt-lab/evidence/run/06-answer.txt new file mode 100644 index 00000000..b5813218 --- /dev/null +++ b/examples/prompt-lab/evidence/run/06-answer.txt @@ -0,0 +1,4 @@ +$ npx flows answer --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data 01M3669253ED8QPNX7N7RE7PXF human-26 yes --by prompt-lab-reviewer +ANSWERED 01M3669253ED8QPNX7N7RE7PXF human-26 yes +Continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data --local-agent 01M3669253ED8QPNX7N7RE7PXF +exit=0 diff --git a/examples/prompt-lab/evidence/run/07-resume.txt b/examples/prompt-lab/evidence/run/07-resume.txt new file mode 100644 index 00000000..7f237607 --- /dev/null +++ b/examples/prompt-lab/evidence/run/07-resume.txt @@ -0,0 +1,59 @@ +$ npx flows resume --no-observer-link --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data --local-agent 01M3669253ED8QPNX7N7RE7PXF +○ run-1 (deterministic) 0.00s +✓ run-1 (deterministic) 0.01s completionReason: success +○ llm-2 (llm) 0.00s +✓ llm-2 (llm) 3.59s completionReason: success +○ llm-3 (llm) 0.00s +✓ llm-3 (llm) 4.20s completionReason: success +○ run-4 (deterministic) 0.00s +✓ run-4 (deterministic) 0.01s completionReason: success +○ llm-5 (llm) 0.00s +✓ llm-5 (llm) 3.79s completionReason: success +○ llm-6 (llm) 0.00s +✓ llm-6 (llm) 5.22s completionReason: success +○ run-7 (deterministic) 0.00s +✓ run-7 (deterministic) 0.01s completionReason: success +○ llm-8 (llm) 0.00s +✓ llm-8 (llm) 3.63s completionReason: success +○ run-9 (deterministic) 0.00s +✓ run-9 (deterministic) 0.01s completionReason: success +○ llm-10 (llm) 0.00s +✓ llm-10 (llm) 4.71s completionReason: success +○ llm-11 (llm) 0.00s +✓ llm-11 (llm) 5.29s completionReason: success +○ llm-12 (llm) 0.00s +✓ llm-12 (llm) 4.56s completionReason: success +○ llm-13 (llm) 0.00s +✓ llm-13 (llm) 3.49s completionReason: success +○ llm-14 (llm) 0.00s +✓ llm-14 (llm) 3.87s completionReason: success +○ llm-15 (llm) 0.00s +✓ llm-15 (llm) 3.79s completionReason: success +○ llm-16 (llm) 0.00s +✓ llm-16 (llm) 3.89s completionReason: success +○ llm-17 (llm) 0.00s +✓ llm-17 (llm) 3.67s completionReason: success +○ llm-18 (llm) 0.00s +✓ llm-18 (llm) 3.70s completionReason: success +○ run-19 (deterministic) 0.00s +✓ run-19 (deterministic) 0.01s completionReason: success +○ human-20 (deterministic) 0.00s +✓ human-20 (deterministic) 0.01s completionReason: success +○ run-21 (deterministic) 0.00s +✓ run-21 (deterministic) 0.01s completionReason: success +○ run-22 (deterministic) 0.00s +✓ run-22 (deterministic) 0.01s completionReason: success +○ llm-23 (llm) 0.00s +✓ llm-23 (llm) 3.93s completionReason: success +○ run-24 (deterministic) 0.00s +✓ run-24 (deterministic) 0.01s completionReason: success +○ run-25 (deterministic) 0.00s +✓ run-25 (deterministic) 0.01s completionReason: success +○ human-26 (deterministic) 0.00s +✓ human-26 (deterministic) 0.02s completionReason: success +○ run-27 (deterministic) 0.00s +✓ run-27 (deterministic) 0.07s completionReason: success +○ run-28 (deterministic) 0.00s +✓ run-28 (deterministic) 0.06s completionReason: success +RUN 01M3669253ED8QPNX7N7RE7PXF completed (29 steps) completionReason: success +exit=0 diff --git a/examples/prompt-lab/evidence/run/08-patient-run.txt b/examples/prompt-lab/evidence/run/08-patient-run.txt new file mode 100644 index 00000000..af19984c --- /dev/null +++ b/examples/prompt-lab/evidence/run/08-patient-run.txt @@ -0,0 +1,28 @@ +$ npx flows run --no-observer-link --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data --local-agent prompt-lab.flow.ts --input {"job":"patient","reviewer":"prompt-lab-reviewer","lab":"evidence/run/lab","briefId":"gap-ostomy-supplies"} +○ run-1 (deterministic) 0.00s +✓ run-1 (deterministic) 0.07s completionReason: success +○ llm-2 (llm) 0.00s +WAITING [worker_lease] Run "01M366JZVB6S0MAJJE29NSHHNP" step "llm-2" (llm) is running under a worker lease until 1790135891103. +↻ llm-2 (llm) 3.61s +WAITING [worker_lease] Run "01M366JZVB6S0MAJJE29NSHHNP" step "llm-2" (llm) is running under a worker lease until 1790135901107. +↻ llm-2 (llm) 13.63s +WAITING [worker_lease] Run "01M366JZVB6S0MAJJE29NSHHNP" step "llm-2" (llm) is running under a worker lease until 1790135911114. +↻ llm-2 (llm) 23.65s +WAITING [worker_lease] Run "01M366JZVB6S0MAJJE29NSHHNP" step "llm-2" (llm) is running under a worker lease until 1790135921122. +↻ llm-2 (llm) 33.65s +WAITING [worker_lease] Run "01M366JZVB6S0MAJJE29NSHHNP" step "llm-2" (llm) is running under a worker lease until 1790135931125. +↻ llm-2 (llm) 43.64s +WAITING [worker_lease] Run "01M366JZVB6S0MAJJE29NSHHNP" step "llm-2" (llm) is running under a worker lease until 1790135941128. +↻ llm-2 (llm) 53.67s +WAITING [worker_lease] Run "01M366JZVB6S0MAJJE29NSHHNP" step "llm-2" (llm) is running under a worker lease until 1790135951133. +↻ llm-2 (llm) 63.69s +✓ llm-2 (llm) 66.29s completionReason: success +○ run-3 (deterministic) 0.00s +✓ run-3 (deterministic) 0.07s completionReason: success +○ human-4 (deterministic) 0.00s +⏸ human-4 (human) 0.00s +PARKED [run_parked] Run "01M366JW806SDNME4SKS4JGWBV" is waiting for prompt-lab-reviewer to answer human-4: "Test patient creator · gap-ostomy-supplies for ostomy-supplies.\nPlan:\nINVENTED SHELF PATIENT — \"Marguerite\" (first-name label only; no surname, DOB, MRN, address, phone, or dates anywhere in the chart; all temporal references are relative, e.g. \"post-op day ~12\", \"two visits ago\", \"last week\").\n\nPURPOSE: exercise the workbench question \"Which ostomy supply does the patient need most this visit?\" across all answer paths. Ships as ONE patient with TWO chart versions (A: new/complicated stoma → specific single supply need; B: established/well-healed stoma with full cupboard → 'None'). Two versions of the same characteristics keep the shelf small and make the discriminating evidence obvious in diff.\n\nCHARACTERISTICS (the only demographics): age band 75–84, older adult; female; lives alone in a one-story apartment; adult child nearby providing intermittent help; independent with ADLs pre-op, now limited by post-op fatigue; mild vision impairment (reading glasses, difficulty with fine print on supply boxes); arthritic hands with reduced pinch strength — relevant because it plausibly explains seal-application technique problems and argues AGAINST answering with a cut-to-fit-only supply.\n\nDIAGNOSES: primary — sigmoid colon adenocarcinoma s/p open sigmoid colectomy with end colostomy. Secondary — type 2 diabetes mellitus (non-insulin, oral agent), hypertension, osteoarthritis of the hands, mild protein-calorie malnutrition post-op, and history of chronic sun-damaged/thin skin. Diabetes and thin skin are deliberate context: they make peristomal skin irritation clinically credible and raise the stakes of the answer without themselves naming a supply.\n\nLIVING SITUATION: alone, apartment with a standard bathroom, no caregiver present during pouch changes; adult child visits ~2×/week, does grocery runs but does not do ostomy care; no home aide authorized yet; transportation limited — patient does not drive, so \"just pick some up today\" is not a realistic self-remedy and the supply gap is a real gap.\n\nREFERRAL PACKET (discharge → home health) SAYS:\n- Reason for referral: post-surgical care and new ostomy self-care teaching; skilled nursing, plus dietitian and OT consults pending.\n- Ostomy description: end colostomy, left lower quadrant, matured, stoma ~28 mm round, budded ~1 cm, beefy red and viable at discharge; peristomal skin intact at discharge.\n- Output character: soft-formed to pasty brown stool, moderate volume, ~2–3 emptyings per day; no high-output/liquid pattern documented; flatus present.\n- Stoma age: new — approximately 10–14 days from creation at the time of the first home visit.\n- Supply list sent home (explicit inventory, this is the core evidence surface): 20 two-piece cut-to-fit skin barriers/wafers; 30 drainable pouches with closure clips; 1 box (10) barrier rings/seals; 1 tube ostomy paste; 1 box (50) adhesive-remover wipes; 1 box (30) skin-prep/barrier-film wipes; 1 bottle pouch deodorant; measuring guide and scissors; ostomy belt (1).\n- Teaching status at discharge: \"patient returned demonstration of pouch emptying; barrier change performed by nursing, patient observed only.\"\n- Referral explicitly states \"all supplies provided for 30 days; no anticipated supply need.\" (This line is the hook for the deliberate conflict below.)\n- Reorder pathway: DME supplier assigned, reorder by phone, 3–5 day delivery — i.e., an unmet need today is actionable but not instantly self-solvable.\n\nTODAY'S VISIT NOTES (VERSION A — the complicated chart) MUST SHOW:\n- Stoma assessment: stoma now ~24 mm (expected post-op shrinkage from the referral's 28 mm), still beefy red, viable, budded, at skin level on one edge due to a shallow crease in the left lower quadrant when the patient sits.\n- Peristomal skin: erythematous, moist, weepy denudement in a crescent along the 4-to-8 o'clock inferior aspect, ~2 cm wide, matching where output tracks under the barrier when seated; no candidal satellite lesions, no ulceration, no mucocutaneous separation. Patient reports burning/stinging under the wafer.\n- Seal failure: barrier lasting ~18–24 hours before leakage, versus expected 3–4 days; patient has changed the appliance 3 times in the last 2 days. Undermining of the wafer adhesive noted on removal, with stool tracking onto the denuded skin.\n- Supply inventory ON HAND, counted at the visit (this is what forces a SPECIFIC answer): barrier rings 8 of 10 remaining; wafers 14 remaining; pouches 22 remaining; paste ~¾ tube; adhesive-remover wipes ~40 remaining; ostomy belt unused in the drawer; SKIN-PREP / BARRIER-FILM WIPES: 0 — box empty, patient has been applying the barrier to wet, weepy skin without any protective film and did not know a refill was needed. Also absent from the home entirely: stoma powder / protective powder (never sent, not on the referral list) — so the chart supports a single, concrete, most-needed item rather than a generic \"more supplies.\"\n- Technique observation: patient cuts the wafer opening to the discharge measurement (28 mm) rather than the current 24 mm, leaving exposed skin; arthritic hands make scissor-cutting slow and imprecise. This is documented as a TEACHING need, not a supply need — it is a deliberate distractor that a weak answer will convert into \"needs pre-cut/moldable barriers.\"\n- Output character today: matches the referral (soft-formed, 2–3× daily) — deliberately NOT high-output, so \"needs high-output/drainable-with-spout pouches\" is unsupported.\n- Vitals/systemic: afebrile, no peri-stomal cellulitis, blood glucose mildly elevated; pain 3/10 burning at the skin only.\n- Patient goal stated in their own words: \"I want it to stay on so I can go to my grandchild's recital without worrying.\"\n\nTHE DEFENSIBLE ANSWER in Version A: the skin-prep / barrier-film wipes (the exhausted item), because the failing link in the chain is unprotected, weeping peristomal skin under an adhesive barrier — rings are on hand and already in use, pouches and wafers are stocked, output does not justify a different pouch system, and the wafer-sizing error is a teaching correction, not a purchase. A close-second defensible answer (stoma/protective powder, never supplied) is intentionally reachable, so graders can distinguish \"picked the empty box\" from \"reasoned about the crusting technique the weepy skin actually needs.\"\n\nVERSION B — the 'None' chart (same patient characteristics, later state; or a second shelf patient in the same age band if the shelf prefers distinct records):\n- Stoma age: established, ~14 months; stoma ~24 mm, stable size, beefy red, budded, no crease interference; patient uses a moldable/pre-sized barrier.\n- Peristomal skin: intact, no erythema, no denudement, no maceration; skin described as fully healed with no breakdown anywhere in the visit note.\n- Wear time: 4 days consistently, no leakage reported since the last two visits; patient independently empties and changes, returns demonstration flawlessly.\n- Supply inventory: full cupboard — wafers 18, pouches 40, barrier rings 9, skin-prep wipes 45, adhesive remover 50, paste ~full, powder 1 unopened bottle, belt available; 30-day reorder already placed with the DME supplier and confirmed in transit.\n- Output character: soft-formed, predictable, 1–2 emptyings daily, consistent with the referral.\n- Visit purpose: routine reassessment/recertification, no complaint.\n- The defensible answer is 'None' — no supply is deficient, no skin or seal problem creates a need. Version B also carries one benign non-supply need (a dietitian question about gas-producing foods) so 'None' has to be chosen on supply grounds, not because the chart is empty.\n\nDELIBERATE REFERRAL-vs-NOTES CONFLICTS (the part that tests the question rather than pattern-matching):\n1. STOMA SIZE DRIFT: referral says 28 mm; today's measurement is 24 mm. A model that trusts the referral packet will endorse the patient's 28 mm cut and miss that exposed skin is why the seal fails. The correct read is that current assessment overrides discharge documentation.\n2. \"NO ANTICIPATED SUPPLY NEED\": the referral asserts 30 days of supplies and no need; the visit count proves one box is at zero and one item was never supplied at all. A model that defers to the referral's blanket statement will answer 'None' on a chart that clearly supports a specific item — this is the primary trap separating Version A from Version B.\n3. PERISTOMAL SKIN STATUS: referral says \"peristomal skin intact at discharge\"; today's note documents weepy denudement. Same field, opposite value, with the visit note being the current truth.\n4. SUPPLY COUNT vs SUPPLY LIST: the referral's list includes skin-prep wipes, so a list-reading model will conclude the patient has them; only the counted inventory in today's note reveals the box is empty. The evidence needed to answer lives in the notes, not the packet.\n5. WEAR-TIME EXPECTATION: referral implies 3–4 day wear; notes show 18–24 hours. The gap is the clinical signal that something under the barrier — skin, not the pouch — is failing.\n\nGUARDRAILS FOR AUTHORING: no surnames, no dates or date-like strings (use relative intervals only), no MRN/account/encounter numbers, no phone numbers, no addresses, no facility or clinician names (use \"the discharging hospital\", \"the home health nurse\", \"the DME supplier\"), no insurance IDs. Keep the referral packet and visit note as separate documents so the conflict is only visible when both are read. Every fact that drives the answer — stoma measurement, skin condition, wear time, and the counted inventory with explicit zeros — must appear verbatim in the chart text, so a grader can point to the line that justifies each answer path.\nYou may edit \"plan\" in evidence/run/lab/work/patients/gap-ostomy-supplies/plan.json first.\nyes = kick generate (Patient QA then locks it onto the shelf; you do not approve the chart); no = leave it on the queue." +Answer with: flows answer --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data 01M366JW806SDNME4SKS4JGWBV human-4 yes|no +Then continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data --local-agent 01M366JW806SDNME4SKS4JGWBV +RUN 01M366JW806SDNME4SKS4JGWBV parked +exit=3 diff --git a/examples/prompt-lab/evidence/run/09-answer.txt b/examples/prompt-lab/evidence/run/09-answer.txt new file mode 100644 index 00000000..d2b2b071 --- /dev/null +++ b/examples/prompt-lab/evidence/run/09-answer.txt @@ -0,0 +1,4 @@ +$ npx flows answer --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data 01M366JW806SDNME4SKS4JGWBV human-4 yes --by prompt-lab-reviewer +ANSWERED 01M366JW806SDNME4SKS4JGWBV human-4 yes +Continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data --local-agent 01M366JW806SDNME4SKS4JGWBV +exit=0 diff --git a/examples/prompt-lab/evidence/run/10-resume.txt b/examples/prompt-lab/evidence/run/10-resume.txt new file mode 100644 index 00000000..cfb6db2f --- /dev/null +++ b/examples/prompt-lab/evidence/run/10-resume.txt @@ -0,0 +1,41 @@ +$ npx flows resume --no-observer-link --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data --local-agent 01M366JW806SDNME4SKS4JGWBV +○ run-1 (deterministic) 0.00s +✓ run-1 (deterministic) 0.01s completionReason: success +○ llm-2 (llm) 0.00s +✓ llm-2 (llm) 4.02s completionReason: success +○ run-3 (deterministic) 0.00s +✓ run-3 (deterministic) 0.01s completionReason: success +○ human-4 (deterministic) 0.00s +✓ human-4 (deterministic) 0.02s completionReason: success +○ run-5 (deterministic) 0.00s +✓ run-5 (deterministic) 0.06s completionReason: success +○ llm-6 (llm) 0.00s +WAITING [worker_lease] Run "01M366N6CC52X7F8GZTYGHCPYA" step "llm-6" (llm) is running under a worker lease until 1790135963330. +↻ llm-6 (llm) 4.09s +WAITING [worker_lease] Run "01M366N6CC52X7F8GZTYGHCPYA" step "llm-6" (llm) is running under a worker lease until 1790135973335. +↻ llm-6 (llm) 14.11s +WAITING [worker_lease] Run "01M366N6CC52X7F8GZTYGHCPYA" step "llm-6" (llm) is running under a worker lease until 1790135983336. +↻ llm-6 (llm) 24.10s +WAITING [worker_lease] Run "01M366N6CC52X7F8GZTYGHCPYA" step "llm-6" (llm) is running under a worker lease until 1790135993341. +↻ llm-6 (llm) 34.14s +✓ llm-6 (llm) 37.72s completionReason: success +○ llm-7 (llm) 0.00s +WAITING [worker_lease] Run "01M366PAR4A4JPY4W5NMYRTWCM" step "llm-7" (llm) is running under a worker lease until 1790136000572. +↻ llm-7 (llm) 3.61s +WAITING [worker_lease] Run "01M366PAR4A4JPY4W5NMYRTWCM" step "llm-7" (llm) is running under a worker lease until 1790136010576. +↻ llm-7 (llm) 13.62s +WAITING [worker_lease] Run "01M366PAR4A4JPY4W5NMYRTWCM" step "llm-7" (llm) is running under a worker lease until 1790136020579. +↻ llm-7 (llm) 23.65s +WAITING [worker_lease] Run "01M366PAR4A4JPY4W5NMYRTWCM" step "llm-7" (llm) is running under a worker lease until 1790136030584. +↻ llm-7 (llm) 33.66s +WAITING [worker_lease] Run "01M366PAR4A4JPY4W5NMYRTWCM" step "llm-7" (llm) is running under a worker lease until 1790136040588. +↻ llm-7 (llm) 43.67s +WAITING [worker_lease] Run "01M366PAR4A4JPY4W5NMYRTWCM" step "llm-7" (llm) is running under a worker lease until 1790136050595. +↻ llm-7 (llm) 53.68s +WAITING [worker_lease] Run "01M366PAR4A4JPY4W5NMYRTWCM" step "llm-7" (llm) is running under a worker lease until 1790136060600. +↻ llm-7 (llm) 63.64s +✓ llm-7 (llm) 67.46s completionReason: success +○ run-8 (deterministic) 0.00s +✓ run-8 (deterministic) 0.06s completionReason: success +RUN 01M366JW806SDNME4SKS4JGWBV completed (9 steps) completionReason: success +exit=0 diff --git a/examples/prompt-lab/evidence/run/11-job2-run.txt b/examples/prompt-lab/evidence/run/11-job2-run.txt new file mode 100644 index 00000000..bb3645c4 --- /dev/null +++ b/examples/prompt-lab/evidence/run/11-job2-run.txt @@ -0,0 +1,34 @@ +$ npx flows run --no-observer-link --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data --local-agent prompt-lab.flow.ts --input {"job":"fix","reviewer":"prompt-lab-reviewer","lab":"evidence/run/lab","issueId":"config-sunrise-wound-status-ce89aea7"} +○ run-1 (deterministic) 0.00s +✓ run-1 (deterministic) 0.07s completionReason: success +○ llm-2 (llm) 0.00s +WAITING [worker_lease] Run "01M366REV3VYNC5GG6AA7TX0FV" step "llm-2" (llm) is running under a worker lease until 1790136070295. +↻ llm-2 (llm) 4.84s +✓ llm-2 (llm) 8.85s completionReason: success +○ llm-3 (llm) 0.00s +WAITING [worker_lease] Run "01M366RPB0C97EW0E7MC134S9V" step "llm-3" (llm) is running under a worker lease until 1790136077971. +↻ llm-3 (llm) 3.66s +✓ llm-3 (llm) 8.60s completionReason: success +○ llm-4 (llm) 0.00s +WAITING [worker_lease] Run "01M366RYM9Z7595ZX07VTH2WAX" step "llm-4" (llm) is running under a worker lease until 1790136086459. +↻ llm-4 (llm) 3.55s +✓ llm-4 (llm) 10.93s completionReason: success +○ llm-5 (llm) 0.00s +WAITING [worker_lease] Run "01M366S9ZG10B4C2C9JE0YAK70" step "llm-5" (llm) is running under a worker lease until 1790136098083. +↻ llm-5 (llm) 4.25s +WAITING [worker_lease] Run "01M366S9ZG10B4C2C9JE0YAK70" step "llm-5" (llm) is running under a worker lease until 1790136108084. +↻ llm-5 (llm) 14.26s +✓ llm-5 (llm) 14.64s completionReason: success +○ llm-6 (llm) 0.00s +WAITING [worker_lease] Run "01M366SQKD7VZ06Z89B990JP84" step "llm-6" (llm) is running under a worker lease until 1790136112032. +↻ llm-6 (llm) 3.55s +✓ llm-6 (llm) 8.11s completionReason: success +○ run-7 (deterministic) 0.00s +✓ run-7 (deterministic) 0.06s completionReason: success +○ human-8 (deterministic) 0.00s +⏸ human-8 (human) 0.00s +PARKED [run_parked] Run "01M366RA1PSQ7295XTPYPV9PXC" is waiting for prompt-lab-reviewer to answer human-8: "Question workbench · wound-status (issue config-sunrise-wound-status-ce89aea7): current-prompt outputs on 4 shelf patient(s) — jordan: No wound/High, marguerite: Ongoing/Medium, pat: Ongoing/High, riley: No wound/High.\nSet gold in evidence/run/lab/work/q-wound-status/d61687f9/grid.json (\"target\" per row; prefilled from persisted gold or the config-level target). Your review persists as gold; later AI runs never overwrite it.\nyes = persist gold and run the iterator; no = stop." +Answer with: flows answer --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data 01M366RA1PSQ7295XTPYPV9PXC human-8 yes|no +Then continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data --local-agent 01M366RA1PSQ7295XTPYPV9PXC +RUN 01M366RA1PSQ7295XTPYPV9PXC parked +exit=3 diff --git a/examples/prompt-lab/evidence/run/12-answer.txt b/examples/prompt-lab/evidence/run/12-answer.txt new file mode 100644 index 00000000..bf3e8337 --- /dev/null +++ b/examples/prompt-lab/evidence/run/12-answer.txt @@ -0,0 +1,4 @@ +$ npx flows answer --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data 01M366RA1PSQ7295XTPYPV9PXC human-8 yes --by prompt-lab-reviewer +ANSWERED 01M366RA1PSQ7295XTPYPV9PXC human-8 yes +Continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data --local-agent 01M366RA1PSQ7295XTPYPV9PXC +exit=0 diff --git a/examples/prompt-lab/evidence/run/13-resume.txt b/examples/prompt-lab/evidence/run/13-resume.txt new file mode 100644 index 00000000..2698a843 --- /dev/null +++ b/examples/prompt-lab/evidence/run/13-resume.txt @@ -0,0 +1,80 @@ +$ npx flows resume --no-observer-link --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data --local-agent 01M366RA1PSQ7295XTPYPV9PXC +○ run-1 (deterministic) 0.00s +✓ run-1 (deterministic) 0.01s completionReason: success +○ llm-2 (llm) 0.00s +✓ llm-2 (llm) 4.15s completionReason: success +○ llm-3 (llm) 0.00s +✓ llm-3 (llm) 3.86s completionReason: success +○ llm-4 (llm) 0.00s +✓ llm-4 (llm) 3.81s completionReason: success +○ llm-5 (llm) 0.00s +✓ llm-5 (llm) 3.66s completionReason: success +○ llm-6 (llm) 0.00s +✓ llm-6 (llm) 3.89s completionReason: success +○ run-7 (deterministic) 0.00s +✓ run-7 (deterministic) 0.01s completionReason: success +○ human-8 (deterministic) 0.00s +✓ human-8 (deterministic) 0.05s completionReason: success +○ run-9 (deterministic) 0.00s +✓ run-9 (deterministic) 0.06s completionReason: success +○ run-10 (deterministic) 0.00s +✓ run-10 (deterministic) 0.08s completionReason: success +○ llm-11 (llm) 0.00s +WAITING [worker_lease] Run "01M366TMJA525EKZT8BXXV264M" step "llm-11" (llm) is running under a worker lease until 1790136141696. +↻ llm-11 (llm) 4.28s +WAITING [worker_lease] Run "01M366TMJA525EKZT8BXXV264M" step "llm-11" (llm) is running under a worker lease until 1790136151698. +↻ llm-11 (llm) 14.31s +WAITING [worker_lease] Run "01M366TMJA525EKZT8BXXV264M" step "llm-11" (llm) is running under a worker lease until 1790136161701. +↻ llm-11 (llm) 24.31s +✓ llm-11 (llm) 25.06s completionReason: success +○ llm-12 (llm) 0.00s +WAITING [worker_lease] Run "01M366VCMJ8MV2Y2S98KGKD9Z8" step "llm-12" (llm) is running under a worker lease until 1790136166341. +↻ llm-12 (llm) 3.87s +WAITING [worker_lease] Run "01M366VCMJ8MV2Y2S98KGKD9Z8" step "llm-12" (llm) is running under a worker lease until 1790136176345. +↻ llm-12 (llm) 13.88s +✓ llm-12 (llm) 22.38s completionReason: success +○ llm-13 (llm) 0.00s +WAITING [worker_lease] Run "01M366W2NG4G1HNGMZC3ZYBQNN" step "llm-13" (llm) is running under a worker lease until 1790136188900. +↻ llm-13 (llm) 4.04s +WAITING [worker_lease] Run "01M366W2NG4G1HNGMZC3ZYBQNN" step "llm-13" (llm) is running under a worker lease until 1790136198904. +↻ llm-13 (llm) 14.09s +WAITING [worker_lease] Run "01M366W2NG4G1HNGMZC3ZYBQNN" step "llm-13" (llm) is running under a worker lease until 1790136208907. +↻ llm-13 (llm) 24.07s +WAITING [worker_lease] Run "01M366W2NG4G1HNGMZC3ZYBQNN" step "llm-13" (llm) is running under a worker lease until 1790136218909. +↻ llm-13 (llm) 34.05s +✓ llm-13 (llm) 35.55s completionReason: success +○ llm-14 (llm) 0.00s +WAITING [worker_lease] Run "01M366X6H9SM8WVMQD4DJ89QMM" step "llm-14" (llm) is running under a worker lease until 1790136225632. +↻ llm-14 (llm) 5.23s +WAITING [worker_lease] Run "01M366X6H9SM8WVMQD4DJ89QMM" step "llm-14" (llm) is running under a worker lease until 1790136235648. +↻ llm-14 (llm) 15.26s +WAITING [worker_lease] Run "01M366X6H9SM8WVMQD4DJ89QMM" step "llm-14" (llm) is running under a worker lease until 1790136245651. +↻ llm-14 (llm) 25.25s +✓ llm-14 (llm) 32.03s completionReason: success +○ run-15 (deterministic) 0.00s +✓ run-15 (deterministic) 0.09s completionReason: success +○ llm-16 (llm) 0.00s +WAITING [worker_lease] Run "01M366Y64AATPSDNA71NA21N1V" step "llm-16" (llm) is running under a worker lease until 1790136257991. +↻ llm-16 (llm) 5.45s +✓ llm-16 (llm) 10.43s completionReason: success +○ llm-17 (llm) 0.00s +WAITING [worker_lease] Run "01M366YERVVD3169AH4H5C304R" step "llm-17" (llm) is running under a worker lease until 1790136266830. +↻ llm-17 (llm) 3.87s +✓ llm-17 (llm) 9.53s completionReason: success +○ llm-18 (llm) 0.00s +WAITING [worker_lease] Run "01M366YS9B14N64W0YHMB5CBQV" step "llm-18" (llm) is running under a worker lease until 1790136277601. +↻ llm-18 (llm) 5.10s +✓ llm-18 (llm) 10.16s completionReason: success +○ llm-19 (llm) 0.00s +WAITING [worker_lease] Run "01M366Z2KF5X7R6ZFNM653ETT0" step "llm-19" (llm) is running under a worker lease until 1790136287140. +↻ llm-19 (llm) 4.48s +✓ llm-19 (llm) 8.99s completionReason: success +○ run-20 (deterministic) 0.00s +✓ run-20 (deterministic) 0.06s completionReason: success +○ human-21 (deterministic) 0.00s +⏸ human-21 (human) 0.00s +PARKED [run_parked] Run "01M366RA1PSQ7295XTPYPV9PXC" is waiting for prompt-lab-reviewer to answer human-21: "Re-run and score · wound-status, new prompt p-wound-status-f41f7ca1:\n4 of 4 golded patients worked (100%)\n jordan: gold No wound · new run No wound · worked\n marguerite: gold Ongoing · new run Ongoing · worked\n pat: gold Ongoing · new run Ongoing · worked\n riley: gold No wound · new run No wound · worked\nSHARED: marking done changes this prompt for every agency that uses it: harbor, maple, sunrise.\nRead the new prompt in evidence/run/lab/bank.json and the outputs in evidence/run/lab/work/q-wound-status/d61687f9/score.json.\nyes = mark done (live in Apricot now, no promote step); no = keep it as a draft and iterate again in a new run." +Answer with: flows answer --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data 01M366RA1PSQ7295XTPYPV9PXC human-21 yes|no +Then continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data --local-agent 01M366RA1PSQ7295XTPYPV9PXC +RUN 01M366RA1PSQ7295XTPYPV9PXC parked +exit=3 diff --git a/examples/prompt-lab/evidence/run/14-answer.txt b/examples/prompt-lab/evidence/run/14-answer.txt new file mode 100644 index 00000000..070992b6 --- /dev/null +++ b/examples/prompt-lab/evidence/run/14-answer.txt @@ -0,0 +1,4 @@ +$ npx flows answer --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data 01M366RA1PSQ7295XTPYPV9PXC human-21 yes --by prompt-lab-reviewer +ANSWERED 01M366RA1PSQ7295XTPYPV9PXC human-21 yes +Continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data --local-agent 01M366RA1PSQ7295XTPYPV9PXC +exit=0 diff --git a/examples/prompt-lab/evidence/run/15-resume.txt b/examples/prompt-lab/evidence/run/15-resume.txt new file mode 100644 index 00000000..e03ce8f1 --- /dev/null +++ b/examples/prompt-lab/evidence/run/15-resume.txt @@ -0,0 +1,49 @@ +$ npx flows resume --no-observer-link --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data --local-agent 01M366RA1PSQ7295XTPYPV9PXC +○ run-1 (deterministic) 0.00s +✓ run-1 (deterministic) 0.01s completionReason: success +○ llm-2 (llm) 0.00s +✓ llm-2 (llm) 5.35s completionReason: success +○ llm-3 (llm) 0.00s +✓ llm-3 (llm) 4.10s completionReason: success +○ llm-4 (llm) 0.00s +✓ llm-4 (llm) 3.83s completionReason: success +○ llm-5 (llm) 0.00s +✓ llm-5 (llm) 3.60s completionReason: success +○ llm-6 (llm) 0.00s +✓ llm-6 (llm) 4.94s completionReason: success +○ run-7 (deterministic) 0.00s +✓ run-7 (deterministic) 0.01s completionReason: success +○ human-8 (deterministic) 0.00s +✓ human-8 (deterministic) 0.01s completionReason: success +○ run-9 (deterministic) 0.00s +✓ run-9 (deterministic) 0.01s completionReason: success +○ run-10 (deterministic) 0.00s +✓ run-10 (deterministic) 0.01s completionReason: success +○ llm-11 (llm) 0.00s +✓ llm-11 (llm) 3.53s completionReason: success +○ llm-12 (llm) 0.00s +✓ llm-12 (llm) 3.97s completionReason: success +○ llm-13 (llm) 0.00s +✓ llm-13 (llm) 4.02s completionReason: success +○ llm-14 (llm) 0.00s +✓ llm-14 (llm) 3.96s completionReason: success +○ run-15 (deterministic) 0.00s +✓ run-15 (deterministic) 0.01s completionReason: success +○ llm-16 (llm) 0.00s +✓ llm-16 (llm) 3.54s completionReason: success +○ llm-17 (llm) 0.00s +✓ llm-17 (llm) 3.47s completionReason: success +○ llm-18 (llm) 0.00s +✓ llm-18 (llm) 6.68s completionReason: success +○ llm-19 (llm) 0.00s +✓ llm-19 (llm) 4.34s completionReason: success +○ run-20 (deterministic) 0.00s +✓ run-20 (deterministic) 0.01s completionReason: success +○ human-21 (deterministic) 0.00s +✓ human-21 (deterministic) 0.02s completionReason: success +○ run-22 (deterministic) 0.00s +✓ run-22 (deterministic) 0.07s completionReason: success +○ run-23 (deterministic) 0.00s +✓ run-23 (deterministic) 0.07s completionReason: success +RUN 01M366RA1PSQ7295XTPYPV9PXC completed (24 steps) completionReason: success +exit=0 diff --git a/examples/prompt-lab/evidence/run/16-lab-state.txt b/examples/prompt-lab/evidence/run/16-lab-state.txt new file mode 100644 index 00000000..9d6a1a3e --- /dev/null +++ b/examples/prompt-lab/evidence/run/16-lab-state.txt @@ -0,0 +1,16 @@ +$ node -e +const b=require('./evidence/run/lab/bank.json'); +for (const [q,v] of Object.entries(b.questions)) console.log(q.padEnd(17),'live',String(v.livePromptId).padEnd(30),'draft',v.draftPromptId??'-'); +console.log('shelf', require('fs').readdirSync('evidence/run/lab/shelf').join(' ')); +console.log('issues', JSON.stringify(require('./evidence/run/lab/queue/issues.json').map(i=>[i.id,i.status]))); +console.log('patient briefs', JSON.stringify(require('./evidence/run/lab/queue/patient-briefs.json').map(i=>[i.id,i.status]))); +console.log('gold', Object.keys(require('./evidence/run/lab/gold.json')).join(' ')); +wound-status live p-wound-status-f41f7ca1 draft - +mood live p-mood-v1 draft - +living-situation live p-living-situation-50cd9fc4 draft - +ostomy-supplies live null draft p-ostomy-supplies-af0f46ca +shelf jordan.json marguerite.json pat.json riley.json +issues [["config-sunrise-wound-status-ce89aea7","done"]] +patient briefs [["gap-ostomy-supplies","locked"]] +gold wound-status|jordan wound-status|marguerite wound-status|pat wound-status|riley +exit=0 diff --git a/examples/prompt-lab/evidence/run/lab/agencies.json b/examples/prompt-lab/evidence/run/lab/agencies.json new file mode 100644 index 00000000..b3306493 --- /dev/null +++ b/examples/prompt-lab/evidence/run/lab/agencies.json @@ -0,0 +1,30 @@ +{ + "harbor": { + "name": "Harbor Home Health", + "visitTypes": { + "soc": [ + { "questionId": "wound-status", "options": ["Healed", "Ongoing", "No wound"] }, + { "questionId": "mood", "options": ["Calm", "Anxious", "Low"] } + ] + } + }, + "maple": { + "name": "Maple Visiting Nurses", + "visitTypes": { + "roc": [ + { "questionId": "wound-status", "options": ["Healed", "Ongoing", "No wound"] } + ] + } + }, + "sunrise": { + "name": "Sunrise Home Care", + "visitTypes": { + "soc": [ + { "questionId": "wound-status", "options": ["Healed", "Ongoing", "No wound"] }, + { "questionId": "mood", "options": ["Calm", "Anxious", "Low", "Agitated"] }, + { "questionId": "living-situation", "options": ["Lives alone", "Lives with spouse", "Lives with family", "Facility"] }, + { "questionId": "ostomy-supplies", "options": ["Pouches", "Barrier rings", "Skin prep wipes", "None"] } + ] + } + } +} diff --git a/examples/prompt-lab/evidence/run/lab/bank.json b/examples/prompt-lab/evidence/run/lab/bank.json new file mode 100644 index 00000000..803436d8 --- /dev/null +++ b/examples/prompt-lab/evidence/run/lab/bank.json @@ -0,0 +1,34 @@ +{ + "prompts": { + "p-wound-status-v1": "Determine the status of the patient's primary wound. Read the referral packet first; if the discharge summary states the wound is closed or healed, answer Healed. Otherwise use the visit notes. Use High confidence when the referral states the status.", + "p-mood-v1": "Describe the patient's mood today from the visit notes. Choose the single option that best matches what the patient says and how the clinician describes them. Use High confidence when the patient states their feelings in their own words, Medium when inferred from behavior, Low when the notes are silent.", + "p-living-situation-50cd9fc4": "You are filling one field of a home-health visit chart: the patient's living situation. Choose exactly one of the options provided to you.\n\nSOURCE OF TRUTH\n- Today's visit notes — what the clinician observed and what the patient or caregiver said during this visit — are the source of truth. Prefer them over the referral packet, intake form, prior visit notes, or demographic fields.\n- When the visit notes conflict with any other document, follow the visit notes and say so in your explanation.\n- Use other documents only when today's notes are silent on where and with whom the patient lives.\n\nHOW TO CHOOSE\n- Pick one option from the list you are given. Never invent an option, never combine two, never return free text. If none fits cleanly, pick the closest one and drop your confidence to Low or Medium, explaining the mismatch.\n- Read the options as a residence question: who is in the home overnight, not who visits or provides care.\n- Guidance for common wording in charts:\n - Lives alone: the patient is the only person residing in the home. Daytime caregivers, aides, visiting relatives, or neighbors who check in do not change this — the patient still lives alone.\n - Lives with spouse: a spouse or long-term partner resides in the home. If a spouse plus other relatives reside there, prefer the spouse option when the chart frames the spouse as the co-resident and primary support; prefer the family option when the notes describe a multi-person household without singling out the spouse.\n - Lives with family: one or more relatives or household members other than a spouse reside with the patient (adult child, sibling, parent, grandchild, or a described household).\n - Facility: the patient resides in a congregate or institutional setting — assisted living, memory care, group home, board and care, skilled nursing, or similar. A private residence in a senior apartment or independent-living community is not a facility by itself; classify by who resides with the patient unless the chart describes on-site staffing or facility-provided care.\n- If the notes describe a recent or planned move, answer for where the patient lives as of this visit, not where they are expected to live later.\n\nCONFIDENCE\n- High: today's visit notes state the living situation directly (an explicit statement of who lives in the home or that the patient resides in a named type of facility).\n- Medium: the answer is inferred — drawn from indirect cues in today's notes (who was present and described as a household member, references to \"her husband in the next room\", facility staff charting the visit), or taken from the referral packet or intake form because today's notes do not address it.\n- Low: the chart is silent on living situation, or sources contradict each other and today's notes do not resolve the conflict.\n\nEXPLANATION\n- Return a one- or two-sentence explanation that cites where in the chart the answer came from — name the document or section (e.g. today's visit note narrative, the social/home environment section, the referral packet) and paraphrase the supporting language. Do not quote or reproduce patient identifiers of any kind. If you overrode another document, state which one and why.", + "p-ostomy-supplies-af0f46ca": "You are filling one field of a home-health visit assessment: which ostomy supply the patient needs most at this visit. You will be given the question, the exact list of answer options, and the patient's chart.\n\nSOURCE OF TRUTH\n- Today's visit notes — what the clinician observed, measured, and was told during this visit — are authoritative. Where they conflict with the referral packet, intake forms, prior visit notes, the medication list, or any standing order, today's visit notes win and you say so in the explanation.\n- Older chart material is usable only as background: it can establish that the patient has an ostomy at all, or what supplies were being used, but it cannot override what today's clinician recorded.\n- Read the whole of today's note, not just an ostomy-specific heading. Supply need often appears in the skin assessment, the wound or peristomal description, the supplies-on-hand or inventory section, the patient/caregiver report, the teaching section, or the plan for next visit.\n\nHOW TO CHOOSE\n- Choose exactly one option, and only from the option list you are given. Do not invent, merge, rename, or split options. If the need you see in the chart has no matching option, pick the option that is the closest true fit and explain the gap in your explanation rather than writing a new option.\n- \"Most needed\" means the single supply whose absence is the most immediate problem for this patient right now. Weigh, in this order: (1) the supply is out, nearly out, or unavailable today; (2) the supply is the one that addresses an active problem documented today, such as leakage, an unsealed or uneven seal, an appliance that will not adhere, or peristomal skin breakdown, irritation, moisture, or denudement; (3) the supply the clinician explicitly asks to be ordered, delivered, refilled, or taught about today.\n- If today's notes document more than one gap, choose the one the clinician treats as the priority — the one tied to a change in the plan of care, an order placed, a teaching intervention, or the stated reason the appliance is failing. If two appear genuinely equal in urgency, choose the one that keeps the appliance attached and contained (containment before skin protection), and name the second one in your explanation.\n- Select the \"none needed\" style option only when today's notes affirmatively support it: the patient has adequate supplies on hand, the appliance is intact and sealed, the peristomal skin is described as intact, or the patient has no ostomy. Do not choose it merely because the notes are quiet about supplies.\n\nCONFIDENCE\n- High: today's visit notes state the answer directly — they name the supply as out, needed, ordered, or requested, or they name it as the fix for a problem documented today.\n- Medium: you inferred the answer from today's notes rather than reading it off them — for example, the note describes the clinical problem and the matching supply follows from it, or a supply count implies the shortfall, but no sentence names the needed supply outright.\n- Low: the chart is silent on ostomy supplies and on peristomal condition, the only relevant information is from the referral packet or a prior visit rather than today, or today's notes contradict themselves and nothing resolves which reading is current. Still return your best single option at Low confidence; do not leave the field unanswered.\n\nEXPLANATION\n- Write one or two sentences saying where in the chart the answer came from — name the section or the kind of entry (for example, today's peristomal skin assessment, today's supply inventory, today's plan of care) and what it said in your own words. If you overrode an older source, or if a close second option exists, say so here.\n\nPRIVACY\n- Never include a patient name, initials, date of birth, medical record or account number, address, phone number, or any other identifier in your explanation. Refer to \"the patient\" and cite chart locations by section, not by quoting identifying text. Quote clinical wording only as briefly as needed to support the answer.", + "p-wound-status-f41f7ca1": "Determine the current status of the patient's primary wound. Answer Healed, Ongoing, or No wound.\n\nRead the whole chart before answering. Rank sources by how directly and how recently they observed the wound site:\n\n1. A direct assessment — a clinician's own observation of the site at a visit, with findings such as measurements, wound bed description, drainage, odor, dressing applied, or an explicit note that the site is intact — is the best evidence of current status. The most recent direct assessment governs.\n2. A referral packet, hospital discharge summary, intake form, or other history describes the wound as of an earlier date. It establishes that a wound exists and what it was, but it never overrides a later direct assessment. Read \"closed\", \"healed\", or \"resolved\" in such a document as the status on that document's date, not as today's status; if a later assessment finds the site open, the document is simply outdated.\n\nChoose the answer:\n- Ongoing — the most recent direct assessment finds an open or unhealed wound (open area, drainage, slough, eschar, undermining, or active wound care), or the chart documents a wound with nothing later indicating it closed.\n- Healed — the most recent direct assessment finds the site closed, intact, or re-epithelialized, or the chart documents closure and no later source reports an open wound.\n- No wound — nothing in the chart documents a wound: no wound history and no wound findings on assessment.\n\nAssign confidence. Exactly one of these applies to every chart; work down the list and take the first that fits:\n- High — the answer rests on a direct assessment of the wound site at a recent visit. This holds even when an earlier document says the opposite: a direct observation of the site settles the question, and being contradicted by stale history does not lower confidence.\n- Medium — no direct assessment of the site is available, so the answer rests on referral or history documents alone; or the only direct assessment is old enough that the status could plausibly have changed since.\n- Low — the chart does not address the wound at all (a silent chart, which yields No wound), or its wound documentation is so sparse, ambiguous, or internally inconsistent that you cannot resolve it.\n\nIn the rationale, cite the deciding observation with its concrete findings and, when an earlier document said something different, name that document in a clause and say its status is outdated. Two sentences at most. Do not speculate beyond what the chart records, and do not let living situation, mood, or unrelated diagnoses influence the wound status." + }, + "questions": { + "wound-status": { + "text": "What is the current status of the patient's primary wound?", + "type": "single", + "livePromptId": "p-wound-status-f41f7ca1", + "draftPromptId": null + }, + "mood": { + "text": "How would you describe the patient's mood today?", + "type": "single", + "livePromptId": "p-mood-v1" + }, + "living-situation": { + "text": "What is the patient's living situation?", + "type": "single", + "livePromptId": "p-living-situation-50cd9fc4", + "draftPromptId": null + }, + "ostomy-supplies": { + "text": "Which ostomy supply does the patient need most this visit?", + "type": "single", + "livePromptId": null, + "draftPromptId": "p-ostomy-supplies-af0f46ca" + } + } +} diff --git a/examples/prompt-lab/evidence/run/lab/gold.json b/examples/prompt-lab/evidence/run/lab/gold.json new file mode 100644 index 00000000..198a0cf6 --- /dev/null +++ b/examples/prompt-lab/evidence/run/lab/gold.json @@ -0,0 +1,22 @@ +{ + "wound-status|jordan": { + "answer": "No wound", + "confidence": "High", + "explanation": "The referral packet's intake form documents \"No wounds documented. Skin intact,\" and today's SOC visit note confirms \"Skin intact, no wounds\" — there is no primary wound to stage, and no discharge summary states a wound was closed or healed." + }, + "wound-status|marguerite": { + "answer": "Ongoing", + "confidence": "Medium", + "explanation": "The referral packet never states the wound is closed or healed — it only records peristomal skin intact at discharge — so the visit notes govern, and today's assessment documents erythematous, moist, weepy denudement about 2 cm wide along the 4-to-8 o'clock peristomal aspect with patient-reported burning and stinging. Confidence is Medium rather than High because the status comes from the visit note, not a referral statement." + }, + "wound-status|pat": { + "answer": "Ongoing", + "confidence": "High", + "explanation": "Today's visit notes describe an open 2.0 x 1.5 cm left heel wound with serous drainage and slough; the discharge summary's closed status is outdated." + }, + "wound-status|riley": { + "answer": "No wound", + "confidence": "High", + "explanation": "The referral packet states 'No wounds' and today's SOC visit note confirms 'No wounds; skin intact around cast,' so there is no primary wound to track." + } +} diff --git a/examples/prompt-lab/evidence/run/lab/guidelines.md b/examples/prompt-lab/evidence/run/lab/guidelines.md new file mode 100644 index 00000000..c383992a --- /dev/null +++ b/examples/prompt-lab/evidence/run/lab/guidelines.md @@ -0,0 +1,9 @@ +# Shared prompt guidelines + +Every question prompt, whoever wrote it, must: + +1. Name the source of truth. When the visit notes (what the clinician saw and heard today) disagree with the referral packet or an intake form, today's visit notes win. +2. Tell the engine how to pick among the options it will be given; never invent an option. +3. Define the confidence levels: High only when today's notes state the answer directly; Medium when it is inferred; Low when the chart is silent or contradictory. +4. Ask for a one- or two-sentence explanation that cites where in the chart the answer came from. +5. Contain no patient identifiers and no examples copied from real charts. diff --git a/examples/prompt-lab/evidence/run/lab/queue/issues.json b/examples/prompt-lab/evidence/run/lab/queue/issues.json new file mode 100644 index 00000000..00859cdb --- /dev/null +++ b/examples/prompt-lab/evidence/run/lab/queue/issues.json @@ -0,0 +1,20 @@ +[ + { + "id": "config-sunrise-wound-status-ce89aea7", + "kind": "config-send", + "questionIds": [ + "wound-status" + ], + "status": "done", + "agency": "sunrise", + "text": "sunrise soc first pass changed 1 row(s) on a shared question (shared with harbor, maple). The prompt makes the referral win over what the nurse saw today. Today's notes must win (shared guideline 1).", + "targets": { + "pat": { + "answer": "Ongoing", + "confidence": "High", + "explanation": "Today's visit notes describe an open 2.0 x 1.5 cm left heel wound with serous drainage and slough; the discharge summary's closed status is outdated." + } + }, + "proposedPrompt": "Determine the status of the patient's primary wound.\n\nToday's visit notes govern. When the visit notes describe the wound, base the answer on what the clinician documented at today's visit, even if the referral packet or discharge summary says something different. A referral or discharge summary reflects the patient's condition at an earlier point in time; it can be outdated by the time of the visit, and a documented open wound today overrides an earlier \"closed\" or \"healed\" note.\n\n- Answer Ongoing when today's visit notes document an open or unhealed wound — for example an open area with measurements, drainage, slough, or an active dressing regimen.\n- Answer Healed when today's visit notes document the wound as closed, resurfaced, or fully healed. If the visit notes do not assess the wound at all, fall back to the referral packet: if the discharge summary states the wound is closed or healed, answer Healed.\n- Answer No wound only when neither the visit notes nor the referral packet document any wound, present or recently resolved.\n\nUse High confidence when today's visit notes directly document the wound's current state, including when they contradict the referral. Use High confidence when the referral states the status and the visit notes do not contradict it. Lower the confidence when the only source is an older referral and the visit notes are silent, or when the documentation within a single source conflicts with itself.\n\nIn the rationale, cite the specific findings from today's visit notes that drive the answer, and when the referral disagrees, say plainly that the earlier status is outdated. Keep the rationale to one or two sentences grounded in the documentation." + } +] diff --git a/examples/prompt-lab/evidence/run/lab/queue/patient-briefs.json b/examples/prompt-lab/evidence/run/lab/queue/patient-briefs.json new file mode 100644 index 00000000..e06f18f5 --- /dev/null +++ b/examples/prompt-lab/evidence/run/lab/queue/patient-briefs.json @@ -0,0 +1,9 @@ +[ + { + "id": "gap-ostomy-supplies", + "questionId": "ostomy-supplies", + "brief": "No shelf patient has any ostomy documented, so every answer path for this question is unexercised. Add an older adult recently discharged after colorectal surgery with a new colostomy: referral packet lists the stoma, output character, and the supply list sent home; today's visit notes show peristomal skin irritation and a pouch seal failing early, with barrier rings on hand but skin prep wipes exhausted — so the chart supports a specific single supply need rather than a generic one. To exercise the remaining options, a second version of the chart (or a second such patient) should show an established, well-healed stoma with a full supply cupboard and no skin breakdown, supporting 'None'. Characteristics only: age band, ostomy type, stoma age, skin condition, current supply inventory — no names, dates, addresses, or record numbers.", + "from": "planner", + "status": "locked" + } +] diff --git a/examples/prompt-lab/evidence/run/lab/shelf/jordan.json b/examples/prompt-lab/evidence/run/lab/shelf/jordan.json new file mode 100644 index 00000000..072aca46 --- /dev/null +++ b/examples/prompt-lab/evidence/run/lab/shelf/jordan.json @@ -0,0 +1,6 @@ +{ + "id": "jordan", "label": "Jordan", "locked": true, "source": "invented", + "ageBand": "65-74", "visitType": "soc", + "referral": "Referral: CHF exacerbation, new diuretic regimen, daily weights. Intake form (completed by hospital case manager): lives alone. No wounds documented. Skin intact.", + "notes": "Skilled nursing SOC visit. Spouse present for the whole visit; patient and spouse share the home and spouse sets up the pill organizer each morning. Skin intact, no wounds. Weight up 1 lb from discharge. Patient relaxed and engaged, asks good questions about the diuretic." +} diff --git a/examples/prompt-lab/evidence/run/lab/shelf/marguerite.json b/examples/prompt-lab/evidence/run/lab/shelf/marguerite.json new file mode 100644 index 00000000..8f2235ee --- /dev/null +++ b/examples/prompt-lab/evidence/run/lab/shelf/marguerite.json @@ -0,0 +1,10 @@ +{ + "id": "marguerite", + "label": "Marguerite", + "ageBand": "75–84", + "visitType": "soc", + "referral": "REFERRAL PACKET — discharging hospital to home health (start of care)\n\nReason for referral: post-surgical care and new ostomy self-care teaching. Skilled nursing ordered; dietitian and occupational therapy consults pending.\n\nCharacteristics: older adult, age band 75–84, female. Lives alone in a one-story apartment with a standard bathroom; no caregiver present during pouch changes. Adult child lives nearby and visits about twice a week for grocery runs but does not perform ostomy care. No home aide authorized. Patient does not drive and has limited transportation. Independent with ADLs before surgery, now limited by post-op fatigue. Mild vision impairment — uses reading glasses, reports difficulty with fine print on supply boxes. Osteoarthritis of the hands with reduced pinch strength.\n\nDiagnoses: primary — sigmoid colon adenocarcinoma, status post open sigmoid colectomy with end colostomy. Secondary — type 2 diabetes mellitus (non-insulin, oral agent), hypertension, osteoarthritis of the hands, mild protein-calorie malnutrition post-op, and history of chronic sun-damaged, thin skin.\n\nOstomy description: end colostomy, left lower quadrant, matured. Stoma measured 28 mm round at discharge, budded approximately 1 cm, beefy red and viable. Peristomal skin intact at discharge.\n\nOutput character: soft-formed to pasty brown stool, moderate volume, approximately 2–3 emptyings per day. No high-output or liquid pattern documented. Flatus present.\n\nStoma age: new — approximately 10 to 14 days from creation at the time of the first home visit.\n\nSUPPLY LIST SENT HOME: 20 two-piece cut-to-fit skin barriers/wafers; 30 drainable pouches with closure clips; 1 box (10) barrier rings/seals; 1 tube ostomy paste; 1 box (50) adhesive-remover wipes; 1 box (30) skin-prep / barrier-film wipes; 1 bottle pouch deodorant; measuring guide and scissors; 1 ostomy belt.\n\nTeaching status at discharge: patient returned demonstration of pouch emptying; barrier change performed by nursing, patient observed only.\n\nSupply status per discharging hospital: \"All supplies provided for 30 days; no anticipated supply need.\"\n\nExpected wear time per discharge instruction: barrier change every 3 to 4 days.\n\nReorder pathway: DME supplier assigned; reorder by phone; 3–5 day delivery.", + "notes": "HOME HEALTH VISIT NOTE — start of care, post-op day ~12\n\nVisit purpose: initial assessment, ostomy assessment, and self-care teaching. Patient alone in the apartment at the time of the visit; adult child was here two days ago for groceries.\n\nStoma assessment: stoma now measures 24 mm round — smaller than the 28 mm recorded in the referral packet, consistent with expected post-op shrinkage. Stoma remains beefy red, viable, and budded. Left lower quadrant placement sits at skin level along one edge because of a shallow abdominal crease that appears when the patient is seated.\n\nPeristomal skin: erythematous with moist, weepy denudement in a crescent along the 4-to-8 o'clock inferior aspect, approximately 2 cm wide, matching the track where output runs under the barrier when the patient sits. No candidal satellite lesions, no ulceration, no mucocutaneous separation. Patient reports burning and stinging under the wafer.\n\nSeal failure: barrier is lasting approximately 18 to 24 hours before leakage, against the referral's expected 3 to 4 days. Patient has changed the appliance 3 times in the last 2 days. Undermining of the wafer adhesive noted on removal, with stool tracking directly onto the denuded skin.\n\nSUPPLY INVENTORY COUNTED IN THE HOME THIS VISIT: barrier rings/seals — 8 of 10 remaining. Cut-to-fit wafers — 14 remaining. Drainable pouches — 22 remaining. Ostomy paste — approximately three-quarters of the tube remaining. Adhesive-remover wipes — approximately 40 remaining. Ostomy belt — 1, unused, still in the drawer. SKIN-PREP / BARRIER-FILM WIPES — 0 remaining; the box is empty. Patient has been applying the barrier directly to wet, weepy skin with no protective film and did not know a refill was needed. Stoma powder / protective powder — none in the home; it was never sent and does not appear on the referral supply list.\n\nTechnique observation: patient cuts the wafer opening to the discharge measurement of 28 mm rather than the current 24 mm, leaving a ring of exposed skin. Arthritic hands make scissor-cutting slow and imprecise, and fine print on the supply boxes is hard to read. Documented as a TEACHING need — correct measuring and cutting to current stoma size — not as a supply need.\n\nOutput character today: soft-formed brown stool, 2–3 emptyings daily, unchanged from the referral. No high-output or liquid pattern.\n\nVitals and systemic: afebrile. No peristomal cellulitis, no surrounding induration. Blood glucose mildly elevated on the patient's own meter. Pain 3/10, burning, localized to the skin only.\n\nPatient goal, in their own words: \"I want it to stay on so I can go to my grandchild's recital without worrying.\"\n\nAccess note: patient does not drive and cannot pick up supplies today. The DME supplier reorder is by phone with 3–5 day delivery, so any gap counted above stands until that order arrives.", + "locked": true, + "source": "invented" +} diff --git a/examples/prompt-lab/evidence/run/lab/shelf/pat.json b/examples/prompt-lab/evidence/run/lab/shelf/pat.json new file mode 100644 index 00000000..9a7b34c2 --- /dev/null +++ b/examples/prompt-lab/evidence/run/lab/shelf/pat.json @@ -0,0 +1,6 @@ +{ + "id": "pat", "label": "Pat", "locked": true, "source": "invented", + "ageBand": "75-84", "visitType": "soc", + "referral": "Hospital discharge summary: admitted for cellulitis of the left lower leg. Left heel pressure injury noted on admission; wound closed per discharge summary. Discharged home with home health for strengthening and medication teaching. Social: lives alone per intake form.", + "notes": "Skilled nursing SOC visit. Left heel: open area 2.0 x 1.5 cm, moderate serous drainage, wound bed pink with yellow slough at edges. Dressing changed with foam per protocol. Patient lives alone in a one-story home; neighbor checks in daily. Patient calm and pleasant, joking about hospital food, says she is glad to be home." +} diff --git a/examples/prompt-lab/evidence/run/lab/shelf/riley.json b/examples/prompt-lab/evidence/run/lab/shelf/riley.json new file mode 100644 index 00000000..dc8555b7 --- /dev/null +++ b/examples/prompt-lab/evidence/run/lab/shelf/riley.json @@ -0,0 +1,6 @@ +{ + "id": "riley", "label": "Riley", "locked": true, "source": "invented", + "ageBand": "85+", "visitType": "soc", + "referral": "Referral: fall at home with right wrist fracture, cast in place. History of hypertension. No wounds. Lives with daughter's family.", + "notes": "Skilled nursing SOC visit. Patient lives with her daughter, son-in-law and two grandchildren. No wounds; skin intact around cast. Patient says 'I can't stop thinking about falling again, I lie awake worrying about it', wrings hands during the medication review." +} diff --git a/examples/prompt-lab/evidence/run/lab/targets.json b/examples/prompt-lab/evidence/run/lab/targets.json new file mode 100644 index 00000000..55bacef4 --- /dev/null +++ b/examples/prompt-lab/evidence/run/lab/targets.json @@ -0,0 +1,47 @@ +{ + "sunrise|living-situation|jordan": { + "answer": "Lives with spouse", + "confidence": "High", + "explanation": "Today's visit note narrative states the spouse was present throughout the SOC visit, that patient and spouse share the home, and that the spouse sets up the pill organizer each morning. This overrides the intake form completed by the hospital case manager, which recorded living alone." + }, + "sunrise|living-situation|pat": { + "answer": "Lives alone", + "confidence": "High", + "explanation": "Today's SOC visit note narrative states directly that the patient lives alone in a one-story home, with a neighbor who checks in daily; a daily check-in visitor does not change residence status. This agrees with the referral packet's intake-form social history, so no override was needed." + }, + "sunrise|living-situation|riley": { + "answer": "Lives with family", + "confidence": "High", + "explanation": "Today's skilled nursing SOC visit note states directly that the patient resides with her daughter, son-in-law and two grandchildren. This agrees with the referral packet, which also describes her living with her daughter's family." + }, + "sunrise|mood|jordan": { + "answer": "Calm", + "confidence": "Medium", + "explanation": "Today's visit notes describe the patient as \"relaxed and engaged\" and asking good questions about the new diuretic, but the mood is inferred from the clinician's behavioral description rather than stated by the patient in their own words." + }, + "sunrise|mood|pat": { + "answer": "Calm", + "confidence": "High", + "explanation": "Today's visit notes describe the patient as \"calm and pleasant, joking about hospital food,\" and she states in her own words that she is \"glad to be home.\"" + }, + "sunrise|mood|riley": { + "answer": "Anxious", + "confidence": "High", + "explanation": "The patient states in her own words, 'I can't stop thinking about falling again, I lie awake worrying about it,' and the clinician notes she wrings her hands during the medication review." + }, + "sunrise|wound-status|jordan": { + "answer": "No wound", + "confidence": "High", + "explanation": "The referral packet's intake form documents 'No wounds documented. Skin intact,' and today's SOC visit note confirms 'Skin intact, no wounds' — no wound was ever present, so there is nothing to have healed." + }, + "sunrise|wound-status|pat": { + "answer": "Ongoing", + "confidence": "High", + "explanation": "Today's visit notes describe an open 2.0 x 1.5 cm left heel wound with serous drainage and slough; the discharge summary's closed status is outdated." + }, + "sunrise|wound-status|riley": { + "answer": "No wound", + "confidence": "High", + "explanation": "The referral packet states \"No wounds\" and today's SOC visit note confirms \"No wounds; skin intact around cast,\" so there is no primary wound to stage or track." + } +} diff --git a/examples/prompt-lab/evidence/run/lab/work/patients/gap-ostomy-supplies/plan.json b/examples/prompt-lab/evidence/run/lab/work/patients/gap-ostomy-supplies/plan.json new file mode 100644 index 00000000..1a00c5b6 --- /dev/null +++ b/examples/prompt-lab/evidence/run/lab/work/patients/gap-ostomy-supplies/plan.json @@ -0,0 +1,4 @@ +{ + "brief": "No shelf patient has any ostomy documented, so every answer path for this question is unexercised. Add an older adult recently discharged after colorectal surgery with a new colostomy: referral packet lists the stoma, output character, and the supply list sent home; today's visit notes show peristomal skin irritation and a pouch seal failing early, with barrier rings on hand but skin prep wipes exhausted — so the chart supports a specific single supply need rather than a generic one. To exercise the remaining options, a second version of the chart (or a second such patient) should show an established, well-healed stoma with a full supply cupboard and no skin breakdown, supporting 'None'. Characteristics only: age band, ostomy type, stoma age, skin condition, current supply inventory — no names, dates, addresses, or record numbers.", + "plan": "INVENTED SHELF PATIENT — \"Marguerite\" (first-name label only; no surname, DOB, MRN, address, phone, or dates anywhere in the chart; all temporal references are relative, e.g. \"post-op day ~12\", \"two visits ago\", \"last week\").\n\nPURPOSE: exercise the workbench question \"Which ostomy supply does the patient need most this visit?\" across all answer paths. Ships as ONE patient with TWO chart versions (A: new/complicated stoma → specific single supply need; B: established/well-healed stoma with full cupboard → 'None'). Two versions of the same characteristics keep the shelf small and make the discriminating evidence obvious in diff.\n\nCHARACTERISTICS (the only demographics): age band 75–84, older adult; female; lives alone in a one-story apartment; adult child nearby providing intermittent help; independent with ADLs pre-op, now limited by post-op fatigue; mild vision impairment (reading glasses, difficulty with fine print on supply boxes); arthritic hands with reduced pinch strength — relevant because it plausibly explains seal-application technique problems and argues AGAINST answering with a cut-to-fit-only supply.\n\nDIAGNOSES: primary — sigmoid colon adenocarcinoma s/p open sigmoid colectomy with end colostomy. Secondary — type 2 diabetes mellitus (non-insulin, oral agent), hypertension, osteoarthritis of the hands, mild protein-calorie malnutrition post-op, and history of chronic sun-damaged/thin skin. Diabetes and thin skin are deliberate context: they make peristomal skin irritation clinically credible and raise the stakes of the answer without themselves naming a supply.\n\nLIVING SITUATION: alone, apartment with a standard bathroom, no caregiver present during pouch changes; adult child visits ~2×/week, does grocery runs but does not do ostomy care; no home aide authorized yet; transportation limited — patient does not drive, so \"just pick some up today\" is not a realistic self-remedy and the supply gap is a real gap.\n\nREFERRAL PACKET (discharge → home health) SAYS:\n- Reason for referral: post-surgical care and new ostomy self-care teaching; skilled nursing, plus dietitian and OT consults pending.\n- Ostomy description: end colostomy, left lower quadrant, matured, stoma ~28 mm round, budded ~1 cm, beefy red and viable at discharge; peristomal skin intact at discharge.\n- Output character: soft-formed to pasty brown stool, moderate volume, ~2–3 emptyings per day; no high-output/liquid pattern documented; flatus present.\n- Stoma age: new — approximately 10–14 days from creation at the time of the first home visit.\n- Supply list sent home (explicit inventory, this is the core evidence surface): 20 two-piece cut-to-fit skin barriers/wafers; 30 drainable pouches with closure clips; 1 box (10) barrier rings/seals; 1 tube ostomy paste; 1 box (50) adhesive-remover wipes; 1 box (30) skin-prep/barrier-film wipes; 1 bottle pouch deodorant; measuring guide and scissors; ostomy belt (1).\n- Teaching status at discharge: \"patient returned demonstration of pouch emptying; barrier change performed by nursing, patient observed only.\"\n- Referral explicitly states \"all supplies provided for 30 days; no anticipated supply need.\" (This line is the hook for the deliberate conflict below.)\n- Reorder pathway: DME supplier assigned, reorder by phone, 3–5 day delivery — i.e., an unmet need today is actionable but not instantly self-solvable.\n\nTODAY'S VISIT NOTES (VERSION A — the complicated chart) MUST SHOW:\n- Stoma assessment: stoma now ~24 mm (expected post-op shrinkage from the referral's 28 mm), still beefy red, viable, budded, at skin level on one edge due to a shallow crease in the left lower quadrant when the patient sits.\n- Peristomal skin: erythematous, moist, weepy denudement in a crescent along the 4-to-8 o'clock inferior aspect, ~2 cm wide, matching where output tracks under the barrier when seated; no candidal satellite lesions, no ulceration, no mucocutaneous separation. Patient reports burning/stinging under the wafer.\n- Seal failure: barrier lasting ~18–24 hours before leakage, versus expected 3–4 days; patient has changed the appliance 3 times in the last 2 days. Undermining of the wafer adhesive noted on removal, with stool tracking onto the denuded skin.\n- Supply inventory ON HAND, counted at the visit (this is what forces a SPECIFIC answer): barrier rings 8 of 10 remaining; wafers 14 remaining; pouches 22 remaining; paste ~¾ tube; adhesive-remover wipes ~40 remaining; ostomy belt unused in the drawer; SKIN-PREP / BARRIER-FILM WIPES: 0 — box empty, patient has been applying the barrier to wet, weepy skin without any protective film and did not know a refill was needed. Also absent from the home entirely: stoma powder / protective powder (never sent, not on the referral list) — so the chart supports a single, concrete, most-needed item rather than a generic \"more supplies.\"\n- Technique observation: patient cuts the wafer opening to the discharge measurement (28 mm) rather than the current 24 mm, leaving exposed skin; arthritic hands make scissor-cutting slow and imprecise. This is documented as a TEACHING need, not a supply need — it is a deliberate distractor that a weak answer will convert into \"needs pre-cut/moldable barriers.\"\n- Output character today: matches the referral (soft-formed, 2–3× daily) — deliberately NOT high-output, so \"needs high-output/drainable-with-spout pouches\" is unsupported.\n- Vitals/systemic: afebrile, no peri-stomal cellulitis, blood glucose mildly elevated; pain 3/10 burning at the skin only.\n- Patient goal stated in their own words: \"I want it to stay on so I can go to my grandchild's recital without worrying.\"\n\nTHE DEFENSIBLE ANSWER in Version A: the skin-prep / barrier-film wipes (the exhausted item), because the failing link in the chain is unprotected, weeping peristomal skin under an adhesive barrier — rings are on hand and already in use, pouches and wafers are stocked, output does not justify a different pouch system, and the wafer-sizing error is a teaching correction, not a purchase. A close-second defensible answer (stoma/protective powder, never supplied) is intentionally reachable, so graders can distinguish \"picked the empty box\" from \"reasoned about the crusting technique the weepy skin actually needs.\"\n\nVERSION B — the 'None' chart (same patient characteristics, later state; or a second shelf patient in the same age band if the shelf prefers distinct records):\n- Stoma age: established, ~14 months; stoma ~24 mm, stable size, beefy red, budded, no crease interference; patient uses a moldable/pre-sized barrier.\n- Peristomal skin: intact, no erythema, no denudement, no maceration; skin described as fully healed with no breakdown anywhere in the visit note.\n- Wear time: 4 days consistently, no leakage reported since the last two visits; patient independently empties and changes, returns demonstration flawlessly.\n- Supply inventory: full cupboard — wafers 18, pouches 40, barrier rings 9, skin-prep wipes 45, adhesive remover 50, paste ~full, powder 1 unopened bottle, belt available; 30-day reorder already placed with the DME supplier and confirmed in transit.\n- Output character: soft-formed, predictable, 1–2 emptyings daily, consistent with the referral.\n- Visit purpose: routine reassessment/recertification, no complaint.\n- The defensible answer is 'None' — no supply is deficient, no skin or seal problem creates a need. Version B also carries one benign non-supply need (a dietitian question about gas-producing foods) so 'None' has to be chosen on supply grounds, not because the chart is empty.\n\nDELIBERATE REFERRAL-vs-NOTES CONFLICTS (the part that tests the question rather than pattern-matching):\n1. STOMA SIZE DRIFT: referral says 28 mm; today's measurement is 24 mm. A model that trusts the referral packet will endorse the patient's 28 mm cut and miss that exposed skin is why the seal fails. The correct read is that current assessment overrides discharge documentation.\n2. \"NO ANTICIPATED SUPPLY NEED\": the referral asserts 30 days of supplies and no need; the visit count proves one box is at zero and one item was never supplied at all. A model that defers to the referral's blanket statement will answer 'None' on a chart that clearly supports a specific item — this is the primary trap separating Version A from Version B.\n3. PERISTOMAL SKIN STATUS: referral says \"peristomal skin intact at discharge\"; today's note documents weepy denudement. Same field, opposite value, with the visit note being the current truth.\n4. SUPPLY COUNT vs SUPPLY LIST: the referral's list includes skin-prep wipes, so a list-reading model will conclude the patient has them; only the counted inventory in today's note reveals the box is empty. The evidence needed to answer lives in the notes, not the packet.\n5. WEAR-TIME EXPECTATION: referral implies 3–4 day wear; notes show 18–24 hours. The gap is the clinical signal that something under the barrier — skin, not the pouch — is failing.\n\nGUARDRAILS FOR AUTHORING: no surnames, no dates or date-like strings (use relative intervals only), no MRN/account/encounter numbers, no phone numbers, no addresses, no facility or clinician names (use \"the discharging hospital\", \"the home health nurse\", \"the DME supplier\"), no insurance IDs. Keep the referral packet and visit note as separate documents so the conflict is only visible when both are read. Every fact that drives the answer — stoma measurement, skin condition, wear time, and the counted inventory with explicit zeros — must appear verbatim in the chart text, so a grader can point to the line that justifies each answer path." +} diff --git a/examples/prompt-lab/evidence/run/lab/work/q-wound-status/d61687f9/grid.json b/examples/prompt-lab/evidence/run/lab/work/q-wound-status/d61687f9/grid.json new file mode 100644 index 00000000..7ea267f8 --- /dev/null +++ b/examples/prompt-lab/evidence/run/lab/work/q-wound-status/d61687f9/grid.json @@ -0,0 +1,72 @@ +[ + { + "questionId": "wound-status", + "patientId": "jordan", + "pile": "shared", + "highlight": [], + "ai": { + "answer": "No wound", + "confidence": "High", + "explanation": "The referral packet's intake form documents \"No wounds documented. Skin intact,\" and today's SOC visit note confirms \"Skin intact, no wounds\" — there is no primary wound to stage, and no discharge summary states a wound was closed or healed." + }, + "target": { + "answer": "No wound", + "confidence": "High", + "explanation": "The referral packet's intake form documents \"No wounds documented. Skin intact,\" and today's SOC visit note confirms \"Skin intact, no wounds\" — there is no primary wound to stage, and no discharge summary states a wound was closed or healed." + }, + "notes": "" + }, + { + "questionId": "wound-status", + "patientId": "marguerite", + "pile": "shared", + "highlight": [ + "medium confidence" + ], + "ai": { + "answer": "Ongoing", + "confidence": "Medium", + "explanation": "The referral packet never states the wound is closed or healed — it only records peristomal skin intact at discharge — so the visit notes govern, and today's assessment documents erythematous, moist, weepy denudement about 2 cm wide along the 4-to-8 o'clock peristomal aspect with patient-reported burning and stinging. Confidence is Medium rather than High because the status comes from the visit note, not a referral statement." + }, + "target": { + "answer": "Ongoing", + "confidence": "Medium", + "explanation": "The referral packet never states the wound is closed or healed — it only records peristomal skin intact at discharge — so the visit notes govern, and today's assessment documents erythematous, moist, weepy denudement about 2 cm wide along the 4-to-8 o'clock peristomal aspect with patient-reported burning and stinging. Confidence is Medium rather than High because the status comes from the visit note, not a referral statement." + }, + "notes": "" + }, + { + "questionId": "wound-status", + "patientId": "pat", + "pile": "shared", + "highlight": [], + "ai": { + "answer": "Ongoing", + "confidence": "High", + "explanation": "Today's SOC visit documents an open left heel wound measuring 2.0 x 1.5 cm with moderate serous drainage, pink wound bed and yellow slough, dressed with foam per protocol. The discharge summary's \"wound closed\" statement is contradicted by direct assessment at this visit, so the current status is Ongoing." + }, + "target": { + "answer": "Ongoing", + "confidence": "High", + "explanation": "Today's visit notes describe an open 2.0 x 1.5 cm left heel wound with serous drainage and slough; the discharge summary's closed status is outdated." + }, + "notes": "" + }, + { + "questionId": "wound-status", + "patientId": "riley", + "pile": "shared", + "highlight": [], + "ai": { + "answer": "No wound", + "confidence": "High", + "explanation": "The referral packet states 'No wounds' and today's SOC visit note confirms 'No wounds; skin intact around cast,' so there is no primary wound to track." + }, + "target": { + "answer": "No wound", + "confidence": "High", + "explanation": "The referral packet states 'No wounds' and today's SOC visit note confirms 'No wounds; skin intact around cast,' so there is no primary wound to track." + }, + "notes": "" + } +] diff --git a/examples/prompt-lab/evidence/run/lab/work/q-wound-status/d61687f9/score.json b/examples/prompt-lab/evidence/run/lab/work/q-wound-status/d61687f9/score.json new file mode 100644 index 00000000..22f032f4 --- /dev/null +++ b/examples/prompt-lab/evidence/run/lab/work/q-wound-status/d61687f9/score.json @@ -0,0 +1,54 @@ +{ + "promptId": "p-wound-status-f41f7ca1", + "rows": [ + { + "patientId": "jordan", + "gold": "No wound", + "rerun": "No wound", + "result": "worked" + }, + { + "patientId": "marguerite", + "gold": "Ongoing", + "rerun": "Ongoing", + "result": "worked" + }, + { + "patientId": "pat", + "gold": "Ongoing", + "rerun": "Ongoing", + "result": "worked" + }, + { + "patientId": "riley", + "gold": "No wound", + "rerun": "No wound", + "result": "worked" + } + ], + "golded": 4, + "worked": 4, + "percent": 100, + "outputs": { + "jordan": { + "answer": "No wound", + "confidence": "High", + "explanation": "Today's skilled nursing SOC visit directly assessed the skin and found it intact with no wounds, and the referral intake form likewise documented no wounds and intact skin. Nothing in the chart records any wound history or wound findings." + }, + "marguerite": { + "answer": "Ongoing", + "confidence": "High", + "explanation": "Today's start-of-care visit directly assessed the site and found erythematous, moist, weepy peristomal denudement in a 2 cm crescent along the 4-to-8 o'clock inferior aspect, with stool tracking under an undermined wafer and the patient reporting burning and stinging. The referral packet's note that peristomal skin was intact at discharge describes the site as of that earlier date and is outdated." + }, + "pat": { + "answer": "Ongoing", + "confidence": "High", + "explanation": "Today's SOC skilled nursing visit directly assessed the left heel and found an open area 2.0 x 1.5 cm with moderate serous drainage and a pink wound bed with yellow slough at the edges, with foam dressing applied. The hospital discharge summary's note that the wound was closed reflects status on that earlier date and is outdated." + }, + "riley": { + "answer": "No wound", + "confidence": "High", + "explanation": "Today's skilled nursing SOC visit directly assessed the skin and documented 'No wounds; skin intact around cast,' and the referral packet likewise records no wounds with only a right wrist fracture in a cast. Nothing in the chart documents any wound history or wound findings." + } + } +} diff --git a/examples/prompt-lab/evidence/run/lab/work/sunrise-soc/62c71a0f/commit.json b/examples/prompt-lab/evidence/run/lab/work/sunrise-soc/62c71a0f/commit.json new file mode 100644 index 00000000..236cb5ee --- /dev/null +++ b/examples/prompt-lab/evidence/run/lab/work/sunrise-soc/62c71a0f/commit.json @@ -0,0 +1,6 @@ +{ + "mode": "except", + "questions": [ + "ostomy-supplies" + ] +} diff --git a/examples/prompt-lab/evidence/run/lab/work/sunrise-soc/62c71a0f/grid.json b/examples/prompt-lab/evidence/run/lab/work/sunrise-soc/62c71a0f/grid.json new file mode 100644 index 00000000..5d19a845 --- /dev/null +++ b/examples/prompt-lab/evidence/run/lab/work/sunrise-soc/62c71a0f/grid.json @@ -0,0 +1,168 @@ +[ + { + "questionId": "living-situation", + "patientId": "jordan", + "pile": "agency-specific", + "highlight": [ + "first-pass prompt, never reviewed" + ], + "ai": { + "answer": "Lives with spouse", + "confidence": "High", + "explanation": "Today's visit note narrative states the spouse was present throughout the SOC visit, that patient and spouse share the home, and that the spouse sets up the pill organizer each morning. This overrides the intake form completed by the hospital case manager, which recorded living alone." + }, + "target": { + "answer": "Lives with spouse", + "confidence": "High", + "explanation": "Today's visit note narrative states the spouse was present throughout the SOC visit, that patient and spouse share the home, and that the spouse sets up the pill organizer each morning. This overrides the intake form completed by the hospital case manager, which recorded living alone." + }, + "notes": "" + }, + { + "questionId": "living-situation", + "patientId": "pat", + "pile": "agency-specific", + "highlight": [ + "first-pass prompt, never reviewed" + ], + "ai": { + "answer": "Lives alone", + "confidence": "High", + "explanation": "Today's SOC visit note narrative states directly that the patient lives alone in a one-story home, with a neighbor who checks in daily; a daily check-in visitor does not change residence status. This agrees with the referral packet's intake-form social history, so no override was needed." + }, + "target": { + "answer": "Lives alone", + "confidence": "High", + "explanation": "Today's SOC visit note narrative states directly that the patient lives alone in a one-story home, with a neighbor who checks in daily; a daily check-in visitor does not change residence status. This agrees with the referral packet's intake-form social history, so no override was needed." + }, + "notes": "" + }, + { + "questionId": "living-situation", + "patientId": "riley", + "pile": "agency-specific", + "highlight": [ + "first-pass prompt, never reviewed" + ], + "ai": { + "answer": "Lives with family", + "confidence": "High", + "explanation": "Today's skilled nursing SOC visit note states directly that the patient resides with her daughter, son-in-law and two grandchildren. This agrees with the referral packet, which also describes her living with her daughter's family." + }, + "target": { + "answer": "Lives with family", + "confidence": "High", + "explanation": "Today's skilled nursing SOC visit note states directly that the patient resides with her daughter, son-in-law and two grandchildren. This agrees with the referral packet, which also describes her living with her daughter's family." + }, + "notes": "" + }, + { + "questionId": "mood", + "patientId": "jordan", + "pile": "mismatch", + "highlight": [ + "medium confidence", + "shared prompt, this agency's menu or ancestry differs" + ], + "ai": { + "answer": "Calm", + "confidence": "Medium", + "explanation": "Today's visit notes describe the patient as \"relaxed and engaged\" and asking good questions about the new diuretic, but the mood is inferred from the clinician's behavioral description rather than stated by the patient in their own words." + }, + "target": { + "answer": "Calm", + "confidence": "Medium", + "explanation": "Today's visit notes describe the patient as \"relaxed and engaged\" and asking good questions about the new diuretic, but the mood is inferred from the clinician's behavioral description rather than stated by the patient in their own words." + }, + "notes": "" + }, + { + "questionId": "mood", + "patientId": "pat", + "pile": "mismatch", + "highlight": [ + "shared prompt, this agency's menu or ancestry differs" + ], + "ai": { + "answer": "Calm", + "confidence": "High", + "explanation": "Today's visit notes describe the patient as \"calm and pleasant, joking about hospital food,\" and she states in her own words that she is \"glad to be home.\"" + }, + "target": { + "answer": "Calm", + "confidence": "High", + "explanation": "Today's visit notes describe the patient as \"calm and pleasant, joking about hospital food,\" and she states in her own words that she is \"glad to be home.\"" + }, + "notes": "" + }, + { + "questionId": "mood", + "patientId": "riley", + "pile": "mismatch", + "highlight": [ + "shared prompt, this agency's menu or ancestry differs" + ], + "ai": { + "answer": "Anxious", + "confidence": "High", + "explanation": "The patient states in her own words, 'I can't stop thinking about falling again, I lie awake worrying about it,' and the clinician notes she wrings her hands during the medication review." + }, + "target": { + "answer": "Anxious", + "confidence": "High", + "explanation": "The patient states in her own words, 'I can't stop thinking about falling again, I lie awake worrying about it,' and the clinician notes she wrings her hands during the medication review." + }, + "notes": "" + }, + { + "questionId": "wound-status", + "patientId": "jordan", + "pile": "shared", + "highlight": [], + "ai": { + "answer": "No wound", + "confidence": "High", + "explanation": "The referral packet's intake form documents 'No wounds documented. Skin intact,' and today's SOC visit note confirms 'Skin intact, no wounds' — no wound was ever present, so there is nothing to have healed." + }, + "target": { + "answer": "No wound", + "confidence": "High", + "explanation": "The referral packet's intake form documents 'No wounds documented. Skin intact,' and today's SOC visit note confirms 'Skin intact, no wounds' — no wound was ever present, so there is nothing to have healed." + }, + "notes": "" + }, + { + "questionId": "wound-status", + "patientId": "pat", + "pile": "shared", + "highlight": [], + "ai": { + "answer": "Ongoing", + "confidence": "High", + "explanation": "Today's SOC visit documents an open left heel wound measuring 2.0 x 1.5 cm with moderate serous drainage and yellow slough, dressed with foam per protocol, so the discharge summary's \"wound closed\" note is outdated and the current assessment governs." + }, + "target": { + "answer": "Ongoing", + "confidence": "High", + "explanation": "Today's visit notes describe an open 2.0 x 1.5 cm left heel wound with serous drainage and slough; the discharge summary's closed status is outdated." + }, + "notes": "The prompt makes the referral win over what the nurse saw today. Today's notes must win (shared guideline 1)." + }, + { + "questionId": "wound-status", + "patientId": "riley", + "pile": "shared", + "highlight": [], + "ai": { + "answer": "No wound", + "confidence": "High", + "explanation": "The referral packet states \"No wounds\" and today's SOC visit note confirms \"No wounds; skin intact around cast,\" so there is no primary wound to stage or track." + }, + "target": { + "answer": "No wound", + "confidence": "High", + "explanation": "The referral packet states \"No wounds\" and today's SOC visit note confirms \"No wounds; skin intact around cast,\" so there is no primary wound to stage or track." + }, + "notes": "" + } +] diff --git a/examples/prompt-lab/evidence/runtime-findings/00-job1-attempt1-fenced-json.txt b/examples/prompt-lab/evidence/runtime-findings/00-job1-attempt1-fenced-json.txt new file mode 100644 index 00000000..938987ab --- /dev/null +++ b/examples/prompt-lab/evidence/runtime-findings/00-job1-attempt1-fenced-json.txt @@ -0,0 +1,57 @@ +$ flows run --local-agent --agent-capacity 4 prompt-lab.flow.ts --input {"job":"new-agency","reviewer":"prompt-lab-reviewer","lab":"evidence/lab","agency":"sunrise","visitType":"soc"} +○ run-1 (deterministic) 0.00s +✓ run-1 (deterministic) 0.23s completionReason: success +○ llm-2 (llm) 0.00s +WAITING [worker_lease] Run "01M364PGHQH88R6M3WYD3ZK1WV" step "llm-2" (llm) is running under a worker lease until 1790133909384. +↻ llm-2 (llm) 4.84s +✓ llm-2 (llm) 14.87s completionReason: success +○ llm-3 (llm) 0.00s +WAITING [worker_lease] Run "01M364PZJZAX3BYGEH3QPG0646" step "llm-3" (llm) is running under a worker lease until 1790133924794. +↻ llm-3 (llm) 5.41s +WAITING [worker_lease] Run "01M364PZJZAX3BYGEH3QPG0646" step "llm-3" (llm) is running under a worker lease until 1790133934901. +↻ llm-3 (llm) 15.48s +✓ llm-3 (llm) 17.42s completionReason: success +○ run-4 (deterministic) 0.00s +✓ run-4 (deterministic) 0.29s completionReason: success +○ llm-5 (llm) 0.00s +WAITING [worker_lease] Run "01M364QG9AW260TM3DTTFPHA32" step "llm-5" (llm) is running under a worker lease until 1790133941942. +↻ llm-5 (llm) 4.98s +WAITING [worker_lease] Run "01M364QG9AW260TM3DTTFPHA32" step "llm-5" (llm) is running under a worker lease until 1790133951954. +↻ llm-5 (llm) 14.84s +✓ llm-5 (llm) 15.38s completionReason: success +○ llm-6 (llm) 0.00s +WAITING [worker_lease] Run "01M364QYTJ1K4P7YJFR4RJ2AGB" step "llm-6" (llm) is running under a worker lease until 1790133956822. +↻ llm-6 (llm) 4.38s +WAITING [worker_lease] Run "01M364QYTJ1K4P7YJFR4RJ2AGB" step "llm-6" (llm) is running under a worker lease until 1790133966890. +↻ llm-6 (llm) 14.44s +✓ llm-6 (llm) 20.39s completionReason: success +○ run-7 (deterministic) 0.00s +✓ run-7 (deterministic) 0.15s completionReason: success +○ llm-8 (llm) 0.00s +WAITING [worker_lease] Run "01M364RKM12W308QV9MJF1GJTK" step "llm-8" (llm) is running under a worker lease until 1790133978095. +↻ llm-8 (llm) 5.08s +WAITING [worker_lease] Run "01M364RKM12W308QV9MJF1GJTK" step "llm-8" (llm) is running under a worker lease until 1790133988117. +↻ llm-8 (llm) 15.07s +WAITING [worker_lease] Run "01M364RKM12W308QV9MJF1GJTK" step "llm-8" (llm) is running under a worker lease until 1790133998128. +↻ llm-8 (llm) 25.14s +WAITING [worker_lease] Run "01M364RKM12W308QV9MJF1GJTK" step "llm-8" (llm) is running under a worker lease until 1790134008171. +↻ llm-8 (llm) 35.08s +✗ llm-8 (llm) 36.20s +FAILED [step_failed] Run "01M364RKM12W308QV9MJF1GJTK" failed with completionReason: step_failed. Step "llm-8" (llm) completionReason: verification_failed attempt=1/1. +Detail: ```json +{ + "coverage": [ + { "questionId": "wound-status", "patientIds": ["jordan", "pat", "riley"] }, + { "questionId": "mood", "patientIds": ["jordan", "pat", "riley"] }, + { "questionId": "living-situation", "patientIds": ["jordan", "pat", "riley"] } + ], + "gaps": [ + { "questionId": "ostomy-supplies", "brief": "No shelf chart mentions an ostomy at all. Needs a patient with a documented colostomy/ileostomy in the referral packet, whose today's visit note specifies the actual supply used or needed at this visit (e.g., pouch change due to leakage, barrier ring reapplied, or skin prep wipe used during appliance change) so each option has real evidence." } + ] +} +``` +Inspect: flows replay 01M364RKM12W308QV9MJF1GJTK --at llm-8 --data-dir '/private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/e2e-data' +Journal: /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/e2e-data/runs/01M364RKM12W308QV9MJF1GJTK.sqlite3 +RUN 01M364RKM12W308QV9MJF1GJTK failed completionReason: step_failed +npx flows run --no-observer-link --data-dir $DD --local-agent --agent-capacit 18.02s user 5.31s system 19% cpu 1:58.98 total +exit=1 diff --git a/examples/prompt-lab/evidence/runtime-findings/00-job1-attempt2-parallel-lease-expired.txt b/examples/prompt-lab/evidence/runtime-findings/00-job1-attempt2-parallel-lease-expired.txt new file mode 100644 index 00000000..17f641c8 --- /dev/null +++ b/examples/prompt-lab/evidence/runtime-findings/00-job1-attempt2-parallel-lease-expired.txt @@ -0,0 +1,56 @@ +$ flows run --local-agent --agent-capacity 4 prompt-lab.flow.ts --input {"job":"new-agency","reviewer":"prompt-lab-reviewer","lab":"evidence/lab","agency":"sunrise","visitType":"soc"} +○ run-1 (deterministic) 0.00s +✓ run-1 (deterministic) 0.07s completionReason: success +○ llm-2 (llm) 0.00s +WAITING [worker_lease] Run "01M364TP1MGMTD1MW0ZXPYGXDA" step "llm-2" (llm) is running under a worker lease until 1790134046058. +↻ llm-2 (llm) 5.98s +✓ llm-2 (llm) 14.69s completionReason: success +○ llm-3 (llm) 0.00s +WAITING [worker_lease] Run "01M364V251DEE26VC670W6R8SW" step "llm-3" (llm) is running under a worker lease until 1790134058454. +↻ llm-3 (llm) 3.69s +✓ llm-3 (llm) 9.97s completionReason: success +○ run-4 (deterministic) 0.00s +✓ run-4 (deterministic) 0.06s completionReason: success +○ llm-5 (llm) 0.00s +WAITING [worker_lease] Run "01M364VDZKVFZFTQKP4Q9X8KNT" step "llm-5" (llm) is running under a worker lease until 1790134070567. +↻ llm-5 (llm) 5.77s +WAITING [worker_lease] Run "01M364VDZKVFZFTQKP4Q9X8KNT" step "llm-5" (llm) is running under a worker lease until 1790134080571. +↻ llm-5 (llm) 15.78s +✓ llm-5 (llm) 15.95s completionReason: success +○ llm-6 (llm) 0.00s +WAITING [worker_lease] Run "01M364VXFEKG6EGVMK5TG88S4F" step "llm-6" (llm) is running under a worker lease until 1790134086434. +↻ llm-6 (llm) 5.68s +✓ llm-6 (llm) 14.59s completionReason: success +○ run-7 (deterministic) 0.00s +✓ run-7 (deterministic) 0.08s completionReason: success +○ llm-8 (llm) 0.00s +WAITING [worker_lease] Run "01M364WBAN8ZCF7ZKG56SF4RR0" step "llm-8" (llm) is running under a worker lease until 1790134100620. +↻ llm-8 (llm) 5.20s +WAITING [worker_lease] Run "01M364WBAN8ZCF7ZKG56SF4RR0" step "llm-8" (llm) is running under a worker lease until 1790134110624. +↻ llm-8 (llm) 15.23s +✓ llm-8 (llm) 20.72s completionReason: success +○ llm-8.gate (deterministic) 0.00s +✓ llm-8.gate (deterministic) 0.01s completionReason: success +○ run-9 (deterministic) 0.00s +✓ run-9 (deterministic) 0.07s completionReason: success +○ llm-10 (llm) 0.00s +○ llm-11 (llm) 0.00s +○ llm-12 (llm) 0.00s +○ llm-13 (llm) 0.00s +○ llm-14 (llm) 0.00s +○ llm-15 (llm) 0.00s +○ llm-16 (llm) 0.00s +○ llm-17 (llm) 0.00s +○ llm-18 (llm) 0.00s +✗ llm-13 (llm) 25.89s +✗ llm-10 (llm) 39.22s +✗ llm-11 (llm) 33.54s +✗ llm-12 (llm) 30.12s +✗ llm-18 (llm) 4.53s +✗ llm-14 (llm) 20.80s +✗ llm-15 (llm) 17.33s +✗ llm-16 (llm) 13.86s +✗ llm-17 (llm) 9.69s +FAILED [protocol_error] relayflowd could not complete the run request: Agent lease is already expired for 01M364X3FB5C3DEF4BX7RN7KS3/llm-11. +npx flows run --no-observer-link --data-dir $DD --local-agent --agent-capacit 41.63s user 12.76s system 41% cpu 2:11.69 total +exit=1 diff --git a/examples/prompt-lab/evidence/runtime-findings/00-job1-attempt3-capacity1-lease-expired.txt b/examples/prompt-lab/evidence/runtime-findings/00-job1-attempt3-capacity1-lease-expired.txt new file mode 100644 index 00000000..2289efa7 --- /dev/null +++ b/examples/prompt-lab/evidence/runtime-findings/00-job1-attempt3-capacity1-lease-expired.txt @@ -0,0 +1,56 @@ +$ flows run --local-agent --agent-capacity 1 prompt-lab.flow.ts --input {"job":"new-agency","reviewer":"prompt-lab-reviewer","lab":"evidence/lab","agency":"sunrise","visitType":"soc"} +○ run-1 (deterministic) 0.00s +✓ run-1 (deterministic) 0.06s completionReason: success +○ llm-2 (llm) 0.00s +WAITING [worker_lease] Run "01M364ZXSF8WWDDF7HACDTKEAY" step "llm-2" (llm) is running under a worker lease until 1790134217827. +↻ llm-2 (llm) 3.74s +✓ llm-2 (llm) 13.24s completionReason: success +○ llm-3 (llm) 0.00s +WAITING [worker_lease] Run "01M3650AKFSSWFVRKYH31PHK5R" step "llm-3" (llm) is running under a worker lease until 1790134230949. +↻ llm-3 (llm) 3.62s +✓ llm-3 (llm) 11.90s completionReason: success +○ run-4 (deterministic) 0.00s +✓ run-4 (deterministic) 0.08s completionReason: success +○ llm-5 (llm) 0.00s +WAITING [worker_lease] Run "01M3650Q6HKHGSG4B3WR7KJQ2V" step "llm-5" (llm) is running under a worker lease until 1790134243849. +↻ llm-5 (llm) 4.54s +✓ llm-5 (llm) 12.38s completionReason: success +○ llm-6 (llm) 0.00s +WAITING [worker_lease] Run "01M36513PZ24BGB93RC697G12H" step "llm-6" (llm) is running under a worker lease until 1790134256661. +↻ llm-6 (llm) 4.98s +✓ llm-6 (llm) 13.52s completionReason: success +○ run-7 (deterministic) 0.00s +✓ run-7 (deterministic) 0.08s completionReason: success +○ llm-8 (llm) 0.00s +WAITING [worker_lease] Run "01M3651HB59NEZDKEFVVF3YT5X" step "llm-8" (llm) is running under a worker lease until 1790134270620. +↻ llm-8 (llm) 5.34s +WAITING [worker_lease] Run "01M3651HB59NEZDKEFVVF3YT5X" step "llm-8" (llm) is running under a worker lease until 1790134280625. +↻ llm-8 (llm) 15.33s +WAITING [worker_lease] Run "01M3651HB59NEZDKEFVVF3YT5X" step "llm-8" (llm) is running under a worker lease until 1790134290628. +↻ llm-8 (llm) 25.34s +✓ llm-8 (llm) 33.09s completionReason: success +○ llm-8.gate (deterministic) 0.00s +✓ llm-8.gate (deterministic) 0.01s completionReason: success +○ run-9 (deterministic) 0.00s +✓ run-9 (deterministic) 0.06s completionReason: success +○ llm-10 (llm) 0.00s +○ llm-11 (llm) 0.00s +○ llm-12 (llm) 0.00s +○ llm-13 (llm) 0.00s +○ llm-14 (llm) 0.00s +○ llm-15 (llm) 0.00s +○ llm-16 (llm) 0.00s +○ llm-17 (llm) 0.00s +○ llm-18 (llm) 0.00s +✗ llm-10 (llm) 44.39s +✗ llm-12 (llm) 37.05s +✗ llm-13 (llm) 32.67s +✗ llm-14 (llm) 26.26s +✗ llm-15 (llm) 21.41s +✗ llm-16 (llm) 16.72s +✗ llm-17 (llm) 10.52s +✗ llm-18 (llm) 4.64s +✗ llm-11 (llm) 40.62s +FAILED [protocol_error] relayflowd could not complete the run request: Agent lease is already expired for 01M3652G7G2N0YS360H4ZVVYK3/llm-10. +npx flows run --no-observer-link --data-dir $DD --local-agent --agent-capacit 38.19s user 11.65s system 37% cpu 2:14.68 total +exit=1 diff --git a/examples/prompt-lab/evidence/runtime-findings/00-job1-attempt4-predicate-gate-resume-conflict.txt b/examples/prompt-lab/evidence/runtime-findings/00-job1-attempt4-predicate-gate-resume-conflict.txt new file mode 100644 index 00000000..9b199a23 --- /dev/null +++ b/examples/prompt-lab/evidence/runtime-findings/00-job1-attempt4-predicate-gate-resume-conflict.txt @@ -0,0 +1,27 @@ +$ flows answer 01M365CQ2342WZDX3T21CN5905 human-18 yes --by prompt-lab-reviewer +ANSWERED 01M365CQ2342WZDX3T21CN5905 human-18 yes +Continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/e2e-data --local-agent 01M365CQ2342WZDX3T21CN5905 +exit=0 +$ flows resume --local-agent 01M365CQ2342WZDX3T21CN5905 +○ run-1 (deterministic) 0.00s +✓ run-1 (deterministic) 0.01s completionReason: success +○ llm-2 (llm) 0.00s +✓ llm-2 (llm) 3.20s completionReason: success +○ llm-3 (llm) 0.00s +✓ llm-3 (llm) 4.19s completionReason: success +○ run-4 (deterministic) 0.00s +✓ run-4 (deterministic) 0.01s completionReason: success +○ llm-5 (llm) 0.00s +✓ llm-5 (llm) 3.36s completionReason: success +○ llm-6 (llm) 0.00s +✓ llm-6 (llm) 3.63s completionReason: success +○ run-7 (deterministic) 0.00s +✓ run-7 (deterministic) 0.01s completionReason: success +○ llm-8 (llm) 0.00s +✓ llm-8 (llm) 6.40s completionReason: success +○ llm-8.gate (deterministic) 0.00s +✗ llm-8.gate (deterministic) 0.00s +FAILED [protocol_error] relayflowd could not complete the resume request: run_admission_conflict: run admission key "authored-child:080c430bb5c4847d0ed3da903df8518e818d16e860cb1e5f16269ca8a2544aba" is already bound to a different spec +RUN 01M365CQ2342WZDX3T21CN5905 unknown +npx flows resume --no-observer-link --data-dir $DD --local-agent $R 8.79s user 2.61s system 23% cpu 47.952 total +exit=1 diff --git a/examples/prompt-lab/evidence/runtime-findings/00-job1-attempt4-run.txt b/examples/prompt-lab/evidence/runtime-findings/00-job1-attempt4-run.txt new file mode 100644 index 00000000..76aba898 --- /dev/null +++ b/examples/prompt-lab/evidence/runtime-findings/00-job1-attempt4-run.txt @@ -0,0 +1,73 @@ +$ flows run --local-agent prompt-lab.flow.ts --input {"job":"new-agency","reviewer":"prompt-lab-reviewer","lab":"evidence/lab","agency":"sunrise","visitType":"soc"} +○ run-1 (deterministic) 0.00s +✓ run-1 (deterministic) 0.06s completionReason: success +○ llm-2 (llm) 0.00s +WAITING [worker_lease] Run "01M365CV7NKTC3HRY8QDGMC830" step "llm-2" (llm) is running under a worker lease until 1790134641216. +↻ llm-2 (llm) 4.23s +✓ llm-2 (llm) 11.49s completionReason: success +○ llm-3 (llm) 0.00s +WAITING [worker_lease] Run "01M365D6R9BJSFJCZAC4CYX1GN" step "llm-3" (llm) is running under a worker lease until 1790134652988. +↻ llm-3 (llm) 4.51s +✓ llm-3 (llm) 9.56s completionReason: success +○ run-4 (deterministic) 0.00s +✓ run-4 (deterministic) 0.06s completionReason: success +○ llm-5 (llm) 0.00s +WAITING [worker_lease] Run "01M365DFGHCMRNH3XX2QV0MYMP" step "llm-5" (llm) is running under a worker lease until 1790134661957. +↻ llm-5 (llm) 3.85s +✓ llm-5 (llm) 10.49s completionReason: success +○ llm-6 (llm) 0.00s +WAITING [worker_lease] Run "01M365DSE8X1P36VKZES2V0936" step "llm-6" (llm) is running under a worker lease until 1790134672122. +↻ llm-6 (llm) 3.53s +✓ llm-6 (llm) 9.62s completionReason: success +○ run-7 (deterministic) 0.00s +✓ run-7 (deterministic) 0.06s completionReason: success +○ llm-8 (llm) 0.00s +WAITING [worker_lease] Run "01M365E3HJHZ8ZPYMC5K3FERJ3" step "llm-8" (llm) is running under a worker lease until 1790134682470. +↻ llm-8 (llm) 4.19s +WAITING [worker_lease] Run "01M365E3HJHZ8ZPYMC5K3FERJ3" step "llm-8" (llm) is running under a worker lease until 1790134692475. +↻ llm-8 (llm) 14.23s +WAITING [worker_lease] Run "01M365E3HJHZ8ZPYMC5K3FERJ3" step "llm-8" (llm) is running under a worker lease until 1790134702478. +↻ llm-8 (llm) 24.23s +✓ llm-8 (llm) 24.61s completionReason: success +○ llm-8.gate (deterministic) 0.00s +✓ llm-8.gate (deterministic) 0.01s completionReason: success +○ run-9 (deterministic) 0.00s +✓ run-9 (deterministic) 0.06s completionReason: success +○ llm-10 (llm) 0.00s +WAITING [worker_lease] Run "01M365ETPDM24HMEFG6Y88TWXY" step "llm-10" (llm) is running under a worker lease until 1790134706175. +↻ llm-10 (llm) 3.22s +✓ llm-10 (llm) 7.01s completionReason: success +○ llm-11 (llm) 0.00s +WAITING [worker_lease] Run "01M365F2V7MJ7XKEA9T0BBKVEE" step "llm-11" (llm) is running under a worker lease until 1790134714523. +↻ llm-11 (llm) 4.56s +✓ llm-11 (llm) 9.08s completionReason: success +○ llm-12 (llm) 0.00s +WAITING [worker_lease] Run "01M365FAV0T6QKC547GFYXYNX4" step "llm-12" (llm) is running under a worker lease until 1790134722707. +↻ llm-12 (llm) 3.66s +✓ llm-12 (llm) 7.68s completionReason: success +○ llm-13 (llm) 0.00s +WAITING [worker_lease] Run "01M365FJ0FFHXGXHXPDEDHX9Q0" step "llm-13" (llm) is running under a worker lease until 1790134730050. +↻ llm-13 (llm) 3.33s +✓ llm-13 (llm) 7.32s completionReason: success +○ llm-14 (llm) 0.00s +WAITING [worker_lease] Run "01M365FSZH6A7WN1NJG538WR44" step "llm-14" (llm) is running under a worker lease until 1790134738212. +↻ llm-14 (llm) 4.17s +✓ llm-14 (llm) 7.85s completionReason: success +○ llm-15 (llm) 0.00s +WAITING [worker_lease] Run "01M365G1HX02X7JWXRRGPC75F0" step "llm-15" (llm) is running under a worker lease until 1790134745968. +↻ llm-15 (llm) 4.07s +✓ llm-15 (llm) 8.33s completionReason: success +○ llm-16 (llm) 0.00s +WAITING [worker_lease] Run "01M365G9BMSER9RRAWKF2SPRCH" step "llm-16" (llm) is running under a worker lease until 1790134753959. +↻ llm-16 (llm) 3.73s +✓ llm-16 (llm) 7.15s completionReason: success +○ run-17 (deterministic) 0.00s +✓ run-17 (deterministic) 0.06s completionReason: success +○ human-18 (deterministic) 0.00s +⏸ human-18 (human) 0.00s +PARKED [run_parked] Run "01M365CQ2342WZDX3T21CN5905" is waiting for prompt-lab-reviewer to answer human-18: "Config workbench · sunrise soc: 7 rows on 3 questions (5 highlighted, listed first); 1 gap brief(s) queued for the test patient manager.\nEdit target answer / confidence / explanation / notes in evidence/lab/work/sunrise-soc/e4694a27/grid.json. Your first pass persists as the target for each question × patient.\nyes = persist targets and run iteration on every changed row; no = stop without persisting." +Answer with: flows answer --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/e2e-data 01M365CQ2342WZDX3T21CN5905 human-18 yes|no +Then continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/e2e-data --local-agent 01M365CQ2342WZDX3T21CN5905 +RUN 01M365CQ2342WZDX3T21CN5905 parked +npx flows run --no-observer-link --data-dir $DD --local-agent --input 40.46s user 12.06s system 43% cpu 2:01.92 total +exit=3 diff --git a/examples/prompt-lab/evidence/runtime-findings/00-patient-attempt1-fenced-json.txt b/examples/prompt-lab/evidence/runtime-findings/00-patient-attempt1-fenced-json.txt new file mode 100644 index 00000000..d99bd58c --- /dev/null +++ b/examples/prompt-lab/evidence/runtime-findings/00-patient-attempt1-fenced-json.txt @@ -0,0 +1,29 @@ +$ flows answer 01M365YAGJSVJH722B5NQC0ZBF human-4 yes --by prompt-lab-reviewer +ANSWERED 01M365YAGJSVJH722B5NQC0ZBF human-4 yes +Continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/e2e-data --local-agent 01M365YAGJSVJH722B5NQC0ZBF +exit=0 +$ flows resume --local-agent 01M365YAGJSVJH722B5NQC0ZBF +○ run-1 (deterministic) 0.00s +✓ run-1 (deterministic) 0.01s completionReason: success +○ llm-2 (llm) 0.00s +✓ llm-2 (llm) 5.54s completionReason: success +○ run-3 (deterministic) 0.00s +✓ run-3 (deterministic) 0.01s completionReason: success +○ human-4 (deterministic) 0.00s +✓ human-4 (deterministic) 0.01s completionReason: success +○ run-5 (deterministic) 0.00s +✓ run-5 (deterministic) 0.06s completionReason: success +○ llm-6 (llm) 0.00s +WAITING [worker_lease] Run "01M365ZFH4ZEPTTD2P9JH0K7W4" step "llm-6" (llm) is running under a worker lease until 1790135251804. +↻ llm-6 (llm) 3.99s +WAITING [worker_lease] Run "01M365ZFH4ZEPTTD2P9JH0K7W4" step "llm-6" (llm) is running under a worker lease until 1790135261808. +↻ llm-6 (llm) 14.02s +✗ llm-6 (llm) 14.35s +FAILED [step_failed] Run "01M365ZFH4ZEPTTD2P9JH0K7W4" failed with completionReason: step_failed. Step "llm-6" (llm) completionReason: verification_failed attempt=1/1. +Detail: ```json +{"id":"walter","label":"Walter","ageBand":"70s","visitType":"soc","referral":"Home-health referral for skilled nursing, ostomy care and teaching, 2x/week x 4 weeks, following laparoscopic sigmoid colectomy with new end colostomy for sigmoid colon cancer. History of hypertension and type 2 diabetes. Discharge summary notes the ostomy is new, with patient and caregiver education initiated inpatient. Ostomy nurse consult recommends a standard two-piece pouching system with a flat, non-convex barrier; stoma was flush with intact peristomal skin at discharge. Authorized supply list includes pouches and adhesive remover wipes only; no barrier rings or barrier paste listed. Patient lives alone in a single-story apartment and is independent with most ADLs; adult daughter visits twice weekly but is not present for cares. Early arthritis in hands affects fine-motor tasks such as pouch clip alignment.","notes":"Visit today for scheduled ostomy care and teaching per referral. Since discharge, stoma output has become looser and higher volume than documented at time of surgery. Peristomal skin now shows moisture-associated skin damage in the lower quadrant where output pools: redness, weeping, and superficial denudement noted on exam. The flat barrier currently authorized is visibly undermined and leaking within 4-6 hours of application instead of the expected 3-day wear time; patient reports frequent overnight leakage and verbalized frustration with disrupted sleep and repeated clothing changes. Assessment: flat barrier no longer adequate seal given change in stoma contour and output consistency since discharge. Stoma remeasured today; dimensions have changed from the discharge measurement, consistent with expected postoperative narrowing but now contributing to poor pouch fit. Pouch change performed this visit; skin barrier powder applied to weeping areas and a barrier ring/seal applied as a stopgap measure to improve adhesion and protect denuded skin until a durable so… (560 bytes truncated) +Inspect: flows replay 01M365ZFH4ZEPTTD2P9JH0K7W4 --at llm-6 --data-dir '/private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/e2e-data' +Journal: /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/e2e-data/runs/01M365ZFH4ZEPTTD2P9JH0K7W4.sqlite3 +RUN 01M365ZFH4ZEPTTD2P9JH0K7W4 failed completionReason: step_failed +npx flows resume --no-observer-link --data-dir $DD --local-agent $R 5.85s user 1.63s system 15% cpu 48.777 total +exit=1 diff --git a/examples/prompt-lab/evidence/runtime-findings/00-patient-attempt1-run.txt b/examples/prompt-lab/evidence/runtime-findings/00-patient-attempt1-run.txt new file mode 100644 index 00000000..8ccaae59 --- /dev/null +++ b/examples/prompt-lab/evidence/runtime-findings/00-patient-attempt1-run.txt @@ -0,0 +1,19 @@ +$ flows run --local-agent prompt-lab.flow.ts --input {"job":"patient","reviewer":"prompt-lab-reviewer","lab":"evidence/lab","briefId":"gap-ostomy-supplies"} +○ run-1 (deterministic) 0.00s +✓ run-1 (deterministic) 0.06s completionReason: success +○ llm-2 (llm) 0.00s +WAITING [worker_lease] Run "01M365YDYWF56D704H60HQ0Z9X" step "llm-2" (llm) is running under a worker lease until 1790135217425. +↻ llm-2 (llm) 3.46s +WAITING [worker_lease] Run "01M365YDYWF56D704H60HQ0Z9X" step "llm-2" (llm) is running under a worker lease until 1790135227427. +↻ llm-2 (llm) 13.48s +✓ llm-2 (llm) 20.62s completionReason: success +○ run-3 (deterministic) 0.00s +✓ run-3 (deterministic) 0.06s completionReason: success +○ human-4 (deterministic) 0.00s +⏸ human-4 (human) 0.00s +PARKED [run_parked] Run "01M365YAGJSVJH722B5NQC0ZBF" is waiting for prompt-lab-reviewer to answer human-4: "Test patient creator · gap-ostomy-supplies for ostomy-supplies.\nPlan:\nInvented patient: 'Walter', age band 70s, home-health referral for post-surgical ostomy care. Diagnoses: sigmoid colon cancer status-post laparoscopic sigmoid colectomy with new end colostomy; hypertension; type 2 diabetes (relevant because peristomal skin healing is slower and infection risk is higher). Living situation: lives alone in a single-story apartment, adult daughter visits twice a week but is not present for cares; patient is independent with most ADLs but has early arthritis in his hands affecting fine-motor tasks like pouch clip alignment. Referral packet contents: discharge summary from the surgical stay stating the ostomy is 'new, patient and caregiver education initiated inpatient,' ostomy nurse consult note recommending a standard two-piece pouching system with a flat (non-convex) barrier, and a home-health referral order for 'skilled nursing, ostomy care and teaching, 2x/week x 4 weeks' — with supply list on the referral limited to pouches and adhesive remover wipes only, no barrier rings or paste listed, since at discharge the stoma was flush and skin was intact. Today's visit notes (the deliberate conflict): stoma output has become looser/higher volume than at discharge, peristomal skin now shows moisture-associated skin damage (redness, weeping, superficial denudement) in the lower quadrant where output pools, and the flat barrier is visibly undermined/leaking within 4-6 hours instead of the expected 3-day wear time; nurse's assessment explicitly notes 'flat barrier no longer adequate seal given change in stoma contour/output consistency' and documents applying a barrier ring/seal plus skin barrier powder as a stopgap, but flags that the durable medical equipment supply order needs to be updated because the current authorized supplies (flat two-piece pouch + wipes only) do not include convex barrier products or barrier rings. Visit note also documents pouch change performed today, measurement of stoma size (changed from discharge measurement), and patient verbalizing frustration with frequent leakage overnight. This creates the intended tension: the referral/DME authorization says 'pouches + wipes are sufficient,' but today's visit note shows the patient's most pressing unmet need this visit is a convex barrier/barrier ring product to manage leakage and protect skin — testing whether the reader correctly identifies the barrier ring (convex, moldable skin barrier) as the supply the patient needs most right now, distinct from what the referral originally authorized. No real names, dates, MRNs, phone numbers, or addresses are used — only invented first-name label 'Walter' and relative/vague timing (e.g., 'since discharge,' 'overnight,' 'this visit').\nYou may edit \"plan\" in evidence/lab/work/patients/gap-ostomy-supplies/plan.json first.\nyes = kick generate (Patient QA then locks it onto the shelf; you do not approve the chart); no = leave it on the queue." +Answer with: flows answer --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/e2e-data 01M365YAGJSVJH722B5NQC0ZBF human-4 yes|no +Then continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/e2e-data --local-agent 01M365YAGJSVJH722B5NQC0ZBF +RUN 01M365YAGJSVJH722B5NQC0ZBF parked +npx flows run --no-observer-link --data-dir $DD --local-agent --input "$I" 4.44s user 1.21s system 26% cpu 21.493 total +exit=3 diff --git a/examples/prompt-lab/evidence/runtime-findings/00-prove-attempt1-lease-conflict-after-success.txt b/examples/prompt-lab/evidence/runtime-findings/00-prove-attempt1-lease-conflict-after-success.txt new file mode 100644 index 00000000..c299b6bb --- /dev/null +++ b/examples/prompt-lab/evidence/runtime-findings/00-prove-attempt1-lease-conflict-after-success.txt @@ -0,0 +1,25 @@ +$ npx flows run --no-observer-link --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data --local-agent prompt-lab.flow.ts --input {"job":"new-agency","reviewer":"prompt-lab-reviewer","lab":"evidence/run/lab","agency":"sunrise","visitType":"soc"} +○ run-1 (deterministic) 0.00s +✓ run-1 (deterministic) 0.06s completionReason: success +○ llm-2 (llm) 0.00s +WAITING [worker_lease] Run "01M3665A2TP7DF8X6PBK5F0Q6H" step "llm-2" (llm) is running under a worker lease until 1790135442831. +↻ llm-2 (llm) 6.11s +WAITING [worker_lease] Run "01M3665A2TP7DF8X6PBK5F0Q6H" step "llm-2" (llm) is running under a worker lease until 1790135452836. +↻ llm-2 (llm) 16.16s +WAITING [worker_lease] Run "01M3665A2TP7DF8X6PBK5F0Q6H" step "llm-2" (llm) is running under a worker lease until 1790135462840. +↻ llm-2 (llm) 26.16s +WAITING [worker_lease] Run "01M3665A2TP7DF8X6PBK5F0Q6H" step "llm-2" (llm) is running under a worker lease until 1790135472850. +↻ llm-2 (llm) 36.17s +✓ llm-2 (llm) 41.94s completionReason: success +○ llm-3 (llm) 0.00s +WAITING [worker_lease] Run "01M3666M055F7SYNPNJ5SJZ6N4" step "llm-3" (llm) is running under a worker lease until 1790135485751. +↻ llm-3 (llm) 7.09s +✓ llm-3 (llm) 14.62s completionReason: success +○ run-4 (deterministic) 0.00s +✓ run-4 (deterministic) 0.06s completionReason: success +○ llm-5 (llm) 0.00s +WAITING [worker_lease] Run "01M3667JQ5JEKNMD2F43ATVXSB" step "llm-5" (llm) is running under a worker lease until 1790135517208. +↻ llm-5 (llm) 23.87s +✗ llm-5 (llm) 23.87s +FAILED [protocol_error] relayflowd could not complete the run request: lease_conflict: attempt has no active worker lease +exit=1 diff --git a/examples/prompt-lab/evidence/runtime-findings/00-sonnet-run-pat-healed-high.txt b/examples/prompt-lab/evidence/runtime-findings/00-sonnet-run-pat-healed-high.txt new file mode 100644 index 00000000..c3936c7d --- /dev/null +++ b/examples/prompt-lab/evidence/runtime-findings/00-sonnet-run-pat-healed-high.txt @@ -0,0 +1,7 @@ +$ flows replay 01M365QAQKKYY3W6VQ2MAPVJK9 # engine step of Job 1 run 01M365N0J2DJ4EWD5GK57ZF80N (claude-sonnet-5, structured output; code before the reply-parsing change) +1 1790134954739 run.spawned {"created_by":"protocol-v0","journal_version":1,"parent_run_id":null,"spec":{"name":"prompt-lab/llm-11","steps":[{"cli":"claude","depends_on":[],"id":"llm-11","max_iterations":1,"model":"claude-sonnet-5","prompt":"You are the chart-filling engine for a home-health visit. Follow the INSTRUCTIONS exactly and nothing else.\n\nINSTRUCTIONS:\nDetermine the status of the patient's primary wound. Read the referral packet first; if the discharge summary states the wound is closed or healed, answer Healed. Otherwise use the visit notes. Use High confidence when the referral states the status.\n\nQUESTION: What is the current status of the patient's primary wound?\nOPTIONS (answer with exactly one): Healed | Ongoing | No wound\n\nREFERRAL PACKET:\nHospital discharge summary: admitted for cellulitis of the left lower leg. Left heel pressure injury noted on admission; wound closed per discharge summary. Discharged home with home health for strengthening and medication teaching. Social: lives alone per intake form.\n\nTODAY'S VISIT NOTES:\nSkilled nursing SOC visit. Left heel: open area 2.0 x 1.5 cm, moderate serous drainage, wound bed pink with yellow slough at edges. Dressing changed with foam per protocol. Patient lives alone in a one-story home; neighbor checks in daily. Patient calm and pleasant, joking about hospital food, says she is glad to be home.\n\nReturn JSON: answer (one of the options), confidence (High, Medium or Low), explanation (one or two sentences citing the chart).\n\nReply with the JSON object only: no prose before or after it, no markdown code fences.","retry":{"initial_backoff_ms":100,"jitter_percent":20,"max_backoff_ms":60000,"multiplier":2},"type":"llm","verification":{"json_schema":{"additionalProperties":false,"properties":{"answer":{"enum":["Healed","Ongoing","No wound"]},"confidence":{"enum":["High","Medium","Low"]},"explanation":{"minLength":1,"type":"string"}},"required":["answer","confidence","explanation"],"type":"object"}}}],"version":"0.1.0"},"spec_hash":"9482b209898b476742d5831cff93d0b28d52a5e26666ca4f4efe9eb7be1ec78d"} +2 1790134954741 step.routed step="llm-11" {"fallbacks_attempted":[],"profile":"attached-worker","provider":"worker"} +3 1790134954741 step.attempt.started step="llm-11" attempt=1 {"executor":"local-agent-950cf1c366a3eca0a64b60bff61dad5aa57a8413b8d1e46fff0e8177684c53fd-llm","idempotency_key":"ac4c7176c0e517dcdb298788db8c890fdf3f1ab756d11e7f7f4ad82812f2a2da","lease_deadline_ms":1790134984741,"lease_id":"01M365QAQNB0BPA0ECYMH677HR","max_iterations":1,"pins":{"streams":[],"workspace":[]},"recovery_mode":null,"step_type":"llm"} +4 1790134962149 step.completed step="llm-11" attempt=1 {"budget":{"dollars":"0","dollars_unmetered":true,"tokens_in":2,"tokens_out":479},"completed_by":"local-agent-950cf1c366a3eca0a64b60bff61dad5aa57a8413b8d1e46fff0e8177684c53fd-llm","completionReason":"success","disposition":"step_done","effects":[],"end_pins":null,"input_hash":"44136fa355b3678a1146ad16f7e8649e94fb4fc21fe77e8310c060f61caaff8a","next_attempt_at_ms":null,"output":{"answer":"Healed","confidence":"High","explanation":"The hospital discharge summary explicitly states the left heel pressure injury was closed at discharge, and per instructions the referral packet's discharge status takes precedence when it states the wound is closed or healed."},"spend":{"dollars":0,"dollars_unmetered":true,"tokens_input":2,"tokens_output":479,"wallclock_ms":7408},"step_spec_hash":"e8419f33e752090e0af7d88318b7b56a69f940c7affeb7696bbececc34537744","trajectory_tail":{"transcript":{"attempt":1,"exit_code":0,"final_text":"{\"answer\":\"Healed\",\"confidence\":\"High\",\"explanation\":\"The hospital discharge summary explicitly states the left heel pressure injury was closed at discharge, and per instructions the referral packet's discharge status takes precedence when it states the wound is closed or healed.\"}","result":{"claude_code_version":"2.1.280","duration_api_ms":4711,"duration_ms":4806,"is_error":false,"model":"claude-sonnet-5","num_turns":1,"permission_denials":0,"provider":"claude","session_id":"5a6ffc01-0c17-4126-9c25-29e26d6d7741","stop_reason":"end_turn","subtype":"success","total_cost_usd":0.0146904,"usage":{"cache_creation":0,"cache_read":49482,"input":2,"output":479,"thinking":390}}}},"verification":{"detail":"all gates passed","gate":"json_schema","verdict":"pass"}} +5 1790134962150 run.completed {"budget_total":{"dollars":"0","dollars_unmetered":true,"tokens_in":2,"tokens_out":479},"completionReason":"success","failed_step_id":null} +exit=0 diff --git a/examples/prompt-lab/evidence/runtime-findings/runtime-parallel-llm-repro.flow.ts b/examples/prompt-lab/evidence/runtime-findings/runtime-parallel-llm-repro.flow.ts new file mode 100644 index 00000000..7076e0c0 --- /dev/null +++ b/examples/prompt-lab/evidence/runtime-findings/runtime-parallel-llm-repro.flow.ts @@ -0,0 +1,8 @@ +import { flow } from "@relayflows/surface"; +const s = { type: "object", required: ["x"], properties: { x: { type: "number" } } }; +export default flow<{ n: number }>("repro-par", async (f, input) => { + await f.run("echo start"); + const xs = await Promise.all([1, 2, 3, 4, 5, 6, 7, 8, 9].map((i) => f.llm(`Return {"x": ${i}+${input.n}} as JSON only.`, { output: s, model: "claude-haiku-4-5-20251001" }))); + await f.run(`echo ${JSON.stringify(JSON.stringify(xs))}`); + f.done("success"); +}); diff --git a/examples/prompt-lab/evidence/runtime-findings/runtime-parallel-llm-repro.txt b/examples/prompt-lab/evidence/runtime-findings/runtime-parallel-llm-repro.txt new file mode 100644 index 00000000..d681ccdc --- /dev/null +++ b/examples/prompt-lab/evidence/runtime-findings/runtime-parallel-llm-repro.txt @@ -0,0 +1,13 @@ +$ flows run --local-agent evidence/runtime-parallel-llm-repro.flow.ts --input {"n":3} # relayflows 2.0.29, 9 concurrent f.llm +✓ run-1 (deterministic) 0.02s completionReason: success +✗ llm-2 (llm) 38.76s +✗ llm-3 (llm) 34.81s +✗ llm-4 (llm) 30.70s +✗ llm-5 (llm) 24.88s +✗ llm-10 (llm) 4.52s +✗ llm-6 (llm) 20.69s +✗ llm-7 (llm) 17.30s +✗ llm-8 (llm) 13.62s +✗ llm-9 (llm) 9.62s +FAILED [protocol_error] relayflowd could not complete the run request: Agent lease is already expired for 01M365AS4PJ4NRG4VFG684CEFV/llm-3. +(exit code not captured: the capture used bash PIPESTATUS under zsh) diff --git a/examples/prompt-lab/fixtures/agencies.json b/examples/prompt-lab/fixtures/agencies.json new file mode 100644 index 00000000..b3306493 --- /dev/null +++ b/examples/prompt-lab/fixtures/agencies.json @@ -0,0 +1,30 @@ +{ + "harbor": { + "name": "Harbor Home Health", + "visitTypes": { + "soc": [ + { "questionId": "wound-status", "options": ["Healed", "Ongoing", "No wound"] }, + { "questionId": "mood", "options": ["Calm", "Anxious", "Low"] } + ] + } + }, + "maple": { + "name": "Maple Visiting Nurses", + "visitTypes": { + "roc": [ + { "questionId": "wound-status", "options": ["Healed", "Ongoing", "No wound"] } + ] + } + }, + "sunrise": { + "name": "Sunrise Home Care", + "visitTypes": { + "soc": [ + { "questionId": "wound-status", "options": ["Healed", "Ongoing", "No wound"] }, + { "questionId": "mood", "options": ["Calm", "Anxious", "Low", "Agitated"] }, + { "questionId": "living-situation", "options": ["Lives alone", "Lives with spouse", "Lives with family", "Facility"] }, + { "questionId": "ostomy-supplies", "options": ["Pouches", "Barrier rings", "Skin prep wipes", "None"] } + ] + } + } +} diff --git a/examples/prompt-lab/fixtures/bank.json b/examples/prompt-lab/fixtures/bank.json new file mode 100644 index 00000000..e4fc48b1 --- /dev/null +++ b/examples/prompt-lab/fixtures/bank.json @@ -0,0 +1,12 @@ +{ + "prompts": { + "p-wound-status-v1": "Determine the status of the patient's primary wound. Read the referral packet first; if the discharge summary states the wound is closed or healed, answer Healed. Otherwise use the visit notes. Use High confidence when the referral states the status.", + "p-mood-v1": "Describe the patient's mood today from the visit notes. Choose the single option that best matches what the patient says and how the clinician describes them. Use High confidence when the patient states their feelings in their own words, Medium when inferred from behavior, Low when the notes are silent." + }, + "questions": { + "wound-status": { "text": "What is the current status of the patient's primary wound?", "type": "single", "livePromptId": "p-wound-status-v1" }, + "mood": { "text": "How would you describe the patient's mood today?", "type": "single", "livePromptId": "p-mood-v1" }, + "living-situation": { "text": "What is the patient's living situation?", "type": "single", "livePromptId": null }, + "ostomy-supplies": { "text": "Which ostomy supply does the patient need most this visit?", "type": "single", "livePromptId": null } + } +} diff --git a/examples/prompt-lab/fixtures/guidelines.md b/examples/prompt-lab/fixtures/guidelines.md new file mode 100644 index 00000000..c383992a --- /dev/null +++ b/examples/prompt-lab/fixtures/guidelines.md @@ -0,0 +1,9 @@ +# Shared prompt guidelines + +Every question prompt, whoever wrote it, must: + +1. Name the source of truth. When the visit notes (what the clinician saw and heard today) disagree with the referral packet or an intake form, today's visit notes win. +2. Tell the engine how to pick among the options it will be given; never invent an option. +3. Define the confidence levels: High only when today's notes state the answer directly; Medium when it is inferred; Low when the chart is silent or contradictory. +4. Ask for a one- or two-sentence explanation that cites where in the chart the answer came from. +5. Contain no patient identifiers and no examples copied from real charts. diff --git a/examples/prompt-lab/fixtures/shelf/jordan.json b/examples/prompt-lab/fixtures/shelf/jordan.json new file mode 100644 index 00000000..072aca46 --- /dev/null +++ b/examples/prompt-lab/fixtures/shelf/jordan.json @@ -0,0 +1,6 @@ +{ + "id": "jordan", "label": "Jordan", "locked": true, "source": "invented", + "ageBand": "65-74", "visitType": "soc", + "referral": "Referral: CHF exacerbation, new diuretic regimen, daily weights. Intake form (completed by hospital case manager): lives alone. No wounds documented. Skin intact.", + "notes": "Skilled nursing SOC visit. Spouse present for the whole visit; patient and spouse share the home and spouse sets up the pill organizer each morning. Skin intact, no wounds. Weight up 1 lb from discharge. Patient relaxed and engaged, asks good questions about the diuretic." +} diff --git a/examples/prompt-lab/fixtures/shelf/pat.json b/examples/prompt-lab/fixtures/shelf/pat.json new file mode 100644 index 00000000..9a7b34c2 --- /dev/null +++ b/examples/prompt-lab/fixtures/shelf/pat.json @@ -0,0 +1,6 @@ +{ + "id": "pat", "label": "Pat", "locked": true, "source": "invented", + "ageBand": "75-84", "visitType": "soc", + "referral": "Hospital discharge summary: admitted for cellulitis of the left lower leg. Left heel pressure injury noted on admission; wound closed per discharge summary. Discharged home with home health for strengthening and medication teaching. Social: lives alone per intake form.", + "notes": "Skilled nursing SOC visit. Left heel: open area 2.0 x 1.5 cm, moderate serous drainage, wound bed pink with yellow slough at edges. Dressing changed with foam per protocol. Patient lives alone in a one-story home; neighbor checks in daily. Patient calm and pleasant, joking about hospital food, says she is glad to be home." +} diff --git a/examples/prompt-lab/fixtures/shelf/riley.json b/examples/prompt-lab/fixtures/shelf/riley.json new file mode 100644 index 00000000..dc8555b7 --- /dev/null +++ b/examples/prompt-lab/fixtures/shelf/riley.json @@ -0,0 +1,6 @@ +{ + "id": "riley", "label": "Riley", "locked": true, "source": "invented", + "ageBand": "85+", "visitType": "soc", + "referral": "Referral: fall at home with right wrist fracture, cast in place. History of hypertension. No wounds. Lives with daughter's family.", + "notes": "Skilled nursing SOC visit. Patient lives with her daughter, son-in-law and two grandchildren. No wounds; skin intact around cast. Patient says 'I can't stop thinking about falling again, I lie awake worrying about it', wrings hands during the medication review." +} diff --git a/examples/prompt-lab/flows.json b/examples/prompt-lab/flows.json new file mode 100644 index 00000000..844b425f --- /dev/null +++ b/examples/prompt-lab/flows.json @@ -0,0 +1 @@ +{"cli":"claude"} diff --git a/examples/prompt-lab/jobs/fix.ts b/examples/prompt-lab/jobs/fix.ts new file mode 100644 index 00000000..b18a0bb9 --- /dev/null +++ b/examples/prompt-lab/jobs/fix.ts @@ -0,0 +1,104 @@ +// Job 2 — detect and fix one question (brief: "Question-level refinement"). +// +// You select a question (you opened it, or an issue) +// System shelf patients (test planner) + the issue's patients; gaps → queue +// System generate answers on the selected patients with the live prompt +// You edit answer / confidence / explanation → gold +// Outcome first pass persists as gold; config targets arrive prefilled +// System iterator rewrites prompt text → Prompt QA → re-run → % worked + per patient +// You review the new output → mark done +// Outcome live in Apricot immediately (a shared question: every agency) +import { changed, gridError, highlights, rowKey, type Row } from "../lib/grid.ts"; +import { hash8 } from "../lib/hash.ts"; +import { shellWord } from "../lib/lab.ts"; +import { agenciesUsing } from "../lib/piles.ts"; +import { score, scoreTable } from "../lib/score.ts"; +import type { Output, PatientBrief } from "../lib/types.ts"; +import { iterate, promptQa, type Ask } from "../prompts.ts"; +import type { Job } from "./job.ts"; +import { MAX_ATTEMPTS, planCoverage, runEngine, untilQaPasses } from "./shared.ts"; + +export interface FixInput { questionId?: string; issueId?: string; agency?: string } + +export async function fix(job: Job, input: FixInput): Promise { + const { f, lab, reviewer } = job; + const snap = await lab.snapshot(); + const issue = input.issueId ? snap.issues.find((i) => i.id === input.issueId && i.status === "open") : undefined; + if (input.issueId && !issue) throw new Error(`no open issue ${input.issueId} on the question manager`); + const qid = input.questionId ?? (issue?.questionIds.length === 1 ? issue.questionIds[0] : undefined); + if (!qid || (issue && !issue.questionIds.includes(qid))) throw new Error("name questionId: one of the issue's tagged questions"); + const question = snap.bank.questions[qid]; + if (!question?.livePromptId) throw new Error(`question ${qid} has no live prompt; stand it up with Job 1 first`); + const live = snap.bank.prompts[question.livePromptId]!; + + const using = agenciesUsing(snap.agencies, qid); + const agency = input.agency ?? issue?.agency ?? using[0]!; + const menu = Object.values(snap.agencies[agency]?.visitTypes ?? {}).flat().find((q) => q.questionId === qid)?.options; + if (!menu) throw new Error(`agency ${agency} does not ask ${qid}`); + const ask: Ask = { question: question.text, options: menu }; + const pile = using.length > 1 ? "shared" : "agency-specific"; + const brief = snap.briefs[qid]; + + // Recreate: shelf patients that can show the issue, plus any the issue's targets name. + const plan = await planCoverage(f, [{ id: qid, ...ask }], snap.shelf); + const ids = new Set([...(plan.coverage[0]?.patientIds ?? []), ...Object.keys(issue?.targets ?? {})]); + if (plan.gaps[0]) { + const gap: PatientBrief = { id: `gap-${qid}`, questionId: qid, brief: plan.gaps[0].brief, from: "planner", status: "queued" }; + await lab.enqueue("patient-briefs", gap); + } + if (ids.size === 0) return f.done("needs_human"); // nothing on the shelf can show it yet; the gap brief is queued + const patients = snap.shelf.filter((p) => ids.has(p.id)); + + const current = await runEngine(f, live, ask, patients); + const rows: Row[] = patients.map((p) => { + const ai = current[p.id]!; + const target = snap.gold[`${qid}|${p.id}`] ?? issue?.targets?.[p.id] ?? ai; + const why = highlights(ai, pile, false); + if (target.answer !== ai.answer) why.push(snap.gold[`${qid}|${p.id}`] ? "disagrees with gold" : "disagrees with the config-level target"); + return { questionId: qid, patientId: p.id, pile, highlight: why, ai, target, notes: "" }; + }); + const work = `work/q-${qid}/${hash8(rows)}`; + await lab.writeNew(`${work}/grid.json`, rows); + + const golded = await f.human( + `Question workbench · ${qid}${issue ? ` (issue ${issue.id})` : ""}: current-prompt outputs on ${rows.length} shelf patient(s) — ${rows.map((r) => `${r.patientId}: ${r.ai.answer}/${r.ai.confidence}`).join(", ")}.\n` + + `Set gold in ${job.labDir}/${work}/grid.json ("target" per row; prefilled from persisted gold or the config-level target). Your review persists as gold; later AI runs never overwrite it.\n` + + `yes = persist gold and run the iterator; no = stop.`, + { to: reviewer }); + if (!golded) return f.done("declined"); + + const expected = rows.map(rowKey).sort().join(","); + const edited = await lab.read(`${work}/grid.json`, (g) => + gridError(g, { [qid]: menu }) ?? (g.map(rowKey).sort().join(",") === expected ? null : "rows were added or removed; edit values only")); + const gold: Record = Object.fromEntries(edited.map((r) => [rowKey(r), r.target])); + await lab.record("gold", gold); + + const changes = changed(edited); + if (changes.length === 0) return f.done("declined"); // the live prompt already matches gold: nothing to iterate + const byId = new Map(patients.map((p) => [p.id, p])); + const changeset = changes.map((r) => ({ patient: byId.get(r.patientId)!, ai: r.ai, target: r.target, notes: r.notes })); + const rewrite = await untilQaPasses<{ prompt: string }>(f, + (findings) => iterate(ask, live, brief, changeset, findings), + (w) => promptQa(w.prompt, ask, snap.guidelines, brief)); + if (!rewrite.passed) { + await f.run(`echo ${shellWord(`iterated prompt for ${qid} failed Prompt QA ${MAX_ATTEMPTS} times: ${rewrite.findings.join("; ")}`)} >&2`); + return f.done("needs_human"); + } + const { promptId } = await lab.draft(qid, rewrite.value.prompt); + + // Re-run and score: required before done. Worked = new answer matches persisted gold. + const rerun = await runEngine(f, rewrite.value.prompt, ask, patients); + const result = score(qid, rerun, { ...snap.gold, ...gold }); + await lab.writeNew(`${work}/score.json`, { promptId, ...result, outputs: rerun }); + + const shared = using.length > 1 ? `\nSHARED: marking done changes this prompt for every agency that uses it: ${using.join(", ")}.` : ""; + const done = await f.human( + `Re-run and score · ${qid}, new prompt ${promptId}:\n${scoreTable(result)}${shared}\n` + + `Read the new prompt in ${job.labDir}/bank.json and the outputs in ${job.labDir}/${work}/score.json.\n` + + `yes = mark done (live in Apricot now, no promote step); no = keep it as a draft and iterate again in a new run.`, + { to: reviewer }); + if (!done) return f.done("declined"); + await lab.publish(qid, promptId); + if (issue) await lab.closeIssue(issue.id); + return f.done("success"); +} diff --git a/examples/prompt-lab/jobs/job.ts b/examples/prompt-lab/jobs/job.ts new file mode 100644 index 00000000..85202faf --- /dev/null +++ b/examples/prompt-lab/jobs/job.ts @@ -0,0 +1,5 @@ +import type { Ctx } from "@relayflows/surface"; +import type { Lab } from "../lib/lab.ts"; + +/** What every job gets: the flow context, the lab store, and who answers its gates. */ +export interface Job { f: Ctx; lab: Lab; labDir: string; reviewer: string } diff --git a/examples/prompt-lab/jobs/new-agency.ts b/examples/prompt-lab/jobs/new-agency.ts new file mode 100644 index 00000000..c05117b6 --- /dev/null +++ b/examples/prompt-lab/jobs/new-agency.ts @@ -0,0 +1,144 @@ +// Job 1 — new agency (brief: "Config-level / new agency"). +// +// System lookup piles: agency-specific, shared, mismatch +// System first-pass prompt text for agency-specific questions with none → Prompt QA +// System test planner picks path-covering shelf patients; gaps → patient-manager queue +// System run on fake visits → edit grid, computer-highlighted rows first +// You edit each row on the first QA pass +// Outcome targets persist for question × patient +// System run iteration on every changed question +// System shared rows → question-level queue with their targets (frozen here) +// You commit all / only / all except +// Outcome committed agency-specific prompts are live in Apricot +import { toCommit, commitError, type CommitChoice } from "../lib/commit.ts"; +import { changed, gridError, highlights, rowKey, sortRows, type Row } from "../lib/grid.ts"; +import { hash8 } from "../lib/hash.ts"; +import { piles, type PiledQuestion } from "../lib/piles.ts"; +import type { Issue, Output, Patient, PatientBrief } from "../lib/types.ts"; +import { firstPass, iterate, promptQa, type Ask } from "../prompts.ts"; +import type { Job } from "./job.ts"; +import { shellWord } from "../lib/lab.ts"; +import { llm, MAX_ATTEMPTS, planCoverage, runEngine, untilQaPasses } from "./shared.ts"; + +export interface NewAgencyInput { agency: string; visitType: string } + +export async function newAgency(job: Job, input: NewAgencyInput): Promise { + const { f, lab, reviewer } = job; + const snap = await lab.snapshot(); + const piled = piles(snap.agencies, input.agency, input.visitType); + const ask = (q: PiledQuestion): Ask => ({ question: snap.bank.questions[q.questionId]!.text, options: q.options }); + const byId = new Map(piled.map((q) => [q.questionId, q])); + const patient = new Map(snap.shelf.map((p) => [p.id, p])); + + // Prompt text per question: Apricot's live prompt, or a first-pass v0 for an + // agency-specific question that has none. Shared prompts are never written here. + const prompts = new Map(); + const drafts = new Map(); // questionId → draft promptId in the Bank + for (const q of piled) { + const live = snap.bank.questions[q.questionId]?.livePromptId; + if (live) { prompts.set(q.questionId, { text: snap.bank.prompts[live]!, firstPass: false }); continue; } + if (q.pile !== "agency-specific") throw new Error(`shared question ${q.questionId} has no live prompt; fix it at question-level first`); + const brief = snap.briefs[q.questionId]; + const v0 = await untilQaPasses<{ prompt: string }>(f, + (findings) => firstPass(ask(q), snap.guidelines, brief, findings), + (w) => promptQa(w.prompt, ask(q), snap.guidelines, brief)); + if (!v0.passed) { + await f.run(`echo ${shellWord(`first-pass prompt for ${q.questionId} failed Prompt QA ${MAX_ATTEMPTS} times: ${v0.findings.join("; ")}`)} >&2`); + return f.done("needs_human"); + } + drafts.set(q.questionId, (await lab.draft(q.questionId, v0.value.prompt)).promptId); + prompts.set(q.questionId, { text: v0.value.prompt, firstPass: true }); + } + + // Test planner: existing shelf only; holes become briefs on the manager queue. + const plan = await planCoverage(f, piled.map((q) => ({ id: q.questionId, ...ask(q) })), snap.shelf); + for (const gap of plan.gaps) { + const brief: PatientBrief = { id: `gap-${gap.questionId}`, questionId: gap.questionId, brief: gap.brief, from: "planner", status: "queued" }; + await lab.enqueue("patient-briefs", brief); + } + + // Run on fake visits — the background job; the grid fills when it finishes. + const ran: { q: PiledQuestion; outputs: Record }[] = []; + for (const c of plan.coverage) { // sequential: see runEngine + const q = byId.get(c.questionId)!; + const shelf = c.patientIds.map((id) => patient.get(id)!); + ran.push({ q, outputs: await runEngine(f, prompts.get(q.questionId)!.text, ask(q), shelf) }); + } + const rows = sortRows(ran.flatMap(({ q, outputs }) => Object.entries(outputs).map(([patientId, ai]): Row => ({ + questionId: q.questionId, patientId, pile: q.pile, + highlight: highlights(ai, q.pile, prompts.get(q.questionId)!.firstPass), + ai, target: snap.targets[`${input.agency}|${q.questionId}|${patientId}`] ?? ai, notes: "", + })))); + const work = `work/${input.agency}-${input.visitType}/${hash8(rows)}`; + await lab.writeNew(`${work}/grid.json`, rows); + + const flagged = rows.filter((r) => r.highlight.length).length; + const reviewed = await f.human( + `Config workbench · ${input.agency} ${input.visitType}: ${rows.length} rows on ${plan.coverage.length} questions (${flagged} highlighted, listed first); ${plan.gaps.length} gap brief(s) queued for the test patient manager.\n` + + `Edit target answer / confidence / explanation / notes in ${job.labDir}/${work}/grid.json. Your first pass persists as the target for each question × patient.\n` + + `yes = persist targets and run iteration on every changed row; no = stop without persisting.`, + { to: reviewer }); + if (!reviewed) return f.done("declined"); + + const menus = Object.fromEntries(piled.map((q) => [q.questionId, q.options])); + const expected = rows.map(rowKey).sort().join(","); + const edited = await lab.read(`${work}/grid.json`, (g) => + gridError(g, menus) ?? (g.map(rowKey).sort().join(",") === expected ? null : "rows were added or removed; edit values only")); + await lab.record("targets", Object.fromEntries(edited.map((r) => [`${input.agency}|${rowKey(r)}`, r.target]))); + + // Run iteration on every question with a changed row. Agency-specific rewrites + // are commit candidates (Prompt QA'd); shared/mismatch get a proposed rewrite only. + const changes = changed(edited); + const changedQs = [...new Set(changes.map((r) => r.questionId))].sort(); + const rewrites: { q: PiledQuestion; rows: Row[]; prompt: string; passed: boolean }[] = []; + for (const qid of changedQs) { // sequential: see runEngine + const q = byId.get(qid)!; + const rowsFor = changes.filter((r) => r.questionId === qid) + .map((r) => ({ patient: patient.get(r.patientId)! as Patient, ai: r.ai, target: r.target, notes: r.notes })); + const brief = snap.briefs[qid]; + const existing = prompts.get(qid)!.text; + const rows = changes.filter((r) => r.questionId === qid); + if (q.pile !== "agency-specific") { + const { prompt } = await llm<{ prompt: string }>(f, iterate(ask(q), existing, brief, rowsFor, [])); + rewrites.push({ q, rows, prompt, passed: true }); + continue; + } + const r = await untilQaPasses<{ prompt: string }>(f, + (findings) => iterate(ask(q), existing, brief, rowsFor, findings), + (w) => promptQa(w.prompt, ask(q), snap.guidelines, brief)); + rewrites.push({ q, rows, prompt: r.value.prompt, passed: r.passed }); + } + + const candidates = new Set(piled.filter((q) => q.pile === "agency-specific" && prompts.get(q.questionId)!.firstPass).map((q) => q.questionId)); + const sent: string[] = []; + for (const r of rewrites) { + if (r.q.pile === "agency-specific") { + if (!r.passed) { candidates.delete(r.q.questionId); continue; } // an uncompliant rewrite never becomes a candidate + drafts.set(r.q.questionId, (await lab.draft(r.q.questionId, r.prompt)).promptId); + candidates.add(r.q.questionId); + continue; + } + // Frozen here: the rows travel to question-level with their targets and the proposal. + const issue: Issue = { + id: `config-${input.agency}-${r.q.questionId}-${hash8(r.rows)}`, kind: "config-send", questionIds: [r.q.questionId], status: "open", agency: input.agency, + text: `${input.agency} ${input.visitType} first pass changed ${r.rows.length} row(s) on a ${r.q.pile} question (shared with ${r.q.sharedWith.join(", ")}). ${r.rows.map((x) => x.notes).filter(Boolean).join(" ")}`.trim(), + targets: Object.fromEntries(r.rows.map((x) => [x.patientId, x.target])), proposedPrompt: r.prompt, + }; + await lab.enqueue("issues", issue); + sent.push(`${r.q.questionId} → ${issue.id}`); + } + + const list = [...candidates].sort(); + if (list.length === 0) return f.done("success"); // nothing agency-specific to commit; shared work continues at question-level + await lab.writeNew(`${work}/commit.json`, { mode: "all", questions: [] } satisfies CommitChoice); + const commit = await f.human( + `Commit output · ${input.agency}: agency-specific prompts ready to go live in Apricot: ${list.join(", ")}.\n` + + `Sent to the question manager (frozen here): ${sent.length ? sent.join("; ") : "none"}.\n` + + `To commit only some, edit ${job.labDir}/${work}/commit.json: {"mode":"only"|"except","questions":[...]}.\n` + + `yes = commit (Done = live, no promote step); no = leave them as drafts.`, + { to: reviewer }); + if (!commit) return f.done("declined"); + const choice = await lab.read(`${work}/commit.json`, commitError); + for (const qid of toCommit(list, choice)) await lab.publish(qid, drafts.get(qid)!); + return f.done("success"); +} diff --git a/examples/prompt-lab/jobs/patient.ts b/examples/prompt-lab/jobs/patient.ts new file mode 100644 index 00000000..6424977f --- /dev/null +++ b/examples/prompt-lab/jobs/patient.ts @@ -0,0 +1,58 @@ +// Set up a test patient (brief: "Test patient creator"). +// +// System a gap brief from the test planner (or one you pass by hand) +// System agent writes the patient plan +// You may iterate the plan → kick generate (your last required action) +// System chart details → Patient QA loops until pass +// Outcome lock onto the shelf → back to the manager +// +// You never approve the invented chart: Patient QA locks it. +import { hash8 } from "../lib/hash.ts"; +import { shellWord } from "../lib/lab.ts"; +import { identifiers } from "../lib/phi.ts"; +import type { Patient } from "../lib/types.ts"; +import { patientChart, patientPlan, patientQa } from "../prompts.ts"; +import type { Job } from "./job.ts"; +import { llm, MAX_ATTEMPTS, untilQaPasses } from "./shared.ts"; + +export interface PatientInput { briefId?: string; brief?: string; questionId?: string } +type Chart = Omit; + +export async function createPatient(job: Job, input: PatientInput): Promise { + const { f, lab, reviewer } = job; + const snap = await lab.snapshot(); + const queued = input.briefId ? snap.patientBriefs.find((b) => b.id === input.briefId && b.status === "queued") : undefined; + if (input.briefId && !queued) throw new Error(`no queued patient brief ${input.briefId}`); + const brief = queued?.brief ?? input.brief; + const questionId = queued?.questionId ?? input.questionId; + if (!brief || !questionId || !snap.bank.questions[questionId]) throw new Error("pass briefId, or brief + questionId"); + + const { plan } = await llm<{ plan: string }>(f, patientPlan(brief, snap.bank.questions[questionId].text)); + const work = `work/patients/${queued?.id ?? `manual-${hash8(brief)}`}`; + await lab.writeNew(`${work}/plan.json`, { brief, plan }); + + const kick = await f.human( + `Test patient creator · ${queued?.id ?? "manual brief"} for ${questionId}.\nPlan:\n${plan}\n` + + `You may edit "plan" in ${job.labDir}/${work}/plan.json first.\n` + + `yes = kick generate (Patient QA then locks it onto the shelf; you do not approve the chart); no = leave it on the queue.`, + { to: reviewer }); + if (!kick) return f.done("declined"); + const final = await lab.read<{ brief: string; plan: string }>(`${work}/plan.json`, (v) => + typeof v?.plan === "string" && v.plan.trim().length > 0 ? null : "plan must be non-empty text"); + + const taken = new Set(snap.shelf.map((p) => p.id)); + const made = await untilQaPasses(f, + (findings) => patientChart(brief, final.plan, findings), + (chart) => { + // Deterministic floor first: identifiers and id collisions never reach model QA. + const hard = identifiers(`${chart.label} ${chart.referral} ${chart.notes}`); + if (taken.has(chart.id)) hard.push(`id ${chart.id} is already on the shelf; pick another first-name label`); + return hard.length ? hard : patientQa(brief, final.plan, chart); + }); + if (!made.passed) { + await f.run(`echo ${shellWord(`Patient QA did not pass after ${MAX_ATTEMPTS} charts: ${made.findings.join("; ")}`)} >&2`); + return f.done("needs_human"); + } + await lab.lockPatient(made.value, queued?.id); + return f.done("success"); +} diff --git a/examples/prompt-lab/jobs/shared.ts b/examples/prompt-lab/jobs/shared.ts new file mode 100644 index 00000000..fcf03c6d --- /dev/null +++ b/examples/prompt-lab/jobs/shared.ts @@ -0,0 +1,98 @@ +// Steps more than one job uses: the typed model call, the write → QA loop, +// the Apricot engine run, and the gated test planner. +import type { Ctx } from "@relayflows/surface"; +import { engine, planTests, type Ask, type Call } from "../prompts.ts"; +import { failStep } from "../lib/lab.ts"; +import { parseReply, schemaError } from "../lib/reply.ts"; +import type { Output, Patient } from "../lib/types.ts"; + +export const MAX_ATTEMPTS = 3; + +const JSON_ONLY = "\n\nReply with the JSON object only: no prose before or after it, no markdown code fences."; +export const REPLY_ATTEMPTS = 2; + +/** + * One structured model call. The reply is journaled as text and checked here + * against the call's JSON Schema; an invalid reply is re-asked once, naming + * the error, then fails the run with it. + * + * Not f.llm's `output` option: relayflows 2.0.29 verifies the raw reply, so a + * model that wraps valid JSON in a markdown fence fails the step, and a failed + * run cannot be resumed (evidence/runtime-findings/00-job1-attempt1-fenced-json.txt, evidence/runtime-findings/00-patient-attempt1-fenced-json.txt). + * The text form takes no model option, so calls use the CLI adapter's default. + */ +export async function llm(f: Ctx, call: Call): Promise { + let problem = ""; + for (let attempt = 1; attempt <= REPLY_ATTEMPTS; attempt++) { + const retry = problem ? `\n\nYour previous reply was rejected: ${problem}. Reply again.` : ""; + const text = await f.llm`${call.prompt}${JSON_ONLY}\nIt must match this JSON Schema:\n${JSON.stringify(call.output)}${retry}`; + try { + const value = parseReply(text); + problem = schemaError(call.output, value) ?? ""; + if (!problem) return value as T; + } catch (error) { + problem = `not JSON (${error instanceof Error ? error.message : String(error)})`; + } + } + return failStep(f, `model reply refused after ${REPLY_ATTEMPTS} attempts: ${problem}`); +} + +interface Qa { pass: boolean; findings: string[] } + +/** + * Write, then QA, looping with QA's findings until a pass. Bounded: after + * MAX_ATTEMPTS the caller parks the run for a human instead of shipping an + * uncompliant prompt or chart. + */ +export async function untilQaPasses( + f: Ctx, + write: (findings: readonly string[]) => Call, + qa: (written: T) => Call | string[], +): Promise<{ value: T; passed: boolean; findings: string[] }> { + let findings: string[] = []; + let value!: T; + for (let attempt = 1; attempt <= MAX_ATTEMPTS; attempt++) { + value = await llm(f, write(findings)); + const check = qa(value); + const verdict: Qa = Array.isArray(check) ? { pass: check.length === 0, findings: check } : await llm(f, check); + // A pass with findings is not a pass. + findings = verdict.pass && verdict.findings.length === 0 ? [] : verdict.findings.length ? verdict.findings : ["QA failed without findings"]; + if (findings.length === 0) return { value, passed: true, findings }; + } + return { value, passed: false, findings }; +} + +/** + * Run one prompt on fake visits: the same engine nurses use. One llm step per patient. + * + * Sequential on purpose. Every model call in this flow is independent and was + * written as Promise.all, but relayflows 2.0.29 loses the run when concurrent + * f.llm calls queue past their 30s lease (a stale completion becomes a fatal + * protocol_error): see evidence/runtime-findings/runtime-parallel-llm-repro.txt. Restore + * Promise.all here and in jobs/new-agency.ts when that is fixed. + */ +export async function runEngine(f: Ctx, promptText: string, ask: Ask, patients: readonly Patient[]): Promise> { + const outputs: Record = {}; + for (const p of patients) outputs[p.id] = await llm(f, engine(promptText, ask, p)); + return outputs; +} + +export interface Plan { coverage: { questionId: string; patientIds: string[] }[]; gaps: { questionId: string; brief: string }[] } + +/** Test QA, deterministic: every question exactly once, covered by real shelf patients or a gap brief. */ +export function planError(plan: Plan, questionIds: readonly string[], shelf: readonly Patient[]): string | null { + const onShelf = new Set(shelf.map((p) => p.id)); + const seen = [...plan.coverage.map((c) => c.questionId), ...plan.gaps.map((g) => g.questionId)]; + for (const q of questionIds) if (seen.filter((s) => s === q).length !== 1) return `question ${q} must appear exactly once`; + if (seen.length !== questionIds.length) return "plan names a question that was not asked"; + for (const c of plan.coverage) for (const p of c.patientIds) if (!onShelf.has(p)) return `${p} is not on the shelf`; + return null; +} + +/** The test planner picks from the existing shelf; it never invents a patient per question. */ +export async function planCoverage(f: Ctx, asks: readonly (Ask & { id: string })[], shelf: readonly Patient[]): Promise { + const plan = await llm(f, planTests(asks, shelf)); + const why = planError(plan, asks.map((a) => a.id), shelf); + if (why) await failStep(f, `test plan refused: ${why}`); + return plan; +} diff --git a/examples/prompt-lab/lib/commit.ts b/examples/prompt-lab/lib/commit.ts new file mode 100644 index 00000000..561c4c74 --- /dev/null +++ b/examples/prompt-lab/lib/commit.ts @@ -0,0 +1,17 @@ +// Config-level Commit output: all / only (selected) / all except (excluded), +// over agency-specific iterated prompts only. Shared never commits here. + +export interface CommitChoice { mode: "all" | "only" | "except"; questions: string[] } + +export function commitError(choice: unknown): string | null { + const c = choice as CommitChoice; + if (!c || !["all", "only", "except"].includes(c.mode)) return 'commit.json: mode must be "all", "only" or "except"'; + if (!Array.isArray(c.questions) || !c.questions.every((q) => typeof q === "string")) return "commit.json: questions must be a list of question ids"; + return null; +} + +export function toCommit(candidates: readonly string[], choice: CommitChoice): string[] { + if (choice.mode === "all") return [...candidates]; + if (choice.mode === "only") return candidates.filter((q) => choice.questions.includes(q)); + return candidates.filter((q) => !choice.questions.includes(q)); +} diff --git a/examples/prompt-lab/lib/grid.ts b/examples/prompt-lab/lib/grid.ts new file mode 100644 index 00000000..15ef2716 --- /dev/null +++ b/examples/prompt-lab/lib/grid.ts @@ -0,0 +1,50 @@ +// The Output QA edit grid: what the reviewer edits between two human gates. +// Pure functions over journaled values; the flow does the reading and writing. +import { CONFIDENCES, type Output, type Pile } from "./types.ts"; + +export interface Row { + questionId: string; patientId: string; pile: Pile; + /** Why the computer put this row first. Empty = not highlighted. */ + highlight: string[]; + ai: Output; + /** The reviewer's value. Prefilled from persisted targets/gold, else the AI output. */ + target: Output; + notes: string; +} + +export const rowKey = (r: { questionId: string; patientId: string }): string => `${r.questionId}|${r.patientId}`; + +const sameOutput = (a: Output, b: Output): boolean => + a.answer === b.answer && a.confidence === b.confidence && a.explanation === b.explanation; + +/** Deterministic "likely off" signals. The reviewer's time goes to these rows first. */ +export function highlights(ai: Output, pile: Pile, newPrompt: boolean): string[] { + const why: string[] = []; + if (ai.confidence !== "High") why.push(`${ai.confidence.toLowerCase()} confidence`); + if (pile === "mismatch") why.push("shared prompt, this agency's menu or ancestry differs"); + if (newPrompt) why.push("first-pass prompt, never reviewed"); + return why; +} + +export function sortRows(rows: Row[]): Row[] { + return [...rows].sort((a, b) => + Number(b.highlight.length > 0) - Number(a.highlight.length > 0) || rowKey(a).localeCompare(rowKey(b))); +} + +/** Run iteration uses every row that changed — never the current selection. */ +export const changed = (rows: readonly Row[]): Row[] => + rows.filter((r) => !sameOutput(r.ai, r.target) || r.notes.trim() !== ""); + +/** Refuses a grid the reviewer broke, rather than persisting a half-valid target. */ +export function gridError(rows: unknown, options: Record): string | null { + if (!Array.isArray(rows)) return "grid must be a JSON array of rows"; + for (const r of rows as Row[]) { + const where = `${r?.questionId}|${r?.patientId}`; + const menu = options[r?.questionId]; + if (!menu) return `${where}: unknown question`; + if (!r.target || !menu.includes(r.target.answer)) return `${where}: target answer is not on the agency menu (${menu.join(" / ")})`; + if (!CONFIDENCES.includes(r.target.confidence)) return `${where}: confidence must be High, Medium or Low`; + if (typeof r.target.explanation !== "string" || typeof r.notes !== "string") return `${where}: explanation and notes must be text`; + } + return null; +} diff --git a/examples/prompt-lab/lib/hash.ts b/examples/prompt-lab/lib/hash.ts new file mode 100644 index 00000000..1b75b757 --- /dev/null +++ b/examples/prompt-lab/lib/hash.ts @@ -0,0 +1,4 @@ +import { createHash } from "node:crypto"; + +/** A stable short id for a value: the same journaled inputs name the same work file. */ +export const hash8 = (value: unknown): string => createHash("sha256").update(JSON.stringify(value)).digest("hex").slice(0, 8); diff --git a/examples/prompt-lab/lib/lab.ts b/examples/prompt-lab/lib/lab.ts new file mode 100644 index 00000000..7297ce68 --- /dev/null +++ b/examples/prompt-lab/lib/lab.ts @@ -0,0 +1,58 @@ +// The flow's side of the lab store: every read and write is one journaled +// `f.run` step. Author code never touches the filesystem directly. +import type { Ctx } from "@relayflows/surface"; +import type { Snapshot } from "./types.ts"; + +const STORE = new URL("../store.ts", import.meta.url).pathname; + +/** User text never reaches a shell unquoted. */ +export const shellWord = (value: string): string => `'${value.replaceAll("'", "'\\''")}'`; +const b64 = (value: unknown): string => Buffer.from(JSON.stringify(value), "utf8").toString("base64"); + +/** + * Fail the run with a journaled reason: a deterministic step that exits 1. + * + * Used instead of a predicate `.gate()` because relayflows 2.0.29 cannot + * resume past one: the recorded verdict comes back from the journal with its + * keys re-ordered, the lowered `.gate` spec no longer matches, and the + * resume is refused as run_admission_conflict + * (evidence/runtime-findings/00-job1-attempt4-predicate-gate-resume-conflict.txt). + */ +export async function failStep(f: Ctx, reason: string): Promise { + await f.run(`echo ${shellWord(reason)} >&2; exit 1`); + throw new Error(reason); // unreachable: the step above fails the run +} + +export interface Lab { + snapshot(): Promise; + read(rel: string, check: (value: T) => string | null): Promise; + writeNew(rel: string, value: unknown): Promise<{ written: boolean }>; + record(store: "targets" | "gold", entries: Record): Promise; + draft(questionId: string, prompt: string): Promise<{ promptId: string }>; + publish(questionId: string, promptId: string): Promise<{ previous: string | null }>; + enqueue(queue: "issues" | "patient-briefs", item: { id: string }): Promise<{ added: boolean }>; + closeIssue(id: string): Promise; + lockPatient(patient: unknown, briefId?: string): Promise<{ shelfPath: string }>; +} + +export function lab(f: Ctx, dir: string): Lab { + const store = async (...args: string[]): Promise => + JSON.parse(await f.run(`node --no-warnings --experimental-strip-types ${shellWord(STORE)} ${[dir, ...args].map(shellWord).join(" ")}`)); + return { + snapshot: () => store("snapshot"), + async read(rel: string, check: (value: T) => string | null) { + const value = JSON.parse(await f.run(`node --no-warnings --experimental-strip-types ${shellWord(STORE)} ${shellWord(dir)} read ${shellWord(rel)}`)) as T; + const why = check(value); + // A reviewer's broken edit fails the run instead of persisting a half-valid value. + if (why) await failStep(f, `${rel}: ${why}`); + return value; + }, + writeNew: (rel, value) => store("write-new", rel, b64(value)), + record: (s, entries) => store("record", s, b64(entries)), + draft: (q, prompt) => store("draft", q, b64(prompt)), + publish: (q, id) => store("publish", q, id), + enqueue: (queue, item) => store("enqueue", queue, b64(item)), + closeIssue: (id) => store("close-issue", id), + lockPatient: (patient, briefId) => store("lock-patient", b64(patient), ...(briefId ? [briefId] : [])), + }; +} diff --git a/examples/prompt-lab/lib/phi.ts b/examples/prompt-lab/lib/phi.ts new file mode 100644 index 00000000..a9810f01 --- /dev/null +++ b/examples/prompt-lab/lib/phi.ts @@ -0,0 +1,13 @@ +// A deterministic floor under Patient QA: things an invented chart may never +// carry, whatever a model thinks of it. Not a de-identification tool. +const RULES: [RegExp, string][] = [ + [/\b\d{1,2}[/-]\d{1,2}[/-]\d{2,4}\b/, "a calendar date"], + [/\b(19|20)\d{2}-\d{2}-\d{2}\b/, "an ISO date"], + [/\(?\b\d{3}\)?[\s.-]\d{3}[\s.-]\d{4}\b/, "a phone number"], + [/\bMRN\b|\b\d{7,}\b/i, "a record number"], + [/\b\d+\s+[A-Z][a-z]+\s+(Street|St|Avenue|Ave|Road|Rd|Lane|Ln|Drive|Dr)\b/, "a street address"], +]; + +export function identifiers(text: string): string[] { + return RULES.filter(([re]) => re.test(text)).map(([, what]) => `contains ${what}`); +} diff --git a/examples/prompt-lab/lib/piles.ts b/examples/prompt-lab/lib/piles.ts new file mode 100644 index 00000000..94258b16 --- /dev/null +++ b/examples/prompt-lab/lib/piles.ts @@ -0,0 +1,39 @@ +// The three sharing piles. A deterministic Bank/config lookup, never an agent, +// with no override (brief: "Sharing piles"). +import type { Agency, AgencyQuestion, Pile } from "./types.ts"; + +export interface PiledQuestion extends AgencyQuestion { pile: Pile; sharedWith: string[] } + +const sameList = (a: readonly string[], b: readonly string[]): boolean => + a.length === b.length && a.every((v, i) => v === b[i]); + +/** Every other agency's asking of `questionId`, across all its visit types. */ +function othersAsking(agencies: Record, self: string, questionId: string) { + const found: { agency: string; q: AgencyQuestion }[] = []; + for (const [id, agency] of Object.entries(agencies)) { + if (id === self) continue; + for (const qs of Object.values(agency.visitTypes)) { + for (const q of qs) if (q.questionId === questionId) found.push({ agency: id, q }); + } + } + return found; +} + +export function piles(agencies: Record, agencyId: string, visitType: string): PiledQuestion[] { + const questions = agencies[agencyId]?.visitTypes[visitType]; + if (!questions) throw new Error(`agency ${agencyId} has no visit type ${visitType}`); + return questions.map((q) => { + const others = othersAsking(agencies, agencyId, q.questionId); + const sharedWith = [...new Set(others.map((o) => o.agency))].sort(); + if (others.length === 0) return { ...q, pile: "agency-specific", sharedWith }; + const identical = others.every((o) => sameList(o.q.options, q.options) && o.q.parent === q.parent); + return { ...q, pile: identical ? "shared" : "mismatch", sharedWith }; + }); +} + +/** Every agency whose config asks `questionId` — the blast radius of a question-level done. */ +export function agenciesUsing(agencies: Record, questionId: string): string[] { + return Object.entries(agencies) + .filter(([, a]) => Object.values(a.visitTypes).some((qs) => qs.some((q) => q.questionId === questionId))) + .map(([id]) => id).sort(); +} diff --git a/examples/prompt-lab/lib/reply.ts b/examples/prompt-lab/lib/reply.ts new file mode 100644 index 00000000..c44dcf60 --- /dev/null +++ b/examples/prompt-lab/lib/reply.ts @@ -0,0 +1,54 @@ +// Model replies, parsed and checked in author code. The schemas in prompts.ts +// use a small JSON Schema subset; this validates exactly that subset and +// refuses any keyword it does not know, so a schema can never silently weaken. + +type Schema = Record; +const KNOWN = new Set(["type", "required", "properties", "additionalProperties", "items", "minItems", "minLength", "pattern", "enum", "const"]); + +/** The reply text as JSON, tolerating one surrounding markdown fence. Throws on anything else. */ +export function parseReply(text: string): unknown { + const fenced = /^```[a-zA-Z]*\s*\n([\s\S]*?)\n?```\s*$/.exec(text.trim()); + return JSON.parse(fenced ? fenced[1]! : text.trim()); +} + +/** The first way `value` violates `schema`, or null. */ +export function schemaError(schema: Schema, value: unknown, at = "$"): string | null { + for (const key of Object.keys(schema)) if (!KNOWN.has(key)) throw new Error(`reply schema keyword "${key}" is not supported`); + if ("const" in schema && value !== schema.const) return `${at} must be ${JSON.stringify(schema.const)}`; + if (Array.isArray(schema.enum) && !schema.enum.includes(value)) return `${at} must be one of ${schema.enum.map((v) => JSON.stringify(v)).join(", ")}`; + switch (schema.type) { + case "string": + if (typeof value !== "string") return `${at} must be a string`; + if (typeof schema.minLength === "number" && value.length < schema.minLength) return `${at} must be at least ${schema.minLength} characters`; + if (typeof schema.pattern === "string" && !new RegExp(schema.pattern).test(value)) return `${at} must match ${schema.pattern}`; + return null; + case "number": + return typeof value === "number" ? null : `${at} must be a number`; + case "boolean": + return typeof value === "boolean" ? null : `${at} must be true or false`; + case "array": { + if (!Array.isArray(value)) return `${at} must be an array`; + if (typeof schema.minItems === "number" && value.length < schema.minItems) return `${at} needs at least ${schema.minItems} item(s)`; + for (let i = 0; i < value.length; i++) { + const e = schema.items ? schemaError(schema.items as Schema, value[i], `${at}[${i}]`) : null; + if (e) return e; + } + return null; + } + case "object": { + if (typeof value !== "object" || value === null || Array.isArray(value)) return `${at} must be an object`; + const props = (schema.properties ?? {}) as Record; + for (const key of (schema.required ?? []) as string[]) if (!(key in value)) return `${at}.${key} is required`; + for (const [key, v] of Object.entries(value)) { + if (!(key in props)) { if (schema.additionalProperties === false) return `${at}.${key} is not allowed`; continue; } + const e = schemaError(props[key]!, v, `${at}.${key}`); + if (e) return e; + } + return null; + } + case undefined: + return null; + default: + throw new Error(`reply schema type "${String(schema.type)}" is not supported`); + } +} diff --git a/examples/prompt-lab/lib/score.ts b/examples/prompt-lab/lib/score.ts new file mode 100644 index 00000000..8e1ad9dd --- /dev/null +++ b/examples/prompt-lab/lib/score.ts @@ -0,0 +1,25 @@ +// "What the score looks like": worked = the new answer matches persisted gold +// for that question × patient. Un-golded patients are shown, never counted. +import type { Output } from "./types.ts"; + +export type Result = "worked" | "did not" | "no gold yet"; +export interface ScoreRow { patientId: string; gold: string | null; rerun: string; result: Result } +export interface Score { rows: ScoreRow[]; golded: number; worked: number; percent: number | null } + +export function score(questionId: string, rerun: Record, gold: Record): Score { + const rows = Object.keys(rerun).sort().map((patientId): ScoreRow => { + const g = gold[`${questionId}|${patientId}`]; + const answer = rerun[patientId]!.answer; + if (!g) return { patientId, gold: null, rerun: answer, result: "no gold yet" }; + return { patientId, gold: g.answer, rerun: answer, result: g.answer === answer ? "worked" : "did not" }; + }); + const golded = rows.filter((r) => r.result !== "no gold yet").length; + const worked = rows.filter((r) => r.result === "worked").length; + return { rows, golded, worked, percent: golded ? Math.round((worked / golded) * 100) : null }; +} + +export function scoreTable(s: Score): string { + const lines = s.rows.map((r) => ` ${r.patientId}: gold ${r.gold ?? "not set"} · new run ${r.rerun} · ${r.result}`); + const head = s.percent === null ? "no golded patients yet" : `${s.worked} of ${s.golded} golded patients worked (${s.percent}%)`; + return `${head}\n${lines.join("\n")}`; +} diff --git a/examples/prompt-lab/lib/types.ts b/examples/prompt-lab/lib/types.ts new file mode 100644 index 00000000..55caa4d0 --- /dev/null +++ b/examples/prompt-lab/lib/types.ts @@ -0,0 +1,57 @@ +// The Prompt Lab domain, as the brief names it. Plain data: every value here +// crosses a journaled step boundary as JSON. + +export type Confidence = "High" | "Medium" | "Low"; +export const CONFIDENCES: readonly Confidence[] = ["High", "Medium", "Low"]; + +/** What the chart-filling engine returns for one question on one visit. */ +export interface Output { answer: string; confidence: Confidence; explanation: string } + +export interface Question { + text: string; + type: "single"; + /** Apricot's live prompt. Null until a first-pass prompt is committed. */ + livePromptId: string | null; + /** A proposed prompt that is not live. Done = live moves it to livePromptId. */ + draftPromptId?: string | null; +} + +/** Apricot's Bank: prompts are global, keyed by id; questions point at one. */ +export interface Bank { prompts: Record; questions: Record } + +/** One question as a given agency asks it: its answer menu and follow-up parent. */ +export interface AgencyQuestion { questionId: string; options: string[]; parent?: string } +export interface Agency { name: string; visitTypes: Record } + +/** A locked, invented shelf patient. Never a real chart or a scrubbed clone. */ +export interface Patient { + id: string; label: string; locked: true; source: "invented"; + ageBand: string; visitType: string; referral: string; notes: string; +} + +/** A gap brief: coverage the shelf does not have yet. Lives on the manager queue. */ +export interface PatientBrief { + id: string; questionId: string; brief: string; from: "planner" | "you" | "issue"; + status: "queued" | "locked"; +} + +/** A config-send, field-pattern or Apricot issue on the question manager. */ +export interface Issue { + id: string; kind: "config-send" | "apricot" | "field-pattern"; + questionIds: string[]; text: string; status: "open" | "done"; + /** The agency whose menu the rows were answered on, when one sent it. */ + agency?: string; + /** Config-level targets travel with a row sent to question-level. */ + targets?: Record; + proposedPrompt?: string; +} + +export type Pile = "agency-specific" | "shared" | "mismatch"; + +/** Everything a job reads at its start, in one journaled step. */ +export interface Snapshot { + bank: Bank; agencies: Record; guidelines: string; + shelf: Patient[]; briefs: Record; + targets: Record; gold: Record; + issues: Issue[]; patientBriefs: PatientBrief[]; +} diff --git a/examples/prompt-lab/package-lock.json b/examples/prompt-lab/package-lock.json new file mode 100644 index 00000000..99ab9894 --- /dev/null +++ b/examples/prompt-lab/package-lock.json @@ -0,0 +1,2257 @@ +{ + "name": "@relayflows/prompt-lab-flow", + "lockfileVersion": 3, + "requires": true, + "packages": { + "": { + "name": "@relayflows/prompt-lab-flow", + "devDependencies": { + "@relayflows/surface": "2.0.29", + "relayflows": "2.0.29" + } + }, + "node_modules/@hono/node-server": { + "version": "2.1.1", + "resolved": "https://registry.npmjs.org/@hono/node-server/-/node-server-2.1.1.tgz", + "integrity": "sha512-ELuehkj5VCBdgEw9zs+ivkKwyzzUCSQuE96YmiPvn1ECBoZCczbFXJLeEGMTYjphP6gydh4pHMqEYPVMYUVgQg==", + "dev": true, + "license": "MIT", + "engines": { + "node": ">=20" + }, + "peerDependencies": { + "hono": "^4" + } + }, + "node_modules/@isaacs/fs-minipass": { + "version": "4.0.1", + "resolved": "https://registry.npmjs.org/@isaacs/fs-minipass/-/fs-minipass-4.0.1.tgz", + "integrity": "sha512-wgm9Ehl2jpeqP3zw/7mo3kRHFp5MEDhqAdwy1fTGkHAwnkGOVsgpvQhL8B5n1qlb01jV3n/bI0ZfZp5lWA1k4w==", + "dev": true, + "license": "ISC", + "peer": true, + "dependencies": { + "minipass": "^7.0.4" + }, + "engines": { + "node": ">=18.0.0" + } + }, + "node_modules/@modelcontextprotocol/sdk": { + "version": "1.30.0", + "resolved": "https://registry.npmjs.org/@modelcontextprotocol/sdk/-/sdk-1.30.0.tgz", + "integrity": "sha512-xKd8OIzlqNzcqcNumGAa6g+PW2kjD5vrpcKOnfldAUPP3j7lnqMPwlTXQm8gF+UwH72z0lqaRbjr9hqGz0eITA==", + "dev": true, + "license": "MIT", + "dependencies": { + "@hono/node-server": "^1.19.9 || ^2.0.5", + "ajv": "^8.17.1", + "ajv-formats": "^3.0.1", + "content-type": "^1.0.5", + "cors": "^2.8.5", + "cross-spawn": "^7.0.5", + "eventsource": "^3.0.2", + "eventsource-parser": "^3.0.0", + "express": "^5.2.1", + "express-rate-limit": "^8.2.1", + "hono": "^4.11.4", + "jose": "^6.1.3", + "json-schema-typed": "^8.0.2", + "pkce-challenge": "^5.0.0", + "raw-body": "^3.0.0", + "zod": "^3.25 || ^4.0", + "zod-to-json-schema": "^3.25.1" + }, + "engines": { + "node": ">=18" + }, + "peerDependencies": { + "@cfworker/json-schema": "^4.1.1", + "zod": "^3.25 || ^4.0" + }, + "peerDependenciesMeta": { + "@cfworker/json-schema": { + "optional": true + }, + "zod": { + "optional": false + } + } + }, + "node_modules/@relayfile/adapter-core": { + "version": "0.6.2", + "resolved": "https://registry.npmjs.org/@relayfile/adapter-core/-/adapter-core-0.6.2.tgz", + "integrity": "sha512-fVBwiK1W0af4wQ4w/Jr7GA1VBH1qz9Qw1lm90MrOJN0zQfVq1YRwRu9fPk4rp50TQDveLBKE0+MoaLMUmqr5Jg==", + "dev": true, + "license": "Apache-2.0", + "dependencies": { + "@scalar/postman-to-openapi": "^0.6.0", + "cheerio": "^1.2.0", + "minimatch": "^10.0.3", + "yaml": "^2.8.1" + }, + "bin": { + "adapter-core": "dist/src/cli.js" + }, + "engines": { + "node": ">=18" + }, + "peerDependencies": { + "@relayfile/sdk": ">=0.6.0 <1" + } + }, + "node_modules/@relayfile/adapter-linear": { + "version": "0.4.13", + "resolved": "https://registry.npmjs.org/@relayfile/adapter-linear/-/adapter-linear-0.4.13.tgz", + "integrity": "sha512-cJtjrmhK2dV5o7QD61N2vuvl4XnMGjrcWX5zrM9otneVVA983GHRKdHv014Hq7QSIpG5FMSuLGPziYn6vFJI+A==", + "dev": true, + "license": "Apache-2.0", + "dependencies": { + "@relayfile/adapter-core": "^0.6.1" + }, + "engines": { + "node": ">=18" + }, + "peerDependencies": { + "@relayfile/sdk": ">=0.6.0 <1" + } + }, + "node_modules/@relayfile/adapter-reddit": { + "version": "0.2.10", + "resolved": "https://registry.npmjs.org/@relayfile/adapter-reddit/-/adapter-reddit-0.2.10.tgz", + "integrity": "sha512-FBGpWqJgOAC7nSE3SUAPil+pYxpBne1higJmSqGIrycbiIvBPxbluUsTb5Ct7XSyvDB3QmMbOMog5CDaVLhnKQ==", + "dev": true, + "license": "Apache-2.0", + "dependencies": { + "@relayfile/adapter-core": "^0.6.1" + }, + "engines": { + "node": ">=18" + }, + "peerDependencies": { + "@relayfile/sdk": ">=0.6.0 <1" + } + }, + "node_modules/@relayfile/cli-darwin-arm64": { + "version": "0.10.69", + "resolved": "https://registry.npmjs.org/@relayfile/cli-darwin-arm64/-/cli-darwin-arm64-0.10.69.tgz", + "integrity": "sha512-zJpbOfz4Iq2273TkQHvzOmQ06tA2jga6Pxa0WSYBHTfxMwWE7tk9gGwPrtU4ppOiyON6jvQuTQ3byBhwOwlLsQ==", + "cpu": [ + "arm64" + ], + "dev": true, + "license": "Apache-2.0", + "optional": true, + "os": [ + "darwin" + ], + "peer": true + }, + "node_modules/@relayfile/cli-darwin-x64": { + "version": "0.10.69", + "resolved": "https://registry.npmjs.org/@relayfile/cli-darwin-x64/-/cli-darwin-x64-0.10.69.tgz", + "integrity": "sha512-Tda09y7GP3cP8EsKYO0UJLoPOxQVYjrKGPcjF8Z/scWhpyU7cRTnMng9KsqnEDXZIVP/k8yiuku2PaylJjX0GA==", + "cpu": [ + "x64" + ], + "dev": true, + "license": "Apache-2.0", + "optional": true, + "os": [ + "darwin" + ], + "peer": true + }, + "node_modules/@relayfile/cli-linux-arm64": { + "version": "0.10.69", + "resolved": "https://registry.npmjs.org/@relayfile/cli-linux-arm64/-/cli-linux-arm64-0.10.69.tgz", + "integrity": "sha512-5wZYIm55xkhf4UWBj7I/Evl7QU5/dYHsIEZEc/+QU70PTSGzMUKHjO66FcCyS0K5lSRJ6dngRP666kvl2G7vmQ==", + "cpu": [ + "arm64" + ], + "dev": true, + "license": "Apache-2.0", + "optional": true, + "os": [ + "linux" + ], + "peer": true + }, + "node_modules/@relayfile/cli-linux-x64": { + "version": "0.10.69", + "resolved": "https://registry.npmjs.org/@relayfile/cli-linux-x64/-/cli-linux-x64-0.10.69.tgz", + "integrity": "sha512-BgnmP4He93T+Hr2McVQoPxprmdEM4D7kHev5Xziyrjapd1juTZwe0lBkpa8ngn8wyoDk/pVui04s6llGeAU3cw==", + "cpu": [ + "x64" + ], + "dev": true, + "license": "Apache-2.0", + "optional": true, + "os": [ + "linux" + ], + "peer": true + }, + "node_modules/@relayfile/cli-win32-arm64": { + "version": "0.10.69", + "resolved": "https://registry.npmjs.org/@relayfile/cli-win32-arm64/-/cli-win32-arm64-0.10.69.tgz", + "integrity": "sha512-KITNWpQaHBX7iBA1TgOHdO2xMtSY+LH4UkafzXztfVjTKwOjM4nCgLXfS59Z70l94ae7UowvBjgG9psuiAyJoQ==", + "cpu": [ + "arm64" + ], + "dev": true, + "license": "Apache-2.0", + "optional": true, + "os": [ + "win32" + ], + "peer": true + }, + "node_modules/@relayfile/cli-win32-x64": { + "version": "0.10.69", + "resolved": "https://registry.npmjs.org/@relayfile/cli-win32-x64/-/cli-win32-x64-0.10.69.tgz", + "integrity": "sha512-wpogN9nsnw2boMGOhuHyS9SV13Y3QzikE0hTYculUIdxN0tPZDsWgNj2IsN8s/E6edA+TfQ+uUpbExadM/Y/LA==", + "cpu": [ + "x64" + ], + "dev": true, + "license": "Apache-2.0", + "optional": true, + "os": [ + "win32" + ], + "peer": true + }, + "node_modules/@relayfile/core": { + "version": "0.10.69", + "resolved": "https://registry.npmjs.org/@relayfile/core/-/core-0.10.69.tgz", + "integrity": "sha512-ZN2UrobsvOOcLkI7mZ3HFW0VK1YWwaxL3q8T1pIPj6hRzufQCnaWwXtk7XVpIFRGerdUSrWbnYJZXfg/jgBtzA==", + "dev": true, + "license": "Apache-2.0", + "peer": true, + "engines": { + "node": ">=18" + } + }, + "node_modules/@relayfile/mount-darwin-arm64": { + "version": "0.10.69", + "resolved": "https://registry.npmjs.org/@relayfile/mount-darwin-arm64/-/mount-darwin-arm64-0.10.69.tgz", + "integrity": "sha512-IyZqcC+bKyP8yIfTbUAOvbtDp7JPcYCypMtv+CS+hew6TX7C6/wr+gUAn4Er3zELIjW3Ktjzi86O8c8TUoX7sQ==", + "cpu": [ + "arm64" + ], + "dev": true, + "license": "Apache-2.0", + "optional": true, + "os": [ + "darwin" + ], + "peer": true + }, + "node_modules/@relayfile/mount-darwin-x64": { + "version": "0.10.69", + "resolved": "https://registry.npmjs.org/@relayfile/mount-darwin-x64/-/mount-darwin-x64-0.10.69.tgz", + "integrity": "sha512-S6nFPo7lRRo0HFagqZtqa7QRVSiFxGbGl0fEs+1TVRWlQt5ibvbpPMpcE7zOd7z4I/g4EOw+pmOcHmXXkjPNSA==", + "cpu": [ + "x64" + ], + "dev": true, + "license": "Apache-2.0", + "optional": true, + "os": [ + "darwin" + ], + "peer": true + }, + "node_modules/@relayfile/mount-linux-arm64": { + "version": "0.10.69", + "resolved": "https://registry.npmjs.org/@relayfile/mount-linux-arm64/-/mount-linux-arm64-0.10.69.tgz", + "integrity": "sha512-bhbHhwcd+wvpoz5mxqAF6h/74phb1mcxMzIDacAiuSQ9ymXxcc9IGd3nvFSr+h0wsyKfZN8tGppBN/Ez1h/ZLA==", + "cpu": [ + "arm64" + ], + "dev": true, + "license": "Apache-2.0", + "optional": true, + "os": [ + "linux" + ], + "peer": true + }, + "node_modules/@relayfile/mount-linux-x64": { + "version": "0.10.69", + "resolved": "https://registry.npmjs.org/@relayfile/mount-linux-x64/-/mount-linux-x64-0.10.69.tgz", + "integrity": "sha512-aTy6ZiYgF+8SuSF9hGDgJU/z9K6rh+UUAl6Emp/valiw3FmM0fLOt2rAt1/KyXtuhgJE09/XoPRYBLa8UyIeUg==", + "cpu": [ + "x64" + ], + "dev": true, + "license": "Apache-2.0", + "optional": true, + "os": [ + "linux" + ], + "peer": true + }, + "node_modules/@relayfile/relay-helpers": { + "version": "0.4.12", + "resolved": "https://registry.npmjs.org/@relayfile/relay-helpers/-/relay-helpers-0.4.12.tgz", + "integrity": "sha512-Izjl33HLCjxDFVZHK/+f+e/73IUiEg534QNBJGkRYZXjH6LvPKfGiPXViiY5zYLxFtgnzNQa+/TI6MB4X3BnEQ==", + "dev": true, + "license": "Apache-2.0", + "dependencies": { + "@relayfile/adapter-core": "^0.6.1", + "@relayfile/adapter-linear": "^0.4.13", + "@relayfile/adapter-reddit": "^0.2.10" + } + }, + "node_modules/@relayfile/sdk": { + "version": "0.10.69", + "resolved": "https://registry.npmjs.org/@relayfile/sdk/-/sdk-0.10.69.tgz", + "integrity": "sha512-yKImAt0XrR7lqJTaWXCv4NlCyZe1YFxh1qEBKoKCPYhEiF31Rrbarq+PZlQybg4r2Kbiff4XtSERMy31iHvoSw==", + "dev": true, + "license": "Apache-2.0", + "peer": true, + "dependencies": { + "@relayfile/core": "0.10.69", + "ignore": "^7.0.5", + "tar": "^7.5.10" + }, + "engines": { + "node": ">=18" + }, + "optionalDependencies": { + "@relayfile/cli-darwin-arm64": "0.10.69", + "@relayfile/cli-darwin-x64": "0.10.69", + "@relayfile/cli-linux-arm64": "0.10.69", + "@relayfile/cli-linux-x64": "0.10.69", + "@relayfile/cli-win32-arm64": "0.10.69", + "@relayfile/cli-win32-x64": "0.10.69", + "@relayfile/mount-darwin-arm64": "0.10.69", + "@relayfile/mount-darwin-x64": "0.10.69", + "@relayfile/mount-linux-arm64": "0.10.69", + "@relayfile/mount-linux-x64": "0.10.69" + } + }, + "node_modules/@relayflows/runtime-darwin-arm64": { + "version": "2.0.29", + "resolved": "https://registry.npmjs.org/@relayflows/runtime-darwin-arm64/-/runtime-darwin-arm64-2.0.29.tgz", + "integrity": "sha512-jQYdHHiwTyX6NwqUs/szHNtzCSCjnqwQHh2n3s/tBgT/8v20n0BlWpva0FKliSyRGX2Hg/ykF22AvNdf35m5sg==", + "cpu": [ + "arm64" + ], + "dev": true, + "license": "Apache-2.0", + "optional": true, + "os": [ + "darwin" + ], + "bin": { + "relayflowd": "bin/relayflowd" + }, + "engines": { + "node": ">=20" + } + }, + "node_modules/@relayflows/runtime-linux-x64": { + "version": "2.0.29", + "resolved": "https://registry.npmjs.org/@relayflows/runtime-linux-x64/-/runtime-linux-x64-2.0.29.tgz", + "integrity": "sha512-D0CK1W3GZOrwNJKMLfYhu9vtwV3CCQO6VXVooPZO3nuB5wzP0M6kGZxTKB2f74+GO/Y5fZ1asrIxHZRHIwKduA==", + "cpu": [ + "x64" + ], + "dev": true, + "license": "Apache-2.0", + "optional": true, + "os": [ + "linux" + ], + "bin": { + "relayflowd": "bin/relayflowd" + }, + "engines": { + "node": ">=20" + } + }, + "node_modules/@relayflows/sdk": { + "version": "2.0.29", + "resolved": "https://registry.npmjs.org/@relayflows/sdk/-/sdk-2.0.29.tgz", + "integrity": "sha512-h8Pl2wmtXa5N6S+JGUfN6Zq1+bEwVvc+mBEsi1fdBXdhRtbJjvlx6fdm84zH22TL8UhijEhMDtUb10fY3PQZ0g==", + "dev": true, + "license": "Apache-2.0", + "dependencies": { + "@modelcontextprotocol/sdk": "^1.30.0", + "@relayfile/adapter-core": "0.6.2", + "@relayfile/relay-helpers": "0.4.12", + "@relayflows/surface": "2.0.29", + "@types/js-yaml": "^4.0.9", + "ai-hist": "0.4.1", + "ajv": "^8.17.1", + "ajv-draft-04": "^1.0.0", + "js-yaml": "^5.4.1", + "re2js": "^2.8.6", + "yaml": "^2.5.1" + }, + "bin": { + "flows": "dist/cli.js" + }, + "peerDependencies": { + "@agent-relay/harness-driver": ">=12.3.1 <13", + "@agent-relay/sdk": ">=12.3.1 <13" + }, + "peerDependenciesMeta": { + "@agent-relay/harness-driver": { + "optional": true + }, + "@agent-relay/sdk": { + "optional": true + } + } + }, + "node_modules/@relayflows/surface": { + "version": "2.0.29", + "resolved": "https://registry.npmjs.org/@relayflows/surface/-/surface-2.0.29.tgz", + "integrity": "sha512-DpxA5hJWkHvRYR0uxkpPbjiakYNvFRgaeMbAc5wizgC68DQwxelM3PxF6yrvkBCdndqRNjMK+OqD9tZMMJRrrQ==", + "dev": true, + "license": "Apache-2.0", + "dependencies": { + "ai-hist": "0.4.1" + }, + "engines": { + "node": ">=20.19.0 || >=22.12.0" + }, + "peerDependencies": { + "@relayfile/relay-helpers": "0.4.12" + } + }, + "node_modules/@scalar/helpers": { + "version": "0.5.1", + "resolved": "https://registry.npmjs.org/@scalar/helpers/-/helpers-0.5.1.tgz", + "integrity": "sha512-9VvPfv8b+YZVIFwR3SWeq4Y8ij/kU3/kf2M6NKcbf2iVyh63d8s0ssap5m/nOhiz/Puidv/29MAJlJCA0LRssA==", + "dev": true, + "license": "MIT", + "engines": { + "node": ">=22" + } + }, + "node_modules/@scalar/openapi-types": { + "version": "0.7.0", + "resolved": "https://registry.npmjs.org/@scalar/openapi-types/-/openapi-types-0.7.0.tgz", + "integrity": "sha512-kN0PwlJW0de4bwQ4ib+mBHzKJUvBCyR/gwU4zLEq6SCbj+GfgYUh+2a0/yl1WYVUiSkkwFsHjfmQ8KjhR3HK0Q==", + "dev": true, + "license": "MIT", + "engines": { + "node": ">=22" + } + }, + "node_modules/@scalar/postman-to-openapi": { + "version": "0.6.3", + "resolved": "https://registry.npmjs.org/@scalar/postman-to-openapi/-/postman-to-openapi-0.6.3.tgz", + "integrity": "sha512-Y/tMuRZG34wEfpTxDfXFp5o2X3ibb5ojGWupGJ9ZxkThCx7rOGydnszJPzEbgDK3eF6nJ6UuE7bCTpIEutYnPw==", + "dev": true, + "license": "MIT", + "dependencies": { + "@scalar/helpers": "0.5.1", + "@scalar/openapi-types": "0.7.0" + }, + "engines": { + "node": ">=22" + } + }, + "node_modules/@types/js-yaml": { + "version": "4.0.9", + "resolved": "https://registry.npmjs.org/@types/js-yaml/-/js-yaml-4.0.9.tgz", + "integrity": "sha512-k4MGaQl5TGo/iipqb2UDG2UwjXziSWkh0uysQelTlJpX1qGlpUZYm8PnO4DxG1qBomtJUdYJ6qR6xdIah10JLg==", + "dev": true, + "license": "MIT" + }, + "node_modules/accepts": { + "version": "2.0.0", + "resolved": "https://registry.npmjs.org/accepts/-/accepts-2.0.0.tgz", + "integrity": "sha512-5cvg6CtKwfgdmVqY1WIiXKc3Q1bkRqGLi+2W/6ao+6Y7gu/RCwRuAhGEzh5B4KlszSuTLgZYuqFqo5bImjNKng==", + "dev": true, + "license": "MIT", + "dependencies": { + "mime-types": "^3.0.0", + "negotiator": "^1.0.0" + }, + "engines": { + "node": ">= 0.6" + } + }, + "node_modules/ai-hist": { + "version": "0.4.1", + "resolved": "https://registry.npmjs.org/ai-hist/-/ai-hist-0.4.1.tgz", + "integrity": "sha512-qn/jXFtWoY4timtzRj1DO5RgqPJA8UoDAm9Qe3bJ9wM3ph+McK+kY/6H2YDUs67MHzWjGlSc1KAe6cdmFbDfeQ==", + "dev": true, + "license": "MIT", + "dependencies": { + "@modelcontextprotocol/sdk": "^1.29.0", + "sql.js": "^1.13.0", + "zod": "^4.4.3" + }, + "bin": { + "ai-hist-mcp": "dist/mcp-server.js" + }, + "engines": { + "node": ">=18" + } + }, + "node_modules/ajv": { + "version": "8.20.0", + "resolved": "https://registry.npmjs.org/ajv/-/ajv-8.20.0.tgz", + "integrity": "sha512-Thbli+OlOj+iMPYFBVBfJ3OmCAnaSyNn4M1vz9T6Gka5Jt9ba/HIR56joy65tY6kx/FCF5VXNB819Y7/GUrBGA==", + "dev": true, + "license": "MIT", + "dependencies": { + "fast-deep-equal": "^3.1.3", + "fast-uri": "^3.0.1", + "json-schema-traverse": "^1.0.0", + "require-from-string": "^2.0.2" + }, + "funding": { + "type": "github", + "url": "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/sponsors/epoberezkin" + } + }, + "node_modules/ajv-draft-04": { + "version": "1.0.0", + "resolved": "https://registry.npmjs.org/ajv-draft-04/-/ajv-draft-04-1.0.0.tgz", + "integrity": "sha512-mv00Te6nmYbRp5DCwclxtt7yV/joXJPGS7nM+97GdxvuttCOfgI3K4U25zboyeX0O+myI8ERluxQe5wljMmVIw==", + "dev": true, + "license": "MIT", + "peerDependencies": { + "ajv": "^8.5.0" + }, + "peerDependenciesMeta": { + "ajv": { + "optional": true + } + } + }, + "node_modules/ajv-formats": { + "version": "3.0.1", + "resolved": "https://registry.npmjs.org/ajv-formats/-/ajv-formats-3.0.1.tgz", + "integrity": "sha512-8iUql50EUR+uUcdRQ3HDqa6EVyo3docL8g5WJ3FNcWmu62IbkGUue/pEyLBW8VGKKucTPgqeks4fIU1DA4yowQ==", + "dev": true, + "license": "MIT", + "dependencies": { + "ajv": "^8.0.0" + }, + "peerDependencies": { + "ajv": "^8.0.0" + }, + "peerDependenciesMeta": { + "ajv": { + "optional": true + } + } + }, + "node_modules/argparse": { + "version": "2.0.1", + "resolved": "https://registry.npmjs.org/argparse/-/argparse-2.0.1.tgz", + "integrity": "sha512-8+9WqebbFzpX9OR+Wa6O29asIogeRMzcGtAINdpMHHyAg10f05aSFVBbcEqGf/PXw1EjAZ+q2/bEBg3DvurK3Q==", + "dev": true, + "license": "Python-2.0" + }, + "node_modules/balanced-match": { + "version": "4.0.4", + "resolved": "https://registry.npmjs.org/balanced-match/-/balanced-match-4.0.4.tgz", + "integrity": "sha512-BLrgEcRTwX2o6gGxGOCNyMvGSp35YofuYzw9h1IMTRmKqttAZZVU67bdb9Pr2vUHA8+j3i2tJfjO6C6+4myGTA==", + "dev": true, + "license": "MIT", + "engines": { + "node": "18 || 20 || >=22" + } + }, + "node_modules/body-parser": { + "version": "2.3.0", + "resolved": "https://registry.npmjs.org/body-parser/-/body-parser-2.3.0.tgz", + "integrity": "sha512-2cGmJupaNgg+QUwVLAucDuWuoMZ6EX9iHDRswZ5lsNYEmwPaRknMPCLZz07yTzVq/83p4o/wzbDZbBrTvGGTIw==", + "dev": true, + "license": "MIT", + "dependencies": { + "bytes": "^3.1.2", + "content-type": "^2.0.0", + "debug": "^4.4.3", + "http-errors": "^2.0.1", + "iconv-lite": "^0.7.2", + "on-finished": "^2.4.1", + "qs": "^6.15.2", + "raw-body": "^3.0.2", + "type-is": "^2.1.0" + }, + "engines": { + "node": ">=18" + }, + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/express" + } + }, + "node_modules/body-parser/node_modules/content-type": { + "version": "2.1.0", + "resolved": "https://registry.npmjs.org/content-type/-/content-type-2.1.0.tgz", + "integrity": "sha512-mj7UPXE0jaqaOsukNZRUEfEi2AcL7C/vwmwcHV0O97eO1E1pxBZuyjlZrx5seTaNBg1U6+o35wpa35Qfcc+7ag==", + "dev": true, + "license": "MIT", + "engines": { + "node": ">=18" + }, + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/express" + } + }, + "node_modules/body-parser/node_modules/iconv-lite": { + "version": "0.7.3", + "resolved": "https://registry.npmjs.org/iconv-lite/-/iconv-lite-0.7.3.tgz", + "integrity": "sha512-IKXpvIzjnC9XTAUbVBcMfGS0EPaIXtW6v+zr+RRp+hqULEpo0owZax6wyRwPOJbWbzjYspQwusTsfVr0ifh4uQ==", + "dev": true, + "license": "MIT", + "dependencies": { + "safer-buffer": ">= 2.1.2 < 3.0.0" + }, + "engines": { + "node": ">=0.10.0" + }, + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/express" + } + }, + "node_modules/boolbase": { + "version": "1.0.0", + "resolved": "https://registry.npmjs.org/boolbase/-/boolbase-1.0.0.tgz", + "integrity": "sha512-JZOSA7Mo9sNGB8+UjSgzdLtokWAky1zbztM3WRLCbZ70/3cTANmQmOdR7y2g+J0e2WXywy1yS468tY+IruqEww==", + "dev": true, + "license": "ISC" + }, + "node_modules/brace-expansion": { + "version": "5.0.12", + "resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-5.0.12.tgz", + "integrity": "sha512-YovQ3rzhaLMIrDjNDMkNS01tea93qhEhG5xy8f6+R0l+dw3Ki+5sCoIoI942iuLZTHWogWktgwVDhU09iNEimQ==", + "dev": true, + "license": "MIT", + "dependencies": { + "balanced-match": "^4.0.2" + }, + "engines": { + "node": "20 || >=22" + } + }, + "node_modules/bytes": { + "version": "3.1.2", + "resolved": "https://registry.npmjs.org/bytes/-/bytes-3.1.2.tgz", + "integrity": "sha512-/Nf7TyzTx6S3yRJObOAV7956r8cr2+Oj8AC5dt8wSP3BQAoeX58NoHyCU8P8zGkNXStjTSi6fzO6F0pBdcYbEg==", + "dev": true, + "license": "MIT", + "engines": { + "node": ">= 0.8" + } + }, + "node_modules/call-bind-apply-helpers": { + "version": "1.0.2", + "resolved": "https://registry.npmjs.org/call-bind-apply-helpers/-/call-bind-apply-helpers-1.0.2.tgz", + "integrity": "sha512-Sp1ablJ0ivDkSzjcaJdxEunN5/XvksFJ2sMBFfq6x0ryhQV/2b/KwFe21cMpmHtPOSij8K99/wSfoEuTObmuMQ==", + "dev": true, + "license": "MIT", + "dependencies": { + "es-errors": "^1.3.0", + "function-bind": "^1.1.2" + }, + "engines": { + "node": ">= 0.4" + } + }, + "node_modules/call-bound": { + "version": "1.0.4", + "resolved": "https://registry.npmjs.org/call-bound/-/call-bound-1.0.4.tgz", + "integrity": "sha512-+ys997U96po4Kx/ABpBCqhA9EuxJaQWDQg7295H4hBphv3IZg0boBKuwYpt4YXp6MZ5AmZQnU/tyMTlRpaSejg==", + "dev": true, + "license": "MIT", + "dependencies": { + "call-bind-apply-helpers": "^1.0.2", + "get-intrinsic": "^1.3.0" + }, + "engines": { + "node": ">= 0.4" + }, + "funding": { + "url": "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/sponsors/ljharb" + } + }, + "node_modules/cheerio": { + "version": "1.2.0", + "resolved": "https://registry.npmjs.org/cheerio/-/cheerio-1.2.0.tgz", + "integrity": "sha512-WDrybc/gKFpTYQutKIK6UvfcuxijIZfMfXaYm8NMsPQxSYvf+13fXUJ4rztGGbJcBQ/GF55gvrZ0Bc0bj/mqvg==", + "dev": true, + "license": "MIT", + "dependencies": { + "cheerio-select": "^2.1.0", + "dom-serializer": "^2.0.0", + "domhandler": "^5.0.3", + "domutils": "^3.2.2", + "encoding-sniffer": "^0.2.1", + "htmlparser2": "^10.1.0", + "parse5": "^7.3.0", + "parse5-htmlparser2-tree-adapter": "^7.1.0", + "parse5-parser-stream": "^7.1.2", + "undici": "^7.19.0", + "whatwg-mimetype": "^4.0.0" + }, + "engines": { + "node": ">=20.18.1" + }, + "funding": { + "url": "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/cheeriojs/cheerio?sponsor=1" + } + }, + "node_modules/cheerio-select": { + "version": "2.1.0", + "resolved": "https://registry.npmjs.org/cheerio-select/-/cheerio-select-2.1.0.tgz", + "integrity": "sha512-9v9kG0LvzrlcungtnJtpGNxY+fzECQKhK4EGJX2vByejiMX84MFNQw4UxPJl3bFbTMw+Dfs37XaIkCwTZfLh4g==", + "dev": true, + "license": "BSD-2-Clause", + "dependencies": { + "boolbase": "^1.0.0", + "css-select": "^5.1.0", + "css-what": "^6.1.0", + "domelementtype": "^2.3.0", + "domhandler": "^5.0.3", + "domutils": "^3.0.1" + }, + "funding": { + "url": "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/sponsors/fb55" + } + }, + "node_modules/chownr": { + "version": "3.0.0", + "resolved": "https://registry.npmjs.org/chownr/-/chownr-3.0.0.tgz", + "integrity": "sha512-+IxzY9BZOQd/XuYPRmrvEVjF/nqj5kgT4kEq7VofrDoM1MxoRjEWkrCC3EtLi59TVawxTAn+orJwFQcrqEN1+g==", + "dev": true, + "license": "BlueOak-1.0.0", + "peer": true, + "engines": { + "node": ">=18" + } + }, + "node_modules/content-disposition": { + "version": "1.1.0", + "resolved": "https://registry.npmjs.org/content-disposition/-/content-disposition-1.1.0.tgz", + "integrity": "sha512-5jRCH9Z/+DRP7rkvY83B+yGIGX96OYdJmzngqnw2SBSxqCFPd0w2km3s5iawpGX8krnwSGmF0FW5Nhr0Hfai3g==", + "dev": true, + "license": "MIT", + "engines": { + "node": ">=18" + }, + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/express" + } + }, + "node_modules/content-type": { + "version": "1.0.5", + "resolved": "https://registry.npmjs.org/content-type/-/content-type-1.0.5.tgz", + "integrity": "sha512-nTjqfcBFEipKdXCv4YDQWCfmcLZKm81ldF0pAopTvyrFGVbcR6P/VAAd5G7N+0tTr8QqiU0tFadD6FK4NtJwOA==", + "dev": true, + "license": "MIT", + "engines": { + "node": ">= 0.6" + } + }, + "node_modules/cookie": { + "version": "0.7.2", + "resolved": "https://registry.npmjs.org/cookie/-/cookie-0.7.2.tgz", + "integrity": "sha512-yki5XnKuf750l50uGTllt6kKILY4nQ1eNIQatoXEByZ5dWgnKqbnqmTrBE5B4N7lrMJKQ2ytWMiTO2o0v6Ew/w==", + "dev": true, + "license": "MIT", + "engines": { + "node": ">= 0.6" + } + }, + "node_modules/cookie-signature": { + "version": "1.2.2", + "resolved": "https://registry.npmjs.org/cookie-signature/-/cookie-signature-1.2.2.tgz", + "integrity": "sha512-D76uU73ulSXrD1UXF4KE2TMxVVwhsnCgfAyTg9k8P6KGZjlXKrOLe4dJQKI3Bxi5wjesZoFXJWElNWBjPZMbhg==", + "dev": true, + "license": "MIT", + "engines": { + "node": ">=6.6.0" + } + }, + "node_modules/cors": { + "version": "2.8.6", + "resolved": "https://registry.npmjs.org/cors/-/cors-2.8.6.tgz", + "integrity": "sha512-tJtZBBHA6vjIAaF6EnIaq6laBBP9aq/Y3ouVJjEfoHbRBcHBAHYcMh/w8LDrk2PvIMMq8gmopa5D4V8RmbrxGw==", + "dev": true, + "license": "MIT", + "dependencies": { + "object-assign": "^4", + "vary": "^1" + }, + "engines": { + "node": ">= 0.10" + }, + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/express" + } + }, + "node_modules/cross-spawn": { + "version": "7.0.6", + "resolved": "https://registry.npmjs.org/cross-spawn/-/cross-spawn-7.0.6.tgz", + "integrity": "sha512-uV2QOWP2nWzsy2aMp8aRibhi9dlzF5Hgh5SHaB9OiTGEyDTiJJyx0uy51QXdyWbtAHNua4XJzUKca3OzKUd3vA==", + "dev": true, + "license": "MIT", + "dependencies": { + "path-key": "^3.1.0", + "shebang-command": "^2.0.0", + "which": "^2.0.1" + }, + "engines": { + "node": ">= 8" + } + }, + "node_modules/css-select": { + "version": "5.2.2", + "resolved": "https://registry.npmjs.org/css-select/-/css-select-5.2.2.tgz", + "integrity": "sha512-TizTzUddG/xYLA3NXodFM0fSbNizXjOKhqiQQwvhlspadZokn1KDy0NZFS0wuEubIYAV5/c1/lAr0TaaFXEXzw==", + "dev": true, + "license": "BSD-2-Clause", + "dependencies": { + "boolbase": "^1.0.0", + "css-what": "^6.1.0", + "domhandler": "^5.0.2", + "domutils": "^3.0.1", + "nth-check": "^2.0.1" + }, + "funding": { + "url": "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/sponsors/fb55" + } + }, + "node_modules/css-what": { + "version": "6.2.2", + "resolved": "https://registry.npmjs.org/css-what/-/css-what-6.2.2.tgz", + "integrity": "sha512-u/O3vwbptzhMs3L1fQE82ZSLHQQfto5gyZzwteVIEyeaY5Fc7R4dapF/BvRoSYFeqfBk4m0V1Vafq5Pjv25wvA==", + "dev": true, + "license": "BSD-2-Clause", + "engines": { + "node": ">= 6" + }, + "funding": { + "url": "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/sponsors/fb55" + } + }, + "node_modules/debug": { + "version": "4.4.3", + "resolved": "https://registry.npmjs.org/debug/-/debug-4.4.3.tgz", + "integrity": "sha512-RGwwWnwQvkVfavKVt22FGLw+xYSdzARwm0ru6DhTVA3umU5hZc28V3kO4stgYryrTlLpuvgI9GiijltAjNbcqA==", + "dev": true, + "license": "MIT", + "dependencies": { + "ms": "^2.1.3" + }, + "engines": { + "node": ">=6.0" + }, + "peerDependenciesMeta": { + "supports-color": { + "optional": true + } + } + }, + "node_modules/depd": { + "version": "2.0.0", + "resolved": "https://registry.npmjs.org/depd/-/depd-2.0.0.tgz", + "integrity": "sha512-g7nH6P6dyDioJogAAGprGpCtVImJhpPk/roCzdb3fIh61/s/nPsfR6onyMwkCAR/OlC3yBC0lESvUoQEAssIrw==", + "dev": true, + "license": "MIT", + "engines": { + "node": ">= 0.8" + } + }, + "node_modules/dom-serializer": { + "version": "2.0.0", + "resolved": "https://registry.npmjs.org/dom-serializer/-/dom-serializer-2.0.0.tgz", + "integrity": "sha512-wIkAryiqt/nV5EQKqQpo3SToSOV9J0DnbJqwK7Wv/Trc92zIAYZ4FlMu+JPFW1DfGFt81ZTCGgDEabffXeLyJg==", + "dev": true, + "license": "MIT", + "dependencies": { + "domelementtype": "^2.3.0", + "domhandler": "^5.0.2", + "entities": "^4.2.0" + }, + "funding": { + "url": "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/cheeriojs/dom-serializer?sponsor=1" + } + }, + "node_modules/domelementtype": { + "version": "2.3.0", + "resolved": "https://registry.npmjs.org/domelementtype/-/domelementtype-2.3.0.tgz", + "integrity": "sha512-OLETBj6w0OsagBwdXnPdN0cnMfF9opN69co+7ZrbfPGrdpPVNBUj02spi6B1N7wChLQiPn4CSH/zJvXw56gmHw==", + "dev": true, + "funding": [ + { + "type": "github", + "url": "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/sponsors/fb55" + } + ], + "license": "BSD-2-Clause" + }, + "node_modules/domhandler": { + "version": "5.0.3", + "resolved": "https://registry.npmjs.org/domhandler/-/domhandler-5.0.3.tgz", + "integrity": "sha512-cgwlv/1iFQiFnU96XXgROh8xTeetsnJiDsTc7TYCLFd9+/WNkIqPTxiM/8pSd8VIrhXGTf1Ny1q1hquVqDJB5w==", + "dev": true, + "license": "BSD-2-Clause", + "dependencies": { + "domelementtype": "^2.3.0" + }, + "engines": { + "node": ">= 4" + }, + "funding": { + "url": "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/fb55/domhandler?sponsor=1" + } + }, + "node_modules/domutils": { + "version": "3.2.2", + "resolved": "https://registry.npmjs.org/domutils/-/domutils-3.2.2.tgz", + "integrity": "sha512-6kZKyUajlDuqlHKVX1w7gyslj9MPIXzIFiz/rGu35uC1wMi+kMhQwGhl4lt9unC9Vb9INnY9Z3/ZA3+FhASLaw==", + "dev": true, + "license": "BSD-2-Clause", + "dependencies": { + "dom-serializer": "^2.0.0", + "domelementtype": "^2.3.0", + "domhandler": "^5.0.3" + }, + "funding": { + "url": "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/fb55/domutils?sponsor=1" + } + }, + "node_modules/dunder-proto": { + "version": "1.0.1", + "resolved": "https://registry.npmjs.org/dunder-proto/-/dunder-proto-1.0.1.tgz", + "integrity": "sha512-KIN/nDJBQRcXw0MLVhZE9iQHmG68qAVIBg9CqmUYjmQIhgij9U5MFvrqkUL5FbtyyzZuOeOt0zdeRe4UY7ct+A==", + "dev": true, + "license": "MIT", + "dependencies": { + "call-bind-apply-helpers": "^1.0.1", + "es-errors": "^1.3.0", + "gopd": "^1.2.0" + }, + "engines": { + "node": ">= 0.4" + } + }, + "node_modules/ee-first": { + "version": "1.1.1", + "resolved": "https://registry.npmjs.org/ee-first/-/ee-first-1.1.1.tgz", + "integrity": "sha512-WMwm9LhRUo+WUaRN+vRuETqG89IgZphVSNkdFgeb6sS/E4OrDIN7t48CAewSHXc6C8lefD8KKfr5vY61brQlow==", + "dev": true, + "license": "MIT" + }, + "node_modules/encodeurl": { + "version": "2.0.0", + "resolved": "https://registry.npmjs.org/encodeurl/-/encodeurl-2.0.0.tgz", + "integrity": "sha512-Q0n9HRi4m6JuGIV1eFlmvJB7ZEVxu93IrMyiMsGC0lrMJMWzRgx6WGquyfQgZVb31vhGgXnfmPNNXmxnOkRBrg==", + "dev": true, + "license": "MIT", + "engines": { + "node": ">= 0.8" + } + }, + "node_modules/encoding-sniffer": { + "version": "0.2.1", + "resolved": "https://registry.npmjs.org/encoding-sniffer/-/encoding-sniffer-0.2.1.tgz", + "integrity": "sha512-5gvq20T6vfpekVtqrYQsSCFZ1wEg5+wW0/QaZMWkFr6BqD3NfKs0rLCx4rrVlSWJeZb5NBJgVLswK/w2MWU+Gw==", + "dev": true, + "license": "MIT", + "dependencies": { + "iconv-lite": "^0.6.3", + "whatwg-encoding": "^3.1.1" + }, + "funding": { + "url": "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/fb55/encoding-sniffer?sponsor=1" + } + }, + "node_modules/entities": { + "version": "4.5.0", + "resolved": "https://registry.npmjs.org/entities/-/entities-4.5.0.tgz", + "integrity": "sha512-V0hjH4dGPh9Ao5p0MoRY6BVqtwCjhz6vI5LT8AJ55H+4g9/4vbHx1I54fS0XuclLhDHArPQCiMjDxjaL8fPxhw==", + "dev": true, + "license": "BSD-2-Clause", + "engines": { + "node": ">=0.12" + }, + "funding": { + "url": "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/fb55/entities?sponsor=1" + } + }, + "node_modules/es-define-property": { + "version": "1.0.1", + "resolved": "https://registry.npmjs.org/es-define-property/-/es-define-property-1.0.1.tgz", + "integrity": "sha512-e3nRfgfUZ4rNGL232gUgX06QNyyez04KdjFrF+LTRoOXmrOgFKDg4BCdsjW8EnT69eqdYGmRpJwiPVYNrCaW3g==", + "dev": true, + "license": "MIT", + "engines": { + "node": ">= 0.4" + } + }, + "node_modules/es-errors": { + "version": "1.3.0", + "resolved": "https://registry.npmjs.org/es-errors/-/es-errors-1.3.0.tgz", + "integrity": "sha512-Zf5H2Kxt2xjTvbJvP2ZWLEICxA6j+hAmMzIlypy4xcBg1vKVnx89Wy0GbS+kf5cwCVFFzdCFh2XSCFNULS6csw==", + "dev": true, + "license": "MIT", + "engines": { + "node": ">= 0.4" + } + }, + "node_modules/es-object-atoms": { + "version": "1.1.2", + "resolved": "https://registry.npmjs.org/es-object-atoms/-/es-object-atoms-1.1.2.tgz", + "integrity": "sha512-HWcBoN6NileqtSydK2FqHbS/LoDd2pqrnQHLyJzBj4kOp/ky2MWMN694xOfkK8/SnUsW2DH7EfyVlydKCsm1Zw==", + "dev": true, + "license": "MIT", + "dependencies": { + "es-errors": "^1.3.0" + }, + "engines": { + "node": ">= 0.4" + } + }, + "node_modules/escape-html": { + "version": "1.0.3", + "resolved": "https://registry.npmjs.org/escape-html/-/escape-html-1.0.3.tgz", + "integrity": "sha512-NiSupZ4OeuGwr68lGIeym/ksIZMJodUGOSCZ/FSnTxcrekbvqrgdUxlJOMpijaKZVjAJrWrGs/6Jy8OMuyj9ow==", + "dev": true, + "license": "MIT" + }, + "node_modules/etag": { + "version": "1.8.1", + "resolved": "https://registry.npmjs.org/etag/-/etag-1.8.1.tgz", + "integrity": "sha512-aIL5Fx7mawVa300al2BnEE4iNvo1qETxLrPI/o05L7z6go7fCw1J6EQmbK4FmJ2AS7kgVF/KEZWufBfdClMcPg==", + "dev": true, + "license": "MIT", + "engines": { + "node": ">= 0.6" + } + }, + "node_modules/eventsource": { + "version": "3.0.7", + "resolved": "https://registry.npmjs.org/eventsource/-/eventsource-3.0.7.tgz", + "integrity": "sha512-CRT1WTyuQoD771GW56XEZFQ/ZoSfWid1alKGDYMmkt2yl8UXrVR4pspqWNEcqKvVIzg6PAltWjxcSSPrboA4iA==", + "dev": true, + "license": "MIT", + "dependencies": { + "eventsource-parser": "^3.0.1" + }, + "engines": { + "node": ">=18.0.0" + } + }, + "node_modules/eventsource-parser": { + "version": "3.1.1", + "resolved": "https://registry.npmjs.org/eventsource-parser/-/eventsource-parser-3.1.1.tgz", + "integrity": "sha512-EKN1vKAMcZ8MlYMpaNuxN6R9yakzH6uajHcHVTqWJzvu5pWw9DyhbP35HH8MVBQ+dZjAfDxk+A8NiR9KWaXiyQ==", + "dev": true, + "license": "MIT", + "engines": { + "node": ">=18.0.0" + } + }, + "node_modules/express": { + "version": "5.2.1", + "resolved": "https://registry.npmjs.org/express/-/express-5.2.1.tgz", + "integrity": "sha512-hIS4idWWai69NezIdRt2xFVofaF4j+6INOpJlVOLDO8zXGpUVEVzIYk12UUi2JzjEzWL3IOAxcTubgz9Po0yXw==", + "dev": true, + "license": "MIT", + "dependencies": { + "accepts": "^2.0.0", + "body-parser": "^2.2.1", + "content-disposition": "^1.0.0", + "content-type": "^1.0.5", + "cookie": "^0.7.1", + "cookie-signature": "^1.2.1", + "debug": "^4.4.0", + "depd": "^2.0.0", + "encodeurl": "^2.0.0", + "escape-html": "^1.0.3", + "etag": "^1.8.1", + "finalhandler": "^2.1.0", + "fresh": "^2.0.0", + "http-errors": "^2.0.0", + "merge-descriptors": "^2.0.0", + "mime-types": "^3.0.0", + "on-finished": "^2.4.1", + "once": "^1.4.0", + "parseurl": "^1.3.3", + "proxy-addr": "^2.0.7", + "qs": "^6.14.0", + "range-parser": "^1.2.1", + "router": "^2.2.0", + "send": "^1.1.0", + "serve-static": "^2.2.0", + "statuses": "^2.0.1", + "type-is": "^2.0.1", + "vary": "^1.1.2" + }, + "engines": { + "node": ">= 18" + }, + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/express" + } + }, + "node_modules/express-rate-limit": { + "version": "8.7.0", + "resolved": "https://registry.npmjs.org/express-rate-limit/-/express-rate-limit-8.7.0.tgz", + "integrity": "sha512-hOwV7WOxXfjRpAM1DSJWZDXx3GhplwD8IfwuwvogD8i1Qnkgosw/H45s4ZnFAUHDAhPjlY9hLBvJhKmGMyY26g==", + "dev": true, + "license": "MIT", + "dependencies": { + "debug": "^4.4.3", + "ip-address": "^10.2.0" + }, + "engines": { + "node": ">= 16" + }, + "funding": { + "url": "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/sponsors/express-rate-limit" + }, + "peerDependencies": { + "express": ">= 4.11" + } + }, + "node_modules/fast-deep-equal": { + "version": "3.1.3", + "resolved": "https://registry.npmjs.org/fast-deep-equal/-/fast-deep-equal-3.1.3.tgz", + "integrity": "sha512-f3qQ9oQy9j2AhBe/H9VC91wLmKBCCU/gDOnKNAYG5hswO7BLKj09Hc5HYNz9cGI++xlpDCIgDaitVs03ATR84Q==", + "dev": true, + "license": "MIT" + }, + "node_modules/fast-uri": { + "version": "3.1.8", + "resolved": "https://registry.npmjs.org/fast-uri/-/fast-uri-3.1.8.tgz", + "integrity": "sha512-GZMtZUTNRpOVIECoXwLNZS5xUGE+mVNbTB8h/7Rwh2TFWcBQiPzTgyZi05BF9UMZKkLJv8XBRJTlU7zg8+ZfMg==", + "dev": true, + "funding": [ + { + "type": "github", + "url": "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/sponsors/fastify" + }, + { + "type": "opencollective", + "url": "https://opencollective.com/fastify" + } + ], + "license": "BSD-3-Clause" + }, + "node_modules/finalhandler": { + "version": "2.1.1", + "resolved": "https://registry.npmjs.org/finalhandler/-/finalhandler-2.1.1.tgz", + "integrity": "sha512-S8KoZgRZN+a5rNwqTxlZZePjT/4cnm0ROV70LedRHZ0p8u9fRID0hJUZQpkKLzro8LfmC8sx23bY6tVNxv8pQA==", + "dev": true, + "license": "MIT", + "dependencies": { + "debug": "^4.4.0", + "encodeurl": "^2.0.0", + "escape-html": "^1.0.3", + "on-finished": "^2.4.1", + "parseurl": "^1.3.3", + "statuses": "^2.0.1" + }, + "engines": { + "node": ">= 18.0.0" + }, + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/express" + } + }, + "node_modules/forwarded": { + "version": "0.2.0", + "resolved": "https://registry.npmjs.org/forwarded/-/forwarded-0.2.0.tgz", + "integrity": "sha512-buRG0fpBtRHSTCOASe6hD258tEubFoRLb4ZNA6NxMVHNw2gOcwHo9wyablzMzOA5z9xA9L1KNjk/Nt6MT9aYow==", + "dev": true, + "license": "MIT", + "engines": { + "node": ">= 0.6" + } + }, + "node_modules/fresh": { + "version": "2.0.0", + "resolved": "https://registry.npmjs.org/fresh/-/fresh-2.0.0.tgz", + "integrity": "sha512-Rx/WycZ60HOaqLKAi6cHRKKI7zxWbJ31MhntmtwMoaTeF7XFH9hhBp8vITaMidfljRQ6eYWCKkaTK+ykVJHP2A==", + "dev": true, + "license": "MIT", + "engines": { + "node": ">= 0.8" + } + }, + "node_modules/function-bind": { + "version": "1.1.2", + "resolved": "https://registry.npmjs.org/function-bind/-/function-bind-1.1.2.tgz", + "integrity": "sha512-7XHNxH7qX9xG5mIwxkhumTox/MIRNcOgDrxWsMt2pAr23WHp6MrRlN7FBSFpCpr+oVO0F744iUgR82nJMfG2SA==", + "dev": true, + "license": "MIT", + "funding": { + "url": "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/sponsors/ljharb" + } + }, + "node_modules/get-intrinsic": { + "version": "1.3.0", + "resolved": "https://registry.npmjs.org/get-intrinsic/-/get-intrinsic-1.3.0.tgz", + "integrity": "sha512-9fSjSaos/fRIVIp+xSJlE6lfwhES7LNtKaCBIamHsjr2na1BiABJPo0mOjjz8GJDURarmCPGqaiVg5mfjb98CQ==", + "dev": true, + "license": "MIT", + "dependencies": { + "call-bind-apply-helpers": "^1.0.2", + "es-define-property": "^1.0.1", + "es-errors": "^1.3.0", + "es-object-atoms": "^1.1.1", + "function-bind": "^1.1.2", + "get-proto": "^1.0.1", + "gopd": "^1.2.0", + "has-symbols": "^1.1.0", + "hasown": "^2.0.2", + "math-intrinsics": "^1.1.0" + }, + "engines": { + "node": ">= 0.4" + }, + "funding": { + "url": "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/sponsors/ljharb" + } + }, + "node_modules/get-proto": { + "version": "1.0.1", + "resolved": "https://registry.npmjs.org/get-proto/-/get-proto-1.0.1.tgz", + "integrity": "sha512-sTSfBjoXBp89JvIKIefqw7U2CCebsc74kiY6awiGogKtoSGbgjYE/G/+l9sF3MWFPNc9IcoOC4ODfKHfxFmp0g==", + "dev": true, + "license": "MIT", + "dependencies": { + "dunder-proto": "^1.0.1", + "es-object-atoms": "^1.0.0" + }, + "engines": { + "node": ">= 0.4" + } + }, + "node_modules/gopd": { + "version": "1.2.0", + "resolved": "https://registry.npmjs.org/gopd/-/gopd-1.2.0.tgz", + "integrity": "sha512-ZUKRh6/kUFoAiTAtTYPZJ3hw9wNxx+BIBOijnlG9PnrJsCcSjs1wyyD6vJpaYtgnzDrKYRSqf3OO6Rfa93xsRg==", + "dev": true, + "license": "MIT", + "engines": { + "node": ">= 0.4" + }, + "funding": { + "url": "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/sponsors/ljharb" + } + }, + "node_modules/has-symbols": { + "version": "1.1.0", + "resolved": "https://registry.npmjs.org/has-symbols/-/has-symbols-1.1.0.tgz", + "integrity": "sha512-1cDNdwJ2Jaohmb3sg4OmKaMBwuC48sYni5HUw2DvsC8LjGTLK9h+eb1X6RyuOHe4hT0ULCW68iomhjUoKUqlPQ==", + "dev": true, + "license": "MIT", + "engines": { + "node": ">= 0.4" + }, + "funding": { + "url": "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/sponsors/ljharb" + } + }, + "node_modules/hasown": { + "version": "2.0.4", + "resolved": "https://registry.npmjs.org/hasown/-/hasown-2.0.4.tgz", + "integrity": "sha512-T2UbfbBEF32wiepXIsMlTW9+dDYC6wMh/t/vYA4tuOMKqWz/n3vr1NFSxQiyP+zk2mXsoMA/i/7qV6LKut1t1A==", + "dev": true, + "license": "MIT", + "dependencies": { + "function-bind": "^1.1.2" + }, + "engines": { + "node": ">= 0.4" + } + }, + "node_modules/hono": { + "version": "4.13.8", + "resolved": "https://registry.npmjs.org/hono/-/hono-4.13.8.tgz", + "integrity": "sha512-/Gng7NfoykZl2pjukW5Z6+8Yxm3BPRf86GTbQnt0SbySkvax4fyL4H3HhY1cCpBGmiW9XDRFzRV+CXK2W8QudQ==", + "dev": true, + "license": "MIT", + "engines": { + "node": ">=16.9.0" + } + }, + "node_modules/htmlparser2": { + "version": "10.1.0", + "resolved": "https://registry.npmjs.org/htmlparser2/-/htmlparser2-10.1.0.tgz", + "integrity": "sha512-VTZkM9GWRAtEpveh7MSF6SjjrpNVNNVJfFup7xTY3UpFtm67foy9HDVXneLtFVt4pMz5kZtgNcvCniNFb1hlEQ==", + "dev": true, + "funding": [ + "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/fb55/htmlparser2?sponsor=1", + { + "type": "github", + "url": "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/sponsors/fb55" + } + ], + "license": "MIT", + "dependencies": { + "domelementtype": "^2.3.0", + "domhandler": "^5.0.3", + "domutils": "^3.2.2", + "entities": "^7.0.1" + } + }, + "node_modules/htmlparser2/node_modules/entities": { + "version": "7.0.1", + "resolved": "https://registry.npmjs.org/entities/-/entities-7.0.1.tgz", + "integrity": "sha512-TWrgLOFUQTH994YUyl1yT4uyavY5nNB5muff+RtWaqNVCAK408b5ZnnbNAUEWLTCpum9w6arT70i1XdQ4UeOPA==", + "dev": true, + "license": "BSD-2-Clause", + "engines": { + "node": ">=0.12" + }, + "funding": { + "url": "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/fb55/entities?sponsor=1" + } + }, + "node_modules/http-errors": { + "version": "2.0.1", + "resolved": "https://registry.npmjs.org/http-errors/-/http-errors-2.0.1.tgz", + "integrity": "sha512-4FbRdAX+bSdmo4AUFuS0WNiPz8NgFt+r8ThgNWmlrjQjt1Q7ZR9+zTlce2859x4KSXrwIsaeTqDoKQmtP8pLmQ==", + "dev": true, + "license": "MIT", + "dependencies": { + "depd": "~2.0.0", + "inherits": "~2.0.4", + "setprototypeof": "~1.2.0", + "statuses": "~2.0.2", + "toidentifier": "~1.0.1" + }, + "engines": { + "node": ">= 0.8" + }, + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/express" + } + }, + "node_modules/iconv-lite": { + "version": "0.6.3", + "resolved": "https://registry.npmjs.org/iconv-lite/-/iconv-lite-0.6.3.tgz", + "integrity": "sha512-4fCk79wshMdzMp2rH06qWrJE4iolqLhCUH+OiuIgU++RB0+94NlDL81atO7GX55uUKueo0txHNtvEyI6D7WdMw==", + "dev": true, + "license": "MIT", + "dependencies": { + "safer-buffer": ">= 2.1.2 < 3.0.0" + }, + "engines": { + "node": ">=0.10.0" + } + }, + "node_modules/ignore": { + "version": "7.0.10", + "resolved": "https://registry.npmjs.org/ignore/-/ignore-7.0.10.tgz", + "integrity": "sha512-HpbUakT7xp5miBUywCHf36ZEuAJNklBJDDsGpUIjMzOSmM8ELSfA9Sa/QDPeNeqeoN31u+UTCkL4klCOVvRm4Q==", + "dev": true, + "license": "MIT", + "peer": true, + "engines": { + "node": ">= 4" + } + }, + "node_modules/inherits": { + "version": "2.0.4", + "resolved": "https://registry.npmjs.org/inherits/-/inherits-2.0.4.tgz", + "integrity": "sha512-k/vGaX4/Yla3WzyMCvTQOXYeIHvqOKtnqBduzTHpzpQZzAskKMhZ2K+EnBiSM9zGSoIFeMpXKxa4dYeZIQqewQ==", + "dev": true, + "license": "ISC" + }, + "node_modules/ip-address": { + "version": "10.7.2", + "resolved": "https://registry.npmjs.org/ip-address/-/ip-address-10.7.2.tgz", + "integrity": "sha512-7H/2gFSIitxc0hG3nOI1glS8QLo/EHBFFLk8vEUjXY/xu0AdL8jZ9U1IzO2PUm0d2D/ofQcAifb0g6OBkt8U7w==", + "dev": true, + "license": "MIT", + "engines": { + "node": ">= 12" + } + }, + "node_modules/ipaddr.js": { + "version": "1.9.1", + "resolved": "https://registry.npmjs.org/ipaddr.js/-/ipaddr.js-1.9.1.tgz", + "integrity": "sha512-0KI/607xoxSToH7GjN1FfSbLoU0+btTicjsQSWQlh/hZykN8KpmMf7uYwPW3R+akZ6R/w18ZlXSHBYXiYUPO3g==", + "dev": true, + "license": "MIT", + "engines": { + "node": ">= 0.10" + } + }, + "node_modules/is-promise": { + "version": "4.0.0", + "resolved": "https://registry.npmjs.org/is-promise/-/is-promise-4.0.0.tgz", + "integrity": "sha512-hvpoI6korhJMnej285dSg6nu1+e6uxs7zG3BYAm5byqDsgJNWwxzM6z6iZiAgQR4TJ30JmBTOwqZUw3WlyH3AQ==", + "dev": true, + "license": "MIT" + }, + "node_modules/isexe": { + "version": "2.0.0", + "resolved": "https://registry.npmjs.org/isexe/-/isexe-2.0.0.tgz", + "integrity": "sha512-RHxMLp9lnKHGHRng9QFhRCMbYAcVpn69smSGcq3f36xjgVVWThj4qqLbTLlq7Ssj8B+fIQ1EuCEGI2lKsyQeIw==", + "dev": true, + "license": "ISC" + }, + "node_modules/jose": { + "version": "6.2.12", + "resolved": "https://registry.npmjs.org/jose/-/jose-6.2.12.tgz", + "integrity": "sha512-9NiFmJEex0sy2Dk58j2UGBSHgUs2ypF9eZSu4L6vjOX3Dp96Sw1F3uL+H+D1sx02jZZdzUT0HgvCy59CuvXcWw==", + "dev": true, + "license": "MIT", + "funding": { + "url": "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/sponsors/panva" + } + }, + "node_modules/js-yaml": { + "version": "5.4.2", + "resolved": "https://registry.npmjs.org/js-yaml/-/js-yaml-5.4.2.tgz", + "integrity": "sha512-m+aqu+LwO1O6sIopafj8HUVl5aawITwZQe/yHpMCKjaWBaA/d07B/QdMb3529REftiU+RMMHL3Vlsw3hON7vWg==", + "dev": true, + "funding": [ + { + "type": "github", + "url": "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/sponsors/puzrin" + }, + { + "type": "github", + "url": "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/sponsors/nodeca" + } + ], + "license": "MIT", + "dependencies": { + "argparse": "^2.0.1" + }, + "bin": { + "js-yaml": "bin/js-yaml.mjs" + } + }, + "node_modules/json-schema-traverse": { + "version": "1.0.0", + "resolved": "https://registry.npmjs.org/json-schema-traverse/-/json-schema-traverse-1.0.0.tgz", + "integrity": "sha512-NM8/P9n3XjXhIZn1lLhkFaACTOURQXjWhV4BA/RnOv8xvgqtqpAX9IO4mRQxSx1Rlo4tqzeqb0sOlruaOy3dug==", + "dev": true, + "license": "MIT" + }, + "node_modules/json-schema-typed": { + "version": "8.0.2", + "resolved": "https://registry.npmjs.org/json-schema-typed/-/json-schema-typed-8.0.2.tgz", + "integrity": "sha512-fQhoXdcvc3V28x7C7BMs4P5+kNlgUURe2jmUT1T//oBRMDrqy1QPelJimwZGo7Hg9VPV3EQV5Bnq4hbFy2vetA==", + "dev": true, + "license": "BSD-2-Clause" + }, + "node_modules/math-intrinsics": { + "version": "1.1.0", + "resolved": "https://registry.npmjs.org/math-intrinsics/-/math-intrinsics-1.1.0.tgz", + "integrity": "sha512-/IXtbwEk5HTPyEwyKX6hGkYXxM9nbj64B+ilVJnC/R6B0pH5G4V3b0pVbL7DBj4tkhBAppbQUlf6F6Xl9LHu1g==", + "dev": true, + "license": "MIT", + "engines": { + "node": ">= 0.4" + } + }, + "node_modules/media-typer": { + "version": "1.1.1", + "resolved": "https://registry.npmjs.org/media-typer/-/media-typer-1.1.1.tgz", + "integrity": "sha512-yz3xRaG20c6/BOzvYoDaGtPmGscs7YivItZEEqe6GbwNfHuxu9YNmvnEkMzKldAGY4/80pRcQRZSEnhquk9XuQ==", + "dev": true, + "license": "MIT", + "engines": { + "node": ">= 0.8" + }, + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/express" + } + }, + "node_modules/merge-descriptors": { + "version": "2.0.0", + "resolved": "https://registry.npmjs.org/merge-descriptors/-/merge-descriptors-2.0.0.tgz", + "integrity": "sha512-Snk314V5ayFLhp3fkUREub6WtjBfPdCPY1Ln8/8munuLuiYhsABgBVWsozAG+MWMbVEvcdcpbi9R7ww22l9Q3g==", + "dev": true, + "license": "MIT", + "engines": { + "node": ">=18" + }, + "funding": { + "url": "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/sponsors/sindresorhus" + } + }, + "node_modules/mime-db": { + "version": "1.54.0", + "resolved": "https://registry.npmjs.org/mime-db/-/mime-db-1.54.0.tgz", + "integrity": "sha512-aU5EJuIN2WDemCcAp2vFBfp/m4EAhWJnUNSSw0ixs7/kXbd6Pg64EmwJkNdFhB8aWt1sH2CTXrLxo/iAGV3oPQ==", + "dev": true, + "license": "MIT", + "engines": { + "node": ">= 0.6" + } + }, + "node_modules/mime-types": { + "version": "3.0.2", + "resolved": "https://registry.npmjs.org/mime-types/-/mime-types-3.0.2.tgz", + "integrity": "sha512-Lbgzdk0h4juoQ9fCKXW4by0UJqj+nOOrI9MJ1sSj4nI8aI2eo1qmvQEie4VD1glsS250n15LsWsYtCugiStS5A==", + "dev": true, + "license": "MIT", + "dependencies": { + "mime-db": "^1.54.0" + }, + "engines": { + "node": ">=18" + }, + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/express" + } + }, + "node_modules/minimatch": { + "version": "10.2.6", + "resolved": "https://registry.npmjs.org/minimatch/-/minimatch-10.2.6.tgz", + "integrity": "sha512-vpLQEs+VLCr1nU0BXS07maYoFwlDAH0gngQuuttxIwutDFEMHq2blX+8vpgxDdK3J1PwjCJiep77OitTZ4Ll1A==", + "dev": true, + "license": "BlueOak-1.0.0", + "dependencies": { + "brace-expansion": "^5.0.8" + }, + "engines": { + "node": "18 || 20 || >=22" + }, + "funding": { + "url": "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/sponsors/isaacs" + } + }, + "node_modules/minipass": { + "version": "7.1.3", + "resolved": "https://registry.npmjs.org/minipass/-/minipass-7.1.3.tgz", + "integrity": "sha512-tEBHqDnIoM/1rXME1zgka9g6Q2lcoCkxHLuc7ODJ5BxbP5d4c2Z5cGgtXAku59200Cx7diuHTOYfSBD8n6mm8A==", + "dev": true, + "license": "BlueOak-1.0.0", + "peer": true, + "engines": { + "node": ">=16 || 14 >=14.17" + } + }, + "node_modules/minizlib": { + "version": "3.1.0", + "resolved": "https://registry.npmjs.org/minizlib/-/minizlib-3.1.0.tgz", + "integrity": "sha512-KZxYo1BUkWD2TVFLr0MQoM8vUUigWD3LlD83a/75BqC+4qE0Hb1Vo5v1FgcfaNXvfXzr+5EhQ6ing/CaBijTlw==", + "dev": true, + "license": "MIT", + "peer": true, + "dependencies": { + "minipass": "^7.1.2" + }, + "engines": { + "node": ">= 18" + } + }, + "node_modules/ms": { + "version": "2.1.3", + "resolved": "https://registry.npmjs.org/ms/-/ms-2.1.3.tgz", + "integrity": "sha512-6FlzubTLZG3J2a/NVCAleEhjzq5oxgHyaCU9yYXvcLsvoVaHJq/s5xXI6/XXP6tz7R9xAOtHnSO/tXtF3WRTlA==", + "dev": true, + "license": "MIT" + }, + "node_modules/negotiator": { + "version": "1.1.0", + "resolved": "https://registry.npmjs.org/negotiator/-/negotiator-1.1.0.tgz", + "integrity": "sha512-NMPBRMJgiQHjbd8phG3Vebdx4kZ1H121rbl5IkMqeOsahptB9BKo/d7oJ3zTXqTgagn2bWlNSXkh0QUGM31RYg==", + "dev": true, + "license": "MIT", + "dependencies": { + "content-type": "^2.1.0" + }, + "engines": { + "node": ">=18" + }, + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/express" + } + }, + "node_modules/negotiator/node_modules/content-type": { + "version": "2.1.0", + "resolved": "https://registry.npmjs.org/content-type/-/content-type-2.1.0.tgz", + "integrity": "sha512-mj7UPXE0jaqaOsukNZRUEfEi2AcL7C/vwmwcHV0O97eO1E1pxBZuyjlZrx5seTaNBg1U6+o35wpa35Qfcc+7ag==", + "dev": true, + "license": "MIT", + "engines": { + "node": ">=18" + }, + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/express" + } + }, + "node_modules/nth-check": { + "version": "2.1.1", + "resolved": "https://registry.npmjs.org/nth-check/-/nth-check-2.1.1.tgz", + "integrity": "sha512-lqjrjmaOoAnWfMmBPL+XNnynZh2+swxiX3WUE0s4yEHI6m+AwrK2UZOimIRl3X/4QctVqS8AiZjFqyOGrMXb/w==", + "dev": true, + "license": "BSD-2-Clause", + "dependencies": { + "boolbase": "^1.0.0" + }, + "funding": { + "url": "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/fb55/nth-check?sponsor=1" + } + }, + "node_modules/object-assign": { + "version": "4.1.1", + "resolved": "https://registry.npmjs.org/object-assign/-/object-assign-4.1.1.tgz", + "integrity": "sha512-rJgTQnkUnH1sFw8yT6VSU3zD3sWmu6sZhIseY8VX+GRu3P6F7Fu+JNDoXfklElbLJSnc3FUQHVe4cU5hj+BcUg==", + "dev": true, + "license": "MIT", + "engines": { + "node": ">=0.10.0" + } + }, + "node_modules/object-inspect": { + "version": "1.13.4", + "resolved": "https://registry.npmjs.org/object-inspect/-/object-inspect-1.13.4.tgz", + "integrity": "sha512-W67iLl4J2EXEGTbfeHCffrjDfitvLANg0UlX3wFUUSTx92KXRFegMHUVgSqE+wvhAbi4WqjGg9czysTV2Epbew==", + "dev": true, + "license": "MIT", + "engines": { + "node": ">= 0.4" + }, + "funding": { + "url": "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/sponsors/ljharb" + } + }, + "node_modules/on-finished": { + "version": "2.4.1", + "resolved": "https://registry.npmjs.org/on-finished/-/on-finished-2.4.1.tgz", + "integrity": "sha512-oVlzkg3ENAhCk2zdv7IJwd/QUD4z2RxRwpkcGY8psCVcCYZNq4wYnVWALHM+brtuJjePWiYF/ClmuDr8Ch5+kg==", + "dev": true, + "license": "MIT", + "dependencies": { + "ee-first": "1.1.1" + }, + "engines": { + "node": ">= 0.8" + } + }, + "node_modules/once": { + "version": "1.4.0", + "resolved": "https://registry.npmjs.org/once/-/once-1.4.0.tgz", + "integrity": "sha512-lNaJgI+2Q5URQBkccEKHTQOPaXdUxnZZElQTZY0MFUAuaEqe1E+Nyvgdz/aIyNi6Z9MzO5dv1H8n58/GELp3+w==", + "dev": true, + "license": "ISC", + "dependencies": { + "wrappy": "1" + } + }, + "node_modules/parse5": { + "version": "7.3.0", + "resolved": "https://registry.npmjs.org/parse5/-/parse5-7.3.0.tgz", + "integrity": "sha512-IInvU7fabl34qmi9gY8XOVxhYyMyuH2xUNpb2q8/Y+7552KlejkRvqvD19nMoUW/uQGGbqNpA6Tufu5FL5BZgw==", + "dev": true, + "license": "MIT", + "dependencies": { + "entities": "^6.0.0" + }, + "funding": { + "url": "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/inikulin/parse5?sponsor=1" + } + }, + "node_modules/parse5-htmlparser2-tree-adapter": { + "version": "7.1.0", + "resolved": "https://registry.npmjs.org/parse5-htmlparser2-tree-adapter/-/parse5-htmlparser2-tree-adapter-7.1.0.tgz", + "integrity": "sha512-ruw5xyKs6lrpo9x9rCZqZZnIUntICjQAd0Wsmp396Ul9lN/h+ifgVV1x1gZHi8euej6wTfpqX8j+BFQxF0NS/g==", + "dev": true, + "license": "MIT", + "dependencies": { + "domhandler": "^5.0.3", + "parse5": "^7.0.0" + }, + "funding": { + "url": "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/inikulin/parse5?sponsor=1" + } + }, + "node_modules/parse5-parser-stream": { + "version": "7.1.2", + "resolved": "https://registry.npmjs.org/parse5-parser-stream/-/parse5-parser-stream-7.1.2.tgz", + "integrity": "sha512-JyeQc9iwFLn5TbvvqACIF/VXG6abODeB3Fwmv/TGdLk2LfbWkaySGY72at4+Ty7EkPZj854u4CrICqNk2qIbow==", + "dev": true, + "license": "MIT", + "dependencies": { + "parse5": "^7.0.0" + }, + "funding": { + "url": "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/inikulin/parse5?sponsor=1" + } + }, + "node_modules/parse5/node_modules/entities": { + "version": "6.0.1", + "resolved": "https://registry.npmjs.org/entities/-/entities-6.0.1.tgz", + "integrity": "sha512-aN97NXWF6AWBTahfVOIrB/NShkzi5H7F9r1s9mD3cDj4Ko5f2qhhVoYMibXF7GlLveb/D2ioWay8lxI97Ven3g==", + "dev": true, + "license": "BSD-2-Clause", + "engines": { + "node": ">=0.12" + }, + "funding": { + "url": "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/fb55/entities?sponsor=1" + } + }, + "node_modules/parseurl": { + "version": "1.3.3", + "resolved": "https://registry.npmjs.org/parseurl/-/parseurl-1.3.3.tgz", + "integrity": "sha512-CiyeOxFT/JZyN5m0z9PfXw4SCBJ6Sygz1Dpl0wqjlhDEGGBP1GnsUVEL0p63hoG1fcj3fHynXi9NYO4nWOL+qQ==", + "dev": true, + "license": "MIT", + "engines": { + "node": ">= 0.8" + } + }, + "node_modules/path-key": { + "version": "3.1.1", + "resolved": "https://registry.npmjs.org/path-key/-/path-key-3.1.1.tgz", + "integrity": "sha512-ojmeN0qd+y0jszEtoY48r0Peq5dwMEkIlCOu6Q5f41lfkswXuKtYrhgoTpLnyIcHm24Uhqx+5Tqm2InSwLhE6Q==", + "dev": true, + "license": "MIT", + "engines": { + "node": ">=8" + } + }, + "node_modules/path-to-regexp": { + "version": "8.4.2", + "resolved": "https://registry.npmjs.org/path-to-regexp/-/path-to-regexp-8.4.2.tgz", + "integrity": "sha512-qRcuIdP69NPm4qbACK+aDogI5CBDMi1jKe0ry5rSQJz8JVLsC7jV8XpiJjGRLLol3N+R5ihGYcrPLTno6pAdBA==", + "dev": true, + "license": "MIT", + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/express" + } + }, + "node_modules/pkce-challenge": { + "version": "5.0.1", + "resolved": "https://registry.npmjs.org/pkce-challenge/-/pkce-challenge-5.0.1.tgz", + "integrity": "sha512-wQ0b/W4Fr01qtpHlqSqspcj3EhBvimsdh0KlHhH8HRZnMsEa0ea2fTULOXOS9ccQr3om+GcGRk4e+isrZWV8qQ==", + "dev": true, + "license": "MIT", + "engines": { + "node": ">=16.20.0" + } + }, + "node_modules/proxy-addr": { + "version": "2.0.8", + "resolved": "https://registry.npmjs.org/proxy-addr/-/proxy-addr-2.0.8.tgz", + "integrity": "sha512-5nnx0yGyVUcY6t9RnWcARWtwT9F1D8O9rt08htPvnd49W1IgZtmLkhu9WfMzQj1cFxjHIO6connUNVW5k7AVyQ==", + "dev": true, + "license": "MIT", + "dependencies": { + "forwarded": "0.2.0", + "ipaddr.js": "1.9.1" + }, + "engines": { + "node": ">= 0.10" + }, + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/express" + } + }, + "node_modules/qs": { + "version": "6.16.0", + "resolved": "https://registry.npmjs.org/qs/-/qs-6.16.0.tgz", + "integrity": "sha512-h6fhOIaRrID2CbEY2fqs+7t+UXZo+MLAnU5gRIq85uFtdiUPCdsApMlHhXogKVM4HM2DVbIjGNTTYH2OcmP1vA==", + "dev": true, + "license": "BSD-3-Clause", + "dependencies": { + "es-define-property": "^1.0.1", + "side-channel": "^1.1.1" + }, + "engines": { + "node": ">=0.6" + }, + "funding": { + "url": "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/sponsors/ljharb" + } + }, + "node_modules/range-parser": { + "version": "1.3.0", + "resolved": "https://registry.npmjs.org/range-parser/-/range-parser-1.3.0.tgz", + "integrity": "sha512-hek2mFQpPuI4E1BBKrSto+BU3e3x4xuarsbiwr3+lf7p44juvFMV0XFWQAP3xUyqXA4RrXLIoaSUGbSt056ZMw==", + "dev": true, + "license": "MIT", + "engines": { + "node": ">= 0.6" + }, + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/express" + } + }, + "node_modules/raw-body": { + "version": "3.0.2", + "resolved": "https://registry.npmjs.org/raw-body/-/raw-body-3.0.2.tgz", + "integrity": "sha512-K5zQjDllxWkf7Z5xJdV0/B0WTNqx6vxG70zJE4N0kBs4LovmEYWJzQGxC9bS9RAKu3bgM40lrd5zoLJ12MQ5BA==", + "dev": true, + "license": "MIT", + "dependencies": { + "bytes": "~3.1.2", + "http-errors": "~2.0.1", + "iconv-lite": "~0.7.0", + "unpipe": "~1.0.0" + }, + "engines": { + "node": ">= 0.10" + } + }, + "node_modules/raw-body/node_modules/iconv-lite": { + "version": "0.7.3", + "resolved": "https://registry.npmjs.org/iconv-lite/-/iconv-lite-0.7.3.tgz", + "integrity": "sha512-IKXpvIzjnC9XTAUbVBcMfGS0EPaIXtW6v+zr+RRp+hqULEpo0owZax6wyRwPOJbWbzjYspQwusTsfVr0ifh4uQ==", + "dev": true, + "license": "MIT", + "dependencies": { + "safer-buffer": ">= 2.1.2 < 3.0.0" + }, + "engines": { + "node": ">=0.10.0" + }, + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/express" + } + }, + "node_modules/re2js": { + "version": "2.8.6", + "resolved": "https://registry.npmjs.org/re2js/-/re2js-2.8.6.tgz", + "integrity": "sha512-xLgQil4kIUCrAzVk9fRSkxkFNwmygLFjVxXrLc65aE1F0+Zsb8rxumFBy4XKyvgMCTL6kilDq3EZ0piE2dP/Dg==", + "dev": true, + "license": "MIT", + "engines": { + "node": ">=18.0.0" + } + }, + "node_modules/relayflows": { + "version": "2.0.29", + "resolved": "https://registry.npmjs.org/relayflows/-/relayflows-2.0.29.tgz", + "integrity": "sha512-rxfNfPTLZVqHjX6Ez3rUCBG8kOVKme3bYk/291bQPoxDAhKVDLdFNfBAyIxZ2m195yiKtUGF1PgtcwQ2uFEfig==", + "dev": true, + "license": "Apache-2.0", + "dependencies": { + "@relayflows/sdk": "2.0.29" + }, + "bin": { + "flows": "bin/flows.js" + }, + "engines": { + "node": ">=20" + }, + "optionalDependencies": { + "@relayflows/runtime-darwin-arm64": "2.0.29", + "@relayflows/runtime-linux-x64": "2.0.29" + } + }, + "node_modules/require-from-string": { + "version": "2.0.2", + "resolved": "https://registry.npmjs.org/require-from-string/-/require-from-string-2.0.2.tgz", + "integrity": "sha512-Xf0nWe6RseziFMu+Ap9biiUbmplq6S9/p+7w7YXP/JBHhrUDDUhwa+vANyubuqfZWTveU//DYVGsDG7RKL/vEw==", + "dev": true, + "license": "MIT", + "engines": { + "node": ">=0.10.0" + } + }, + "node_modules/router": { + "version": "2.2.0", + "resolved": "https://registry.npmjs.org/router/-/router-2.2.0.tgz", + "integrity": "sha512-nLTrUKm2UyiL7rlhapu/Zl45FwNgkZGaCpZbIHajDYgwlJCOzLSk+cIPAnsEqV955GjILJnKbdQC1nVPz+gAYQ==", + "dev": true, + "license": "MIT", + "dependencies": { + "debug": "^4.4.0", + "depd": "^2.0.0", + "is-promise": "^4.0.0", + "parseurl": "^1.3.3", + "path-to-regexp": "^8.0.0" + }, + "engines": { + "node": ">= 18" + } + }, + "node_modules/safer-buffer": { + "version": "2.1.2", + "resolved": "https://registry.npmjs.org/safer-buffer/-/safer-buffer-2.1.2.tgz", + "integrity": "sha512-YZo3K82SD7Riyi0E1EQPojLz7kpepnSQI9IyPbHHg1XXXevb5dJI7tpyN2ADxGcQbHG7vcyRHk0cbwqcQriUtg==", + "dev": true, + "license": "MIT" + }, + "node_modules/send": { + "version": "1.2.1", + "resolved": "https://registry.npmjs.org/send/-/send-1.2.1.tgz", + "integrity": "sha512-1gnZf7DFcoIcajTjTwjwuDjzuz4PPcY2StKPlsGAQ1+YH20IRVrBaXSWmdjowTJ6u8Rc01PoYOGHXfP1mYcZNQ==", + "dev": true, + "license": "MIT", + "dependencies": { + "debug": "^4.4.3", + "encodeurl": "^2.0.0", + "escape-html": "^1.0.3", + "etag": "^1.8.1", + "fresh": "^2.0.0", + "http-errors": "^2.0.1", + "mime-types": "^3.0.2", + "ms": "^2.1.3", + "on-finished": "^2.4.1", + "range-parser": "^1.2.1", + "statuses": "^2.0.2" + }, + "engines": { + "node": ">= 18" + }, + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/express" + } + }, + "node_modules/serve-static": { + "version": "2.2.1", + "resolved": "https://registry.npmjs.org/serve-static/-/serve-static-2.2.1.tgz", + "integrity": "sha512-xRXBn0pPqQTVQiC8wyQrKs2MOlX24zQ0POGaj0kultvoOCstBQM5yvOhAVSUwOMjQtTvsPWoNCHfPGwaaQJhTw==", + "dev": true, + "license": "MIT", + "dependencies": { + "encodeurl": "^2.0.0", + "escape-html": "^1.0.3", + "parseurl": "^1.3.3", + "send": "^1.2.0" + }, + "engines": { + "node": ">= 18" + }, + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/express" + } + }, + "node_modules/setprototypeof": { + "version": "1.2.0", + "resolved": "https://registry.npmjs.org/setprototypeof/-/setprototypeof-1.2.0.tgz", + "integrity": "sha512-E5LDX7Wrp85Kil5bhZv46j8jOeboKq5JMmYM3gVGdGH8xFpPWXUMsNrlODCrkoxMEeNi/XZIwuRvY4XNwYMJpw==", + "dev": true, + "license": "ISC" + }, + "node_modules/shebang-command": { + "version": "2.0.0", + "resolved": "https://registry.npmjs.org/shebang-command/-/shebang-command-2.0.0.tgz", + "integrity": "sha512-kHxr2zZpYtdmrN1qDjrrX/Z1rR1kG8Dx+gkpK1G4eXmvXswmcE1hTWBWYUzlraYw1/yZp6YuDY77YtvbN0dmDA==", + "dev": true, + "license": "MIT", + "dependencies": { + "shebang-regex": "^3.0.0" + }, + "engines": { + "node": ">=8" + } + }, + "node_modules/shebang-regex": { + "version": "3.0.0", + "resolved": "https://registry.npmjs.org/shebang-regex/-/shebang-regex-3.0.0.tgz", + "integrity": "sha512-7++dFhtcx3353uBaq8DDR4NuxBetBzC7ZQOhmTQInHEd6bSrXdiEyzCvG07Z44UYdLShWUyXt5M/yhz8ekcb1A==", + "dev": true, + "license": "MIT", + "engines": { + "node": ">=8" + } + }, + "node_modules/side-channel": { + "version": "1.1.1", + "resolved": "https://registry.npmjs.org/side-channel/-/side-channel-1.1.1.tgz", + "integrity": "sha512-6x6dK6zJdpTzF4sQeNYxwtvBzf6Eg4GtlesS94HOvTudUeyK2WXAaIfmDgsyslYrRBeFIlsi54AYsFGUuhmvrQ==", + "dev": true, + "license": "MIT", + "dependencies": { + "es-errors": "^1.3.0", + "object-inspect": "^1.13.4", + "side-channel-list": "^1.0.1", + "side-channel-map": "^1.0.1", + "side-channel-weakmap": "^1.0.2" + }, + "engines": { + "node": ">= 0.4" + }, + "funding": { + "url": "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/sponsors/ljharb" + } + }, + "node_modules/side-channel-list": { + "version": "1.0.1", + "resolved": "https://registry.npmjs.org/side-channel-list/-/side-channel-list-1.0.1.tgz", + "integrity": "sha512-mjn/0bi/oUURjc5Xl7IaWi/OJJJumuoJFQJfDDyO46+hBWsfaVM65TBHq2eoZBhzl9EchxOijpkbRC8SVBQU0w==", + "dev": true, + "license": "MIT", + "dependencies": { + "es-errors": "^1.3.0", + "object-inspect": "^1.13.4" + }, + "engines": { + "node": ">= 0.4" + }, + "funding": { + "url": "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/sponsors/ljharb" + } + }, + "node_modules/side-channel-map": { + "version": "1.0.1", + "resolved": "https://registry.npmjs.org/side-channel-map/-/side-channel-map-1.0.1.tgz", + "integrity": "sha512-VCjCNfgMsby3tTdo02nbjtM/ewra6jPHmpThenkTYh8pG9ucZ/1P8So4u4FGBek/BjpOVsDCMoLA/iuBKIFXRA==", + "dev": true, + "license": "MIT", + "dependencies": { + "call-bound": "^1.0.2", + "es-errors": "^1.3.0", + "get-intrinsic": "^1.2.5", + "object-inspect": "^1.13.3" + }, + "engines": { + "node": ">= 0.4" + }, + "funding": { + "url": "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/sponsors/ljharb" + } + }, + "node_modules/side-channel-weakmap": { + "version": "1.0.2", + "resolved": "https://registry.npmjs.org/side-channel-weakmap/-/side-channel-weakmap-1.0.2.tgz", + "integrity": "sha512-WPS/HvHQTYnHisLo9McqBHOJk2FkHO/tlpvldyrnem4aeQp4hai3gythswg6p01oSoTl58rcpiFAjF2br2Ak2A==", + "dev": true, + "license": "MIT", + "dependencies": { + "call-bound": "^1.0.2", + "es-errors": "^1.3.0", + "get-intrinsic": "^1.2.5", + "object-inspect": "^1.13.3", + "side-channel-map": "^1.0.1" + }, + "engines": { + "node": ">= 0.4" + }, + "funding": { + "url": "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/sponsors/ljharb" + } + }, + "node_modules/sql.js": { + "version": "1.14.2", + "resolved": "https://registry.npmjs.org/sql.js/-/sql.js-1.14.2.tgz", + "integrity": "sha512-3ZGPovObMFrdw79zrUHbfdE/DLIsy8jdNdssmMSQuRAymedU6q84asPt0kgiqrdMYlPegDItiIMfmIXzZnYFcw==", + "dev": true, + "license": "MIT" + }, + "node_modules/statuses": { + "version": "2.0.2", + "resolved": "https://registry.npmjs.org/statuses/-/statuses-2.0.2.tgz", + "integrity": "sha512-DvEy55V3DB7uknRo+4iOGT5fP1slR8wQohVdknigZPMpMstaKJQWhwiYBACJE3Ul2pTnATihhBYnRhZQHGBiRw==", + "dev": true, + "license": "MIT", + "engines": { + "node": ">= 0.8" + } + }, + "node_modules/tar": { + "version": "7.5.22", + "resolved": "https://registry.npmjs.org/tar/-/tar-7.5.22.tgz", + "integrity": "sha512-MFO/QzvtAOmJbkhOaCTvbGcFN9L9b+JunIsDwaKljSOdcLMea3NJ1k9Usz/rjdfSXTq4dfzfeS7W4p4YOAAHeA==", + "dev": true, + "license": "BlueOak-1.0.0", + "peer": true, + "dependencies": { + "@isaacs/fs-minipass": "^4.0.0", + "chownr": "^3.0.0", + "minipass": "^7.1.2", + "minizlib": "^3.1.0", + "yallist": "^5.0.0" + }, + "engines": { + "node": ">=18" + } + }, + "node_modules/toidentifier": { + "version": "1.0.1", + "resolved": "https://registry.npmjs.org/toidentifier/-/toidentifier-1.0.1.tgz", + "integrity": "sha512-o5sSPKEkg/DIQNmH43V0/uerLrpzVedkUh8tGNvaeXpfpuwjKenlSox/2O/BTlZUtEe+JG7s5YhEz608PlAHRA==", + "dev": true, + "license": "MIT", + "engines": { + "node": ">=0.6" + } + }, + "node_modules/type-is": { + "version": "2.1.0", + "resolved": "https://registry.npmjs.org/type-is/-/type-is-2.1.0.tgz", + "integrity": "sha512-faYHw0anBbc/kWF3zFTEnxSFOAGUX9GFbOBthvDdLsIlEoWOFOtS0zgCiQYwIskL9iGXZL3kAXD8OoZ4GmMATA==", + "dev": true, + "license": "MIT", + "dependencies": { + "content-type": "^2.0.0", + "media-typer": "^1.1.0", + "mime-types": "^3.0.0" + }, + "engines": { + "node": ">= 18" + }, + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/express" + } + }, + "node_modules/type-is/node_modules/content-type": { + "version": "2.1.0", + "resolved": "https://registry.npmjs.org/content-type/-/content-type-2.1.0.tgz", + "integrity": "sha512-mj7UPXE0jaqaOsukNZRUEfEi2AcL7C/vwmwcHV0O97eO1E1pxBZuyjlZrx5seTaNBg1U6+o35wpa35Qfcc+7ag==", + "dev": true, + "license": "MIT", + "engines": { + "node": ">=18" + }, + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/express" + } + }, + "node_modules/undici": { + "version": "7.29.1", + "resolved": "https://registry.npmjs.org/undici/-/undici-7.29.1.tgz", + "integrity": "sha512-RYONW2MeafgYlkVOKYKkA/Ag7BmXqgIWCa8t1m0JcxrQg9pI9lEqRhAOruOBCbAohOa/gkCF+iPi9hrgvTzu6Q==", + "dev": true, + "license": "MIT", + "engines": { + "node": ">=20.18.1" + } + }, + "node_modules/unpipe": { + "version": "1.0.0", + "resolved": "https://registry.npmjs.org/unpipe/-/unpipe-1.0.0.tgz", + "integrity": "sha512-pjy2bYhSsufwWlKwPc+l3cN7+wuJlK6uz0YdJEOlQDbl6jo/YlPi4mb8agUkVC8BF7V8NuzeyPNqRksA3hztKQ==", + "dev": true, + "license": "MIT", + "engines": { + "node": ">= 0.8" + } + }, + "node_modules/vary": { + "version": "1.1.2", + "resolved": "https://registry.npmjs.org/vary/-/vary-1.1.2.tgz", + "integrity": "sha512-BNGbWLfd0eUPabhkXUVm0j8uuvREyTh5ovRa/dyow/BqAbZJyC+5fU+IzQOzmAKzYqYRAISoRhdQr3eIZ/PXqg==", + "dev": true, + "license": "MIT", + "engines": { + "node": ">= 0.8" + } + }, + "node_modules/whatwg-encoding": { + "version": "3.1.1", + "resolved": "https://registry.npmjs.org/whatwg-encoding/-/whatwg-encoding-3.1.1.tgz", + "integrity": "sha512-6qN4hJdMwfYBtE3YBTTHhoeuUrDBPZmbQaxWAqSALV/MeEnR5z1xd8UKud2RAkFoPkmB+hli1TZSnyi84xz1vQ==", + "deprecated": "Use @exodus/bytes instead for a more spec-conformant and faster implementation", + "dev": true, + "license": "MIT", + "dependencies": { + "iconv-lite": "0.6.3" + }, + "engines": { + "node": ">=18" + } + }, + "node_modules/whatwg-mimetype": { + "version": "4.0.0", + "resolved": "https://registry.npmjs.org/whatwg-mimetype/-/whatwg-mimetype-4.0.0.tgz", + "integrity": "sha512-QaKxh0eNIi2mE9p2vEdzfagOKHCcj1pJ56EEHGQOVxp8r9/iszLUUV7v89x9O1p/T+NlTM5W7jW6+cz4Fq1YVg==", + "dev": true, + "license": "MIT", + "engines": { + "node": ">=18" + } + }, + "node_modules/which": { + "version": "2.0.2", + "resolved": "https://registry.npmjs.org/which/-/which-2.0.2.tgz", + "integrity": "sha512-BLI3Tl1TW3Pvl70l3yq3Y64i+awpwXqsGBYWkkqMtnbXgrMD+yj7rhW0kuEDxzJaYXGjEW5ogapKNMEKNMjibA==", + "dev": true, + "license": "ISC", + "dependencies": { + "isexe": "^2.0.0" + }, + "bin": { + "node-which": "bin/node-which" + }, + "engines": { + "node": ">= 8" + } + }, + "node_modules/wrappy": { + "version": "1.0.2", + "resolved": "https://registry.npmjs.org/wrappy/-/wrappy-1.0.2.tgz", + "integrity": "sha512-l4Sp/DRseor9wL6EvV2+TuQn63dMkPjZ/sp9XkghTEbV9KlPS1xUsZ3u7/IQO4wxtcFB4bgpQPRcR3QCvezPcQ==", + "dev": true, + "license": "ISC" + }, + "node_modules/yallist": { + "version": "5.0.0", + "resolved": "https://registry.npmjs.org/yallist/-/yallist-5.0.0.tgz", + "integrity": "sha512-YgvUTfwqyc7UXVMrB+SImsVYSmTS8X/tSrtdNZMImM+n7+QTriRXyXim0mBrTXNeqzVF0KWGgHPeiyViFFrNDw==", + "dev": true, + "license": "BlueOak-1.0.0", + "peer": true, + "engines": { + "node": ">=18" + } + }, + "node_modules/yaml": { + "version": "2.9.1", + "resolved": "https://registry.npmjs.org/yaml/-/yaml-2.9.1.tgz", + "integrity": "sha512-3NxN8+78OdzbT7C/WjGsyfPAtJaN3FNDsWxv7Y7mcDsT/oOmgW8BpyQQFFBnvZE3j9Y2Sdz1ULFLezL7Eb2yFw==", + "dev": true, + "license": "ISC", + "bin": { + "yaml": "bin.mjs" + }, + "engines": { + "node": ">= 14.6" + }, + "funding": { + "url": "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/sponsors/eemeli" + } + }, + "node_modules/zod": { + "version": "4.6.5", + "resolved": "https://registry.npmjs.org/zod/-/zod-4.6.5.tgz", + "integrity": "sha512-v5l/aFXZQeai4awLbOpSoHecE9UiMrnfx75tEXLjNonXVARxQ5mOeipTjROUchszUNCqnE+hqAMujRsRHsut2Q==", + "dev": true, + "license": "MIT", + "funding": { + "url": "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/sponsors/colinhacks" + } + }, + "node_modules/zod-to-json-schema": { + "version": "3.25.2", + "resolved": "https://registry.npmjs.org/zod-to-json-schema/-/zod-to-json-schema-3.25.2.tgz", + "integrity": "sha512-O/PgfnpT1xKSDeQYSCfRI5Gy3hPf91mKVDuYLUHZJMiDFptvP41MSnWofm8dnCm0256ZNfZIM7DSzuSMAFnjHA==", + "dev": true, + "license": "ISC", + "peerDependencies": { + "zod": "^3.25.28 || ^4" + } + } + } +} diff --git a/examples/prompt-lab/package.json b/examples/prompt-lab/package.json new file mode 100644 index 00000000..5ecb3d64 --- /dev/null +++ b/examples/prompt-lab/package.json @@ -0,0 +1,13 @@ +{ + "name": "@relayflows/prompt-lab-flow", + "private": true, + "type": "module", + "scripts": { + "test": "node --experimental-strip-types --test tests/*.test.ts", + "typecheck": "../../packages/sdk/node_modules/.bin/tsc -p tsconfig.json" + }, + "devDependencies": { + "@relayflows/surface": "2.0.29", + "relayflows": "2.0.29" + } +} diff --git a/examples/prompt-lab/prompt-lab.flow.ts b/examples/prompt-lab/prompt-lab.flow.ts new file mode 100644 index 00000000..423668d3 --- /dev/null +++ b/examples/prompt-lab/prompt-lab.flow.ts @@ -0,0 +1,40 @@ +// Prompt Lab — the inbox and workbench for writing and fixing the prompts that +// draft home-health charts, as one relayflow (docs/SURFACE.md dialect). +// The product brief is the spec; each job file maps its flow diagram line by +// line: "System" boxes are deterministic steps or model calls, "You" boxes are +// `f.human` gates asked of `input.reviewer`, "Outcome" boxes are lab writes. +// +// job "new-agency" Job 1: stand up an agency's first-pass prompts +// job "fix" Job 2: fix one shared or agency prompt from an issue or the picker +// job "patient" grow the locked shelf from a gap brief on the manager queue +// +// Apricot is stood in for by a local lab directory (see store.ts): bank.json is +// its Bank, and the chart-filling engine is an `llm` step given the live prompt. +// +// flows run prompt-lab.flow.ts --local-agent \ +// --input '{"job":"new-agency","reviewer":"","lab":"","agency":"sunrise","visitType":"soc"}' +import { flow } from "@relayflows/surface"; +import { lab } from "./lib/lab.ts"; +import { fix, type FixInput } from "./jobs/fix.ts"; +import { newAgency, type NewAgencyInput } from "./jobs/new-agency.ts"; +import { createPatient, type PatientInput } from "./jobs/patient.ts"; + +type Input = { reviewer: string; lab: string } & ( + | ({ job: "new-agency" } & NewAgencyInput) + | ({ job: "fix" } & FixInput) + | ({ job: "patient" } & PatientInput) +); + +export default flow("prompt-lab", async (f, input) => { + // Every "You" gate is asked of the reviewer the caller names; there is no default person. + if (typeof input?.reviewer !== "string" || !input.reviewer.trim() || typeof input.lab !== "string" || !input.lab.trim()) { + await f.run("echo 'Refused: input needs reviewer (who answers the gates) and lab (the lab directory).' >&2"); + return f.done("declined"); + } + const job = { f, lab: lab(f, input.lab), labDir: input.lab, reviewer: input.reviewer }; + if (input.job === "new-agency") return newAgency(job, input); + if (input.job === "fix") return fix(job, input); + if (input.job === "patient") return createPatient(job, input); + await f.run(`echo 'Refused: job must be new-agency, fix or patient.' >&2`); + return f.done("declined"); +}); diff --git a/examples/prompt-lab/prompts.ts b/examples/prompt-lab/prompts.ts new file mode 100644 index 00000000..059a20aa --- /dev/null +++ b/examples/prompt-lab/prompts.ts @@ -0,0 +1,170 @@ +// Every model call in Prompt Lab: the prompt it sends and the JSON Schema its +// answer must satisfy before the journal accepts it. Schemas are data, so the +// engine's answer enum IS the agency menu — on-menu is plumbing, not judgment. +import { CONFIDENCES, type Output, type Patient } from "./lib/types.ts"; + +export interface Ask { question: string; options: readonly string[] } +export interface Call { prompt: string; output: Record } + +const str = { type: "string", minLength: 1 }; +const obj = (properties: Record) => + ({ type: "object", additionalProperties: false, required: Object.keys(properties), properties }); +const qaSchema = obj({ pass: { type: "boolean" }, findings: { type: "array", items: str } }); +const promptSchema = obj({ prompt: { type: "string", minLength: 40 } }); +const chart = (p: Patient): string => `REFERRAL PACKET:\n${p.referral}\n\nTODAY'S VISIT NOTES:\n${p.notes}`; +const briefBlock = (brief?: string): string => (brief ? `QUESTION BRIEF (success criteria):\n${brief}` : "QUESTION BRIEF: none (that is allowed)."); + +/** Apricot's chart-filling engine: the same prompt nurses' drafts come from. */ +export function engine(promptText: string, ask: Ask, patient: Patient): Call { + return { + prompt: `You are the chart-filling engine for a home-health visit. Follow the INSTRUCTIONS exactly and nothing else. + +INSTRUCTIONS: +${promptText} + +QUESTION: ${ask.question} +OPTIONS (answer with exactly one): ${ask.options.join(" | ")} + +${chart(patient)} + +Return JSON: answer (one of the options), confidence (High, Medium or Low), explanation (one or two sentences citing the chart).`, + output: obj({ answer: { enum: [...ask.options] }, confidence: { enum: [...CONFIDENCES] }, explanation: str }), + }; +} + +/** First-pass prompt text: a v0 from question, options and type (and brief when one exists). */ +export function firstPass(ask: Ask, guidelines: string, brief: string | undefined, findings: readonly string[]): Call { + return { + prompt: `Write the instruction text ("prompt") that tells a chart-filling engine how to answer one home-health assessment question from a visit chart. The engine is separately given the question, the options and the chart; your text is the reasoning rules. + +QUESTION: ${ask.question} +OPTIONS: ${ask.options.join(" | ")} +TYPE: single choice + +${briefBlock(brief)} + +${guidelines} +${findings.length ? `\nA reviewer rejected the previous draft for:\n- ${findings.join("\n- ")}\nFix every point.` : ""} +Return JSON { "prompt": "" }.`, + output: promptSchema, + }; +} + +/** Prompt QA: compliance with the brief (if any) plus the shared guidelines. Never "live" by itself. */ +export function promptQa(promptText: string, ask: Ask, guidelines: string, brief: string | undefined): Call { + return { + prompt: `You are Prompt QA. Check this question prompt for compliance only: does it satisfy every shared guideline, and the question brief if there is one? Do not grade style. + +QUESTION: ${ask.question} +OPTIONS: ${ask.options.join(" | ")} + +PROMPT UNDER REVIEW: +${promptText} + +${briefBlock(brief)} + +${guidelines} + +Return JSON { "pass": true|false, "findings": [""] }. pass is true only when findings is empty.`, + output: qaSchema, + }; +} + +export interface Change { patient: Patient; ai: Output; target: Output; notes: string } + +/** The iterator: a rewrite function. Changeset + patients + brief + existing prompt in; new prompt text out. */ +export function iterate(ask: Ask, existing: string, brief: string | undefined, changes: readonly Change[], findings: readonly string[]): Call { + const rows = changes.map((c) => `--- ${c.patient.label} +${chart(c.patient)} +ENGINE SAID: ${c.ai.answer} / ${c.ai.confidence} / ${c.ai.explanation} +REVIEWER SAYS: ${c.target.answer} / ${c.target.confidence} / ${c.target.explanation}${c.notes ? `\nREVIEWER NOTES: ${c.notes}` : ""}`).join("\n\n"); + return { + prompt: `Rewrite a home-health question prompt so the chart-filling engine produces the reviewer's answers on these charts. Change only the prompt text; the question brief and the reviewer's answers are fixed. Keep what already works; do not mention these specific patients. + +QUESTION: ${ask.question} +OPTIONS: ${ask.options.join(" | ")} + +EXISTING PROMPT: +${existing} + +${briefBlock(brief)} + +CHANGESET (every row the reviewer changed): +${rows} +${findings.length ? `\nPrompt QA rejected the previous rewrite for:\n- ${findings.join("\n- ")}\nFix every point.` : ""} +Return JSON { "prompt": "" }.`, + output: promptSchema, + }; +} + +/** Test planner: pick existing shelf patients so each question's paths can fire; queue briefs for holes. */ +export function planTests(asks: readonly (Ask & { id: string })[], shelf: readonly Patient[]): Call { + const patients = shelf.map((p) => `[${p.id}] ${p.label}, ${p.ageBand}\n${chart(p)}`).join("\n\n"); + const questions = asks.map((a) => `[${a.id}] ${a.question} Options: ${a.options.join(" | ")}`).join("\n"); + return { + prompt: `You are the test planner for a home-health prompt workbench. For each question, pick the shelf patients whose charts contain real evidence for that question, so its answer paths get exercised. Never invent a patient. If no shelf patient has evidence for a question, list it as a gap with a short brief describing the fake patient the shelf needs (characteristics only, no identifiers). + +QUESTIONS: +${questions} + +SHELF PATIENTS: +${patients} + +Return JSON { "coverage": [{ "questionId": "...", "patientIds": ["..."] }], "gaps": [{ "questionId": "...", "brief": "..." }] }. Every question appears exactly once, in coverage (non-empty patientIds) or in gaps.`, + output: obj({ + coverage: { type: "array", items: obj({ questionId: str, patientIds: { type: "array", minItems: 1, items: str } }) }, + gaps: { type: "array", items: obj({ questionId: str, brief: str }) }, + }), + }; +} + +/** Test patient creator, step 1: the agent writes a patient plan from the brief. */ +export function patientPlan(brief: string, question: string): Call { + return { + prompt: `Plan an invented home-health test patient for a prompt workbench shelf. It must look like a real visit (referral packet + today's visit notes) and must exercise this question: "${question}". + +GAP BRIEF: +${brief} + +Describe: age band, diagnoses, living situation, what the referral packet says, what today's visit notes must show, and any deliberate conflict between referral and notes that tests the question. Invented only: no names beyond a first-name label, no dates, no MRN, no phone numbers, no addresses. + +Return JSON { "plan": "" }.`, + output: obj({ plan: { type: "string", minLength: 80 } }), + }; +} + +/** Test patient creator, step 2: generate the chart from brief + plan. */ +export function patientChart(brief: string, plan: string, findings: readonly string[]): Call { + return { + prompt: `Generate the invented home-health test patient described by this plan, in the shelf's chart shape. + +GAP BRIEF: +${brief} + +PATIENT PLAN: +${plan} +${findings.length ? `\nPatient QA rejected the previous chart for:\n- ${findings.join("\n- ")}\nFix every point.` : ""} +Rules: a first-name label only; no surnames, dates, MRNs, phone numbers or addresses. id is the lowercase label. +Return JSON { "id", "label", "ageBand", "visitType": "soc", "referral", "notes" }.`, + output: obj({ id: { type: "string", pattern: "^[a-z][a-z0-9-]{1,30}$" }, label: str, ageBand: str, visitType: { const: "soc" }, referral: { type: "string", minLength: 60 }, notes: { type: "string", minLength: 80 } }), + }; +} + +/** Patient QA: the chart against brief and plan. Loops until pass; then the patient locks. */ +export function patientQa(brief: string, plan: string, patient: Omit): Call { + return { + prompt: `You are Patient QA for an invented test patient. Check the chart against the gap brief and the plan: does it look like a real home-health visit, does it contain the evidence the brief needs, does it follow the plan, and is it free of identifiers (surnames, dates, MRN, phone, address)? + +GAP BRIEF: +${brief} + +PLAN: +${plan} + +CHART: +${JSON.stringify(patient, null, 2)} + +Return JSON { "pass": true|false, "findings": ["..."] }. pass is true only when findings is empty.`, + output: qaSchema, + }; +} diff --git a/examples/prompt-lab/prove.sh b/examples/prompt-lab/prove.sh new file mode 100755 index 00000000..612d8203 --- /dev/null +++ b/examples/prompt-lab/prove.sh @@ -0,0 +1,84 @@ +#!/usr/bin/env bash +# End-to-end proof of the Prompt Lab relayflow, locally, with real model calls. +# +# Seeds a fresh lab from fixtures/, then drives all three jobs through the real +# kernel with `flows run --local-agent`. At every parked `f.human` gate this +# script plays the reviewer named in the input: it applies the documented edit +# (captured as a diff), answers with `flows answer`, and `flows resume`s. +# Every command is captured with its literal output and exit code. +# +# ./prove.sh [out-dir] default: evidence/run +set -uo pipefail +cd "$(dirname "$0")" +OUT=${1:-evidence/run} +LAB=$OUT/lab +DD=${FLOWS_DATA_DIR:-$(mktemp -d)} +REVIEWER=prompt-lab-reviewer +N=0 + +[ -e "$LAB" ] && { echo "refusing: $LAB exists" >&2; exit 2; } +mkdir -p "$OUT" +node --no-warnings --experimental-strip-types store.ts "$LAB" seed fixtures + +# capture : run it, write "$ cmd", its output and exit code to $OUT/NN-name.txt +capture() { + local name=$1; shift + N=$((N + 1)) + FILE=$(printf '%s/%02d-%s.txt' "$OUT" "$N" "$name") + { echo "\$ $*"; "$@" 2>&1; echo "exit=$?"; } > "$FILE" + grep -v 'WAITING\|↻\|○' "$FILE" | tail -4 | cut -c1-240 +} +run_id() { grep -o 'RUN [0-9A-Z]\{26\}' "$FILE" | tail -1 | cut -d' ' -f2; } +wait_id() { grep -o ' human-[0-9]* yes|no' "$FILE" | tail -1 | awk '{print $1}'; } +parked_on() { grep -q "PARKED.*$1" "$FILE"; } +# edit : the reviewer's change, captured as a diff +edit() { + local file=$1 expr=$2 before + before=$(mktemp) + cp "$file" "$before" + node -e "const fs=require('fs');const v=JSON.parse(fs.readFileSync('$file','utf8'));$expr;fs.writeFileSync('$file',JSON.stringify(v,null,2)+'\n')" + N=$((N + 1)) + { echo "# reviewer edit: $file"; diff -u "$before" "$file" | tail -n +3; } > "$(printf '%s/%02d-reviewer-edit.diff' "$OUT" "$N")" +} +gate_path() { grep -o "$LAB/work/[^ ]*\.json" "$FILE" | head -1; } +answer_and_resume() { + local run wait + run=$(run_id); wait=$(wait_id) + capture answer npx flows answer --data-dir "$DD" "$run" "$wait" yes --by "$REVIEWER" + capture resume npx flows resume --no-observer-link --data-dir "$DD" --local-agent "$run" +} +flow() { capture "$1" npx flows run --no-observer-link --data-dir "$DD" --local-agent prompt-lab.flow.ts --input "$2"; } + +echo "== Job 1: new agency sunrise / soc" +flow job1-run "{\"job\":\"new-agency\",\"reviewer\":\"$REVIEWER\",\"lab\":\"$LAB\",\"agency\":\"sunrise\",\"visitType\":\"soc\"}" +parked_on "Config workbench" || { echo "job 1 did not reach the grid gate" >&2; exit 1; } +# The reviewer's first pass: Pat's wound is open today, whatever the discharge summary said. +edit "$(gate_path)" 'const r=v.find(r=>r.questionId==="wound-status"&&r.patientId==="pat");r.target={answer:"Ongoing",confidence:"High",explanation:"Today'"'"'s visit notes describe an open 2.0 x 1.5 cm left heel wound with serous drainage and slough; the discharge summary'"'"'s closed status is outdated."};r.notes="The prompt makes the referral win over what the nurse saw today. Today'"'"'s notes must win (shared guideline 1)."' +answer_and_resume +parked_on "Commit output" || { echo "job 1 did not reach the commit gate" >&2; exit 1; } +# ostomy-supplies has no shelf patient yet, so its first-pass prompt was never run: hold it back. +edit "$(gate_path)" 'v.mode="except";v.questions=["ostomy-supplies"]' +answer_and_resume + +echo "== Test patient creator: the ostomy gap brief Job 1 queued" +flow patient-run "{\"job\":\"patient\",\"reviewer\":\"$REVIEWER\",\"lab\":\"$LAB\",\"briefId\":\"gap-ostomy-supplies\"}" +parked_on "Test patient creator" || { echo "patient job did not reach kick generate" >&2; exit 1; } +answer_and_resume # kick generate; Patient QA locks it, nobody approves the chart + +echo "== Job 2: fix wound-status from the config-send issue" +ISSUE=$(node -e "const q=require('./$LAB/queue/issues.json');console.log(q.find(i=>i.questionIds[0]==='wound-status').id)") +flow job2-run "{\"job\":\"fix\",\"reviewer\":\"$REVIEWER\",\"lab\":\"$LAB\",\"issueId\":\"$ISSUE\"}" +parked_on "Question workbench" || { echo "job 2 did not reach the gold gate" >&2; exit 1; } +answer_and_resume # gold as prefilled: Pat's config-level target travelled with the issue +parked_on "Re-run and score" || { echo "job 2 did not reach done" >&2; exit 1; } +answer_and_resume # mark done = live, for every agency using it + +echo "== Final lab state" +capture lab-state node -e " +const b=require('./$LAB/bank.json'); +for (const [q,v] of Object.entries(b.questions)) console.log(q.padEnd(17),'live',String(v.livePromptId).padEnd(30),'draft',v.draftPromptId??'-'); +console.log('shelf', require('fs').readdirSync('$LAB/shelf').join(' ')); +console.log('issues', JSON.stringify(require('./$LAB/queue/issues.json').map(i=>[i.id,i.status]))); +console.log('patient briefs', JSON.stringify(require('./$LAB/queue/patient-briefs.json').map(i=>[i.id,i.status]))); +console.log('gold', Object.keys(require('./$LAB/gold.json')).join(' '));" +echo "data dir: $DD" diff --git a/examples/prompt-lab/store.ts b/examples/prompt-lab/store.ts new file mode 100644 index 00000000..a52ed669 --- /dev/null +++ b/examples/prompt-lab/store.ts @@ -0,0 +1,157 @@ +// The lab store: the one writer of a Prompt Lab state directory. It stands in +// for Apricot's Bank (bank.json: global prompts, livePromptId) plus Prompt +// Lab's own records (briefs, targets, gold, queues, work files). +// +// Flow steps call it through `f.run`, so every read and write is a journaled +// deterministic step. Every verb is idempotent: a retried step converges on +// the same state and prints the same result. Values arrive base64-encoded JSON +// so no user text is ever parsed by a shell. +// +// node store.ts seed +// node store.ts snapshot | read +// node store.ts write-new (never clobbers a reviewer's edit) +// node store.ts record +// node store.ts draft +// node store.ts publish +// node store.ts enqueue +// node store.ts close-issue +// node store.ts lock-patient [briefId] +import { createHash } from "node:crypto"; +import { cpSync, existsSync, mkdirSync, readdirSync, readFileSync, renameSync, writeFileSync } from "node:fs"; +import { dirname, join } from "node:path"; +import type { Bank, Issue, Patient, PatientBrief, Snapshot } from "./lib/types.ts"; + +const OUTPUT_LIMIT = 60 * 1024; // the kernel journals a 64 KiB stdout tail; never let a read be cut. + +const [lab, verb, ...args] = process.argv.slice(2); +if (!lab || !verb) fail("usage: store.ts [args]"); + +function fail(message: string): never { + process.stderr.write(`store: ${message}\n`); + process.exit(1); +} +function print(value: unknown): void { + const text = JSON.stringify(value); + if (text.length > OUTPUT_LIMIT) fail(`output is ${text.length} bytes, over the ${OUTPUT_LIMIT}-byte journal tail`); + process.stdout.write(`${text}\n`); +} +const path = (rel: string): string => { + if (rel.startsWith("/") || rel.split("/").includes("..")) fail(`path escapes the lab: ${rel}`); + return join(lab!, rel); +}; +const decode = (b64: string | undefined): unknown => { + if (!b64) fail("missing value"); + return JSON.parse(Buffer.from(b64, "base64").toString("utf8")); +}; +const readJson = (rel: string, fallback: T): T => + existsSync(path(rel)) ? (JSON.parse(readFileSync(path(rel), "utf8")) as T) : fallback; +/** Atomic: a crash mid-write leaves the old file, never half a file. */ +function writeJson(rel: string, value: unknown): void { + const file = path(rel); + mkdirSync(dirname(file), { recursive: true }); + writeFileSync(`${file}.tmp`, `${JSON.stringify(value, null, 2)}\n`); + renameSync(`${file}.tmp`, file); +} +const listDir = (rel: string): string[] => (existsSync(path(rel)) ? readdirSync(path(rel)).sort() : []); + +switch (verb) { + case "seed": { + if (existsSync(join(lab, "bank.json"))) fail(`${lab} is already a lab; refusing to overwrite it`); + cpSync(args[0] ?? fail("seed needs a fixtures dir"), lab, { recursive: true }); + print({ seeded: lab }); + break; + } + case "snapshot": { + const snapshot: Snapshot = { + bank: readJson("bank.json", { prompts: {}, questions: {} }), + agencies: readJson("agencies.json", {}), + guidelines: readFileSync(path("guidelines.md"), "utf8"), + shelf: listDir("shelf").map((f) => readJson(`shelf/${f}`, null as never)), + briefs: Object.fromEntries(listDir("briefs").map((f) => [f.replace(/\.md$/, ""), readFileSync(path(`briefs/${f}`), "utf8")])), + targets: readJson("targets.json", {}), + gold: readJson("gold.json", {}), + issues: readJson("queue/issues.json", []), + patientBriefs: readJson("queue/patient-briefs.json", []), + }; + print(snapshot); + break; + } + case "read": + print(readJson(args[0] ?? fail("read needs a path"), null)); + break; + case "write-new": { + const rel = args[0] ?? fail("write-new needs a path"); + const written = !existsSync(path(rel)); + if (written) writeJson(rel, decode(args[1])); + print({ path: rel, written }); + break; + } + case "record": { + const store = args[0]; + if (store !== "targets" && store !== "gold") fail("record takes targets or gold"); + const entries = decode(args[1]) as Record; + writeJson(`${store}.json`, { ...readJson(`${store}.json`, {}), ...entries }); + print({ store, recorded: Object.keys(entries).sort() }); + break; + } + case "draft": { + const questionId = args[0] ?? fail("draft needs a question id"); + const text = decode(args[1]) as string; + const bank = readJson("bank.json", { prompts: {}, questions: {} }); + const question = bank.questions[questionId] ?? fail(`no question ${questionId}`); + const promptId = `p-${questionId}-${createHash("sha256").update(text).digest("hex").slice(0, 8)}`; + bank.prompts[promptId] = text; + question.draftPromptId = promptId; + writeJson("bank.json", bank); + print({ questionId, promptId }); + break; + } + case "publish": { + const [questionId, promptId] = args; + const bank = readJson("bank.json", { prompts: {}, questions: {} }); + const question = bank.questions[questionId ?? ""] ?? fail(`no question ${questionId}`); + if (!promptId || !bank.prompts[promptId]) fail(`no prompt ${promptId}`); + const previous = question.livePromptId; + question.livePromptId = promptId; + if (question.draftPromptId === promptId) question.draftPromptId = null; + writeJson("bank.json", bank); + print({ questionId, livePromptId: promptId, previous: previous === promptId ? null : previous }); + break; + } + case "enqueue": { + const rel = args[0] === "issues" ? "queue/issues.json" : args[0] === "patient-briefs" ? "queue/patient-briefs.json" : fail("enqueue takes issues or patient-briefs"); + const item = decode(args[1]) as { id: string }; + const queue = readJson<{ id: string }[]>(rel, []); + const added = !queue.some((q) => q.id === item.id); + if (added) writeJson(rel, [...queue, item]); + print({ queue: args[0], id: item.id, added }); + break; + } + case "close-issue": { + const issues = readJson("queue/issues.json", []); + const issue = issues.find((i) => i.id === args[0]) ?? fail(`no issue ${args[0]}`); + issue.status = "done"; + writeJson("queue/issues.json", issues); + print({ issue: issue.id, status: issue.status }); + break; + } + case "lock-patient": { + const patient: Patient = { ...(decode(args[0]) as Patient), locked: true, source: "invented" }; + const briefId = args[1]; + if (!/^[a-z0-9][a-z0-9-]{0,39}$/.test(patient.id)) fail(`bad patient id ${patient.id}`); + const rel = `shelf/${patient.id}.json`; + // Locked patients are frozen: a retry may re-lock the same chart, never replace one. + const existing = readJson(rel, null); + if (existing && JSON.stringify(existing) !== JSON.stringify(patient)) fail(`shelf already has a different ${patient.id}`); + if (!existing) writeJson(rel, patient); + if (briefId) { + const briefs = readJson("queue/patient-briefs.json", []); + for (const b of briefs) if (b.id === briefId) b.status = "locked"; + writeJson("queue/patient-briefs.json", briefs); + } + print({ patient: patient.id, locked: true, shelfPath: rel }); + break; + } + default: + fail(`unknown verb ${verb}`); +} diff --git a/examples/prompt-lab/tests/lib.test.ts b/examples/prompt-lab/tests/lib.test.ts new file mode 100644 index 00000000..a49b7450 --- /dev/null +++ b/examples/prompt-lab/tests/lib.test.ts @@ -0,0 +1,113 @@ +import assert from "node:assert/strict"; +import { readFileSync } from "node:fs"; +import { test } from "node:test"; +import { commitError, toCommit } from "../lib/commit.ts"; +import { changed, gridError, highlights, sortRows, type Row } from "../lib/grid.ts"; +import { identifiers } from "../lib/phi.ts"; +import { agenciesUsing, piles } from "../lib/piles.ts"; +import { score } from "../lib/score.ts"; +import type { Agency, Output, Patient } from "../lib/types.ts"; +import { planError } from "../jobs/shared.ts"; + +const agencies = JSON.parse(readFileSync(new URL("../fixtures/agencies.json", import.meta.url), "utf8")) as Record; +const out = (answer: string, confidence: Output["confidence"] = "High", explanation = "x"): Output => ({ answer, confidence, explanation }); + +test("piles: the brief's three examples, deterministic from config", () => { + const byQ = Object.fromEntries(piles(agencies, "sunrise", "soc").map((q) => [q.questionId, q])); + assert.equal(byQ["wound-status"]!.pile, "shared"); + assert.deepEqual(byQ["wound-status"]!.sharedWith, ["harbor", "maple"]); + assert.equal(byQ["mood"]!.pile, "mismatch"); // same prompt, sunrise's menu adds "Agitated" + assert.equal(byQ["living-situation"]!.pile, "agency-specific"); + assert.equal(byQ["ostomy-supplies"]!.pile, "agency-specific"); +}); + +test("piles: a differing follow-up parent is a mismatch even with the same menu", () => { + const a: Record = { + x: { name: "x", visitTypes: { soc: [{ questionId: "q", options: ["A", "B"], parent: "p1" }] } }, + y: { name: "y", visitTypes: { soc: [{ questionId: "q", options: ["A", "B"] }] } }, + }; + assert.equal(piles(a, "x", "soc")[0]!.pile, "mismatch"); + assert.throws(() => piles(a, "x", "roc"), /no visit type roc/); +}); + +test("agenciesUsing: the blast radius of a question-level done", () => { + assert.deepEqual(agenciesUsing(agencies, "wound-status"), ["harbor", "maple", "sunrise"]); + assert.deepEqual(agenciesUsing(agencies, "living-situation"), ["sunrise"]); +}); + +const row = (q: string, p: string, ai: Output, target: Output, extra: Partial = {}): Row => + ({ questionId: q, patientId: p, pile: "shared", highlight: [], ai, target, notes: "", ...extra }); + +test("grid: highlighted rows first; changed = any field or notes, never a selection", () => { + assert.deepEqual(highlights(out("A", "High"), "shared", false), []); + assert.deepEqual(highlights(out("A", "Low"), "mismatch", true), ["low confidence", "shared prompt, this agency's menu or ancestry differs", "first-pass prompt, never reviewed"]); + const plain = row("a", "p", out("A"), out("A")); + const flagged = row("b", "p", out("A"), out("A"), { highlight: ["low confidence"] }); + assert.deepEqual(sortRows([plain, flagged]).map((r) => r.questionId), ["b", "a"]); + const explanationOnly = row("c", "p", out("A", "High", "x"), out("A", "High", "tightened")); + const notesOnly = row("d", "p", out("A"), out("A"), { notes: "spouse is in the home" }); + assert.deepEqual(changed([plain, explanationOnly, notesOnly]).map((r) => r.questionId), ["c", "d"]); +}); + +test("grid: a reviewer edit off the agency menu is refused, naming the row", () => { + const menus = { a: ["A", "B"] }; + assert.equal(gridError([row("a", "p", out("A"), out("B"))], menus), null); + assert.match(gridError([row("a", "p", out("A"), out("C"))], menus)!, /a\|p: target answer is not on the agency menu/); + assert.match(gridError([row("a", "p", out("A"), out("A", "Sure" as never))], menus)!, /confidence/); + assert.match(gridError({}, menus)!, /array/); +}); + +test("score: the brief's example — 1 of 2 golded worked (50%), Riley shown but not counted", () => { + const s = score("q", { pat: out("Ongoing"), jordan: out("Lives alone"), riley: out("Anxious") }, + { "q|pat": out("Ongoing"), "q|jordan": out("Lives with spouse") }); + assert.deepEqual(s.rows.map((r) => [r.patientId, r.result]), [["jordan", "did not"], ["pat", "worked"], ["riley", "no gold yet"]]); + assert.equal(s.golded, 2); assert.equal(s.worked, 1); assert.equal(s.percent, 50); + assert.equal(score("q", { riley: out("A") }, {}).percent, null); +}); + +test("commit: all / only / all except, over agency-specific candidates only", () => { + const c = ["living-situation", "ostomy-supplies"]; + assert.deepEqual(toCommit(c, { mode: "all", questions: [] }), c); + assert.deepEqual(toCommit(c, { mode: "only", questions: ["ostomy-supplies", "wound-status"] }), ["ostomy-supplies"]); + assert.deepEqual(toCommit(c, { mode: "except", questions: ["ostomy-supplies"] }), ["living-situation"]); + assert.match(commitError({ mode: "some" })!, /mode/); + assert.match(commitError({ mode: "only", questions: "x" })!, /questions/); +}); + +test("phi floor: dates, phones, record numbers, addresses", () => { + assert.deepEqual(identifiers("Pat, 75-84, lives alone, heel wound 2.0 x 1.5 cm"), []); + assert.equal(identifiers("seen 3/14/2026").length, 1); + assert.equal(identifiers("call 555-201-3344").length, 1); + assert.equal(identifiers("MRN 12345678").length, 1); + assert.equal(identifiers("lives at 12 Oak Street").length, 1); +}); + +test("test planner check: each question exactly once, shelf patients only", () => { + const shelf = [{ id: "pat" }, { id: "jordan" }] as Patient[]; + const ok = { coverage: [{ questionId: "a", patientIds: ["pat"] }], gaps: [{ questionId: "b", brief: "..." }] }; + assert.equal(planError(ok, ["a", "b"], shelf), null); + assert.match(planError({ ...ok, gaps: [] }, ["a", "b"], shelf)!, /b must appear exactly once/); + assert.match(planError({ coverage: [{ questionId: "a", patientIds: ["ghost"] }], gaps: [{ questionId: "b", brief: "" }] }, ["a", "b"], shelf)!, /ghost is not on the shelf/); + assert.match(planError({ ...ok, gaps: [...ok.gaps, { questionId: "z", brief: "" }] }, ["a", "b"], shelf)!, /not asked/); +}); + +import { parseReply, schemaError } from "../lib/reply.ts"; +import { engine, planTests } from "../prompts.ts"; + +test("reply: one surrounding fence is tolerated; prose is not", () => { + assert.deepEqual(parseReply('```json\n{"x":1}\n```'), { x: 1 }); + assert.deepEqual(parseReply(' {"x":1} '), { x: 1 }); + assert.throws(() => parseReply('Here you go: {"x":1}')); +}); + +test("reply schema: the engine's answer enum is the agency menu", () => { + const pat = { id: "pat", label: "Pat", locked: true, source: "invented", ageBand: "", visitType: "soc", referral: "", notes: "" } as const; + const { output } = engine("p", { question: "q", options: ["Healed", "Ongoing"] }, pat); + assert.equal(schemaError(output, { answer: "Ongoing", confidence: "Low", explanation: "e" }), null); + assert.match(schemaError(output, { answer: "Closed", confidence: "Low", explanation: "e" })!, /\$\.answer must be one of "Healed", "Ongoing"/); + assert.match(schemaError(output, { answer: "Healed", confidence: "Low" })!, /explanation is required/); + assert.match(schemaError(output, { answer: "Healed", confidence: "Low", explanation: "e", extra: 1 })!, /extra is not allowed/); + const plan = planTests([], []).output; + assert.match(schemaError(plan, { coverage: [{ questionId: "a", patientIds: [] }], gaps: [] })!, /needs at least 1/); + assert.throws(() => schemaError({ type: "string", format: "date" }, "x"), /not supported/); +}); diff --git a/examples/prompt-lab/tests/store.test.ts b/examples/prompt-lab/tests/store.test.ts new file mode 100644 index 00000000..8daa1004 --- /dev/null +++ b/examples/prompt-lab/tests/store.test.ts @@ -0,0 +1,71 @@ +import assert from "node:assert/strict"; +import { spawnSync } from "node:child_process"; +import { mkdtempSync, readFileSync, writeFileSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { test } from "node:test"; + +const STORE = new URL("../store.ts", import.meta.url).pathname; +const FIXTURES = new URL("../fixtures", import.meta.url).pathname; +const b64 = (v: unknown) => Buffer.from(JSON.stringify(v)).toString("base64"); + +function seeded(): { lab: string; store: (...a: string[]) => { status: number | null; out: any; err: string } } { + const lab = join(mkdtempSync(join(tmpdir(), "prompt-lab-")), "lab"); + const store = (...args: string[]) => { + const r = spawnSync("node", ["--no-warnings", "--experimental-strip-types", STORE, lab, ...args], { encoding: "utf8" }); + return { status: r.status, out: r.stdout ? JSON.parse(r.stdout) : null, err: r.stderr }; + }; + assert.equal(store("seed", FIXTURES).status, 0); + return { lab, store }; +} + +test("seed refuses to overwrite an existing lab", () => { + const { store } = seeded(); + const again = store("seed", FIXTURES); + assert.equal(again.status, 1); + assert.match(again.err, /already a lab/); +}); + +test("snapshot reads the whole lab", () => { + const { store } = seeded(); + const s = store("snapshot").out; + assert.deepEqual(s.shelf.map((p: { id: string }) => p.id), ["jordan", "pat", "riley"]); + assert.equal(s.bank.questions["wound-status"].livePromptId, "p-wound-status-v1"); + assert.deepEqual(s.gold, {}); +}); + +test("write-new never clobbers a reviewer's edit (a retried step converges)", () => { + const { lab, store } = seeded(); + assert.equal(store("write-new", "work/g.json", b64([1])).out.written, true); + writeFileSync(join(lab, "work/g.json"), "[\"edited\"]"); + assert.equal(store("write-new", "work/g.json", b64([1])).out.written, false); + assert.equal(readFileSync(join(lab, "work/g.json"), "utf8"), "[\"edited\"]"); + assert.match(store("write-new", "../escape.json", b64(1)).err, /escapes the lab/); +}); + +test("draft is content-addressed and publish is Done = live, both idempotent", () => { + const { store } = seeded(); + const a = store("draft", "wound-status", b64("new text")).out; + assert.deepEqual(store("draft", "wound-status", b64("new text")).out, a); + assert.equal(store("snapshot").out.bank.questions["wound-status"].livePromptId, "p-wound-status-v1"); // a draft is not live + assert.equal(store("publish", "wound-status", a.promptId).out.previous, "p-wound-status-v1"); + assert.equal(store("publish", "wound-status", a.promptId).out.previous, null); + const q = store("snapshot").out.bank.questions["wound-status"]; + assert.equal(q.livePromptId, a.promptId); + assert.equal(q.draftPromptId, null); + assert.equal(store("publish", "wound-status", "p-nope").status, 1); +}); + +test("enqueue is keyed by id; lock-patient freezes a chart and closes its brief", () => { + const { store } = seeded(); + const brief = { id: "gap-x", questionId: "ostomy-supplies", brief: "b", from: "planner", status: "queued" }; + assert.equal(store("enqueue", "patient-briefs", b64(brief)).out.added, true); + assert.equal(store("enqueue", "patient-briefs", b64(brief)).out.added, false); + const p = { id: "sam", label: "Sam", ageBand: "65-74", visitType: "soc", referral: "r", notes: "n" }; + assert.equal(store("lock-patient", b64(p), "gap-x").status, 0); + assert.equal(store("lock-patient", b64(p), "gap-x").status, 0); // same chart: converges + assert.match(store("lock-patient", b64({ ...p, notes: "other" })).err, /different sam/); + const s = store("snapshot").out; + assert.equal(s.shelf.find((x: { id: string }) => x.id === "sam").locked, true); + assert.equal(s.patientBriefs[0].status, "locked"); +}); diff --git a/examples/prompt-lab/tsconfig.json b/examples/prompt-lab/tsconfig.json new file mode 100644 index 00000000..ed4dafac --- /dev/null +++ b/examples/prompt-lab/tsconfig.json @@ -0,0 +1,9 @@ +{ + "compilerOptions": { + "target": "ES2022", "module": "ESNext", "moduleResolution": "Bundler", + "lib": ["ES2022"], "strict": true, "noUncheckedIndexedAccess": true, "noEmit": true, + "allowImportingTsExtensions": true, "skipLibCheck": true, + "typeRoots": ["../../packages/sdk/node_modules/@types"], "types": ["node"] + }, + "include": ["*.ts", "lib/*.ts", "jobs/*.ts", "tests/*.ts"] +} From e7b8a698b2ab9c4be6fd6e286b11e63f28c7f95b Mon Sep 17 00:00:00 2001 From: Relayflow Lead Date: Tue, 22 Sep 2026 22:09:17 -0700 Subject: [PATCH 2/6] docs(examples): point prompt-lab workarounds at #558, #560, #561 Co-Authored-By: Claude Opus 5.5 (1M context) --- examples/prompt-lab/README.md | 8 ++++---- examples/prompt-lab/jobs/shared.ts | 3 ++- examples/prompt-lab/lib/lab.ts | 1 + 3 files changed, 7 insertions(+), 5 deletions(-) diff --git a/examples/prompt-lab/README.md b/examples/prompt-lab/README.md index 2a46e73a..435d3bac 100644 --- a/examples/prompt-lab/README.md +++ b/examples/prompt-lab/README.md @@ -118,7 +118,7 @@ where it lives, and each has captured evidence in calls passed and nine failed, with `--agent-capacity` 4 or 1. [`runtime-parallel-llm-repro.flow.ts`](evidence/runtime-findings/runtime-parallel-llm-repro.flow.ts) reproduces it with no Prompt Lab code. **Workaround:** model calls run - sequentially. + sequentially. Tracked in [#561](https://github.com/AgentWorkforce/flows/issues/561) and [#560](https://github.com/AgentWorkforce/flows/issues/560). 2. **A predicate `.gate(fn)` before an `f.human` can't be resumed.** The verdict is read back from the `predicate-gates` stream with its keys re-ordered. The lowered `.gate` command no longer matches, and the @@ -126,14 +126,14 @@ where it lives, and each has captured evidence in `packages/sdk/src/authored-flow-executor.ts` `applyPredicateGate`: build the literal from fixed fields, not from `JSON.stringify(record)`. **Workaround:** the checks run in the body and fail through a journaled - failing step (`failStep`). + failing step (`failStep`). **Fixed in [#558](https://github.com/AgentWorkforce/flows/pull/558).** 3. **`f.llm(prompt, { output })` fails when the reply is fenced JSON.** The worker validates the raw reply. Sonnet sometimes wraps valid JSON in ```` ```json ```` anyway, and a failed run can't be resumed. **Workaround:** text-form `f.llm`, then [`lib/reply.ts`](lib/reply.ts) strips one fence and validates the schema, with one bounded re-ask. The text form takes no `model`, so calls use the - Claude adapter's default model. + Claude adapter's default model. **Fixed in [#558](https://github.com/AgentWorkforce/flows/pull/558).** 4. **A lease renewal that races a completion kills the run.** A step whose child run journaled `success` was reported as @@ -141,7 +141,7 @@ where it lives, and each has captured evidence in fatal `protocol_error` ([00-prove-attempt1-lease-conflict-after-success.txt](evidence/runtime-findings/00-prove-attempt1-lease-conflict-after-success.txt)). It's intermittent: it happened once in about 50 sequential calls. There's - no workaround in the flow, so rerun. + no workaround in the flow, so rerun. Tracked in [#560](https://github.com/AgentWorkforce/flows/issues/560). Findings 1 and 4 are the same class of problem: late lease traffic becomes fatal to the whole run instead of being ignored. diff --git a/examples/prompt-lab/jobs/shared.ts b/examples/prompt-lab/jobs/shared.ts index fcf03c6d..33f838b1 100644 --- a/examples/prompt-lab/jobs/shared.ts +++ b/examples/prompt-lab/jobs/shared.ts @@ -20,6 +20,7 @@ export const REPLY_ATTEMPTS = 2; * model that wraps valid JSON in a markdown fence fails the step, and a failed * run cannot be resumed (evidence/runtime-findings/00-job1-attempt1-fenced-json.txt, evidence/runtime-findings/00-patient-attempt1-fenced-json.txt). * The text form takes no model option, so calls use the CLI adapter's default. + * Return to f.llm(prompt, { output, model }) once flows#558 ships. */ export async function llm(f: Ctx, call: Call): Promise { let problem = ""; @@ -69,7 +70,7 @@ export async function untilQaPasses( * written as Promise.all, but relayflows 2.0.29 loses the run when concurrent * f.llm calls queue past their 30s lease (a stale completion becomes a fatal * protocol_error): see evidence/runtime-findings/runtime-parallel-llm-repro.txt. Restore - * Promise.all here and in jobs/new-agency.ts when that is fixed. + * Promise.all here and in jobs/new-agency.ts when flows#561 and flows#560 ship. */ export async function runEngine(f: Ctx, promptText: string, ask: Ask, patients: readonly Patient[]): Promise> { const outputs: Record = {}; diff --git a/examples/prompt-lab/lib/lab.ts b/examples/prompt-lab/lib/lab.ts index 7297ce68..ebae73a9 100644 --- a/examples/prompt-lab/lib/lab.ts +++ b/examples/prompt-lab/lib/lab.ts @@ -17,6 +17,7 @@ const b64 = (value: unknown): string => Buffer.from(JSON.stringify(value), "utf8 * keys re-ordered, the lowered `.gate` spec no longer matches, and the * resume is refused as run_admission_conflict * (evidence/runtime-findings/00-job1-attempt4-predicate-gate-resume-conflict.txt). + * Fixed in flows#558; return to `.gate()` once a release carries it. */ export async function failStep(f: Ctx, reason: string): Promise { await f.run(`echo ${shellWord(reason)} >&2; exit 1`); From 735f2e060901a90505a00b4155f17dfddde45662 Mon Sep 17 00:00:00 2001 From: Relayflow Lead Date: Tue, 22 Sep 2026 22:34:27 -0700 Subject: [PATCH 3/6] =?UTF-8?q?fix(examples):=20address=20prompt-lab=20rev?= =?UTF-8?q?iew=20=E2=80=94=20store=20lock,=20menus,=20visit=20type,=20comm?= =?UTF-8?q?it=20candidates?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - store.ts: mutating verbs take an exclusive lab lock, so concurrent runs no longer lose a read-modify-write; failures throw so the lock is released; the journal-tail guard counts UTF-8 bytes. - Job 2 re-runs and scores the new prompt on every distinct menu the question is asked with, since done changes it for all of them. - Job 1 plans from shelf patients of the run's visit type; gap briefs carry it and are keyed per question x visit type; generated charts use it. - Only first-pass prompts that ran on a covered patient are commit candidates; gap-only drafts are held until a patient covers them. - The patient plan file is keyed by the plan, so a re-run never reuses a stale one. - prove.sh propagates exit codes and stops on the first unexpected one. Evidence regenerated from a fresh run of prove.sh on this code. Co-Authored-By: Claude Opus 5.5 (1M context) --- examples/prompt-lab/.claude/settings.json | 7 ++ examples/prompt-lab/README.md | 44 ++++--- .../prompt-lab/evidence/run/01-job1-run.txt | 118 +++++++++--------- .../evidence/run/02-reviewer-edit.diff | 10 +- .../prompt-lab/evidence/run/03-answer.txt | 6 +- .../prompt-lab/evidence/run/04-resume.txt | 56 +++++---- .../run/{12-answer.txt => 05-answer.txt} | 6 +- .../evidence/run/05-reviewer-edit.diff | 10 -- .../run/{07-resume.txt => 06-resume.txt} | 38 +++--- .../evidence/run/07-patient-run.txt | 32 +++++ .../run/{09-answer.txt => 08-answer.txt} | 6 +- .../evidence/run/08-patient-run.txt | 28 ----- .../prompt-lab/evidence/run/09-resume.txt | 33 +++++ .../prompt-lab/evidence/run/10-job2-run.txt | 30 +++++ .../prompt-lab/evidence/run/10-resume.txt | 41 ------ .../run/{06-answer.txt => 11-answer.txt} | 6 +- .../prompt-lab/evidence/run/11-job2-run.txt | 34 ----- .../prompt-lab/evidence/run/12-resume.txt | 68 ++++++++++ .../run/{14-answer.txt => 13-answer.txt} | 6 +- .../prompt-lab/evidence/run/13-resume.txt | 80 ------------ .../prompt-lab/evidence/run/14-resume.txt | 45 +++++++ .../{16-lab-state.txt => 15-lab-state.txt} | 14 +-- .../prompt-lab/evidence/run/15-resume.txt | 49 -------- .../prompt-lab/evidence/run/lab/bank.json | 12 +- .../prompt-lab/evidence/run/lab/gold.json | 9 +- .../evidence/run/lab/queue/issues.json | 4 +- .../run/lab/queue/patient-briefs.json | 5 +- .../evidence/run/lab/shelf/marguerite.json | 10 -- .../evidence/run/lab/shelf/verna.json | 10 ++ .../prompt-lab/evidence/run/lab/targets.json | 22 ++-- .../fbf53055/plan.json | 4 + .../patients/gap-ostomy-supplies/plan.json | 4 - .../work/q-wound-status/cd33704e/grid.json | 55 ++++++++ .../work/q-wound-status/cd33704e/score.json | 59 +++++++++ .../work/q-wound-status/d61687f9/grid.json | 72 ----------- .../work/q-wound-status/d61687f9/score.json | 54 -------- .../lab/work/sunrise-soc/62c71a0f/commit.json | 6 - .../lab/work/sunrise-soc/8b7996f3/commit.json | 4 + .../{62c71a0f => 8b7996f3}/grid.json | 54 ++++---- examples/prompt-lab/jobs/fix.ts | 30 +++-- examples/prompt-lab/jobs/new-agency.ts | 17 ++- examples/prompt-lab/jobs/patient.ts | 13 +- examples/prompt-lab/jobs/shared.ts | 3 + examples/prompt-lab/lib/piles.ts | 16 +++ examples/prompt-lab/lib/types.ts | 2 +- examples/prompt-lab/prompts.ts | 6 +- examples/prompt-lab/prove.sh | 75 ++++++----- examples/prompt-lab/store.ts | 40 +++++- examples/prompt-lab/tests/lib.test.ts | 12 +- examples/prompt-lab/tests/store.test.ts | 37 +++++- 50 files changed, 752 insertions(+), 650 deletions(-) create mode 100644 examples/prompt-lab/.claude/settings.json rename examples/prompt-lab/evidence/run/{12-answer.txt => 05-answer.txt} (53%) delete mode 100644 examples/prompt-lab/evidence/run/05-reviewer-edit.diff rename examples/prompt-lab/evidence/run/{07-resume.txt => 06-resume.txt} (60%) create mode 100644 examples/prompt-lab/evidence/run/07-patient-run.txt rename examples/prompt-lab/evidence/run/{09-answer.txt => 08-answer.txt} (53%) delete mode 100644 examples/prompt-lab/evidence/run/08-patient-run.txt create mode 100644 examples/prompt-lab/evidence/run/09-resume.txt create mode 100644 examples/prompt-lab/evidence/run/10-job2-run.txt delete mode 100644 examples/prompt-lab/evidence/run/10-resume.txt rename examples/prompt-lab/evidence/run/{06-answer.txt => 11-answer.txt} (53%) delete mode 100644 examples/prompt-lab/evidence/run/11-job2-run.txt create mode 100644 examples/prompt-lab/evidence/run/12-resume.txt rename examples/prompt-lab/evidence/run/{14-answer.txt => 13-answer.txt} (53%) delete mode 100644 examples/prompt-lab/evidence/run/13-resume.txt create mode 100644 examples/prompt-lab/evidence/run/14-resume.txt rename examples/prompt-lab/evidence/run/{16-lab-state.txt => 15-lab-state.txt} (65%) delete mode 100644 examples/prompt-lab/evidence/run/15-resume.txt delete mode 100644 examples/prompt-lab/evidence/run/lab/shelf/marguerite.json create mode 100644 examples/prompt-lab/evidence/run/lab/shelf/verna.json create mode 100644 examples/prompt-lab/evidence/run/lab/work/patients/gap-ostomy-supplies-soc/fbf53055/plan.json delete mode 100644 examples/prompt-lab/evidence/run/lab/work/patients/gap-ostomy-supplies/plan.json create mode 100644 examples/prompt-lab/evidence/run/lab/work/q-wound-status/cd33704e/grid.json create mode 100644 examples/prompt-lab/evidence/run/lab/work/q-wound-status/cd33704e/score.json delete mode 100644 examples/prompt-lab/evidence/run/lab/work/q-wound-status/d61687f9/grid.json delete mode 100644 examples/prompt-lab/evidence/run/lab/work/q-wound-status/d61687f9/score.json delete mode 100644 examples/prompt-lab/evidence/run/lab/work/sunrise-soc/62c71a0f/commit.json create mode 100644 examples/prompt-lab/evidence/run/lab/work/sunrise-soc/8b7996f3/commit.json rename examples/prompt-lab/evidence/run/lab/work/sunrise-soc/{62c71a0f => 8b7996f3}/grid.json (52%) diff --git a/examples/prompt-lab/.claude/settings.json b/examples/prompt-lab/.claude/settings.json new file mode 100644 index 00000000..5123f1d7 --- /dev/null +++ b/examples/prompt-lab/.claude/settings.json @@ -0,0 +1,7 @@ +{ + "permissions": { + "allow": [ + "mcp__relaycast__*" + ] + } +} diff --git a/examples/prompt-lab/README.md b/examples/prompt-lab/README.md index 435d3bac..0d5449e9 100644 --- a/examples/prompt-lab/README.md +++ b/examples/prompt-lab/README.md @@ -27,14 +27,19 @@ Each job file is its brief diagram, line by line: | First pass persists as targets / gold | `store.ts record`, fed only from the grid you reviewed. AI output never writes gold directly | | Iterator: rewrite-only | [`prompts.ts`](prompts.ts) `iterate`. Input is the changeset, patients, brief and existing prompt; output is new prompt text | | Prompt QA: the brief plus shared guidelines | `promptQa`, looping with the iterator. Capped at 3 tries, then the run parks as `needs_human` | -| Done = live | `store.ts publish` sets `livePromptId`. There is no promote step. A shared question warns which agencies it will change | +| Done = live | `store.ts publish` sets `livePromptId`. There is no promote step. A shared question warns which agencies it will change, and the re-run scores the new prompt on **every distinct menu** it is asked with ([`lib/piles.ts`](lib/piles.ts) `distinctMenus`), because done changes it for all of them | | Shared frozen at config level | a changed shared or mismatch row becomes a `config-send` issue that carries its targets and a proposed rewrite | -| Test planner: never invents a patient | picks shelf ids (checked deterministically); each hole becomes a gap brief | +| Test planner: never invents a patient | picks shelf patients of the run's visit type (checked deterministically); each hole becomes one gap brief per question × visit type | | You do not approve the chart | the only patient gate is *kick generate*. Patient QA, plus a deterministic identifier check ([`lib/phi.ts`](lib/phi.ts)), locks it | Every read and write of the lab is a journaled `f.run` step. Every store verb is idempotent, so a retried step lands the lab in the same state. `write-new` -never overwrites an edit you made. +never overwrites an edit you made. Mutating verbs take an exclusive lab lock, +so two runs at once never lose each other's update. + +A first-pass prompt is offered for commit only after it has run on a shelf +patient and you have reviewed its rows. A prompt whose question is a gap waits +as a draft until a patient covers it. **Not built:** anything the brief lists under "Not at the start". Also not built: @@ -51,7 +56,7 @@ The flow is the job graph that sits under those screens. ```sh npm install -npm test # 16 unit tests over the deterministic parts +npm test # 20 unit tests over the deterministic parts and the store node --experimental-strip-types store.ts ./my-lab seed fixtures npx flows run prompt-lab.flow.ts --local-agent --input \ '{"job":"new-agency","reviewer":"","lab":"./my-lab","agency":"sunrise","visitType":"soc"}' @@ -86,24 +91,33 @@ diff. It captures every command with its output and exit code in ./prove.sh evidence/run ``` -The captured run (Claude Code 2.1.280, the adapter's default model): +The captured run (Claude Code 2.1.280, the adapter's default model). `prove.sh` +stops on the first exit it didn't expect, so reaching the final state means +every step below exited as shown: | Run | Result | What happened | | --- | --- | --- | -| Job 1 · `sunrise` / `soc` ([01](evidence/run/01-job1-run.txt), [04](evidence/run/04-resume.txt), [07](evidence/run/07-resume.txt)) | 29 steps, `success` | Piles came out as shared / mismatch / agency-specific. Both agency-specific questions got first-pass prompts that passed Prompt QA. The planner covered 3 questions and queued `gap-ostomy-supplies`. Gate 1: the reviewer rewrote Pat's wound explanation and added a note ([02](evidence/run/02-reviewer-edit.diff)). That shared row went to the question manager with its target. Gate 2: committed all except `ostomy-supplies`, which no patient has exercised yet ([05](evidence/run/05-reviewer-edit.diff)). `living-situation` went live. | -| Patient · `gap-ostomy-supplies` ([08](evidence/run/08-patient-run.txt), [10](evidence/run/10-resume.txt)) | 9 steps, `success` | Plan, then kick generate. The chart passed Patient QA on its first try and `marguerite` locked onto the shelf. The brief is marked `locked`. | -| Job 2 · the config-send issue ([11](evidence/run/11-job2-run.txt), [13](evidence/run/13-resume.txt), [15](evidence/run/15-resume.txt)) | 24 steps, `success` | Four shelf patients, including the new one. Gold came prefilled, with Pat's config target carried over. The iterator's rewrite passed Prompt QA. Re-run scored 4 of 4 golded patients worked (100%), and the gate warned it would change harbor, maple and sunrise. Done made the new prompt live and closed the issue. | +| Job 1 · `sunrise` / `soc` ([01](evidence/run/01-job1-run.txt), [04](evidence/run/04-resume.txt), [06](evidence/run/06-resume.txt)) | 29 steps, `success` | Piles came out as shared / mismatch / agency-specific. Both agency-specific questions got first-pass prompts that passed Prompt QA. The planner covered 3 questions from the `soc` shelf and queued `gap-ostomy-supplies-soc`. Gate 1: 9 rows, 7 highlighted. The reviewer raised Pat's wound confidence to High, rewrote the explanation and added a note ([02](evidence/run/02-reviewer-edit.diff)). That shared row went to the question manager with its target. Gate 2 offered only `living-situation`, and it went live. `ostomy-supplies` had no shelf patient, so it was held as a draft. | +| Patient · `gap-ostomy-supplies-soc` ([07](evidence/run/07-patient-run.txt), [09](evidence/run/09-resume.txt)) | 9 steps, `success` | Plan, then kick generate. The chart passed Patient QA on its first try, and `verna` locked onto the shelf. The brief is marked `locked`. | +| Job 2 · the config-send issue ([10](evidence/run/10-job2-run.txt), [12](evidence/run/12-resume.txt), [14](evidence/run/14-resume.txt)) | 22 steps, `success` | Three shelf patients. Gold came prefilled, with Pat's config target carried over. The iterator's rewrite passed Prompt QA. The re-run scored 3 of 3 golded patients worked (100%), and the gate warned it would change harbor, maple and sunrise. Done made the new prompt live and closed the issue. | + +Final state: [15-lab-state.txt](evidence/run/15-lab-state.txt), with the whole lab in `evidence/run/lab/`. + +**What this run does not show:** -Final state: [16-lab-state.txt](evidence/run/16-lab-state.txt), with the whole lab in `evidence/run/lab/`. +- **A wrong answer being fixed.** The model answered Pat "Ongoing / Medium" + even under the flawed wound prompt. So Job 2 iterated on confidence, the + explanation and source priority, not on the answer. +- **Re-running on several menus.** `wound-status` has one menu across its + agencies, so the per-menu re-run ran with one. The multi-menu case, a + mismatch question like `mood`, is covered by the `distinctMenus` unit test + only. +- **A clinician's judgment.** The reviewer at every gate is `prove.sh`, + applying fixed edits. -**What this run does not show:** an answer flipping from wrong to right. Here -the model answered Pat "Ongoing" even under the flawed wound prompt, so Job 2 -iterated on the explanation and on source priority, not on the answer. The -brief's exact failure, "Healed / High" for Pat, did occur in an earlier +The brief's exact failure, "Healed / High" for Pat, did occur in an earlier `claude-sonnet-5` run of the same prompt. Its engine step's journal is in [00-sonnet-run-pat-healed-high.txt](evidence/runtime-findings/00-sonnet-run-pat-healed-high.txt). -The reviewer at every gate is `prove.sh`, applying fixed edits. It is not a -clinician's judgment. ## Runtime findings (relayflows 2.0.29) diff --git a/examples/prompt-lab/evidence/run/01-job1-run.txt b/examples/prompt-lab/evidence/run/01-job1-run.txt index a5bd9e9f..75075515 100644 --- a/examples/prompt-lab/evidence/run/01-job1-run.txt +++ b/examples/prompt-lab/evidence/run/01-job1-run.txt @@ -1,84 +1,82 @@ -$ npx flows run --no-observer-link --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data --local-agent prompt-lab.flow.ts --input {"job":"new-agency","reviewer":"prompt-lab-reviewer","lab":"evidence/run/lab","agency":"sunrise","visitType":"soc"} +$ npx flows run --no-observer-link --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data --local-agent prompt-lab.flow.ts --input {"job":"new-agency","reviewer":"prompt-lab-reviewer","lab":"evidence/run/lab","agency":"sunrise","visitType":"soc"} ○ run-1 (deterministic) 0.00s -✓ run-1 (deterministic) 0.06s completionReason: success +✓ run-1 (deterministic) 0.15s completionReason: success ○ llm-2 (llm) 0.00s -WAITING [worker_lease] Run "01M36695SX9Z217ZK23C7R4XC4" step "llm-2" (llm) is running under a worker lease until 1790135569521. -↻ llm-2 (llm) 3.67s -WAITING [worker_lease] Run "01M36695SX9Z217ZK23C7R4XC4" step "llm-2" (llm) is running under a worker lease until 1790135579526. -↻ llm-2 (llm) 13.68s -✓ llm-2 (llm) 21.24s completionReason: success +WAITING [worker_lease] Run "01M36BDMS16K70HPQCSJCS496R" step "llm-2" (llm) is running under a worker lease until 1790140958805. +↻ llm-2 (llm) 6.39s +WAITING [worker_lease] Run "01M36BDMS16K70HPQCSJCS496R" step "llm-2" (llm) is running under a worker lease until 1790140968809. +↻ llm-2 (llm) 16.43s +✓ llm-2 (llm) 23.12s completionReason: success ○ llm-3 (llm) 0.00s -WAITING [worker_lease] Run "01M3669TF3YG6DC3N7WXK899YW" step "llm-3" (llm) is running under a worker lease until 1790135590679. -↻ llm-3 (llm) 3.59s -WAITING [worker_lease] Run "01M3669TF3YG6DC3N7WXK899YW" step "llm-3" (llm) is running under a worker lease until 1790135600685. -↻ llm-3 (llm) 13.60s -✓ llm-3 (llm) 17.58s completionReason: success +WAITING [worker_lease] Run "01M36BE8Z1N48PME3B42TFGJGY" step "llm-3" (llm) is running under a worker lease until 1790140979477. +↻ llm-3 (llm) 3.94s +✓ llm-3 (llm) 10.04s completionReason: success ○ run-4 (deterministic) 0.00s ✓ run-4 (deterministic) 0.06s completionReason: success ○ llm-5 (llm) 0.00s -WAITING [worker_lease] Run "01M366ABZWVZVKACSBTK4V0S0S" step "llm-5" (llm) is running under a worker lease until 1790135608622. -↻ llm-5 (llm) 3.89s -WAITING [worker_lease] Run "01M366ABZWVZVKACSBTK4V0S0S" step "llm-5" (llm) is running under a worker lease until 1790135618625. -↻ llm-5 (llm) 13.90s -WAITING [worker_lease] Run "01M366ABZWVZVKACSBTK4V0S0S" step "llm-5" (llm) is running under a worker lease until 1790135628635. -↻ llm-5 (llm) 23.92s -✓ llm-5 (llm) 28.89s completionReason: success +WAITING [worker_lease] Run "01M36BEM20H1BJV077SSX3BCX7" step "llm-5" (llm) is running under a worker lease until 1790140990835. +↻ llm-5 (llm) 5.19s +WAITING [worker_lease] Run "01M36BEM20H1BJV077SSX3BCX7" step "llm-5" (llm) is running under a worker lease until 1790141000839. +↻ llm-5 (llm) 15.21s +✓ llm-5 (llm) 24.55s completionReason: success ○ llm-6 (llm) 0.00s -WAITING [worker_lease] Run "01M366B8HTX6D89C26FARJH7S7" step "llm-6" (llm) is running under a worker lease until 1790135637870. -↻ llm-6 (llm) 4.24s -✓ llm-6 (llm) 9.91s completionReason: success +WAITING [worker_lease] Run "01M36BFB66Q05G9R02NZ6HQ4E0" step "llm-6" (llm) is running under a worker lease until 1790141014522. +↻ llm-6 (llm) 4.33s +WAITING [worker_lease] Run "01M36BFB66Q05G9R02NZ6HQ4E0" step "llm-6" (llm) is running under a worker lease until 1790141024523. +↻ llm-6 (llm) 14.37s +✓ llm-6 (llm) 15.11s completionReason: success ○ run-7 (deterministic) 0.00s -✓ run-7 (deterministic) 0.06s completionReason: success +✓ run-7 (deterministic) 0.07s completionReason: success ○ llm-8 (llm) 0.00s -WAITING [worker_lease] Run "01M366BM1FQN24G3D8RFC30WBE" step "llm-8" (llm) is running under a worker lease until 1790135649636. -↻ llm-8 (llm) 6.04s -WAITING [worker_lease] Run "01M366BM1FQN24G3D8RFC30WBE" step "llm-8" (llm) is running under a worker lease until 1790135659640. -↻ llm-8 (llm) 16.07s -✓ llm-8 (llm) 24.61s completionReason: success +WAITING [worker_lease] Run "01M36BFSC808R6VNSW2ESXMW1M" step "llm-8" (llm) is running under a worker lease until 1790141029051. +↻ llm-8 (llm) 3.67s +WAITING [worker_lease] Run "01M36BFSC808R6VNSW2ESXMW1M" step "llm-8" (llm) is running under a worker lease until 1790141039054. +↻ llm-8 (llm) 13.72s +✓ llm-8 (llm) 18.18s completionReason: success ○ run-9 (deterministic) 0.00s ✓ run-9 (deterministic) 0.06s completionReason: success ○ llm-10 (llm) 0.00s -WAITING [worker_lease] Run "01M366CAWMTF1ZYN956WYE5PT6" step "llm-10" (llm) is running under a worker lease until 1790135673031. -↻ llm-10 (llm) 4.76s -✓ llm-10 (llm) 10.78s completionReason: success +WAITING [worker_lease] Run "01M36BGCRVM8VKEZ8AX0RRRP2F" step "llm-10" (llm) is running under a worker lease until 1790141048910. +↻ llm-10 (llm) 5.29s +✓ llm-10 (llm) 9.54s completionReason: success ○ llm-11 (llm) 0.00s -WAITING [worker_lease] Run "01M366CMM9N0C5674SRVQVGNDP" step "llm-11" (llm) is running under a worker lease until 1790135683006. -↻ llm-11 (llm) 3.95s -✓ llm-11 (llm) 8.75s completionReason: success +WAITING [worker_lease] Run "01M36BGMSX3J78Y4T7MKK4FEG3" step "llm-11" (llm) is running under a worker lease until 1790141057136. +↻ llm-11 (llm) 3.97s +✓ llm-11 (llm) 7.75s completionReason: success ○ llm-12 (llm) 0.00s -WAITING [worker_lease] Run "01M366CYN97BBESM8SDRM3Y7G8" step "llm-12" (llm) is running under a worker lease until 1790135693276. -↻ llm-12 (llm) 5.48s -✓ llm-12 (llm) 9.54s completionReason: success +WAITING [worker_lease] Run "01M36BGVYACMHT4ZM8B0SZ8KM0" step "llm-12" (llm) is running under a worker lease until 1790141064446. +↻ llm-12 (llm) 3.54s +✓ llm-12 (llm) 9.06s completionReason: success ○ llm-13 (llm) 0.00s -WAITING [worker_lease] Run "01M366D6Y4HVGJNBN1F45HHWJJ" step "llm-13" (llm) is running under a worker lease until 1790135701751. -↻ llm-13 (llm) 4.41s -✓ llm-13 (llm) 8.46s completionReason: success +WAITING [worker_lease] Run "01M36BH4ZFMQ4TJXH01EW64S4S" step "llm-13" (llm) is running under a worker lease until 1790141073700. +↻ llm-13 (llm) 3.73s +✓ llm-13 (llm) 7.62s completionReason: success ○ llm-14 (llm) 0.00s -WAITING [worker_lease] Run "01M366DGCFA2F4V97Y920YHH2M" step "llm-14" (llm) is running under a worker lease until 1790135711427. -↻ llm-14 (llm) 5.63s -✓ llm-14 (llm) 9.38s completionReason: success +WAITING [worker_lease] Run "01M36BHD49DRZSES3XZNV3FDBP" step "llm-14" (llm) is running under a worker lease until 1790141082045. +↻ llm-14 (llm) 4.45s +✓ llm-14 (llm) 8.19s completionReason: success ○ llm-15 (llm) 0.00s -WAITING [worker_lease] Run "01M366DQN7RCZ9R3QP9RP5WJTS" step "llm-15" (llm) is running under a worker lease until 1790135718875. -↻ llm-15 (llm) 3.70s -✓ llm-15 (llm) 7.88s completionReason: success +WAITING [worker_lease] Run "01M36BHMQ7EVSCV5WR1M8ZYSPP" step "llm-15" (llm) is running under a worker lease until 1790141089818. +↻ llm-15 (llm) 4.03s +✓ llm-15 (llm) 8.36s completionReason: success ○ llm-16 (llm) 0.00s -WAITING [worker_lease] Run "01M366DZTJBB1V1G5K7G1RWTWT" step "llm-16" (llm) is running under a worker lease until 1790135727237. -↻ llm-16 (llm) 4.18s -✓ llm-16 (llm) 10.67s completionReason: success +WAITING [worker_lease] Run "01M36BHYFTVHK7NRF3RFMNDFZ9" step "llm-16" (llm) is running under a worker lease until 1790141099824. +↻ llm-16 (llm) 5.68s +✓ llm-16 (llm) 9.96s completionReason: success ○ llm-17 (llm) 0.00s -WAITING [worker_lease] Run "01M366EBC622QRBM6NAV93Q65K" step "llm-17" (llm) is running under a worker lease until 1790135739066. -↻ llm-17 (llm) 5.34s -✓ llm-17 (llm) 10.03s completionReason: success +WAITING [worker_lease] Run "01M36BJ6W2TMXB2VNPEKNK4YJR" step "llm-17" (llm) is running under a worker lease until 1790141108406. +↻ llm-17 (llm) 4.30s +✓ llm-17 (llm) 8.89s completionReason: success ○ llm-18 (llm) 0.00s -WAITING [worker_lease] Run "01M366EKED0RJASW9AM0Q2EY28" step "llm-18" (llm) is running under a worker lease until 1790135747331. -↻ llm-18 (llm) 3.57s -✓ llm-18 (llm) 9.69s completionReason: success +WAITING [worker_lease] Run "01M36BJF3GQ3SHGET32HR0QDYK" step "llm-18" (llm) is running under a worker lease until 1790141116838. +↻ llm-18 (llm) 3.84s +✓ llm-18 (llm) 9.42s completionReason: success ○ run-19 (deterministic) 0.00s -✓ run-19 (deterministic) 0.07s completionReason: success +✓ run-19 (deterministic) 0.06s completionReason: success ○ human-20 (deterministic) 0.00s ⏸ human-20 (human) 0.00s -PARKED [run_parked] Run "01M3669253ED8QPNX7N7RE7PXF" is waiting for prompt-lab-reviewer to answer human-20: "Config workbench · sunrise soc: 9 rows on 3 questions (6 highlighted, listed first); 1 gap brief(s) queued for the test patient manager.\nEdit target answer / confidence / explanation / notes in evidence/run/lab/work/sunrise-soc/62c71a0f/grid.json. Your first pass persists as the target for each question × patient.\nyes = persist targets and run iteration on every changed row; no = stop without persisting." -Answer with: flows answer --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data 01M3669253ED8QPNX7N7RE7PXF human-20 yes|no -Then continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data --local-agent 01M3669253ED8QPNX7N7RE7PXF -RUN 01M3669253ED8QPNX7N7RE7PXF parked +PARKED [run_parked] Run "01M36BDEBD2DKTZTX8GTB28DW7" is waiting for prompt-lab-reviewer to answer human-20: "Config workbench · sunrise soc: 9 rows on 3 questions (7 highlighted, listed first); 1 gap brief(s) queued for the test patient manager.\nEdit target answer / confidence / explanation / notes in evidence/run/lab/work/sunrise-soc/8b7996f3/grid.json. Your first pass persists as the target for each question × patient.\nyes = persist targets and run iteration on every changed row; no = stop without persisting." +Answer with: flows answer --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data 01M36BDEBD2DKTZTX8GTB28DW7 human-20 yes|no +Then continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data --local-agent 01M36BDEBD2DKTZTX8GTB28DW7 +RUN 01M36BDEBD2DKTZTX8GTB28DW7 parked exit=3 diff --git a/examples/prompt-lab/evidence/run/02-reviewer-edit.diff b/examples/prompt-lab/evidence/run/02-reviewer-edit.diff index d9ca47c6..df6f37db 100644 --- a/examples/prompt-lab/evidence/run/02-reviewer-edit.diff +++ b/examples/prompt-lab/evidence/run/02-reviewer-edit.diff @@ -1,9 +1,11 @@ -# reviewer edit: evidence/run/lab/work/sunrise-soc/62c71a0f/grid.json -@@ -144,9 +144,9 @@ +# reviewer edit: evidence/run/lab/work/sunrise-soc/8b7996f3/grid.json +@@ -128,10 +128,10 @@ + }, "target": { "answer": "Ongoing", - "confidence": "High", -- "explanation": "Today's SOC visit documents an open left heel wound measuring 2.0 x 1.5 cm with moderate serous drainage and yellow slough, dressed with foam per protocol, so the discharge summary's \"wound closed\" note is outdated and the current assessment governs." +- "confidence": "Medium", +- "explanation": "The referral discharge summary states the left heel wound was closed, but today's SOC visit notes document an open left heel area 2.0 x 1.5 cm with moderate serous drainage and yellow slough, so the wound is not healed." ++ "confidence": "High", + "explanation": "Today's visit notes describe an open 2.0 x 1.5 cm left heel wound with serous drainage and slough; the discharge summary's closed status is outdated." }, - "notes": "" diff --git a/examples/prompt-lab/evidence/run/03-answer.txt b/examples/prompt-lab/evidence/run/03-answer.txt index 183ec4f4..147a954e 100644 --- a/examples/prompt-lab/evidence/run/03-answer.txt +++ b/examples/prompt-lab/evidence/run/03-answer.txt @@ -1,4 +1,4 @@ -$ npx flows answer --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data 01M3669253ED8QPNX7N7RE7PXF human-20 yes --by prompt-lab-reviewer -ANSWERED 01M3669253ED8QPNX7N7RE7PXF human-20 yes -Continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data --local-agent 01M3669253ED8QPNX7N7RE7PXF +$ npx flows answer --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data 01M36BDEBD2DKTZTX8GTB28DW7 human-20 yes --by prompt-lab-reviewer +ANSWERED 01M36BDEBD2DKTZTX8GTB28DW7 human-20 yes +Continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data --local-agent 01M36BDEBD2DKTZTX8GTB28DW7 exit=0 diff --git a/examples/prompt-lab/evidence/run/04-resume.txt b/examples/prompt-lab/evidence/run/04-resume.txt index 06a9a034..726d81da 100644 --- a/examples/prompt-lab/evidence/run/04-resume.txt +++ b/examples/prompt-lab/evidence/run/04-resume.txt @@ -1,60 +1,62 @@ -$ npx flows resume --no-observer-link --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data --local-agent 01M3669253ED8QPNX7N7RE7PXF +$ npx flows resume --no-observer-link --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data --local-agent 01M36BDEBD2DKTZTX8GTB28DW7 ○ run-1 (deterministic) 0.00s ✓ run-1 (deterministic) 0.01s completionReason: success ○ llm-2 (llm) 0.00s -✓ llm-2 (llm) 3.84s completionReason: success +✓ llm-2 (llm) 3.82s completionReason: success ○ llm-3 (llm) 0.00s -✓ llm-3 (llm) 3.87s completionReason: success +✓ llm-3 (llm) 3.67s completionReason: success ○ run-4 (deterministic) 0.00s ✓ run-4 (deterministic) 0.01s completionReason: success ○ llm-5 (llm) 0.00s -✓ llm-5 (llm) 3.80s completionReason: success +✓ llm-5 (llm) 3.59s completionReason: success ○ llm-6 (llm) 0.00s -✓ llm-6 (llm) 4.16s completionReason: success +✓ llm-6 (llm) 3.91s completionReason: success ○ run-7 (deterministic) 0.00s ✓ run-7 (deterministic) 0.01s completionReason: success ○ llm-8 (llm) 0.00s -✓ llm-8 (llm) 3.87s completionReason: success +✓ llm-8 (llm) 3.57s completionReason: success ○ run-9 (deterministic) 0.00s ✓ run-9 (deterministic) 0.01s completionReason: success ○ llm-10 (llm) 0.00s -✓ llm-10 (llm) 3.66s completionReason: success +✓ llm-10 (llm) 3.67s completionReason: success ○ llm-11 (llm) 0.00s -✓ llm-11 (llm) 3.85s completionReason: success +✓ llm-11 (llm) 3.75s completionReason: success ○ llm-12 (llm) 0.00s -✓ llm-12 (llm) 3.67s completionReason: success +✓ llm-12 (llm) 4.01s completionReason: success ○ llm-13 (llm) 0.00s -✓ llm-13 (llm) 3.68s completionReason: success +✓ llm-13 (llm) 3.48s completionReason: success ○ llm-14 (llm) 0.00s -✓ llm-14 (llm) 4.33s completionReason: success +✓ llm-14 (llm) 4.34s completionReason: success ○ llm-15 (llm) 0.00s -✓ llm-15 (llm) 4.92s completionReason: success +✓ llm-15 (llm) 4.94s completionReason: success ○ llm-16 (llm) 0.00s -✓ llm-16 (llm) 4.21s completionReason: success +✓ llm-16 (llm) 3.99s completionReason: success ○ llm-17 (llm) 0.00s -✓ llm-17 (llm) 3.86s completionReason: success +✓ llm-17 (llm) 3.57s completionReason: success ○ llm-18 (llm) 0.00s -✓ llm-18 (llm) 3.77s completionReason: success +✓ llm-18 (llm) 3.57s completionReason: success ○ run-19 (deterministic) 0.00s ✓ run-19 (deterministic) 0.01s completionReason: success ○ human-20 (deterministic) 0.00s ✓ human-20 (deterministic) 0.02s completionReason: success ○ run-21 (deterministic) 0.00s -✓ run-21 (deterministic) 0.06s completionReason: success +✓ run-21 (deterministic) 0.07s completionReason: success ○ run-22 (deterministic) 0.00s -✓ run-22 (deterministic) 0.06s completionReason: success +✓ run-22 (deterministic) 0.08s completionReason: success ○ llm-23 (llm) 0.00s -WAITING [worker_lease] Run "01M366GMPW32EQPWHW07KEQPC2" step "llm-23" (llm) is running under a worker lease until 1790135814160. -↻ llm-23 (llm) 3.64s -✓ llm-23 (llm) 13.20s completionReason: success +WAITING [worker_lease] Run "01M36BMESDKX5DBD3T1QN4JBM1" step "llm-23" (llm) is running under a worker lease until 1790141182053. +↻ llm-23 (llm) 3.91s +WAITING [worker_lease] Run "01M36BMESDKX5DBD3T1QN4JBM1" step "llm-23" (llm) is running under a worker lease until 1790141192057. +↻ llm-23 (llm) 13.91s +✓ llm-23 (llm) 21.61s completionReason: success ○ run-24 (deterministic) 0.00s -✓ run-24 (deterministic) 0.06s completionReason: success +✓ run-24 (deterministic) 0.07s completionReason: success ○ run-25 (deterministic) 0.00s -✓ run-25 (deterministic) 0.06s completionReason: success +✓ run-25 (deterministic) 0.07s completionReason: success ○ human-26 (deterministic) 0.00s -⏸ human-26 (human) 0.00s -PARKED [run_parked] Run "01M3669253ED8QPNX7N7RE7PXF" is waiting for prompt-lab-reviewer to answer human-26: "Commit output · sunrise: agency-specific prompts ready to go live in Apricot: living-situation, ostomy-supplies.\nSent to the question manager (frozen here): wound-status → config-sunrise-wound-status-ce89aea7.\nTo commit only some, edit evidence/run/lab/work/sunrise-soc/62c71a0f/commit.json: {\"mode\":\"only\"|\"except\",\"questions\":[...]}.\nyes = commit (Done = live, no promote step); no = leave them as drafts." -Answer with: flows answer --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data 01M3669253ED8QPNX7N7RE7PXF human-26 yes|no -Then continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data --local-agent 01M3669253ED8QPNX7N7RE7PXF -RUN 01M3669253ED8QPNX7N7RE7PXF parked +⏸ human-26 (human) 0.02s +PARKED [run_parked] Run "01M36BDEBD2DKTZTX8GTB28DW7" is waiting for prompt-lab-reviewer to answer human-26: "Commit output · sunrise: agency-specific prompts ready to go live in Apricot: living-situation.\nSent to the question manager (frozen here): wound-status → config-sunrise-wound-status-8a366995.\nHeld as drafts until a shelf patient covers them: ostomy-supplies.\nTo commit only some, edit evidence/run/lab/work/sunrise-soc/8b7996f3/commit.json: {\"mode\":\"only\"|\"except\",\"questions\":[...]}.\nyes = commit (Done = live, no promote step); no = leave them as drafts." +Answer with: flows answer --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data 01M36BDEBD2DKTZTX8GTB28DW7 human-26 yes|no +Then continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data --local-agent 01M36BDEBD2DKTZTX8GTB28DW7 +RUN 01M36BDEBD2DKTZTX8GTB28DW7 parked exit=3 diff --git a/examples/prompt-lab/evidence/run/12-answer.txt b/examples/prompt-lab/evidence/run/05-answer.txt similarity index 53% rename from examples/prompt-lab/evidence/run/12-answer.txt rename to examples/prompt-lab/evidence/run/05-answer.txt index bf3e8337..b356c5c2 100644 --- a/examples/prompt-lab/evidence/run/12-answer.txt +++ b/examples/prompt-lab/evidence/run/05-answer.txt @@ -1,4 +1,4 @@ -$ npx flows answer --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data 01M366RA1PSQ7295XTPYPV9PXC human-8 yes --by prompt-lab-reviewer -ANSWERED 01M366RA1PSQ7295XTPYPV9PXC human-8 yes -Continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data --local-agent 01M366RA1PSQ7295XTPYPV9PXC +$ npx flows answer --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data 01M36BDEBD2DKTZTX8GTB28DW7 human-26 yes --by prompt-lab-reviewer +ANSWERED 01M36BDEBD2DKTZTX8GTB28DW7 human-26 yes +Continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data --local-agent 01M36BDEBD2DKTZTX8GTB28DW7 exit=0 diff --git a/examples/prompt-lab/evidence/run/05-reviewer-edit.diff b/examples/prompt-lab/evidence/run/05-reviewer-edit.diff deleted file mode 100644 index 3b99d02a..00000000 --- a/examples/prompt-lab/evidence/run/05-reviewer-edit.diff +++ /dev/null @@ -1,10 +0,0 @@ -# reviewer edit: evidence/run/lab/work/sunrise-soc/62c71a0f/commit.json -@@ -1,4 +1,6 @@ - { -- "mode": "all", -- "questions": [] -+ "mode": "except", -+ "questions": [ -+ "ostomy-supplies" -+ ] - } diff --git a/examples/prompt-lab/evidence/run/07-resume.txt b/examples/prompt-lab/evidence/run/06-resume.txt similarity index 60% rename from examples/prompt-lab/evidence/run/07-resume.txt rename to examples/prompt-lab/evidence/run/06-resume.txt index 7f237607..026b85a2 100644 --- a/examples/prompt-lab/evidence/run/07-resume.txt +++ b/examples/prompt-lab/evidence/run/06-resume.txt @@ -1,40 +1,40 @@ -$ npx flows resume --no-observer-link --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data --local-agent 01M3669253ED8QPNX7N7RE7PXF +$ npx flows resume --no-observer-link --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data --local-agent 01M36BDEBD2DKTZTX8GTB28DW7 ○ run-1 (deterministic) 0.00s ✓ run-1 (deterministic) 0.01s completionReason: success ○ llm-2 (llm) 0.00s -✓ llm-2 (llm) 3.59s completionReason: success +✓ llm-2 (llm) 5.30s completionReason: success ○ llm-3 (llm) 0.00s -✓ llm-3 (llm) 4.20s completionReason: success +✓ llm-3 (llm) 4.16s completionReason: success ○ run-4 (deterministic) 0.00s ✓ run-4 (deterministic) 0.01s completionReason: success ○ llm-5 (llm) 0.00s -✓ llm-5 (llm) 3.79s completionReason: success +✓ llm-5 (llm) 3.85s completionReason: success ○ llm-6 (llm) 0.00s -✓ llm-6 (llm) 5.22s completionReason: success +✓ llm-6 (llm) 3.74s completionReason: success ○ run-7 (deterministic) 0.00s ✓ run-7 (deterministic) 0.01s completionReason: success ○ llm-8 (llm) 0.00s -✓ llm-8 (llm) 3.63s completionReason: success +✓ llm-8 (llm) 3.36s completionReason: success ○ run-9 (deterministic) 0.00s ✓ run-9 (deterministic) 0.01s completionReason: success ○ llm-10 (llm) 0.00s -✓ llm-10 (llm) 4.71s completionReason: success +✓ llm-10 (llm) 3.65s completionReason: success ○ llm-11 (llm) 0.00s -✓ llm-11 (llm) 5.29s completionReason: success +✓ llm-11 (llm) 3.82s completionReason: success ○ llm-12 (llm) 0.00s -✓ llm-12 (llm) 4.56s completionReason: success +✓ llm-12 (llm) 3.74s completionReason: success ○ llm-13 (llm) 0.00s -✓ llm-13 (llm) 3.49s completionReason: success +✓ llm-13 (llm) 5.21s completionReason: success ○ llm-14 (llm) 0.00s -✓ llm-14 (llm) 3.87s completionReason: success +✓ llm-14 (llm) 4.68s completionReason: success ○ llm-15 (llm) 0.00s -✓ llm-15 (llm) 3.79s completionReason: success +✓ llm-15 (llm) 3.81s completionReason: success ○ llm-16 (llm) 0.00s -✓ llm-16 (llm) 3.89s completionReason: success +✓ llm-16 (llm) 3.57s completionReason: success ○ llm-17 (llm) 0.00s -✓ llm-17 (llm) 3.67s completionReason: success +✓ llm-17 (llm) 3.89s completionReason: success ○ llm-18 (llm) 0.00s -✓ llm-18 (llm) 3.70s completionReason: success +✓ llm-18 (llm) 3.58s completionReason: success ○ run-19 (deterministic) 0.00s ✓ run-19 (deterministic) 0.01s completionReason: success ○ human-20 (deterministic) 0.00s @@ -44,16 +44,16 @@ $ npx flows resume --no-observer-link --data-dir /private/tmp/claude-501/-Users- ○ run-22 (deterministic) 0.00s ✓ run-22 (deterministic) 0.01s completionReason: success ○ llm-23 (llm) 0.00s -✓ llm-23 (llm) 3.93s completionReason: success +✓ llm-23 (llm) 3.76s completionReason: success ○ run-24 (deterministic) 0.00s ✓ run-24 (deterministic) 0.01s completionReason: success ○ run-25 (deterministic) 0.00s ✓ run-25 (deterministic) 0.01s completionReason: success ○ human-26 (deterministic) 0.00s -✓ human-26 (deterministic) 0.02s completionReason: success +✓ human-26 (deterministic) 0.01s completionReason: success ○ run-27 (deterministic) 0.00s ✓ run-27 (deterministic) 0.07s completionReason: success ○ run-28 (deterministic) 0.00s -✓ run-28 (deterministic) 0.06s completionReason: success -RUN 01M3669253ED8QPNX7N7RE7PXF completed (29 steps) completionReason: success +✓ run-28 (deterministic) 0.07s completionReason: success +RUN 01M36BDEBD2DKTZTX8GTB28DW7 completed (29 steps) completionReason: success exit=0 diff --git a/examples/prompt-lab/evidence/run/07-patient-run.txt b/examples/prompt-lab/evidence/run/07-patient-run.txt new file mode 100644 index 00000000..dd019654 --- /dev/null +++ b/examples/prompt-lab/evidence/run/07-patient-run.txt @@ -0,0 +1,32 @@ +$ npx flows run --no-observer-link --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data --local-agent prompt-lab.flow.ts --input {"job":"patient","reviewer":"prompt-lab-reviewer","lab":"evidence/run/lab","briefId":"gap-ostomy-supplies-soc"} +○ run-1 (deterministic) 0.00s +✓ run-1 (deterministic) 0.07s completionReason: success +○ llm-2 (llm) 0.00s +WAITING [worker_lease] Run "01M36BQ2J79QTAJJSZF42ZRV0A" step "llm-2" (llm) is running under a worker lease until 1790141267834. +↻ llm-2 (llm) 3.72s +WAITING [worker_lease] Run "01M36BQ2J79QTAJJSZF42ZRV0A" step "llm-2" (llm) is running under a worker lease until 1790141277836. +↻ llm-2 (llm) 13.76s +WAITING [worker_lease] Run "01M36BQ2J79QTAJJSZF42ZRV0A" step "llm-2" (llm) is running under a worker lease until 1790141287838. +↻ llm-2 (llm) 23.73s +WAITING [worker_lease] Run "01M36BQ2J79QTAJJSZF42ZRV0A" step "llm-2" (llm) is running under a worker lease until 1790141297840. +↻ llm-2 (llm) 33.74s +WAITING [worker_lease] Run "01M36BQ2J79QTAJJSZF42ZRV0A" step "llm-2" (llm) is running under a worker lease until 1790141307842. +↻ llm-2 (llm) 43.72s +WAITING [worker_lease] Run "01M36BQ2J79QTAJJSZF42ZRV0A" step "llm-2" (llm) is running under a worker lease until 1790141317844. +↻ llm-2 (llm) 53.77s +WAITING [worker_lease] Run "01M36BQ2J79QTAJJSZF42ZRV0A" step "llm-2" (llm) is running under a worker lease until 1790141327845. +↻ llm-2 (llm) 63.77s +WAITING [worker_lease] Run "01M36BQ2J79QTAJJSZF42ZRV0A" step "llm-2" (llm) is running under a worker lease until 1790141337847. +↻ llm-2 (llm) 73.75s +WAITING [worker_lease] Run "01M36BQ2J79QTAJJSZF42ZRV0A" step "llm-2" (llm) is running under a worker lease until 1790141347850. +↻ llm-2 (llm) 83.75s +✓ llm-2 (llm) 84.90s completionReason: success +○ run-3 (deterministic) 0.00s +✓ run-3 (deterministic) 0.06s completionReason: success +○ human-4 (deterministic) 0.00s +⏸ human-4 (human) 0.00s +PARKED [run_parked] Run "01M36BPYVQFDCG2WEP7V7H04X4" is waiting for prompt-lab-reviewer to answer human-4: "Test patient creator · gap-ostomy-supplies-soc for ostomy-supplies (soc visit).\nPlan:\nINVENTED SHELF PATIENT — VARIANT A (the load-bearing one): first-name label \"Verna\". Age band 60-69. No surname, no dates, no MRN, no phone, no address, no facility or clinician names anywhere — the referral packet is from \"the surgical service\" and \"the inpatient WOC nurse\", supplies come from \"the DME supplier\", and all timing is relative (\"about three weeks post-op\", \"since discharge\", \"this visit\").\n\nDIAGNOSES: sigmoid adenocarcinoma s/p open sigmoid colectomy with permanent end colostomy; type 2 diabetes (skin fragility, slow healing); hypertension; obesity with a pendulous lower abdomen and a skin crease below the stoma; osteoarthritis of the hands limiting fine dexterity; post-op anemia.\n\nLIVING SITUATION: lives alone in a one-bedroom second-floor walk-up, no elevator. One adult child works full time and visits on weekends. No home aide authorized. Changes the pouch standing at the bathroom sink, no mirror, cannot see the inferior edge of the stoma over her abdomen when seated.\n\nWHAT THE REFERRAL PACKET SAYS: discharge summary describing an uneventful post-op course; stoma at discharge \"beefy red, budded 2 cm, 32 mm round, peristomal skin intact\"; output \"pasty to formed\"; one supervised pouch change with return demonstration charted as \"partially independent\"; pouching schedule of every 3 days. Starter-kit checklist: 10 one-piece drainable pouches with a PRE-CUT 32 mm barrier, skin barrier wipes, adhesive remover wipes, deodorant drops, and — this is the trap — \"barrier rings x10\". Orders: skilled nursing 2x/week for ostomy care and teaching, WOC consult PRN, DME resupply order placed with no confirmed delivery.\n\nWHAT TODAY'S VISIT NOTES MUST SHOW: stoma now budded only ~0.5 cm and flush-to-retracted at the 4 o'clock position as post-op edema resolved; measures ~25 mm and slightly oval; sits at the lip of the abdominal crease when she sits. Peristomal skin denuded, excoriated and weeping serous moisture across roughly the inferior half of the wafer field, 2-4 cm, with an irregular scalloped border tracking exactly where effluent ran; she reports burning. Explicitly NO satellite lesions, no induration, no purulence, no fever — so this reads as irritant contact dermatitis from effluent, not infection or fungus. Pouch has been on ~5 hours and the adhesive is already lifting along the inferior border with effluent visible under the edge; she has been taping it down with paper tape. Two to three leaks a day; she is changing daily instead of every 3 days. Output looser and higher-volume than the packet describes. Inventory counted WITH the patient and documented item by item: 2 pouches left in the box on the counter, and their pre-cut 32 mm opening now leaves 3-4 mm of bare skin around a 25 mm stoma; barrier wipes nearly full; adhesive remover full; deodorant drops unopened; NO barrier rings anywhere — cabinet, bag and counter searched; no barrier powder, no convex product, no belt.\n\nINTENDED ANSWER: a moldable barrier ring/seal — it fills the crease and the oversized opening, stops effluent contact with denuded skin, and restores wear time. It is the only supply on the shelf that addresses the mechanism rather than a symptom.\n\nDELIBERATE DISTRACTORS (so the question is a real choice, not a lookup): (1) the nearly empty pouch box makes \"pouches\" look urgent, but more pouches at the wrong size leak the same way — resupply is a workflow action, not this visit's supply need; (2) weeping skin invites \"barrier powder/crusting\", but the notes make ongoing effluent contact explicit, so protection beats treatment; (3) a near-flush stoma invites \"convex wafer\" or \"belt\", but neither is in the home, no WOC convexity assessment has happened, and convexity on denuded skin without a ring first is second-line; (4) an answer path that defers to \"WOC referral\" instead of naming a supply should visibly under-answer, since the packet already carries a PRN consult.\n\nDELIBERATE CONFLICTS BETWEEN PACKET AND NOTES, each testing one thing: (C1 inventory) the packet checklist says 10 barrier rings went home; the in-home count finds zero and she says she never saw any — tests whether the answer trusts today's observation over discharge paperwork, and a packet-trusting path lands on \"pouches\" or \"None\". (C2 sizing) 32 mm round in the packet vs ~25 mm oval today, with pre-cut wafers still cut to the discharge size — tests whether the model connects resolving edema to the leak mechanism instead of blaming her technique. (C3 skin) \"peristomal skin intact\" and q3-day changes in the packet vs denuded skin and daily changes today — tests recency. (C4 output) formed/pasty in the packet vs looser and higher-volume today — a quiet detail that makes leakage more corrosive and further undercuts \"just reorder pouches\".\n\nVARIANT B — THE 'NONE' PATH: first-name label \"Roland\", age band 70-79. Mature end colostomy of several years after colectomy for diverticular disease; lives with a spouse who does the changes competently. The referral packet is for something else entirely — skilled nursing after a heart-failure hospitalization for medication management and teaching — and notes the colostomy as long-standing and self-managed, with a stale discharge-planner line claiming \"patient reports running out of pouches\" so the question must still be answered rather than skipped. Visit notes: stoma mature, budded 2 cm, beefy red, on a flat surface, 28 mm and unchanged for years; peristomal skin fully intact, no erythema or denudement; two-piece system worn 4 days with an intact seal, no leaks this month or historically; supply closet inventoried with the patient — 30+ pouches, 20+ wafers already cut to size, a full box of barrier rings, unopened barrier powder, adhesive remover, an unused belt; monthly DME auto-ship arrived complete. Conflict for B mirrors A's in the opposite direction: the stale \"running out of pouches\" note contradicts the counted inventory. That symmetry is the point — the same rule (observed notes over packet) yields \"barrier ring\" in A and \"None\" in B, so neither shortcut (\"always name a supply\", \"always trust the packet\") can pass both.\n\nSHELF MECHANICS: keep packet and visit notes as two separately quotable documents in the shelf's existing envelope shape — the conflicts only test anything if an answer path can cite one against the other. Measurements and counts are the load-bearing detail and must stay exact; everything identifying stays absent.\nYou may edit \"plan\" in evidence/run/lab/work/patients/gap-ostomy-supplies-soc/fbf53055/plan.json first.\nyes = kick generate (Patient QA then locks it onto the shelf; you do not approve the chart); no = leave it on the queue." +Answer with: flows answer --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data 01M36BPYVQFDCG2WEP7V7H04X4 human-4 yes|no +Then continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data --local-agent 01M36BPYVQFDCG2WEP7V7H04X4 +RUN 01M36BPYVQFDCG2WEP7V7H04X4 parked +exit=3 diff --git a/examples/prompt-lab/evidence/run/09-answer.txt b/examples/prompt-lab/evidence/run/08-answer.txt similarity index 53% rename from examples/prompt-lab/evidence/run/09-answer.txt rename to examples/prompt-lab/evidence/run/08-answer.txt index d2b2b071..c01e6b8c 100644 --- a/examples/prompt-lab/evidence/run/09-answer.txt +++ b/examples/prompt-lab/evidence/run/08-answer.txt @@ -1,4 +1,4 @@ -$ npx flows answer --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data 01M366JW806SDNME4SKS4JGWBV human-4 yes --by prompt-lab-reviewer -ANSWERED 01M366JW806SDNME4SKS4JGWBV human-4 yes -Continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data --local-agent 01M366JW806SDNME4SKS4JGWBV +$ npx flows answer --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data 01M36BPYVQFDCG2WEP7V7H04X4 human-4 yes --by prompt-lab-reviewer +ANSWERED 01M36BPYVQFDCG2WEP7V7H04X4 human-4 yes +Continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data --local-agent 01M36BPYVQFDCG2WEP7V7H04X4 exit=0 diff --git a/examples/prompt-lab/evidence/run/08-patient-run.txt b/examples/prompt-lab/evidence/run/08-patient-run.txt deleted file mode 100644 index af19984c..00000000 --- a/examples/prompt-lab/evidence/run/08-patient-run.txt +++ /dev/null @@ -1,28 +0,0 @@ -$ npx flows run --no-observer-link --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data --local-agent prompt-lab.flow.ts --input {"job":"patient","reviewer":"prompt-lab-reviewer","lab":"evidence/run/lab","briefId":"gap-ostomy-supplies"} -○ run-1 (deterministic) 0.00s -✓ run-1 (deterministic) 0.07s completionReason: success -○ llm-2 (llm) 0.00s -WAITING [worker_lease] Run "01M366JZVB6S0MAJJE29NSHHNP" step "llm-2" (llm) is running under a worker lease until 1790135891103. -↻ llm-2 (llm) 3.61s -WAITING [worker_lease] Run "01M366JZVB6S0MAJJE29NSHHNP" step "llm-2" (llm) is running under a worker lease until 1790135901107. -↻ llm-2 (llm) 13.63s -WAITING [worker_lease] Run "01M366JZVB6S0MAJJE29NSHHNP" step "llm-2" (llm) is running under a worker lease until 1790135911114. -↻ llm-2 (llm) 23.65s -WAITING [worker_lease] Run "01M366JZVB6S0MAJJE29NSHHNP" step "llm-2" (llm) is running under a worker lease until 1790135921122. -↻ llm-2 (llm) 33.65s -WAITING [worker_lease] Run "01M366JZVB6S0MAJJE29NSHHNP" step "llm-2" (llm) is running under a worker lease until 1790135931125. -↻ llm-2 (llm) 43.64s -WAITING [worker_lease] Run "01M366JZVB6S0MAJJE29NSHHNP" step "llm-2" (llm) is running under a worker lease until 1790135941128. -↻ llm-2 (llm) 53.67s -WAITING [worker_lease] Run "01M366JZVB6S0MAJJE29NSHHNP" step "llm-2" (llm) is running under a worker lease until 1790135951133. -↻ llm-2 (llm) 63.69s -✓ llm-2 (llm) 66.29s completionReason: success -○ run-3 (deterministic) 0.00s -✓ run-3 (deterministic) 0.07s completionReason: success -○ human-4 (deterministic) 0.00s -⏸ human-4 (human) 0.00s -PARKED [run_parked] Run "01M366JW806SDNME4SKS4JGWBV" is waiting for prompt-lab-reviewer to answer human-4: "Test patient creator · gap-ostomy-supplies for ostomy-supplies.\nPlan:\nINVENTED SHELF PATIENT — \"Marguerite\" (first-name label only; no surname, DOB, MRN, address, phone, or dates anywhere in the chart; all temporal references are relative, e.g. \"post-op day ~12\", \"two visits ago\", \"last week\").\n\nPURPOSE: exercise the workbench question \"Which ostomy supply does the patient need most this visit?\" across all answer paths. Ships as ONE patient with TWO chart versions (A: new/complicated stoma → specific single supply need; B: established/well-healed stoma with full cupboard → 'None'). Two versions of the same characteristics keep the shelf small and make the discriminating evidence obvious in diff.\n\nCHARACTERISTICS (the only demographics): age band 75–84, older adult; female; lives alone in a one-story apartment; adult child nearby providing intermittent help; independent with ADLs pre-op, now limited by post-op fatigue; mild vision impairment (reading glasses, difficulty with fine print on supply boxes); arthritic hands with reduced pinch strength — relevant because it plausibly explains seal-application technique problems and argues AGAINST answering with a cut-to-fit-only supply.\n\nDIAGNOSES: primary — sigmoid colon adenocarcinoma s/p open sigmoid colectomy with end colostomy. Secondary — type 2 diabetes mellitus (non-insulin, oral agent), hypertension, osteoarthritis of the hands, mild protein-calorie malnutrition post-op, and history of chronic sun-damaged/thin skin. Diabetes and thin skin are deliberate context: they make peristomal skin irritation clinically credible and raise the stakes of the answer without themselves naming a supply.\n\nLIVING SITUATION: alone, apartment with a standard bathroom, no caregiver present during pouch changes; adult child visits ~2×/week, does grocery runs but does not do ostomy care; no home aide authorized yet; transportation limited — patient does not drive, so \"just pick some up today\" is not a realistic self-remedy and the supply gap is a real gap.\n\nREFERRAL PACKET (discharge → home health) SAYS:\n- Reason for referral: post-surgical care and new ostomy self-care teaching; skilled nursing, plus dietitian and OT consults pending.\n- Ostomy description: end colostomy, left lower quadrant, matured, stoma ~28 mm round, budded ~1 cm, beefy red and viable at discharge; peristomal skin intact at discharge.\n- Output character: soft-formed to pasty brown stool, moderate volume, ~2–3 emptyings per day; no high-output/liquid pattern documented; flatus present.\n- Stoma age: new — approximately 10–14 days from creation at the time of the first home visit.\n- Supply list sent home (explicit inventory, this is the core evidence surface): 20 two-piece cut-to-fit skin barriers/wafers; 30 drainable pouches with closure clips; 1 box (10) barrier rings/seals; 1 tube ostomy paste; 1 box (50) adhesive-remover wipes; 1 box (30) skin-prep/barrier-film wipes; 1 bottle pouch deodorant; measuring guide and scissors; ostomy belt (1).\n- Teaching status at discharge: \"patient returned demonstration of pouch emptying; barrier change performed by nursing, patient observed only.\"\n- Referral explicitly states \"all supplies provided for 30 days; no anticipated supply need.\" (This line is the hook for the deliberate conflict below.)\n- Reorder pathway: DME supplier assigned, reorder by phone, 3–5 day delivery — i.e., an unmet need today is actionable but not instantly self-solvable.\n\nTODAY'S VISIT NOTES (VERSION A — the complicated chart) MUST SHOW:\n- Stoma assessment: stoma now ~24 mm (expected post-op shrinkage from the referral's 28 mm), still beefy red, viable, budded, at skin level on one edge due to a shallow crease in the left lower quadrant when the patient sits.\n- Peristomal skin: erythematous, moist, weepy denudement in a crescent along the 4-to-8 o'clock inferior aspect, ~2 cm wide, matching where output tracks under the barrier when seated; no candidal satellite lesions, no ulceration, no mucocutaneous separation. Patient reports burning/stinging under the wafer.\n- Seal failure: barrier lasting ~18–24 hours before leakage, versus expected 3–4 days; patient has changed the appliance 3 times in the last 2 days. Undermining of the wafer adhesive noted on removal, with stool tracking onto the denuded skin.\n- Supply inventory ON HAND, counted at the visit (this is what forces a SPECIFIC answer): barrier rings 8 of 10 remaining; wafers 14 remaining; pouches 22 remaining; paste ~¾ tube; adhesive-remover wipes ~40 remaining; ostomy belt unused in the drawer; SKIN-PREP / BARRIER-FILM WIPES: 0 — box empty, patient has been applying the barrier to wet, weepy skin without any protective film and did not know a refill was needed. Also absent from the home entirely: stoma powder / protective powder (never sent, not on the referral list) — so the chart supports a single, concrete, most-needed item rather than a generic \"more supplies.\"\n- Technique observation: patient cuts the wafer opening to the discharge measurement (28 mm) rather than the current 24 mm, leaving exposed skin; arthritic hands make scissor-cutting slow and imprecise. This is documented as a TEACHING need, not a supply need — it is a deliberate distractor that a weak answer will convert into \"needs pre-cut/moldable barriers.\"\n- Output character today: matches the referral (soft-formed, 2–3× daily) — deliberately NOT high-output, so \"needs high-output/drainable-with-spout pouches\" is unsupported.\n- Vitals/systemic: afebrile, no peri-stomal cellulitis, blood glucose mildly elevated; pain 3/10 burning at the skin only.\n- Patient goal stated in their own words: \"I want it to stay on so I can go to my grandchild's recital without worrying.\"\n\nTHE DEFENSIBLE ANSWER in Version A: the skin-prep / barrier-film wipes (the exhausted item), because the failing link in the chain is unprotected, weeping peristomal skin under an adhesive barrier — rings are on hand and already in use, pouches and wafers are stocked, output does not justify a different pouch system, and the wafer-sizing error is a teaching correction, not a purchase. A close-second defensible answer (stoma/protective powder, never supplied) is intentionally reachable, so graders can distinguish \"picked the empty box\" from \"reasoned about the crusting technique the weepy skin actually needs.\"\n\nVERSION B — the 'None' chart (same patient characteristics, later state; or a second shelf patient in the same age band if the shelf prefers distinct records):\n- Stoma age: established, ~14 months; stoma ~24 mm, stable size, beefy red, budded, no crease interference; patient uses a moldable/pre-sized barrier.\n- Peristomal skin: intact, no erythema, no denudement, no maceration; skin described as fully healed with no breakdown anywhere in the visit note.\n- Wear time: 4 days consistently, no leakage reported since the last two visits; patient independently empties and changes, returns demonstration flawlessly.\n- Supply inventory: full cupboard — wafers 18, pouches 40, barrier rings 9, skin-prep wipes 45, adhesive remover 50, paste ~full, powder 1 unopened bottle, belt available; 30-day reorder already placed with the DME supplier and confirmed in transit.\n- Output character: soft-formed, predictable, 1–2 emptyings daily, consistent with the referral.\n- Visit purpose: routine reassessment/recertification, no complaint.\n- The defensible answer is 'None' — no supply is deficient, no skin or seal problem creates a need. Version B also carries one benign non-supply need (a dietitian question about gas-producing foods) so 'None' has to be chosen on supply grounds, not because the chart is empty.\n\nDELIBERATE REFERRAL-vs-NOTES CONFLICTS (the part that tests the question rather than pattern-matching):\n1. STOMA SIZE DRIFT: referral says 28 mm; today's measurement is 24 mm. A model that trusts the referral packet will endorse the patient's 28 mm cut and miss that exposed skin is why the seal fails. The correct read is that current assessment overrides discharge documentation.\n2. \"NO ANTICIPATED SUPPLY NEED\": the referral asserts 30 days of supplies and no need; the visit count proves one box is at zero and one item was never supplied at all. A model that defers to the referral's blanket statement will answer 'None' on a chart that clearly supports a specific item — this is the primary trap separating Version A from Version B.\n3. PERISTOMAL SKIN STATUS: referral says \"peristomal skin intact at discharge\"; today's note documents weepy denudement. Same field, opposite value, with the visit note being the current truth.\n4. SUPPLY COUNT vs SUPPLY LIST: the referral's list includes skin-prep wipes, so a list-reading model will conclude the patient has them; only the counted inventory in today's note reveals the box is empty. The evidence needed to answer lives in the notes, not the packet.\n5. WEAR-TIME EXPECTATION: referral implies 3–4 day wear; notes show 18–24 hours. The gap is the clinical signal that something under the barrier — skin, not the pouch — is failing.\n\nGUARDRAILS FOR AUTHORING: no surnames, no dates or date-like strings (use relative intervals only), no MRN/account/encounter numbers, no phone numbers, no addresses, no facility or clinician names (use \"the discharging hospital\", \"the home health nurse\", \"the DME supplier\"), no insurance IDs. Keep the referral packet and visit note as separate documents so the conflict is only visible when both are read. Every fact that drives the answer — stoma measurement, skin condition, wear time, and the counted inventory with explicit zeros — must appear verbatim in the chart text, so a grader can point to the line that justifies each answer path.\nYou may edit \"plan\" in evidence/run/lab/work/patients/gap-ostomy-supplies/plan.json first.\nyes = kick generate (Patient QA then locks it onto the shelf; you do not approve the chart); no = leave it on the queue." -Answer with: flows answer --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data 01M366JW806SDNME4SKS4JGWBV human-4 yes|no -Then continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data --local-agent 01M366JW806SDNME4SKS4JGWBV -RUN 01M366JW806SDNME4SKS4JGWBV parked -exit=3 diff --git a/examples/prompt-lab/evidence/run/09-resume.txt b/examples/prompt-lab/evidence/run/09-resume.txt new file mode 100644 index 00000000..6f48ee94 --- /dev/null +++ b/examples/prompt-lab/evidence/run/09-resume.txt @@ -0,0 +1,33 @@ +$ npx flows resume --no-observer-link --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data --local-agent 01M36BPYVQFDCG2WEP7V7H04X4 +○ run-1 (deterministic) 0.00s +✓ run-1 (deterministic) 0.01s completionReason: success +○ llm-2 (llm) 0.00s +✓ llm-2 (llm) 4.25s completionReason: success +○ run-3 (deterministic) 0.00s +✓ run-3 (deterministic) 0.01s completionReason: success +○ human-4 (deterministic) 0.00s +✓ human-4 (deterministic) 0.01s completionReason: success +○ run-5 (deterministic) 0.00s +✓ run-5 (deterministic) 0.06s completionReason: success +○ llm-6 (llm) 0.00s +WAITING [worker_lease] Run "01M36BSWPMV69TB4YAE3K4SA7V" step "llm-6" (llm) is running under a worker lease until 1790141360145. +↻ llm-6 (llm) 4.23s +WAITING [worker_lease] Run "01M36BSWPMV69TB4YAE3K4SA7V" step "llm-6" (llm) is running under a worker lease until 1790141370147. +↻ llm-6 (llm) 14.27s +WAITING [worker_lease] Run "01M36BSWPMV69TB4YAE3K4SA7V" step "llm-6" (llm) is running under a worker lease until 1790141380150. +↻ llm-6 (llm) 24.24s +WAITING [worker_lease] Run "01M36BSWPMV69TB4YAE3K4SA7V" step "llm-6" (llm) is running under a worker lease until 1790141390152. +↻ llm-6 (llm) 34.24s +✓ llm-6 (llm) 41.18s completionReason: success +○ llm-7 (llm) 0.00s +WAITING [worker_lease] Run "01M36BV4YJ8GEC3YTNNHH5QCQJ" step "llm-7" (llm) is running under a worker lease until 1790141401349. +↻ llm-7 (llm) 4.25s +WAITING [worker_lease] Run "01M36BV4YJ8GEC3YTNNHH5QCQJ" step "llm-7" (llm) is running under a worker lease until 1790141411351. +↻ llm-7 (llm) 14.27s +WAITING [worker_lease] Run "01M36BV4YJ8GEC3YTNNHH5QCQJ" step "llm-7" (llm) is running under a worker lease until 1790141421353. +↻ llm-7 (llm) 24.28s +✓ llm-7 (llm) 24.33s completionReason: success +○ run-8 (deterministic) 0.00s +✓ run-8 (deterministic) 0.07s completionReason: success +RUN 01M36BPYVQFDCG2WEP7V7H04X4 completed (9 steps) completionReason: success +exit=0 diff --git a/examples/prompt-lab/evidence/run/10-job2-run.txt b/examples/prompt-lab/evidence/run/10-job2-run.txt new file mode 100644 index 00000000..0c283c5f --- /dev/null +++ b/examples/prompt-lab/evidence/run/10-job2-run.txt @@ -0,0 +1,30 @@ +$ npx flows run --no-observer-link --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data --local-agent prompt-lab.flow.ts --input {"job":"fix","reviewer":"prompt-lab-reviewer","lab":"evidence/run/lab","issueId":"config-sunrise-wound-status-8a366995"} +○ run-1 (deterministic) 0.00s +✓ run-1 (deterministic) 0.07s completionReason: success +○ llm-2 (llm) 0.00s +WAITING [worker_lease] Run "01M36BVY7T3HKS7YM1DWYCC9XT" step "llm-2" (llm) is running under a worker lease until 1790141427245. +↻ llm-2 (llm) 3.87s +WAITING [worker_lease] Run "01M36BVY7T3HKS7YM1DWYCC9XT" step "llm-2" (llm) is running under a worker lease until 1790141437248. +↻ llm-2 (llm) 13.90s +✓ llm-2 (llm) 16.01s completionReason: success +○ llm-3 (llm) 0.00s +WAITING [worker_lease] Run "01M36BWGDVRNR4NGN1GREEXR2H" step "llm-3" (llm) is running under a worker lease until 1790141445870. +↻ llm-3 (llm) 6.49s +✓ llm-3 (llm) 10.72s completionReason: success +○ llm-4 (llm) 0.00s +WAITING [worker_lease] Run "01M36BWR19D9GP7XWB2HX39F20" step "llm-4" (llm) is running under a worker lease until 1790141453661. +↻ llm-4 (llm) 3.56s +✓ llm-4 (llm) 7.99s completionReason: success +○ llm-5 (llm) 0.00s +WAITING [worker_lease] Run "01M36BX06ZZAG345PMZD9TT43T" step "llm-5" (llm) is running under a worker lease until 1790141462034. +↻ llm-5 (llm) 3.94s +✓ llm-5 (llm) 8.27s completionReason: success +○ run-6 (deterministic) 0.00s +✓ run-6 (deterministic) 0.06s completionReason: success +○ human-7 (deterministic) 0.00s +⏸ human-7 (human) 0.00s +PARKED [run_parked] Run "01M36BVTCK80EV8P4FWA8WZZHK" is waiting for prompt-lab-reviewer to answer human-7: "Question workbench · wound-status (issue config-sunrise-wound-status-8a366995): current-prompt outputs on 3 shelf patient(s) — jordan: No wound/High, pat: Ongoing/Medium, riley: No wound/High.\nSet gold in evidence/run/lab/work/q-wound-status/cd33704e/grid.json (\"target\" per row; prefilled from persisted gold or the config-level target). Your review persists as gold; later AI runs never overwrite it.\nyes = persist gold and run the iterator; no = stop." +Answer with: flows answer --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data 01M36BVTCK80EV8P4FWA8WZZHK human-7 yes|no +Then continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data --local-agent 01M36BVTCK80EV8P4FWA8WZZHK +RUN 01M36BVTCK80EV8P4FWA8WZZHK parked +exit=3 diff --git a/examples/prompt-lab/evidence/run/10-resume.txt b/examples/prompt-lab/evidence/run/10-resume.txt deleted file mode 100644 index cfb6db2f..00000000 --- a/examples/prompt-lab/evidence/run/10-resume.txt +++ /dev/null @@ -1,41 +0,0 @@ -$ npx flows resume --no-observer-link --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data --local-agent 01M366JW806SDNME4SKS4JGWBV -○ run-1 (deterministic) 0.00s -✓ run-1 (deterministic) 0.01s completionReason: success -○ llm-2 (llm) 0.00s -✓ llm-2 (llm) 4.02s completionReason: success -○ run-3 (deterministic) 0.00s -✓ run-3 (deterministic) 0.01s completionReason: success -○ human-4 (deterministic) 0.00s -✓ human-4 (deterministic) 0.02s completionReason: success -○ run-5 (deterministic) 0.00s -✓ run-5 (deterministic) 0.06s completionReason: success -○ llm-6 (llm) 0.00s -WAITING [worker_lease] Run "01M366N6CC52X7F8GZTYGHCPYA" step "llm-6" (llm) is running under a worker lease until 1790135963330. -↻ llm-6 (llm) 4.09s -WAITING [worker_lease] Run "01M366N6CC52X7F8GZTYGHCPYA" step "llm-6" (llm) is running under a worker lease until 1790135973335. -↻ llm-6 (llm) 14.11s -WAITING [worker_lease] Run "01M366N6CC52X7F8GZTYGHCPYA" step "llm-6" (llm) is running under a worker lease until 1790135983336. -↻ llm-6 (llm) 24.10s -WAITING [worker_lease] Run "01M366N6CC52X7F8GZTYGHCPYA" step "llm-6" (llm) is running under a worker lease until 1790135993341. -↻ llm-6 (llm) 34.14s -✓ llm-6 (llm) 37.72s completionReason: success -○ llm-7 (llm) 0.00s -WAITING [worker_lease] Run "01M366PAR4A4JPY4W5NMYRTWCM" step "llm-7" (llm) is running under a worker lease until 1790136000572. -↻ llm-7 (llm) 3.61s -WAITING [worker_lease] Run "01M366PAR4A4JPY4W5NMYRTWCM" step "llm-7" (llm) is running under a worker lease until 1790136010576. -↻ llm-7 (llm) 13.62s -WAITING [worker_lease] Run "01M366PAR4A4JPY4W5NMYRTWCM" step "llm-7" (llm) is running under a worker lease until 1790136020579. -↻ llm-7 (llm) 23.65s -WAITING [worker_lease] Run "01M366PAR4A4JPY4W5NMYRTWCM" step "llm-7" (llm) is running under a worker lease until 1790136030584. -↻ llm-7 (llm) 33.66s -WAITING [worker_lease] Run "01M366PAR4A4JPY4W5NMYRTWCM" step "llm-7" (llm) is running under a worker lease until 1790136040588. -↻ llm-7 (llm) 43.67s -WAITING [worker_lease] Run "01M366PAR4A4JPY4W5NMYRTWCM" step "llm-7" (llm) is running under a worker lease until 1790136050595. -↻ llm-7 (llm) 53.68s -WAITING [worker_lease] Run "01M366PAR4A4JPY4W5NMYRTWCM" step "llm-7" (llm) is running under a worker lease until 1790136060600. -↻ llm-7 (llm) 63.64s -✓ llm-7 (llm) 67.46s completionReason: success -○ run-8 (deterministic) 0.00s -✓ run-8 (deterministic) 0.06s completionReason: success -RUN 01M366JW806SDNME4SKS4JGWBV completed (9 steps) completionReason: success -exit=0 diff --git a/examples/prompt-lab/evidence/run/06-answer.txt b/examples/prompt-lab/evidence/run/11-answer.txt similarity index 53% rename from examples/prompt-lab/evidence/run/06-answer.txt rename to examples/prompt-lab/evidence/run/11-answer.txt index b5813218..a0e3c59b 100644 --- a/examples/prompt-lab/evidence/run/06-answer.txt +++ b/examples/prompt-lab/evidence/run/11-answer.txt @@ -1,4 +1,4 @@ -$ npx flows answer --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data 01M3669253ED8QPNX7N7RE7PXF human-26 yes --by prompt-lab-reviewer -ANSWERED 01M3669253ED8QPNX7N7RE7PXF human-26 yes -Continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data --local-agent 01M3669253ED8QPNX7N7RE7PXF +$ npx flows answer --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data 01M36BVTCK80EV8P4FWA8WZZHK human-7 yes --by prompt-lab-reviewer +ANSWERED 01M36BVTCK80EV8P4FWA8WZZHK human-7 yes +Continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data --local-agent 01M36BVTCK80EV8P4FWA8WZZHK exit=0 diff --git a/examples/prompt-lab/evidence/run/11-job2-run.txt b/examples/prompt-lab/evidence/run/11-job2-run.txt deleted file mode 100644 index bb3645c4..00000000 --- a/examples/prompt-lab/evidence/run/11-job2-run.txt +++ /dev/null @@ -1,34 +0,0 @@ -$ npx flows run --no-observer-link --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data --local-agent prompt-lab.flow.ts --input {"job":"fix","reviewer":"prompt-lab-reviewer","lab":"evidence/run/lab","issueId":"config-sunrise-wound-status-ce89aea7"} -○ run-1 (deterministic) 0.00s -✓ run-1 (deterministic) 0.07s completionReason: success -○ llm-2 (llm) 0.00s -WAITING [worker_lease] Run "01M366REV3VYNC5GG6AA7TX0FV" step "llm-2" (llm) is running under a worker lease until 1790136070295. -↻ llm-2 (llm) 4.84s -✓ llm-2 (llm) 8.85s completionReason: success -○ llm-3 (llm) 0.00s -WAITING [worker_lease] Run "01M366RPB0C97EW0E7MC134S9V" step "llm-3" (llm) is running under a worker lease until 1790136077971. -↻ llm-3 (llm) 3.66s -✓ llm-3 (llm) 8.60s completionReason: success -○ llm-4 (llm) 0.00s -WAITING [worker_lease] Run "01M366RYM9Z7595ZX07VTH2WAX" step "llm-4" (llm) is running under a worker lease until 1790136086459. -↻ llm-4 (llm) 3.55s -✓ llm-4 (llm) 10.93s completionReason: success -○ llm-5 (llm) 0.00s -WAITING [worker_lease] Run "01M366S9ZG10B4C2C9JE0YAK70" step "llm-5" (llm) is running under a worker lease until 1790136098083. -↻ llm-5 (llm) 4.25s -WAITING [worker_lease] Run "01M366S9ZG10B4C2C9JE0YAK70" step "llm-5" (llm) is running under a worker lease until 1790136108084. -↻ llm-5 (llm) 14.26s -✓ llm-5 (llm) 14.64s completionReason: success -○ llm-6 (llm) 0.00s -WAITING [worker_lease] Run "01M366SQKD7VZ06Z89B990JP84" step "llm-6" (llm) is running under a worker lease until 1790136112032. -↻ llm-6 (llm) 3.55s -✓ llm-6 (llm) 8.11s completionReason: success -○ run-7 (deterministic) 0.00s -✓ run-7 (deterministic) 0.06s completionReason: success -○ human-8 (deterministic) 0.00s -⏸ human-8 (human) 0.00s -PARKED [run_parked] Run "01M366RA1PSQ7295XTPYPV9PXC" is waiting for prompt-lab-reviewer to answer human-8: "Question workbench · wound-status (issue config-sunrise-wound-status-ce89aea7): current-prompt outputs on 4 shelf patient(s) — jordan: No wound/High, marguerite: Ongoing/Medium, pat: Ongoing/High, riley: No wound/High.\nSet gold in evidence/run/lab/work/q-wound-status/d61687f9/grid.json (\"target\" per row; prefilled from persisted gold or the config-level target). Your review persists as gold; later AI runs never overwrite it.\nyes = persist gold and run the iterator; no = stop." -Answer with: flows answer --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data 01M366RA1PSQ7295XTPYPV9PXC human-8 yes|no -Then continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data --local-agent 01M366RA1PSQ7295XTPYPV9PXC -RUN 01M366RA1PSQ7295XTPYPV9PXC parked -exit=3 diff --git a/examples/prompt-lab/evidence/run/12-resume.txt b/examples/prompt-lab/evidence/run/12-resume.txt new file mode 100644 index 00000000..482a0f11 --- /dev/null +++ b/examples/prompt-lab/evidence/run/12-resume.txt @@ -0,0 +1,68 @@ +$ npx flows resume --no-observer-link --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data --local-agent 01M36BVTCK80EV8P4FWA8WZZHK +○ run-1 (deterministic) 0.00s +✓ run-1 (deterministic) 0.01s completionReason: success +○ llm-2 (llm) 0.00s +✓ llm-2 (llm) 3.82s completionReason: success +○ llm-3 (llm) 0.00s +✓ llm-3 (llm) 4.13s completionReason: success +○ llm-4 (llm) 0.00s +✓ llm-4 (llm) 3.87s completionReason: success +○ llm-5 (llm) 0.00s +✓ llm-5 (llm) 4.40s completionReason: success +○ run-6 (deterministic) 0.00s +✓ run-6 (deterministic) 0.01s completionReason: success +○ human-7 (deterministic) 0.00s +✓ human-7 (deterministic) 0.01s completionReason: success +○ run-8 (deterministic) 0.00s +✓ run-8 (deterministic) 0.06s completionReason: success +○ run-9 (deterministic) 0.00s +✓ run-9 (deterministic) 0.07s completionReason: success +○ llm-10 (llm) 0.00s +WAITING [worker_lease] Run "01M36BXT4CQ15K7D7BGKN0WT5G" step "llm-10" (llm) is running under a worker lease until 1790141488575. +↻ llm-10 (llm) 3.83s +WAITING [worker_lease] Run "01M36BXT4CQ15K7D7BGKN0WT5G" step "llm-10" (llm) is running under a worker lease until 1790141498577. +↻ llm-10 (llm) 13.87s +✓ llm-10 (llm) 19.22s completionReason: success +○ llm-11 (llm) 0.00s +WAITING [worker_lease] Run "01M36BYEV7RNQ0ZRM8NPD94ZS6" step "llm-11" (llm) is running under a worker lease until 1790141509788. +↻ llm-11 (llm) 5.83s +WAITING [worker_lease] Run "01M36BYEV7RNQ0ZRM8NPD94ZS6" step "llm-11" (llm) is running under a worker lease until 1790141519790. +↻ llm-11 (llm) 15.85s +✓ llm-11 (llm) 22.17s completionReason: success +○ llm-12 (llm) 0.00s +WAITING [worker_lease] Run "01M36BZ2JR3WBGG3FETBMR66FX" step "llm-12" (llm) is running under a worker lease until 1790141530006. +↻ llm-12 (llm) 3.88s +WAITING [worker_lease] Run "01M36BZ2JR3WBGG3FETBMR66FX" step "llm-12" (llm) is running under a worker lease until 1790141540009. +↻ llm-12 (llm) 13.91s +WAITING [worker_lease] Run "01M36BZ2JR3WBGG3FETBMR66FX" step "llm-12" (llm) is running under a worker lease until 1790141550011. +↻ llm-12 (llm) 23.92s +✓ llm-12 (llm) 31.46s completionReason: success +○ llm-13 (llm) 0.00s +WAITING [worker_lease] Run "01M36C01XRV3F9DKHN4JWCV3XW" step "llm-13" (llm) is running under a worker lease until 1790141562091. +↻ llm-13 (llm) 4.50s +WAITING [worker_lease] Run "01M36C01XRV3F9DKHN4JWCV3XW" step "llm-13" (llm) is running under a worker lease until 1790141572092. +↻ llm-13 (llm) 14.51s +✓ llm-13 (llm) 15.14s completionReason: success +○ run-14 (deterministic) 0.00s +✓ run-14 (deterministic) 0.06s completionReason: success +○ llm-15 (llm) 0.00s +WAITING [worker_lease] Run "01M36C0GPZTJGQTMD14Y6TV9WG" step "llm-15" (llm) is running under a worker lease until 1790141577235. +↻ llm-15 (llm) 4.44s +✓ llm-15 (llm) 8.57s completionReason: success +○ llm-16 (llm) 0.00s +WAITING [worker_lease] Run "01M36C0RKC7NK9XXY5AGQ74TYH" step "llm-16" (llm) is running under a worker lease until 1790141585311. +↻ llm-16 (llm) 3.95s +✓ llm-16 (llm) 8.04s completionReason: success +○ llm-17 (llm) 0.00s +WAITING [worker_lease] Run "01M36C108ACT174RZXCWBQCGR5" step "llm-17" (llm) is running under a worker lease until 1790141593150. +↻ llm-17 (llm) 3.74s +✓ llm-17 (llm) 8.03s completionReason: success +○ run-18 (deterministic) 0.00s +✓ run-18 (deterministic) 0.07s completionReason: success +○ human-19 (deterministic) 0.00s +⏸ human-19 (human) 0.00s +PARKED [run_parked] Run "01M36BVTCK80EV8P4FWA8WZZHK" is waiting for prompt-lab-reviewer to answer human-19: "Re-run and score · wound-status, new prompt p-wound-status-f1305d28:\n3 of 3 golded patients worked (100%)\n jordan: gold No wound · new run No wound · worked\n pat: gold Ongoing · new run Ongoing · worked\n riley: gold No wound · new run No wound · worked\nSHARED: marking done changes this prompt for every agency that uses it: harbor, maple, sunrise.\nRead the new prompt in evidence/run/lab/bank.json and the outputs in evidence/run/lab/work/q-wound-status/cd33704e/score.json.\nyes = mark done (live in Apricot now, no promote step); no = keep it as a draft and iterate again in a new run." +Answer with: flows answer --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data 01M36BVTCK80EV8P4FWA8WZZHK human-19 yes|no +Then continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data --local-agent 01M36BVTCK80EV8P4FWA8WZZHK +RUN 01M36BVTCK80EV8P4FWA8WZZHK parked +exit=3 diff --git a/examples/prompt-lab/evidence/run/14-answer.txt b/examples/prompt-lab/evidence/run/13-answer.txt similarity index 53% rename from examples/prompt-lab/evidence/run/14-answer.txt rename to examples/prompt-lab/evidence/run/13-answer.txt index 070992b6..6bd71893 100644 --- a/examples/prompt-lab/evidence/run/14-answer.txt +++ b/examples/prompt-lab/evidence/run/13-answer.txt @@ -1,4 +1,4 @@ -$ npx flows answer --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data 01M366RA1PSQ7295XTPYPV9PXC human-21 yes --by prompt-lab-reviewer -ANSWERED 01M366RA1PSQ7295XTPYPV9PXC human-21 yes -Continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data --local-agent 01M366RA1PSQ7295XTPYPV9PXC +$ npx flows answer --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data 01M36BVTCK80EV8P4FWA8WZZHK human-19 yes --by prompt-lab-reviewer +ANSWERED 01M36BVTCK80EV8P4FWA8WZZHK human-19 yes +Continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data --local-agent 01M36BVTCK80EV8P4FWA8WZZHK exit=0 diff --git a/examples/prompt-lab/evidence/run/13-resume.txt b/examples/prompt-lab/evidence/run/13-resume.txt deleted file mode 100644 index 2698a843..00000000 --- a/examples/prompt-lab/evidence/run/13-resume.txt +++ /dev/null @@ -1,80 +0,0 @@ -$ npx flows resume --no-observer-link --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data --local-agent 01M366RA1PSQ7295XTPYPV9PXC -○ run-1 (deterministic) 0.00s -✓ run-1 (deterministic) 0.01s completionReason: success -○ llm-2 (llm) 0.00s -✓ llm-2 (llm) 4.15s completionReason: success -○ llm-3 (llm) 0.00s -✓ llm-3 (llm) 3.86s completionReason: success -○ llm-4 (llm) 0.00s -✓ llm-4 (llm) 3.81s completionReason: success -○ llm-5 (llm) 0.00s -✓ llm-5 (llm) 3.66s completionReason: success -○ llm-6 (llm) 0.00s -✓ llm-6 (llm) 3.89s completionReason: success -○ run-7 (deterministic) 0.00s -✓ run-7 (deterministic) 0.01s completionReason: success -○ human-8 (deterministic) 0.00s -✓ human-8 (deterministic) 0.05s completionReason: success -○ run-9 (deterministic) 0.00s -✓ run-9 (deterministic) 0.06s completionReason: success -○ run-10 (deterministic) 0.00s -✓ run-10 (deterministic) 0.08s completionReason: success -○ llm-11 (llm) 0.00s -WAITING [worker_lease] Run "01M366TMJA525EKZT8BXXV264M" step "llm-11" (llm) is running under a worker lease until 1790136141696. -↻ llm-11 (llm) 4.28s -WAITING [worker_lease] Run "01M366TMJA525EKZT8BXXV264M" step "llm-11" (llm) is running under a worker lease until 1790136151698. -↻ llm-11 (llm) 14.31s -WAITING [worker_lease] Run "01M366TMJA525EKZT8BXXV264M" step "llm-11" (llm) is running under a worker lease until 1790136161701. -↻ llm-11 (llm) 24.31s -✓ llm-11 (llm) 25.06s completionReason: success -○ llm-12 (llm) 0.00s -WAITING [worker_lease] Run "01M366VCMJ8MV2Y2S98KGKD9Z8" step "llm-12" (llm) is running under a worker lease until 1790136166341. -↻ llm-12 (llm) 3.87s -WAITING [worker_lease] Run "01M366VCMJ8MV2Y2S98KGKD9Z8" step "llm-12" (llm) is running under a worker lease until 1790136176345. -↻ llm-12 (llm) 13.88s -✓ llm-12 (llm) 22.38s completionReason: success -○ llm-13 (llm) 0.00s -WAITING [worker_lease] Run "01M366W2NG4G1HNGMZC3ZYBQNN" step "llm-13" (llm) is running under a worker lease until 1790136188900. -↻ llm-13 (llm) 4.04s -WAITING [worker_lease] Run "01M366W2NG4G1HNGMZC3ZYBQNN" step "llm-13" (llm) is running under a worker lease until 1790136198904. -↻ llm-13 (llm) 14.09s -WAITING [worker_lease] Run "01M366W2NG4G1HNGMZC3ZYBQNN" step "llm-13" (llm) is running under a worker lease until 1790136208907. -↻ llm-13 (llm) 24.07s -WAITING [worker_lease] Run "01M366W2NG4G1HNGMZC3ZYBQNN" step "llm-13" (llm) is running under a worker lease until 1790136218909. -↻ llm-13 (llm) 34.05s -✓ llm-13 (llm) 35.55s completionReason: success -○ llm-14 (llm) 0.00s -WAITING [worker_lease] Run "01M366X6H9SM8WVMQD4DJ89QMM" step "llm-14" (llm) is running under a worker lease until 1790136225632. -↻ llm-14 (llm) 5.23s -WAITING [worker_lease] Run "01M366X6H9SM8WVMQD4DJ89QMM" step "llm-14" (llm) is running under a worker lease until 1790136235648. -↻ llm-14 (llm) 15.26s -WAITING [worker_lease] Run "01M366X6H9SM8WVMQD4DJ89QMM" step "llm-14" (llm) is running under a worker lease until 1790136245651. -↻ llm-14 (llm) 25.25s -✓ llm-14 (llm) 32.03s completionReason: success -○ run-15 (deterministic) 0.00s -✓ run-15 (deterministic) 0.09s completionReason: success -○ llm-16 (llm) 0.00s -WAITING [worker_lease] Run "01M366Y64AATPSDNA71NA21N1V" step "llm-16" (llm) is running under a worker lease until 1790136257991. -↻ llm-16 (llm) 5.45s -✓ llm-16 (llm) 10.43s completionReason: success -○ llm-17 (llm) 0.00s -WAITING [worker_lease] Run "01M366YERVVD3169AH4H5C304R" step "llm-17" (llm) is running under a worker lease until 1790136266830. -↻ llm-17 (llm) 3.87s -✓ llm-17 (llm) 9.53s completionReason: success -○ llm-18 (llm) 0.00s -WAITING [worker_lease] Run "01M366YS9B14N64W0YHMB5CBQV" step "llm-18" (llm) is running under a worker lease until 1790136277601. -↻ llm-18 (llm) 5.10s -✓ llm-18 (llm) 10.16s completionReason: success -○ llm-19 (llm) 0.00s -WAITING [worker_lease] Run "01M366Z2KF5X7R6ZFNM653ETT0" step "llm-19" (llm) is running under a worker lease until 1790136287140. -↻ llm-19 (llm) 4.48s -✓ llm-19 (llm) 8.99s completionReason: success -○ run-20 (deterministic) 0.00s -✓ run-20 (deterministic) 0.06s completionReason: success -○ human-21 (deterministic) 0.00s -⏸ human-21 (human) 0.00s -PARKED [run_parked] Run "01M366RA1PSQ7295XTPYPV9PXC" is waiting for prompt-lab-reviewer to answer human-21: "Re-run and score · wound-status, new prompt p-wound-status-f41f7ca1:\n4 of 4 golded patients worked (100%)\n jordan: gold No wound · new run No wound · worked\n marguerite: gold Ongoing · new run Ongoing · worked\n pat: gold Ongoing · new run Ongoing · worked\n riley: gold No wound · new run No wound · worked\nSHARED: marking done changes this prompt for every agency that uses it: harbor, maple, sunrise.\nRead the new prompt in evidence/run/lab/bank.json and the outputs in evidence/run/lab/work/q-wound-status/d61687f9/score.json.\nyes = mark done (live in Apricot now, no promote step); no = keep it as a draft and iterate again in a new run." -Answer with: flows answer --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data 01M366RA1PSQ7295XTPYPV9PXC human-21 yes|no -Then continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data --local-agent 01M366RA1PSQ7295XTPYPV9PXC -RUN 01M366RA1PSQ7295XTPYPV9PXC parked -exit=3 diff --git a/examples/prompt-lab/evidence/run/14-resume.txt b/examples/prompt-lab/evidence/run/14-resume.txt new file mode 100644 index 00000000..0f29a0e9 --- /dev/null +++ b/examples/prompt-lab/evidence/run/14-resume.txt @@ -0,0 +1,45 @@ +$ npx flows resume --no-observer-link --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data --local-agent 01M36BVTCK80EV8P4FWA8WZZHK +○ run-1 (deterministic) 0.00s +✓ run-1 (deterministic) 0.01s completionReason: success +○ llm-2 (llm) 0.00s +✓ llm-2 (llm) 4.01s completionReason: success +○ llm-3 (llm) 0.00s +✓ llm-3 (llm) 3.67s completionReason: success +○ llm-4 (llm) 0.00s +✓ llm-4 (llm) 5.00s completionReason: success +○ llm-5 (llm) 0.00s +✓ llm-5 (llm) 3.61s completionReason: success +○ run-6 (deterministic) 0.00s +✓ run-6 (deterministic) 0.01s completionReason: success +○ human-7 (deterministic) 0.00s +✓ human-7 (deterministic) 0.01s completionReason: success +○ run-8 (deterministic) 0.00s +✓ run-8 (deterministic) 0.01s completionReason: success +○ run-9 (deterministic) 0.00s +✓ run-9 (deterministic) 0.01s completionReason: success +○ llm-10 (llm) 0.00s +✓ llm-10 (llm) 3.56s completionReason: success +○ llm-11 (llm) 0.00s +✓ llm-11 (llm) 3.87s completionReason: success +○ llm-12 (llm) 0.00s +✓ llm-12 (llm) 4.36s completionReason: success +○ llm-13 (llm) 0.00s +✓ llm-13 (llm) 3.70s completionReason: success +○ run-14 (deterministic) 0.00s +✓ run-14 (deterministic) 0.01s completionReason: success +○ llm-15 (llm) 0.00s +✓ llm-15 (llm) 3.73s completionReason: success +○ llm-16 (llm) 0.00s +✓ llm-16 (llm) 3.29s completionReason: success +○ llm-17 (llm) 0.00s +✓ llm-17 (llm) 3.76s completionReason: success +○ run-18 (deterministic) 0.00s +✓ run-18 (deterministic) 0.01s completionReason: success +○ human-19 (deterministic) 0.00s +✓ human-19 (deterministic) 0.01s completionReason: success +○ run-20 (deterministic) 0.00s +✓ run-20 (deterministic) 0.06s completionReason: success +○ run-21 (deterministic) 0.00s +✓ run-21 (deterministic) 0.06s completionReason: success +RUN 01M36BVTCK80EV8P4FWA8WZZHK completed (22 steps) completionReason: success +exit=0 diff --git a/examples/prompt-lab/evidence/run/16-lab-state.txt b/examples/prompt-lab/evidence/run/15-lab-state.txt similarity index 65% rename from examples/prompt-lab/evidence/run/16-lab-state.txt rename to examples/prompt-lab/evidence/run/15-lab-state.txt index 9d6a1a3e..e0534cf9 100644 --- a/examples/prompt-lab/evidence/run/16-lab-state.txt +++ b/examples/prompt-lab/evidence/run/15-lab-state.txt @@ -5,12 +5,12 @@ console.log('shelf', require('fs').readdirSync('evidence/run/lab/shelf').join(' console.log('issues', JSON.stringify(require('./evidence/run/lab/queue/issues.json').map(i=>[i.id,i.status]))); console.log('patient briefs', JSON.stringify(require('./evidence/run/lab/queue/patient-briefs.json').map(i=>[i.id,i.status]))); console.log('gold', Object.keys(require('./evidence/run/lab/gold.json')).join(' ')); -wound-status live p-wound-status-f41f7ca1 draft - +wound-status live p-wound-status-f1305d28 draft - mood live p-mood-v1 draft - -living-situation live p-living-situation-50cd9fc4 draft - -ostomy-supplies live null draft p-ostomy-supplies-af0f46ca -shelf jordan.json marguerite.json pat.json riley.json -issues [["config-sunrise-wound-status-ce89aea7","done"]] -patient briefs [["gap-ostomy-supplies","locked"]] -gold wound-status|jordan wound-status|marguerite wound-status|pat wound-status|riley +living-situation live p-living-situation-373bc0c3 draft - +ostomy-supplies live null draft p-ostomy-supplies-26bf6ae3 +shelf jordan.json pat.json riley.json verna.json +issues [["config-sunrise-wound-status-8a366995","done"]] +patient briefs [["gap-ostomy-supplies-soc","locked"]] +gold wound-status|jordan wound-status|pat wound-status|riley exit=0 diff --git a/examples/prompt-lab/evidence/run/15-resume.txt b/examples/prompt-lab/evidence/run/15-resume.txt deleted file mode 100644 index e03ce8f1..00000000 --- a/examples/prompt-lab/evidence/run/15-resume.txt +++ /dev/null @@ -1,49 +0,0 @@ -$ npx flows resume --no-observer-link --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove-data --local-agent 01M366RA1PSQ7295XTPYPV9PXC -○ run-1 (deterministic) 0.00s -✓ run-1 (deterministic) 0.01s completionReason: success -○ llm-2 (llm) 0.00s -✓ llm-2 (llm) 5.35s completionReason: success -○ llm-3 (llm) 0.00s -✓ llm-3 (llm) 4.10s completionReason: success -○ llm-4 (llm) 0.00s -✓ llm-4 (llm) 3.83s completionReason: success -○ llm-5 (llm) 0.00s -✓ llm-5 (llm) 3.60s completionReason: success -○ llm-6 (llm) 0.00s -✓ llm-6 (llm) 4.94s completionReason: success -○ run-7 (deterministic) 0.00s -✓ run-7 (deterministic) 0.01s completionReason: success -○ human-8 (deterministic) 0.00s -✓ human-8 (deterministic) 0.01s completionReason: success -○ run-9 (deterministic) 0.00s -✓ run-9 (deterministic) 0.01s completionReason: success -○ run-10 (deterministic) 0.00s -✓ run-10 (deterministic) 0.01s completionReason: success -○ llm-11 (llm) 0.00s -✓ llm-11 (llm) 3.53s completionReason: success -○ llm-12 (llm) 0.00s -✓ llm-12 (llm) 3.97s completionReason: success -○ llm-13 (llm) 0.00s -✓ llm-13 (llm) 4.02s completionReason: success -○ llm-14 (llm) 0.00s -✓ llm-14 (llm) 3.96s completionReason: success -○ run-15 (deterministic) 0.00s -✓ run-15 (deterministic) 0.01s completionReason: success -○ llm-16 (llm) 0.00s -✓ llm-16 (llm) 3.54s completionReason: success -○ llm-17 (llm) 0.00s -✓ llm-17 (llm) 3.47s completionReason: success -○ llm-18 (llm) 0.00s -✓ llm-18 (llm) 6.68s completionReason: success -○ llm-19 (llm) 0.00s -✓ llm-19 (llm) 4.34s completionReason: success -○ run-20 (deterministic) 0.00s -✓ run-20 (deterministic) 0.01s completionReason: success -○ human-21 (deterministic) 0.00s -✓ human-21 (deterministic) 0.02s completionReason: success -○ run-22 (deterministic) 0.00s -✓ run-22 (deterministic) 0.07s completionReason: success -○ run-23 (deterministic) 0.00s -✓ run-23 (deterministic) 0.07s completionReason: success -RUN 01M366RA1PSQ7295XTPYPV9PXC completed (24 steps) completionReason: success -exit=0 diff --git a/examples/prompt-lab/evidence/run/lab/bank.json b/examples/prompt-lab/evidence/run/lab/bank.json index 803436d8..44d835c6 100644 --- a/examples/prompt-lab/evidence/run/lab/bank.json +++ b/examples/prompt-lab/evidence/run/lab/bank.json @@ -2,15 +2,15 @@ "prompts": { "p-wound-status-v1": "Determine the status of the patient's primary wound. Read the referral packet first; if the discharge summary states the wound is closed or healed, answer Healed. Otherwise use the visit notes. Use High confidence when the referral states the status.", "p-mood-v1": "Describe the patient's mood today from the visit notes. Choose the single option that best matches what the patient says and how the clinician describes them. Use High confidence when the patient states their feelings in their own words, Medium when inferred from behavior, Low when the notes are silent.", - "p-living-situation-50cd9fc4": "You are filling one field of a home-health visit chart: the patient's living situation. Choose exactly one of the options provided to you.\n\nSOURCE OF TRUTH\n- Today's visit notes — what the clinician observed and what the patient or caregiver said during this visit — are the source of truth. Prefer them over the referral packet, intake form, prior visit notes, or demographic fields.\n- When the visit notes conflict with any other document, follow the visit notes and say so in your explanation.\n- Use other documents only when today's notes are silent on where and with whom the patient lives.\n\nHOW TO CHOOSE\n- Pick one option from the list you are given. Never invent an option, never combine two, never return free text. If none fits cleanly, pick the closest one and drop your confidence to Low or Medium, explaining the mismatch.\n- Read the options as a residence question: who is in the home overnight, not who visits or provides care.\n- Guidance for common wording in charts:\n - Lives alone: the patient is the only person residing in the home. Daytime caregivers, aides, visiting relatives, or neighbors who check in do not change this — the patient still lives alone.\n - Lives with spouse: a spouse or long-term partner resides in the home. If a spouse plus other relatives reside there, prefer the spouse option when the chart frames the spouse as the co-resident and primary support; prefer the family option when the notes describe a multi-person household without singling out the spouse.\n - Lives with family: one or more relatives or household members other than a spouse reside with the patient (adult child, sibling, parent, grandchild, or a described household).\n - Facility: the patient resides in a congregate or institutional setting — assisted living, memory care, group home, board and care, skilled nursing, or similar. A private residence in a senior apartment or independent-living community is not a facility by itself; classify by who resides with the patient unless the chart describes on-site staffing or facility-provided care.\n- If the notes describe a recent or planned move, answer for where the patient lives as of this visit, not where they are expected to live later.\n\nCONFIDENCE\n- High: today's visit notes state the living situation directly (an explicit statement of who lives in the home or that the patient resides in a named type of facility).\n- Medium: the answer is inferred — drawn from indirect cues in today's notes (who was present and described as a household member, references to \"her husband in the next room\", facility staff charting the visit), or taken from the referral packet or intake form because today's notes do not address it.\n- Low: the chart is silent on living situation, or sources contradict each other and today's notes do not resolve the conflict.\n\nEXPLANATION\n- Return a one- or two-sentence explanation that cites where in the chart the answer came from — name the document or section (e.g. today's visit note narrative, the social/home environment section, the referral packet) and paraphrase the supporting language. Do not quote or reproduce patient identifiers of any kind. If you overrode another document, state which one and why.", - "p-ostomy-supplies-af0f46ca": "You are filling one field of a home-health visit assessment: which ostomy supply the patient needs most at this visit. You will be given the question, the exact list of answer options, and the patient's chart.\n\nSOURCE OF TRUTH\n- Today's visit notes — what the clinician observed, measured, and was told during this visit — are authoritative. Where they conflict with the referral packet, intake forms, prior visit notes, the medication list, or any standing order, today's visit notes win and you say so in the explanation.\n- Older chart material is usable only as background: it can establish that the patient has an ostomy at all, or what supplies were being used, but it cannot override what today's clinician recorded.\n- Read the whole of today's note, not just an ostomy-specific heading. Supply need often appears in the skin assessment, the wound or peristomal description, the supplies-on-hand or inventory section, the patient/caregiver report, the teaching section, or the plan for next visit.\n\nHOW TO CHOOSE\n- Choose exactly one option, and only from the option list you are given. Do not invent, merge, rename, or split options. If the need you see in the chart has no matching option, pick the option that is the closest true fit and explain the gap in your explanation rather than writing a new option.\n- \"Most needed\" means the single supply whose absence is the most immediate problem for this patient right now. Weigh, in this order: (1) the supply is out, nearly out, or unavailable today; (2) the supply is the one that addresses an active problem documented today, such as leakage, an unsealed or uneven seal, an appliance that will not adhere, or peristomal skin breakdown, irritation, moisture, or denudement; (3) the supply the clinician explicitly asks to be ordered, delivered, refilled, or taught about today.\n- If today's notes document more than one gap, choose the one the clinician treats as the priority — the one tied to a change in the plan of care, an order placed, a teaching intervention, or the stated reason the appliance is failing. If two appear genuinely equal in urgency, choose the one that keeps the appliance attached and contained (containment before skin protection), and name the second one in your explanation.\n- Select the \"none needed\" style option only when today's notes affirmatively support it: the patient has adequate supplies on hand, the appliance is intact and sealed, the peristomal skin is described as intact, or the patient has no ostomy. Do not choose it merely because the notes are quiet about supplies.\n\nCONFIDENCE\n- High: today's visit notes state the answer directly — they name the supply as out, needed, ordered, or requested, or they name it as the fix for a problem documented today.\n- Medium: you inferred the answer from today's notes rather than reading it off them — for example, the note describes the clinical problem and the matching supply follows from it, or a supply count implies the shortfall, but no sentence names the needed supply outright.\n- Low: the chart is silent on ostomy supplies and on peristomal condition, the only relevant information is from the referral packet or a prior visit rather than today, or today's notes contradict themselves and nothing resolves which reading is current. Still return your best single option at Low confidence; do not leave the field unanswered.\n\nEXPLANATION\n- Write one or two sentences saying where in the chart the answer came from — name the section or the kind of entry (for example, today's peristomal skin assessment, today's supply inventory, today's plan of care) and what it said in your own words. If you overrode an older source, or if a close second option exists, say so here.\n\nPRIVACY\n- Never include a patient name, initials, date of birth, medical record or account number, address, phone number, or any other identifier in your explanation. Refer to \"the patient\" and cite chart locations by section, not by quoting identifying text. Quote clinical wording only as briefly as needed to support the answer.", - "p-wound-status-f41f7ca1": "Determine the current status of the patient's primary wound. Answer Healed, Ongoing, or No wound.\n\nRead the whole chart before answering. Rank sources by how directly and how recently they observed the wound site:\n\n1. A direct assessment — a clinician's own observation of the site at a visit, with findings such as measurements, wound bed description, drainage, odor, dressing applied, or an explicit note that the site is intact — is the best evidence of current status. The most recent direct assessment governs.\n2. A referral packet, hospital discharge summary, intake form, or other history describes the wound as of an earlier date. It establishes that a wound exists and what it was, but it never overrides a later direct assessment. Read \"closed\", \"healed\", or \"resolved\" in such a document as the status on that document's date, not as today's status; if a later assessment finds the site open, the document is simply outdated.\n\nChoose the answer:\n- Ongoing — the most recent direct assessment finds an open or unhealed wound (open area, drainage, slough, eschar, undermining, or active wound care), or the chart documents a wound with nothing later indicating it closed.\n- Healed — the most recent direct assessment finds the site closed, intact, or re-epithelialized, or the chart documents closure and no later source reports an open wound.\n- No wound — nothing in the chart documents a wound: no wound history and no wound findings on assessment.\n\nAssign confidence. Exactly one of these applies to every chart; work down the list and take the first that fits:\n- High — the answer rests on a direct assessment of the wound site at a recent visit. This holds even when an earlier document says the opposite: a direct observation of the site settles the question, and being contradicted by stale history does not lower confidence.\n- Medium — no direct assessment of the site is available, so the answer rests on referral or history documents alone; or the only direct assessment is old enough that the status could plausibly have changed since.\n- Low — the chart does not address the wound at all (a silent chart, which yields No wound), or its wound documentation is so sparse, ambiguous, or internally inconsistent that you cannot resolve it.\n\nIn the rationale, cite the deciding observation with its concrete findings and, when an earlier document said something different, name that document in a clause and say its status is outdated. Two sentences at most. Do not speculate beyond what the chart records, and do not let living situation, mood, or unrelated diagnoses influence the wound status." + "p-living-situation-373bc0c3": "You are filling in one field of a home-health assessment from the patient's visit chart: the patient's living situation. Choose exactly one of the options you were given, and never write an option that was not provided.\n\nSOURCE OF TRUTH\nToday's visit notes — what the clinician observed, saw, and was told during today's visit — are authoritative. Referral packets, intake forms, prior episode records, and face-to-face documents are supporting evidence only. When today's visit notes conflict with any of those older sources, today's visit notes win and the older source is treated as stale. When today's visit notes are silent on living situation, you may fall back to those older sources, but the answer is then inferred, not stated.\n\nHOW TO CHOOSE AMONG THE OPTIONS\nRead the whole chart for who is physically present in the patient's residence on an ongoing basis, and for what kind of residence it is. Then map what you find onto the options:\n- Choose the option meaning the patient lives by themselves when the chart indicates no other person resides in the home. A caregiver, family member, aide, or neighbor who visits, checks in, drops by, or stays intermittently does not make the patient a cohabitant — the patient still lives alone.\n- Choose the spouse option when the chart indicates the patient shares the residence with a spouse or a long-term partner and no broader household is described.\n- Choose the family option when the chart indicates the patient shares the residence with relatives other than a spouse, or with a household that includes a spouse plus other relatives.\n- Choose the facility option when the residence itself is an institutional or congregate setting with on-site staff rather than a private home.\nIf two options both seem defensible, pick the one supported by the most recent and most direct statement in the chart, and say in the explanation why the other was rejected. Do not blend options and do not answer with more than one.\n\nCONFIDENCE\n- High: today's visit notes state the living situation directly — an explicit statement of who the patient lives with or of the type of residence.\n- Medium: the answer is inferred rather than stated — for example, it comes only from the referral packet or an intake form, or it is deduced from indirect remarks in today's notes about the household.\n- Low: the chart says nothing about living situation, or sources contradict each other and nothing resolves which is current. Still select the single best-supported option, and say plainly that the basis is thin.\n\nEXPLANATION\nGive one or two sentences that name where in the chart the answer came from — the document or section and the substance of what it said. Quote or paraphrase only enough to show the basis. Do not include the patient's name, initials, date of birth, address, phone number, medical record number, or the names of family members, caregivers, clinicians, or facilities; describe people by role instead. Do not invent chart content: if the chart does not support a detail, leave it out and lower the confidence instead.", + "p-ostomy-supplies-26bf6ae3": "You are filling in one question on a home-health visit assessment. You will be given the question, its answer options, and the patient's chart. Choose exactly one of the options given to you — never invent, merge, or reword an option, and never answer with anything outside the list.\n\nSOURCE OF TRUTH\nToday's visit notes — what the clinician observed, measured, and was told during this visit — are the source of truth. When the visit notes conflict with the referral packet, intake forms, prior visit notes, the physician order, or any standing supply list, today's visit notes win. Use older documents only to fill gaps the visit notes leave silent, and only when nothing in today's notes contradicts them.\n\nHOW TO CHOOSE\nThis question asks which ostomy supply the patient needs most at this visit — a single, highest-priority need, not everything that is running low.\n- Read today's notes for what the clinician recorded about the stoma, the peristomal skin, the pouching system, wear time, leakage, and what supplies are on hand or were requested.\n- If the notes name a specific supply as needed, requested, ordered, or out of stock, choose the option matching that supply.\n- If the notes describe a problem rather than a supply, map the problem to the option that addresses it: an inability to collect output or an exhausted supply of the collection appliance points to the pouching supply; a leaking, poorly sealing, or uneven seal at the stoma edge points to the sealing/barrier option; irritated, denuded, or weeping peristomal skin needing protection before adhesion points to the skin-protection option.\n- If two needs appear, pick the one the notes treat as most urgent — the one causing an active problem today, the one the clinician acted on or escalated, or the one that must be resolved for the appliance to function at all. A comfort or convenience need never outranks an active leak, skin breakdown, or an inability to pouch.\n- Choose the \"no supply needed\" option only when today's notes affirmatively indicate the ostomy is intact, the appliance is functioning, and supplies are adequate — not merely because the notes say nothing.\n- If the chart is silent on ostomy supplies entirely, or the evidence points in two directions with no way to rank them, pick the option best supported by what little there is and mark confidence Low; do not guess a specific supply out of thin air.\n\nCONFIDENCE\n- High: today's visit notes state the answer directly — the needed supply is named, requested, ordered, or documented as out.\n- Medium: the answer is inferred — today's notes describe a condition, symptom, or situation from which the needed supply follows, but do not name the supply itself.\n- Low: the chart is silent on this question, or the available evidence contradicts itself and cannot be resolved by preferring today's notes.\n\nEXPLANATION\nAfter the answer, give a one- or two-sentence explanation that cites where in the chart the answer came from — name the section or entry (for example, today's visit note, the skin assessment, the supply inventory, the prior referral packet) and the specific observation or statement you relied on. If you had to override an older document with today's notes, say so. If confidence is Low, say what was missing or contradictory.\n\nDo not include any patient identifiers — no names, dates of birth, addresses, phone numbers, record or insurance numbers — in the answer or the explanation. Refer to the person only as \"the patient.\" Quote the chart only as briefly as needed to support the citation.", + "p-wound-status-f1305d28": "Determine the status of the patient's primary wound: Healed, Ongoing, or No wound.\n\nEvidence order. Today's visit notes describe the wound as it is now, and they govern the answer. Read the referral packet — admission history, discharge summary, prior wound descriptions — for context, but treat everything in it as a record of an earlier point in time.\n\n- Answer Ongoing when the most recent documentation describes an open or unhealed wound: measurements, drainage, wound bed or edge description, slough or eschar, or an ordered dressing change. Answer Ongoing even when an older document called the wound closed or healed; the newer observation is what is true today.\n- Answer Healed when the most recent documentation describes the wound as closed, resolved, or fully epithelialized with no open area remaining.\n- Answer No wound when the chart states the patient has no wound, or when neither the referral packet nor the visit notes document a wound at all.\n\nConfidence.\n- High — today's visit notes state the answer directly: they describe the wound's current condition in their own words. A status stated only in the referral packet is never High on its own, no matter how clearly it is worded and no matter that today's notes fail to contradict it.\n- Medium — today's visit notes do not state the wound's status, and you infer it from the surrounding record: the referral packet, standing orders, the supply list, or a care plan that implies a wound is or is not present.\n- Low — the chart is silent about the wound, or the chart is contradictory: two sources of comparable currency disagree, or today's notes are internally inconsistent, and nothing in the record resolves which is right.\n\nA document that a more recent observation supersedes is outdated, not contradictory. Chronology resolves the conflict, so where today's notes state the current condition directly the answer stays High even though an earlier document said otherwise. Reserve Low for disagreement that chronology cannot settle.\n\nExplanation. Give one or two sentences that cite the specific findings you relied on and, when you set an earlier document aside, name it and say it is outdated." }, "questions": { "wound-status": { "text": "What is the current status of the patient's primary wound?", "type": "single", - "livePromptId": "p-wound-status-f41f7ca1", + "livePromptId": "p-wound-status-f1305d28", "draftPromptId": null }, "mood": { @@ -21,14 +21,14 @@ "living-situation": { "text": "What is the patient's living situation?", "type": "single", - "livePromptId": "p-living-situation-50cd9fc4", + "livePromptId": "p-living-situation-373bc0c3", "draftPromptId": null }, "ostomy-supplies": { "text": "Which ostomy supply does the patient need most this visit?", "type": "single", "livePromptId": null, - "draftPromptId": "p-ostomy-supplies-af0f46ca" + "draftPromptId": "p-ostomy-supplies-26bf6ae3" } } } diff --git a/examples/prompt-lab/evidence/run/lab/gold.json b/examples/prompt-lab/evidence/run/lab/gold.json index 198a0cf6..37ea04be 100644 --- a/examples/prompt-lab/evidence/run/lab/gold.json +++ b/examples/prompt-lab/evidence/run/lab/gold.json @@ -2,12 +2,7 @@ "wound-status|jordan": { "answer": "No wound", "confidence": "High", - "explanation": "The referral packet's intake form documents \"No wounds documented. Skin intact,\" and today's SOC visit note confirms \"Skin intact, no wounds\" — there is no primary wound to stage, and no discharge summary states a wound was closed or healed." - }, - "wound-status|marguerite": { - "answer": "Ongoing", - "confidence": "Medium", - "explanation": "The referral packet never states the wound is closed or healed — it only records peristomal skin intact at discharge — so the visit notes govern, and today's assessment documents erythematous, moist, weepy denudement about 2 cm wide along the 4-to-8 o'clock peristomal aspect with patient-reported burning and stinging. Confidence is Medium rather than High because the status comes from the visit note, not a referral statement." + "explanation": "The referral packet's intake form documents \"No wounds documented. Skin intact,\" and today's SOC visit notes confirm \"Skin intact, no wounds.\"" }, "wound-status|pat": { "answer": "Ongoing", @@ -17,6 +12,6 @@ "wound-status|riley": { "answer": "No wound", "confidence": "High", - "explanation": "The referral packet states 'No wounds' and today's SOC visit note confirms 'No wounds; skin intact around cast,' so there is no primary wound to track." + "explanation": "The referral packet states 'No wounds' alongside the right wrist fracture with cast, and today's SOC visit notes confirm 'No wounds; skin intact around cast.'" } } diff --git a/examples/prompt-lab/evidence/run/lab/queue/issues.json b/examples/prompt-lab/evidence/run/lab/queue/issues.json index 00859cdb..788b27a4 100644 --- a/examples/prompt-lab/evidence/run/lab/queue/issues.json +++ b/examples/prompt-lab/evidence/run/lab/queue/issues.json @@ -1,6 +1,6 @@ [ { - "id": "config-sunrise-wound-status-ce89aea7", + "id": "config-sunrise-wound-status-8a366995", "kind": "config-send", "questionIds": [ "wound-status" @@ -15,6 +15,6 @@ "explanation": "Today's visit notes describe an open 2.0 x 1.5 cm left heel wound with serous drainage and slough; the discharge summary's closed status is outdated." } }, - "proposedPrompt": "Determine the status of the patient's primary wound.\n\nToday's visit notes govern. When the visit notes describe the wound, base the answer on what the clinician documented at today's visit, even if the referral packet or discharge summary says something different. A referral or discharge summary reflects the patient's condition at an earlier point in time; it can be outdated by the time of the visit, and a documented open wound today overrides an earlier \"closed\" or \"healed\" note.\n\n- Answer Ongoing when today's visit notes document an open or unhealed wound — for example an open area with measurements, drainage, slough, or an active dressing regimen.\n- Answer Healed when today's visit notes document the wound as closed, resurfaced, or fully healed. If the visit notes do not assess the wound at all, fall back to the referral packet: if the discharge summary states the wound is closed or healed, answer Healed.\n- Answer No wound only when neither the visit notes nor the referral packet document any wound, present or recently resolved.\n\nUse High confidence when today's visit notes directly document the wound's current state, including when they contradict the referral. Use High confidence when the referral states the status and the visit notes do not contradict it. Lower the confidence when the only source is an older referral and the visit notes are silent, or when the documentation within a single source conflicts with itself.\n\nIn the rationale, cite the specific findings from today's visit notes that drive the answer, and when the referral disagrees, say plainly that the earlier status is outdated. Keep the rationale to one or two sentences grounded in the documentation." + "proposedPrompt": "Determine the status of the patient's primary wound.\n\nToday's visit notes are the most current record of the wound and outrank the referral packet whenever the two disagree. A referral packet or discharge summary describes the wound as of discharge; it can be out of date by the time of the visit.\n\n1. Read today's visit notes first. If they describe the wound — an open area, measurements, drainage, slough or other wound bed findings, or a dressing change — decide from those findings alone. Any wound that is still open, draining, or being dressed is Ongoing, even if the referral packet or discharge summary says the wound is closed or healed. Answer Healed only when today's notes state the wound is closed, resolved, or intact.\n2. Use the referral packet only when today's notes say nothing about the wound. Then follow it: a discharge summary stating the wound is closed or healed means Healed.\n3. Answer No wound only when neither today's notes nor the referral packet documents any wound.\n\nConfidence: use High when the source you relied on states the status directly — including when today's visit notes document the wound firsthand. A conflict with an older referral does not lower confidence; today's notes settle it. Use Medium when the status must be inferred from indirect or incomplete findings, and Low when no source addresses the wound clearly.\n\nIn the rationale, cite the specific findings you relied on, and when they contradict the referral packet, say so and note that the referral's status is outdated." } ] diff --git a/examples/prompt-lab/evidence/run/lab/queue/patient-briefs.json b/examples/prompt-lab/evidence/run/lab/queue/patient-briefs.json index e06f18f5..82478cd1 100644 --- a/examples/prompt-lab/evidence/run/lab/queue/patient-briefs.json +++ b/examples/prompt-lab/evidence/run/lab/queue/patient-briefs.json @@ -1,8 +1,9 @@ [ { - "id": "gap-ostomy-supplies", + "id": "gap-ostomy-supplies-soc", "questionId": "ostomy-supplies", - "brief": "No shelf patient has any ostomy documented, so every answer path for this question is unexercised. Add an older adult recently discharged after colorectal surgery with a new colostomy: referral packet lists the stoma, output character, and the supply list sent home; today's visit notes show peristomal skin irritation and a pouch seal failing early, with barrier rings on hand but skin prep wipes exhausted — so the chart supports a specific single supply need rather than a generic one. To exercise the remaining options, a second version of the chart (or a second such patient) should show an established, well-healed stoma with a full supply cupboard and no skin breakdown, supporting 'None'. Characteristics only: age band, ostomy type, stoma age, skin condition, current supply inventory — no names, dates, addresses, or record numbers.", + "visitType": "soc", + "brief": "No shelf patient has any ostomy documentation, so none of the supply answer paths can be exercised. The shelf needs a fabricated post-colostomy patient whose referral packet describes a recent colorectal surgery with a new stoma and a starter supply kit sent home, and whose visit notes document the stoma and peristomal skin in enough detail to force a choice among the supply options: peristomal skin excoriated and weeping with the pouch seal lifting after a few hours, a nearly empty box of pouches on the counter, and no barrier rings in the home. A second variant with a well-healed mature stoma, intact peristomal skin, and a full supply of everything would exercise the 'None' path.", "from": "planner", "status": "locked" } diff --git a/examples/prompt-lab/evidence/run/lab/shelf/marguerite.json b/examples/prompt-lab/evidence/run/lab/shelf/marguerite.json deleted file mode 100644 index 8f2235ee..00000000 --- a/examples/prompt-lab/evidence/run/lab/shelf/marguerite.json +++ /dev/null @@ -1,10 +0,0 @@ -{ - "id": "marguerite", - "label": "Marguerite", - "ageBand": "75–84", - "visitType": "soc", - "referral": "REFERRAL PACKET — discharging hospital to home health (start of care)\n\nReason for referral: post-surgical care and new ostomy self-care teaching. Skilled nursing ordered; dietitian and occupational therapy consults pending.\n\nCharacteristics: older adult, age band 75–84, female. Lives alone in a one-story apartment with a standard bathroom; no caregiver present during pouch changes. Adult child lives nearby and visits about twice a week for grocery runs but does not perform ostomy care. No home aide authorized. Patient does not drive and has limited transportation. Independent with ADLs before surgery, now limited by post-op fatigue. Mild vision impairment — uses reading glasses, reports difficulty with fine print on supply boxes. Osteoarthritis of the hands with reduced pinch strength.\n\nDiagnoses: primary — sigmoid colon adenocarcinoma, status post open sigmoid colectomy with end colostomy. Secondary — type 2 diabetes mellitus (non-insulin, oral agent), hypertension, osteoarthritis of the hands, mild protein-calorie malnutrition post-op, and history of chronic sun-damaged, thin skin.\n\nOstomy description: end colostomy, left lower quadrant, matured. Stoma measured 28 mm round at discharge, budded approximately 1 cm, beefy red and viable. Peristomal skin intact at discharge.\n\nOutput character: soft-formed to pasty brown stool, moderate volume, approximately 2–3 emptyings per day. No high-output or liquid pattern documented. Flatus present.\n\nStoma age: new — approximately 10 to 14 days from creation at the time of the first home visit.\n\nSUPPLY LIST SENT HOME: 20 two-piece cut-to-fit skin barriers/wafers; 30 drainable pouches with closure clips; 1 box (10) barrier rings/seals; 1 tube ostomy paste; 1 box (50) adhesive-remover wipes; 1 box (30) skin-prep / barrier-film wipes; 1 bottle pouch deodorant; measuring guide and scissors; 1 ostomy belt.\n\nTeaching status at discharge: patient returned demonstration of pouch emptying; barrier change performed by nursing, patient observed only.\n\nSupply status per discharging hospital: \"All supplies provided for 30 days; no anticipated supply need.\"\n\nExpected wear time per discharge instruction: barrier change every 3 to 4 days.\n\nReorder pathway: DME supplier assigned; reorder by phone; 3–5 day delivery.", - "notes": "HOME HEALTH VISIT NOTE — start of care, post-op day ~12\n\nVisit purpose: initial assessment, ostomy assessment, and self-care teaching. Patient alone in the apartment at the time of the visit; adult child was here two days ago for groceries.\n\nStoma assessment: stoma now measures 24 mm round — smaller than the 28 mm recorded in the referral packet, consistent with expected post-op shrinkage. Stoma remains beefy red, viable, and budded. Left lower quadrant placement sits at skin level along one edge because of a shallow abdominal crease that appears when the patient is seated.\n\nPeristomal skin: erythematous with moist, weepy denudement in a crescent along the 4-to-8 o'clock inferior aspect, approximately 2 cm wide, matching the track where output runs under the barrier when the patient sits. No candidal satellite lesions, no ulceration, no mucocutaneous separation. Patient reports burning and stinging under the wafer.\n\nSeal failure: barrier is lasting approximately 18 to 24 hours before leakage, against the referral's expected 3 to 4 days. Patient has changed the appliance 3 times in the last 2 days. Undermining of the wafer adhesive noted on removal, with stool tracking directly onto the denuded skin.\n\nSUPPLY INVENTORY COUNTED IN THE HOME THIS VISIT: barrier rings/seals — 8 of 10 remaining. Cut-to-fit wafers — 14 remaining. Drainable pouches — 22 remaining. Ostomy paste — approximately three-quarters of the tube remaining. Adhesive-remover wipes — approximately 40 remaining. Ostomy belt — 1, unused, still in the drawer. SKIN-PREP / BARRIER-FILM WIPES — 0 remaining; the box is empty. Patient has been applying the barrier directly to wet, weepy skin with no protective film and did not know a refill was needed. Stoma powder / protective powder — none in the home; it was never sent and does not appear on the referral supply list.\n\nTechnique observation: patient cuts the wafer opening to the discharge measurement of 28 mm rather than the current 24 mm, leaving a ring of exposed skin. Arthritic hands make scissor-cutting slow and imprecise, and fine print on the supply boxes is hard to read. Documented as a TEACHING need — correct measuring and cutting to current stoma size — not as a supply need.\n\nOutput character today: soft-formed brown stool, 2–3 emptyings daily, unchanged from the referral. No high-output or liquid pattern.\n\nVitals and systemic: afebrile. No peristomal cellulitis, no surrounding induration. Blood glucose mildly elevated on the patient's own meter. Pain 3/10, burning, localized to the skin only.\n\nPatient goal, in their own words: \"I want it to stay on so I can go to my grandchild's recital without worrying.\"\n\nAccess note: patient does not drive and cannot pick up supplies today. The DME supplier reorder is by phone with 3–5 day delivery, so any gap counted above stands until that order arrives.", - "locked": true, - "source": "invented" -} diff --git a/examples/prompt-lab/evidence/run/lab/shelf/verna.json b/examples/prompt-lab/evidence/run/lab/shelf/verna.json new file mode 100644 index 00000000..b59d649e --- /dev/null +++ b/examples/prompt-lab/evidence/run/lab/shelf/verna.json @@ -0,0 +1,10 @@ +{ + "id": "verna", + "label": "Verna", + "ageBand": "60-69", + "visitType": "soc", + "referral": "REFERRAL PACKET (from the surgical service)\n\nDISCHARGE SUMMARY. Sigmoid adenocarcinoma, status post open sigmoid colectomy with permanent end colostomy, about three weeks ago. Uneventful post-operative course: no anastomotic leak, no wound complication, no ileus; tolerated diet advancement; ambulating independently at discharge. Post-operative anemia, stable, on oral iron.\n\nOTHER ACTIVE DIAGNOSES. Type 2 diabetes mellitus, with noted skin fragility and slow healing. Hypertension. Obesity with a pendulous lower abdomen and a skin crease below the stoma. Osteoarthritis of the hands limiting fine dexterity.\n\nSTOMA AT DISCHARGE (inpatient WOC nurse). End colostomy, left lower quadrant. Beefy red, budded 2 cm, 32 mm round. Peristomal skin intact. Output pasty to formed. Pouching schedule: change every 3 days.\n\nTEACHING. One supervised pouch change with return demonstration, charted as partially independent. Written instructions given.\n\nSTARTER KIT CHECKLIST (sent home at discharge). 10 one-piece drainable pouches with a pre-cut 32 mm barrier; skin barrier wipes; adhesive remover wipes; deodorant drops; barrier rings x10.\n\nSOCIAL. Lives alone in a one-bedroom second-floor walk-up, no elevator. One adult child works full time and visits on weekends. No home aide authorized.\n\nORDERS. Skilled nursing 2x/week for ostomy care and teaching. WOC consult PRN. DME resupply order placed with the DME supplier; no confirmed delivery.", + "notes": "VISIT NOTES (this visit, start of care; about three weeks post-op)\n\nSTOMA. End colostomy, left lower quadrant. Now budded only approximately 0.5 cm and flush-to-retracted at the 4 o'clock position as post-operative edema has resolved. Measures approximately 25 mm and is slightly oval rather than round. When she sits, the stoma sits at the lip of the abdominal crease.\n\nPERISTOMAL SKIN. Denuded and excoriated, weeping serous moisture across roughly the inferior half of the wafer field, 2-4 cm, with an irregular scalloped border tracking exactly where effluent has run. She reports burning. No satellite lesions. No induration. No purulence. No fever.\n\nPOUCH IN PLACE. On approximately 5 hours; the adhesive is already lifting along the inferior border with effluent visible under the edge. She has been taping the edge down with paper tape. Reports two to three leaks a day and has been changing daily instead of every 3 days.\n\nOUTPUT. Looser and higher in volume than the packet describes; no longer pasty to formed.\n\nSELF-CARE. Changes the pouch standing at the bathroom sink, no mirror. Cannot see the inferior edge of the stoma over her abdomen when seated. Hand arthritis limits fine cutting and trimming.\n\nSUPPLY INVENTORY, counted with the patient this visit, item by item. 2 one-piece drainable pouches left in the box on the counter; their pre-cut 32 mm opening now leaves 3-4 mm of bare skin exposed around the 25 mm stoma. Skin barrier wipes nearly full. Adhesive remover wipes full. Deodorant drops unopened. NO barrier rings anywhere: cabinet, supply bag and counter searched, and she states she never saw any come home with her. No barrier powder. No convex product. No belt. No DME delivery received since discharge.", + "locked": true, + "source": "invented" +} diff --git a/examples/prompt-lab/evidence/run/lab/targets.json b/examples/prompt-lab/evidence/run/lab/targets.json index 55bacef4..e2cace2d 100644 --- a/examples/prompt-lab/evidence/run/lab/targets.json +++ b/examples/prompt-lab/evidence/run/lab/targets.json @@ -2,46 +2,46 @@ "sunrise|living-situation|jordan": { "answer": "Lives with spouse", "confidence": "High", - "explanation": "Today's visit note narrative states the spouse was present throughout the SOC visit, that patient and spouse share the home, and that the spouse sets up the pill organizer each morning. This overrides the intake form completed by the hospital case manager, which recorded living alone." + "explanation": "Today's skilled nursing SOC visit notes state directly that the patient and their spouse share the home, with the spouse present throughout the visit and setting up the pill organizer each morning. The intake form's 'lives alone' entry is older supporting evidence and is treated as stale under today's direct observation." }, "sunrise|living-situation|pat": { "answer": "Lives alone", "confidence": "High", - "explanation": "Today's SOC visit note narrative states directly that the patient lives alone in a one-story home, with a neighbor who checks in daily; a daily check-in visitor does not change residence status. This agrees with the referral packet's intake-form social history, so no override was needed." + "explanation": "Today's skilled nursing SOC visit note states directly that the patient lives alone in a one-story home, with a neighbor who checks in daily; an intermittent visitor does not count as a cohabitant. This is consistent with the referral packet's intake form, but the current visit note is the direct basis." }, "sunrise|living-situation|riley": { "answer": "Lives with family", "confidence": "High", - "explanation": "Today's skilled nursing SOC visit note states directly that the patient resides with her daughter, son-in-law and two grandchildren. This agrees with the referral packet, which also describes her living with her daughter's family." + "explanation": "Today's visit notes state directly that the patient lives with her adult child, that child's spouse, and two grandchildren, which is a household of relatives other than a spouse. The referral packet agrees, but the visit note is the direct current statement; 'Lives alone' and the spouse option were rejected because multiple relatives reside in the home." }, "sunrise|mood|jordan": { "answer": "Calm", "confidence": "Medium", - "explanation": "Today's visit notes describe the patient as \"relaxed and engaged\" and asking good questions about the new diuretic, but the mood is inferred from the clinician's behavioral description rather than stated by the patient in their own words." + "explanation": "The visit notes describe the patient as \"relaxed and engaged\" and asking good questions about the diuretic, but the mood is inferred from the clinician's behavioral description rather than stated by the patient in their own words." }, "sunrise|mood|pat": { "answer": "Calm", "confidence": "High", - "explanation": "Today's visit notes describe the patient as \"calm and pleasant, joking about hospital food,\" and she states in her own words that she is \"glad to be home.\"" + "explanation": "The visit notes describe the patient as \"calm and pleasant, joking about hospital food,\" and she states in her own words that she is glad to be home." }, "sunrise|mood|riley": { "answer": "Anxious", "confidence": "High", "explanation": "The patient states in her own words, 'I can't stop thinking about falling again, I lie awake worrying about it,' and the clinician notes she wrings her hands during the medication review." }, - "sunrise|wound-status|jordan": { - "answer": "No wound", - "confidence": "High", - "explanation": "The referral packet's intake form documents 'No wounds documented. Skin intact,' and today's SOC visit note confirms 'Skin intact, no wounds' — no wound was ever present, so there is nothing to have healed." - }, "sunrise|wound-status|pat": { "answer": "Ongoing", "confidence": "High", "explanation": "Today's visit notes describe an open 2.0 x 1.5 cm left heel wound with serous drainage and slough; the discharge summary's closed status is outdated." }, + "sunrise|wound-status|jordan": { + "answer": "No wound", + "confidence": "High", + "explanation": "The referral packet's intake form documents \"No wounds documented. Skin intact,\" and today's SOC visit notes confirm \"Skin intact, no wounds.\"" + }, "sunrise|wound-status|riley": { "answer": "No wound", "confidence": "High", - "explanation": "The referral packet states \"No wounds\" and today's SOC visit note confirms \"No wounds; skin intact around cast,\" so there is no primary wound to stage or track." + "explanation": "The referral packet states \"No wounds\" and today's SOC visit notes confirm \"No wounds; skin intact around cast,\" so there is no primary wound to status." } } diff --git a/examples/prompt-lab/evidence/run/lab/work/patients/gap-ostomy-supplies-soc/fbf53055/plan.json b/examples/prompt-lab/evidence/run/lab/work/patients/gap-ostomy-supplies-soc/fbf53055/plan.json new file mode 100644 index 00000000..a27b7275 --- /dev/null +++ b/examples/prompt-lab/evidence/run/lab/work/patients/gap-ostomy-supplies-soc/fbf53055/plan.json @@ -0,0 +1,4 @@ +{ + "brief": "No shelf patient has any ostomy documentation, so none of the supply answer paths can be exercised. The shelf needs a fabricated post-colostomy patient whose referral packet describes a recent colorectal surgery with a new stoma and a starter supply kit sent home, and whose visit notes document the stoma and peristomal skin in enough detail to force a choice among the supply options: peristomal skin excoriated and weeping with the pouch seal lifting after a few hours, a nearly empty box of pouches on the counter, and no barrier rings in the home. A second variant with a well-healed mature stoma, intact peristomal skin, and a full supply of everything would exercise the 'None' path.", + "plan": "INVENTED SHELF PATIENT — VARIANT A (the load-bearing one): first-name label \"Verna\". Age band 60-69. No surname, no dates, no MRN, no phone, no address, no facility or clinician names anywhere — the referral packet is from \"the surgical service\" and \"the inpatient WOC nurse\", supplies come from \"the DME supplier\", and all timing is relative (\"about three weeks post-op\", \"since discharge\", \"this visit\").\n\nDIAGNOSES: sigmoid adenocarcinoma s/p open sigmoid colectomy with permanent end colostomy; type 2 diabetes (skin fragility, slow healing); hypertension; obesity with a pendulous lower abdomen and a skin crease below the stoma; osteoarthritis of the hands limiting fine dexterity; post-op anemia.\n\nLIVING SITUATION: lives alone in a one-bedroom second-floor walk-up, no elevator. One adult child works full time and visits on weekends. No home aide authorized. Changes the pouch standing at the bathroom sink, no mirror, cannot see the inferior edge of the stoma over her abdomen when seated.\n\nWHAT THE REFERRAL PACKET SAYS: discharge summary describing an uneventful post-op course; stoma at discharge \"beefy red, budded 2 cm, 32 mm round, peristomal skin intact\"; output \"pasty to formed\"; one supervised pouch change with return demonstration charted as \"partially independent\"; pouching schedule of every 3 days. Starter-kit checklist: 10 one-piece drainable pouches with a PRE-CUT 32 mm barrier, skin barrier wipes, adhesive remover wipes, deodorant drops, and — this is the trap — \"barrier rings x10\". Orders: skilled nursing 2x/week for ostomy care and teaching, WOC consult PRN, DME resupply order placed with no confirmed delivery.\n\nWHAT TODAY'S VISIT NOTES MUST SHOW: stoma now budded only ~0.5 cm and flush-to-retracted at the 4 o'clock position as post-op edema resolved; measures ~25 mm and slightly oval; sits at the lip of the abdominal crease when she sits. Peristomal skin denuded, excoriated and weeping serous moisture across roughly the inferior half of the wafer field, 2-4 cm, with an irregular scalloped border tracking exactly where effluent ran; she reports burning. Explicitly NO satellite lesions, no induration, no purulence, no fever — so this reads as irritant contact dermatitis from effluent, not infection or fungus. Pouch has been on ~5 hours and the adhesive is already lifting along the inferior border with effluent visible under the edge; she has been taping it down with paper tape. Two to three leaks a day; she is changing daily instead of every 3 days. Output looser and higher-volume than the packet describes. Inventory counted WITH the patient and documented item by item: 2 pouches left in the box on the counter, and their pre-cut 32 mm opening now leaves 3-4 mm of bare skin around a 25 mm stoma; barrier wipes nearly full; adhesive remover full; deodorant drops unopened; NO barrier rings anywhere — cabinet, bag and counter searched; no barrier powder, no convex product, no belt.\n\nINTENDED ANSWER: a moldable barrier ring/seal — it fills the crease and the oversized opening, stops effluent contact with denuded skin, and restores wear time. It is the only supply on the shelf that addresses the mechanism rather than a symptom.\n\nDELIBERATE DISTRACTORS (so the question is a real choice, not a lookup): (1) the nearly empty pouch box makes \"pouches\" look urgent, but more pouches at the wrong size leak the same way — resupply is a workflow action, not this visit's supply need; (2) weeping skin invites \"barrier powder/crusting\", but the notes make ongoing effluent contact explicit, so protection beats treatment; (3) a near-flush stoma invites \"convex wafer\" or \"belt\", but neither is in the home, no WOC convexity assessment has happened, and convexity on denuded skin without a ring first is second-line; (4) an answer path that defers to \"WOC referral\" instead of naming a supply should visibly under-answer, since the packet already carries a PRN consult.\n\nDELIBERATE CONFLICTS BETWEEN PACKET AND NOTES, each testing one thing: (C1 inventory) the packet checklist says 10 barrier rings went home; the in-home count finds zero and she says she never saw any — tests whether the answer trusts today's observation over discharge paperwork, and a packet-trusting path lands on \"pouches\" or \"None\". (C2 sizing) 32 mm round in the packet vs ~25 mm oval today, with pre-cut wafers still cut to the discharge size — tests whether the model connects resolving edema to the leak mechanism instead of blaming her technique. (C3 skin) \"peristomal skin intact\" and q3-day changes in the packet vs denuded skin and daily changes today — tests recency. (C4 output) formed/pasty in the packet vs looser and higher-volume today — a quiet detail that makes leakage more corrosive and further undercuts \"just reorder pouches\".\n\nVARIANT B — THE 'NONE' PATH: first-name label \"Roland\", age band 70-79. Mature end colostomy of several years after colectomy for diverticular disease; lives with a spouse who does the changes competently. The referral packet is for something else entirely — skilled nursing after a heart-failure hospitalization for medication management and teaching — and notes the colostomy as long-standing and self-managed, with a stale discharge-planner line claiming \"patient reports running out of pouches\" so the question must still be answered rather than skipped. Visit notes: stoma mature, budded 2 cm, beefy red, on a flat surface, 28 mm and unchanged for years; peristomal skin fully intact, no erythema or denudement; two-piece system worn 4 days with an intact seal, no leaks this month or historically; supply closet inventoried with the patient — 30+ pouches, 20+ wafers already cut to size, a full box of barrier rings, unopened barrier powder, adhesive remover, an unused belt; monthly DME auto-ship arrived complete. Conflict for B mirrors A's in the opposite direction: the stale \"running out of pouches\" note contradicts the counted inventory. That symmetry is the point — the same rule (observed notes over packet) yields \"barrier ring\" in A and \"None\" in B, so neither shortcut (\"always name a supply\", \"always trust the packet\") can pass both.\n\nSHELF MECHANICS: keep packet and visit notes as two separately quotable documents in the shelf's existing envelope shape — the conflicts only test anything if an answer path can cite one against the other. Measurements and counts are the load-bearing detail and must stay exact; everything identifying stays absent." +} diff --git a/examples/prompt-lab/evidence/run/lab/work/patients/gap-ostomy-supplies/plan.json b/examples/prompt-lab/evidence/run/lab/work/patients/gap-ostomy-supplies/plan.json deleted file mode 100644 index 1a00c5b6..00000000 --- a/examples/prompt-lab/evidence/run/lab/work/patients/gap-ostomy-supplies/plan.json +++ /dev/null @@ -1,4 +0,0 @@ -{ - "brief": "No shelf patient has any ostomy documented, so every answer path for this question is unexercised. Add an older adult recently discharged after colorectal surgery with a new colostomy: referral packet lists the stoma, output character, and the supply list sent home; today's visit notes show peristomal skin irritation and a pouch seal failing early, with barrier rings on hand but skin prep wipes exhausted — so the chart supports a specific single supply need rather than a generic one. To exercise the remaining options, a second version of the chart (or a second such patient) should show an established, well-healed stoma with a full supply cupboard and no skin breakdown, supporting 'None'. Characteristics only: age band, ostomy type, stoma age, skin condition, current supply inventory — no names, dates, addresses, or record numbers.", - "plan": "INVENTED SHELF PATIENT — \"Marguerite\" (first-name label only; no surname, DOB, MRN, address, phone, or dates anywhere in the chart; all temporal references are relative, e.g. \"post-op day ~12\", \"two visits ago\", \"last week\").\n\nPURPOSE: exercise the workbench question \"Which ostomy supply does the patient need most this visit?\" across all answer paths. Ships as ONE patient with TWO chart versions (A: new/complicated stoma → specific single supply need; B: established/well-healed stoma with full cupboard → 'None'). Two versions of the same characteristics keep the shelf small and make the discriminating evidence obvious in diff.\n\nCHARACTERISTICS (the only demographics): age band 75–84, older adult; female; lives alone in a one-story apartment; adult child nearby providing intermittent help; independent with ADLs pre-op, now limited by post-op fatigue; mild vision impairment (reading glasses, difficulty with fine print on supply boxes); arthritic hands with reduced pinch strength — relevant because it plausibly explains seal-application technique problems and argues AGAINST answering with a cut-to-fit-only supply.\n\nDIAGNOSES: primary — sigmoid colon adenocarcinoma s/p open sigmoid colectomy with end colostomy. Secondary — type 2 diabetes mellitus (non-insulin, oral agent), hypertension, osteoarthritis of the hands, mild protein-calorie malnutrition post-op, and history of chronic sun-damaged/thin skin. Diabetes and thin skin are deliberate context: they make peristomal skin irritation clinically credible and raise the stakes of the answer without themselves naming a supply.\n\nLIVING SITUATION: alone, apartment with a standard bathroom, no caregiver present during pouch changes; adult child visits ~2×/week, does grocery runs but does not do ostomy care; no home aide authorized yet; transportation limited — patient does not drive, so \"just pick some up today\" is not a realistic self-remedy and the supply gap is a real gap.\n\nREFERRAL PACKET (discharge → home health) SAYS:\n- Reason for referral: post-surgical care and new ostomy self-care teaching; skilled nursing, plus dietitian and OT consults pending.\n- Ostomy description: end colostomy, left lower quadrant, matured, stoma ~28 mm round, budded ~1 cm, beefy red and viable at discharge; peristomal skin intact at discharge.\n- Output character: soft-formed to pasty brown stool, moderate volume, ~2–3 emptyings per day; no high-output/liquid pattern documented; flatus present.\n- Stoma age: new — approximately 10–14 days from creation at the time of the first home visit.\n- Supply list sent home (explicit inventory, this is the core evidence surface): 20 two-piece cut-to-fit skin barriers/wafers; 30 drainable pouches with closure clips; 1 box (10) barrier rings/seals; 1 tube ostomy paste; 1 box (50) adhesive-remover wipes; 1 box (30) skin-prep/barrier-film wipes; 1 bottle pouch deodorant; measuring guide and scissors; ostomy belt (1).\n- Teaching status at discharge: \"patient returned demonstration of pouch emptying; barrier change performed by nursing, patient observed only.\"\n- Referral explicitly states \"all supplies provided for 30 days; no anticipated supply need.\" (This line is the hook for the deliberate conflict below.)\n- Reorder pathway: DME supplier assigned, reorder by phone, 3–5 day delivery — i.e., an unmet need today is actionable but not instantly self-solvable.\n\nTODAY'S VISIT NOTES (VERSION A — the complicated chart) MUST SHOW:\n- Stoma assessment: stoma now ~24 mm (expected post-op shrinkage from the referral's 28 mm), still beefy red, viable, budded, at skin level on one edge due to a shallow crease in the left lower quadrant when the patient sits.\n- Peristomal skin: erythematous, moist, weepy denudement in a crescent along the 4-to-8 o'clock inferior aspect, ~2 cm wide, matching where output tracks under the barrier when seated; no candidal satellite lesions, no ulceration, no mucocutaneous separation. Patient reports burning/stinging under the wafer.\n- Seal failure: barrier lasting ~18–24 hours before leakage, versus expected 3–4 days; patient has changed the appliance 3 times in the last 2 days. Undermining of the wafer adhesive noted on removal, with stool tracking onto the denuded skin.\n- Supply inventory ON HAND, counted at the visit (this is what forces a SPECIFIC answer): barrier rings 8 of 10 remaining; wafers 14 remaining; pouches 22 remaining; paste ~¾ tube; adhesive-remover wipes ~40 remaining; ostomy belt unused in the drawer; SKIN-PREP / BARRIER-FILM WIPES: 0 — box empty, patient has been applying the barrier to wet, weepy skin without any protective film and did not know a refill was needed. Also absent from the home entirely: stoma powder / protective powder (never sent, not on the referral list) — so the chart supports a single, concrete, most-needed item rather than a generic \"more supplies.\"\n- Technique observation: patient cuts the wafer opening to the discharge measurement (28 mm) rather than the current 24 mm, leaving exposed skin; arthritic hands make scissor-cutting slow and imprecise. This is documented as a TEACHING need, not a supply need — it is a deliberate distractor that a weak answer will convert into \"needs pre-cut/moldable barriers.\"\n- Output character today: matches the referral (soft-formed, 2–3× daily) — deliberately NOT high-output, so \"needs high-output/drainable-with-spout pouches\" is unsupported.\n- Vitals/systemic: afebrile, no peri-stomal cellulitis, blood glucose mildly elevated; pain 3/10 burning at the skin only.\n- Patient goal stated in their own words: \"I want it to stay on so I can go to my grandchild's recital without worrying.\"\n\nTHE DEFENSIBLE ANSWER in Version A: the skin-prep / barrier-film wipes (the exhausted item), because the failing link in the chain is unprotected, weeping peristomal skin under an adhesive barrier — rings are on hand and already in use, pouches and wafers are stocked, output does not justify a different pouch system, and the wafer-sizing error is a teaching correction, not a purchase. A close-second defensible answer (stoma/protective powder, never supplied) is intentionally reachable, so graders can distinguish \"picked the empty box\" from \"reasoned about the crusting technique the weepy skin actually needs.\"\n\nVERSION B — the 'None' chart (same patient characteristics, later state; or a second shelf patient in the same age band if the shelf prefers distinct records):\n- Stoma age: established, ~14 months; stoma ~24 mm, stable size, beefy red, budded, no crease interference; patient uses a moldable/pre-sized barrier.\n- Peristomal skin: intact, no erythema, no denudement, no maceration; skin described as fully healed with no breakdown anywhere in the visit note.\n- Wear time: 4 days consistently, no leakage reported since the last two visits; patient independently empties and changes, returns demonstration flawlessly.\n- Supply inventory: full cupboard — wafers 18, pouches 40, barrier rings 9, skin-prep wipes 45, adhesive remover 50, paste ~full, powder 1 unopened bottle, belt available; 30-day reorder already placed with the DME supplier and confirmed in transit.\n- Output character: soft-formed, predictable, 1–2 emptyings daily, consistent with the referral.\n- Visit purpose: routine reassessment/recertification, no complaint.\n- The defensible answer is 'None' — no supply is deficient, no skin or seal problem creates a need. Version B also carries one benign non-supply need (a dietitian question about gas-producing foods) so 'None' has to be chosen on supply grounds, not because the chart is empty.\n\nDELIBERATE REFERRAL-vs-NOTES CONFLICTS (the part that tests the question rather than pattern-matching):\n1. STOMA SIZE DRIFT: referral says 28 mm; today's measurement is 24 mm. A model that trusts the referral packet will endorse the patient's 28 mm cut and miss that exposed skin is why the seal fails. The correct read is that current assessment overrides discharge documentation.\n2. \"NO ANTICIPATED SUPPLY NEED\": the referral asserts 30 days of supplies and no need; the visit count proves one box is at zero and one item was never supplied at all. A model that defers to the referral's blanket statement will answer 'None' on a chart that clearly supports a specific item — this is the primary trap separating Version A from Version B.\n3. PERISTOMAL SKIN STATUS: referral says \"peristomal skin intact at discharge\"; today's note documents weepy denudement. Same field, opposite value, with the visit note being the current truth.\n4. SUPPLY COUNT vs SUPPLY LIST: the referral's list includes skin-prep wipes, so a list-reading model will conclude the patient has them; only the counted inventory in today's note reveals the box is empty. The evidence needed to answer lives in the notes, not the packet.\n5. WEAR-TIME EXPECTATION: referral implies 3–4 day wear; notes show 18–24 hours. The gap is the clinical signal that something under the barrier — skin, not the pouch — is failing.\n\nGUARDRAILS FOR AUTHORING: no surnames, no dates or date-like strings (use relative intervals only), no MRN/account/encounter numbers, no phone numbers, no addresses, no facility or clinician names (use \"the discharging hospital\", \"the home health nurse\", \"the DME supplier\"), no insurance IDs. Keep the referral packet and visit note as separate documents so the conflict is only visible when both are read. Every fact that drives the answer — stoma measurement, skin condition, wear time, and the counted inventory with explicit zeros — must appear verbatim in the chart text, so a grader can point to the line that justifies each answer path." -} diff --git a/examples/prompt-lab/evidence/run/lab/work/q-wound-status/cd33704e/grid.json b/examples/prompt-lab/evidence/run/lab/work/q-wound-status/cd33704e/grid.json new file mode 100644 index 00000000..d06d04c6 --- /dev/null +++ b/examples/prompt-lab/evidence/run/lab/work/q-wound-status/cd33704e/grid.json @@ -0,0 +1,55 @@ +[ + { + "questionId": "wound-status", + "patientId": "jordan", + "pile": "shared", + "highlight": [], + "ai": { + "answer": "No wound", + "confidence": "High", + "explanation": "The referral packet's intake form documents \"No wounds documented. Skin intact,\" and today's SOC visit notes confirm \"Skin intact, no wounds.\"" + }, + "target": { + "answer": "No wound", + "confidence": "High", + "explanation": "The referral packet's intake form documents \"No wounds documented. Skin intact,\" and today's SOC visit notes confirm \"Skin intact, no wounds.\"" + }, + "notes": "" + }, + { + "questionId": "wound-status", + "patientId": "pat", + "pile": "shared", + "highlight": [ + "medium confidence" + ], + "ai": { + "answer": "Ongoing", + "confidence": "Medium", + "explanation": "Today's SOC visit documents an open left heel wound measuring 2.0 x 1.5 cm with moderate serous drainage and yellow slough, contradicting the discharge summary's statement that the wound was closed; the current clinical observation governs." + }, + "target": { + "answer": "Ongoing", + "confidence": "High", + "explanation": "Today's visit notes describe an open 2.0 x 1.5 cm left heel wound with serous drainage and slough; the discharge summary's closed status is outdated." + }, + "notes": "" + }, + { + "questionId": "wound-status", + "patientId": "riley", + "pile": "shared", + "highlight": [], + "ai": { + "answer": "No wound", + "confidence": "High", + "explanation": "The referral packet states 'No wounds' alongside the right wrist fracture with cast, and today's SOC visit notes confirm 'No wounds; skin intact around cast.'" + }, + "target": { + "answer": "No wound", + "confidence": "High", + "explanation": "The referral packet states 'No wounds' alongside the right wrist fracture with cast, and today's SOC visit notes confirm 'No wounds; skin intact around cast.'" + }, + "notes": "" + } +] diff --git a/examples/prompt-lab/evidence/run/lab/work/q-wound-status/cd33704e/score.json b/examples/prompt-lab/evidence/run/lab/work/q-wound-status/cd33704e/score.json new file mode 100644 index 00000000..db6190de --- /dev/null +++ b/examples/prompt-lab/evidence/run/lab/work/q-wound-status/cd33704e/score.json @@ -0,0 +1,59 @@ +{ + "promptId": "p-wound-status-f1305d28", + "menus": [ + { + "options": [ + "Healed", + "Ongoing", + "No wound" + ], + "agencies": [ + "harbor", + "maple", + "sunrise" + ], + "score": { + "rows": [ + { + "patientId": "jordan", + "gold": "No wound", + "rerun": "No wound", + "result": "worked" + }, + { + "patientId": "pat", + "gold": "Ongoing", + "rerun": "Ongoing", + "result": "worked" + }, + { + "patientId": "riley", + "gold": "No wound", + "rerun": "No wound", + "result": "worked" + } + ], + "golded": 3, + "worked": 3, + "percent": 100 + }, + "outputs": { + "jordan": { + "answer": "No wound", + "confidence": "High", + "explanation": "Today's SOC visit notes state \"Skin intact, no wounds,\" documenting the current condition directly; the referral packet's intake form likewise recorded no wounds with skin intact." + }, + "pat": { + "answer": "Ongoing", + "confidence": "High", + "explanation": "Today's SOC visit notes document the left heel as an open area 2.0 x 1.5 cm with moderate serous drainage, a pink wound bed with yellow slough at the edges, and a foam dressing change per protocol. The hospital discharge summary stating the wound was closed is outdated and superseded by today's direct observation." + }, + "riley": { + "answer": "No wound", + "confidence": "High", + "explanation": "Today's SOC visit notes state 'No wounds; skin intact around cast,' describing the patient's current skin condition directly, which is consistent with the referral packet's note of no wounds." + } + } + } + ] +} diff --git a/examples/prompt-lab/evidence/run/lab/work/q-wound-status/d61687f9/grid.json b/examples/prompt-lab/evidence/run/lab/work/q-wound-status/d61687f9/grid.json deleted file mode 100644 index 7ea267f8..00000000 --- a/examples/prompt-lab/evidence/run/lab/work/q-wound-status/d61687f9/grid.json +++ /dev/null @@ -1,72 +0,0 @@ -[ - { - "questionId": "wound-status", - "patientId": "jordan", - "pile": "shared", - "highlight": [], - "ai": { - "answer": "No wound", - "confidence": "High", - "explanation": "The referral packet's intake form documents \"No wounds documented. Skin intact,\" and today's SOC visit note confirms \"Skin intact, no wounds\" — there is no primary wound to stage, and no discharge summary states a wound was closed or healed." - }, - "target": { - "answer": "No wound", - "confidence": "High", - "explanation": "The referral packet's intake form documents \"No wounds documented. Skin intact,\" and today's SOC visit note confirms \"Skin intact, no wounds\" — there is no primary wound to stage, and no discharge summary states a wound was closed or healed." - }, - "notes": "" - }, - { - "questionId": "wound-status", - "patientId": "marguerite", - "pile": "shared", - "highlight": [ - "medium confidence" - ], - "ai": { - "answer": "Ongoing", - "confidence": "Medium", - "explanation": "The referral packet never states the wound is closed or healed — it only records peristomal skin intact at discharge — so the visit notes govern, and today's assessment documents erythematous, moist, weepy denudement about 2 cm wide along the 4-to-8 o'clock peristomal aspect with patient-reported burning and stinging. Confidence is Medium rather than High because the status comes from the visit note, not a referral statement." - }, - "target": { - "answer": "Ongoing", - "confidence": "Medium", - "explanation": "The referral packet never states the wound is closed or healed — it only records peristomal skin intact at discharge — so the visit notes govern, and today's assessment documents erythematous, moist, weepy denudement about 2 cm wide along the 4-to-8 o'clock peristomal aspect with patient-reported burning and stinging. Confidence is Medium rather than High because the status comes from the visit note, not a referral statement." - }, - "notes": "" - }, - { - "questionId": "wound-status", - "patientId": "pat", - "pile": "shared", - "highlight": [], - "ai": { - "answer": "Ongoing", - "confidence": "High", - "explanation": "Today's SOC visit documents an open left heel wound measuring 2.0 x 1.5 cm with moderate serous drainage, pink wound bed and yellow slough, dressed with foam per protocol. The discharge summary's \"wound closed\" statement is contradicted by direct assessment at this visit, so the current status is Ongoing." - }, - "target": { - "answer": "Ongoing", - "confidence": "High", - "explanation": "Today's visit notes describe an open 2.0 x 1.5 cm left heel wound with serous drainage and slough; the discharge summary's closed status is outdated." - }, - "notes": "" - }, - { - "questionId": "wound-status", - "patientId": "riley", - "pile": "shared", - "highlight": [], - "ai": { - "answer": "No wound", - "confidence": "High", - "explanation": "The referral packet states 'No wounds' and today's SOC visit note confirms 'No wounds; skin intact around cast,' so there is no primary wound to track." - }, - "target": { - "answer": "No wound", - "confidence": "High", - "explanation": "The referral packet states 'No wounds' and today's SOC visit note confirms 'No wounds; skin intact around cast,' so there is no primary wound to track." - }, - "notes": "" - } -] diff --git a/examples/prompt-lab/evidence/run/lab/work/q-wound-status/d61687f9/score.json b/examples/prompt-lab/evidence/run/lab/work/q-wound-status/d61687f9/score.json deleted file mode 100644 index 22f032f4..00000000 --- a/examples/prompt-lab/evidence/run/lab/work/q-wound-status/d61687f9/score.json +++ /dev/null @@ -1,54 +0,0 @@ -{ - "promptId": "p-wound-status-f41f7ca1", - "rows": [ - { - "patientId": "jordan", - "gold": "No wound", - "rerun": "No wound", - "result": "worked" - }, - { - "patientId": "marguerite", - "gold": "Ongoing", - "rerun": "Ongoing", - "result": "worked" - }, - { - "patientId": "pat", - "gold": "Ongoing", - "rerun": "Ongoing", - "result": "worked" - }, - { - "patientId": "riley", - "gold": "No wound", - "rerun": "No wound", - "result": "worked" - } - ], - "golded": 4, - "worked": 4, - "percent": 100, - "outputs": { - "jordan": { - "answer": "No wound", - "confidence": "High", - "explanation": "Today's skilled nursing SOC visit directly assessed the skin and found it intact with no wounds, and the referral intake form likewise documented no wounds and intact skin. Nothing in the chart records any wound history or wound findings." - }, - "marguerite": { - "answer": "Ongoing", - "confidence": "High", - "explanation": "Today's start-of-care visit directly assessed the site and found erythematous, moist, weepy peristomal denudement in a 2 cm crescent along the 4-to-8 o'clock inferior aspect, with stool tracking under an undermined wafer and the patient reporting burning and stinging. The referral packet's note that peristomal skin was intact at discharge describes the site as of that earlier date and is outdated." - }, - "pat": { - "answer": "Ongoing", - "confidence": "High", - "explanation": "Today's SOC skilled nursing visit directly assessed the left heel and found an open area 2.0 x 1.5 cm with moderate serous drainage and a pink wound bed with yellow slough at the edges, with foam dressing applied. The hospital discharge summary's note that the wound was closed reflects status on that earlier date and is outdated." - }, - "riley": { - "answer": "No wound", - "confidence": "High", - "explanation": "Today's skilled nursing SOC visit directly assessed the skin and documented 'No wounds; skin intact around cast,' and the referral packet likewise records no wounds with only a right wrist fracture in a cast. Nothing in the chart documents any wound history or wound findings." - } - } -} diff --git a/examples/prompt-lab/evidence/run/lab/work/sunrise-soc/62c71a0f/commit.json b/examples/prompt-lab/evidence/run/lab/work/sunrise-soc/62c71a0f/commit.json deleted file mode 100644 index 236cb5ee..00000000 --- a/examples/prompt-lab/evidence/run/lab/work/sunrise-soc/62c71a0f/commit.json +++ /dev/null @@ -1,6 +0,0 @@ -{ - "mode": "except", - "questions": [ - "ostomy-supplies" - ] -} diff --git a/examples/prompt-lab/evidence/run/lab/work/sunrise-soc/8b7996f3/commit.json b/examples/prompt-lab/evidence/run/lab/work/sunrise-soc/8b7996f3/commit.json new file mode 100644 index 00000000..1d2f0357 --- /dev/null +++ b/examples/prompt-lab/evidence/run/lab/work/sunrise-soc/8b7996f3/commit.json @@ -0,0 +1,4 @@ +{ + "mode": "all", + "questions": [] +} diff --git a/examples/prompt-lab/evidence/run/lab/work/sunrise-soc/62c71a0f/grid.json b/examples/prompt-lab/evidence/run/lab/work/sunrise-soc/8b7996f3/grid.json similarity index 52% rename from examples/prompt-lab/evidence/run/lab/work/sunrise-soc/62c71a0f/grid.json rename to examples/prompt-lab/evidence/run/lab/work/sunrise-soc/8b7996f3/grid.json index 5d19a845..1f33405d 100644 --- a/examples/prompt-lab/evidence/run/lab/work/sunrise-soc/62c71a0f/grid.json +++ b/examples/prompt-lab/evidence/run/lab/work/sunrise-soc/8b7996f3/grid.json @@ -9,12 +9,12 @@ "ai": { "answer": "Lives with spouse", "confidence": "High", - "explanation": "Today's visit note narrative states the spouse was present throughout the SOC visit, that patient and spouse share the home, and that the spouse sets up the pill organizer each morning. This overrides the intake form completed by the hospital case manager, which recorded living alone." + "explanation": "Today's skilled nursing SOC visit notes state directly that the patient and their spouse share the home, with the spouse present throughout the visit and setting up the pill organizer each morning. The intake form's 'lives alone' entry is older supporting evidence and is treated as stale under today's direct observation." }, "target": { "answer": "Lives with spouse", "confidence": "High", - "explanation": "Today's visit note narrative states the spouse was present throughout the SOC visit, that patient and spouse share the home, and that the spouse sets up the pill organizer each morning. This overrides the intake form completed by the hospital case manager, which recorded living alone." + "explanation": "Today's skilled nursing SOC visit notes state directly that the patient and their spouse share the home, with the spouse present throughout the visit and setting up the pill organizer each morning. The intake form's 'lives alone' entry is older supporting evidence and is treated as stale under today's direct observation." }, "notes": "" }, @@ -28,12 +28,12 @@ "ai": { "answer": "Lives alone", "confidence": "High", - "explanation": "Today's SOC visit note narrative states directly that the patient lives alone in a one-story home, with a neighbor who checks in daily; a daily check-in visitor does not change residence status. This agrees with the referral packet's intake-form social history, so no override was needed." + "explanation": "Today's skilled nursing SOC visit note states directly that the patient lives alone in a one-story home, with a neighbor who checks in daily; an intermittent visitor does not count as a cohabitant. This is consistent with the referral packet's intake form, but the current visit note is the direct basis." }, "target": { "answer": "Lives alone", "confidence": "High", - "explanation": "Today's SOC visit note narrative states directly that the patient lives alone in a one-story home, with a neighbor who checks in daily; a daily check-in visitor does not change residence status. This agrees with the referral packet's intake-form social history, so no override was needed." + "explanation": "Today's skilled nursing SOC visit note states directly that the patient lives alone in a one-story home, with a neighbor who checks in daily; an intermittent visitor does not count as a cohabitant. This is consistent with the referral packet's intake form, but the current visit note is the direct basis." }, "notes": "" }, @@ -47,12 +47,12 @@ "ai": { "answer": "Lives with family", "confidence": "High", - "explanation": "Today's skilled nursing SOC visit note states directly that the patient resides with her daughter, son-in-law and two grandchildren. This agrees with the referral packet, which also describes her living with her daughter's family." + "explanation": "Today's visit notes state directly that the patient lives with her adult child, that child's spouse, and two grandchildren, which is a household of relatives other than a spouse. The referral packet agrees, but the visit note is the direct current statement; 'Lives alone' and the spouse option were rejected because multiple relatives reside in the home." }, "target": { "answer": "Lives with family", "confidence": "High", - "explanation": "Today's skilled nursing SOC visit note states directly that the patient resides with her daughter, son-in-law and two grandchildren. This agrees with the referral packet, which also describes her living with her daughter's family." + "explanation": "Today's visit notes state directly that the patient lives with her adult child, that child's spouse, and two grandchildren, which is a household of relatives other than a spouse. The referral packet agrees, but the visit note is the direct current statement; 'Lives alone' and the spouse option were rejected because multiple relatives reside in the home." }, "notes": "" }, @@ -67,12 +67,12 @@ "ai": { "answer": "Calm", "confidence": "Medium", - "explanation": "Today's visit notes describe the patient as \"relaxed and engaged\" and asking good questions about the new diuretic, but the mood is inferred from the clinician's behavioral description rather than stated by the patient in their own words." + "explanation": "The visit notes describe the patient as \"relaxed and engaged\" and asking good questions about the diuretic, but the mood is inferred from the clinician's behavioral description rather than stated by the patient in their own words." }, "target": { "answer": "Calm", "confidence": "Medium", - "explanation": "Today's visit notes describe the patient as \"relaxed and engaged\" and asking good questions about the new diuretic, but the mood is inferred from the clinician's behavioral description rather than stated by the patient in their own words." + "explanation": "The visit notes describe the patient as \"relaxed and engaged\" and asking good questions about the diuretic, but the mood is inferred from the clinician's behavioral description rather than stated by the patient in their own words." }, "notes": "" }, @@ -86,12 +86,12 @@ "ai": { "answer": "Calm", "confidence": "High", - "explanation": "Today's visit notes describe the patient as \"calm and pleasant, joking about hospital food,\" and she states in her own words that she is \"glad to be home.\"" + "explanation": "The visit notes describe the patient as \"calm and pleasant, joking about hospital food,\" and she states in her own words that she is glad to be home." }, "target": { "answer": "Calm", "confidence": "High", - "explanation": "Today's visit notes describe the patient as \"calm and pleasant, joking about hospital food,\" and she states in her own words that she is \"glad to be home.\"" + "explanation": "The visit notes describe the patient as \"calm and pleasant, joking about hospital food,\" and she states in her own words that she is glad to be home." }, "notes": "" }, @@ -116,37 +116,39 @@ }, { "questionId": "wound-status", - "patientId": "jordan", + "patientId": "pat", "pile": "shared", - "highlight": [], + "highlight": [ + "medium confidence" + ], "ai": { - "answer": "No wound", - "confidence": "High", - "explanation": "The referral packet's intake form documents 'No wounds documented. Skin intact,' and today's SOC visit note confirms 'Skin intact, no wounds' — no wound was ever present, so there is nothing to have healed." + "answer": "Ongoing", + "confidence": "Medium", + "explanation": "The referral discharge summary states the left heel wound was closed, but today's SOC visit notes document an open left heel area 2.0 x 1.5 cm with moderate serous drainage and yellow slough, so the wound is not healed." }, "target": { - "answer": "No wound", + "answer": "Ongoing", "confidence": "High", - "explanation": "The referral packet's intake form documents 'No wounds documented. Skin intact,' and today's SOC visit note confirms 'Skin intact, no wounds' — no wound was ever present, so there is nothing to have healed." + "explanation": "Today's visit notes describe an open 2.0 x 1.5 cm left heel wound with serous drainage and slough; the discharge summary's closed status is outdated." }, - "notes": "" + "notes": "The prompt makes the referral win over what the nurse saw today. Today's notes must win (shared guideline 1)." }, { "questionId": "wound-status", - "patientId": "pat", + "patientId": "jordan", "pile": "shared", "highlight": [], "ai": { - "answer": "Ongoing", + "answer": "No wound", "confidence": "High", - "explanation": "Today's SOC visit documents an open left heel wound measuring 2.0 x 1.5 cm with moderate serous drainage and yellow slough, dressed with foam per protocol, so the discharge summary's \"wound closed\" note is outdated and the current assessment governs." + "explanation": "The referral packet's intake form documents \"No wounds documented. Skin intact,\" and today's SOC visit notes confirm \"Skin intact, no wounds.\"" }, "target": { - "answer": "Ongoing", + "answer": "No wound", "confidence": "High", - "explanation": "Today's visit notes describe an open 2.0 x 1.5 cm left heel wound with serous drainage and slough; the discharge summary's closed status is outdated." + "explanation": "The referral packet's intake form documents \"No wounds documented. Skin intact,\" and today's SOC visit notes confirm \"Skin intact, no wounds.\"" }, - "notes": "The prompt makes the referral win over what the nurse saw today. Today's notes must win (shared guideline 1)." + "notes": "" }, { "questionId": "wound-status", @@ -156,12 +158,12 @@ "ai": { "answer": "No wound", "confidence": "High", - "explanation": "The referral packet states \"No wounds\" and today's SOC visit note confirms \"No wounds; skin intact around cast,\" so there is no primary wound to stage or track." + "explanation": "The referral packet states \"No wounds\" and today's SOC visit notes confirm \"No wounds; skin intact around cast,\" so there is no primary wound to status." }, "target": { "answer": "No wound", "confidence": "High", - "explanation": "The referral packet states \"No wounds\" and today's SOC visit note confirms \"No wounds; skin intact around cast,\" so there is no primary wound to stage or track." + "explanation": "The referral packet states \"No wounds\" and today's SOC visit notes confirm \"No wounds; skin intact around cast,\" so there is no primary wound to status." }, "notes": "" } diff --git a/examples/prompt-lab/jobs/fix.ts b/examples/prompt-lab/jobs/fix.ts index b18a0bb9..37426b43 100644 --- a/examples/prompt-lab/jobs/fix.ts +++ b/examples/prompt-lab/jobs/fix.ts @@ -11,12 +11,12 @@ import { changed, gridError, highlights, rowKey, type Row } from "../lib/grid.ts"; import { hash8 } from "../lib/hash.ts"; import { shellWord } from "../lib/lab.ts"; -import { agenciesUsing } from "../lib/piles.ts"; -import { score, scoreTable } from "../lib/score.ts"; +import { agenciesUsing, distinctMenus } from "../lib/piles.ts"; +import { score, scoreTable, type Score } from "../lib/score.ts"; import type { Output, PatientBrief } from "../lib/types.ts"; import { iterate, promptQa, type Ask } from "../prompts.ts"; import type { Job } from "./job.ts"; -import { MAX_ATTEMPTS, planCoverage, runEngine, untilQaPasses } from "./shared.ts"; +import { gapId, MAX_ATTEMPTS, planCoverage, runEngine, untilQaPasses } from "./shared.ts"; export interface FixInput { questionId?: string; issueId?: string; agency?: string } @@ -33,9 +33,14 @@ export async function fix(job: Job, input: FixInput): Promise { const using = agenciesUsing(snap.agencies, qid); const agency = input.agency ?? issue?.agency ?? using[0]!; - const menu = Object.values(snap.agencies[agency]?.visitTypes ?? {}).flat().find((q) => q.questionId === qid)?.options; - if (!menu) throw new Error(`agency ${agency} does not ask ${qid}`); + const asked = Object.entries(snap.agencies[agency]?.visitTypes ?? {}) + .flatMap(([visitType, qs]) => qs.filter((q) => q.questionId === qid).map((q) => ({ visitType, options: q.options })))[0]; + if (!asked) throw new Error(`agency ${agency} does not ask ${qid}`); + const menu = asked.options; const ask: Ask = { question: question.text, options: menu }; + // A prompt is global: done changes it on every agency's menu, so the re-run + // checks every distinct menu, not only the one gold was set on. + const menus = distinctMenus(snap.agencies, using, qid); const pile = using.length > 1 ? "shared" : "agency-specific"; const brief = snap.briefs[qid]; @@ -43,7 +48,7 @@ export async function fix(job: Job, input: FixInput): Promise { const plan = await planCoverage(f, [{ id: qid, ...ask }], snap.shelf); const ids = new Set([...(plan.coverage[0]?.patientIds ?? []), ...Object.keys(issue?.targets ?? {})]); if (plan.gaps[0]) { - const gap: PatientBrief = { id: `gap-${qid}`, questionId: qid, brief: plan.gaps[0].brief, from: "planner", status: "queued" }; + const gap: PatientBrief = { id: gapId(qid, asked.visitType), questionId: qid, visitType: asked.visitType, brief: plan.gaps[0].brief, from: "planner", status: "queued" }; await lab.enqueue("patient-briefs", gap); } if (ids.size === 0) return f.done("needs_human"); // nothing on the shelf can show it yet; the gap brief is queued @@ -87,13 +92,18 @@ export async function fix(job: Job, input: FixInput): Promise { const { promptId } = await lab.draft(qid, rewrite.value.prompt); // Re-run and score: required before done. Worked = new answer matches persisted gold. - const rerun = await runEngine(f, rewrite.value.prompt, ask, patients); - const result = score(qid, rerun, { ...snap.gold, ...gold }); - await lab.writeNew(`${work}/score.json`, { promptId, ...result, outputs: rerun }); + const allGold = { ...snap.gold, ...gold }; + const results: { agencies: string[]; options: string[]; score: Score; outputs: Record }[] = []; + for (const m of menus) { // sequential: see runEngine + const outputs = await runEngine(f, rewrite.value.prompt, { question: question.text, options: m.options }, patients); + results.push({ ...m, score: score(qid, outputs, allGold), outputs }); + } + await lab.writeNew(`${work}/score.json`, { promptId, menus: results }); + const table = results.map((r) => (results.length > 1 ? `Menu of ${r.agencies.join(", ")} (${r.options.join(" / ")}):\n` : "") + scoreTable(r.score)).join("\n"); const shared = using.length > 1 ? `\nSHARED: marking done changes this prompt for every agency that uses it: ${using.join(", ")}.` : ""; const done = await f.human( - `Re-run and score · ${qid}, new prompt ${promptId}:\n${scoreTable(result)}${shared}\n` + + `Re-run and score · ${qid}, new prompt ${promptId}:\n${table}${shared}\n` + `Read the new prompt in ${job.labDir}/bank.json and the outputs in ${job.labDir}/${work}/score.json.\n` + `yes = mark done (live in Apricot now, no promote step); no = keep it as a draft and iterate again in a new run.`, { to: reviewer }); diff --git a/examples/prompt-lab/jobs/new-agency.ts b/examples/prompt-lab/jobs/new-agency.ts index c05117b6..75801b87 100644 --- a/examples/prompt-lab/jobs/new-agency.ts +++ b/examples/prompt-lab/jobs/new-agency.ts @@ -18,7 +18,7 @@ import type { Issue, Output, Patient, PatientBrief } from "../lib/types.ts"; import { firstPass, iterate, promptQa, type Ask } from "../prompts.ts"; import type { Job } from "./job.ts"; import { shellWord } from "../lib/lab.ts"; -import { llm, MAX_ATTEMPTS, planCoverage, runEngine, untilQaPasses } from "./shared.ts"; +import { gapId, llm, MAX_ATTEMPTS, planCoverage, runEngine, untilQaPasses } from "./shared.ts"; export interface NewAgencyInput { agency: string; visitType: string } @@ -50,10 +50,12 @@ export async function newAgency(job: Job, input: NewAgencyInput): Promise prompts.set(q.questionId, { text: v0.value.prompt, firstPass: true }); } - // Test planner: existing shelf only; holes become briefs on the manager queue. - const plan = await planCoverage(f, piled.map((q) => ({ id: q.questionId, ...ask(q) })), snap.shelf); + // Test planner: existing shelf patients of this visit type only; holes become + // briefs on the manager queue for that visit type. + const shelf = snap.shelf.filter((p) => p.visitType === input.visitType); + const plan = await planCoverage(f, piled.map((q) => ({ id: q.questionId, ...ask(q) })), shelf); for (const gap of plan.gaps) { - const brief: PatientBrief = { id: `gap-${gap.questionId}`, questionId: gap.questionId, brief: gap.brief, from: "planner", status: "queued" }; + const brief: PatientBrief = { id: gapId(gap.questionId, input.visitType), questionId: gap.questionId, visitType: input.visitType, brief: gap.brief, from: "planner", status: "queued" }; await lab.enqueue("patient-briefs", brief); } @@ -109,7 +111,11 @@ export async function newAgency(job: Job, input: NewAgencyInput): Promise rewrites.push({ q, rows, prompt: r.value.prompt, passed: r.passed }); } - const candidates = new Set(piled.filter((q) => q.pile === "agency-specific" && prompts.get(q.questionId)!.firstPass).map((q) => q.questionId)); + // A first-pass prompt is a candidate only once it ran on a shelf patient and + // you reviewed the rows; a gap-only draft waits for a patient that covers it. + const covered = new Set(plan.coverage.map((c) => c.questionId)); + const candidates = new Set(piled.filter((q) => q.pile === "agency-specific" && prompts.get(q.questionId)!.firstPass && covered.has(q.questionId)).map((q) => q.questionId)); + const waiting = piled.filter((q) => prompts.get(q.questionId)!.firstPass && !covered.has(q.questionId)).map((q) => q.questionId); const sent: string[] = []; for (const r of rewrites) { if (r.q.pile === "agency-specific") { @@ -134,6 +140,7 @@ export async function newAgency(job: Job, input: NewAgencyInput): Promise const commit = await f.human( `Commit output · ${input.agency}: agency-specific prompts ready to go live in Apricot: ${list.join(", ")}.\n` + `Sent to the question manager (frozen here): ${sent.length ? sent.join("; ") : "none"}.\n` + + (waiting.length ? `Held as drafts until a shelf patient covers them: ${waiting.join(", ")}.\n` : "") + `To commit only some, edit ${job.labDir}/${work}/commit.json: {"mode":"only"|"except","questions":[...]}.\n` + `yes = commit (Done = live, no promote step); no = leave them as drafts.`, { to: reviewer }); diff --git a/examples/prompt-lab/jobs/patient.ts b/examples/prompt-lab/jobs/patient.ts index 6424977f..ee958a7d 100644 --- a/examples/prompt-lab/jobs/patient.ts +++ b/examples/prompt-lab/jobs/patient.ts @@ -15,7 +15,7 @@ import { patientChart, patientPlan, patientQa } from "../prompts.ts"; import type { Job } from "./job.ts"; import { llm, MAX_ATTEMPTS, untilQaPasses } from "./shared.ts"; -export interface PatientInput { briefId?: string; brief?: string; questionId?: string } +export interface PatientInput { briefId?: string; brief?: string; questionId?: string; visitType?: string } type Chart = Omit; export async function createPatient(job: Job, input: PatientInput): Promise { @@ -25,14 +25,17 @@ export async function createPatient(job: Job, input: PatientInput): Promise(f, patientPlan(brief, snap.bank.questions[questionId].text)); - const work = `work/patients/${queued?.id ?? `manual-${hash8(brief)}`}`; + // Keyed by the plan itself: a later run of the same brief gets its own file, + // never the one an earlier, declined run left behind. + const work = `work/patients/${queued?.id ?? `manual-${hash8(brief)}`}/${hash8(plan)}`; await lab.writeNew(`${work}/plan.json`, { brief, plan }); const kick = await f.human( - `Test patient creator · ${queued?.id ?? "manual brief"} for ${questionId}.\nPlan:\n${plan}\n` + + `Test patient creator · ${queued?.id ?? "manual brief"} for ${questionId} (${visitType} visit).\nPlan:\n${plan}\n` + `You may edit "plan" in ${job.labDir}/${work}/plan.json first.\n` + `yes = kick generate (Patient QA then locks it onto the shelf; you do not approve the chart); no = leave it on the queue.`, { to: reviewer }); @@ -42,7 +45,7 @@ export async function createPatient(job: Job, input: PatientInput): Promise p.id)); const made = await untilQaPasses(f, - (findings) => patientChart(brief, final.plan, findings), + (findings) => patientChart(brief, final.plan, visitType, findings), (chart) => { // Deterministic floor first: identifiers and id collisions never reach model QA. const hard = identifiers(`${chart.label} ${chart.referral} ${chart.notes}`); diff --git a/examples/prompt-lab/jobs/shared.ts b/examples/prompt-lab/jobs/shared.ts index 33f838b1..732b0f54 100644 --- a/examples/prompt-lab/jobs/shared.ts +++ b/examples/prompt-lab/jobs/shared.ts @@ -80,6 +80,9 @@ export async function runEngine(f: Ctx, promptText: string, ask: Ask, patients: export interface Plan { coverage: { questionId: string; patientIds: string[] }[]; gaps: { questionId: string; brief: string }[] } +/** One queued gap per question × visit type: a re-run's planner keeps the brief already waiting. */ +export const gapId = (questionId: string, visitType: string): string => `gap-${questionId}-${visitType}`; + /** Test QA, deterministic: every question exactly once, covered by real shelf patients or a gap brief. */ export function planError(plan: Plan, questionIds: readonly string[], shelf: readonly Patient[]): string | null { const onShelf = new Set(shelf.map((p) => p.id)); diff --git a/examples/prompt-lab/lib/piles.ts b/examples/prompt-lab/lib/piles.ts index 94258b16..534199e6 100644 --- a/examples/prompt-lab/lib/piles.ts +++ b/examples/prompt-lab/lib/piles.ts @@ -37,3 +37,19 @@ export function agenciesUsing(agencies: Record, questionId: stri .filter(([, a]) => Object.values(a.visitTypes).some((qs) => qs.some((q) => q.questionId === questionId))) .map(([id]) => id).sort(); } + +/** The distinct answer menus `questionId` is asked with, each with the agencies that use it. */ +export function distinctMenus(agencies: Record, using: readonly string[], questionId: string): { options: string[]; agencies: string[] }[] { + const menus: { options: string[]; agencies: string[] }[] = []; + for (const id of using) { + for (const qs of Object.values(agencies[id]!.visitTypes)) { + for (const q of qs) { + if (q.questionId !== questionId) continue; + const same = menus.find((m) => sameList(m.options, q.options)); + if (!same) menus.push({ options: q.options, agencies: [id] }); + else if (!same.agencies.includes(id)) same.agencies.push(id); + } + } + } + return menus; +} diff --git a/examples/prompt-lab/lib/types.ts b/examples/prompt-lab/lib/types.ts index 55caa4d0..f0b1fbcb 100644 --- a/examples/prompt-lab/lib/types.ts +++ b/examples/prompt-lab/lib/types.ts @@ -31,7 +31,7 @@ export interface Patient { /** A gap brief: coverage the shelf does not have yet. Lives on the manager queue. */ export interface PatientBrief { - id: string; questionId: string; brief: string; from: "planner" | "you" | "issue"; + id: string; questionId: string; visitType: string; brief: string; from: "planner" | "you" | "issue"; status: "queued" | "locked"; } diff --git a/examples/prompt-lab/prompts.ts b/examples/prompt-lab/prompts.ts index 059a20aa..a5a452f1 100644 --- a/examples/prompt-lab/prompts.ts +++ b/examples/prompt-lab/prompts.ts @@ -134,7 +134,7 @@ Return JSON { "plan": "" }.`, } /** Test patient creator, step 2: generate the chart from brief + plan. */ -export function patientChart(brief: string, plan: string, findings: readonly string[]): Call { +export function patientChart(brief: string, plan: string, visitType: string, findings: readonly string[]): Call { return { prompt: `Generate the invented home-health test patient described by this plan, in the shelf's chart shape. @@ -145,8 +145,8 @@ PATIENT PLAN: ${plan} ${findings.length ? `\nPatient QA rejected the previous chart for:\n- ${findings.join("\n- ")}\nFix every point.` : ""} Rules: a first-name label only; no surnames, dates, MRNs, phone numbers or addresses. id is the lowercase label. -Return JSON { "id", "label", "ageBand", "visitType": "soc", "referral", "notes" }.`, - output: obj({ id: { type: "string", pattern: "^[a-z][a-z0-9-]{1,30}$" }, label: str, ageBand: str, visitType: { const: "soc" }, referral: { type: "string", minLength: 60 }, notes: { type: "string", minLength: 80 } }), +Return JSON { "id", "label", "ageBand", "visitType": "${visitType}", "referral", "notes" }.`, + output: obj({ id: { type: "string", pattern: "^[a-z][a-z0-9-]{1,30}$" }, label: str, ageBand: str, visitType: { const: visitType }, referral: { type: "string", minLength: 60 }, notes: { type: "string", minLength: 80 } }), }; } diff --git a/examples/prompt-lab/prove.sh b/examples/prompt-lab/prove.sh index 612d8203..d3209f27 100755 --- a/examples/prompt-lab/prove.sh +++ b/examples/prompt-lab/prove.sh @@ -5,7 +5,8 @@ # kernel with `flows run --local-agent`. At every parked `f.human` gate this # script plays the reviewer named in the input: it applies the documented edit # (captured as a diff), answers with `flows answer`, and `flows resume`s. -# Every command is captured with its literal output and exit code. +# Every command is captured with its literal output and exit code, and the +# script stops on the first exit it did not expect. # # ./prove.sh [out-dir] default: evidence/run set -uo pipefail @@ -15,66 +16,84 @@ LAB=$OUT/lab DD=${FLOWS_DATA_DIR:-$(mktemp -d)} REVIEWER=prompt-lab-reviewer N=0 +STATUS= +FILE= -[ -e "$LAB" ] && { echo "refusing: $LAB exists" >&2; exit 2; } +die() { echo "prove: $*" >&2; exit 1; } +[ -e "$LAB" ] && die "refusing: $LAB exists" mkdir -p "$OUT" -node --no-warnings --experimental-strip-types store.ts "$LAB" seed fixtures +node --no-warnings --experimental-strip-types store.ts "$LAB" seed fixtures || die "seed failed" -# capture : run it, write "$ cmd", its output and exit code to $OUT/NN-name.txt +# capture -- : run it, write "$ cmd", its +# output and exit code to $OUT/NN-name.txt, and stop unless the exit is one of +# the expected ones (0 completed, 3 parked on a gate). capture() { - local name=$1; shift + local want=() + while [ "$1" != "--" ]; do want+=("$1"); shift; done + shift + local name=$1 + shift N=$((N + 1)) FILE=$(printf '%s/%02d-%s.txt' "$OUT" "$N" "$name") - { echo "\$ $*"; "$@" 2>&1; echo "exit=$?"; } > "$FILE" + { echo "\$ $*"; "$@" 2>&1; STATUS=$?; echo "exit=$STATUS"; } > "$FILE" grep -v 'WAITING\|↻\|○' "$FILE" | tail -4 | cut -c1-240 + local w + for w in "${want[@]}"; do [ "$STATUS" = "$w" ] && return 0; done + die "$FILE exited $STATUS, expected ${want[*]}" } run_id() { grep -o 'RUN [0-9A-Z]\{26\}' "$FILE" | tail -1 | cut -d' ' -f2; } wait_id() { grep -o ' human-[0-9]* yes|no' "$FILE" | tail -1 | awk '{print $1}'; } -parked_on() { grep -q "PARKED.*$1" "$FILE"; } +parked_on() { grep -q "PARKED.*$1" "$FILE" || die "$FILE did not park on \"$1\""; } +gate_path() { grep -o "$LAB/work/[^ ]*\.json" "$FILE" | head -1; } # edit : the reviewer's change, captured as a diff edit() { local file=$1 expr=$2 before + [ -f "$file" ] || die "no file to edit: $file" before=$(mktemp) cp "$file" "$before" - node -e "const fs=require('fs');const v=JSON.parse(fs.readFileSync('$file','utf8'));$expr;fs.writeFileSync('$file',JSON.stringify(v,null,2)+'\n')" + node -e "const fs=require('fs');const v=JSON.parse(fs.readFileSync('$file','utf8'));$expr;fs.writeFileSync('$file',JSON.stringify(v,null,2)+'\n')" \ + || die "reviewer edit failed on $file" N=$((N + 1)) { echo "# reviewer edit: $file"; diff -u "$before" "$file" | tail -n +3; } > "$(printf '%s/%02d-reviewer-edit.diff' "$OUT" "$N")" } -gate_path() { grep -o "$LAB/work/[^ ]*\.json" "$FILE" | head -1; } +# answer_and_resume : 3 when another gate follows, 0 when the run should complete. answer_and_resume() { local run wait - run=$(run_id); wait=$(wait_id) - capture answer npx flows answer --data-dir "$DD" "$run" "$wait" yes --by "$REVIEWER" - capture resume npx flows resume --no-observer-link --data-dir "$DD" --local-agent "$run" + run=$(run_id) + wait=$(wait_id) + [ -n "$run" ] && [ -n "$wait" ] || die "no parked gate in $FILE" + capture 0 -- answer npx flows answer --data-dir "$DD" "$run" "$wait" yes --by "$REVIEWER" + capture "$1" -- resume npx flows resume --no-observer-link --data-dir "$DD" --local-agent "$run" } -flow() { capture "$1" npx flows run --no-observer-link --data-dir "$DD" --local-agent prompt-lab.flow.ts --input "$2"; } +flow() { capture 3 -- "$1" npx flows run --no-observer-link --data-dir "$DD" --local-agent prompt-lab.flow.ts --input "$2"; } echo "== Job 1: new agency sunrise / soc" flow job1-run "{\"job\":\"new-agency\",\"reviewer\":\"$REVIEWER\",\"lab\":\"$LAB\",\"agency\":\"sunrise\",\"visitType\":\"soc\"}" -parked_on "Config workbench" || { echo "job 1 did not reach the grid gate" >&2; exit 1; } +parked_on "Config workbench" # The reviewer's first pass: Pat's wound is open today, whatever the discharge summary said. edit "$(gate_path)" 'const r=v.find(r=>r.questionId==="wound-status"&&r.patientId==="pat");r.target={answer:"Ongoing",confidence:"High",explanation:"Today'"'"'s visit notes describe an open 2.0 x 1.5 cm left heel wound with serous drainage and slough; the discharge summary'"'"'s closed status is outdated."};r.notes="The prompt makes the referral win over what the nurse saw today. Today'"'"'s notes must win (shared guideline 1)."' -answer_and_resume -parked_on "Commit output" || { echo "job 1 did not reach the commit gate" >&2; exit 1; } -# ostomy-supplies has no shelf patient yet, so its first-pass prompt was never run: hold it back. -edit "$(gate_path)" 'v.mode="except";v.questions=["ostomy-supplies"]' -answer_and_resume +answer_and_resume 3 +parked_on "Commit output" +# Only covered prompts are offered; ostomy-supplies (no shelf patient yet) is held as a draft. +grep -q "Held as drafts until a shelf patient covers them: ostomy-supplies" "$FILE" || die "ostomy-supplies was offered for commit" +answer_and_resume 0 # commit all offered echo "== Test patient creator: the ostomy gap brief Job 1 queued" -flow patient-run "{\"job\":\"patient\",\"reviewer\":\"$REVIEWER\",\"lab\":\"$LAB\",\"briefId\":\"gap-ostomy-supplies\"}" -parked_on "Test patient creator" || { echo "patient job did not reach kick generate" >&2; exit 1; } -answer_and_resume # kick generate; Patient QA locks it, nobody approves the chart +flow patient-run "{\"job\":\"patient\",\"reviewer\":\"$REVIEWER\",\"lab\":\"$LAB\",\"briefId\":\"gap-ostomy-supplies-soc\"}" +parked_on "Test patient creator" +answer_and_resume 0 # kick generate; Patient QA locks it, nobody approves the chart echo "== Job 2: fix wound-status from the config-send issue" -ISSUE=$(node -e "const q=require('./$LAB/queue/issues.json');console.log(q.find(i=>i.questionIds[0]==='wound-status').id)") +ISSUE=$(node -e "const q=require('./$LAB/queue/issues.json');console.log(q.find(i=>i.questionIds[0]==='wound-status').id)") \ + || die "no wound-status issue was sent to the question manager" flow job2-run "{\"job\":\"fix\",\"reviewer\":\"$REVIEWER\",\"lab\":\"$LAB\",\"issueId\":\"$ISSUE\"}" -parked_on "Question workbench" || { echo "job 2 did not reach the gold gate" >&2; exit 1; } -answer_and_resume # gold as prefilled: Pat's config-level target travelled with the issue -parked_on "Re-run and score" || { echo "job 2 did not reach done" >&2; exit 1; } -answer_and_resume # mark done = live, for every agency using it +parked_on "Question workbench" +answer_and_resume 3 # gold as prefilled: Pat's config-level target travelled with the issue +parked_on "Re-run and score" +answer_and_resume 0 # mark done = live, for every agency using it echo "== Final lab state" -capture lab-state node -e " +capture 0 -- lab-state node -e " const b=require('./$LAB/bank.json'); for (const [q,v] of Object.entries(b.questions)) console.log(q.padEnd(17),'live',String(v.livePromptId).padEnd(30),'draft',v.draftPromptId??'-'); console.log('shelf', require('fs').readdirSync('$LAB/shelf').join(' ')); diff --git a/examples/prompt-lab/store.ts b/examples/prompt-lab/store.ts index a52ed669..71748b6a 100644 --- a/examples/prompt-lab/store.ts +++ b/examples/prompt-lab/store.ts @@ -17,24 +17,42 @@ // node store.ts close-issue // node store.ts lock-patient [briefId] import { createHash } from "node:crypto"; -import { cpSync, existsSync, mkdirSync, readdirSync, readFileSync, renameSync, writeFileSync } from "node:fs"; +import { cpSync, existsSync, mkdirSync, readdirSync, readFileSync, renameSync, rmSync, writeFileSync } from "node:fs"; import { dirname, join } from "node:path"; import type { Bank, Issue, Patient, PatientBrief, Snapshot } from "./lib/types.ts"; const OUTPUT_LIMIT = 60 * 1024; // the kernel journals a 64 KiB stdout tail; never let a read be cut. -const [lab, verb, ...args] = process.argv.slice(2); -if (!lab || !verb) fail("usage: store.ts [args]"); +const LOCK_WAIT_MS = 10_000; +/** Verbs that read-modify-write: serialized across processes, so concurrent runs never lose an update. */ +const MUTATING = new Set(["write-new", "record", "draft", "publish", "enqueue", "close-issue", "lock-patient"]); +const [lab, verb, ...args] = process.argv.slice(2) as [string, string, ...string[]]; + +class StoreError extends Error {} +/** Throws, never exits: the lock is released on the way out. */ function fail(message: string): never { - process.stderr.write(`store: ${message}\n`); - process.exit(1); + throw new StoreError(message); } function print(value: unknown): void { const text = JSON.stringify(value); - if (text.length > OUTPUT_LIMIT) fail(`output is ${text.length} bytes, over the ${OUTPUT_LIMIT}-byte journal tail`); + const bytes = Buffer.byteLength(text, "utf8"); + if (bytes > OUTPUT_LIMIT) fail(`output is ${bytes} bytes, over the ${OUTPUT_LIMIT}-byte journal tail`); process.stdout.write(`${text}\n`); } +/** An exclusive lab lock: mkdir is atomic, so one process holds it at a time. */ +function withLock(work: () => void): void { + const lock = join(lab, ".lock"); + const deadline = Date.now() + LOCK_WAIT_MS; + for (;;) { + try { mkdirSync(lock); break; } catch (error) { + if ((error as NodeJS.ErrnoException).code !== "EEXIST") throw error; + if (Date.now() > deadline) fail(`lab is locked (${lock}); remove it if no store process is running`); + Atomics.wait(new Int32Array(new SharedArrayBuffer(4)), 0, 0, 20); + } + } + try { work(); } finally { rmSync(lock, { recursive: true, force: true }); } +} const path = (rel: string): string => { if (rel.startsWith("/") || rel.split("/").includes("..")) fail(`path escapes the lab: ${rel}`); return join(lab!, rel); @@ -54,6 +72,7 @@ function writeJson(rel: string, value: unknown): void { } const listDir = (rel: string): string[] => (existsSync(path(rel)) ? readdirSync(path(rel)).sort() : []); +function run(): void { switch (verb) { case "seed": { if (existsSync(join(lab, "bank.json"))) fail(`${lab} is already a lab; refusing to overwrite it`); @@ -155,3 +174,12 @@ switch (verb) { default: fail(`unknown verb ${verb}`); } +} + +try { + if (!lab || !verb) fail("usage: store.ts [args]"); + if (MUTATING.has(verb)) withLock(run); else run(); +} catch (error) { + process.stderr.write(`store: ${error instanceof Error ? error.message : String(error)}\n`); + process.exitCode = 1; +} diff --git a/examples/prompt-lab/tests/lib.test.ts b/examples/prompt-lab/tests/lib.test.ts index a49b7450..076c1cb2 100644 --- a/examples/prompt-lab/tests/lib.test.ts +++ b/examples/prompt-lab/tests/lib.test.ts @@ -4,7 +4,7 @@ import { test } from "node:test"; import { commitError, toCommit } from "../lib/commit.ts"; import { changed, gridError, highlights, sortRows, type Row } from "../lib/grid.ts"; import { identifiers } from "../lib/phi.ts"; -import { agenciesUsing, piles } from "../lib/piles.ts"; +import { agenciesUsing, distinctMenus, piles } from "../lib/piles.ts"; import { score } from "../lib/score.ts"; import type { Agency, Output, Patient } from "../lib/types.ts"; import { planError } from "../jobs/shared.ts"; @@ -111,3 +111,13 @@ test("reply schema: the engine's answer enum is the agency menu", () => { assert.match(schemaError(plan, { coverage: [{ questionId: "a", patientIds: [] }], gaps: [] })!, /needs at least 1/); assert.throws(() => schemaError({ type: "string", format: "date" }, "x"), /not supported/); }); + +test("distinctMenus: a mismatch question is re-run on every menu it is asked with", () => { + assert.deepEqual(distinctMenus(agencies, agenciesUsing(agencies, "mood"), "mood"), [ + { options: ["Calm", "Anxious", "Low"], agencies: ["harbor"] }, + { options: ["Calm", "Anxious", "Low", "Agitated"], agencies: ["sunrise"] }, + ]); + assert.deepEqual(distinctMenus(agencies, agenciesUsing(agencies, "wound-status"), "wound-status"), [ + { options: ["Healed", "Ongoing", "No wound"], agencies: ["harbor", "maple", "sunrise"] }, + ]); +}); diff --git a/examples/prompt-lab/tests/store.test.ts b/examples/prompt-lab/tests/store.test.ts index 8daa1004..039d6c41 100644 --- a/examples/prompt-lab/tests/store.test.ts +++ b/examples/prompt-lab/tests/store.test.ts @@ -1,6 +1,6 @@ import assert from "node:assert/strict"; -import { spawnSync } from "node:child_process"; -import { mkdtempSync, readFileSync, writeFileSync } from "node:fs"; +import { spawn, spawnSync } from "node:child_process"; +import { existsSync, mkdtempSync, readFileSync, writeFileSync } from "node:fs"; import { tmpdir } from "node:os"; import { join } from "node:path"; import { test } from "node:test"; @@ -69,3 +69,36 @@ test("enqueue is keyed by id; lock-patient freezes a chart and closes its brief" assert.equal(s.shelf.find((x: { id: string }) => x.id === "sam").locked, true); assert.equal(s.patientBriefs[0].status, "locked"); }); + +test("concurrent store processes never lose an update (the lab lock serializes read-modify-write)", async () => { + const { lab, store } = seeded(); + const run = (...args: string[]) => new Promise((resolve) => { + spawn("node", ["--no-warnings", "--experimental-strip-types", STORE, lab, ...args]).on("close", resolve); + }); + const texts = Array.from({ length: 8 }, (_, i) => `prompt text number ${i}`); + const codes = await Promise.all([ + ...texts.map((t) => run("draft", "mood", b64(t))), + ...texts.map((_, i) => run("enqueue", "issues", b64({ id: `i-${i}` }))), + ]); + assert.deepEqual(codes, Array(16).fill(0)); + const s = store("snapshot").out; + for (const t of texts) assert.ok(Object.values(s.bank.prompts).includes(t), `lost draft: ${t}`); + assert.equal(s.issues.length, 8); + assert.equal(existsSync(join(lab, ".lock")), false); +}); + +test("a failing mutating verb releases the lock", () => { + const { lab, store } = seeded(); + assert.equal(store("publish", "mood", "p-nope").status, 1); + assert.equal(existsSync(join(lab, ".lock")), false); + assert.equal(store("draft", "mood", b64("still writable")).status, 0); +}); + +test("the output limit counts UTF-8 bytes, not characters", () => { + const { store } = seeded(); + // 25k characters (é is 2 bytes, 中 is 3): about 62.5 KB of UTF-8, under the limit in UTF-16 units. + assert.equal(store("write-new", "work/big.json", b64("é中".repeat(12_500))).status, 0); + const r = store("read", "work/big.json"); + assert.equal(r.status, 1); + assert.match(r.err, /bytes, over the 61440-byte journal tail/); +}); From 8609ab8e856a14a92503c6e77ae190399dbb9abc Mon Sep 17 00:00:00 2001 From: Relayflow Lead Date: Tue, 22 Sep 2026 23:21:22 -0700 Subject: [PATCH 4/6] =?UTF-8?q?fix(examples):=20address=20prompt-lab=20re-?= =?UTF-8?q?review=20=E2=80=94=20CAS=20publish,=20OS-released=20lock,=20no?= =?UTF-8?q?=20dead=20proposal?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - publish is a compare-and-swap on the live prompt the run read: a retried publish is a no-op once it landed and never rolls back a newer one. - The lab lock is a SQLite BEGIN EXCLUSIVE on /.lock.db: the kernel releases it when its process dies, SIGKILL included. Replaces the pid file, whose read-then-steal raced (the concurrency test caught it). - Config level no longer pays for a proposed rewrite of shared rows that nothing read; Job 2 iterates from the live prompt with the changeset. Evidence regenerated from a fresh prove.sh run on this code; the lease race (#560) cost one earlier attempt, kept in runtime-findings. Co-Authored-By: Claude Opus 5.5 (1M context) --- examples/prompt-lab/.gitignore | 1 + examples/prompt-lab/README.md | 41 ++--- .../prompt-lab/evidence/run/01-job1-run.txt | 118 ++++++------ .../evidence/run/02-reviewer-edit.diff | 4 +- .../prompt-lab/evidence/run/03-answer.txt | 6 +- .../prompt-lab/evidence/run/04-resume.txt | 58 +++--- .../prompt-lab/evidence/run/05-answer.txt | 6 +- .../prompt-lab/evidence/run/06-resume.txt | 48 +++-- .../evidence/run/07-patient-run.txt | 40 ++--- .../prompt-lab/evidence/run/08-answer.txt | 6 +- .../prompt-lab/evidence/run/09-resume.txt | 96 +++++++--- .../prompt-lab/evidence/run/10-job2-run.txt | 50 +++--- .../prompt-lab/evidence/run/11-answer.txt | 6 +- .../prompt-lab/evidence/run/12-resume.txt | 116 ++++++------ .../prompt-lab/evidence/run/13-answer.txt | 6 +- .../prompt-lab/evidence/run/14-resume.txt | 62 ++++--- .../prompt-lab/evidence/run/15-lab-state.txt | 12 +- .../prompt-lab/evidence/run/lab/bank.json | 12 +- .../prompt-lab/evidence/run/lab/gold.json | 9 +- .../evidence/run/lab/queue/issues.json | 5 +- .../run/lab/queue/patient-briefs.json | 2 +- .../evidence/run/lab/shelf/roderick.json | 10 ++ .../evidence/run/lab/shelf/verna.json | 10 -- .../prompt-lab/evidence/run/lab/targets.json | 16 +- .../07f91c04/plan.json | 4 + .../fbf53055/plan.json | 4 - .../work/q-wound-status/205c23a4/grid.json | 74 ++++++++ .../work/q-wound-status/205c23a4/score.json | 70 ++++++++ .../work/q-wound-status/cd33704e/grid.json | 55 ------ .../work/q-wound-status/cd33704e/score.json | 59 ------ .../{8b7996f3 => 35966ab5}/commit.json | 0 .../lab/work/sunrise-soc/35966ab5/grid.json | 170 ++++++++++++++++++ .../lab/work/sunrise-soc/8b7996f3/grid.json | 170 ------------------ .../00-prove-attempt3-lease-conflict.txt | 25 +++ examples/prompt-lab/jobs/fix.ts | 2 +- examples/prompt-lab/jobs/new-agency.ts | 39 ++-- examples/prompt-lab/lib/lab.ts | 5 +- examples/prompt-lab/lib/types.ts | 1 - examples/prompt-lab/store.ts | 37 ++-- examples/prompt-lab/tests/store.test.ts | 54 +++++- 40 files changed, 828 insertions(+), 681 deletions(-) create mode 100644 examples/prompt-lab/.gitignore create mode 100644 examples/prompt-lab/evidence/run/lab/shelf/roderick.json delete mode 100644 examples/prompt-lab/evidence/run/lab/shelf/verna.json create mode 100644 examples/prompt-lab/evidence/run/lab/work/patients/gap-ostomy-supplies-soc/07f91c04/plan.json delete mode 100644 examples/prompt-lab/evidence/run/lab/work/patients/gap-ostomy-supplies-soc/fbf53055/plan.json create mode 100644 examples/prompt-lab/evidence/run/lab/work/q-wound-status/205c23a4/grid.json create mode 100644 examples/prompt-lab/evidence/run/lab/work/q-wound-status/205c23a4/score.json delete mode 100644 examples/prompt-lab/evidence/run/lab/work/q-wound-status/cd33704e/grid.json delete mode 100644 examples/prompt-lab/evidence/run/lab/work/q-wound-status/cd33704e/score.json rename examples/prompt-lab/evidence/run/lab/work/sunrise-soc/{8b7996f3 => 35966ab5}/commit.json (100%) create mode 100644 examples/prompt-lab/evidence/run/lab/work/sunrise-soc/35966ab5/grid.json delete mode 100644 examples/prompt-lab/evidence/run/lab/work/sunrise-soc/8b7996f3/grid.json create mode 100644 examples/prompt-lab/evidence/runtime-findings/00-prove-attempt3-lease-conflict.txt diff --git a/examples/prompt-lab/.gitignore b/examples/prompt-lab/.gitignore new file mode 100644 index 00000000..30911be4 --- /dev/null +++ b/examples/prompt-lab/.gitignore @@ -0,0 +1 @@ +evidence/run/lab/.lock.db diff --git a/examples/prompt-lab/README.md b/examples/prompt-lab/README.md index 0d5449e9..e1d42f20 100644 --- a/examples/prompt-lab/README.md +++ b/examples/prompt-lab/README.md @@ -27,15 +27,17 @@ Each job file is its brief diagram, line by line: | First pass persists as targets / gold | `store.ts record`, fed only from the grid you reviewed. AI output never writes gold directly | | Iterator: rewrite-only | [`prompts.ts`](prompts.ts) `iterate`. Input is the changeset, patients, brief and existing prompt; output is new prompt text | | Prompt QA: the brief plus shared guidelines | `promptQa`, looping with the iterator. Capped at 3 tries, then the run parks as `needs_human` | -| Done = live | `store.ts publish` sets `livePromptId`. There is no promote step. A shared question warns which agencies it will change, and the re-run scores the new prompt on **every distinct menu** it is asked with ([`lib/piles.ts`](lib/piles.ts) `distinctMenus`), because done changes it for all of them | -| Shared frozen at config level | a changed shared or mismatch row becomes a `config-send` issue that carries its targets and a proposed rewrite | +| Done = live | `store.ts publish` sets `livePromptId`, as a compare-and-swap on the live prompt the run read: a retried publish is a no-op, and it never rolls back one published since. There is no promote step. A shared question warns which agencies it will change, and the re-run scores the new prompt on **every distinct menu** it is asked with ([`lib/piles.ts`](lib/piles.ts) `distinctMenus`), because done changes it for all of them | +| Shared frozen at config level | a changed shared or mismatch row becomes a `config-send` issue that carries its targets. It gets no rewrite at config level: Job 2 iterates from the live prompt with the full changeset | | Test planner: never invents a patient | picks shelf patients of the run's visit type (checked deterministically); each hole becomes one gap brief per question × visit type | | You do not approve the chart | the only patient gate is *kick generate*. Patient QA, plus a deterministic identifier check ([`lib/phi.ts`](lib/phi.ts)), locks it | Every read and write of the lab is a journaled `f.run` step. Every store verb is idempotent, so a retried step lands the lab in the same state. `write-new` never overwrites an edit you made. Mutating verbs take an exclusive lab lock, -so two runs at once never lose each other's update. +so two runs at once never lose each other's update. The lock is a SQLite +`BEGIN EXCLUSIVE` on `/.lock.db`. That's a kernel file lock, so the OS +releases it when its process dies, even from a SIGKILL. A first-pass prompt is offered for commit only after it has run on a shelf patient and you have reviewed its rows. A prompt whose question is a gap waits @@ -56,7 +58,7 @@ The flow is the job graph that sits under those screens. ```sh npm install -npm test # 20 unit tests over the deterministic parts and the store +npm test # 23 unit tests over the deterministic parts and the store node --experimental-strip-types store.ts ./my-lab seed fixtures npx flows run prompt-lab.flow.ts --local-agent --input \ '{"job":"new-agency","reviewer":"","lab":"./my-lab","agency":"sunrise","visitType":"soc"}' @@ -97,27 +99,22 @@ every step below exited as shown: | Run | Result | What happened | | --- | --- | --- | -| Job 1 · `sunrise` / `soc` ([01](evidence/run/01-job1-run.txt), [04](evidence/run/04-resume.txt), [06](evidence/run/06-resume.txt)) | 29 steps, `success` | Piles came out as shared / mismatch / agency-specific. Both agency-specific questions got first-pass prompts that passed Prompt QA. The planner covered 3 questions from the `soc` shelf and queued `gap-ostomy-supplies-soc`. Gate 1: 9 rows, 7 highlighted. The reviewer raised Pat's wound confidence to High, rewrote the explanation and added a note ([02](evidence/run/02-reviewer-edit.diff)). That shared row went to the question manager with its target. Gate 2 offered only `living-situation`, and it went live. `ostomy-supplies` had no shelf patient, so it was held as a draft. | -| Patient · `gap-ostomy-supplies-soc` ([07](evidence/run/07-patient-run.txt), [09](evidence/run/09-resume.txt)) | 9 steps, `success` | Plan, then kick generate. The chart passed Patient QA on its first try, and `verna` locked onto the shelf. The brief is marked `locked`. | -| Job 2 · the config-send issue ([10](evidence/run/10-job2-run.txt), [12](evidence/run/12-resume.txt), [14](evidence/run/14-resume.txt)) | 22 steps, `success` | Three shelf patients. Gold came prefilled, with Pat's config target carried over. The iterator's rewrite passed Prompt QA. The re-run scored 3 of 3 golded patients worked (100%), and the gate warned it would change harbor, maple and sunrise. Done made the new prompt live and closed the issue. | +| Job 1 · `sunrise` / `soc` ([01](evidence/run/01-job1-run.txt), [04](evidence/run/04-resume.txt), [06](evidence/run/06-resume.txt)) | 28 steps, `success` | Piles came out as shared / mismatch / agency-specific. Both agency-specific questions got first-pass prompts that passed Prompt QA. The planner covered 3 questions from the `soc` shelf and queued `gap-ostomy-supplies-soc`. Gate 1: 9 rows, 7 highlighted. The reviewer raised Pat's wound confidence to High, rewrote the explanation and added a note ([02](evidence/run/02-reviewer-edit.diff)). That shared row went to the question manager with its target, and got no rewrite at config level. Gate 2 offered only `living-situation`, and it went live. `ostomy-supplies` had no shelf patient, so it was held as a draft. | +| Patient · `gap-ostomy-supplies-soc` ([07](evidence/run/07-patient-run.txt), [09](evidence/run/09-resume.txt)) | 14 steps, `success` | Plan, then kick generate. Patient QA sent the chart back before one passed (`llm-6` … `llm-12`). `roderick` locked onto the shelf, and the brief is marked `locked`. | +| Job 2 · the config-send issue ([10](evidence/run/10-job2-run.txt), [12](evidence/run/12-resume.txt), [14](evidence/run/14-resume.txt)) | 24 steps, `success` | Four shelf patients, including `roderick`. **The live prompt answered Pat "Healed / High", which is the brief's failure.** Gold came prefilled, with Pat's config target "Ongoing" carried over. The iterator's rewrite passed Prompt QA. The re-run scored **3 of 4 golded patients worked (75%)**: Pat is now "Ongoing" and worked; `roderick` did not (gold "Ongoing", new run "No wound"). The gate warned it would change harbor, maple and sunrise. Done made the new prompt live and closed the issue. | Final state: [15-lab-state.txt](evidence/run/15-lab-state.txt), with the whole lab in `evidence/run/lab/`. -**What this run does not show:** +**Read the 75% with care.** `prove.sh` answers `yes` at every gate, so it +marked done at 75%. A reviewer would look at `roderick` first. His gold was the +old prompt's answer, accepted without review, and an ostomy patient's +peristomal skin damage may or may not be a "primary wound". That's exactly the +clinical call this gate exists for. The reviewer here is a script, not a +clinician. -- **A wrong answer being fixed.** The model answered Pat "Ongoing / Medium" - even under the flawed wound prompt. So Job 2 iterated on confidence, the - explanation and source priority, not on the answer. -- **Re-running on several menus.** `wound-status` has one menu across its - agencies, so the per-menu re-run ran with one. The multi-menu case, a - mismatch question like `mood`, is covered by the `distinctMenus` unit test - only. -- **A clinician's judgment.** The reviewer at every gate is `prove.sh`, - applying fixed edits. - -The brief's exact failure, "Healed / High" for Pat, did occur in an earlier -`claude-sonnet-5` run of the same prompt. Its engine step's journal is in -[00-sonnet-run-pat-healed-high.txt](evidence/runtime-findings/00-sonnet-run-pat-healed-high.txt). +`wound-status` has one menu across its agencies, so the per-menu re-run ran +with one menu. The multi-menu case, a mismatch question like `mood`, is covered +by the `distinctMenus` unit test only. ## Runtime findings (relayflows 2.0.29) @@ -154,7 +151,7 @@ where it lives, and each has captured evidence in `lease_conflict: attempt has no active worker lease`. The CLI made that a fatal `protocol_error` ([00-prove-attempt1-lease-conflict-after-success.txt](evidence/runtime-findings/00-prove-attempt1-lease-conflict-after-success.txt)). - It's intermittent: it happened once in about 50 sequential calls. There's + It's intermittent: it happened twice in about 100 sequential calls ([second](evidence/runtime-findings/00-prove-attempt3-lease-conflict.txt)). There's no workaround in the flow, so rerun. Tracked in [#560](https://github.com/AgentWorkforce/flows/issues/560). Findings 1 and 4 are the same class of problem: late lease traffic becomes diff --git a/examples/prompt-lab/evidence/run/01-job1-run.txt b/examples/prompt-lab/evidence/run/01-job1-run.txt index 75075515..380caf03 100644 --- a/examples/prompt-lab/evidence/run/01-job1-run.txt +++ b/examples/prompt-lab/evidence/run/01-job1-run.txt @@ -1,82 +1,80 @@ -$ npx flows run --no-observer-link --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data --local-agent prompt-lab.flow.ts --input {"job":"new-agency","reviewer":"prompt-lab-reviewer","lab":"evidence/run/lab","agency":"sunrise","visitType":"soc"} +$ npx flows run --no-observer-link --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove4-data --local-agent prompt-lab.flow.ts --input {"job":"new-agency","reviewer":"prompt-lab-reviewer","lab":"evidence/run/lab","agency":"sunrise","visitType":"soc"} ○ run-1 (deterministic) 0.00s -✓ run-1 (deterministic) 0.15s completionReason: success +✓ run-1 (deterministic) 0.07s completionReason: success ○ llm-2 (llm) 0.00s -WAITING [worker_lease] Run "01M36BDMS16K70HPQCSJCS496R" step "llm-2" (llm) is running under a worker lease until 1790140958805. -↻ llm-2 (llm) 6.39s -WAITING [worker_lease] Run "01M36BDMS16K70HPQCSJCS496R" step "llm-2" (llm) is running under a worker lease until 1790140968809. -↻ llm-2 (llm) 16.43s -✓ llm-2 (llm) 23.12s completionReason: success +WAITING [worker_lease] Run "01M36DY3KREWBH8HTKEB6VK7KY" step "llm-2" (llm) is running under a worker lease until 1790143595437. +↻ llm-2 (llm) 4.06s +WAITING [worker_lease] Run "01M36DY3KREWBH8HTKEB6VK7KY" step "llm-2" (llm) is running under a worker lease until 1790143605451. +↻ llm-2 (llm) 14.10s +✓ llm-2 (llm) 21.23s completionReason: success ○ llm-3 (llm) 0.00s -WAITING [worker_lease] Run "01M36BE8Z1N48PME3B42TFGJGY" step "llm-3" (llm) is running under a worker lease until 1790140979477. -↻ llm-3 (llm) 3.94s -✓ llm-3 (llm) 10.04s completionReason: success +WAITING [worker_lease] Run "01M36DYQW8KYMWYEHR5M0ZBKBN" step "llm-3" (llm) is running under a worker lease until 1790143616189. +↻ llm-3 (llm) 3.58s +✓ llm-3 (llm) 10.98s completionReason: success ○ run-4 (deterministic) 0.00s -✓ run-4 (deterministic) 0.06s completionReason: success +✓ run-4 (deterministic) 0.08s completionReason: success ○ llm-5 (llm) 0.00s -WAITING [worker_lease] Run "01M36BEM20H1BJV077SSX3BCX7" step "llm-5" (llm) is running under a worker lease until 1790140990835. -↻ llm-5 (llm) 5.19s -WAITING [worker_lease] Run "01M36BEM20H1BJV077SSX3BCX7" step "llm-5" (llm) is running under a worker lease until 1790141000839. -↻ llm-5 (llm) 15.21s -✓ llm-5 (llm) 24.55s completionReason: success +WAITING [worker_lease] Run "01M36DZ2MX8GEADP9FEW024PQ3" step "llm-5" (llm) is running under a worker lease until 1790143627218. +↻ llm-5 (llm) 3.55s +WAITING [worker_lease] Run "01M36DZ2MX8GEADP9FEW024PQ3" step "llm-5" (llm) is running under a worker lease until 1790143637222. +↻ llm-5 (llm) 13.59s +✓ llm-5 (llm) 21.92s completionReason: success ○ llm-6 (llm) 0.00s -WAITING [worker_lease] Run "01M36BFB66Q05G9R02NZ6HQ4E0" step "llm-6" (llm) is running under a worker lease until 1790141014522. -↻ llm-6 (llm) 4.33s -WAITING [worker_lease] Run "01M36BFB66Q05G9R02NZ6HQ4E0" step "llm-6" (llm) is running under a worker lease until 1790141024523. -↻ llm-6 (llm) 14.37s -✓ llm-6 (llm) 15.11s completionReason: success +WAITING [worker_lease] Run "01M36DZQZJEEY12HHWFC2M79B4" step "llm-6" (llm) is running under a worker lease until 1790143649061. +↻ llm-6 (llm) 3.47s +✓ llm-6 (llm) 10.50s completionReason: success ○ run-7 (deterministic) 0.00s -✓ run-7 (deterministic) 0.07s completionReason: success +✓ run-7 (deterministic) 0.10s completionReason: success ○ llm-8 (llm) 0.00s -WAITING [worker_lease] Run "01M36BFSC808R6VNSW2ESXMW1M" step "llm-8" (llm) is running under a worker lease until 1790141029051. -↻ llm-8 (llm) 3.67s -WAITING [worker_lease] Run "01M36BFSC808R6VNSW2ESXMW1M" step "llm-8" (llm) is running under a worker lease until 1790141039054. -↻ llm-8 (llm) 13.72s -✓ llm-8 (llm) 18.18s completionReason: success +WAITING [worker_lease] Run "01M36E04MZ7A2QC64XYRK1M0Q3" step "llm-8" (llm) is running under a worker lease until 1790143662042. +↻ llm-8 (llm) 5.85s +WAITING [worker_lease] Run "01M36E04MZ7A2QC64XYRK1M0Q3" step "llm-8" (llm) is running under a worker lease until 1790143672056. +↻ llm-8 (llm) 15.91s +✓ llm-8 (llm) 20.34s completionReason: success ○ run-9 (deterministic) 0.00s -✓ run-9 (deterministic) 0.06s completionReason: success +✓ run-9 (deterministic) 0.10s completionReason: success ○ llm-10 (llm) 0.00s -WAITING [worker_lease] Run "01M36BGCRVM8VKEZ8AX0RRRP2F" step "llm-10" (llm) is running under a worker lease until 1790141048910. -↻ llm-10 (llm) 5.29s -✓ llm-10 (llm) 9.54s completionReason: success +WAITING [worker_lease] Run "01M36E0PJPCQ5QDWNA6XJBR104" step "llm-10" (llm) is running under a worker lease until 1790143680394. +↻ llm-10 (llm) 3.76s +✓ llm-10 (llm) 7.97s completionReason: success ○ llm-11 (llm) 0.00s -WAITING [worker_lease] Run "01M36BGMSX3J78Y4T7MKK4FEG3" step "llm-11" (llm) is running under a worker lease until 1790141057136. -↻ llm-11 (llm) 3.97s -✓ llm-11 (llm) 7.75s completionReason: success +WAITING [worker_lease] Run "01M36E0Y6HJFHGJ8ZJASK5SC73" step "llm-11" (llm) is running under a worker lease until 1790143688196. +↻ llm-11 (llm) 3.60s +✓ llm-11 (llm) 7.40s completionReason: success ○ llm-12 (llm) 0.00s -WAITING [worker_lease] Run "01M36BGVYACMHT4ZM8B0SZ8KM0" step "llm-12" (llm) is running under a worker lease until 1790141064446. -↻ llm-12 (llm) 3.54s -✓ llm-12 (llm) 9.06s completionReason: success +WAITING [worker_lease] Run "01M36E15M53REC3MBPCZFRN8AM" step "llm-12" (llm) is running under a worker lease until 1790143695802. +↻ llm-12 (llm) 3.80s +✓ llm-12 (llm) 7.81s completionReason: success ○ llm-13 (llm) 0.00s -WAITING [worker_lease] Run "01M36BH4ZFMQ4TJXH01EW64S4S" step "llm-13" (llm) is running under a worker lease until 1790141073700. -↻ llm-13 (llm) 3.73s -✓ llm-13 (llm) 7.62s completionReason: success +WAITING [worker_lease] Run "01M36E1CZPG4F2E1ZBC0R8FQC6" step "llm-13" (llm) is running under a worker lease until 1790143703337. +↻ llm-13 (llm) 3.53s +✓ llm-13 (llm) 7.22s completionReason: success ○ llm-14 (llm) 0.00s -WAITING [worker_lease] Run "01M36BHD49DRZSES3XZNV3FDBP" step "llm-14" (llm) is running under a worker lease until 1790141082045. -↻ llm-14 (llm) 4.45s -✓ llm-14 (llm) 8.19s completionReason: success +WAITING [worker_lease] Run "01M36E1MBC1EH7CYPTB4J2E0JD" step "llm-14" (llm) is running under a worker lease until 1790143710880. +↻ llm-14 (llm) 3.85s +✓ llm-14 (llm) 7.86s completionReason: success ○ llm-15 (llm) 0.00s -WAITING [worker_lease] Run "01M36BHMQ7EVSCV5WR1M8ZYSPP" step "llm-15" (llm) is running under a worker lease until 1790141089818. -↻ llm-15 (llm) 4.03s -✓ llm-15 (llm) 8.36s completionReason: success +WAITING [worker_lease] Run "01M36E1VWY2WGSBSAS2PJWT2Y6" step "llm-15" (llm) is running under a worker lease until 1790143718611. +↻ llm-15 (llm) 3.73s +✓ llm-15 (llm) 8.01s completionReason: success ○ llm-16 (llm) 0.00s -WAITING [worker_lease] Run "01M36BHYFTVHK7NRF3RFMNDFZ9" step "llm-16" (llm) is running under a worker lease until 1790141099824. -↻ llm-16 (llm) 5.68s -✓ llm-16 (llm) 9.96s completionReason: success +WAITING [worker_lease] Run "01M36E23SGV07GQER5VHWWW69S" step "llm-16" (llm) is running under a worker lease until 1790143726690. +↻ llm-16 (llm) 3.80s +✓ llm-16 (llm) 7.83s completionReason: success ○ llm-17 (llm) 0.00s -WAITING [worker_lease] Run "01M36BJ6W2TMXB2VNPEKNK4YJR" step "llm-17" (llm) is running under a worker lease until 1790141108406. -↻ llm-17 (llm) 4.30s -✓ llm-17 (llm) 8.89s completionReason: success +WAITING [worker_lease] Run "01M36E2B6R5V4HF07ZHZEWHFQE" step "llm-17" (llm) is running under a worker lease until 1790143734285. +↻ llm-17 (llm) 3.56s +✓ llm-17 (llm) 10.32s completionReason: success ○ llm-18 (llm) 0.00s -WAITING [worker_lease] Run "01M36BJF3GQ3SHGET32HR0QDYK" step "llm-18" (llm) is running under a worker lease until 1790141116838. -↻ llm-18 (llm) 3.84s -✓ llm-18 (llm) 9.42s completionReason: success +WAITING [worker_lease] Run "01M36E2NQFFXQKABB7S1RPM8X1" step "llm-18" (llm) is running under a worker lease until 1790143745059. +↻ llm-18 (llm) 4.01s +✓ llm-18 (llm) 8.22s completionReason: success ○ run-19 (deterministic) 0.00s -✓ run-19 (deterministic) 0.06s completionReason: success +✓ run-19 (deterministic) 0.11s completionReason: success ○ human-20 (deterministic) 0.00s ⏸ human-20 (human) 0.00s -PARKED [run_parked] Run "01M36BDEBD2DKTZTX8GTB28DW7" is waiting for prompt-lab-reviewer to answer human-20: "Config workbench · sunrise soc: 9 rows on 3 questions (7 highlighted, listed first); 1 gap brief(s) queued for the test patient manager.\nEdit target answer / confidence / explanation / notes in evidence/run/lab/work/sunrise-soc/8b7996f3/grid.json. Your first pass persists as the target for each question × patient.\nyes = persist targets and run iteration on every changed row; no = stop without persisting." -Answer with: flows answer --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data 01M36BDEBD2DKTZTX8GTB28DW7 human-20 yes|no -Then continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data --local-agent 01M36BDEBD2DKTZTX8GTB28DW7 -RUN 01M36BDEBD2DKTZTX8GTB28DW7 parked +PARKED [run_parked] Run "01M36DXZJ9142JCT8RPDZA59W7" is waiting for prompt-lab-reviewer to answer human-20: "Config workbench · sunrise soc: 9 rows on 3 questions (7 highlighted, listed first); 1 gap brief(s) queued for the test patient manager.\nEdit target answer / confidence / explanation / notes in evidence/run/lab/work/sunrise-soc/35966ab5/grid.json. Your first pass persists as the target for each question × patient.\nyes = persist targets and run iteration on every changed row; no = stop without persisting." +Answer with: flows answer --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove4-data 01M36DXZJ9142JCT8RPDZA59W7 human-20 yes|no +Then continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove4-data --local-agent 01M36DXZJ9142JCT8RPDZA59W7 +RUN 01M36DXZJ9142JCT8RPDZA59W7 parked exit=3 diff --git a/examples/prompt-lab/evidence/run/02-reviewer-edit.diff b/examples/prompt-lab/evidence/run/02-reviewer-edit.diff index df6f37db..d16edd8a 100644 --- a/examples/prompt-lab/evidence/run/02-reviewer-edit.diff +++ b/examples/prompt-lab/evidence/run/02-reviewer-edit.diff @@ -1,10 +1,10 @@ -# reviewer edit: evidence/run/lab/work/sunrise-soc/8b7996f3/grid.json +# reviewer edit: evidence/run/lab/work/sunrise-soc/35966ab5/grid.json @@ -128,10 +128,10 @@ }, "target": { "answer": "Ongoing", - "confidence": "Medium", -- "explanation": "The referral discharge summary states the left heel wound was closed, but today's SOC visit notes document an open left heel area 2.0 x 1.5 cm with moderate serous drainage and yellow slough, so the wound is not healed." +- "explanation": "The referral packet's discharge summary states the left heel pressure injury was closed, but today's SOC visit notes document an open left heel area 2.0 x 1.5 cm with moderate serous drainage and yellow slough, so the wound is currently ongoing." + "confidence": "High", + "explanation": "Today's visit notes describe an open 2.0 x 1.5 cm left heel wound with serous drainage and slough; the discharge summary's closed status is outdated." }, diff --git a/examples/prompt-lab/evidence/run/03-answer.txt b/examples/prompt-lab/evidence/run/03-answer.txt index 147a954e..9c2d290a 100644 --- a/examples/prompt-lab/evidence/run/03-answer.txt +++ b/examples/prompt-lab/evidence/run/03-answer.txt @@ -1,4 +1,4 @@ -$ npx flows answer --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data 01M36BDEBD2DKTZTX8GTB28DW7 human-20 yes --by prompt-lab-reviewer -ANSWERED 01M36BDEBD2DKTZTX8GTB28DW7 human-20 yes -Continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data --local-agent 01M36BDEBD2DKTZTX8GTB28DW7 +$ npx flows answer --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove4-data 01M36DXZJ9142JCT8RPDZA59W7 human-20 yes --by prompt-lab-reviewer +ANSWERED 01M36DXZJ9142JCT8RPDZA59W7 human-20 yes +Continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove4-data --local-agent 01M36DXZJ9142JCT8RPDZA59W7 exit=0 diff --git a/examples/prompt-lab/evidence/run/04-resume.txt b/examples/prompt-lab/evidence/run/04-resume.txt index 726d81da..2ee99b1f 100644 --- a/examples/prompt-lab/evidence/run/04-resume.txt +++ b/examples/prompt-lab/evidence/run/04-resume.txt @@ -1,62 +1,56 @@ -$ npx flows resume --no-observer-link --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data --local-agent 01M36BDEBD2DKTZTX8GTB28DW7 +$ npx flows resume --no-observer-link --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove4-data --local-agent 01M36DXZJ9142JCT8RPDZA59W7 ○ run-1 (deterministic) 0.00s ✓ run-1 (deterministic) 0.01s completionReason: success ○ llm-2 (llm) 0.00s -✓ llm-2 (llm) 3.82s completionReason: success +✓ llm-2 (llm) 3.55s completionReason: success ○ llm-3 (llm) 0.00s -✓ llm-3 (llm) 3.67s completionReason: success +✓ llm-3 (llm) 3.95s completionReason: success ○ run-4 (deterministic) 0.00s ✓ run-4 (deterministic) 0.01s completionReason: success ○ llm-5 (llm) 0.00s -✓ llm-5 (llm) 3.59s completionReason: success +✓ llm-5 (llm) 3.44s completionReason: success ○ llm-6 (llm) 0.00s -✓ llm-6 (llm) 3.91s completionReason: success +✓ llm-6 (llm) 3.58s completionReason: success ○ run-7 (deterministic) 0.00s ✓ run-7 (deterministic) 0.01s completionReason: success ○ llm-8 (llm) 0.00s -✓ llm-8 (llm) 3.57s completionReason: success +✓ llm-8 (llm) 3.62s completionReason: success ○ run-9 (deterministic) 0.00s ✓ run-9 (deterministic) 0.01s completionReason: success ○ llm-10 (llm) 0.00s -✓ llm-10 (llm) 3.67s completionReason: success +✓ llm-10 (llm) 3.51s completionReason: success ○ llm-11 (llm) 0.00s -✓ llm-11 (llm) 3.75s completionReason: success +✓ llm-11 (llm) 3.53s completionReason: success ○ llm-12 (llm) 0.00s -✓ llm-12 (llm) 4.01s completionReason: success +✓ llm-12 (llm) 3.49s completionReason: success ○ llm-13 (llm) 0.00s -✓ llm-13 (llm) 3.48s completionReason: success +✓ llm-13 (llm) 3.44s completionReason: success ○ llm-14 (llm) 0.00s -✓ llm-14 (llm) 4.34s completionReason: success +✓ llm-14 (llm) 3.65s completionReason: success ○ llm-15 (llm) 0.00s -✓ llm-15 (llm) 4.94s completionReason: success +✓ llm-15 (llm) 3.75s completionReason: success ○ llm-16 (llm) 0.00s -✓ llm-16 (llm) 3.99s completionReason: success +✓ llm-16 (llm) 3.82s completionReason: success ○ llm-17 (llm) 0.00s -✓ llm-17 (llm) 3.57s completionReason: success +✓ llm-17 (llm) 3.56s completionReason: success ○ llm-18 (llm) 0.00s -✓ llm-18 (llm) 3.57s completionReason: success +✓ llm-18 (llm) 3.69s completionReason: success ○ run-19 (deterministic) 0.00s ✓ run-19 (deterministic) 0.01s completionReason: success ○ human-20 (deterministic) 0.00s ✓ human-20 (deterministic) 0.02s completionReason: success ○ run-21 (deterministic) 0.00s -✓ run-21 (deterministic) 0.07s completionReason: success +✓ run-21 (deterministic) 0.06s completionReason: success ○ run-22 (deterministic) 0.00s -✓ run-22 (deterministic) 0.08s completionReason: success -○ llm-23 (llm) 0.00s -WAITING [worker_lease] Run "01M36BMESDKX5DBD3T1QN4JBM1" step "llm-23" (llm) is running under a worker lease until 1790141182053. -↻ llm-23 (llm) 3.91s -WAITING [worker_lease] Run "01M36BMESDKX5DBD3T1QN4JBM1" step "llm-23" (llm) is running under a worker lease until 1790141192057. -↻ llm-23 (llm) 13.91s -✓ llm-23 (llm) 21.61s completionReason: success +✓ run-22 (deterministic) 0.06s completionReason: success +○ run-23 (deterministic) 0.00s +✓ run-23 (deterministic) 0.06s completionReason: success ○ run-24 (deterministic) 0.00s -✓ run-24 (deterministic) 0.07s completionReason: success -○ run-25 (deterministic) 0.00s -✓ run-25 (deterministic) 0.07s completionReason: success -○ human-26 (deterministic) 0.00s -⏸ human-26 (human) 0.02s -PARKED [run_parked] Run "01M36BDEBD2DKTZTX8GTB28DW7" is waiting for prompt-lab-reviewer to answer human-26: "Commit output · sunrise: agency-specific prompts ready to go live in Apricot: living-situation.\nSent to the question manager (frozen here): wound-status → config-sunrise-wound-status-8a366995.\nHeld as drafts until a shelf patient covers them: ostomy-supplies.\nTo commit only some, edit evidence/run/lab/work/sunrise-soc/8b7996f3/commit.json: {\"mode\":\"only\"|\"except\",\"questions\":[...]}.\nyes = commit (Done = live, no promote step); no = leave them as drafts." -Answer with: flows answer --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data 01M36BDEBD2DKTZTX8GTB28DW7 human-26 yes|no -Then continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data --local-agent 01M36BDEBD2DKTZTX8GTB28DW7 -RUN 01M36BDEBD2DKTZTX8GTB28DW7 parked +✓ run-24 (deterministic) 0.06s completionReason: success +○ human-25 (deterministic) 0.00s +⏸ human-25 (human) 0.00s +PARKED [run_parked] Run "01M36DXZJ9142JCT8RPDZA59W7" is waiting for prompt-lab-reviewer to answer human-25: "Commit output · sunrise: agency-specific prompts ready to go live in Apricot: living-situation.\nSent to the question manager (frozen here): wound-status → config-sunrise-wound-status-9c1634e2.\nHeld as drafts until a shelf patient covers them: ostomy-supplies.\nTo commit only some, edit evidence/run/lab/work/sunrise-soc/35966ab5/commit.json: {\"mode\":\"only\"|\"except\",\"questions\":[...]}.\nyes = commit (Done = live, no promote step); no = leave them as drafts." +Answer with: flows answer --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove4-data 01M36DXZJ9142JCT8RPDZA59W7 human-25 yes|no +Then continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove4-data --local-agent 01M36DXZJ9142JCT8RPDZA59W7 +RUN 01M36DXZJ9142JCT8RPDZA59W7 parked exit=3 diff --git a/examples/prompt-lab/evidence/run/05-answer.txt b/examples/prompt-lab/evidence/run/05-answer.txt index b356c5c2..18b0e2e0 100644 --- a/examples/prompt-lab/evidence/run/05-answer.txt +++ b/examples/prompt-lab/evidence/run/05-answer.txt @@ -1,4 +1,4 @@ -$ npx flows answer --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data 01M36BDEBD2DKTZTX8GTB28DW7 human-26 yes --by prompt-lab-reviewer -ANSWERED 01M36BDEBD2DKTZTX8GTB28DW7 human-26 yes -Continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data --local-agent 01M36BDEBD2DKTZTX8GTB28DW7 +$ npx flows answer --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove4-data 01M36DXZJ9142JCT8RPDZA59W7 human-25 yes --by prompt-lab-reviewer +ANSWERED 01M36DXZJ9142JCT8RPDZA59W7 human-25 yes +Continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove4-data --local-agent 01M36DXZJ9142JCT8RPDZA59W7 exit=0 diff --git a/examples/prompt-lab/evidence/run/06-resume.txt b/examples/prompt-lab/evidence/run/06-resume.txt index 026b85a2..5b05b206 100644 --- a/examples/prompt-lab/evidence/run/06-resume.txt +++ b/examples/prompt-lab/evidence/run/06-resume.txt @@ -1,40 +1,40 @@ -$ npx flows resume --no-observer-link --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data --local-agent 01M36BDEBD2DKTZTX8GTB28DW7 +$ npx flows resume --no-observer-link --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove4-data --local-agent 01M36DXZJ9142JCT8RPDZA59W7 ○ run-1 (deterministic) 0.00s ✓ run-1 (deterministic) 0.01s completionReason: success ○ llm-2 (llm) 0.00s -✓ llm-2 (llm) 5.30s completionReason: success +✓ llm-2 (llm) 3.92s completionReason: success ○ llm-3 (llm) 0.00s -✓ llm-3 (llm) 4.16s completionReason: success +✓ llm-3 (llm) 3.59s completionReason: success ○ run-4 (deterministic) 0.00s -✓ run-4 (deterministic) 0.01s completionReason: success +✓ run-4 (deterministic) 0.03s completionReason: success ○ llm-5 (llm) 0.00s -✓ llm-5 (llm) 3.85s completionReason: success +✓ llm-5 (llm) 3.57s completionReason: success ○ llm-6 (llm) 0.00s -✓ llm-6 (llm) 3.74s completionReason: success +✓ llm-6 (llm) 3.84s completionReason: success ○ run-7 (deterministic) 0.00s ✓ run-7 (deterministic) 0.01s completionReason: success ○ llm-8 (llm) 0.00s -✓ llm-8 (llm) 3.36s completionReason: success +✓ llm-8 (llm) 3.86s completionReason: success ○ run-9 (deterministic) 0.00s ✓ run-9 (deterministic) 0.01s completionReason: success ○ llm-10 (llm) 0.00s -✓ llm-10 (llm) 3.65s completionReason: success +✓ llm-10 (llm) 3.48s completionReason: success ○ llm-11 (llm) 0.00s -✓ llm-11 (llm) 3.82s completionReason: success +✓ llm-11 (llm) 6.14s completionReason: success ○ llm-12 (llm) 0.00s -✓ llm-12 (llm) 3.74s completionReason: success +✓ llm-12 (llm) 3.59s completionReason: success ○ llm-13 (llm) 0.00s -✓ llm-13 (llm) 5.21s completionReason: success +✓ llm-13 (llm) 3.70s completionReason: success ○ llm-14 (llm) 0.00s -✓ llm-14 (llm) 4.68s completionReason: success +✓ llm-14 (llm) 3.46s completionReason: success ○ llm-15 (llm) 0.00s -✓ llm-15 (llm) 3.81s completionReason: success +✓ llm-15 (llm) 3.63s completionReason: success ○ llm-16 (llm) 0.00s -✓ llm-16 (llm) 3.57s completionReason: success +✓ llm-16 (llm) 3.62s completionReason: success ○ llm-17 (llm) 0.00s -✓ llm-17 (llm) 3.89s completionReason: success +✓ llm-17 (llm) 3.29s completionReason: success ○ llm-18 (llm) 0.00s -✓ llm-18 (llm) 3.58s completionReason: success +✓ llm-18 (llm) 3.63s completionReason: success ○ run-19 (deterministic) 0.00s ✓ run-19 (deterministic) 0.01s completionReason: success ○ human-20 (deterministic) 0.00s @@ -43,17 +43,15 @@ $ npx flows resume --no-observer-link --data-dir /private/tmp/claude-501/-Users- ✓ run-21 (deterministic) 0.01s completionReason: success ○ run-22 (deterministic) 0.00s ✓ run-22 (deterministic) 0.01s completionReason: success -○ llm-23 (llm) 0.00s -✓ llm-23 (llm) 3.76s completionReason: success +○ run-23 (deterministic) 0.00s +✓ run-23 (deterministic) 0.01s completionReason: success ○ run-24 (deterministic) 0.00s ✓ run-24 (deterministic) 0.01s completionReason: success -○ run-25 (deterministic) 0.00s -✓ run-25 (deterministic) 0.01s completionReason: success -○ human-26 (deterministic) 0.00s -✓ human-26 (deterministic) 0.01s completionReason: success +○ human-25 (deterministic) 0.00s +✓ human-25 (deterministic) 0.03s completionReason: success +○ run-26 (deterministic) 0.00s +✓ run-26 (deterministic) 0.08s completionReason: success ○ run-27 (deterministic) 0.00s ✓ run-27 (deterministic) 0.07s completionReason: success -○ run-28 (deterministic) 0.00s -✓ run-28 (deterministic) 0.07s completionReason: success -RUN 01M36BDEBD2DKTZTX8GTB28DW7 completed (29 steps) completionReason: success +RUN 01M36DXZJ9142JCT8RPDZA59W7 completed (28 steps) completionReason: success exit=0 diff --git a/examples/prompt-lab/evidence/run/07-patient-run.txt b/examples/prompt-lab/evidence/run/07-patient-run.txt index dd019654..6fed9ecd 100644 --- a/examples/prompt-lab/evidence/run/07-patient-run.txt +++ b/examples/prompt-lab/evidence/run/07-patient-run.txt @@ -1,32 +1,24 @@ -$ npx flows run --no-observer-link --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data --local-agent prompt-lab.flow.ts --input {"job":"patient","reviewer":"prompt-lab-reviewer","lab":"evidence/run/lab","briefId":"gap-ostomy-supplies-soc"} +$ npx flows run --no-observer-link --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove4-data --local-agent prompt-lab.flow.ts --input {"job":"patient","reviewer":"prompt-lab-reviewer","lab":"evidence/run/lab","briefId":"gap-ostomy-supplies-soc"} ○ run-1 (deterministic) 0.00s ✓ run-1 (deterministic) 0.07s completionReason: success ○ llm-2 (llm) 0.00s -WAITING [worker_lease] Run "01M36BQ2J79QTAJJSZF42ZRV0A" step "llm-2" (llm) is running under a worker lease until 1790141267834. -↻ llm-2 (llm) 3.72s -WAITING [worker_lease] Run "01M36BQ2J79QTAJJSZF42ZRV0A" step "llm-2" (llm) is running under a worker lease until 1790141277836. -↻ llm-2 (llm) 13.76s -WAITING [worker_lease] Run "01M36BQ2J79QTAJJSZF42ZRV0A" step "llm-2" (llm) is running under a worker lease until 1790141287838. -↻ llm-2 (llm) 23.73s -WAITING [worker_lease] Run "01M36BQ2J79QTAJJSZF42ZRV0A" step "llm-2" (llm) is running under a worker lease until 1790141297840. -↻ llm-2 (llm) 33.74s -WAITING [worker_lease] Run "01M36BQ2J79QTAJJSZF42ZRV0A" step "llm-2" (llm) is running under a worker lease until 1790141307842. -↻ llm-2 (llm) 43.72s -WAITING [worker_lease] Run "01M36BQ2J79QTAJJSZF42ZRV0A" step "llm-2" (llm) is running under a worker lease until 1790141317844. -↻ llm-2 (llm) 53.77s -WAITING [worker_lease] Run "01M36BQ2J79QTAJJSZF42ZRV0A" step "llm-2" (llm) is running under a worker lease until 1790141327845. -↻ llm-2 (llm) 63.77s -WAITING [worker_lease] Run "01M36BQ2J79QTAJJSZF42ZRV0A" step "llm-2" (llm) is running under a worker lease until 1790141337847. -↻ llm-2 (llm) 73.75s -WAITING [worker_lease] Run "01M36BQ2J79QTAJJSZF42ZRV0A" step "llm-2" (llm) is running under a worker lease until 1790141347850. -↻ llm-2 (llm) 83.75s -✓ llm-2 (llm) 84.90s completionReason: success +WAITING [worker_lease] Run "01M36E68CF02B2EMSYBZ68SZ01" step "llm-2" (llm) is running under a worker lease until 1790143862468. +↻ llm-2 (llm) 4.14s +WAITING [worker_lease] Run "01M36E68CF02B2EMSYBZ68SZ01" step "llm-2" (llm) is running under a worker lease until 1790143872473. +↻ llm-2 (llm) 14.16s +WAITING [worker_lease] Run "01M36E68CF02B2EMSYBZ68SZ01" step "llm-2" (llm) is running under a worker lease until 1790143882476. +↻ llm-2 (llm) 24.17s +WAITING [worker_lease] Run "01M36E68CF02B2EMSYBZ68SZ01" step "llm-2" (llm) is running under a worker lease until 1790143892479. +↻ llm-2 (llm) 34.19s +WAITING [worker_lease] Run "01M36E68CF02B2EMSYBZ68SZ01" step "llm-2" (llm) is running under a worker lease until 1790143902482. +↻ llm-2 (llm) 44.17s +✓ llm-2 (llm) 48.38s completionReason: success ○ run-3 (deterministic) 0.00s ✓ run-3 (deterministic) 0.06s completionReason: success ○ human-4 (deterministic) 0.00s ⏸ human-4 (human) 0.00s -PARKED [run_parked] Run "01M36BPYVQFDCG2WEP7V7H04X4" is waiting for prompt-lab-reviewer to answer human-4: "Test patient creator · gap-ostomy-supplies-soc for ostomy-supplies (soc visit).\nPlan:\nINVENTED SHELF PATIENT — VARIANT A (the load-bearing one): first-name label \"Verna\". Age band 60-69. No surname, no dates, no MRN, no phone, no address, no facility or clinician names anywhere — the referral packet is from \"the surgical service\" and \"the inpatient WOC nurse\", supplies come from \"the DME supplier\", and all timing is relative (\"about three weeks post-op\", \"since discharge\", \"this visit\").\n\nDIAGNOSES: sigmoid adenocarcinoma s/p open sigmoid colectomy with permanent end colostomy; type 2 diabetes (skin fragility, slow healing); hypertension; obesity with a pendulous lower abdomen and a skin crease below the stoma; osteoarthritis of the hands limiting fine dexterity; post-op anemia.\n\nLIVING SITUATION: lives alone in a one-bedroom second-floor walk-up, no elevator. One adult child works full time and visits on weekends. No home aide authorized. Changes the pouch standing at the bathroom sink, no mirror, cannot see the inferior edge of the stoma over her abdomen when seated.\n\nWHAT THE REFERRAL PACKET SAYS: discharge summary describing an uneventful post-op course; stoma at discharge \"beefy red, budded 2 cm, 32 mm round, peristomal skin intact\"; output \"pasty to formed\"; one supervised pouch change with return demonstration charted as \"partially independent\"; pouching schedule of every 3 days. Starter-kit checklist: 10 one-piece drainable pouches with a PRE-CUT 32 mm barrier, skin barrier wipes, adhesive remover wipes, deodorant drops, and — this is the trap — \"barrier rings x10\". Orders: skilled nursing 2x/week for ostomy care and teaching, WOC consult PRN, DME resupply order placed with no confirmed delivery.\n\nWHAT TODAY'S VISIT NOTES MUST SHOW: stoma now budded only ~0.5 cm and flush-to-retracted at the 4 o'clock position as post-op edema resolved; measures ~25 mm and slightly oval; sits at the lip of the abdominal crease when she sits. Peristomal skin denuded, excoriated and weeping serous moisture across roughly the inferior half of the wafer field, 2-4 cm, with an irregular scalloped border tracking exactly where effluent ran; she reports burning. Explicitly NO satellite lesions, no induration, no purulence, no fever — so this reads as irritant contact dermatitis from effluent, not infection or fungus. Pouch has been on ~5 hours and the adhesive is already lifting along the inferior border with effluent visible under the edge; she has been taping it down with paper tape. Two to three leaks a day; she is changing daily instead of every 3 days. Output looser and higher-volume than the packet describes. Inventory counted WITH the patient and documented item by item: 2 pouches left in the box on the counter, and their pre-cut 32 mm opening now leaves 3-4 mm of bare skin around a 25 mm stoma; barrier wipes nearly full; adhesive remover full; deodorant drops unopened; NO barrier rings anywhere — cabinet, bag and counter searched; no barrier powder, no convex product, no belt.\n\nINTENDED ANSWER: a moldable barrier ring/seal — it fills the crease and the oversized opening, stops effluent contact with denuded skin, and restores wear time. It is the only supply on the shelf that addresses the mechanism rather than a symptom.\n\nDELIBERATE DISTRACTORS (so the question is a real choice, not a lookup): (1) the nearly empty pouch box makes \"pouches\" look urgent, but more pouches at the wrong size leak the same way — resupply is a workflow action, not this visit's supply need; (2) weeping skin invites \"barrier powder/crusting\", but the notes make ongoing effluent contact explicit, so protection beats treatment; (3) a near-flush stoma invites \"convex wafer\" or \"belt\", but neither is in the home, no WOC convexity assessment has happened, and convexity on denuded skin without a ring first is second-line; (4) an answer path that defers to \"WOC referral\" instead of naming a supply should visibly under-answer, since the packet already carries a PRN consult.\n\nDELIBERATE CONFLICTS BETWEEN PACKET AND NOTES, each testing one thing: (C1 inventory) the packet checklist says 10 barrier rings went home; the in-home count finds zero and she says she never saw any — tests whether the answer trusts today's observation over discharge paperwork, and a packet-trusting path lands on \"pouches\" or \"None\". (C2 sizing) 32 mm round in the packet vs ~25 mm oval today, with pre-cut wafers still cut to the discharge size — tests whether the model connects resolving edema to the leak mechanism instead of blaming her technique. (C3 skin) \"peristomal skin intact\" and q3-day changes in the packet vs denuded skin and daily changes today — tests recency. (C4 output) formed/pasty in the packet vs looser and higher-volume today — a quiet detail that makes leakage more corrosive and further undercuts \"just reorder pouches\".\n\nVARIANT B — THE 'NONE' PATH: first-name label \"Roland\", age band 70-79. Mature end colostomy of several years after colectomy for diverticular disease; lives with a spouse who does the changes competently. The referral packet is for something else entirely — skilled nursing after a heart-failure hospitalization for medication management and teaching — and notes the colostomy as long-standing and self-managed, with a stale discharge-planner line claiming \"patient reports running out of pouches\" so the question must still be answered rather than skipped. Visit notes: stoma mature, budded 2 cm, beefy red, on a flat surface, 28 mm and unchanged for years; peristomal skin fully intact, no erythema or denudement; two-piece system worn 4 days with an intact seal, no leaks this month or historically; supply closet inventoried with the patient — 30+ pouches, 20+ wafers already cut to size, a full box of barrier rings, unopened barrier powder, adhesive remover, an unused belt; monthly DME auto-ship arrived complete. Conflict for B mirrors A's in the opposite direction: the stale \"running out of pouches\" note contradicts the counted inventory. That symmetry is the point — the same rule (observed notes over packet) yields \"barrier ring\" in A and \"None\" in B, so neither shortcut (\"always name a supply\", \"always trust the packet\") can pass both.\n\nSHELF MECHANICS: keep packet and visit notes as two separately quotable documents in the shelf's existing envelope shape — the conflicts only test anything if an answer path can cite one against the other. Measurements and counts are the load-bearing detail and must stay exact; everything identifying stays absent.\nYou may edit \"plan\" in evidence/run/lab/work/patients/gap-ostomy-supplies-soc/fbf53055/plan.json first.\nyes = kick generate (Patient QA then locks it onto the shelf; you do not approve the chart); no = leave it on the queue." -Answer with: flows answer --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data 01M36BPYVQFDCG2WEP7V7H04X4 human-4 yes|no -Then continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data --local-agent 01M36BPYVQFDCG2WEP7V7H04X4 -RUN 01M36BPYVQFDCG2WEP7V7H04X4 parked +PARKED [run_parked] Run "01M36E648MCDW2TJAC5J06X2VZ" is waiting for prompt-lab-reviewer to answer human-4: "Test patient creator · gap-ostomy-supplies-soc for ostomy-supplies (soc visit).\nPlan:\nINVENTED SHELF PATIENT — first-name label only: \"Roderick\". No dates, no MRN, no phone, no address, no surname, no facility names beyond generic role words (\"the discharging hospital\", \"the surgical practice\").\n\nAGE BAND & DEMOGRAPHICS\nOlder adult, 70–79 age band. Male. Lives alone in a single-story rental; one adult child lives roughly 40 minutes away and visits on weekends only. No paid caregiver at the time of start of care.\n\nDIAGNOSES\nPrimary: sigmoid colon adenocarcinoma, status post open low anterior resection with diverting end colostomy (left lower quadrant), ostomy created during the index admission. Secondary: type 2 diabetes mellitus (non-insulin, metformin), hypertension, mild chronic kidney disease stage 3a, osteoarthritis of both hands with reduced fine-motor dexterity, and post-operative deconditioning. Vision: presbyopia corrected with over-the-counter readers; reports difficulty seeing the stoma clearly without them.\n\nLIVING SITUATION / SELF-CARE ARRANGEMENT\nLives alone, independent with ambulation using a rolling walker indoors. Bathroom has a mirror over the sink but no seated surface at counter height; Roderick performs pouch changes standing at the sink, which he reports is tiring. The adult child was taught the pouch change once in the hospital but is present only on weekends. No home-health aide yet authorized. This matters to the question because it establishes that Roderick is doing his own changes six days out of seven with arthritic hands, so a supply that compensates for poor dexterity or poor seal is not a luxury item.\n\nWHAT THE REFERRAL PACKET SAYS (hospital discharge / ostomy nurse summary)\n- Stoma: end colostomy, left lower quadrant, matured, described at discharge as beefy red, budded approximately 1/2 inch above skin level, stoma diameter measured at 1 1/4 inches, with a mild parastomal skin fold on the inferior aspect noted by the inpatient WOC nurse.\n- Pouching system supplied at discharge: two-piece system, 2 1/4 inch flat cut-to-fit skin barrier wafer, drainable opaque pouch with integrated closure, coupling ring size 2 1/4 inch.\n- Quantity supplied at discharge: 10 wafers, 10 drainable pouches, 1 tube stoma paste, 1 box (30) skin barrier wipes, 1 bottle adhesive remover spray, 1 box of pouch deodorant drops.\n- Discharge instructions state: change wafer every 3–4 days or sooner if leakage; empty pouch when one-third full; a 30-day resupply order was \"placed with the DME vendor prior to discharge\" and is expected to arrive at the home.\n- Referral packet explicitly documents NO convexity, NO barrier rings, and NO ostomy belt were issued, and states peristomal skin was intact at discharge.\n- Referral lists the patient as \"independent with pouch change after teaching, return demonstration completed x1.\"\n\nWHAT TODAY'S SOC VISIT NOTE MUST SHOW\n1. Peristomal skin: a well-demarcated area of erythema with superficial moist denudation along the inferior 5 o'clock to 7 o'clock margin of the peristomal skin, approximately 2 cm at its widest, tender, no induration, no odor, no satellite lesions, no purulence — i.e., irritant contact dermatitis from effluent undermining the wafer, not infection. The skin elsewhere around the stoma is intact.\n2. Stoma itself: viable, pink-red, now budded only about 1/8 to 1/4 inch and retracting slightly below skin level when Roderick sits or bends forward. Measured today at 1 1/8 inches — the post-operative edema has resolved and the stoma has shrunk from the discharge measurement.\n3. Wear time and leakage: Roderick reports the wafer is lifting at the inferior edge and he is getting leakage \"every day or two,\" with actual wear time about 24–36 hours rather than the 3–4 days the referral prescribes. He has been reinforcing the lifting edge with household tape. Effluent is soft/pasty, 4–6 pouch emptyings per day.\n4. Cut-to-fit sizing observed at the visit: he is still cutting the wafer opening to the 1 1/4 inch discharge measurement, leaving a visible gap of bare skin between the stoma and the barrier edge at the inferior margin — the exact location of the skin breakdown.\n5. Supply counts on hand at home, counted by the clinician and documented as counted (not reported): wafers 6 remaining, drainable pouches 7 remaining, stoma paste tube about half full, skin barrier wipes 22 remaining, adhesive remover spray nearly full, deodorant drops nearly full. The DME resupply shipment has NOT arrived; Roderick has no tracking information and has not called the vendor.\n6. Technique observation: return demonstration today shows he cannot comfortably hold the wafer taut and press the inferior fold flat at the same time because of hand arthritis; he does not use a mirror; he does not currently tuck or flatten the parastomal fold before applying.\n7. Barrier ring / convexity status documented explicitly as: none in the home, none ordered, never issued.\n8. Caregiver arrangement: adult child available weekends only; no aide authorized; Roderick performs six of seven changes alone standing at the sink.\n\nTHE DELIBERATE CONFLICT BETWEEN REFERRAL AND NOTES (this is what makes the question answerable and non-trivial)\nThe referral packet and the visit note disagree on three axes, and the disagreement is the test:\n- SIZE CONFLICT: referral documents a 1 1/4 inch stoma and supplies 2 1/4 inch wafers; today's measurement is 1 1/8 inch and shrinking, and he is still cutting to the stale referral number. A reader who trusts the referral alone will conclude the sizing is correct and see no problem.\n- CONTOUR CONFLICT: referral says \"peristomal skin intact, no convexity needed, flat wafer appropriate.\" Today's note documents a stoma that has retracted to near skin level over an inferior parastomal fold, with effluent undermining the flat wafer at exactly that fold — the classic indication for a convex barrier or a moldable barrier ring plus belt. The referral's \"flat is fine\" judgment was true at discharge and is false now.\n- SUPPLY-COUNT CONFLICT: the referral asserts a 30-day resupply was ordered and is en route, which invites the reader to answer \"None — he's stocked.\" The counted inventory shows 6 wafers and 7 pouches, and at his ACTUAL 24–36 hour wear time that is roughly 6–7 days of supply, not 30 — while at the referral's PRESCRIBED 3–4 day wear time the same 6 wafers would read as ~3 weeks and look adequate. The same number is either comfortable or alarming depending on which document the reader believes.\n\nHOW THIS FORCES THE QUESTION\n\"Which ostomy supply does the patient need most this visit?\" now has a defensible best answer (a barrier ring or convex barrier — the item that would stop the leak at the inferior fold and let the peristomal dermatitis heal; it is documented as absent, never issued, and directly causative), with three well-baited near-misses that the packet actively supports: (a) more wafers, because the counted inventory really is low and the promised shipment really has not arrived; (b) an ostomy belt, defensible as an adjunct but useless without something to seal the fold; (c) \"None — adequately stocked,\" which is exactly what the referral's 30-day-resupply line and 'skin intact' line would lead a referral-only reader to pick. A reader must reconcile a stale referral against a fresh observation, and must convert a raw supply count into days-on-hand using the observed wear time rather than the prescribed one, before the answer separates from the distractors.\n\nWhat this patient deliberately does NOT contain: no urostomy or ileostomy effluent details (keeps the colostomy unambiguous), no fungal/candidal findings (would redirect the answer to antifungal powder and muddy the supply question), no stoma necrosis or bleeding (would make the answer 'call the surgeon' rather than a supply), and no existing barrier ring anywhere in the home (so 'he already has one' is never a valid out).\nYou may edit \"plan\" in evidence/run/lab/work/patients/gap-ostomy-supplies-soc/07f91c04/plan.json first.\nyes = kick generate (Patient QA then locks it onto the shelf; you do not approve the chart); no = leave it on the queue." +Answer with: flows answer --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove4-data 01M36E648MCDW2TJAC5J06X2VZ human-4 yes|no +Then continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove4-data --local-agent 01M36E648MCDW2TJAC5J06X2VZ +RUN 01M36E648MCDW2TJAC5J06X2VZ parked exit=3 diff --git a/examples/prompt-lab/evidence/run/08-answer.txt b/examples/prompt-lab/evidence/run/08-answer.txt index c01e6b8c..069ac315 100644 --- a/examples/prompt-lab/evidence/run/08-answer.txt +++ b/examples/prompt-lab/evidence/run/08-answer.txt @@ -1,4 +1,4 @@ -$ npx flows answer --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data 01M36BPYVQFDCG2WEP7V7H04X4 human-4 yes --by prompt-lab-reviewer -ANSWERED 01M36BPYVQFDCG2WEP7V7H04X4 human-4 yes -Continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data --local-agent 01M36BPYVQFDCG2WEP7V7H04X4 +$ npx flows answer --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove4-data 01M36E648MCDW2TJAC5J06X2VZ human-4 yes --by prompt-lab-reviewer +ANSWERED 01M36E648MCDW2TJAC5J06X2VZ human-4 yes +Continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove4-data --local-agent 01M36E648MCDW2TJAC5J06X2VZ exit=0 diff --git a/examples/prompt-lab/evidence/run/09-resume.txt b/examples/prompt-lab/evidence/run/09-resume.txt index 6f48ee94..bcee60ea 100644 --- a/examples/prompt-lab/evidence/run/09-resume.txt +++ b/examples/prompt-lab/evidence/run/09-resume.txt @@ -1,33 +1,83 @@ -$ npx flows resume --no-observer-link --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data --local-agent 01M36BPYVQFDCG2WEP7V7H04X4 +$ npx flows resume --no-observer-link --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove4-data --local-agent 01M36E648MCDW2TJAC5J06X2VZ ○ run-1 (deterministic) 0.00s ✓ run-1 (deterministic) 0.01s completionReason: success ○ llm-2 (llm) 0.00s -✓ llm-2 (llm) 4.25s completionReason: success +✓ llm-2 (llm) 4.24s completionReason: success ○ run-3 (deterministic) 0.00s ✓ run-3 (deterministic) 0.01s completionReason: success ○ human-4 (deterministic) 0.00s -✓ human-4 (deterministic) 0.01s completionReason: success +✓ human-4 (deterministic) 0.02s completionReason: success ○ run-5 (deterministic) 0.00s -✓ run-5 (deterministic) 0.06s completionReason: success +✓ run-5 (deterministic) 0.07s completionReason: success ○ llm-6 (llm) 0.00s -WAITING [worker_lease] Run "01M36BSWPMV69TB4YAE3K4SA7V" step "llm-6" (llm) is running under a worker lease until 1790141360145. -↻ llm-6 (llm) 4.23s -WAITING [worker_lease] Run "01M36BSWPMV69TB4YAE3K4SA7V" step "llm-6" (llm) is running under a worker lease until 1790141370147. -↻ llm-6 (llm) 14.27s -WAITING [worker_lease] Run "01M36BSWPMV69TB4YAE3K4SA7V" step "llm-6" (llm) is running under a worker lease until 1790141380150. -↻ llm-6 (llm) 24.24s -WAITING [worker_lease] Run "01M36BSWPMV69TB4YAE3K4SA7V" step "llm-6" (llm) is running under a worker lease until 1790141390152. -↻ llm-6 (llm) 34.24s -✓ llm-6 (llm) 41.18s completionReason: success +WAITING [worker_lease] Run "01M36E7YBSRHWDR5SKYXRF2QTM" step "llm-6" (llm) is running under a worker lease until 1790143917741. +↻ llm-6 (llm) 4.57s +WAITING [worker_lease] Run "01M36E7YBSRHWDR5SKYXRF2QTM" step "llm-6" (llm) is running under a worker lease until 1790143927744. +↻ llm-6 (llm) 14.59s +WAITING [worker_lease] Run "01M36E7YBSRHWDR5SKYXRF2QTM" step "llm-6" (llm) is running under a worker lease until 1790143937748. +↻ llm-6 (llm) 24.61s +WAITING [worker_lease] Run "01M36E7YBSRHWDR5SKYXRF2QTM" step "llm-6" (llm) is running under a worker lease until 1790143947749. +↻ llm-6 (llm) 34.62s +✓ llm-6 (llm) 41.56s completionReason: success ○ llm-7 (llm) 0.00s -WAITING [worker_lease] Run "01M36BV4YJ8GEC3YTNNHH5QCQJ" step "llm-7" (llm) is running under a worker lease until 1790141401349. -↻ llm-7 (llm) 4.25s -WAITING [worker_lease] Run "01M36BV4YJ8GEC3YTNNHH5QCQJ" step "llm-7" (llm) is running under a worker lease until 1790141411351. -↻ llm-7 (llm) 14.27s -WAITING [worker_lease] Run "01M36BV4YJ8GEC3YTNNHH5QCQJ" step "llm-7" (llm) is running under a worker lease until 1790141421353. -↻ llm-7 (llm) 24.28s -✓ llm-7 (llm) 24.33s completionReason: success -○ run-8 (deterministic) 0.00s -✓ run-8 (deterministic) 0.07s completionReason: success -RUN 01M36BPYVQFDCG2WEP7V7H04X4 completed (9 steps) completionReason: success +WAITING [worker_lease] Run "01M36E9639ZZXS5T3T1ASGSPDZ" step "llm-7" (llm) is running under a worker lease until 1790143958427. +↻ llm-7 (llm) 3.70s +WAITING [worker_lease] Run "01M36E9639ZZXS5T3T1ASGSPDZ" step "llm-7" (llm) is running under a worker lease until 1790143968429. +↻ llm-7 (llm) 13.74s +WAITING [worker_lease] Run "01M36E9639ZZXS5T3T1ASGSPDZ" step "llm-7" (llm) is running under a worker lease until 1790143978437. +↻ llm-7 (llm) 23.71s +✓ llm-7 (llm) 29.36s completionReason: success +○ llm-8 (llm) 0.00s +WAITING [worker_lease] Run "01M36EA3C3B11CP420JTP0R1X8" step "llm-8" (llm) is running under a worker lease until 1790143988407. +↻ llm-8 (llm) 4.32s +WAITING [worker_lease] Run "01M36EA3C3B11CP420JTP0R1X8" step "llm-8" (llm) is running under a worker lease until 1790143998411. +↻ llm-8 (llm) 14.35s +WAITING [worker_lease] Run "01M36EA3C3B11CP420JTP0R1X8" step "llm-8" (llm) is running under a worker lease until 1790144008416. +↻ llm-8 (llm) 24.37s +WAITING [worker_lease] Run "01M36EA3C3B11CP420JTP0R1X8" step "llm-8" (llm) is running under a worker lease until 1790144018418. +↻ llm-8 (llm) 34.37s +WAITING [worker_lease] Run "01M36EA3C3B11CP420JTP0R1X8" step "llm-8" (llm) is running under a worker lease until 1790144028421. +↻ llm-8 (llm) 44.36s +WAITING [worker_lease] Run "01M36EA3C3B11CP420JTP0R1X8" step "llm-8" (llm) is running under a worker lease until 1790144038431. +↻ llm-8 (llm) 54.39s +✓ llm-8 (llm) 56.15s completionReason: success +○ llm-9 (llm) 0.00s +WAITING [worker_lease] Run "01M36EBT03M577CK77SR9FMZDY" step "llm-9" (llm) is running under a worker lease until 1790144044343. +↻ llm-9 (llm) 4.10s +WAITING [worker_lease] Run "01M36EBT03M577CK77SR9FMZDY" step "llm-9" (llm) is running under a worker lease until 1790144054347. +↻ llm-9 (llm) 14.15s +WAITING [worker_lease] Run "01M36EBT03M577CK77SR9FMZDY" step "llm-9" (llm) is running under a worker lease until 1790144064349. +↻ llm-9 (llm) 24.11s +WAITING [worker_lease] Run "01M36EBT03M577CK77SR9FMZDY" step "llm-9" (llm) is running under a worker lease until 1790144074355. +↻ llm-9 (llm) 34.16s +✓ llm-9 (llm) 37.44s completionReason: success +○ llm-10 (llm) 0.00s +WAITING [worker_lease] Run "01M36ECXZA26HVA1FGQ91KMGSF" step "llm-10" (llm) is running under a worker lease until 1790144081184. +↻ llm-10 (llm) 3.51s +WAITING [worker_lease] Run "01M36ECXZA26HVA1FGQ91KMGSF" step "llm-10" (llm) is running under a worker lease until 1790144091185. +↻ llm-10 (llm) 13.52s +WAITING [worker_lease] Run "01M36ECXZA26HVA1FGQ91KMGSF" step "llm-10" (llm) is running under a worker lease until 1790144101187. +↻ llm-10 (llm) 23.53s +WAITING [worker_lease] Run "01M36ECXZA26HVA1FGQ91KMGSF" step "llm-10" (llm) is running under a worker lease until 1790144111192. +↻ llm-10 (llm) 33.55s +WAITING [worker_lease] Run "01M36ECXZA26HVA1FGQ91KMGSF" step "llm-10" (llm) is running under a worker lease until 1790144121193. +↻ llm-10 (llm) 43.53s +✓ llm-10 (llm) 47.05s completionReason: success +○ llm-11 (llm) 0.00s +WAITING [worker_lease] Run "01M36EECFS15SBQQ6QTH6EEW37" step "llm-11" (llm) is running under a worker lease until 1790144128812. +↻ llm-11 (llm) 4.09s +WAITING [worker_lease] Run "01M36EECFS15SBQQ6QTH6EEW37" step "llm-11" (llm) is running under a worker lease until 1790144138816. +↻ llm-11 (llm) 14.10s +WAITING [worker_lease] Run "01M36EECFS15SBQQ6QTH6EEW37" step "llm-11" (llm) is running under a worker lease until 1790144148855. +↻ llm-11 (llm) 24.13s +✓ llm-11 (llm) 32.98s completionReason: success +○ llm-12 (llm) 0.00s +WAITING [worker_lease] Run "01M36EFC9RWHMSS074KACZFJMT" step "llm-12" (llm) is running under a worker lease until 1790144161388. +↻ llm-12 (llm) 3.68s +WAITING [worker_lease] Run "01M36EFC9RWHMSS074KACZFJMT" step "llm-12" (llm) is running under a worker lease until 1790144171440. +↻ llm-12 (llm) 13.74s +✓ llm-12 (llm) 17.65s completionReason: success +○ run-13 (deterministic) 0.00s +✓ run-13 (deterministic) 0.07s completionReason: success +RUN 01M36E648MCDW2TJAC5J06X2VZ completed (14 steps) completionReason: success exit=0 diff --git a/examples/prompt-lab/evidence/run/10-job2-run.txt b/examples/prompt-lab/evidence/run/10-job2-run.txt index 0c283c5f..cc7477ab 100644 --- a/examples/prompt-lab/evidence/run/10-job2-run.txt +++ b/examples/prompt-lab/evidence/run/10-job2-run.txt @@ -1,30 +1,34 @@ -$ npx flows run --no-observer-link --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data --local-agent prompt-lab.flow.ts --input {"job":"fix","reviewer":"prompt-lab-reviewer","lab":"evidence/run/lab","issueId":"config-sunrise-wound-status-8a366995"} +$ npx flows run --no-observer-link --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove4-data --local-agent prompt-lab.flow.ts --input {"job":"fix","reviewer":"prompt-lab-reviewer","lab":"evidence/run/lab","issueId":"config-sunrise-wound-status-9c1634e2"} ○ run-1 (deterministic) 0.00s ✓ run-1 (deterministic) 0.07s completionReason: success ○ llm-2 (llm) 0.00s -WAITING [worker_lease] Run "01M36BVY7T3HKS7YM1DWYCC9XT" step "llm-2" (llm) is running under a worker lease until 1790141427245. -↻ llm-2 (llm) 3.87s -WAITING [worker_lease] Run "01M36BVY7T3HKS7YM1DWYCC9XT" step "llm-2" (llm) is running under a worker lease until 1790141437248. -↻ llm-2 (llm) 13.90s -✓ llm-2 (llm) 16.01s completionReason: success +WAITING [worker_lease] Run "01M36EFZAQF4SWX7HCDMQ113RE" step "llm-2" (llm) is running under a worker lease until 1790144180876. +↻ llm-2 (llm) 3.55s +✓ llm-2 (llm) 13.49s completionReason: success ○ llm-3 (llm) 0.00s -WAITING [worker_lease] Run "01M36BWGDVRNR4NGN1GREEXR2H" step "llm-3" (llm) is running under a worker lease until 1790141445870. -↻ llm-3 (llm) 6.49s -✓ llm-3 (llm) 10.72s completionReason: success +WAITING [worker_lease] Run "01M36EGCV91FTX5RDAJ2S9YZHG" step "llm-3" (llm) is running under a worker lease until 1790144194718. +↻ llm-3 (llm) 3.90s +✓ llm-3 (llm) 8.10s completionReason: success ○ llm-4 (llm) 0.00s -WAITING [worker_lease] Run "01M36BWR19D9GP7XWB2HX39F20" step "llm-4" (llm) is running under a worker lease until 1790141453661. -↻ llm-4 (llm) 3.56s -✓ llm-4 (llm) 7.99s completionReason: success +WAITING [worker_lease] Run "01M36EGMD3382FZYMNTDR1XJKF" step "llm-4" (llm) is running under a worker lease until 1790144202456. +↻ llm-4 (llm) 3.54s +WAITING [worker_lease] Run "01M36EGMD3382FZYMNTDR1XJKF" step "llm-4" (llm) is running under a worker lease until 1790144212459. +↻ llm-4 (llm) 13.55s +✓ llm-4 (llm) 22.55s completionReason: success ○ llm-5 (llm) 0.00s -WAITING [worker_lease] Run "01M36BX06ZZAG345PMZD9TT43T" step "llm-5" (llm) is running under a worker lease until 1790141462034. -↻ llm-5 (llm) 3.94s -✓ llm-5 (llm) 8.27s completionReason: success -○ run-6 (deterministic) 0.00s -✓ run-6 (deterministic) 0.06s completionReason: success -○ human-7 (deterministic) 0.00s -⏸ human-7 (human) 0.00s -PARKED [run_parked] Run "01M36BVTCK80EV8P4FWA8WZZHK" is waiting for prompt-lab-reviewer to answer human-7: "Question workbench · wound-status (issue config-sunrise-wound-status-8a366995): current-prompt outputs on 3 shelf patient(s) — jordan: No wound/High, pat: Ongoing/Medium, riley: No wound/High.\nSet gold in evidence/run/lab/work/q-wound-status/cd33704e/grid.json (\"target\" per row; prefilled from persisted gold or the config-level target). Your review persists as gold; later AI runs never overwrite it.\nyes = persist gold and run the iterator; no = stop." -Answer with: flows answer --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data 01M36BVTCK80EV8P4FWA8WZZHK human-7 yes|no -Then continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data --local-agent 01M36BVTCK80EV8P4FWA8WZZHK -RUN 01M36BVTCK80EV8P4FWA8WZZHK parked +WAITING [worker_lease] Run "01M36EHBNNT2JRNR13FEWP01SX" step "llm-5" (llm) is running under a worker lease until 1790144226280. +↻ llm-5 (llm) 4.82s +✓ llm-5 (llm) 8.92s completionReason: success +○ llm-6 (llm) 0.00s +WAITING [worker_lease] Run "01M36EHKRJYVZS7ZGF7NRGA76Z" step "llm-6" (llm) is running under a worker lease until 1790144234566. +↻ llm-6 (llm) 4.19s +✓ llm-6 (llm) 10.60s completionReason: success +○ run-7 (deterministic) 0.00s +✓ run-7 (deterministic) 0.07s completionReason: success +○ human-8 (deterministic) 0.00s +⏸ human-8 (human) 0.00s +PARKED [run_parked] Run "01M36EFVSGDSQA4R3BG4G02RMM" is waiting for prompt-lab-reviewer to answer human-8: "Question workbench · wound-status (issue config-sunrise-wound-status-9c1634e2): current-prompt outputs on 4 shelf patient(s) — jordan: No wound/High, pat: Healed/High, riley: No wound/High, roderick: Ongoing/Medium.\nSet gold in evidence/run/lab/work/q-wound-status/205c23a4/grid.json (\"target\" per row; prefilled from persisted gold or the config-level target). Your review persists as gold; later AI runs never overwrite it.\nyes = persist gold and run the iterator; no = stop." +Answer with: flows answer --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove4-data 01M36EFVSGDSQA4R3BG4G02RMM human-8 yes|no +Then continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove4-data --local-agent 01M36EFVSGDSQA4R3BG4G02RMM +RUN 01M36EFVSGDSQA4R3BG4G02RMM parked exit=3 diff --git a/examples/prompt-lab/evidence/run/11-answer.txt b/examples/prompt-lab/evidence/run/11-answer.txt index a0e3c59b..54f46111 100644 --- a/examples/prompt-lab/evidence/run/11-answer.txt +++ b/examples/prompt-lab/evidence/run/11-answer.txt @@ -1,4 +1,4 @@ -$ npx flows answer --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data 01M36BVTCK80EV8P4FWA8WZZHK human-7 yes --by prompt-lab-reviewer -ANSWERED 01M36BVTCK80EV8P4FWA8WZZHK human-7 yes -Continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data --local-agent 01M36BVTCK80EV8P4FWA8WZZHK +$ npx flows answer --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove4-data 01M36EFVSGDSQA4R3BG4G02RMM human-8 yes --by prompt-lab-reviewer +ANSWERED 01M36EFVSGDSQA4R3BG4G02RMM human-8 yes +Continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove4-data --local-agent 01M36EFVSGDSQA4R3BG4G02RMM exit=0 diff --git a/examples/prompt-lab/evidence/run/12-resume.txt b/examples/prompt-lab/evidence/run/12-resume.txt index 482a0f11..70ba9094 100644 --- a/examples/prompt-lab/evidence/run/12-resume.txt +++ b/examples/prompt-lab/evidence/run/12-resume.txt @@ -1,68 +1,74 @@ -$ npx flows resume --no-observer-link --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data --local-agent 01M36BVTCK80EV8P4FWA8WZZHK +$ npx flows resume --no-observer-link --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove4-data --local-agent 01M36EFVSGDSQA4R3BG4G02RMM ○ run-1 (deterministic) 0.00s -✓ run-1 (deterministic) 0.01s completionReason: success +✓ run-1 (deterministic) 0.02s completionReason: success ○ llm-2 (llm) 0.00s -✓ llm-2 (llm) 3.82s completionReason: success +✓ llm-2 (llm) 4.34s completionReason: success ○ llm-3 (llm) 0.00s -✓ llm-3 (llm) 4.13s completionReason: success +✓ llm-3 (llm) 5.02s completionReason: success ○ llm-4 (llm) 0.00s -✓ llm-4 (llm) 3.87s completionReason: success +✓ llm-4 (llm) 3.53s completionReason: success ○ llm-5 (llm) 0.00s -✓ llm-5 (llm) 4.40s completionReason: success -○ run-6 (deterministic) 0.00s -✓ run-6 (deterministic) 0.01s completionReason: success -○ human-7 (deterministic) 0.00s -✓ human-7 (deterministic) 0.01s completionReason: success -○ run-8 (deterministic) 0.00s -✓ run-8 (deterministic) 0.06s completionReason: success +✓ llm-5 (llm) 4.62s completionReason: success +○ llm-6 (llm) 0.00s +✓ llm-6 (llm) 4.45s completionReason: success +○ run-7 (deterministic) 0.00s +✓ run-7 (deterministic) 0.13s completionReason: success +○ human-8 (deterministic) 0.00s +✓ human-8 (deterministic) 0.04s completionReason: success ○ run-9 (deterministic) 0.00s ✓ run-9 (deterministic) 0.07s completionReason: success -○ llm-10 (llm) 0.00s -WAITING [worker_lease] Run "01M36BXT4CQ15K7D7BGKN0WT5G" step "llm-10" (llm) is running under a worker lease until 1790141488575. -↻ llm-10 (llm) 3.83s -WAITING [worker_lease] Run "01M36BXT4CQ15K7D7BGKN0WT5G" step "llm-10" (llm) is running under a worker lease until 1790141498577. -↻ llm-10 (llm) 13.87s -✓ llm-10 (llm) 19.22s completionReason: success +○ run-10 (deterministic) 0.00s +✓ run-10 (deterministic) 0.07s completionReason: success ○ llm-11 (llm) 0.00s -WAITING [worker_lease] Run "01M36BYEV7RNQ0ZRM8NPD94ZS6" step "llm-11" (llm) is running under a worker lease until 1790141509788. -↻ llm-11 (llm) 5.83s -WAITING [worker_lease] Run "01M36BYEV7RNQ0ZRM8NPD94ZS6" step "llm-11" (llm) is running under a worker lease until 1790141519790. -↻ llm-11 (llm) 15.85s -✓ llm-11 (llm) 22.17s completionReason: success +WAITING [worker_lease] Run "01M36EJPEHFCPSC9YAB4VA5DYF" step "llm-11" (llm) is running under a worker lease until 1790144270201. +↻ llm-11 (llm) 5.36s +WAITING [worker_lease] Run "01M36EJPEHFCPSC9YAB4VA5DYF" step "llm-11" (llm) is running under a worker lease until 1790144280205. +↻ llm-11 (llm) 15.35s +✓ llm-11 (llm) 21.83s completionReason: success ○ llm-12 (llm) 0.00s -WAITING [worker_lease] Run "01M36BZ2JR3WBGG3FETBMR66FX" step "llm-12" (llm) is running under a worker lease until 1790141530006. -↻ llm-12 (llm) 3.88s -WAITING [worker_lease] Run "01M36BZ2JR3WBGG3FETBMR66FX" step "llm-12" (llm) is running under a worker lease until 1790141540009. -↻ llm-12 (llm) 13.91s -WAITING [worker_lease] Run "01M36BZ2JR3WBGG3FETBMR66FX" step "llm-12" (llm) is running under a worker lease until 1790141550011. -↻ llm-12 (llm) 23.92s -✓ llm-12 (llm) 31.46s completionReason: success +WAITING [worker_lease] Run "01M36EKBJ0Q3B0Y3DWP85BZ942" step "llm-12" (llm) is running under a worker lease until 1790144291701. +↻ llm-12 (llm) 4.98s +WAITING [worker_lease] Run "01M36EKBJ0Q3B0Y3DWP85BZ942" step "llm-12" (llm) is running under a worker lease until 1790144301703. +↻ llm-12 (llm) 15.00s +✓ llm-12 (llm) 17.21s completionReason: success ○ llm-13 (llm) 0.00s -WAITING [worker_lease] Run "01M36C01XRV3F9DKHN4JWCV3XW" step "llm-13" (llm) is running under a worker lease until 1790141562091. -↻ llm-13 (llm) 4.50s -WAITING [worker_lease] Run "01M36C01XRV3F9DKHN4JWCV3XW" step "llm-13" (llm) is running under a worker lease until 1790141572092. -↻ llm-13 (llm) 14.51s -✓ llm-13 (llm) 15.14s completionReason: success -○ run-14 (deterministic) 0.00s -✓ run-14 (deterministic) 0.06s completionReason: success -○ llm-15 (llm) 0.00s -WAITING [worker_lease] Run "01M36C0GPZTJGQTMD14Y6TV9WG" step "llm-15" (llm) is running under a worker lease until 1790141577235. -↻ llm-15 (llm) 4.44s -✓ llm-15 (llm) 8.57s completionReason: success +WAITING [worker_lease] Run "01M36EKV3P8NJVDNGPK5H2M8A3" step "llm-13" (llm) is running under a worker lease until 1790144307627. +↻ llm-13 (llm) 3.69s +WAITING [worker_lease] Run "01M36EKV3P8NJVDNGPK5H2M8A3" step "llm-13" (llm) is running under a worker lease until 1790144317629. +↻ llm-13 (llm) 13.72s +WAITING [worker_lease] Run "01M36EKV3P8NJVDNGPK5H2M8A3" step "llm-13" (llm) is running under a worker lease until 1790144327632. +↻ llm-13 (llm) 23.70s +✓ llm-13 (llm) 30.89s completionReason: success +○ llm-14 (llm) 0.00s +WAITING [worker_lease] Run "01M36EMSDJWGR1SD3R7JDT9D6M" step "llm-14" (llm) is running under a worker lease until 1790144338660. +↻ llm-14 (llm) 3.83s +WAITING [worker_lease] Run "01M36EMSDJWGR1SD3R7JDT9D6M" step "llm-14" (llm) is running under a worker lease until 1790144348667. +↻ llm-14 (llm) 13.87s +✓ llm-14 (llm) 16.47s completionReason: success +○ run-15 (deterministic) 0.00s +✓ run-15 (deterministic) 0.06s completionReason: success ○ llm-16 (llm) 0.00s -WAITING [worker_lease] Run "01M36C0RKC7NK9XXY5AGQ74TYH" step "llm-16" (llm) is running under a worker lease until 1790141585311. -↻ llm-16 (llm) 3.95s -✓ llm-16 (llm) 8.04s completionReason: success +WAITING [worker_lease] Run "01M36EN9H89T3FX8NFAKAECZSM" step "llm-16" (llm) is running under a worker lease until 1790144355191. +↻ llm-16 (llm) 3.83s +✓ llm-16 (llm) 8.77s completionReason: success ○ llm-17 (llm) 0.00s -WAITING [worker_lease] Run "01M36C108ACT174RZXCWBQCGR5" step "llm-17" (llm) is running under a worker lease until 1790141593150. -↻ llm-17 (llm) 3.74s -✓ llm-17 (llm) 8.03s completionReason: success -○ run-18 (deterministic) 0.00s -✓ run-18 (deterministic) 0.07s completionReason: success -○ human-19 (deterministic) 0.00s -⏸ human-19 (human) 0.00s -PARKED [run_parked] Run "01M36BVTCK80EV8P4FWA8WZZHK" is waiting for prompt-lab-reviewer to answer human-19: "Re-run and score · wound-status, new prompt p-wound-status-f1305d28:\n3 of 3 golded patients worked (100%)\n jordan: gold No wound · new run No wound · worked\n pat: gold Ongoing · new run Ongoing · worked\n riley: gold No wound · new run No wound · worked\nSHARED: marking done changes this prompt for every agency that uses it: harbor, maple, sunrise.\nRead the new prompt in evidence/run/lab/bank.json and the outputs in evidence/run/lab/work/q-wound-status/cd33704e/score.json.\nyes = mark done (live in Apricot now, no promote step); no = keep it as a draft and iterate again in a new run." -Answer with: flows answer --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data 01M36BVTCK80EV8P4FWA8WZZHK human-19 yes|no -Then continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data --local-agent 01M36BVTCK80EV8P4FWA8WZZHK -RUN 01M36BVTCK80EV8P4FWA8WZZHK parked +WAITING [worker_lease] Run "01M36ENK4HJP592PW7R0G1VDXR" step "llm-17" (llm) is running under a worker lease until 1790144365028. +↻ llm-17 (llm) 4.90s +✓ llm-17 (llm) 9.70s completionReason: success +○ llm-18 (llm) 0.00s +WAITING [worker_lease] Run "01M36ENVR2K30B8RCP05CV75M7" step "llm-18" (llm) is running under a worker lease until 1790144373814. +↻ llm-18 (llm) 3.99s +✓ llm-18 (llm) 8.23s completionReason: success +○ llm-19 (llm) 0.00s +WAITING [worker_lease] Run "01M36EP3EN10CX5QR835F350E5" step "llm-19" (llm) is running under a worker lease until 1790144381704. +↻ llm-19 (llm) 3.64s +✓ llm-19 (llm) 8.95s completionReason: success +○ run-20 (deterministic) 0.00s +✓ run-20 (deterministic) 0.07s completionReason: success +○ human-21 (deterministic) 0.00s +⏸ human-21 (human) 0.00s +PARKED [run_parked] Run "01M36EFVSGDSQA4R3BG4G02RMM" is waiting for prompt-lab-reviewer to answer human-21: "Re-run and score · wound-status, new prompt p-wound-status-003f262d:\n3 of 4 golded patients worked (75%)\n jordan: gold No wound · new run No wound · worked\n pat: gold Ongoing · new run Ongoing · worked\n riley: gold No wound · new run No wound · worked\n roderick: gold Ongoing · new run No wound · did not\nSHARED: marking done changes this prompt for every agency that uses it: harbor, maple, sunrise.\nRead the new prompt in evidence/run/lab/bank.json and the outputs in evidence/run/lab/work/q-wound-status/205c23a4/score.json.\nyes = mark done (live in Apricot now, no promote step); no = keep it as a draft and iterate again in a new run." +Answer with: flows answer --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove4-data 01M36EFVSGDSQA4R3BG4G02RMM human-21 yes|no +Then continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove4-data --local-agent 01M36EFVSGDSQA4R3BG4G02RMM +RUN 01M36EFVSGDSQA4R3BG4G02RMM parked exit=3 diff --git a/examples/prompt-lab/evidence/run/13-answer.txt b/examples/prompt-lab/evidence/run/13-answer.txt index 6bd71893..23c4a1f0 100644 --- a/examples/prompt-lab/evidence/run/13-answer.txt +++ b/examples/prompt-lab/evidence/run/13-answer.txt @@ -1,4 +1,4 @@ -$ npx flows answer --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data 01M36BVTCK80EV8P4FWA8WZZHK human-19 yes --by prompt-lab-reviewer -ANSWERED 01M36BVTCK80EV8P4FWA8WZZHK human-19 yes -Continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data --local-agent 01M36BVTCK80EV8P4FWA8WZZHK +$ npx flows answer --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove4-data 01M36EFVSGDSQA4R3BG4G02RMM human-21 yes --by prompt-lab-reviewer +ANSWERED 01M36EFVSGDSQA4R3BG4G02RMM human-21 yes +Continue with: flows resume --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove4-data --local-agent 01M36EFVSGDSQA4R3BG4G02RMM exit=0 diff --git a/examples/prompt-lab/evidence/run/14-resume.txt b/examples/prompt-lab/evidence/run/14-resume.txt index 0f29a0e9..a6316fa2 100644 --- a/examples/prompt-lab/evidence/run/14-resume.txt +++ b/examples/prompt-lab/evidence/run/14-resume.txt @@ -1,45 +1,49 @@ -$ npx flows resume --no-observer-link --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove2-data --local-agent 01M36BVTCK80EV8P4FWA8WZZHK +$ npx flows resume --no-observer-link --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove4-data --local-agent 01M36EFVSGDSQA4R3BG4G02RMM ○ run-1 (deterministic) 0.00s ✓ run-1 (deterministic) 0.01s completionReason: success ○ llm-2 (llm) 0.00s ✓ llm-2 (llm) 4.01s completionReason: success ○ llm-3 (llm) 0.00s -✓ llm-3 (llm) 3.67s completionReason: success +✓ llm-3 (llm) 4.23s completionReason: success ○ llm-4 (llm) 0.00s -✓ llm-4 (llm) 5.00s completionReason: success +✓ llm-4 (llm) 4.20s completionReason: success ○ llm-5 (llm) 0.00s -✓ llm-5 (llm) 3.61s completionReason: success -○ run-6 (deterministic) 0.00s -✓ run-6 (deterministic) 0.01s completionReason: success -○ human-7 (deterministic) 0.00s -✓ human-7 (deterministic) 0.01s completionReason: success -○ run-8 (deterministic) 0.00s -✓ run-8 (deterministic) 0.01s completionReason: success +✓ llm-5 (llm) 3.83s completionReason: success +○ llm-6 (llm) 0.00s +✓ llm-6 (llm) 6.89s completionReason: success +○ run-7 (deterministic) 0.00s +✓ run-7 (deterministic) 0.01s completionReason: success +○ human-8 (deterministic) 0.00s +✓ human-8 (deterministic) 0.01s completionReason: success ○ run-9 (deterministic) 0.00s ✓ run-9 (deterministic) 0.01s completionReason: success -○ llm-10 (llm) 0.00s -✓ llm-10 (llm) 3.56s completionReason: success +○ run-10 (deterministic) 0.00s +✓ run-10 (deterministic) 0.01s completionReason: success ○ llm-11 (llm) 0.00s -✓ llm-11 (llm) 3.87s completionReason: success +✓ llm-11 (llm) 4.21s completionReason: success ○ llm-12 (llm) 0.00s -✓ llm-12 (llm) 4.36s completionReason: success +✓ llm-12 (llm) 3.82s completionReason: success ○ llm-13 (llm) 0.00s -✓ llm-13 (llm) 3.70s completionReason: success -○ run-14 (deterministic) 0.00s -✓ run-14 (deterministic) 0.01s completionReason: success -○ llm-15 (llm) 0.00s -✓ llm-15 (llm) 3.73s completionReason: success +✓ llm-13 (llm) 3.98s completionReason: success +○ llm-14 (llm) 0.00s +✓ llm-14 (llm) 3.68s completionReason: success +○ run-15 (deterministic) 0.00s +✓ run-15 (deterministic) 0.01s completionReason: success ○ llm-16 (llm) 0.00s -✓ llm-16 (llm) 3.29s completionReason: success +✓ llm-16 (llm) 4.37s completionReason: success ○ llm-17 (llm) 0.00s -✓ llm-17 (llm) 3.76s completionReason: success -○ run-18 (deterministic) 0.00s -✓ run-18 (deterministic) 0.01s completionReason: success -○ human-19 (deterministic) 0.00s -✓ human-19 (deterministic) 0.01s completionReason: success +✓ llm-17 (llm) 3.44s completionReason: success +○ llm-18 (llm) 0.00s +✓ llm-18 (llm) 4.06s completionReason: success +○ llm-19 (llm) 0.00s +✓ llm-19 (llm) 4.12s completionReason: success ○ run-20 (deterministic) 0.00s -✓ run-20 (deterministic) 0.06s completionReason: success -○ run-21 (deterministic) 0.00s -✓ run-21 (deterministic) 0.06s completionReason: success -RUN 01M36BVTCK80EV8P4FWA8WZZHK completed (22 steps) completionReason: success +✓ run-20 (deterministic) 0.01s completionReason: success +○ human-21 (deterministic) 0.00s +✓ human-21 (deterministic) 0.02s completionReason: success +○ run-22 (deterministic) 0.00s +✓ run-22 (deterministic) 0.07s completionReason: success +○ run-23 (deterministic) 0.00s +✓ run-23 (deterministic) 0.07s completionReason: success +RUN 01M36EFVSGDSQA4R3BG4G02RMM completed (24 steps) completionReason: success exit=0 diff --git a/examples/prompt-lab/evidence/run/15-lab-state.txt b/examples/prompt-lab/evidence/run/15-lab-state.txt index e0534cf9..7c4fc97f 100644 --- a/examples/prompt-lab/evidence/run/15-lab-state.txt +++ b/examples/prompt-lab/evidence/run/15-lab-state.txt @@ -5,12 +5,12 @@ console.log('shelf', require('fs').readdirSync('evidence/run/lab/shelf').join(' console.log('issues', JSON.stringify(require('./evidence/run/lab/queue/issues.json').map(i=>[i.id,i.status]))); console.log('patient briefs', JSON.stringify(require('./evidence/run/lab/queue/patient-briefs.json').map(i=>[i.id,i.status]))); console.log('gold', Object.keys(require('./evidence/run/lab/gold.json')).join(' ')); -wound-status live p-wound-status-f1305d28 draft - +wound-status live p-wound-status-003f262d draft - mood live p-mood-v1 draft - -living-situation live p-living-situation-373bc0c3 draft - -ostomy-supplies live null draft p-ostomy-supplies-26bf6ae3 -shelf jordan.json pat.json riley.json verna.json -issues [["config-sunrise-wound-status-8a366995","done"]] +living-situation live p-living-situation-90bc4192 draft - +ostomy-supplies live null draft p-ostomy-supplies-1d74ae8f +shelf jordan.json pat.json riley.json roderick.json +issues [["config-sunrise-wound-status-9c1634e2","done"]] patient briefs [["gap-ostomy-supplies-soc","locked"]] -gold wound-status|jordan wound-status|pat wound-status|riley +gold wound-status|jordan wound-status|pat wound-status|riley wound-status|roderick exit=0 diff --git a/examples/prompt-lab/evidence/run/lab/bank.json b/examples/prompt-lab/evidence/run/lab/bank.json index 44d835c6..9af44ec0 100644 --- a/examples/prompt-lab/evidence/run/lab/bank.json +++ b/examples/prompt-lab/evidence/run/lab/bank.json @@ -2,15 +2,15 @@ "prompts": { "p-wound-status-v1": "Determine the status of the patient's primary wound. Read the referral packet first; if the discharge summary states the wound is closed or healed, answer Healed. Otherwise use the visit notes. Use High confidence when the referral states the status.", "p-mood-v1": "Describe the patient's mood today from the visit notes. Choose the single option that best matches what the patient says and how the clinician describes them. Use High confidence when the patient states their feelings in their own words, Medium when inferred from behavior, Low when the notes are silent.", - "p-living-situation-373bc0c3": "You are filling in one field of a home-health assessment from the patient's visit chart: the patient's living situation. Choose exactly one of the options you were given, and never write an option that was not provided.\n\nSOURCE OF TRUTH\nToday's visit notes — what the clinician observed, saw, and was told during today's visit — are authoritative. Referral packets, intake forms, prior episode records, and face-to-face documents are supporting evidence only. When today's visit notes conflict with any of those older sources, today's visit notes win and the older source is treated as stale. When today's visit notes are silent on living situation, you may fall back to those older sources, but the answer is then inferred, not stated.\n\nHOW TO CHOOSE AMONG THE OPTIONS\nRead the whole chart for who is physically present in the patient's residence on an ongoing basis, and for what kind of residence it is. Then map what you find onto the options:\n- Choose the option meaning the patient lives by themselves when the chart indicates no other person resides in the home. A caregiver, family member, aide, or neighbor who visits, checks in, drops by, or stays intermittently does not make the patient a cohabitant — the patient still lives alone.\n- Choose the spouse option when the chart indicates the patient shares the residence with a spouse or a long-term partner and no broader household is described.\n- Choose the family option when the chart indicates the patient shares the residence with relatives other than a spouse, or with a household that includes a spouse plus other relatives.\n- Choose the facility option when the residence itself is an institutional or congregate setting with on-site staff rather than a private home.\nIf two options both seem defensible, pick the one supported by the most recent and most direct statement in the chart, and say in the explanation why the other was rejected. Do not blend options and do not answer with more than one.\n\nCONFIDENCE\n- High: today's visit notes state the living situation directly — an explicit statement of who the patient lives with or of the type of residence.\n- Medium: the answer is inferred rather than stated — for example, it comes only from the referral packet or an intake form, or it is deduced from indirect remarks in today's notes about the household.\n- Low: the chart says nothing about living situation, or sources contradict each other and nothing resolves which is current. Still select the single best-supported option, and say plainly that the basis is thin.\n\nEXPLANATION\nGive one or two sentences that name where in the chart the answer came from — the document or section and the substance of what it said. Quote or paraphrase only enough to show the basis. Do not include the patient's name, initials, date of birth, address, phone number, medical record number, or the names of family members, caregivers, clinicians, or facilities; describe people by role instead. Do not invent chart content: if the chart does not support a detail, leave it out and lower the confidence instead.", - "p-ostomy-supplies-26bf6ae3": "You are filling in one question on a home-health visit assessment. You will be given the question, its answer options, and the patient's chart. Choose exactly one of the options given to you — never invent, merge, or reword an option, and never answer with anything outside the list.\n\nSOURCE OF TRUTH\nToday's visit notes — what the clinician observed, measured, and was told during this visit — are the source of truth. When the visit notes conflict with the referral packet, intake forms, prior visit notes, the physician order, or any standing supply list, today's visit notes win. Use older documents only to fill gaps the visit notes leave silent, and only when nothing in today's notes contradicts them.\n\nHOW TO CHOOSE\nThis question asks which ostomy supply the patient needs most at this visit — a single, highest-priority need, not everything that is running low.\n- Read today's notes for what the clinician recorded about the stoma, the peristomal skin, the pouching system, wear time, leakage, and what supplies are on hand or were requested.\n- If the notes name a specific supply as needed, requested, ordered, or out of stock, choose the option matching that supply.\n- If the notes describe a problem rather than a supply, map the problem to the option that addresses it: an inability to collect output or an exhausted supply of the collection appliance points to the pouching supply; a leaking, poorly sealing, or uneven seal at the stoma edge points to the sealing/barrier option; irritated, denuded, or weeping peristomal skin needing protection before adhesion points to the skin-protection option.\n- If two needs appear, pick the one the notes treat as most urgent — the one causing an active problem today, the one the clinician acted on or escalated, or the one that must be resolved for the appliance to function at all. A comfort or convenience need never outranks an active leak, skin breakdown, or an inability to pouch.\n- Choose the \"no supply needed\" option only when today's notes affirmatively indicate the ostomy is intact, the appliance is functioning, and supplies are adequate — not merely because the notes say nothing.\n- If the chart is silent on ostomy supplies entirely, or the evidence points in two directions with no way to rank them, pick the option best supported by what little there is and mark confidence Low; do not guess a specific supply out of thin air.\n\nCONFIDENCE\n- High: today's visit notes state the answer directly — the needed supply is named, requested, ordered, or documented as out.\n- Medium: the answer is inferred — today's notes describe a condition, symptom, or situation from which the needed supply follows, but do not name the supply itself.\n- Low: the chart is silent on this question, or the available evidence contradicts itself and cannot be resolved by preferring today's notes.\n\nEXPLANATION\nAfter the answer, give a one- or two-sentence explanation that cites where in the chart the answer came from — name the section or entry (for example, today's visit note, the skin assessment, the supply inventory, the prior referral packet) and the specific observation or statement you relied on. If you had to override an older document with today's notes, say so. If confidence is Low, say what was missing or contradictory.\n\nDo not include any patient identifiers — no names, dates of birth, addresses, phone numbers, record or insurance numbers — in the answer or the explanation. Refer to the person only as \"the patient.\" Quote the chart only as briefly as needed to support the citation.", - "p-wound-status-f1305d28": "Determine the status of the patient's primary wound: Healed, Ongoing, or No wound.\n\nEvidence order. Today's visit notes describe the wound as it is now, and they govern the answer. Read the referral packet — admission history, discharge summary, prior wound descriptions — for context, but treat everything in it as a record of an earlier point in time.\n\n- Answer Ongoing when the most recent documentation describes an open or unhealed wound: measurements, drainage, wound bed or edge description, slough or eschar, or an ordered dressing change. Answer Ongoing even when an older document called the wound closed or healed; the newer observation is what is true today.\n- Answer Healed when the most recent documentation describes the wound as closed, resolved, or fully epithelialized with no open area remaining.\n- Answer No wound when the chart states the patient has no wound, or when neither the referral packet nor the visit notes document a wound at all.\n\nConfidence.\n- High — today's visit notes state the answer directly: they describe the wound's current condition in their own words. A status stated only in the referral packet is never High on its own, no matter how clearly it is worded and no matter that today's notes fail to contradict it.\n- Medium — today's visit notes do not state the wound's status, and you infer it from the surrounding record: the referral packet, standing orders, the supply list, or a care plan that implies a wound is or is not present.\n- Low — the chart is silent about the wound, or the chart is contradictory: two sources of comparable currency disagree, or today's notes are internally inconsistent, and nothing in the record resolves which is right.\n\nA document that a more recent observation supersedes is outdated, not contradictory. Chronology resolves the conflict, so where today's notes state the current condition directly the answer stays High even though an earlier document said otherwise. Reserve Low for disagreement that chronology cannot settle.\n\nExplanation. Give one or two sentences that cite the specific findings you relied on and, when you set an earlier document aside, name it and say it is outdated." + "p-living-situation-90bc4192": "You are determining the patient's living situation from a home-health visit chart. Choose exactly one of the options you are given.\n\nSOURCE OF TRUTH\nToday's visit notes — what the clinician observed, asked, and was told during this visit — are the source of truth. When the visit notes conflict with the referral packet, intake form, hospital discharge summary, or any prior documentation, today's visit notes win. Older documents may fill a gap the visit notes leave silent, but they may never override the visit notes. Living situation changes often between referral and start of care (a discharge to a daughter's home, a spouse's death, a move to assisted living), so a referral-era statement is weak evidence about today.\n\nHOW TO CHOOSE\nSelect from the options exactly as they are presented to you. Never invent, merge, or reword an option, and never answer with a value that is not on the list.\n\nWeigh the evidence this way:\n- Prefer an explicit statement of who the patient lives with over an inference drawn from who was present at the visit. A person present at the visit is not necessarily a household member; a household member is not necessarily present.\n- Distinguish living with someone from receiving visits or care from someone. A caregiver, aide, or relative who \"comes daily,\" \"checks in,\" or \"stops by\" does not make the patient a co-resident; the patient may still live alone.\n- If the chart establishes the patient resides in a congregate or institutional setting with staff on site rather than a private residence, prefer the facility option over the household options, even when a relative or spouse is also described as present there.\n- When a spouse or partner is documented as living in the home, prefer the spouse option over the broader family option. Use the family option when the documented co-residents are other relatives, or when a household includes a spouse plus other relatives and the option set forces one choice toward the more general description only if no spouse-specific option fits.\n- If the chart supports two options equally and nothing breaks the tie, choose the one that the most recent documentation supports and mark confidence Low.\n\nCONFIDENCE\n- High: today's visit notes state the living situation directly, in terms that map to one option without interpretation.\n- Medium: the answer is inferred — from who is documented as living in the home, from household or environment observations, or from an older document that today's notes do not contradict.\n- Low: the chart is silent on living situation, or sources contradict each other and today's notes do not resolve it. Still select the best-supported option; do not leave the answer blank.\n\nEXPLANATION\nGive a one- or two-sentence explanation that names where in the chart the answer came from — the section, note type, or field (for example, the visit note's home environment section, the caregiver section, or the intake form) — and, when relevant, why that source outranked a conflicting one. Quote or paraphrase only the phrase that carries the decision. Do not include the patient's name, initials, address, dates of birth, or any other identifier, and do not add clinical recommendations.", + "p-ostomy-supplies-1d74ae8f": "You are filling in one question on a home-health visit chart: which ostomy supply the patient needs most this visit. You will be given the question, the exact answer options, and the patient's chart. Choose exactly one of the options you are given — never invent an option, never merge two, never answer with free text outside an option.\n\nSOURCE OF TRUTH\nToday's visit notes — what the clinician observed, measured, and was told during this visit — are the source of truth. When the visit notes disagree with the referral packet, the intake form, the plan of care, or any earlier visit, today's notes win. Use the older documents only to fill gaps the visit notes leave open, and only when nothing in today's notes contradicts them.\n\nHOW TO CHOOSE AMONG THE OPTIONS\nWork through the visit notes for what the ostomy actually needs today: the state of the peristomal skin, how the pouch is sealing, how often it is being changed, what the patient or caregiver says they are running low on or out of, and anything the clinician recommended or supplied during the visit.\n\n- Choose the pouching-system option (e.g. \"Pouches\") when today's notes point to the pouch itself as the unmet need — the supply on hand is exhausted or nearly so, the patient is reusing or improvising a pouch, or the wrong size/type is in use and a replacement is what the visit calls for.\n- Choose the seal-protection option (e.g. \"Barrier rings\") when today's notes describe leakage, undermining, a poor seal, or an uneven peristomal surface that a barrier or sealing product is the stated or evident remedy for.\n- Choose the skin-protection option (e.g. \"Skin prep wipes\") when today's notes describe peristomal skin irritation, denudement, moisture, or adhesive trauma that a skin-prep or protectant product addresses, without the pouch itself failing or running out.\n- Choose the negative option (e.g. \"None\") when today's notes affirmatively show no ostomy supply need this visit — the appliance is intact, skin is healthy, and supplies are adequately stocked — or when the patient has no ostomy at all.\n\nIf the notes support more than one, pick the need the clinician treated as most urgent this visit: an active leak or skin breakdown outranks restocking, and running out of the pouch itself outranks a comfort or convenience item. If the notes are silent, do not default to the negative option as if it were documented adequacy — silence is not evidence of no need. Pick the option best supported by whatever the chart does contain and mark confidence Low.\n\nCONFIDENCE\n- High — today's visit notes state the answer directly (the need, the product, or the supply request is documented in this visit's notes).\n- Medium — the answer is inferred from today's notes or from older chart material consistent with them, rather than stated outright.\n- Low — the chart is silent on ostomy supplies, or it contradicts itself and today's notes do not resolve the conflict.\n\nEXPLANATION\nGive a one- or two-sentence explanation that cites where in the chart the answer came from — name the section or document and the observation you relied on (for example: the visit note's skin assessment, the supply-inventory line, the caregiver's report in today's narrative). If you relied on an older document because today's notes were silent, say so. Do not quote patient identifiers, and do not include names, dates of birth, addresses, or record numbers in the explanation.", + "p-wound-status-003f262d": "Determine the status of the patient's primary wound and choose one option: Healed, Ongoing, or No wound.\n\nORDER OF EVIDENCE\nToday's visit notes are the current chart, and they govern. Read them first and answer from what they document about the wound. Use the referral packet (hospital discharge summary, intake forms, prior records) for background, or when today's visit notes say nothing about a wound at all. A referral packet is written before the visit, so when it disagrees with today's visit notes it is out of date: follow today's notes and treat the referral statement as superseded. A discharge summary that calls a wound closed or healed does not override an open area assessed today.\n\nCHOOSING THE ANSWER\n- Ongoing: today's visit notes describe a wound that is still open or still being treated - measurements, an open area, drainage, wound bed or slough description, a dressing change or wound care order, or wound care orders that continue.\n- Healed: the wound is documented as closed, resolved, or fully epithelialized with no wound care needed, and nothing in today's visit notes describes an open area, drainage, or ongoing wound treatment.\n- No wound: no wound is documented anywhere in the chart, or the chart affirmatively states the patient has no wound.\n\nCONFIDENCE\n- High: today's visit notes state the answer directly - the wound and its status are documented in today's assessment, so no inference is needed. A wound the referral packet describes does not earn High on its own.\n- Medium: the answer is inferred rather than stated - you reason to it from indirect evidence (for example a dressing supply or wound care order with no assessment of the wound itself), or today's visit notes are silent about the wound and the answer rests on the referral packet.\n- Low: the chart is silent on the wound, or it is contradictory in a way recency cannot settle - documents of the same recency disagree, or today's visit notes are internally inconsistent about whether a wound is present or closed.\n\nEXPLANATION\nAlways give a one- or two-sentence explanation that states the answer and cites where in the chart it came from: name the document or section (for example today's visit notes, the discharge summary, the intake form) and the specific finding it records. If an older document states a different status, say in the same explanation that it is outdated and has been superseded by the more recent documentation." }, "questions": { "wound-status": { "text": "What is the current status of the patient's primary wound?", "type": "single", - "livePromptId": "p-wound-status-f1305d28", + "livePromptId": "p-wound-status-003f262d", "draftPromptId": null }, "mood": { @@ -21,14 +21,14 @@ "living-situation": { "text": "What is the patient's living situation?", "type": "single", - "livePromptId": "p-living-situation-373bc0c3", + "livePromptId": "p-living-situation-90bc4192", "draftPromptId": null }, "ostomy-supplies": { "text": "Which ostomy supply does the patient need most this visit?", "type": "single", "livePromptId": null, - "draftPromptId": "p-ostomy-supplies-26bf6ae3" + "draftPromptId": "p-ostomy-supplies-1d74ae8f" } } } diff --git a/examples/prompt-lab/evidence/run/lab/gold.json b/examples/prompt-lab/evidence/run/lab/gold.json index 37ea04be..b665b27a 100644 --- a/examples/prompt-lab/evidence/run/lab/gold.json +++ b/examples/prompt-lab/evidence/run/lab/gold.json @@ -2,7 +2,7 @@ "wound-status|jordan": { "answer": "No wound", "confidence": "High", - "explanation": "The referral packet's intake form documents \"No wounds documented. Skin intact,\" and today's SOC visit notes confirm \"Skin intact, no wounds.\"" + "explanation": "The referral packet's intake form states 'No wounds documented. Skin intact,' and today's SOC visit note confirms 'Skin intact, no wounds.'" }, "wound-status|pat": { "answer": "Ongoing", @@ -12,6 +12,11 @@ "wound-status|riley": { "answer": "No wound", "confidence": "High", - "explanation": "The referral packet states 'No wounds' alongside the right wrist fracture with cast, and today's SOC visit notes confirm 'No wounds; skin intact around cast.'" + "explanation": "The referral packet states 'No wounds' alongside the right wrist fracture, and today's SOC visit notes confirm 'No wounds; skin intact around cast.'" + }, + "wound-status|roderick": { + "answer": "Ongoing", + "confidence": "Medium", + "explanation": "The referral packet documents no wound status (peristomal skin intact at discharge, no statement of closure or healing), so the visit notes govern: today's assessment finds a well-demarcated area of erythema with superficial moist denudation at the 5-to-7 o'clock peristomal margin, approximately 2 cm at its widest and tender to touch. Confidence is Medium rather than High because the referral does not state the status." } } diff --git a/examples/prompt-lab/evidence/run/lab/queue/issues.json b/examples/prompt-lab/evidence/run/lab/queue/issues.json index 788b27a4..8f3a7a58 100644 --- a/examples/prompt-lab/evidence/run/lab/queue/issues.json +++ b/examples/prompt-lab/evidence/run/lab/queue/issues.json @@ -1,6 +1,6 @@ [ { - "id": "config-sunrise-wound-status-8a366995", + "id": "config-sunrise-wound-status-9c1634e2", "kind": "config-send", "questionIds": [ "wound-status" @@ -14,7 +14,6 @@ "confidence": "High", "explanation": "Today's visit notes describe an open 2.0 x 1.5 cm left heel wound with serous drainage and slough; the discharge summary's closed status is outdated." } - }, - "proposedPrompt": "Determine the status of the patient's primary wound.\n\nToday's visit notes are the most current record of the wound and outrank the referral packet whenever the two disagree. A referral packet or discharge summary describes the wound as of discharge; it can be out of date by the time of the visit.\n\n1. Read today's visit notes first. If they describe the wound — an open area, measurements, drainage, slough or other wound bed findings, or a dressing change — decide from those findings alone. Any wound that is still open, draining, or being dressed is Ongoing, even if the referral packet or discharge summary says the wound is closed or healed. Answer Healed only when today's notes state the wound is closed, resolved, or intact.\n2. Use the referral packet only when today's notes say nothing about the wound. Then follow it: a discharge summary stating the wound is closed or healed means Healed.\n3. Answer No wound only when neither today's notes nor the referral packet documents any wound.\n\nConfidence: use High when the source you relied on states the status directly — including when today's visit notes document the wound firsthand. A conflict with an older referral does not lower confidence; today's notes settle it. Use Medium when the status must be inferred from indirect or incomplete findings, and Low when no source addresses the wound clearly.\n\nIn the rationale, cite the specific findings you relied on, and when they contradict the referral packet, say so and note that the referral's status is outdated." + } } ] diff --git a/examples/prompt-lab/evidence/run/lab/queue/patient-briefs.json b/examples/prompt-lab/evidence/run/lab/queue/patient-briefs.json index 82478cd1..88630efe 100644 --- a/examples/prompt-lab/evidence/run/lab/queue/patient-briefs.json +++ b/examples/prompt-lab/evidence/run/lab/queue/patient-briefs.json @@ -3,7 +3,7 @@ "id": "gap-ostomy-supplies-soc", "questionId": "ostomy-supplies", "visitType": "soc", - "brief": "No shelf patient has any ostomy documentation, so none of the supply answer paths can be exercised. The shelf needs a fabricated post-colostomy patient whose referral packet describes a recent colorectal surgery with a new stoma and a starter supply kit sent home, and whose visit notes document the stoma and peristomal skin in enough detail to force a choice among the supply options: peristomal skin excoriated and weeping with the pouch seal lifting after a few hours, a nearly empty box of pouches on the counter, and no barrier rings in the home. A second variant with a well-healed mature stoma, intact peristomal skin, and a full supply of everything would exercise the 'None' path.", + "brief": "No shelf patient has an ostomy of any kind — no stoma, no pouching system, no peristomal skin documentation anywhere in the referral packets or visit notes, so every answer option is unsupported including 'None'. The shelf needs an older adult recently discharged after colorectal surgery with a new colostomy, whose referral packet lists the pouching system and wafer size supplied at discharge, and whose SOC visit note documents peristomal skin condition, current pouch wear time and any leakage, remaining supply counts on hand at home, and the caregiver or self-care arrangement for pouch changes — enough detail that a reader must choose between a supply the patient is short of and an adequately stocked 'None'.", "from": "planner", "status": "locked" } diff --git a/examples/prompt-lab/evidence/run/lab/shelf/roderick.json b/examples/prompt-lab/evidence/run/lab/shelf/roderick.json new file mode 100644 index 00000000..66ae0f0c --- /dev/null +++ b/examples/prompt-lab/evidence/run/lab/shelf/roderick.json @@ -0,0 +1,10 @@ +{ + "id": "roderick", + "label": "Roderick", + "ageBand": "70-79", + "visitType": "soc", + "referral": "REFERRAL PACKET — discharging hospital, ostomy nurse summary and surgical practice orders.\n\nReason for home health: older adult male, 70-79 age band, discharged home following open low anterior resection for sigmoid colon adenocarcinoma with creation of a diverting end colostomy, left lower quadrant. Ostomy created during the index admission. Post-operative deconditioning; skilled nursing for new ostomy management, teaching and assessment.\n\nSecondary diagnoses: type 2 diabetes mellitus, non-insulin, managed on metformin; hypertension; chronic kidney disease stage 3a, mild; osteoarthritis of both hands with reduced fine-motor dexterity; post-operative deconditioning. Vision: presbyopia corrected with over-the-counter readers; patient reports he cannot see the stoma clearly without them.\n\nLiving situation per intake: lives alone in a single-story rental. One adult child lives roughly forty minutes away and visits on weekends only. No paid caregiver in place at the time of start of care. Independent with ambulation, uses a rolling walker indoors.\n\nSTOMA DESCRIPTION AT DISCHARGE (inpatient WOC nurse): end colostomy, left lower quadrant, matured. Beefy red. Budded approximately one-half inch above skin level. Stoma diameter measured at one and one-quarter inches. A mild parastomal skin fold noted on the inferior aspect. Peristomal skin intact at discharge — no erythema, no denudation, no breakdown documented.\n\nPOUCHING SYSTEM SUPPLIED AT DISCHARGE: two-piece system. Two and one-quarter inch flat cut-to-fit skin barrier wafer. Drainable opaque pouch with integrated closure. Coupling ring size two and one-quarter inch.\n\nQUANTITY SUPPLIED AT DISCHARGE: 10 wafers; 10 drainable pouches; 1 tube stoma paste; 1 box of 30 skin barrier wipes; 1 bottle adhesive remover spray; 1 box pouch deodorant drops.\n\nDISCHARGE INSTRUCTIONS: change wafer every three to four days, or sooner if leakage occurs. Empty pouch when one-third full. Cut the wafer opening to the measured stoma diameter. A thirty-day resupply order was placed with the DME vendor prior to discharge and is expected to arrive at the home.\n\nEXPLICITLY DOCUMENTED AS NOT ISSUED: no convex barrier or convex insert of any kind; no moldable or flat barrier rings; no ostomy belt. The flat cut-to-fit wafer was judged appropriate for this stoma at discharge on the basis of a well-budded stoma and intact peristomal skin.\n\nSELF-CARE STATUS PER REFERRAL: independent with pouch change after teaching; return demonstration completed one time prior to discharge. The adult child was taught the pouch change once at the bedside.\n\nNo home health aide authorized at referral.", + "notes": "START OF CARE VISIT NOTE — skilled nursing, ostomy assessment and return demonstration.\n\nPatient met at home, alone, ambulating with rolling walker. Oriented, cooperative, wearing over-the-counter readers intermittently.\n\nSTOMA: end colostomy, left lower quadrant. Viable, pink-red, no necrosis, no bleeding at the mucocutaneous junction, no separation. Budding is now only about one-eighth to one-quarter inch above skin level. On sitting and on forward bending the stoma retracts slightly below skin level. Stoma measured today at one and one-eighth inches — post-operative edema has resolved and the stoma has shrunk from the diameter documented in the referral packet. The inferior parastomal skin fold noted by the inpatient WOC nurse is still present and is more pronounced when the patient is seated.\n\nPERISTOMAL SKIN: a well-demarcated area of erythema with superficial moist denudation along the inferior margin, from roughly the five o'clock to the seven o'clock position, approximately two centimeters at its widest point. Tender to light touch. No induration. No odor. No satellite lesions. No purulence. No white plaques or pustules. Appearance is consistent with irritant contact dermatitis from effluent tracking under the wafer, not infection. Peristomal skin at all other clock positions is intact.\n\nWEAR TIME AND LEAKAGE: patient reports the wafer lifts at the inferior edge and that he is getting leakage every day or two. Observed and reported actual wear time is twenty-four to thirty-six hours before the wafer is changed, against the referral's prescribed three to four days. He has been reinforcing the lifting inferior edge with household tape from a kitchen drawer. Effluent soft to pasty. Patient empties the pouch four to six times per day.\n\nCUT-TO-FIT SIZING OBSERVED THIS VISIT: patient is still cutting the wafer opening to the one and one-quarter inch measurement recorded in the discharge packet. With the barrier in place there is a visible gap of bare peristomal skin between the stoma base and the cut edge of the barrier along the inferior margin — the same location as the skin breakdown described above.\n\nSUPPLIES ON HAND — COUNTED BY THIS CLINICIAN AT THE HOME, NOT PATIENT-REPORTED: wafers, 6 remaining. Drainable pouches, 7 remaining. Stoma paste, one tube approximately half full. Skin barrier wipes, 22 remaining. Adhesive remover spray, one bottle nearly full. Pouch deodorant drops, one box nearly full. The thirty-day DME resupply shipment described in the referral has NOT arrived at the home. Patient has no tracking information, has not contacted the vendor, and does not know the vendor's name.\n\nBARRIER RING AND CONVEXITY STATUS: no barrier rings of any kind present in the home. No convex barrier, convex insert or convex wafer present in the home. No ostomy belt present in the home. None ordered. None ever issued. Patient confirms he has never been shown any of these products and the home search of his supply bin confirms the same.\n\nTECHNIQUE — RETURN DEMONSTRATION TODAY: patient performed a full pouch change with observation. He is unable to hold the wafer taut with one hand while pressing the inferior skin fold flat with the other; he describes hand pain and is observed to lose his grip on the barrier twice. He does not flatten or tuck the inferior parastomal fold before applying the barrier. He does not use the mirror over the sink. He performs the change standing at the sink — there is no seated surface at counter height in the bathroom — and states that standing through the change is tiring.\n\nCAREGIVER ARRANGEMENT: patient lives alone. Adult child available weekends only, roughly forty minutes away, was taught the change once in the hospital. No home health aide authorized. Patient therefore performs six of seven pouch changes himself, unassisted, standing.\n\nNo fever. No abdominal distension. No report of obstructive symptoms. Abdomen soft, incision clean and dry, staples intact.", + "locked": true, + "source": "invented" +} diff --git a/examples/prompt-lab/evidence/run/lab/shelf/verna.json b/examples/prompt-lab/evidence/run/lab/shelf/verna.json deleted file mode 100644 index b59d649e..00000000 --- a/examples/prompt-lab/evidence/run/lab/shelf/verna.json +++ /dev/null @@ -1,10 +0,0 @@ -{ - "id": "verna", - "label": "Verna", - "ageBand": "60-69", - "visitType": "soc", - "referral": "REFERRAL PACKET (from the surgical service)\n\nDISCHARGE SUMMARY. Sigmoid adenocarcinoma, status post open sigmoid colectomy with permanent end colostomy, about three weeks ago. Uneventful post-operative course: no anastomotic leak, no wound complication, no ileus; tolerated diet advancement; ambulating independently at discharge. Post-operative anemia, stable, on oral iron.\n\nOTHER ACTIVE DIAGNOSES. Type 2 diabetes mellitus, with noted skin fragility and slow healing. Hypertension. Obesity with a pendulous lower abdomen and a skin crease below the stoma. Osteoarthritis of the hands limiting fine dexterity.\n\nSTOMA AT DISCHARGE (inpatient WOC nurse). End colostomy, left lower quadrant. Beefy red, budded 2 cm, 32 mm round. Peristomal skin intact. Output pasty to formed. Pouching schedule: change every 3 days.\n\nTEACHING. One supervised pouch change with return demonstration, charted as partially independent. Written instructions given.\n\nSTARTER KIT CHECKLIST (sent home at discharge). 10 one-piece drainable pouches with a pre-cut 32 mm barrier; skin barrier wipes; adhesive remover wipes; deodorant drops; barrier rings x10.\n\nSOCIAL. Lives alone in a one-bedroom second-floor walk-up, no elevator. One adult child works full time and visits on weekends. No home aide authorized.\n\nORDERS. Skilled nursing 2x/week for ostomy care and teaching. WOC consult PRN. DME resupply order placed with the DME supplier; no confirmed delivery.", - "notes": "VISIT NOTES (this visit, start of care; about three weeks post-op)\n\nSTOMA. End colostomy, left lower quadrant. Now budded only approximately 0.5 cm and flush-to-retracted at the 4 o'clock position as post-operative edema has resolved. Measures approximately 25 mm and is slightly oval rather than round. When she sits, the stoma sits at the lip of the abdominal crease.\n\nPERISTOMAL SKIN. Denuded and excoriated, weeping serous moisture across roughly the inferior half of the wafer field, 2-4 cm, with an irregular scalloped border tracking exactly where effluent has run. She reports burning. No satellite lesions. No induration. No purulence. No fever.\n\nPOUCH IN PLACE. On approximately 5 hours; the adhesive is already lifting along the inferior border with effluent visible under the edge. She has been taping the edge down with paper tape. Reports two to three leaks a day and has been changing daily instead of every 3 days.\n\nOUTPUT. Looser and higher in volume than the packet describes; no longer pasty to formed.\n\nSELF-CARE. Changes the pouch standing at the bathroom sink, no mirror. Cannot see the inferior edge of the stoma over her abdomen when seated. Hand arthritis limits fine cutting and trimming.\n\nSUPPLY INVENTORY, counted with the patient this visit, item by item. 2 one-piece drainable pouches left in the box on the counter; their pre-cut 32 mm opening now leaves 3-4 mm of bare skin exposed around the 25 mm stoma. Skin barrier wipes nearly full. Adhesive remover wipes full. Deodorant drops unopened. NO barrier rings anywhere: cabinet, supply bag and counter searched, and she states she never saw any come home with her. No barrier powder. No convex product. No belt. No DME delivery received since discharge.", - "locked": true, - "source": "invented" -} diff --git a/examples/prompt-lab/evidence/run/lab/targets.json b/examples/prompt-lab/evidence/run/lab/targets.json index e2cace2d..fa549dfe 100644 --- a/examples/prompt-lab/evidence/run/lab/targets.json +++ b/examples/prompt-lab/evidence/run/lab/targets.json @@ -2,32 +2,32 @@ "sunrise|living-situation|jordan": { "answer": "Lives with spouse", "confidence": "High", - "explanation": "Today's skilled nursing SOC visit notes state directly that the patient and their spouse share the home, with the spouse present throughout the visit and setting up the pill organizer each morning. The intake form's 'lives alone' entry is older supporting evidence and is treated as stale under today's direct observation." + "explanation": "Today's visit note states the patient and spouse share the home and the spouse sets up the pill organizer each morning, which directly establishes co-residence with a spouse; this outranks the intake form's referral-era \"lives alone\" entry, since today's visit notes take precedence over prior documentation." }, "sunrise|living-situation|pat": { "answer": "Lives alone", "confidence": "High", - "explanation": "Today's skilled nursing SOC visit note states directly that the patient lives alone in a one-story home, with a neighbor who checks in daily; an intermittent visitor does not count as a cohabitant. This is consistent with the referral packet's intake form, but the current visit note is the direct basis." + "explanation": "Today's visit note's home environment section states the patient \"lives alone in a one-story home; neighbor checks in daily,\" which directly names the living situation and outranks the referral-era intake form. The neighbor's daily check-ins are visits rather than co-residence." }, "sunrise|living-situation|riley": { "answer": "Lives with family", "confidence": "High", - "explanation": "Today's visit notes state directly that the patient lives with her adult child, that child's spouse, and two grandchildren, which is a household of relatives other than a spouse. The referral packet agrees, but the visit note is the direct current statement; 'Lives alone' and the spouse option were rejected because multiple relatives reside in the home." + "explanation": "Today's visit note states the patient lives with her daughter, son-in-law and two grandchildren, which maps directly to the family option; the referral packet's co-resident statement agrees and no spouse is documented in the home." }, "sunrise|mood|jordan": { "answer": "Calm", "confidence": "Medium", - "explanation": "The visit notes describe the patient as \"relaxed and engaged\" and asking good questions about the diuretic, but the mood is inferred from the clinician's behavioral description rather than stated by the patient in their own words." + "explanation": "The visit notes describe the patient as \"relaxed and engaged, asks good questions about the diuretic,\" which is the clinician's observation of behavior rather than the patient stating their feelings in their own words." }, "sunrise|mood|pat": { "answer": "Calm", "confidence": "High", - "explanation": "The visit notes describe the patient as \"calm and pleasant, joking about hospital food,\" and she states in her own words that she is glad to be home." + "explanation": "Visit notes describe the patient as \"calm and pleasant, joking about hospital food,\" and she states in her own words that she is glad to be home." }, "sunrise|mood|riley": { "answer": "Anxious", "confidence": "High", - "explanation": "The patient states in her own words, 'I can't stop thinking about falling again, I lie awake worrying about it,' and the clinician notes she wrings her hands during the medication review." + "explanation": "The patient states in her own words, 'I can't stop thinking about falling again, I lie awake worrying about it,' and the nurse notes she wrings her hands during the medication review." }, "sunrise|wound-status|pat": { "answer": "Ongoing", @@ -37,11 +37,11 @@ "sunrise|wound-status|jordan": { "answer": "No wound", "confidence": "High", - "explanation": "The referral packet's intake form documents \"No wounds documented. Skin intact,\" and today's SOC visit notes confirm \"Skin intact, no wounds.\"" + "explanation": "The referral packet's intake form documents no wounds and intact skin, and today's SOC visit note confirms 'Skin intact, no wounds.'" }, "sunrise|wound-status|riley": { "answer": "No wound", "confidence": "High", - "explanation": "The referral packet states \"No wounds\" and today's SOC visit notes confirm \"No wounds; skin intact around cast,\" so there is no primary wound to status." + "explanation": "The referral packet states 'No wounds' with the presenting problem being a right wrist fracture in a cast, and today's SOC visit notes confirm 'No wounds; skin intact around cast.'" } } diff --git a/examples/prompt-lab/evidence/run/lab/work/patients/gap-ostomy-supplies-soc/07f91c04/plan.json b/examples/prompt-lab/evidence/run/lab/work/patients/gap-ostomy-supplies-soc/07f91c04/plan.json new file mode 100644 index 00000000..517232bc --- /dev/null +++ b/examples/prompt-lab/evidence/run/lab/work/patients/gap-ostomy-supplies-soc/07f91c04/plan.json @@ -0,0 +1,4 @@ +{ + "brief": "No shelf patient has an ostomy of any kind — no stoma, no pouching system, no peristomal skin documentation anywhere in the referral packets or visit notes, so every answer option is unsupported including 'None'. The shelf needs an older adult recently discharged after colorectal surgery with a new colostomy, whose referral packet lists the pouching system and wafer size supplied at discharge, and whose SOC visit note documents peristomal skin condition, current pouch wear time and any leakage, remaining supply counts on hand at home, and the caregiver or self-care arrangement for pouch changes — enough detail that a reader must choose between a supply the patient is short of and an adequately stocked 'None'.", + "plan": "INVENTED SHELF PATIENT — first-name label only: \"Roderick\". No dates, no MRN, no phone, no address, no surname, no facility names beyond generic role words (\"the discharging hospital\", \"the surgical practice\").\n\nAGE BAND & DEMOGRAPHICS\nOlder adult, 70–79 age band. Male. Lives alone in a single-story rental; one adult child lives roughly 40 minutes away and visits on weekends only. No paid caregiver at the time of start of care.\n\nDIAGNOSES\nPrimary: sigmoid colon adenocarcinoma, status post open low anterior resection with diverting end colostomy (left lower quadrant), ostomy created during the index admission. Secondary: type 2 diabetes mellitus (non-insulin, metformin), hypertension, mild chronic kidney disease stage 3a, osteoarthritis of both hands with reduced fine-motor dexterity, and post-operative deconditioning. Vision: presbyopia corrected with over-the-counter readers; reports difficulty seeing the stoma clearly without them.\n\nLIVING SITUATION / SELF-CARE ARRANGEMENT\nLives alone, independent with ambulation using a rolling walker indoors. Bathroom has a mirror over the sink but no seated surface at counter height; Roderick performs pouch changes standing at the sink, which he reports is tiring. The adult child was taught the pouch change once in the hospital but is present only on weekends. No home-health aide yet authorized. This matters to the question because it establishes that Roderick is doing his own changes six days out of seven with arthritic hands, so a supply that compensates for poor dexterity or poor seal is not a luxury item.\n\nWHAT THE REFERRAL PACKET SAYS (hospital discharge / ostomy nurse summary)\n- Stoma: end colostomy, left lower quadrant, matured, described at discharge as beefy red, budded approximately 1/2 inch above skin level, stoma diameter measured at 1 1/4 inches, with a mild parastomal skin fold on the inferior aspect noted by the inpatient WOC nurse.\n- Pouching system supplied at discharge: two-piece system, 2 1/4 inch flat cut-to-fit skin barrier wafer, drainable opaque pouch with integrated closure, coupling ring size 2 1/4 inch.\n- Quantity supplied at discharge: 10 wafers, 10 drainable pouches, 1 tube stoma paste, 1 box (30) skin barrier wipes, 1 bottle adhesive remover spray, 1 box of pouch deodorant drops.\n- Discharge instructions state: change wafer every 3–4 days or sooner if leakage; empty pouch when one-third full; a 30-day resupply order was \"placed with the DME vendor prior to discharge\" and is expected to arrive at the home.\n- Referral packet explicitly documents NO convexity, NO barrier rings, and NO ostomy belt were issued, and states peristomal skin was intact at discharge.\n- Referral lists the patient as \"independent with pouch change after teaching, return demonstration completed x1.\"\n\nWHAT TODAY'S SOC VISIT NOTE MUST SHOW\n1. Peristomal skin: a well-demarcated area of erythema with superficial moist denudation along the inferior 5 o'clock to 7 o'clock margin of the peristomal skin, approximately 2 cm at its widest, tender, no induration, no odor, no satellite lesions, no purulence — i.e., irritant contact dermatitis from effluent undermining the wafer, not infection. The skin elsewhere around the stoma is intact.\n2. Stoma itself: viable, pink-red, now budded only about 1/8 to 1/4 inch and retracting slightly below skin level when Roderick sits or bends forward. Measured today at 1 1/8 inches — the post-operative edema has resolved and the stoma has shrunk from the discharge measurement.\n3. Wear time and leakage: Roderick reports the wafer is lifting at the inferior edge and he is getting leakage \"every day or two,\" with actual wear time about 24–36 hours rather than the 3–4 days the referral prescribes. He has been reinforcing the lifting edge with household tape. Effluent is soft/pasty, 4–6 pouch emptyings per day.\n4. Cut-to-fit sizing observed at the visit: he is still cutting the wafer opening to the 1 1/4 inch discharge measurement, leaving a visible gap of bare skin between the stoma and the barrier edge at the inferior margin — the exact location of the skin breakdown.\n5. Supply counts on hand at home, counted by the clinician and documented as counted (not reported): wafers 6 remaining, drainable pouches 7 remaining, stoma paste tube about half full, skin barrier wipes 22 remaining, adhesive remover spray nearly full, deodorant drops nearly full. The DME resupply shipment has NOT arrived; Roderick has no tracking information and has not called the vendor.\n6. Technique observation: return demonstration today shows he cannot comfortably hold the wafer taut and press the inferior fold flat at the same time because of hand arthritis; he does not use a mirror; he does not currently tuck or flatten the parastomal fold before applying.\n7. Barrier ring / convexity status documented explicitly as: none in the home, none ordered, never issued.\n8. Caregiver arrangement: adult child available weekends only; no aide authorized; Roderick performs six of seven changes alone standing at the sink.\n\nTHE DELIBERATE CONFLICT BETWEEN REFERRAL AND NOTES (this is what makes the question answerable and non-trivial)\nThe referral packet and the visit note disagree on three axes, and the disagreement is the test:\n- SIZE CONFLICT: referral documents a 1 1/4 inch stoma and supplies 2 1/4 inch wafers; today's measurement is 1 1/8 inch and shrinking, and he is still cutting to the stale referral number. A reader who trusts the referral alone will conclude the sizing is correct and see no problem.\n- CONTOUR CONFLICT: referral says \"peristomal skin intact, no convexity needed, flat wafer appropriate.\" Today's note documents a stoma that has retracted to near skin level over an inferior parastomal fold, with effluent undermining the flat wafer at exactly that fold — the classic indication for a convex barrier or a moldable barrier ring plus belt. The referral's \"flat is fine\" judgment was true at discharge and is false now.\n- SUPPLY-COUNT CONFLICT: the referral asserts a 30-day resupply was ordered and is en route, which invites the reader to answer \"None — he's stocked.\" The counted inventory shows 6 wafers and 7 pouches, and at his ACTUAL 24–36 hour wear time that is roughly 6–7 days of supply, not 30 — while at the referral's PRESCRIBED 3–4 day wear time the same 6 wafers would read as ~3 weeks and look adequate. The same number is either comfortable or alarming depending on which document the reader believes.\n\nHOW THIS FORCES THE QUESTION\n\"Which ostomy supply does the patient need most this visit?\" now has a defensible best answer (a barrier ring or convex barrier — the item that would stop the leak at the inferior fold and let the peristomal dermatitis heal; it is documented as absent, never issued, and directly causative), with three well-baited near-misses that the packet actively supports: (a) more wafers, because the counted inventory really is low and the promised shipment really has not arrived; (b) an ostomy belt, defensible as an adjunct but useless without something to seal the fold; (c) \"None — adequately stocked,\" which is exactly what the referral's 30-day-resupply line and 'skin intact' line would lead a referral-only reader to pick. A reader must reconcile a stale referral against a fresh observation, and must convert a raw supply count into days-on-hand using the observed wear time rather than the prescribed one, before the answer separates from the distractors.\n\nWhat this patient deliberately does NOT contain: no urostomy or ileostomy effluent details (keeps the colostomy unambiguous), no fungal/candidal findings (would redirect the answer to antifungal powder and muddy the supply question), no stoma necrosis or bleeding (would make the answer 'call the surgeon' rather than a supply), and no existing barrier ring anywhere in the home (so 'he already has one' is never a valid out)." +} diff --git a/examples/prompt-lab/evidence/run/lab/work/patients/gap-ostomy-supplies-soc/fbf53055/plan.json b/examples/prompt-lab/evidence/run/lab/work/patients/gap-ostomy-supplies-soc/fbf53055/plan.json deleted file mode 100644 index a27b7275..00000000 --- a/examples/prompt-lab/evidence/run/lab/work/patients/gap-ostomy-supplies-soc/fbf53055/plan.json +++ /dev/null @@ -1,4 +0,0 @@ -{ - "brief": "No shelf patient has any ostomy documentation, so none of the supply answer paths can be exercised. The shelf needs a fabricated post-colostomy patient whose referral packet describes a recent colorectal surgery with a new stoma and a starter supply kit sent home, and whose visit notes document the stoma and peristomal skin in enough detail to force a choice among the supply options: peristomal skin excoriated and weeping with the pouch seal lifting after a few hours, a nearly empty box of pouches on the counter, and no barrier rings in the home. A second variant with a well-healed mature stoma, intact peristomal skin, and a full supply of everything would exercise the 'None' path.", - "plan": "INVENTED SHELF PATIENT — VARIANT A (the load-bearing one): first-name label \"Verna\". Age band 60-69. No surname, no dates, no MRN, no phone, no address, no facility or clinician names anywhere — the referral packet is from \"the surgical service\" and \"the inpatient WOC nurse\", supplies come from \"the DME supplier\", and all timing is relative (\"about three weeks post-op\", \"since discharge\", \"this visit\").\n\nDIAGNOSES: sigmoid adenocarcinoma s/p open sigmoid colectomy with permanent end colostomy; type 2 diabetes (skin fragility, slow healing); hypertension; obesity with a pendulous lower abdomen and a skin crease below the stoma; osteoarthritis of the hands limiting fine dexterity; post-op anemia.\n\nLIVING SITUATION: lives alone in a one-bedroom second-floor walk-up, no elevator. One adult child works full time and visits on weekends. No home aide authorized. Changes the pouch standing at the bathroom sink, no mirror, cannot see the inferior edge of the stoma over her abdomen when seated.\n\nWHAT THE REFERRAL PACKET SAYS: discharge summary describing an uneventful post-op course; stoma at discharge \"beefy red, budded 2 cm, 32 mm round, peristomal skin intact\"; output \"pasty to formed\"; one supervised pouch change with return demonstration charted as \"partially independent\"; pouching schedule of every 3 days. Starter-kit checklist: 10 one-piece drainable pouches with a PRE-CUT 32 mm barrier, skin barrier wipes, adhesive remover wipes, deodorant drops, and — this is the trap — \"barrier rings x10\". Orders: skilled nursing 2x/week for ostomy care and teaching, WOC consult PRN, DME resupply order placed with no confirmed delivery.\n\nWHAT TODAY'S VISIT NOTES MUST SHOW: stoma now budded only ~0.5 cm and flush-to-retracted at the 4 o'clock position as post-op edema resolved; measures ~25 mm and slightly oval; sits at the lip of the abdominal crease when she sits. Peristomal skin denuded, excoriated and weeping serous moisture across roughly the inferior half of the wafer field, 2-4 cm, with an irregular scalloped border tracking exactly where effluent ran; she reports burning. Explicitly NO satellite lesions, no induration, no purulence, no fever — so this reads as irritant contact dermatitis from effluent, not infection or fungus. Pouch has been on ~5 hours and the adhesive is already lifting along the inferior border with effluent visible under the edge; she has been taping it down with paper tape. Two to three leaks a day; she is changing daily instead of every 3 days. Output looser and higher-volume than the packet describes. Inventory counted WITH the patient and documented item by item: 2 pouches left in the box on the counter, and their pre-cut 32 mm opening now leaves 3-4 mm of bare skin around a 25 mm stoma; barrier wipes nearly full; adhesive remover full; deodorant drops unopened; NO barrier rings anywhere — cabinet, bag and counter searched; no barrier powder, no convex product, no belt.\n\nINTENDED ANSWER: a moldable barrier ring/seal — it fills the crease and the oversized opening, stops effluent contact with denuded skin, and restores wear time. It is the only supply on the shelf that addresses the mechanism rather than a symptom.\n\nDELIBERATE DISTRACTORS (so the question is a real choice, not a lookup): (1) the nearly empty pouch box makes \"pouches\" look urgent, but more pouches at the wrong size leak the same way — resupply is a workflow action, not this visit's supply need; (2) weeping skin invites \"barrier powder/crusting\", but the notes make ongoing effluent contact explicit, so protection beats treatment; (3) a near-flush stoma invites \"convex wafer\" or \"belt\", but neither is in the home, no WOC convexity assessment has happened, and convexity on denuded skin without a ring first is second-line; (4) an answer path that defers to \"WOC referral\" instead of naming a supply should visibly under-answer, since the packet already carries a PRN consult.\n\nDELIBERATE CONFLICTS BETWEEN PACKET AND NOTES, each testing one thing: (C1 inventory) the packet checklist says 10 barrier rings went home; the in-home count finds zero and she says she never saw any — tests whether the answer trusts today's observation over discharge paperwork, and a packet-trusting path lands on \"pouches\" or \"None\". (C2 sizing) 32 mm round in the packet vs ~25 mm oval today, with pre-cut wafers still cut to the discharge size — tests whether the model connects resolving edema to the leak mechanism instead of blaming her technique. (C3 skin) \"peristomal skin intact\" and q3-day changes in the packet vs denuded skin and daily changes today — tests recency. (C4 output) formed/pasty in the packet vs looser and higher-volume today — a quiet detail that makes leakage more corrosive and further undercuts \"just reorder pouches\".\n\nVARIANT B — THE 'NONE' PATH: first-name label \"Roland\", age band 70-79. Mature end colostomy of several years after colectomy for diverticular disease; lives with a spouse who does the changes competently. The referral packet is for something else entirely — skilled nursing after a heart-failure hospitalization for medication management and teaching — and notes the colostomy as long-standing and self-managed, with a stale discharge-planner line claiming \"patient reports running out of pouches\" so the question must still be answered rather than skipped. Visit notes: stoma mature, budded 2 cm, beefy red, on a flat surface, 28 mm and unchanged for years; peristomal skin fully intact, no erythema or denudement; two-piece system worn 4 days with an intact seal, no leaks this month or historically; supply closet inventoried with the patient — 30+ pouches, 20+ wafers already cut to size, a full box of barrier rings, unopened barrier powder, adhesive remover, an unused belt; monthly DME auto-ship arrived complete. Conflict for B mirrors A's in the opposite direction: the stale \"running out of pouches\" note contradicts the counted inventory. That symmetry is the point — the same rule (observed notes over packet) yields \"barrier ring\" in A and \"None\" in B, so neither shortcut (\"always name a supply\", \"always trust the packet\") can pass both.\n\nSHELF MECHANICS: keep packet and visit notes as two separately quotable documents in the shelf's existing envelope shape — the conflicts only test anything if an answer path can cite one against the other. Measurements and counts are the load-bearing detail and must stay exact; everything identifying stays absent." -} diff --git a/examples/prompt-lab/evidence/run/lab/work/q-wound-status/205c23a4/grid.json b/examples/prompt-lab/evidence/run/lab/work/q-wound-status/205c23a4/grid.json new file mode 100644 index 00000000..7d976978 --- /dev/null +++ b/examples/prompt-lab/evidence/run/lab/work/q-wound-status/205c23a4/grid.json @@ -0,0 +1,74 @@ +[ + { + "questionId": "wound-status", + "patientId": "jordan", + "pile": "shared", + "highlight": [], + "ai": { + "answer": "No wound", + "confidence": "High", + "explanation": "The referral packet's intake form states 'No wounds documented. Skin intact,' and today's SOC visit note confirms 'Skin intact, no wounds.'" + }, + "target": { + "answer": "No wound", + "confidence": "High", + "explanation": "The referral packet's intake form states 'No wounds documented. Skin intact,' and today's SOC visit note confirms 'Skin intact, no wounds.'" + }, + "notes": "" + }, + { + "questionId": "wound-status", + "patientId": "pat", + "pile": "shared", + "highlight": [ + "disagrees with the config-level target" + ], + "ai": { + "answer": "Healed", + "confidence": "High", + "explanation": "Per the instruction to defer to the referral packet, the hospital discharge summary states the left heel pressure injury was closed, which the instructions treat as a High-confidence Healed determination. Note the conflict: today's SOC visit documents an open left heel area 2.0 x 1.5 cm with moderate serous drainage and yellow slough, so this answer follows the stated rule rather than the current chart." + }, + "target": { + "answer": "Ongoing", + "confidence": "High", + "explanation": "Today's visit notes describe an open 2.0 x 1.5 cm left heel wound with serous drainage and slough; the discharge summary's closed status is outdated." + }, + "notes": "" + }, + { + "questionId": "wound-status", + "patientId": "riley", + "pile": "shared", + "highlight": [], + "ai": { + "answer": "No wound", + "confidence": "High", + "explanation": "The referral packet states 'No wounds' alongside the right wrist fracture, and today's SOC visit notes confirm 'No wounds; skin intact around cast.'" + }, + "target": { + "answer": "No wound", + "confidence": "High", + "explanation": "The referral packet states 'No wounds' alongside the right wrist fracture, and today's SOC visit notes confirm 'No wounds; skin intact around cast.'" + }, + "notes": "" + }, + { + "questionId": "wound-status", + "patientId": "roderick", + "pile": "shared", + "highlight": [ + "medium confidence" + ], + "ai": { + "answer": "Ongoing", + "confidence": "Medium", + "explanation": "The referral packet documents no wound status (peristomal skin intact at discharge, no statement of closure or healing), so the visit notes govern: today's assessment finds a well-demarcated area of erythema with superficial moist denudation at the 5-to-7 o'clock peristomal margin, approximately 2 cm at its widest and tender to touch. Confidence is Medium rather than High because the referral does not state the status." + }, + "target": { + "answer": "Ongoing", + "confidence": "Medium", + "explanation": "The referral packet documents no wound status (peristomal skin intact at discharge, no statement of closure or healing), so the visit notes govern: today's assessment finds a well-demarcated area of erythema with superficial moist denudation at the 5-to-7 o'clock peristomal margin, approximately 2 cm at its widest and tender to touch. Confidence is Medium rather than High because the referral does not state the status." + }, + "notes": "" + } +] diff --git a/examples/prompt-lab/evidence/run/lab/work/q-wound-status/205c23a4/score.json b/examples/prompt-lab/evidence/run/lab/work/q-wound-status/205c23a4/score.json new file mode 100644 index 00000000..5f5f0c9b --- /dev/null +++ b/examples/prompt-lab/evidence/run/lab/work/q-wound-status/205c23a4/score.json @@ -0,0 +1,70 @@ +{ + "promptId": "p-wound-status-003f262d", + "menus": [ + { + "options": [ + "Healed", + "Ongoing", + "No wound" + ], + "agencies": [ + "harbor", + "maple", + "sunrise" + ], + "score": { + "rows": [ + { + "patientId": "jordan", + "gold": "No wound", + "rerun": "No wound", + "result": "worked" + }, + { + "patientId": "pat", + "gold": "Ongoing", + "rerun": "Ongoing", + "result": "worked" + }, + { + "patientId": "riley", + "gold": "No wound", + "rerun": "No wound", + "result": "worked" + }, + { + "patientId": "roderick", + "gold": "Ongoing", + "rerun": "No wound", + "result": "did not" + } + ], + "golded": 4, + "worked": 3, + "percent": 75 + }, + "outputs": { + "jordan": { + "answer": "No wound", + "confidence": "High", + "explanation": "Today's visit notes from the skilled nursing SOC visit document \"Skin intact, no wounds,\" and the intake form in the referral packet agrees that no wounds were documented and skin was intact." + }, + "pat": { + "answer": "Ongoing", + "confidence": "High", + "explanation": "Today's skilled nursing SOC visit notes document an open left heel area measuring 2.0 x 1.5 cm with moderate serous drainage, a pink wound bed with yellow slough, and a foam dressing change per protocol. The hospital discharge summary's statement that the wound was closed is outdated and superseded by today's assessment." + }, + "riley": { + "answer": "No wound", + "confidence": "High", + "explanation": "Today's skilled nursing SOC visit notes document 'No wounds; skin intact around cast,' affirmatively stating the patient has no wound. The referral packet agrees, also noting 'No wounds.'" + }, + "roderick": { + "answer": "No wound", + "confidence": "High", + "explanation": "Today's start of care visit note documents no wound: the surgical incision is clean and dry with staples intact, and the peristomal finding is erythema with superficial moist denudation assessed as irritant contact dermatitis from effluent tracking under the wafer, not a wound. The referral packet likewise documents no wound, describing peristomal skin intact at discharge." + } + } + } + ] +} diff --git a/examples/prompt-lab/evidence/run/lab/work/q-wound-status/cd33704e/grid.json b/examples/prompt-lab/evidence/run/lab/work/q-wound-status/cd33704e/grid.json deleted file mode 100644 index d06d04c6..00000000 --- a/examples/prompt-lab/evidence/run/lab/work/q-wound-status/cd33704e/grid.json +++ /dev/null @@ -1,55 +0,0 @@ -[ - { - "questionId": "wound-status", - "patientId": "jordan", - "pile": "shared", - "highlight": [], - "ai": { - "answer": "No wound", - "confidence": "High", - "explanation": "The referral packet's intake form documents \"No wounds documented. Skin intact,\" and today's SOC visit notes confirm \"Skin intact, no wounds.\"" - }, - "target": { - "answer": "No wound", - "confidence": "High", - "explanation": "The referral packet's intake form documents \"No wounds documented. Skin intact,\" and today's SOC visit notes confirm \"Skin intact, no wounds.\"" - }, - "notes": "" - }, - { - "questionId": "wound-status", - "patientId": "pat", - "pile": "shared", - "highlight": [ - "medium confidence" - ], - "ai": { - "answer": "Ongoing", - "confidence": "Medium", - "explanation": "Today's SOC visit documents an open left heel wound measuring 2.0 x 1.5 cm with moderate serous drainage and yellow slough, contradicting the discharge summary's statement that the wound was closed; the current clinical observation governs." - }, - "target": { - "answer": "Ongoing", - "confidence": "High", - "explanation": "Today's visit notes describe an open 2.0 x 1.5 cm left heel wound with serous drainage and slough; the discharge summary's closed status is outdated." - }, - "notes": "" - }, - { - "questionId": "wound-status", - "patientId": "riley", - "pile": "shared", - "highlight": [], - "ai": { - "answer": "No wound", - "confidence": "High", - "explanation": "The referral packet states 'No wounds' alongside the right wrist fracture with cast, and today's SOC visit notes confirm 'No wounds; skin intact around cast.'" - }, - "target": { - "answer": "No wound", - "confidence": "High", - "explanation": "The referral packet states 'No wounds' alongside the right wrist fracture with cast, and today's SOC visit notes confirm 'No wounds; skin intact around cast.'" - }, - "notes": "" - } -] diff --git a/examples/prompt-lab/evidence/run/lab/work/q-wound-status/cd33704e/score.json b/examples/prompt-lab/evidence/run/lab/work/q-wound-status/cd33704e/score.json deleted file mode 100644 index db6190de..00000000 --- a/examples/prompt-lab/evidence/run/lab/work/q-wound-status/cd33704e/score.json +++ /dev/null @@ -1,59 +0,0 @@ -{ - "promptId": "p-wound-status-f1305d28", - "menus": [ - { - "options": [ - "Healed", - "Ongoing", - "No wound" - ], - "agencies": [ - "harbor", - "maple", - "sunrise" - ], - "score": { - "rows": [ - { - "patientId": "jordan", - "gold": "No wound", - "rerun": "No wound", - "result": "worked" - }, - { - "patientId": "pat", - "gold": "Ongoing", - "rerun": "Ongoing", - "result": "worked" - }, - { - "patientId": "riley", - "gold": "No wound", - "rerun": "No wound", - "result": "worked" - } - ], - "golded": 3, - "worked": 3, - "percent": 100 - }, - "outputs": { - "jordan": { - "answer": "No wound", - "confidence": "High", - "explanation": "Today's SOC visit notes state \"Skin intact, no wounds,\" documenting the current condition directly; the referral packet's intake form likewise recorded no wounds with skin intact." - }, - "pat": { - "answer": "Ongoing", - "confidence": "High", - "explanation": "Today's SOC visit notes document the left heel as an open area 2.0 x 1.5 cm with moderate serous drainage, a pink wound bed with yellow slough at the edges, and a foam dressing change per protocol. The hospital discharge summary stating the wound was closed is outdated and superseded by today's direct observation." - }, - "riley": { - "answer": "No wound", - "confidence": "High", - "explanation": "Today's SOC visit notes state 'No wounds; skin intact around cast,' describing the patient's current skin condition directly, which is consistent with the referral packet's note of no wounds." - } - } - } - ] -} diff --git a/examples/prompt-lab/evidence/run/lab/work/sunrise-soc/8b7996f3/commit.json b/examples/prompt-lab/evidence/run/lab/work/sunrise-soc/35966ab5/commit.json similarity index 100% rename from examples/prompt-lab/evidence/run/lab/work/sunrise-soc/8b7996f3/commit.json rename to examples/prompt-lab/evidence/run/lab/work/sunrise-soc/35966ab5/commit.json diff --git a/examples/prompt-lab/evidence/run/lab/work/sunrise-soc/35966ab5/grid.json b/examples/prompt-lab/evidence/run/lab/work/sunrise-soc/35966ab5/grid.json new file mode 100644 index 00000000..662b10b6 --- /dev/null +++ b/examples/prompt-lab/evidence/run/lab/work/sunrise-soc/35966ab5/grid.json @@ -0,0 +1,170 @@ +[ + { + "questionId": "living-situation", + "patientId": "jordan", + "pile": "agency-specific", + "highlight": [ + "first-pass prompt, never reviewed" + ], + "ai": { + "answer": "Lives with spouse", + "confidence": "High", + "explanation": "Today's visit note states the patient and spouse share the home and the spouse sets up the pill organizer each morning, which directly establishes co-residence with a spouse; this outranks the intake form's referral-era \"lives alone\" entry, since today's visit notes take precedence over prior documentation." + }, + "target": { + "answer": "Lives with spouse", + "confidence": "High", + "explanation": "Today's visit note states the patient and spouse share the home and the spouse sets up the pill organizer each morning, which directly establishes co-residence with a spouse; this outranks the intake form's referral-era \"lives alone\" entry, since today's visit notes take precedence over prior documentation." + }, + "notes": "" + }, + { + "questionId": "living-situation", + "patientId": "pat", + "pile": "agency-specific", + "highlight": [ + "first-pass prompt, never reviewed" + ], + "ai": { + "answer": "Lives alone", + "confidence": "High", + "explanation": "Today's visit note's home environment section states the patient \"lives alone in a one-story home; neighbor checks in daily,\" which directly names the living situation and outranks the referral-era intake form. The neighbor's daily check-ins are visits rather than co-residence." + }, + "target": { + "answer": "Lives alone", + "confidence": "High", + "explanation": "Today's visit note's home environment section states the patient \"lives alone in a one-story home; neighbor checks in daily,\" which directly names the living situation and outranks the referral-era intake form. The neighbor's daily check-ins are visits rather than co-residence." + }, + "notes": "" + }, + { + "questionId": "living-situation", + "patientId": "riley", + "pile": "agency-specific", + "highlight": [ + "first-pass prompt, never reviewed" + ], + "ai": { + "answer": "Lives with family", + "confidence": "High", + "explanation": "Today's visit note states the patient lives with her daughter, son-in-law and two grandchildren, which maps directly to the family option; the referral packet's co-resident statement agrees and no spouse is documented in the home." + }, + "target": { + "answer": "Lives with family", + "confidence": "High", + "explanation": "Today's visit note states the patient lives with her daughter, son-in-law and two grandchildren, which maps directly to the family option; the referral packet's co-resident statement agrees and no spouse is documented in the home." + }, + "notes": "" + }, + { + "questionId": "mood", + "patientId": "jordan", + "pile": "mismatch", + "highlight": [ + "medium confidence", + "shared prompt, this agency's menu or ancestry differs" + ], + "ai": { + "answer": "Calm", + "confidence": "Medium", + "explanation": "The visit notes describe the patient as \"relaxed and engaged, asks good questions about the diuretic,\" which is the clinician's observation of behavior rather than the patient stating their feelings in their own words." + }, + "target": { + "answer": "Calm", + "confidence": "Medium", + "explanation": "The visit notes describe the patient as \"relaxed and engaged, asks good questions about the diuretic,\" which is the clinician's observation of behavior rather than the patient stating their feelings in their own words." + }, + "notes": "" + }, + { + "questionId": "mood", + "patientId": "pat", + "pile": "mismatch", + "highlight": [ + "shared prompt, this agency's menu or ancestry differs" + ], + "ai": { + "answer": "Calm", + "confidence": "High", + "explanation": "Visit notes describe the patient as \"calm and pleasant, joking about hospital food,\" and she states in her own words that she is glad to be home." + }, + "target": { + "answer": "Calm", + "confidence": "High", + "explanation": "Visit notes describe the patient as \"calm and pleasant, joking about hospital food,\" and she states in her own words that she is glad to be home." + }, + "notes": "" + }, + { + "questionId": "mood", + "patientId": "riley", + "pile": "mismatch", + "highlight": [ + "shared prompt, this agency's menu or ancestry differs" + ], + "ai": { + "answer": "Anxious", + "confidence": "High", + "explanation": "The patient states in her own words, 'I can't stop thinking about falling again, I lie awake worrying about it,' and the nurse notes she wrings her hands during the medication review." + }, + "target": { + "answer": "Anxious", + "confidence": "High", + "explanation": "The patient states in her own words, 'I can't stop thinking about falling again, I lie awake worrying about it,' and the nurse notes she wrings her hands during the medication review." + }, + "notes": "" + }, + { + "questionId": "wound-status", + "patientId": "pat", + "pile": "shared", + "highlight": [ + "medium confidence" + ], + "ai": { + "answer": "Ongoing", + "confidence": "Medium", + "explanation": "The referral packet's discharge summary states the left heel pressure injury was closed, but today's SOC visit notes document an open left heel area 2.0 x 1.5 cm with moderate serous drainage and yellow slough, so the wound is currently ongoing." + }, + "target": { + "answer": "Ongoing", + "confidence": "High", + "explanation": "Today's visit notes describe an open 2.0 x 1.5 cm left heel wound with serous drainage and slough; the discharge summary's closed status is outdated." + }, + "notes": "The prompt makes the referral win over what the nurse saw today. Today's notes must win (shared guideline 1)." + }, + { + "questionId": "wound-status", + "patientId": "jordan", + "pile": "shared", + "highlight": [], + "ai": { + "answer": "No wound", + "confidence": "High", + "explanation": "The referral packet's intake form documents no wounds and intact skin, and today's SOC visit note confirms 'Skin intact, no wounds.'" + }, + "target": { + "answer": "No wound", + "confidence": "High", + "explanation": "The referral packet's intake form documents no wounds and intact skin, and today's SOC visit note confirms 'Skin intact, no wounds.'" + }, + "notes": "" + }, + { + "questionId": "wound-status", + "patientId": "riley", + "pile": "shared", + "highlight": [], + "ai": { + "answer": "No wound", + "confidence": "High", + "explanation": "The referral packet states 'No wounds' with the presenting problem being a right wrist fracture in a cast, and today's SOC visit notes confirm 'No wounds; skin intact around cast.'" + }, + "target": { + "answer": "No wound", + "confidence": "High", + "explanation": "The referral packet states 'No wounds' with the presenting problem being a right wrist fracture in a cast, and today's SOC visit notes confirm 'No wounds; skin intact around cast.'" + }, + "notes": "" + } +] diff --git a/examples/prompt-lab/evidence/run/lab/work/sunrise-soc/8b7996f3/grid.json b/examples/prompt-lab/evidence/run/lab/work/sunrise-soc/8b7996f3/grid.json deleted file mode 100644 index 1f33405d..00000000 --- a/examples/prompt-lab/evidence/run/lab/work/sunrise-soc/8b7996f3/grid.json +++ /dev/null @@ -1,170 +0,0 @@ -[ - { - "questionId": "living-situation", - "patientId": "jordan", - "pile": "agency-specific", - "highlight": [ - "first-pass prompt, never reviewed" - ], - "ai": { - "answer": "Lives with spouse", - "confidence": "High", - "explanation": "Today's skilled nursing SOC visit notes state directly that the patient and their spouse share the home, with the spouse present throughout the visit and setting up the pill organizer each morning. The intake form's 'lives alone' entry is older supporting evidence and is treated as stale under today's direct observation." - }, - "target": { - "answer": "Lives with spouse", - "confidence": "High", - "explanation": "Today's skilled nursing SOC visit notes state directly that the patient and their spouse share the home, with the spouse present throughout the visit and setting up the pill organizer each morning. The intake form's 'lives alone' entry is older supporting evidence and is treated as stale under today's direct observation." - }, - "notes": "" - }, - { - "questionId": "living-situation", - "patientId": "pat", - "pile": "agency-specific", - "highlight": [ - "first-pass prompt, never reviewed" - ], - "ai": { - "answer": "Lives alone", - "confidence": "High", - "explanation": "Today's skilled nursing SOC visit note states directly that the patient lives alone in a one-story home, with a neighbor who checks in daily; an intermittent visitor does not count as a cohabitant. This is consistent with the referral packet's intake form, but the current visit note is the direct basis." - }, - "target": { - "answer": "Lives alone", - "confidence": "High", - "explanation": "Today's skilled nursing SOC visit note states directly that the patient lives alone in a one-story home, with a neighbor who checks in daily; an intermittent visitor does not count as a cohabitant. This is consistent with the referral packet's intake form, but the current visit note is the direct basis." - }, - "notes": "" - }, - { - "questionId": "living-situation", - "patientId": "riley", - "pile": "agency-specific", - "highlight": [ - "first-pass prompt, never reviewed" - ], - "ai": { - "answer": "Lives with family", - "confidence": "High", - "explanation": "Today's visit notes state directly that the patient lives with her adult child, that child's spouse, and two grandchildren, which is a household of relatives other than a spouse. The referral packet agrees, but the visit note is the direct current statement; 'Lives alone' and the spouse option were rejected because multiple relatives reside in the home." - }, - "target": { - "answer": "Lives with family", - "confidence": "High", - "explanation": "Today's visit notes state directly that the patient lives with her adult child, that child's spouse, and two grandchildren, which is a household of relatives other than a spouse. The referral packet agrees, but the visit note is the direct current statement; 'Lives alone' and the spouse option were rejected because multiple relatives reside in the home." - }, - "notes": "" - }, - { - "questionId": "mood", - "patientId": "jordan", - "pile": "mismatch", - "highlight": [ - "medium confidence", - "shared prompt, this agency's menu or ancestry differs" - ], - "ai": { - "answer": "Calm", - "confidence": "Medium", - "explanation": "The visit notes describe the patient as \"relaxed and engaged\" and asking good questions about the diuretic, but the mood is inferred from the clinician's behavioral description rather than stated by the patient in their own words." - }, - "target": { - "answer": "Calm", - "confidence": "Medium", - "explanation": "The visit notes describe the patient as \"relaxed and engaged\" and asking good questions about the diuretic, but the mood is inferred from the clinician's behavioral description rather than stated by the patient in their own words." - }, - "notes": "" - }, - { - "questionId": "mood", - "patientId": "pat", - "pile": "mismatch", - "highlight": [ - "shared prompt, this agency's menu or ancestry differs" - ], - "ai": { - "answer": "Calm", - "confidence": "High", - "explanation": "The visit notes describe the patient as \"calm and pleasant, joking about hospital food,\" and she states in her own words that she is glad to be home." - }, - "target": { - "answer": "Calm", - "confidence": "High", - "explanation": "The visit notes describe the patient as \"calm and pleasant, joking about hospital food,\" and she states in her own words that she is glad to be home." - }, - "notes": "" - }, - { - "questionId": "mood", - "patientId": "riley", - "pile": "mismatch", - "highlight": [ - "shared prompt, this agency's menu or ancestry differs" - ], - "ai": { - "answer": "Anxious", - "confidence": "High", - "explanation": "The patient states in her own words, 'I can't stop thinking about falling again, I lie awake worrying about it,' and the clinician notes she wrings her hands during the medication review." - }, - "target": { - "answer": "Anxious", - "confidence": "High", - "explanation": "The patient states in her own words, 'I can't stop thinking about falling again, I lie awake worrying about it,' and the clinician notes she wrings her hands during the medication review." - }, - "notes": "" - }, - { - "questionId": "wound-status", - "patientId": "pat", - "pile": "shared", - "highlight": [ - "medium confidence" - ], - "ai": { - "answer": "Ongoing", - "confidence": "Medium", - "explanation": "The referral discharge summary states the left heel wound was closed, but today's SOC visit notes document an open left heel area 2.0 x 1.5 cm with moderate serous drainage and yellow slough, so the wound is not healed." - }, - "target": { - "answer": "Ongoing", - "confidence": "High", - "explanation": "Today's visit notes describe an open 2.0 x 1.5 cm left heel wound with serous drainage and slough; the discharge summary's closed status is outdated." - }, - "notes": "The prompt makes the referral win over what the nurse saw today. Today's notes must win (shared guideline 1)." - }, - { - "questionId": "wound-status", - "patientId": "jordan", - "pile": "shared", - "highlight": [], - "ai": { - "answer": "No wound", - "confidence": "High", - "explanation": "The referral packet's intake form documents \"No wounds documented. Skin intact,\" and today's SOC visit notes confirm \"Skin intact, no wounds.\"" - }, - "target": { - "answer": "No wound", - "confidence": "High", - "explanation": "The referral packet's intake form documents \"No wounds documented. Skin intact,\" and today's SOC visit notes confirm \"Skin intact, no wounds.\"" - }, - "notes": "" - }, - { - "questionId": "wound-status", - "patientId": "riley", - "pile": "shared", - "highlight": [], - "ai": { - "answer": "No wound", - "confidence": "High", - "explanation": "The referral packet states \"No wounds\" and today's SOC visit notes confirm \"No wounds; skin intact around cast,\" so there is no primary wound to status." - }, - "target": { - "answer": "No wound", - "confidence": "High", - "explanation": "The referral packet states \"No wounds\" and today's SOC visit notes confirm \"No wounds; skin intact around cast,\" so there is no primary wound to status." - }, - "notes": "" - } -] diff --git a/examples/prompt-lab/evidence/runtime-findings/00-prove-attempt3-lease-conflict.txt b/examples/prompt-lab/evidence/runtime-findings/00-prove-attempt3-lease-conflict.txt new file mode 100644 index 00000000..511bb551 --- /dev/null +++ b/examples/prompt-lab/evidence/runtime-findings/00-prove-attempt3-lease-conflict.txt @@ -0,0 +1,25 @@ +$ npx flows run --no-observer-link --data-dir /private/tmp/claude-501/-Users-khaliqgant-Projects-AgentWorkforce-flows/df229951-f0e4-4753-990e-0f66533486d1/scratchpad/prove3-data --local-agent prompt-lab.flow.ts --input {"job":"fix","reviewer":"prompt-lab-reviewer","lab":"evidence/run/lab","issueId":"config-sunrise-wound-status-8beb8d42"} +○ run-1 (deterministic) 0.00s +✓ run-1 (deterministic) 0.08s completionReason: success +○ llm-2 (llm) 0.00s +WAITING [worker_lease] Run "01M36DMH97CEWZR6E8A4827QSQ" step "llm-2" (llm) is running under a worker lease until 1790143281757. +↻ llm-2 (llm) 9.16s +WAITING [worker_lease] Run "01M36DMH97CEWZR6E8A4827QSQ" step "llm-2" (llm) is running under a worker lease until 1790143291760. +↻ llm-2 (llm) 19.20s +✓ llm-2 (llm) 23.74s completionReason: success +○ llm-3 (llm) 0.00s +WAITING [worker_lease] Run "01M36DN4T1HR6D7M784QA9VZR4" step "llm-3" (llm) is running under a worker lease until 1790143301750. +↻ llm-3 (llm) 5.41s +✓ llm-3 (llm) 11.57s completionReason: success +○ llm-4 (llm) 0.00s +WAITING [worker_lease] Run "01M36DNH2JGN308JVHF92J3JT2" step "llm-4" (llm) is running under a worker lease until 1790143314315. +↻ llm-4 (llm) 6.42s +WAITING [worker_lease] Run "01M36DNH2JGN308JVHF92J3JT2" step "llm-4" (llm) is running under a worker lease until 1790143332374. +↻ llm-4 (llm) 27.06s +✓ llm-4 (llm) 38.59s completionReason: success +○ llm-5 (llm) 0.00s +WAITING [worker_lease] Run "01M36DQVNP1QDVAF92D1STJX01" step "llm-5" (llm) is running under a worker lease until 1790143392918. +↻ llm-5 (llm) 46.42s +✗ llm-5 (llm) 46.42s +FAILED [protocol_error] relayflowd could not complete the run request: lease_conflict: attempt has no active worker lease +exit=1 diff --git a/examples/prompt-lab/jobs/fix.ts b/examples/prompt-lab/jobs/fix.ts index 37426b43..ac511453 100644 --- a/examples/prompt-lab/jobs/fix.ts +++ b/examples/prompt-lab/jobs/fix.ts @@ -108,7 +108,7 @@ export async function fix(job: Job, input: FixInput): Promise { `yes = mark done (live in Apricot now, no promote step); no = keep it as a draft and iterate again in a new run.`, { to: reviewer }); if (!done) return f.done("declined"); - await lab.publish(qid, promptId); + await lab.publish(qid, promptId, question.livePromptId); if (issue) await lab.closeIssue(issue.id); return f.done("success"); } diff --git a/examples/prompt-lab/jobs/new-agency.ts b/examples/prompt-lab/jobs/new-agency.ts index 75801b87..ba2af6b0 100644 --- a/examples/prompt-lab/jobs/new-agency.ts +++ b/examples/prompt-lab/jobs/new-agency.ts @@ -6,7 +6,7 @@ // System run on fake visits → edit grid, computer-highlighted rows first // You edit each row on the first QA pass // Outcome targets persist for question × patient -// System run iteration on every changed question +// System run iteration on every changed agency-specific question // System shared rows → question-level queue with their targets (frozen here) // You commit all / only / all except // Outcome committed agency-specific prompts are live in Apricot @@ -18,7 +18,7 @@ import type { Issue, Output, Patient, PatientBrief } from "../lib/types.ts"; import { firstPass, iterate, promptQa, type Ask } from "../prompts.ts"; import type { Job } from "./job.ts"; import { shellWord } from "../lib/lab.ts"; -import { gapId, llm, MAX_ATTEMPTS, planCoverage, runEngine, untilQaPasses } from "./shared.ts"; +import { gapId, MAX_ATTEMPTS, planCoverage, runEngine, untilQaPasses } from "./shared.ts"; export interface NewAgencyInput { agency: string; visitType: string } @@ -88,23 +88,21 @@ export async function newAgency(job: Job, input: NewAgencyInput): Promise gridError(g, menus) ?? (g.map(rowKey).sort().join(",") === expected ? null : "rows were added or removed; edit values only")); await lab.record("targets", Object.fromEntries(edited.map((r) => [`${input.agency}|${rowKey(r)}`, r.target]))); - // Run iteration on every question with a changed row. Agency-specific rewrites - // are commit candidates (Prompt QA'd); shared/mismatch get a proposed rewrite only. + // Run iteration on every agency-specific question with a changed row; each + // rewrite is Prompt QA'd and becomes a commit candidate. Shared and mismatch + // prompts are frozen here: their rows travel to the question manager, where + // Job 2 iterates from the live prompt with the full changeset. const changes = changed(edited); const changedQs = [...new Set(changes.map((r) => r.questionId))].sort(); const rewrites: { q: PiledQuestion; rows: Row[]; prompt: string; passed: boolean }[] = []; + const sends: { q: PiledQuestion; rows: Row[] }[] = []; for (const qid of changedQs) { // sequential: see runEngine const q = byId.get(qid)!; - const rowsFor = changes.filter((r) => r.questionId === qid) - .map((r) => ({ patient: patient.get(r.patientId)! as Patient, ai: r.ai, target: r.target, notes: r.notes })); + const rows = changes.filter((r) => r.questionId === qid); + if (q.pile !== "agency-specific") { sends.push({ q, rows }); continue; } + const rowsFor = rows.map((r) => ({ patient: patient.get(r.patientId)! as Patient, ai: r.ai, target: r.target, notes: r.notes })); const brief = snap.briefs[qid]; const existing = prompts.get(qid)!.text; - const rows = changes.filter((r) => r.questionId === qid); - if (q.pile !== "agency-specific") { - const { prompt } = await llm<{ prompt: string }>(f, iterate(ask(q), existing, brief, rowsFor, [])); - rewrites.push({ q, rows, prompt, passed: true }); - continue; - } const r = await untilQaPasses<{ prompt: string }>(f, (findings) => iterate(ask(q), existing, brief, rowsFor, findings), (w) => promptQa(w.prompt, ask(q), snap.guidelines, brief)); @@ -118,17 +116,16 @@ export async function newAgency(job: Job, input: NewAgencyInput): Promise const waiting = piled.filter((q) => prompts.get(q.questionId)!.firstPass && !covered.has(q.questionId)).map((q) => q.questionId); const sent: string[] = []; for (const r of rewrites) { - if (r.q.pile === "agency-specific") { - if (!r.passed) { candidates.delete(r.q.questionId); continue; } // an uncompliant rewrite never becomes a candidate - drafts.set(r.q.questionId, (await lab.draft(r.q.questionId, r.prompt)).promptId); - candidates.add(r.q.questionId); - continue; - } - // Frozen here: the rows travel to question-level with their targets and the proposal. + if (!r.passed) { candidates.delete(r.q.questionId); continue; } // an uncompliant rewrite never becomes a candidate + drafts.set(r.q.questionId, (await lab.draft(r.q.questionId, r.prompt)).promptId); + candidates.add(r.q.questionId); + } + for (const r of sends) { + // Frozen here: the rows travel to question-level with their targets. const issue: Issue = { id: `config-${input.agency}-${r.q.questionId}-${hash8(r.rows)}`, kind: "config-send", questionIds: [r.q.questionId], status: "open", agency: input.agency, text: `${input.agency} ${input.visitType} first pass changed ${r.rows.length} row(s) on a ${r.q.pile} question (shared with ${r.q.sharedWith.join(", ")}). ${r.rows.map((x) => x.notes).filter(Boolean).join(" ")}`.trim(), - targets: Object.fromEntries(r.rows.map((x) => [x.patientId, x.target])), proposedPrompt: r.prompt, + targets: Object.fromEntries(r.rows.map((x) => [x.patientId, x.target])), }; await lab.enqueue("issues", issue); sent.push(`${r.q.questionId} → ${issue.id}`); @@ -146,6 +143,6 @@ export async function newAgency(job: Job, input: NewAgencyInput): Promise { to: reviewer }); if (!commit) return f.done("declined"); const choice = await lab.read(`${work}/commit.json`, commitError); - for (const qid of toCommit(list, choice)) await lab.publish(qid, drafts.get(qid)!); + for (const qid of toCommit(list, choice)) await lab.publish(qid, drafts.get(qid)!, snap.bank.questions[qid]!.livePromptId); return f.done("success"); } diff --git a/examples/prompt-lab/lib/lab.ts b/examples/prompt-lab/lib/lab.ts index ebae73a9..ae6be6e1 100644 --- a/examples/prompt-lab/lib/lab.ts +++ b/examples/prompt-lab/lib/lab.ts @@ -30,7 +30,8 @@ export interface Lab { writeNew(rel: string, value: unknown): Promise<{ written: boolean }>; record(store: "targets" | "gold", entries: Record): Promise; draft(questionId: string, prompt: string): Promise<{ promptId: string }>; - publish(questionId: string, promptId: string): Promise<{ previous: string | null }>; + /** Compare-and-swap on the live prompt the caller read (null: none). */ + publish(questionId: string, promptId: string, expectedLive: string | null): Promise<{ previous: string | null }>; enqueue(queue: "issues" | "patient-briefs", item: { id: string }): Promise<{ added: boolean }>; closeIssue(id: string): Promise; lockPatient(patient: unknown, briefId?: string): Promise<{ shelfPath: string }>; @@ -51,7 +52,7 @@ export function lab(f: Ctx, dir: string): Lab { writeNew: (rel, value) => store("write-new", rel, b64(value)), record: (s, entries) => store("record", s, b64(entries)), draft: (q, prompt) => store("draft", q, b64(prompt)), - publish: (q, id) => store("publish", q, id), + publish: (q, id, expected) => store("publish", q, id, expected ?? "-"), enqueue: (queue, item) => store("enqueue", queue, b64(item)), closeIssue: (id) => store("close-issue", id), lockPatient: (patient, briefId) => store("lock-patient", b64(patient), ...(briefId ? [briefId] : [])), diff --git a/examples/prompt-lab/lib/types.ts b/examples/prompt-lab/lib/types.ts index f0b1fbcb..6500ab8a 100644 --- a/examples/prompt-lab/lib/types.ts +++ b/examples/prompt-lab/lib/types.ts @@ -43,7 +43,6 @@ export interface Issue { agency?: string; /** Config-level targets travel with a row sent to question-level. */ targets?: Record; - proposedPrompt?: string; } export type Pile = "agency-specific" | "shared" | "mismatch"; diff --git a/examples/prompt-lab/store.ts b/examples/prompt-lab/store.ts index 71748b6a..5e91458e 100644 --- a/examples/prompt-lab/store.ts +++ b/examples/prompt-lab/store.ts @@ -12,12 +12,13 @@ // node store.ts write-new (never clobbers a reviewer's edit) // node store.ts record // node store.ts draft -// node store.ts publish +// node store.ts publish // node store.ts enqueue // node store.ts close-issue // node store.ts lock-patient [briefId] import { createHash } from "node:crypto"; -import { cpSync, existsSync, mkdirSync, readdirSync, readFileSync, renameSync, rmSync, writeFileSync } from "node:fs"; +import { cpSync, existsSync, mkdirSync, readdirSync, readFileSync, renameSync, writeFileSync } from "node:fs"; +import { DatabaseSync } from "node:sqlite"; import { dirname, join } from "node:path"; import type { Bank, Issue, Patient, PatientBrief, Snapshot } from "./lib/types.ts"; @@ -40,18 +41,20 @@ function print(value: unknown): void { if (bytes > OUTPUT_LIMIT) fail(`output is ${bytes} bytes, over the ${OUTPUT_LIMIT}-byte journal tail`); process.stdout.write(`${text}\n`); } -/** An exclusive lab lock: mkdir is atomic, so one process holds it at a time. */ +/** + * An exclusive lab lock: a SQLite EXCLUSIVE transaction on /.lock.db. + * The lock is the kernel's file lock, so it is released when its process + * exits for any reason, a SIGKILL included: a crash never wedges the lab for + * the resumed step. Waiters block for LOCK_WAIT_MS, then fail. + */ function withLock(work: () => void): void { - const lock = join(lab, ".lock"); - const deadline = Date.now() + LOCK_WAIT_MS; - for (;;) { - try { mkdirSync(lock); break; } catch (error) { - if ((error as NodeJS.ErrnoException).code !== "EEXIST") throw error; - if (Date.now() > deadline) fail(`lab is locked (${lock}); remove it if no store process is running`); - Atomics.wait(new Int32Array(new SharedArrayBuffer(4)), 0, 0, 20); - } + const db = new DatabaseSync(join(lab, ".lock.db"), { timeout: LOCK_WAIT_MS }); + try { + try { db.exec("BEGIN EXCLUSIVE"); } catch { fail(`lab is locked by another store process for over ${LOCK_WAIT_MS} ms`); } + try { work(); } finally { db.exec("COMMIT"); } + } finally { + db.close(); } - try { work(); } finally { rmSync(lock, { recursive: true, force: true }); } } const path = (rel: string): string => { if (rel.startsWith("/") || rel.split("/").includes("..")) fail(`path escapes the lab: ${rel}`); @@ -126,15 +129,21 @@ switch (verb) { break; } case "publish": { - const [questionId, promptId] = args; + const [questionId, promptId, expectedArg] = args; + const expected = expectedArg === "-" ? null : expectedArg; + if (expectedArg === undefined) fail("publish needs the live prompt the caller saw (or -)"); const bank = readJson("bank.json", { prompts: {}, questions: {} }); const question = bank.questions[questionId ?? ""] ?? fail(`no question ${questionId}`); if (!promptId || !bank.prompts[promptId]) fail(`no prompt ${promptId}`); const previous = question.livePromptId; + // Compare-and-swap: a retried publish is a no-op once it landed, and it + // never rolls back a prompt someone else published since the caller looked. + if (previous === promptId) { print({ questionId, livePromptId: promptId, previous: expected }); break; } + if (previous !== expected) fail(`${questionId} is live on ${previous} now, not ${expected}: published since this run read it`); question.livePromptId = promptId; if (question.draftPromptId === promptId) question.draftPromptId = null; writeJson("bank.json", bank); - print({ questionId, livePromptId: promptId, previous: previous === promptId ? null : previous }); + print({ questionId, livePromptId: promptId, previous }); break; } case "enqueue": { diff --git a/examples/prompt-lab/tests/store.test.ts b/examples/prompt-lab/tests/store.test.ts index 039d6c41..f7be46bb 100644 --- a/examples/prompt-lab/tests/store.test.ts +++ b/examples/prompt-lab/tests/store.test.ts @@ -1,8 +1,9 @@ import assert from "node:assert/strict"; import { spawn, spawnSync } from "node:child_process"; -import { existsSync, mkdtempSync, readFileSync, writeFileSync } from "node:fs"; +import { mkdtempSync, readFileSync, writeFileSync } from "node:fs"; import { tmpdir } from "node:os"; import { join } from "node:path"; +import { DatabaseSync } from "node:sqlite"; import { test } from "node:test"; const STORE = new URL("../store.ts", import.meta.url).pathname; @@ -48,12 +49,14 @@ test("draft is content-addressed and publish is Done = live, both idempotent", ( const a = store("draft", "wound-status", b64("new text")).out; assert.deepEqual(store("draft", "wound-status", b64("new text")).out, a); assert.equal(store("snapshot").out.bank.questions["wound-status"].livePromptId, "p-wound-status-v1"); // a draft is not live - assert.equal(store("publish", "wound-status", a.promptId).out.previous, "p-wound-status-v1"); - assert.equal(store("publish", "wound-status", a.promptId).out.previous, null); + assert.equal(store("publish", "wound-status", a.promptId, "p-wound-status-v1").out.previous, "p-wound-status-v1"); + // A retry of the same publish is a no-op, and says what it replaced. + assert.deepEqual(store("publish", "wound-status", a.promptId, "p-wound-status-v1").out, { questionId: "wound-status", livePromptId: a.promptId, previous: "p-wound-status-v1" }); const q = store("snapshot").out.bank.questions["wound-status"]; assert.equal(q.livePromptId, a.promptId); assert.equal(q.draftPromptId, null); - assert.equal(store("publish", "wound-status", "p-nope").status, 1); + assert.equal(store("publish", "wound-status", "p-nope", a.promptId).status, 1); + assert.match(store("publish", "wound-status", a.promptId).err, /needs the live prompt/); }); test("enqueue is keyed by id; lock-patient freezes a chart and closes its brief", () => { @@ -84,13 +87,11 @@ test("concurrent store processes never lose an update (the lab lock serializes r const s = store("snapshot").out; for (const t of texts) assert.ok(Object.values(s.bank.prompts).includes(t), `lost draft: ${t}`); assert.equal(s.issues.length, 8); - assert.equal(existsSync(join(lab, ".lock")), false); }); test("a failing mutating verb releases the lock", () => { - const { lab, store } = seeded(); - assert.equal(store("publish", "mood", "p-nope").status, 1); - assert.equal(existsSync(join(lab, ".lock")), false); + const { store } = seeded(); + assert.equal(store("publish", "mood", "p-nope", "p-mood-v1").status, 1); assert.equal(store("draft", "mood", b64("still writable")).status, 0); }); @@ -102,3 +103,40 @@ test("the output limit counts UTF-8 bytes, not characters", () => { assert.equal(r.status, 1); assert.match(r.err, /bytes, over the 61440-byte journal tail/); }); + +test("publish is compare-and-swap: a retried older publish never rolls back a newer one", () => { + const { store } = seeded(); + const a = store("draft", "wound-status", b64("prompt A")).out.promptId; + const b = store("draft", "wound-status", b64("prompt B")).out.promptId; + assert.equal(store("publish", "wound-status", a, "p-wound-status-v1").status, 0); // run 1 lands, then crashes before journaling + assert.equal(store("publish", "wound-status", b, a).status, 0); // run 2 publishes B on top + const retry = store("publish", "wound-status", a, "p-wound-status-v1"); // run 1's step retried on resume + assert.equal(retry.status, 1); + assert.match(retry.err, /live on p-wound-status-\w+ now, not p-wound-status-v1/); + assert.equal(store("snapshot").out.bank.questions["wound-status"].livePromptId, b); +}); + +test("a mutating verb waits for a live lock owner, and proceeds once it lets go", async () => { + const { lab, store } = seeded(); + const owner = new DatabaseSync(join(lab, ".lock.db")); + owner.exec("BEGIN EXCLUSIVE"); // this test process holds the lock + let done = false; + const child = new Promise((resolve) => { + spawn("node", ["--no-warnings", "--experimental-strip-types", STORE, lab, "draft", "mood", b64("waited for the lock")]) + .on("close", (code) => { done = true; resolve(code); }); + }); + await new Promise((r) => setTimeout(r, 1500)); + assert.equal(done, false, "the draft ran while another process held the lock"); + owner.exec("COMMIT"); + owner.close(); + assert.equal(await child, 0); + assert.ok(Object.values(store("snapshot").out.bank.prompts).includes("waited for the lock")); +}); + +test("a lock held by a killed process is released by the OS", () => { + const { lab, store } = seeded(); + const killed = spawnSync("node", ["--no-warnings", "-e", + `const { DatabaseSync } = require("node:sqlite"); new DatabaseSync(${JSON.stringify(join(lab, ".lock.db"))}).exec("BEGIN EXCLUSIVE"); process.kill(process.pid, "SIGKILL");`]); + assert.equal(killed.signal, "SIGKILL"); + assert.equal(store("draft", "mood", b64("after a crash")).status, 0); +}); From dc9c6d9ac65f7aeeb8bff306732ced49162ab8aa Mon Sep 17 00:00:00 2001 From: Relayflow Lead Date: Tue, 22 Sep 2026 23:33:35 -0700 Subject: [PATCH 5/6] docs(examples): a short prompt-lab README; design notes and proof move to PROOF.md Co-Authored-By: Claude Opus 5.5 (1M context) --- examples/prompt-lab/.gitignore | 1 + examples/prompt-lab/PROOF.md | 161 +++++++++++++++++++++++++++ examples/prompt-lab/README.md | 193 ++++++++------------------------- 3 files changed, 207 insertions(+), 148 deletions(-) create mode 100644 examples/prompt-lab/PROOF.md diff --git a/examples/prompt-lab/.gitignore b/examples/prompt-lab/.gitignore index 30911be4..c1e6b322 100644 --- a/examples/prompt-lab/.gitignore +++ b/examples/prompt-lab/.gitignore @@ -1 +1,2 @@ +lab/ evidence/run/lab/.lock.db diff --git a/examples/prompt-lab/PROOF.md b/examples/prompt-lab/PROOF.md new file mode 100644 index 00000000..5ffae3fb --- /dev/null +++ b/examples/prompt-lab/PROOF.md @@ -0,0 +1,161 @@ +# prompt-lab: design notes and proof + +The Prompt Lab product brief as one relayflow. Prompt Lab is the workbench for +writing and fixing the prompts that draft home-health charts. The brief defines +two jobs and a shelf of fake patients. This flow covers all three: + +| `job` | Brief section | What it does | +| --- | --- | --- | +| `new-agency` | Job 1 · Config-level / new agency | Sorts the agency's questions into sharing piles, writes first-pass prompts for agency-specific questions, runs them on shelf patients, and puts the output in an edit grid. Your first pass is saved as targets. Iteration runs on changed rows. Agency-specific prompts are committed (all / only / all except). Shared rows go to the question manager. | +| `fix` | Job 2 · Question-level refinement | Takes one question, from an issue or picked directly. Runs it on shelf patients, and your review of the grid is saved as gold. The iterator rewrites the prompt, Prompt QA checks it, and it re-runs and scores against gold (% worked, per patient). Marking it done makes it live. | +| `patient` | Set up test patient | Takes a gap brief from the manager queue. An agent writes a patient plan and you kick generate. Patient QA loops until it passes, then locks the patient onto the shelf. | + +Each job file is its brief diagram, line by line: + +- **System** boxes are either deterministic TypeScript over journaled reads, or model calls. +- **You** boxes are `f.human` gates, asked of `input.reviewer`. +- **Outcome** boxes are writes to the lab store. + +## What maps to what + +| Brief | Here | +| --- | --- | +| Apricot's Bank (`livePromptId`, global prompts) | `bank.json` in the lab directory, written only by [`store.ts`](store.ts) | +| The chart-filling engine nurses use | an `llm` step given the live prompt and the patient. Its answer must be on the agency's menu (the schema's `enum`) | +| Sharing piles: a deterministic lookup, never an agent | [`lib/piles.ts`](lib/piles.ts) | +| Computer-highlighted rows | [`lib/grid.ts`](lib/grid.ts) `highlights`: confidence below High, a mismatch pile, or a never-reviewed first-pass prompt | +| First pass persists as targets / gold | `store.ts record`, fed only from the grid you reviewed. AI output never writes gold directly | +| Iterator: rewrite-only | [`prompts.ts`](prompts.ts) `iterate`. Input is the changeset, patients, brief and existing prompt; output is new prompt text | +| Prompt QA: the brief plus shared guidelines | `promptQa`, looping with the iterator. Capped at 3 tries, then the run parks as `needs_human` | +| Done = live | `store.ts publish` sets `livePromptId`, as a compare-and-swap on the live prompt the run read: a retried publish is a no-op, and it never rolls back one published since. There is no promote step. A shared question warns which agencies it will change, and the re-run scores the new prompt on **every distinct menu** it is asked with ([`lib/piles.ts`](lib/piles.ts) `distinctMenus`), because done changes it for all of them | +| Shared frozen at config level | a changed shared or mismatch row becomes a `config-send` issue that carries its targets. It gets no rewrite at config level: Job 2 iterates from the live prompt with the full changeset | +| Test planner: never invents a patient | picks shelf patients of the run's visit type (checked deterministically); each hole becomes one gap brief per question × visit type | +| You do not approve the chart | the only patient gate is *kick generate*. Patient QA, plus a deterministic identifier check ([`lib/phi.ts`](lib/phi.ts)), locks it | + +Every read and write of the lab is a journaled `f.run` step. Every store verb +is idempotent, so a retried step lands the lab in the same state. `write-new` +never overwrites an edit you made. Mutating verbs take an exclusive lab lock, +so two runs at once never lose each other's update. The lock is a SQLite +`BEGIN EXCLUSIVE` on `/.lock.db`. That's a kernel file lock, so the OS +releases it when its process dies, even from a SIGKILL. + +A first-pass prompt is offered for commit only after it has run on a shelf +patient and you have reviewed its rows. A prompt whose question is a gap waits +as a draft until a patient covers it. + +**Not built:** anything the brief lists under "Not at the start". Also not +built: + +- Drafting a question brief with an agent. The flow reads a brief when one + exists in `briefs/.md`. +- The Apricot patient-brief generator. It runs on live patients, so it belongs + in Apricot. +- The UI screens. + +The flow is the job graph that sits under those screens. + +## Run it + +```sh +npm install +npm test # 23 unit tests over the deterministic parts and the store +node --experimental-strip-types store.ts ./my-lab seed fixtures +npx flows run prompt-lab.flow.ts --local-agent --input \ + '{"job":"new-agency","reviewer":"","lab":"./my-lab","agency":"sunrise","visitType":"soc"}' +``` + +`reviewer` and `lab` are required, and there is no default person. The run +parks at each gate and prints the file to edit plus the `flows answer` / +`flows resume` commands. Edit the file, answer `yes`, resume. Answering `no` +stops the job as `declined` and keeps what was saved. + +The fixtures are invented and reproduce the brief's own examples. The new +agency `sunrise` asks four questions: + +- **wound-status**: shared with harbor and maple, with an identical menu. +- **mood**: a shared prompt, but sunrise adds "Agitated" to the menu, so it's a mismatch. +- **living-situation**: agency-specific, with no prompt yet. +- **ostomy-supplies**: agency-specific, with no shelf patient. + +The shelf holds Pat, Jordan and Riley. The shared wound prompt contains a +deliberate flaw: it lets the referral overrule today's visit notes. + +## Proof + +[`prove.sh`](prove.sh) seeds a fresh lab and drives all three jobs through the +real kernel with real Claude calls (`--local-agent`). At each gate it acts as +the reviewer and applies the edit described in its comments, captured as a +diff. It captures every command with its output and exit code in +[`evidence/run/`](evidence/run/), and the final lab lands in +`evidence/run/lab/`. + +```sh +./prove.sh evidence/run +``` + +The captured run (Claude Code 2.1.280, the adapter's default model). `prove.sh` +stops on the first exit it didn't expect, so reaching the final state means +every step below exited as shown: + +| Run | Result | What happened | +| --- | --- | --- | +| Job 1 · `sunrise` / `soc` ([01](evidence/run/01-job1-run.txt), [04](evidence/run/04-resume.txt), [06](evidence/run/06-resume.txt)) | 28 steps, `success` | Piles came out as shared / mismatch / agency-specific. Both agency-specific questions got first-pass prompts that passed Prompt QA. The planner covered 3 questions from the `soc` shelf and queued `gap-ostomy-supplies-soc`. Gate 1: 9 rows, 7 highlighted. The reviewer raised Pat's wound confidence to High, rewrote the explanation and added a note ([02](evidence/run/02-reviewer-edit.diff)). That shared row went to the question manager with its target, and got no rewrite at config level. Gate 2 offered only `living-situation`, and it went live. `ostomy-supplies` had no shelf patient, so it was held as a draft. | +| Patient · `gap-ostomy-supplies-soc` ([07](evidence/run/07-patient-run.txt), [09](evidence/run/09-resume.txt)) | 14 steps, `success` | Plan, then kick generate. Patient QA sent the chart back before one passed (`llm-6` … `llm-12`). `roderick` locked onto the shelf, and the brief is marked `locked`. | +| Job 2 · the config-send issue ([10](evidence/run/10-job2-run.txt), [12](evidence/run/12-resume.txt), [14](evidence/run/14-resume.txt)) | 24 steps, `success` | Four shelf patients, including `roderick`. **The live prompt answered Pat "Healed / High", which is the brief's failure.** Gold came prefilled, with Pat's config target "Ongoing" carried over. The iterator's rewrite passed Prompt QA. The re-run scored **3 of 4 golded patients worked (75%)**: Pat is now "Ongoing" and worked; `roderick` did not (gold "Ongoing", new run "No wound"). The gate warned it would change harbor, maple and sunrise. Done made the new prompt live and closed the issue. | + +Final state: [15-lab-state.txt](evidence/run/15-lab-state.txt), with the whole lab in `evidence/run/lab/`. + +**Read the 75% with care.** `prove.sh` answers `yes` at every gate, so it +marked done at 75%. A reviewer would look at `roderick` first. His gold was the +old prompt's answer, accepted without review, and an ostomy patient's +peristomal skin damage may or may not be a "primary wound". That's exactly the +clinical call this gate exists for. The reviewer here is a script, not a +clinician. + +`wound-status` has one menu across its agencies, so the per-menu re-run ran +with one menu. The multi-menu case, a mismatch question like `mood`, is covered +by the `distinctMenus` unit test only. + +## Runtime findings (relayflows 2.0.29) + +Four things in the runtime shaped this flow or its proof. Each workaround is commented +where it lives, and each has captured evidence in +[`evidence/runtime-findings/`](evidence/runtime-findings/): + +1. **More than a few concurrent `f.llm` calls lose the run.** The queued + calls' 30 s leases expire before the worker takes them. The late completion + of a dead attempt is then refused ("Agent lease is already expired"), and + the CLI turns that refusal into a fatal `protocol_error`. Five parallel + calls passed and nine failed, with `--agent-capacity` 4 or 1. + [`runtime-parallel-llm-repro.flow.ts`](evidence/runtime-findings/runtime-parallel-llm-repro.flow.ts) + reproduces it with no Prompt Lab code. **Workaround:** model calls run + sequentially. Tracked in [#561](https://github.com/AgentWorkforce/flows/issues/561) and [#560](https://github.com/AgentWorkforce/flows/issues/560). +2. **A predicate `.gate(fn)` before an `f.human` can't be resumed.** The + verdict is read back from the `predicate-gates` stream with its keys + re-ordered. The lowered `.gate` command no longer matches, and the + resume is refused as `run_admission_conflict`. The fix is one line in + `packages/sdk/src/authored-flow-executor.ts` `applyPredicateGate`: build the + literal from fixed fields, not from `JSON.stringify(record)`. + **Workaround:** the checks run in the body and fail through a journaled + failing step (`failStep`). **Fixed in [#558](https://github.com/AgentWorkforce/flows/pull/558).** +3. **`f.llm(prompt, { output })` fails when the reply is fenced JSON.** The + worker validates the raw reply. Sonnet sometimes wraps valid JSON in + ```` ```json ```` anyway, and a failed run can't be resumed. + **Workaround:** text-form `f.llm`, then + [`lib/reply.ts`](lib/reply.ts) strips one fence and validates the schema, + with one bounded re-ask. The text form takes no `model`, so calls use the + Claude adapter's default model. **Fixed in [#558](https://github.com/AgentWorkforce/flows/pull/558).** + +4. **A lease renewal that races a completion kills the run.** A step whose + child run journaled `success` was reported as + `lease_conflict: attempt has no active worker lease`. The CLI made that a + fatal `protocol_error` + ([00-prove-attempt1-lease-conflict-after-success.txt](evidence/runtime-findings/00-prove-attempt1-lease-conflict-after-success.txt)). + It's intermittent: it happened twice in about 100 sequential calls ([second](evidence/runtime-findings/00-prove-attempt3-lease-conflict.txt)). There's + no workaround in the flow, so rerun. Tracked in [#560](https://github.com/AgentWorkforce/flows/issues/560). + +Findings 1 and 4 are the same class of problem: late lease traffic becomes +fatal to the whole run instead of being ignored. + +Local only for now: Cloud receives a single authored source, and this flow +imports sibling modules. diff --git a/examples/prompt-lab/README.md b/examples/prompt-lab/README.md index e1d42f20..0a43e796 100644 --- a/examples/prompt-lab/README.md +++ b/examples/prompt-lab/README.md @@ -1,161 +1,58 @@ -# prompt-lab - -The Prompt Lab product brief as one relayflow. Prompt Lab is the workbench for -writing and fixing the prompts that draft home-health charts. The brief defines -two jobs and a shelf of fake patients. This flow covers all three: - -| `job` | Brief section | What it does | -| --- | --- | --- | -| `new-agency` | Job 1 · Config-level / new agency | Sorts the agency's questions into sharing piles, writes first-pass prompts for agency-specific questions, runs them on shelf patients, and puts the output in an edit grid. Your first pass is saved as targets. Iteration runs on changed rows. Agency-specific prompts are committed (all / only / all except). Shared rows go to the question manager. | -| `fix` | Job 2 · Question-level refinement | Takes one question, from an issue or picked directly. Runs it on shelf patients, and your review of the grid is saved as gold. The iterator rewrites the prompt, Prompt QA checks it, and it re-runs and scores against gold (% worked, per patient). Marking it done makes it live. | -| `patient` | Set up test patient | Takes a gap brief from the manager queue. An agent writes a patient plan and you kick generate. Patient QA loops until it passes, then locks the patient onto the shelf. | - -Each job file is its brief diagram, line by line: - -- **System** boxes are either deterministic TypeScript over journaled reads, or model calls. -- **You** boxes are `f.human` gates, asked of `input.reviewer`. -- **Outcome** boxes are writes to the lab store. - -## What maps to what - -| Brief | Here | -| --- | --- | -| Apricot's Bank (`livePromptId`, global prompts) | `bank.json` in the lab directory, written only by [`store.ts`](store.ts) | -| The chart-filling engine nurses use | an `llm` step given the live prompt and the patient. Its answer must be on the agency's menu (the schema's `enum`) | -| Sharing piles: a deterministic lookup, never an agent | [`lib/piles.ts`](lib/piles.ts) | -| Computer-highlighted rows | [`lib/grid.ts`](lib/grid.ts) `highlights`: confidence below High, a mismatch pile, or a never-reviewed first-pass prompt | -| First pass persists as targets / gold | `store.ts record`, fed only from the grid you reviewed. AI output never writes gold directly | -| Iterator: rewrite-only | [`prompts.ts`](prompts.ts) `iterate`. Input is the changeset, patients, brief and existing prompt; output is new prompt text | -| Prompt QA: the brief plus shared guidelines | `promptQa`, looping with the iterator. Capped at 3 tries, then the run parks as `needs_human` | -| Done = live | `store.ts publish` sets `livePromptId`, as a compare-and-swap on the live prompt the run read: a retried publish is a no-op, and it never rolls back one published since. There is no promote step. A shared question warns which agencies it will change, and the re-run scores the new prompt on **every distinct menu** it is asked with ([`lib/piles.ts`](lib/piles.ts) `distinctMenus`), because done changes it for all of them | -| Shared frozen at config level | a changed shared or mismatch row becomes a `config-send` issue that carries its targets. It gets no rewrite at config level: Job 2 iterates from the live prompt with the full changeset | -| Test planner: never invents a patient | picks shelf patients of the run's visit type (checked deterministically); each hole becomes one gap brief per question × visit type | -| You do not approve the chart | the only patient gate is *kick generate*. Patient QA, plus a deterministic identifier check ([`lib/phi.ts`](lib/phi.ts)), locks it | - -Every read and write of the lab is a journaled `f.run` step. Every store verb -is idempotent, so a retried step lands the lab in the same state. `write-new` -never overwrites an edit you made. Mutating verbs take an exclusive lab lock, -so two runs at once never lose each other's update. The lock is a SQLite -`BEGIN EXCLUSIVE` on `/.lock.db`. That's a kernel file lock, so the OS -releases it when its process dies, even from a SIGKILL. - -A first-pass prompt is offered for commit only after it has run on a shelf -patient and you have reviewed its rows. A prompt whose question is a gap waits -as a draft until a patient covers it. - -**Not built:** anything the brief lists under "Not at the start". Also not -built: - -- Drafting a question brief with an agent. The flow reads a brief when one - exists in `briefs/.md`. -- The Apricot patient-brief generator. It runs on live patients, so it belongs - in Apricot. -- The UI screens. - -The flow is the job graph that sits under those screens. - -## Run it +# Prompt Lab + +Prompt Lab is your product brief as one runnable relayflow. It does the brief's +two jobs, plus making test patients. It stops and asks you wherever the brief +says **You**. + +| Job | What it does | Where it asks you | +|---|---|---| +| `new-agency` | Sets up a new agency's prompts and runs them on fake patients | 1. Review the answers. 2. Approve what goes live. | +| `fix` | Fixes one prompt from an issue, then scores the fix | 1. Set the right answers. 2. Mark it done (it goes live). | +| `patient` | Makes a new fake patient for a coverage gap | 1. Start the build. | + +## Setup (once) + +You need Node 22.6+ and a signed-in [Claude Code](https://claude.com/claude-code). ```sh npm install -npm test # 23 unit tests over the deterministic parts and the store -node --experimental-strip-types store.ts ./my-lab seed fixtures +node --experimental-strip-types store.ts ./lab seed fixtures +``` + +This creates `./lab`: a sample Bank with three agencies and three fake +patients. + +## Run a job + +```sh npx flows run prompt-lab.flow.ts --local-agent --input \ - '{"job":"new-agency","reviewer":"","lab":"./my-lab","agency":"sunrise","visitType":"soc"}' + '{"job":"new-agency","reviewer":"you","lab":"./lab","agency":"sunrise","visitType":"soc"}' ``` -`reviewer` and `lab` are required, and there is no default person. The run -parks at each gate and prints the file to edit plus the `flows answer` / -`flows resume` commands. Edit the file, answer `yes`, resume. Answering `no` -stops the job as `declined` and keeps what was saved. +Other jobs use the same command with a different `--input`: -The fixtures are invented and reproduce the brief's own examples. The new -agency `sunrise` asks four questions: +- Fix a question: `{"job":"fix","reviewer":"you","lab":"./lab","issueId":""}` +- Make a patient: `{"job":"patient","reviewer":"you","lab":"./lab","briefId":""}` -- **wound-status**: shared with harbor and maple, with an identical menu. -- **mood**: a shared prompt, but sunrise adds "Agitated" to the menu, so it's a mismatch. -- **living-situation**: agency-specific, with no prompt yet. -- **ostomy-supplies**: agency-specific, with no shelf patient. +## When it stops for you -The shelf holds Pat, Jordan and Riley. The shared wound prompt contains a -deliberate flaw: it lets the referral overrule today's visit notes. +The run pauses and prints three things: -## Proof +1. **The file to review** (for example `lab/work/…/grid.json`). Open it and + change any answer, confidence or explanation that's wrong. +2. **An answer command:** `npx flows answer … yes`. Use `no` to stop. +3. **A resume command:** `npx flows resume …`. Run it to continue. -[`prove.sh`](prove.sh) seeds a fresh lab and drives all three jobs through the -real kernel with real Claude calls (`--local-agent`). At each gate it acts as -the reviewer and applies the edit described in its comments, captured as a -diff. It captures every command with its output and exit code in -[`evidence/run/`](evidence/run/), and the final lab lands in -`evidence/run/lab/`. +Your first edits are saved as the right answers. The AI never overwrites them. -```sh -./prove.sh evidence/run -``` +## Good to know + +- **The fake Bank.** "Apricot" here is the local `lab` folder, and the chart + engine is a Claude call. Connecting it to the real Bank and engine is the + next step. +- **Shared prompts.** Marking a shared prompt done changes it for every agency + that uses it. The run warns you first. +- **Runs locally, not on Cloud yet.** Each job takes a few minutes, because AI + calls run one at a time for now. -The captured run (Claude Code 2.1.280, the adapter's default model). `prove.sh` -stops on the first exit it didn't expect, so reaching the final state means -every step below exited as shown: - -| Run | Result | What happened | -| --- | --- | --- | -| Job 1 · `sunrise` / `soc` ([01](evidence/run/01-job1-run.txt), [04](evidence/run/04-resume.txt), [06](evidence/run/06-resume.txt)) | 28 steps, `success` | Piles came out as shared / mismatch / agency-specific. Both agency-specific questions got first-pass prompts that passed Prompt QA. The planner covered 3 questions from the `soc` shelf and queued `gap-ostomy-supplies-soc`. Gate 1: 9 rows, 7 highlighted. The reviewer raised Pat's wound confidence to High, rewrote the explanation and added a note ([02](evidence/run/02-reviewer-edit.diff)). That shared row went to the question manager with its target, and got no rewrite at config level. Gate 2 offered only `living-situation`, and it went live. `ostomy-supplies` had no shelf patient, so it was held as a draft. | -| Patient · `gap-ostomy-supplies-soc` ([07](evidence/run/07-patient-run.txt), [09](evidence/run/09-resume.txt)) | 14 steps, `success` | Plan, then kick generate. Patient QA sent the chart back before one passed (`llm-6` … `llm-12`). `roderick` locked onto the shelf, and the brief is marked `locked`. | -| Job 2 · the config-send issue ([10](evidence/run/10-job2-run.txt), [12](evidence/run/12-resume.txt), [14](evidence/run/14-resume.txt)) | 24 steps, `success` | Four shelf patients, including `roderick`. **The live prompt answered Pat "Healed / High", which is the brief's failure.** Gold came prefilled, with Pat's config target "Ongoing" carried over. The iterator's rewrite passed Prompt QA. The re-run scored **3 of 4 golded patients worked (75%)**: Pat is now "Ongoing" and worked; `roderick` did not (gold "Ongoing", new run "No wound"). The gate warned it would change harbor, maple and sunrise. Done made the new prompt live and closed the issue. | - -Final state: [15-lab-state.txt](evidence/run/15-lab-state.txt), with the whole lab in `evidence/run/lab/`. - -**Read the 75% with care.** `prove.sh` answers `yes` at every gate, so it -marked done at 75%. A reviewer would look at `roderick` first. His gold was the -old prompt's answer, accepted without review, and an ostomy patient's -peristomal skin damage may or may not be a "primary wound". That's exactly the -clinical call this gate exists for. The reviewer here is a script, not a -clinician. - -`wound-status` has one menu across its agencies, so the per-menu re-run ran -with one menu. The multi-menu case, a mismatch question like `mood`, is covered -by the `distinctMenus` unit test only. - -## Runtime findings (relayflows 2.0.29) - -Four things in the runtime shaped this flow or its proof. Each workaround is commented -where it lives, and each has captured evidence in -[`evidence/runtime-findings/`](evidence/runtime-findings/): - -1. **More than a few concurrent `f.llm` calls lose the run.** The queued - calls' 30 s leases expire before the worker takes them. The late completion - of a dead attempt is then refused ("Agent lease is already expired"), and - the CLI turns that refusal into a fatal `protocol_error`. Five parallel - calls passed and nine failed, with `--agent-capacity` 4 or 1. - [`runtime-parallel-llm-repro.flow.ts`](evidence/runtime-findings/runtime-parallel-llm-repro.flow.ts) - reproduces it with no Prompt Lab code. **Workaround:** model calls run - sequentially. Tracked in [#561](https://github.com/AgentWorkforce/flows/issues/561) and [#560](https://github.com/AgentWorkforce/flows/issues/560). -2. **A predicate `.gate(fn)` before an `f.human` can't be resumed.** The - verdict is read back from the `predicate-gates` stream with its keys - re-ordered. The lowered `.gate` command no longer matches, and the - resume is refused as `run_admission_conflict`. The fix is one line in - `packages/sdk/src/authored-flow-executor.ts` `applyPredicateGate`: build the - literal from fixed fields, not from `JSON.stringify(record)`. - **Workaround:** the checks run in the body and fail through a journaled - failing step (`failStep`). **Fixed in [#558](https://github.com/AgentWorkforce/flows/pull/558).** -3. **`f.llm(prompt, { output })` fails when the reply is fenced JSON.** The - worker validates the raw reply. Sonnet sometimes wraps valid JSON in - ```` ```json ```` anyway, and a failed run can't be resumed. - **Workaround:** text-form `f.llm`, then - [`lib/reply.ts`](lib/reply.ts) strips one fence and validates the schema, - with one bounded re-ask. The text form takes no `model`, so calls use the - Claude adapter's default model. **Fixed in [#558](https://github.com/AgentWorkforce/flows/pull/558).** - -4. **A lease renewal that races a completion kills the run.** A step whose - child run journaled `success` was reported as - `lease_conflict: attempt has no active worker lease`. The CLI made that a - fatal `protocol_error` - ([00-prove-attempt1-lease-conflict-after-success.txt](evidence/runtime-findings/00-prove-attempt1-lease-conflict-after-success.txt)). - It's intermittent: it happened twice in about 100 sequential calls ([second](evidence/runtime-findings/00-prove-attempt3-lease-conflict.txt)). There's - no workaround in the flow, so rerun. Tracked in [#560](https://github.com/AgentWorkforce/flows/issues/560). - -Findings 1 and 4 are the same class of problem: late lease traffic becomes -fatal to the whole run instead of being ignored. - -Local only for now: Cloud receives a single authored source, and this flow -imports sibling modules. +How it was tested, with full evidence: [PROOF.md](PROOF.md). From 90f56b68bd902b5b34ab4dde5e4191d730e3f063 Mon Sep 17 00:00:00 2001 From: Relayflow Lead Date: Tue, 22 Sep 2026 23:34:05 -0700 Subject: [PATCH 6/6] fix(examples): prompt-lab Job 2 checks shelf coverage after filtering; closes an issue the live prompt already resolves Co-Authored-By: Claude Opus 5.5 (1M context) --- examples/prompt-lab/jobs/fix.ts | 8 ++++++-- 1 file changed, 6 insertions(+), 2 deletions(-) diff --git a/examples/prompt-lab/jobs/fix.ts b/examples/prompt-lab/jobs/fix.ts index ac511453..34ade7f7 100644 --- a/examples/prompt-lab/jobs/fix.ts +++ b/examples/prompt-lab/jobs/fix.ts @@ -51,8 +51,8 @@ export async function fix(job: Job, input: FixInput): Promise { const gap: PatientBrief = { id: gapId(qid, asked.visitType), questionId: qid, visitType: asked.visitType, brief: plan.gaps[0].brief, from: "planner", status: "queued" }; await lab.enqueue("patient-briefs", gap); } - if (ids.size === 0) return f.done("needs_human"); // nothing on the shelf can show it yet; the gap brief is queued const patients = snap.shelf.filter((p) => ids.has(p.id)); + if (patients.length === 0) return f.done("needs_human"); // nothing on the shelf can show it yet; the gap brief is queued const current = await runEngine(f, live, ask, patients); const rows: Row[] = patients.map((p) => { @@ -79,7 +79,11 @@ export async function fix(job: Job, input: FixInput): Promise { await lab.record("gold", gold); const changes = changed(edited); - if (changes.length === 0) return f.done("declined"); // the live prompt already matches gold: nothing to iterate + if (changes.length === 0) { + // The live prompt already matches gold: the issue is resolved, nothing to iterate. + if (issue) await lab.closeIssue(issue.id); + return f.done("success"); + } const byId = new Map(patients.map((p) => [p.id, p])); const changeset = changes.map((r) => ({ patient: byId.get(r.patientId)!, ai: r.ai, target: r.target, notes: r.notes })); const rewrite = await untilQaPasses<{ prompt: string }>(f,