cowork-harness 1.1.0 → 1.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (60) hide show
  1. package/.claude/skills/cowork-harness/SKILL.md +72 -14
  2. package/.claude/skills/cowork-harness/references/ci-recipe.md +5 -5
  3. package/.claude/skills/cowork-harness/references/fidelity-and-answers.md +13 -7
  4. package/.claude/skills/cowork-harness/references/scenario-schema.md +14 -10
  5. package/.claude/skills/cowork-harness/references/task-recipes.md +27 -0
  6. package/.claude/skills/cowork-harness/scripts/assertion-keys.json +3 -0
  7. package/.claude/skills/cowork-harness/scripts/scenario.py +5 -0
  8. package/CHANGELOG.md +286 -0
  9. package/README.md +26 -21
  10. package/SPEC.md +40 -8
  11. package/dist/agent/session.js +77 -8
  12. package/dist/assert.js +339 -40
  13. package/dist/baseline.js +36 -8
  14. package/dist/cli.js +129 -35
  15. package/dist/decide/decider.js +61 -7
  16. package/dist/decide/external-channel.js +19 -6
  17. package/dist/hostloop/canusetool-gate.js +86 -2
  18. package/dist/hostloop/provenance.js +7 -2
  19. package/dist/hostloop/workspace-handler.js +4 -2
  20. package/dist/run/analyze-artifact-runtime.js +309 -65
  21. package/dist/run/analyze-artifact.js +786 -142
  22. package/dist/run/analyze-skill.js +182 -37
  23. package/dist/run/artifacts.js +105 -22
  24. package/dist/run/cassette.js +74 -7
  25. package/dist/run/chat-result.js +7 -2
  26. package/dist/run/doctor.js +93 -16
  27. package/dist/run/execute.js +71 -16
  28. package/dist/run/inspect-view.js +19 -0
  29. package/dist/run/latest-run.js +43 -0
  30. package/dist/run/pre-run-manifest.js +42 -10
  31. package/dist/run/renderer.js +15 -0
  32. package/dist/run/run-index.js +16 -5
  33. package/dist/run/run-status.js +1 -0
  34. package/dist/run/run.js +49 -7
  35. package/dist/run/status-target.js +28 -0
  36. package/dist/run/trace-view.js +28 -8
  37. package/dist/run/verdict.js +19 -0
  38. package/dist/runtime/hostloop.js +23 -4
  39. package/dist/runtime/microvm.js +44 -7
  40. package/dist/scan.js +17 -14
  41. package/dist/sync/cowork-sync.js +10 -1
  42. package/dist/types.js +17 -4
  43. package/docs/README.md +5 -1
  44. package/docs/cassette.md +27 -11
  45. package/docs/debugging.md +34 -0
  46. package/docs/decisions/README.md +7 -0
  47. package/docs/fidelity-gaps.md +5 -0
  48. package/docs/gotchas.md +9 -0
  49. package/docs/maintenance.md +2 -0
  50. package/docs/run-status.md +5 -0
  51. package/docs/scenario.md +67 -9
  52. package/docs/stats.md +5 -1
  53. package/docs/subagents.md +35 -1
  54. package/examples/replays/README.md +1 -1
  55. package/llms.txt +1 -0
  56. package/package.json +1 -1
  57. package/python/README.md +13 -4
  58. package/schema/run-result.json +19 -1
  59. package/schema/scenario.schema.json +18 -3
  60. package/scripts/check-baseline-staleness.ts +147 -0
@@ -1,10 +1,10 @@
1
1
  ---
2
2
  name: cowork-harness
3
- description: Test or debug a Claude Code skill/plugin under Claude Cowork's runtime — sandboxed agent, default-deny egress, the can_use_tool permission/question protocol — using the cowork-harness CLI. Use when validating or regression-testing a skill, authoring or debugging a scenario YAML (prompt + scripted answers + assert:), choosing a fidelity tier, scripting AskUserQuestion / tool-permission answers, or asserting artifacts, egress, or sub-agent dispatch. Especially when a harness run no-ops an assertion, fails on an unanswered gate, false-greens, a steered answer never reaches the model, or a web_fetch is unexpectedly denied or gated. NOT for generic unit testing (pytest/vitest of your own scripts) or non-Cowork CI. Covers the skill / run / chat / record / replay / trace / decide / assertions / scaffold commands and the session-vs-scenario split.
3
+ description: Test or debug a Claude Code skill/plugin under Claude Cowork's runtime — sandboxed agent, default-deny egress, the can_use_tool permission/question protocol — using the cowork-harness CLI. Use when validating or regression-testing a skill, authoring or debugging a scenario YAML (prompt + scripted answers + assert:), choosing a fidelity tier, scripting AskUserQuestion / tool-permission answers, or asserting artifacts, egress, or sub-agent dispatch. Especially when a harness run no-ops an assertion, fails on an unanswered gate, false-greens, a steered answer never reaches the model, or a web_fetch is unexpectedly denied or gated. Also when iterating or hardening a skill across fixes, or grounding a skill's self-critique against its own run evidence. NOT for generic unit testing (pytest/vitest of your own scripts) or non-Cowork CI. Covers the skill / run / chat / record / replay / trace / decide / assertions / scaffold commands and the session-vs-scenario split.
4
4
  metadata:
5
5
  author: cowork-harness
6
- version: 1.1.0
7
- tracks-harness: cowork-harness 1.1.0 (baseline desktop-1.21459.0)
6
+ version: 1.3.0
7
+ tracks-harness: cowork-harness 1.3.0 (baseline desktop-1.21459.0)
8
8
  ---
9
9
 
10
10
  # cowork-harness
@@ -22,7 +22,7 @@ flagged with a loud `::warning::`, not silent — auto-answer a gate, observe an
22
22
  allowlist). This skill exists mostly to keep you out of those traps — the Gotchas section below is
23
23
  the highest-value part. Read it.
24
24
 
25
- > **Version note:** the facts and `file:line` pointers here track `cowork-harness 1.1.0` (baseline
25
+ > **Version note:** the facts and `file:line` pointers here track `cowork-harness 1.3.0` (baseline
26
26
  > `desktop-1.21459.0`). If your checkout is newer, prefer the live `--help` and — in a repo checkout —
27
27
  > `SPEC.md` / `docs/*.md` over this snapshot, and re-run the bundled linter.
28
28
 
@@ -39,9 +39,9 @@ Before the first command, confirm the CLI is reachable and **fail loud** (never
39
39
 
40
40
  - **One-shot check.** Run `cowork-harness doctor [--tier <tier>]` first — a read-only prerequisite check that inspects Docker, the staged agent, the token, and the baseline in one pass. The bullets below explain each thing it checks (and how to fix it).
41
41
  - **Replay-only? Skip `doctor`.** Replaying committed cassettes needs no Docker, no staged agent, and no token — and every tier's `doctor` validates the auth token (the live tiers also Docker + the staged agent), so a ✗ there is expected, not a blocker. Go straight to `cowork-harness replay <cassette>`.
42
- - **CLI on PATH, recent enough?** Run `cowork-harness --version` — this skill needs **≥ 1.1.0**. If it's missing or older, prefix every command with the version floor `npx "cowork-harness@>=1.1.0" <cmd>` (Node ≥ 20), or install once with `npm i -g "cowork-harness@>=1.1.0"`. **Pin `@>=1.1.0`, never `@latest`** — `@latest` can silently fetch an older CLI and the new commands fail as "unknown command", whereas the floor **fails loud** if no compatible version is published.
42
+ - **CLI on PATH, recent enough?** Run `cowork-harness --version` — this skill needs **≥ 1.3.0**. If it's missing or older, prefix every command with the version floor `npx "cowork-harness@>=1.3.0" <cmd>` (Node ≥ 20), or install once with `npm i -g "cowork-harness@>=1.3.0"`. **Pin `@>=1.3.0`, never `@latest`** — `@latest` can silently fetch an older CLI and the new commands fail as "unknown command", whereas the floor **fails loud** if no compatible version is published.
43
43
 
44
- What the ≥ 1.1.0 floor gates, by release:
44
+ What the ≥ 1.3.0 floor gates, by release:
45
45
 
46
46
  - **core set (pre-0.21.0 vintage, or mixed):** `assertions --list`, `scaffold <run-id>`, `trace --view dispatches`, `artifact_json` incl. the `in:` operator (passes when the resolved value deep-equals one of the listed members — value ∈ your list, not the reverse), `verify-cassettes` incl. the `--allow-domain`/`--allow-email`/`--allow-patterns-file` allows (`--allow-patterns-file <path>` is a FILE of patterns, one regex per line — not a path to allow, unlike `--allow <regex>`), batch `record <dir>`/`--rerecord-stale`, `record --concurrency <N>`, record-time redaction, multiSelect/`answer:`, `verify-run` answer-coverage, `record --max-artifact-bytes`, live record-time deciders, scenario `skills:` staleness scoping with `COWORK_HARNESS_AGENT_SCOPE=skill`, `chat --plugin`, and `/help` in the REPL.
47
47
  - **0.21.0:** `verify-cassettes --allow-path` (`path` — local absolute filesystem paths — is the scanner's 4th class), and `hostloop`'s native host/VM process split with its `allow_host_writes:` consent field.
@@ -53,6 +53,8 @@ Before the first command, confirm the CLI is reachable and **fail loud** (never
53
53
  - **0.33.0:** the `redacted` marker on display-omitted reasoning — `subagents[].reasoning` and the top-level `thinking[]` now carry `{text:"", redacted:true}` when the model returns a signed-but-empty thought (so "reasoned, text omitted" is distinct from "no thought"), plus the fenced `debug.thinking_display` escape hatch.
54
54
  - **1.0.0:** first stable release — the SPEC §12 compatibility contract takes effect (covered CLI/schema/env/Action surfaces are now stable; breaking changes need a major bump). No new author-facing command; the floor simply tracks the 1.0 release.
55
55
  - **1.1.0:** `analyze-skill` now also flags **interactive-artifact write-backs lost under Cowork** — a relative `fetch`/XHR/`sendBeacon`/`<form method=post>` in an emitted `.html` (or its `.py`/`.js` generator) that silently fails when the artifact is served from Cowork's own origin. `artifact-write-back-lost` gates under `--strict`; `artifact-write-back-suspect` is advisory; an unanalyzable candidate is a could-not-verify exit 3. An optional **`analyze-skill --runtime`** drives the artifact in a headless DOM (needs `jsdom`) to *observe* the lost write-back — enrichment only, never changes the exit code. Plus a `lint` check for a container-only assertion key (`no_scratchpad_leak`/`present_files_called`) used off the `container` tier, and the `doctor --output-format json` envelope frozen as a covered SPEC §12 surface (`schema/doctor.json`).
56
+ - **1.2.0:** three new assertion keys — `no_lost_write_back: true` (the write-back detector above wired as a per-scenario gate over the run's authored files, live-only) and the regex siblings `tool_result_matches`/`tool_result_not_matches` (case-insensitive per-result, for an error-signature *family* a literal substring can't express). **`microvm` outputs are now observable** — its session tree is snapshotted from the VM into the run dir, so `file_exists`/`artifact_json`/`user_visible_artifact`/`no_unexpected_files`/`input_unmodified`/`no_lost_write_back`/`semantic_matches` all work there, no longer `container`/`hostloop`-only. `status <dir>` also resolves the newest session under a `--run-dir` root; the completion footer prints a `→ result: …/result.json` pointer; the `on_unanswered=fail` error also points to `on_unanswered: llm`. `analyze-skill` hardening: a phantom `<script>` prose block no longer sinks a real verdict, a delete/remove flow claiming success classifies as lost (error) not just suspect, and write-back detection widened (optional-call `?.` spellings, member-spelled/aliased `fetch`/`sendBeacon`, axios instance/config forms).
57
+ - **1.3.0:** run-identity for the iterate-across-fixes loop — `skill`/`run` take `--label <tag>` (a generation tag surfaced in `result.json` `runLabel`, the run-index row, `inspect`, and `status.json`) alongside an auto-recorded `skillCommit`, layered on the **authoritative** content-exact `fingerprint.skillHash` a harvest step should pair critiques by (`inspect`/index surface a short `skillHash` prefix). `trace --full-results` captures the full input+result of **every** tool call — successful ones too, not just errors — so an external grader can ground a self-critique finding against the call it cites. `verify-run` now **warns** on skillHash drift for answer-less scenarios (was silent unless the scenario declared scripted `answers`). Plus `skill --allow-missing-capability` (the open-ended-run opt-out for a capability FALSE-NEGATIVE on the lean `core` image), a new **warn-severity `ended_with_question`** verdict signal (the agent's final answer contains a question and the run wrote no `outputs/` deliverable — the lenient sibling of the strict `stalled`), and LLM-decider `OTHER:` free-text answers now marked `[via Other free-text]` in gate provenance.
56
58
  - **Agent binary (sandboxed live tiers — `container`/`microvm`/`hostloop`/`cowork`).** The staged Claude Code agent is **bind-mounted** from a local Claude Desktop install, or point `COWORK_AGENT_BINARY` at a `claude-code-vm/<ver>/claude` ELF. Nothing is bundled. `protocol` (L0) and `replay` need no staged agent; for the sandboxed tiers, no agent → no run; report that, don't skip silently.
57
59
  - **Docker / Lima.** Only `--fidelity protocol` (L0) runs without them. `container` / `microvm` / `hostloop` / `cowork` need Docker (Lima for L2). If they're absent, drop to `--fidelity protocol` and **say so** — a green that never exercised the sandbox is not a sandbox pass.
58
60
  - **Auth.** `CLAUDE_CODE_OAUTH_TOKEN` (preferred), or `ANTHROPIC_API_KEY` / `ANTHROPIC_AUTH_TOKEN`, via env or `.env`. Minting an OAuth token needs the **`claude` CLI** (`npm i -g @anthropic-ai/claude-code`, then `claude setup-token`).
@@ -221,7 +223,8 @@ them by what you're trying to prove:
221
223
  | a skill actually **ran** (or must NOT) | `skill_triggered: <regex>`, `no_skill_triggered: <regex>` |
222
224
  | a tool ran **inside** a skill's scope | `skill_tool_used: {skill, tool}` |
223
225
  | a sub-agent did the work | `subagent_output_contains: {contains}`, `subagent_dispatched: <regex>`, `dispatch_count_max: <N>` |
224
- | a pre-existing input wasn't mutated (incl. `uploads/**`) | `input_unmodified: <glob>` or `[<glob>, …]` (live/verify-run; not microvm) |
226
+ | a pre-existing input wasn't mutated (incl. `uploads/**`) | `input_unmodified: <glob>` or `[<glob>, …]` (live/verify-run) |
227
+ | no authored interactive artifact silently loses its Submit under Cowork | `no_lost_write_back: true` (**live-only**; static Tier A over the run's authored `.html`/`.py`/`.js`; per-scenario gate for the same class `analyze-skill` scans) |
225
228
  | a resource ceiling held | `max_peak_rss_bytes: <N>` (**live-only**) |
226
229
  | a hook blocked / didn't block a tool | `hook_blocked: <regex>`, `no_hook_blocked: true` (replay needs a `controlOut` cassette) |
227
230
  | every MCP round-trip succeeded | `no_mcp_error: true` (**live-only**) |
@@ -309,8 +312,10 @@ your `answers:` against that kept run, then record once. **But the kept run is a
309
312
  skill's gate phrasing afterward, re-`--keep` — verify-run's answer-coverage *refuses* (exit 2, "predates the
310
313
  current skill") rather than vouch against stale labels, but the trace/inspect path can't warn you, so re-keep
311
314
  deliberately. (Same fail-closed family: corrupt gate evidence — unparseable `events.jsonl` lines, or fewer
312
- gates than `trace.json` recorded questions — and a structurally invalid `result.json` also refuse rather
313
- than certify.) (A token-free probe of "which gates fire" isn't possible — gates are model-decided per run.)
315
+ gates than `trace.json` recorded questions — a structurally invalid `result.json`, a `command:"replay"`
316
+ result (a replay is a re-check of a recorded cassette, not run evidence — verify the original live run dir),
317
+ and a `mode:"chat"` result (chat carries no assertions or verdict by contract) also refuse rather than
318
+ certify.) (A token-free probe of "which gates fire" isn't possible — gates are model-decided per run.)
314
319
 
315
320
  Run artifacts are written to `~/.cowork-harness/runs/…` by default — **outside any working tree**, so a run
316
321
  launched from a repo root never drops sensitive skill inputs/outputs into it. Pass `--run-dir <path>` (or set
@@ -352,6 +357,15 @@ cassette — has its own recipe:
352
357
  unless the scenario asserts `allow_missing_capability: true`, which downgrades it to a notice and
353
358
  proceeds. Rebuild with `--build-arg COWORK_FULL_PARITY=1` and point `COWORK_AGENT_IMAGE` at it for those
354
359
  skills.
360
+ 6. **Iterate across fixes — verify before you trust, and don't cross-pair generations.** A green run is
361
+ not a correct run, and a skill's self-reported finding is not real until its cited evidence is found in
362
+ the run's own output. Ground each finding against `result.json` (`finalMessage` = the skill's own
363
+ answer/critique; `toolResults` = tool outputs) and the tool-call stream via
364
+ `cowork-harness trace <run-dir> --output-format json` — add `--full-results` so a successful call's full
365
+ input + result are captured, not just errored ones. When iterating, tag generations with `--label` and
366
+ pair a critique only with a `result.json` whose `fingerprint.skillHash` **matches** the skill that
367
+ produced it (`inspect`/the run-index row surface a short `skillHash` prefix; `verify-run` warns when a
368
+ kept run predates the current skill). See `docs/debugging.md` (repo-only) for the full loop.
355
369
 
356
370
  #### Interpreting verdict signals
357
371
 
@@ -362,6 +376,42 @@ The run verdict may include `WARN`-severity signals in addition to pass/fail. On
362
376
  the run can still green. If you see it, fix the asset path — a green with a missing asset is
363
377
  not a valid pass.
364
378
 
379
+ **False negatives — signals that are tier/image artifacts, not skill defects.** Some fail-severity
380
+ signals read like a skill gap but are really a property of the reduced test image or the fidelity tier.
381
+ Recognize these before "fixing" a non-bug:
382
+
383
+ - **`missing_capability`** — the lean `core` agent image is a deliberate partial mirror of real Cowork's
384
+ rootfs, so a skill that used `soffice`/LibreOffice (`office_convert`), `tesseract` (`ocr`),
385
+ `markitdown`/`magika` (`ml_extract`), `cv2` (`cv`), `camelot`/`tabula` (`pdf_tables`), or `wand`
386
+ (`magick`) can trip this even though real Cowork **ships** those. The message says so ("likely a FALSE
387
+ NEGATIVE (real Cowork ships them)"). Fix: rebuild full parity (`--build-arg COWORK_FULL_PARITY=1`, point
388
+ `COWORK_AGENT_IMAGE` at it), or — if the skill's fallback is genuinely equivalent — assert
389
+ `allow_missing_capability: true`. (Two sources: a skill *observed using* an omitted family, live lane;
390
+ or a declared `requires_capabilities` the tier can't provide, both lanes — an unknown family name
391
+ hard-fails rather than silently passing.) **On an open-ended `skill` run** (no `assert:` block to carry
392
+ the modifier), pass **`--allow-missing-capability`** — the CLI equivalent of the assertion.
393
+ - **`ended_with_question`** (`WARN`, live lane) — a heuristic: the agent's final answer contains a
394
+ question and the run wrote **no deliverable to `outputs/`** — it may have ended on a request for input
395
+ instead of finishing. Warn-only; the fix is scripting/steering the answer (`answer:` / `--answer` / a
396
+ decider, or `--decider-llm --intent`), not editing the skill's prose. The strict, fail-severity sibling
397
+ `stalled` already catches a *trailing*-`?` final turn that did no tool work after the last gate; this
398
+ covers the residual (a mid-message `?`, or tool work after the last gate that still ended asking). Read
399
+ the final message before acting — a legitimate question-posing answer that wrote a file never fires.
400
+ Assert `allow_stall: true` if ending on a question is the intended terminal state.
401
+ - **`host_path_leak`** — skipped at **`hostloop` and `protocol`** fidelity (the agent runs on real host
402
+ paths there, so a host path in model-visible text is expected, not a leak); it is *armed* at
403
+ `container`/`microvm`, but only *fires* on an actual scanned leak with no authored
404
+ `transcript_no_host_path`. At `fidelity: cowork` the skip follows the **resolved** tier, so a `cowork`
405
+ run that lands on `container` is armed. Author `transcript_no_host_path` to enforce cleanliness where
406
+ it's valid.
407
+ - **`scan_unavailable`** (`WARN`) — emitted only on the live lane: `events.jsonl` was missing/corrupt, so
408
+ `RunResult.scan` is undefined and the host-path + outputs-delete guards **did not run this run**. Not a
409
+ pass or a defect — assert `no_delete_in_outputs` / `transcript_no_host_path` to hard-fail on it instead.
410
+
411
+ The full 14-code signal table (severity + per-signal opt-out) is in
412
+ [`references/scenario-schema.md`](./references/scenario-schema.md); `docs/scenario.md` (repo-only) carries
413
+ the fuller narrative.
414
+
365
415
  ### Checking whether a background run is alive
366
416
 
367
417
  Never use `ps aux` to check on a `cowork-harness` run you launched in the background — it only sees
@@ -374,7 +424,9 @@ harness writes/updates throughout the run's lifecycle (including a crash-safety
374
424
  error/`SIGTERM`, AND staleness detection for a hard `SIGKILL`/OOM-kill that no exit handler can catch —
375
425
  either way you get `"error"`/`stale` instead of a permanently-trusted `"running"`), so liveness is
376
426
  checkable regardless of PID namespace. The harness prints `[status] <outDir>` to stderr as soon as the
377
- run starts, so capture stderr to get the exact directory. `--follow` fails loud on a timeout/staleness
427
+ run starts, so capture stderr to get the exact directory — but `<dir>` also accepts the run-dir root
428
+ passed to `--run-dir` (a directory without its own `status.json`): it scans up to two levels down for the
429
+ newest session's `status.json` and reads that. `--follow` fails loud on a timeout/staleness
378
430
  rather than hanging forever. (Fuller recipe in `docs/run-status.md` — repo-only, not in the installed
379
431
  payload; `cowork-harness status --help` has the flags.)
380
432
 
@@ -468,7 +520,11 @@ decide which assertions from *Assertions: two orthogonal axes* are worth adding)
468
520
  - **Debugging a wrong Cowork UI panel.** Each panel is reconstructed in `result.json`: **Progress** =
469
521
  `tasks[]`, **Working folder** = `workspaceFiles[]` (classified output/mount/input, with a
470
522
  `trace --view files` diff), **Context / Connectors** = `context` (tools / mcpServers / availableSkills),
471
- **Scratch-pad → outputs** = `presentedFiles[]`. If a panel looks wrong in a run, read its field.
523
+ **Scratch-pad → outputs** = `presentedFiles[]`. If a panel looks wrong in a run, read its field. An
524
+ **absent** `workspaceFiles`/`artifacts` (a replay result, or a run whose workspace root was missing at
525
+ collection) is evidence **UNAVAILABLE**, not an empty run — `trace --view files` reports a loud
526
+ UNAVAILABLE marker (`workspaceFilesRecorded: false` in JSON, and no phantom "removed" diff rows) and
527
+ `inspect` prints `artifacts: UNAVAILABLE` (`artifactsRecorded: false`) instead of `artifacts (0):`.
472
528
 
473
529
  ### Debugging with `chat`
474
530
 
@@ -611,9 +667,11 @@ repeats the assertion/replay-relevant ones alongside the schema (a scoped subset
611
667
  - `on_unanswered` governs **unanswered** `AskUserQuestion` gates; the `stalled` signal covers
612
668
  stalling *after* one is answered — two different failure modes.
613
669
  - **Free-text aside:** a "type-it-in-notes" option has **no scripted deterministic answer** today
614
- (the `OTHER:` directive works only on the LLM-decider path, not scripted `choose:`; on an
615
- options-bearing gate a bare out-of-set LLM answer fails loud (exit 2) — see the LLM-decider
616
- free-text note in `references/fidelity-and-answers.md`).
670
+ (the `OTHER:` directive works only on the LLM-decider path, not scripted `choose:`, and only on
671
+ **single-select** gates — a **multi-select** gate is index-only, so `OTHER:` fails loud there; on an
672
+ options-bearing single-select gate a bare out-of-set LLM answer also fails loud (exit 2) — see the
673
+ LLM-decider free-text note in `references/fidelity-and-answers.md`). An LLM decision answered via
674
+ `OTHER:` is marked `[via Other free-text]` in its `gateProvenance` rationale.
617
675
  14. **A positional `choose` (`first` / index) is order-dependent.** `choose: "2"` survives label drift
618
676
  but NOT option *re-ordering* — if the gate presents its options in a different order run-to-run, the
619
677
  index lands on a different option (a silent re-record flake). Prefer an exact label when order is
@@ -1,6 +1,6 @@
1
1
  # CI recipe — replay vs live lanes
2
2
 
3
- Self-contained reference. Tracks `cowork-harness 1.1.0` (baseline `desktop-1.20186.1`).
3
+ Self-contained reference. Tracks `cowork-harness 1.3.0` (baseline `desktop-1.20186.1`).
4
4
 
5
5
  **Fastest path: the packaged Action.** One step gets you `replay`/`lint`/`verify-cassettes` plus a PR
6
6
  job-summary reporter (verdict table, staleness findings, cost/turns when available):
@@ -13,7 +13,7 @@ job-summary reporter (verdict table, staleness findings, cost/turns when availab
13
13
  ```
14
14
 
15
15
  The Action's `version` input defaults to `latest` — intentional so a copy-pasted recipe tracks the current
16
- release; pin an exact version (e.g. `version: "1.1.0"`) for reproducible CI.
16
+ release; pin an exact version (e.g. `version: "1.3.0"`) for reproducible CI.
17
17
 
18
18
  Reach for the manual multi-step form below only when you need per-step control the Action's inputs don't
19
19
  cover (a custom flag combination, a different runner matrix per step, or `lint`/`verify-cassettes` gated
@@ -57,7 +57,7 @@ sha256-*checked* but not hard-blocking on mismatch — it's advisory for an inte
57
57
  GitHub-hosted runners, no token/Docker/agent:
58
58
 
59
59
  ```yaml
60
- - run: npm i -g "cowork-harness@>=1.1.0"
60
+ - run: npm i -g "cowork-harness@>=1.3.0"
61
61
  - run: cowork-harness lint scenarios/*.yaml # no silent false-greens
62
62
  - run: cowork-harness verify-cassettes cassettes/ # privacy + staleness
63
63
  - run: cowork-harness replay cassettes/ # token-free content/structure
@@ -197,7 +197,7 @@ jobs:
197
197
  with: { node-version: '20' }
198
198
  - uses: actions/setup-python@v5
199
199
  with: { python-version: '3.x' } # python3 only — PyYAML is bundled with the linter
200
- - run: npm i -g "cowork-harness@>=1.1.0"
200
+ - run: npm i -g "cowork-harness@>=1.3.0"
201
201
  - run: cowork-harness lint scenarios/*.yaml # no-silent-false-green (needs python3; PyYAML bundled)
202
202
  - run: cowork-harness verify-cassettes cassettes/ --output-format json # privacy + staleness gate
203
203
  - run: cowork-harness replay cassettes/ --output-format json # token-free content/structure
@@ -226,7 +226,7 @@ jobs:
226
226
  echo "live=true" >> "$GITHUB_OUTPUT"
227
227
  fi
228
228
  - if: steps.guard.outputs.live == 'true'
229
- run: npm i -g "cowork-harness@>=1.1.0"
229
+ run: npm i -g "cowork-harness@>=1.3.0"
230
230
  - if: steps.guard.outputs.live == 'true'
231
231
  run: cowork-harness run scenarios/ --output-format json
232
232
  env:
@@ -1,6 +1,6 @@
1
1
  # Fidelity tiers & answer paths
2
2
 
3
- Self-contained reference. Tracks `cowork-harness 1.1.0` (baseline `desktop-1.20186.1`).
3
+ Self-contained reference. Tracks `cowork-harness 1.3.0` (baseline `desktop-1.20186.1`).
4
4
 
5
5
  ## Fidelity tiers (`fidelity:` in the scenario)
6
6
 
@@ -130,7 +130,10 @@ hands `fn` exactly this dict.
130
130
  ### Determinism contract
131
131
 
132
132
  - `fail` — the default for `run`. On an unscripted gate it hard-errors; the error names the exact
133
- `--answer`/`choose` to add. Correct, but flaky for skills whose gates appear stochastically.
133
+ `--answer`/`choose` to add, and also suggests `on_unanswered: llm` (in the scenario YAML) as a
134
+ secondary escape valve for a gate whose wording drifts run-to-run — a regex chases a moving target,
135
+ at the cost of non-determinism (one model call per gate). Correct, but flaky for skills whose gates
136
+ appear stochastically.
134
137
  - `first` — picks option 1 and warns loudly. **Flagged `nonDeterministic`** — not a deterministic
135
138
  substitute for scripted answers. (For a web_fetch approval gate it abstains → fail-closed.)
136
139
  - `prompt` — asks at the TTY (`skill` only).
@@ -161,10 +164,13 @@ Caveat: `decide` only builds a **single-select** sample (set choices with `--opt
161
164
  multiSelect flag), so its printed request shows `options[].label` but never `multiSelect:true` — to
162
165
  exercise the array reply path, run a real multiSelect gate or unit-test the helper directly.
163
166
 
164
- LLM-decider free-text goes via `OTHER: <value>` on an **options-bearing** gate; a bare out-of-set
165
- answer (no matching label, no `OTHER:`) fails loud (`UnansweredError` → exit 2) — it never stalls or
166
- guesses an option. Open-ended (no-option) gates need no `OTHER:` prefix: free text is delivered
167
- verbatim. (Scripted scenarios use the separate `answer:` escape hatch.)
167
+ LLM-decider free-text goes via `OTHER: <value>` on an **options-bearing single-select** gate; a bare
168
+ out-of-set answer (no matching label, no `OTHER:`) fails loud (`UnansweredError` → exit 2) — it never
169
+ stalls or guesses an option. A **multi-select** gate is **index-only**: it accepts comma-separated option
170
+ numbers, and `OTHER:` fails loud there (no free-text escape on that path). Open-ended (no-option) gates
171
+ need no `OTHER:` prefix: free text is delivered verbatim. A decision answered via any free-text path is
172
+ marked `[via Other free-text]` in its `gateProvenance` rationale, so a `result.json` consumer can tell it
173
+ from an offered-option pick. (Scripted scenarios use the separate `answer:` escape hatch.)
168
174
 
169
175
  A gate that fails loud (`on_unanswered: fail`, the default) still **salvages a PARTIAL run**: the harness
170
176
  writes a `result.json` (marked `partial: true`) with the artifacts the agent produced before the whiff, so
@@ -180,7 +186,7 @@ up often enough to spell out:
180
186
  a filesystem or network — it re-evaluates assertions from the frozen cassette. A fixed set of
181
187
  keys is live-only and **skipped outright** on replay (absent from `assertions[]`, not vacuously
182
188
  passed): `no_delete_in_outputs`, `self_heal_ran`, `transcript_no_host_path`, `egress_denied`,
183
- `egress_allowed`, `no_mcp_error`, `max_peak_rss_bytes`, `semantic_matches`, and `expect_denied`.
189
+ `egress_allowed`, `no_mcp_error`, `max_peak_rss_bytes`, `semantic_matches`, `no_lost_write_back`, and `expect_denied`.
184
190
  Everything else that *is* evaluated is checked against the **recording**, not fresh behavior — a
185
191
  green replay says the skill produced these events when it was recorded, not that it still does
186
192
  (`staleness[]` flags skill/baseline drift as a hint; only a live `run` re-confirms current
@@ -1,6 +1,6 @@
1
1
  # Scenario & session schema, assertion catalog, web_fetch, full gotchas
2
2
 
3
- Self-contained reference for authoring `cowork-harness` scenarios. Tracks `cowork-harness 1.1.0`
3
+ Self-contained reference for authoring `cowork-harness` scenarios. Tracks `cowork-harness 1.3.0`
4
4
  (baseline `desktop-1.20186.1`). If your checkout is newer, prefer the live `docs/scenario.md`,
5
5
  `docs/session.md`, and `SPEC.md`.
6
6
 
@@ -251,13 +251,16 @@ same set live from the schema.
251
251
  | `file_exists: <path>` | the path exists under the run's `work/` (anchored at `mnt/`, e.g. `outputs/x.md`). For a user-facing deliverable prefer `user_visible_artifact` — with a connected folder the file lands in `mnt/<folder>` (= `{{workspaceFolder}}`), not `mnt/outputs`, so `file_exists: outputs/x.md` misses it |
252
252
  | `user_visible_artifact: <path>` | exists **and** under a user-visible root (`outputs/` + each connected folder's mount name) — the right primitive for a workspace deliverable when a folder is connected |
253
253
  | `no_delete_in_outputs: true` | no delete op touched `mnt/outputs` — **only `true` is valid**; `false` is rejected (omit to allow deletes) |
254
- | `no_unexpected_files: [<glob>, …]` | every **newly created** file under a user-visible root matches ≥1 glob (workRoot-relative paths; `**` = whole path segment for any depth — use `outputs/handoff/**` for per-run subdirs); `[]` = no new files; **new-files-only** — overwriting a pre-existing file in place is invisible (use content-level producer stamping); live/verify-run without a pre-run manifest ⇒ evidence-unavailable hard-fail (live runs capture the baseline only when this key is asserted; recordings always capture); an **incomplete post-run filesystem walk** (an unreadable subtree — a permission/I-O error) also fails evidence-unavailable rather than reporting "no strays" over a partial tree — distinct from the missing-manifest/microvm case, which fails for a different reason; **microvm cannot capture** (use container/hostloop); replay needs `cassette.preRunPaths` (≥0.24 container/hostloop recordings) — cassettes without it **exclude** the key with a loud warning |
255
- | `input_unmodified: <glob>` or `[<glob>, …]` | a single glob or a list; every **pre-existing** file (incl. uploaded files under `uploads/**`) whose workRoot-relative path matches ≥1 glob keeps an unchanged content hash after the run — the in-place-mutation companion to `no_unexpected_files`'s new-files check (`[]` is rejected by the schema — list at least one glob); a matched file that was deleted counts as a content change (fails); live/verify-run without a pre-run hash manifest ⇒ evidence-unavailable hard-fail; **microvm cannot capture**; replay needs `cassette.preRunHashes` — cassettes without it **exclude** the key with a loud warning; on replay it compares against the manifest's recorded `sha256`, never a re-hash of the materialized tree |
254
+ | `no_unexpected_files: [<glob>, …]` | every **newly created** file under a user-visible root matches ≥1 glob (workRoot-relative paths; `**` = whole path segment for any depth — use `outputs/handoff/**` for per-run subdirs); `[]` = no new files; **new-files-only** — overwriting a pre-existing file in place is invisible (use content-level producer stamping); live/verify-run without a pre-run manifest ⇒ evidence-unavailable hard-fail (live runs capture the baseline only when this key is asserted; recordings always capture); an **incomplete post-run filesystem walk** (an unreadable subtree — a permission/I-O error) also fails evidence-unavailable rather than reporting "no strays" over a partial tree — distinct from the missing-manifest case (a `--resume` run), which fails for a different reason; captured on every live sandbox tier including microvm (its outputs are snapshotted from the VM into the run dir); replay needs `cassette.preRunPaths` (≥0.24 recordings) — cassettes without it **exclude** the key with a loud warning |
255
+ | `input_unmodified: <glob>` or `[<glob>, …]` | a single glob or a list; every **pre-existing** file (incl. uploaded files under `uploads/**`) whose workRoot-relative path matches ≥1 glob keeps an unchanged content hash after the run — the in-place-mutation companion to `no_unexpected_files`'s new-files check (`[]` is rejected by the schema — list at least one glob); a glob that matches **no** pre-run path fails loud (a typo or renamed mount would otherwise verify zero files and pass vacuously); a matched file that was deleted counts as a content change (fails); live/verify-run without a pre-run hash manifest ⇒ evidence-unavailable hard-fail (a `--resume` run); captured on every live sandbox tier including microvm; replay needs `cassette.preRunHashes` — cassettes without it **exclude** the key with a loud warning; on replay it compares against the manifest's recorded `sha256`, never a re-hash of the materialized tree |
256
256
  | `self_heal_ran: <bool>` | a plugin-root self-heal script was (not) invoked |
257
+ | `no_lost_write_back: true` | fails if the run authored an interactive HTML artifact (or a `.py`/`.js` generator of one) whose **relative** Submit/POST write-back is lost under Cowork (served from Cowork's own origin → resolves non-ok, a "Saved!" is silently false). Runs the shipped **static Tier A** analyzer over the files the run authored (diffed vs the pre-run manifest). A lost write-back on an **added** agent-authored source (`outputs/`, scratchpad) **fails**; a **pre-existing** file the skill only modified on a read-write mount is **advisory**; `-suspect` findings surface but pass. **Only `true` is valid**. **Live/verify-run only** — skipped-loud on replay; runs on every live sandbox tier including **microvm** (its outputs are snapshotted from the VM into the run dir); could-not-verify (fail-closed) on a `--resume` scratchpad or an unanalyzable candidate |
257
258
  | `tool_called: <glob>` | a tool the agent ran matched this **glob** — `*` = any run, `?` = one char, exact when literal, anchored + case-sensitive. Exact name (`Write`) matches only that tool; `mcp__workspace__*` matches any workspace tool. GLOB, not regex (`.` is literal) — an empty glob, or one containing a regex/brace-expansion metacharacter (`.*`, `.+`, `\|`, `()`, `[]`, `+`, `^`, `$`, `{}`, `\d`/`\w`/`\s`/`\b`), is **rejected at load** (a hard schema error, not a runtime warning) — it would match no real tool name and pass a `_not_`/`_absent` assert vacuously. Applies whether the glob comes from an authored scenario or a recorded cassette's frozen assert. The bundled `scenario.py lint` does NOT perform this check — only the harness enforces it, at actual load (`run`/`skill`/`record`) |
258
259
  | `tool_not_called: <glob>` | NO tool the agent ran matched this glob (`mcp__*` = "no MCP tool ran"). Same glob semantics as `tool_called`, including the empty/regex-ish rejection |
259
260
  | `tool_result_contains: <str>` | a tool result includes the literal string (content / replay-checkable — substring match) |
260
261
  | `tool_result_not_contains: <str>` | no tool result includes the literal string (content / replay-checkable; fails loud when tool results are absent) |
262
+ | `tool_result_matches: <regex>` | the regex sibling of `tool_result_contains` — case-insensitive regex matches at least one tool result; use for an error-signature family, not just one literal string |
263
+ | `tool_result_not_matches: <regex>` | the regex sibling of `tool_result_not_contains` — same fails-loud-on-absent-evidence semantics |
261
264
  | `subagent_tool_used: <glob>` | a sub-agent used a tool matching this glob (same `*`/`?`, anchored, case-sensitive semantics as `tool_called`, including the empty/regex-ish rejection) |
262
265
  | `subagent_tool_absent: <glob>` | no sub-agent used a tool matching this glob (same rejection) |
263
266
  | `no_vm_path_file_op: true` | **`fidelity: hostloop` only** — NO gated file tool attempted a `/sessions`(-prefixed) path (`RunResult.fileToolAttempts`) — content-class, replay-checkable without `controlOut`; any other tier FAILS "cannot verify" (`/sessions/...` is valid there). **Only `true` is valid** |
@@ -312,7 +315,7 @@ same set live from the schema.
312
315
  | `egress_allowed: <host>` | the host was allowed through |
313
316
  | `no_mcp_error: true` | no MCP round-trip failed (`RunResult.mcpErrors` is empty — no unhandled server, no handler throw) — live-only: MCP round-trips are harness-computed, not in the SDK stdout stream, so evidence-unavailable on replay (never a vacuous pass). **Only `true` is valid** |
314
317
  | `max_peak_rss_bytes: <N>` | peak sampled RSS of the agent sandbox ≤ N bytes (`RunResult.resources.peakRssBytes`) — live-only: replay never spawns a sandbox to sample, so evidence-unavailable on replay/protocol (never a vacuous pass); also evidence-unavailable when sampling captured no RSS value |
315
- | `semantic_matches: {rubric: [...], min_pass?, judge_model?}` | a pinned LLM judge grades each fixed `rubric` claim against the run's answer — the **union of the agent's final result text (`RunResult.finalMessage`), the transcript, and the final on-disk content of any files the agent authored during the run** — so a claim about content the skill led the agent to *write to a file* grades as reliably as one about inlined prose (authored-file evidence is unavailable on the **microvm** tier — no pre-run manifest to diff against, so no files are captured; `container`/`hostloop` do capture them). Beyond the microvm case, when the authored-file evidence backing the judged document is **incomplete** — a file dropped at the capture-size cap, unreadable at read-back, or (on `--resume`) the scratchpad walk skipped — the assert fails evidence-unavailable rather than trusting a judge grade over a partial document; this is separate from the malformed-grade `judgeInvalid` path below. The assert passes iff ≥ `min_pass` claims pass (default: all — avoid for a gating scenario). Results align by claim index and are recorded per-claim in `RunResult.assertions[].semanticClaims` (`[{index, claim, pass}]`, so a consumer can diff the per-claim profile across runs); a rep whose grade can't be parsed (after one retry) is marked `RunResult.assertions[].judgeInvalid` and **never silently dropped** — it is excluded from the pass denominator, and the guard against a misleading score from that exclusion is the gate's minimum-valid-rep floor (`MIN_VALID` ≥ 4) plus this visibility, not a claim that denominator-shrinking inflation is impossible. Within a rep, a grade that's still unparseable after the retry **fails that assert outright** (evidence-unavailable, not a vacuous pass) — a persistently-flaky judge reds the run rather than silently passing. `judge_model` pins the grader (default when neither it nor `COWORK_HARNESS_JUDGE_MODEL` is set: `claude-opus-4-8`; a dated id keeps a before/after comparison reproducible). Live-only: the judge is a live model call, so evidence-unavailable / skipped-loud on replay (never a vacuous pass) |
318
+ | `semantic_matches: {rubric: [...], min_pass?, judge_model?}` | a pinned LLM judge grades each fixed `rubric` claim against the run's answer — the **union of the agent's final result text (`RunResult.finalMessage`), the transcript, and the final on-disk content of any files the agent authored during the run** — so a claim about content the skill led the agent to *write to a file* grades as reliably as one about inlined prose (authored-file evidence is captured on every live sandbox tier including **microvm** — its session tree is snapshotted from the VM into the run dir). When the authored-file evidence backing the judged document is **incomplete** — a file dropped at the capture-size cap, unreadable at read-back, or (on `--resume`) the scratchpad walk skipped — the assert fails evidence-unavailable rather than trusting a judge grade over a partial document; this is separate from the malformed-grade `judgeInvalid` path below. The assert passes iff ≥ `min_pass` claims pass (default: all — avoid for a gating scenario). Results align by claim index and are recorded per-claim in `RunResult.assertions[].semanticClaims` (`[{index, claim, pass}]`, so a consumer can diff the per-claim profile across runs); a rep whose grade can't be parsed (after one retry) is marked `RunResult.assertions[].judgeInvalid` and **never silently dropped** — it is excluded from the pass denominator, and the guard against a misleading score from that exclusion is the gate's minimum-valid-rep floor (`MIN_VALID` ≥ 4) plus this visibility, not a claim that denominator-shrinking inflation is impossible. Within a rep, a grade that's still unparseable after the retry **fails that assert outright** (evidence-unavailable, not a vacuous pass) — a persistently-flaky judge reds the run rather than silently passing. `judge_model` pins the grader (default when neither it nor `COWORK_HARNESS_JUDGE_MODEL` is set: `claude-opus-4-8`; a dated id keeps a before/after comparison reproducible). Live-only: the judge is a live model call, so evidence-unavailable / skipped-loud on replay (never a vacuous pass) |
316
319
  | `artifact_json: {artifact, path, …}` | assert a JSON artifact's contents — `equals`/`gt`/`in`/`exists`/`absent`/`is_null` over a dotted `path` (`in` = membership in a list, for a stochastic/LLM value; `absent` ≠ `is_null`; an unresolved intermediate fails loud) |
317
320
  | `computer_links_resolve: true` | every `computer://` link in the model-visible transcript resolves to an artifact that exists in the run's collected outputs/mounts — a dangling link fails, naming which target was checked (a live host path, the collected work tree, or the replay manifest). **Requires ≥1 link** (zero links fails — use `computer_links_resolve_if_present` for the presence-free variant). **Only `true` is valid** (`false` is rejected by the schema) |
318
321
  | `computer_links_resolve_if_present: true` | like `computer_links_resolve` but passes vacuously when the transcript has zero `computer://` links — the presence-free variant. **Only `true` is valid** |
@@ -331,7 +334,7 @@ dotted path.
331
334
 
332
335
  **VerdictSignals in `result.verdict.signals`:** `computeVerdict` pushes signals into `result.verdict.signals`; most
333
336
  are **fail**-severity (they flip the run's pass/exit code even though `result.result` itself stays
334
- `"success"`) and only three are **warn**-severity (informational, never flip pass/fail). Current signal
337
+ `"success"`) and only four are **warn**-severity (informational, never flip pass/fail). Current signal
335
338
  codes (`VerdictSignal["code"]` in `src/run/verdict.ts`):
336
339
 
337
340
  | Code | Severity | Meaning |
@@ -344,16 +347,17 @@ codes (`VerdictSignal["code"]` in `src/run/verdict.ts`):
344
347
  | `outputs_delete` | fail | An unauthorized delete touched `mnt/outputs` (opt out: author `no_delete_in_outputs`) |
345
348
  | `host_path_leak` | fail | A host path leaked into model-visible text (opt out: author `transcript_no_host_path`) |
346
349
  | `l0_plugin_divergence` | fail | L0/protocol plugin loading diverged from Cowork (opt out: `allow_l0_plugin_divergence`) |
347
- | `missing_capability` | fail | A `requires_capabilities` need was unmet, or the skill used a capability the image omits (opt out: `allow_missing_capability`) |
350
+ | `missing_capability` | fail | A `requires_capabilities` need was unmet, or the skill used a capability the image omits (opt out: `allow_missing_capability`, or `skill --allow-missing-capability` on an open-ended run) |
348
351
  | `infra_error` | fail | A VM/egress sidecar crashed mid-run — not author-suppressible |
349
- | `stalled` | fail | The run ended on an unanswered question (opt out: `allow_stall`) |
352
+ | `stalled` | fail | The run ended on an unanswered question, or on a trailing-`?` final turn with no tool work after the last gate (opt out: `allow_stall`) |
350
353
  | `non_deterministic` | warn | The run was LLM/external/human-decided — not reproducible |
351
354
  | `prompt_asset_missing` | warn | The run proceeded with a missing prompt asset (`COWORK_HARNESS_ALLOW_MISSING_PROMPT=1`); fidelity is degraded |
352
355
  | `scan_unavailable` | warn | Post-run scan evidence unavailable (`RunResult.scan` undefined) — the host-path and outputs-delete guards did not run this run |
356
+ | `ended_with_question` | warn | Live-lane heuristic: the final answer contains a question and the run wrote no deliverable to `outputs/` — the lenient sibling of `stalled` (covers a mid-message `?`, or tool work after the last gate that still ended asking). Opt out: `allow_stall` |
353
357
 
354
358
  A **fail**-severity signal does not change `result.result` (still `"success"`), but it DOES fail the
355
359
  overall run verdict and exit code — `assert result: success` alone won't catch it; check
356
- `result.verdict.signals[].severity` or the run's exit code. Only the three **warn** codes are truly benign.
360
+ `result.verdict.signals[].severity` or the run's exit code. Only the four **warn** codes are truly benign.
357
361
 
358
362
  ## Replay class
359
363
 
@@ -407,8 +411,8 @@ staleness `fingerprint` shows ANY skill/baseline drift, or `replay --fail-on-ski
407
411
  skill-source drift; every replay result also reports it class-tagged in `staleness[]` for a JSON gate.
408
412
 
409
413
  **Egress + other filesystem — still skipped on replay (live-only):** `no_delete_in_outputs`,
410
- `self_heal_ran`, `transcript_no_host_path`, `egress_*` / `expect_denied`, `no_mcp_error`, `max_peak_rss_bytes`.
411
- These run only on a live `run`/`record`.
414
+ `self_heal_ran`, `transcript_no_host_path`, `egress_*` / `expect_denied`, `no_mcp_error`, `max_peak_rss_bytes`,
415
+ `no_lost_write_back`. These run only on a live `run`/`record`.
412
416
 
413
417
  **Mixed assertions on replay:** before evaluating, `replay` strips each assertion to its replay-checkable
414
418
  keys and drops any left empty. So `{result, egress_denied}` evaluates on replay as `{result}` alone — its
@@ -163,3 +163,30 @@ degrade the advice. It is real work to calibrate; these steps are the traps that
163
163
  **Lane note:** `semantic_matches` is **live-only** (the judge is a live model call), so these scenarios
164
164
  run on the `run` lane, never token-free `replay` — the linter's "all assertions live-only" warning is
165
165
  expected and correct here.
166
+
167
+ ## Recipe 6 — Iterate a skill across fixes (ground findings, don't cross-pair generations)
168
+
169
+ Hardening a skill is a loop: run → read what it did → fix → run again. Two disciplines keep it honest.
170
+
171
+ 1. **Verify before you trust.** A green run is not a correct run, and a skill's self-reported finding (a
172
+ self-critique appendix, "I extracted X") is not real until its cited evidence is found in the run's own
173
+ output. The harness emits the substrate; the grader is yours (it lives outside the harness):
174
+ - `result.json` → `finalMessage` (the skill's own answer/critique) + `toolResults[]` (tool outputs).
175
+ - `cowork-harness trace <run-dir> --output-format json` → the tool-call stream. Add `--full-results` so
176
+ a **successful** call's full input + result are captured (the default view slices them to ~100/120
177
+ chars) — this is what lets your grader confirm "the skill claims it read X and derived Y" against the
178
+ actual call.
179
+ - `cowork-harness inspect <run-dir>` → what the run produced, plus the run's `label` and `skillHash`.
180
+ - In-run alternative: dispatch a checker **sub-agent** (maker/checker) whose result folds into the
181
+ verdict.
182
+ 2. **Don't cross-pair generations.** When you run the same skill across fixes, never pair a *pre-fix*
183
+ `result.json` with a *post-fix* critique. The authoritative version key is `fingerprint.skillHash` —
184
+ content-exact, on every live run, changes on any tracked edit. **Group/pair on it** (`inspect` and the
185
+ run-index row surface a short prefix). Add `--label <tag>` for a human-readable generation name
186
+ (skillHash is the correctness key; the label is ergonomics). `cowork-harness verify-run <run-dir>
187
+ <scenario.yaml>` is the native staleness guard: it **warns** when a kept run predates the current
188
+ skill, and with scripted `answers` **hard-fails** rather than vouch for a stale gate snapshot.
189
+
190
+ **Lane note:** the exploratory driver is `skill <dir> --decider-llm --intent "<what this run tests>"`,
191
+ which is flagged non-deterministic (a green here is exploration, not a scripted pass) — pin the
192
+ load-bearing gates with `--answer` once you know which fire.
@@ -27,6 +27,7 @@
27
27
  "max_turns",
28
28
  "no_delete_in_outputs",
29
29
  "no_hook_blocked",
30
+ "no_lost_write_back",
30
31
  "no_mcp_error",
31
32
  "no_path_denied",
32
33
  "no_scratchpad_leak",
@@ -60,7 +61,9 @@
60
61
  "tool_no_error_if_called",
61
62
  "tool_not_called",
62
63
  "tool_result_contains",
64
+ "tool_result_matches",
63
65
  "tool_result_not_contains",
66
+ "tool_result_not_matches",
64
67
  "transcript_contains",
65
68
  "transcript_matches",
66
69
  "transcript_no_host_path",
@@ -62,6 +62,8 @@ CONTENT_KEYS = {
62
62
  "transcript_not_matches",
63
63
  "tool_result_contains",
64
64
  "tool_result_not_contains",
65
+ "tool_result_matches",
66
+ "tool_result_not_matches",
65
67
  "tool_called",
66
68
  "tool_not_called",
67
69
  "subagent_tool_used",
@@ -129,6 +131,7 @@ LIVE_ONLY_KEYS = {
129
131
  "no_mcp_error",
130
132
  "max_peak_rss_bytes",
131
133
  "semantic_matches",
134
+ "no_lost_write_back",
132
135
  }
133
136
  EGRESS_KEYS = {"egress_denied", "egress_allowed"}
134
137
  # container-only: served only at fidelity: container (present_files / the scratchpad promotion path
@@ -229,6 +232,8 @@ REGEX_KEYS = {
229
232
  "subagent_dispatched",
230
233
  "question_asked",
231
234
  "hook_blocked",
235
+ "tool_result_matches",
236
+ "tool_result_not_matches",
232
237
  }
233
238
  VALID_ON_UNANSWERED = {"fail", "prompt", "first", "llm"}
234
239
  VALID_TIERS = ("protocol", "container", "microvm", "hostloop", "cowork")