cowork-harness 4.1.1 → 4.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (89) hide show
  1. package/.claude/skills/cowork-harness/SKILL.md +14 -10
  2. package/.claude/skills/cowork-harness/references/assertion-catalog.md +4 -4
  3. package/.claude/skills/cowork-harness/references/assertions-guide.md +2 -2
  4. package/.claude/skills/cowork-harness/references/authoring.md +2 -1
  5. package/.claude/skills/cowork-harness/references/ci-recipe.md +5 -5
  6. package/.claude/skills/cowork-harness/references/critique.md +1 -1
  7. package/.claude/skills/cowork-harness/references/debugging.md +1 -1
  8. package/.claude/skills/cowork-harness/references/eval.md +79 -0
  9. package/.claude/skills/cowork-harness/references/fidelity-and-answers.md +1 -1
  10. package/.claude/skills/cowork-harness/references/gotchas.md +15 -2
  11. package/.claude/skills/cowork-harness/references/measurement.md +6 -1
  12. package/.claude/skills/cowork-harness/references/run-record-replay.md +2 -2
  13. package/.claude/skills/cowork-harness/references/scenario-schema.md +2 -2
  14. package/.claude/skills/cowork-harness/references/task-recipes.md +24 -10
  15. package/.claude/skills/cowork-harness/scripts/scenario.py +56 -1
  16. package/CHANGELOG.md +310 -0
  17. package/DESIGN.md +2 -2
  18. package/README.md +10 -5
  19. package/SPEC.md +12 -4
  20. package/baselines/desktop-2.16120.0.json +1148 -0
  21. package/baselines/prompts/cowork-system-prompt-fingerprints.json +10 -1
  22. package/dist/agent/session.js +26 -0
  23. package/dist/assert.js +102 -13
  24. package/dist/baseline.js +7 -0
  25. package/dist/cli.js +143 -365
  26. package/dist/critique/command.js +19 -9
  27. package/dist/critique/skill-invocation.js +61 -1
  28. package/dist/decide/decider.js +25 -3
  29. package/dist/decide/semantic-judge.js +170 -37
  30. package/dist/decide/usage.js +52 -0
  31. package/dist/eval/classify.js +308 -0
  32. package/dist/eval/command.js +591 -0
  33. package/dist/eval/invocation.js +44 -0
  34. package/dist/eval/job-runner.js +50 -0
  35. package/dist/eval/manifest.js +11 -0
  36. package/dist/eval/pins.js +45 -0
  37. package/dist/eval/report.js +479 -0
  38. package/dist/eval/runs.js +127 -0
  39. package/dist/eval/schedule.js +31 -0
  40. package/dist/eval/snapshot.js +328 -0
  41. package/dist/eval/stats.js +240 -0
  42. package/dist/eval/usage.js +62 -0
  43. package/dist/hillclimb/schema-check.js +657 -0
  44. package/dist/run/api-retries.js +31 -0
  45. package/dist/run/artifacts.js +5 -4
  46. package/dist/run/authored-capture-opts.js +23 -0
  47. package/dist/run/cassette.js +149 -12
  48. package/dist/run/chat-result.js +9 -1
  49. package/dist/run/chat.js +125 -63
  50. package/dist/run/command-globals.js +15 -2
  51. package/dist/run/doctor.js +62 -45
  52. package/dist/run/execute.js +194 -61
  53. package/dist/run/model-provenance.js +50 -3
  54. package/dist/run/provenance.js +29 -11
  55. package/dist/run/renderer.js +15 -1
  56. package/dist/run/run-index.js +6 -0
  57. package/dist/run/run.js +8 -0
  58. package/dist/run/runs-gc.js +26 -1
  59. package/dist/run/verify-context.js +403 -0
  60. package/dist/runtime/agent-tree.js +480 -0
  61. package/dist/runtime/hostloop.js +6 -5
  62. package/dist/runtime/protocol.js +6 -1
  63. package/dist/sync/cowork-sync.js +266 -5
  64. package/dist/termination.js +76 -10
  65. package/dist/types.js +2 -2
  66. package/docs/README.md +2 -1
  67. package/docs/boundary.md +7 -0
  68. package/docs/cassette.md +13 -6
  69. package/docs/chat.md +7 -1
  70. package/docs/ci.md +25 -1
  71. package/docs/cli.md +22 -11
  72. package/docs/companion-skill.md +2 -2
  73. package/docs/critique.md +2 -1
  74. package/docs/debugging.md +12 -4
  75. package/docs/eval.md +244 -0
  76. package/docs/fidelity-gaps.md +78 -9
  77. package/docs/run-status.md +7 -3
  78. package/docs/scenario.md +6 -6
  79. package/docs/session.md +1 -1
  80. package/docs/stats.md +15 -3
  81. package/examples/replays/README.md +1 -1
  82. package/examples/replays/example-multiselect-gate.cassette.json +1 -1
  83. package/examples/replays/example-pdf-skill.cassette.json +1 -1
  84. package/examples/replays/hostloop-computer-links.cassette.json +1 -1
  85. package/llms.txt +3 -2
  86. package/package.json +1 -1
  87. package/python/test_scenario_lint.py +71 -0
  88. package/schema/run-result.json +77 -1
  89. package/schema/scenario.schema.json +2 -2
@@ -1,10 +1,10 @@
1
1
  ---
2
2
  name: cowork-harness
3
- description: Test or debug a Claude Code skill/plugin under Claude Cowork's runtime — sandboxed agent, default-deny egress, the can_use_tool permission/question protocol — using the cowork-harness CLI. Use when validating or regression-testing a skill, authoring or debugging a scenario YAML (prompt + scripted answers + assert:), choosing a fidelity tier, scripting AskUserQuestion / tool-permission answers, asserting artifacts, egress, or sub-agent dispatch, measuring how long each tool call took (toolDurations / trace), or debugging a failed run or verdict from its result.json or transcript. Especially when a harness run no-ops an assertion, fails on an unanswered gate, false-greens, a steered answer never reaches the model, or a web_fetch is unexpectedly denied or gated. Also when iterating or hardening a skill across fixes, or grounding a skill's self-critique against its own run evidence — including a document-analysis skill (cap table, deck, financial model, transcript) that needs an uploaded file attached to be critiqued at all. NOT for generic unit testing (pytest/vitest of your own scripts) or non-Cowork CI. Covers the skill / run / chat / record / replay / trace / decide / assertions / scaffold commands and the session-vs-scenario split.
3
+ description: Test or debug a Claude Code skill/plugin under Claude Cowork's runtime — sandboxed agent, default-deny egress, the can_use_tool permission/question protocol — using the cowork-harness CLI. Use when validating or regression-testing a skill, authoring or debugging a scenario YAML (prompt + scripted answers + assert:), choosing a fidelity tier, scripting AskUserQuestion / tool-permission answers, asserting artifacts, egress, or sub-agent dispatch, measuring how long each tool call took (toolDurations / trace), or debugging a failed run or verdict from its result.json or transcript. Especially when a harness run no-ops an assertion, fails on an unanswered gate, false-greens, a steered answer never reaches the model, or a web_fetch is unexpectedly denied or gated. Also when iterating or hardening a skill across fixes, or grounding a skill's self-critique against its own run evidence — including a document-analysis skill (cap table, deck, financial model, transcript) that needs an uploaded file attached to be critiqued at all. Also for comparing two versions of a skill before merging an edit — did the change make its answers worse? (`eval`: paired, interleaved A/B of two plugin versions, pinned models). NOT for generic unit testing (pytest/vitest of your own scripts) or non-Cowork CI. Covers the skill / run / chat / record / replay / trace / decide / assertions / scaffold / critique / stats / eval commands and the session-vs-scenario split.
4
4
  metadata:
5
5
  author: cowork-harness
6
- version: 4.1.1
7
- tracks-harness: cowork-harness 4.1.1 (baseline desktop-2.9939.4)
6
+ version: 4.2.0
7
+ tracks-harness: cowork-harness 4.2.0 (baseline desktop-2.16120.0)
8
8
  ---
9
9
 
10
10
  # cowork-harness
@@ -26,8 +26,8 @@ allowlist). This skill exists mostly to keep you out of those traps — the *Inv
26
26
  full landmine catalog in [`references/gotchas.md`](references/gotchas.md) are the highest-value part.
27
27
  Read them.
28
28
 
29
- > **Version note:** the facts and `file:line` pointers here track `cowork-harness 4.1.1` (baseline
30
- > `desktop-2.9939.4`). If your checkout is newer, prefer the live `--help` and — in a repo checkout —
29
+ > **Version note:** the facts and `file:line` pointers here track `cowork-harness 4.2.0` (baseline
30
+ > `desktop-2.16120.0`). If your checkout is newer, prefer the live `--help` and — in a repo checkout —
31
31
  > `SPEC.md` / `docs/*.md` over this snapshot, and re-run the bundled linter.
32
32
 
33
33
  ## Preflight — make sure the harness can actually run
@@ -43,7 +43,7 @@ Before the first command, confirm the CLI is reachable and **fail loud** (never
43
43
 
44
44
  - **One-shot check.** Run `cowork-harness doctor [--tier <tier>]` first — a read-only prerequisite check that inspects Docker, the staged agent, the token, and the baseline in one pass. The bullets below explain each thing it checks (and how to fix it).
45
45
  - **Replay-only? Skip `doctor`.** Replaying committed cassettes needs no Docker, no staged agent, and no token — and every tier's `doctor` validates the auth token (the live tiers also Docker + the staged agent), so a ✗ there is expected, not a blocker. Go straight to `cowork-harness replay <cassette>`.
46
- - **CLI on PATH, recent enough?** Run `cowork-harness --version` — this skill needs **≥ 4.1.1**. If it's missing or older, prefix every command with the version floor `npx "cowork-harness@^4.1.1" <cmd>` (Node ≥ 22), or install once with `npm i -g "cowork-harness@^4.1.1"`. **Pin `@^4.1.1`, never `@latest`** — `@latest` can silently fetch an older CLI and the new commands fail as "unknown command", whereas the floor **fails loud** if no compatible version is published.
46
+ - **CLI on PATH, recent enough?** Run `cowork-harness --version` — this skill needs **≥ 4.2.0**. If it's missing or older, prefix every command with the version floor `npx "cowork-harness@^4.2.0" <cmd>` (Node ≥ 22), or install once with `npm i -g "cowork-harness@^4.2.0"`. **Pin `@^4.2.0`, never `@latest`** — `@latest` can silently fetch an older CLI and the new commands fail as "unknown command", whereas the floor **fails loud** if no compatible version is published.
47
47
 
48
48
  This skill documents the CURRENT surface, not release history. If `cowork-harness --version` is
49
49
  OLDER than the floor, the per-release record of what you are missing is [CHANGELOG.md](https://github.com/yaniv-golan/cowork-harness/blob/main/CHANGELOG.md)
@@ -76,8 +76,10 @@ CI-grade scenario, and the post-hoc debug loop; the rest are narrower tools that
76
76
  grades against a different artifact, `critique-evidence-package.txt`, which none of these tools
77
77
  surface; see `references/critique.md`.
78
78
  - **Regression-test your skill's ANSWER quality** (not just its behavior — does its guidance still lead to
79
- correct answers after you edit it?) → author `semantic_matches` scenarios and gate on the per-claim
80
- profile. See **Recipe 5** in `references/task-recipes.md` (validity, N≥3, discrimination — the traps).
79
+ correct answers after you edit it?) → author `semantic_matches` scenarios, then compare the version
80
+ before your edit with the one after using `cowork-harness eval` (EXPERIMENTAL, live: 10 runs per
81
+ scenario at the defaults). See **Recipe 5** step 6 in `references/task-recipes.md` (validity,
82
+ discrimination — the traps) and [`references/eval.md`](references/eval.md).
81
83
  - **"What is WRONG with this skill?"** (a graded critique, not a pass/fail) → `cowork-harness critique
82
84
  <folder> --prompt "<probe>"`. Up to four model workloads (zero with `--corpus-only`; pass 2 is skipped with no self-report) and 10–20 minutes; budget from
83
85
  `report.costUsd.totalUsd`. Reach for it when you want **findings**. **For "what does this skill
@@ -97,7 +99,7 @@ CI-grade scenario, and the post-hoc debug loop; the rest are narrower tools that
97
99
 
98
100
  Full command set: `skill · run · chat · record · replay · verify-cassettes · rehash · prune · migrate-run-dir · lint ·
99
101
  lint-skill · analyze-skill · probe-dispatch ·
100
- verify-run · trace · inspect · diff · critique · stats · decide · gates · answer · scaffold · assertions --list · sync ·
102
+ verify-run · trace · inspect · diff · critique · eval · eval report · stats · decide · gates · answer · scaffold · assertions --list · sync ·
101
103
  list · boundary-check · status · vm <init|status|delete|prune> · doctor · init-redact`. Always check `cowork-harness <cmd> --help`.
102
104
 
103
105
  ## Invariants — how a green run lies
@@ -107,7 +109,8 @@ behind each, is [`references/gotchas.md`](references/gotchas.md).
107
109
 
108
110
  1. **`result: success` is not "the task completed".** It means the agent didn't error. Assert the
109
111
  deliverable (`file_exists` / `artifact_json` / `transcript_matches`). A `skill`-lane `PASS` only means
110
- no guard fired: read `skillsInvoked`, `models` and `ablated` before concluding anything from it.
112
+ no guard fired: read `skillsInvoked` (plus `slashInvokedSkills` — a `/<skill>` prompt runs the skill
113
+ with no `Skill` call), `models` and `ablated` before concluding anything from it.
111
114
  2. **`replay` skips live-only keys.** Filesystem and egress keys are skipped on replay (loudly), so a
112
115
  mixed item like `{result, egress_denied}` greens on its content half. Keep one concern per `assert:`
113
116
  item, put live-only checks on a live gate, and run `cowork-harness lint`.
@@ -154,4 +157,5 @@ behind each, is [`references/gotchas.md`](references/gotchas.md).
154
157
  | [`references/fidelity-and-answers.md`](references/fidelity-and-answers.md) | tier semantics, answer paths, the determinism contract |
155
158
  | [`references/ci-recipe.md`](references/ci-recipe.md) | the GitHub Action, replay-vs-live lanes, the four-stage pipeline |
156
159
  | [`references/critique.md`](references/critique.md) | `critique` report and evidence-package shapes |
160
+ | [`references/eval.md`](references/eval.md) | `eval`: paired before/after of two plugin versions — labels, refusals, exit codes, files |
157
161
  | `scripts/scenario.py` | `scaffold`, `lint`, `lint-skill`, `resolve-agent-types <plugin-dir>` (validates a pinned `subagent_type` against `plugin.json` + `agents/*.md`) |
@@ -1,6 +1,6 @@
1
1
  # Assertion catalog
2
2
 
3
- Tracks `cowork-harness 4.1.1` (baseline `desktop-2.9939.4`). Every `assert:` key with its semantics, and the
3
+ Tracks `cowork-harness 4.2.0` (baseline `desktop-2.16120.0`). Every `assert:` key with its semantics, and the
4
4
  verdict-signal table. Which keys survive `replay` is in [`scenario-schema.md`](./scenario-schema.md#replay-class);
5
5
  the scenario and session YAML fields are there too.
6
6
 
@@ -55,8 +55,8 @@ same set live from the schema.
55
55
  | `subagent_declared_but_unused: <Tool>` | a sub-agent declared the tool but never used **that** tool (even if it used others) |
56
56
  | `subagent_output_contains: {match?, contains}` | a dispatched sub-agent's own output contains the substring `contains` — `match` (optional regex over `dispatchAgentType`/`resolvedAgentType`/`description`) narrows to specific dispatch(es); omitted, checks whether ANY dispatch's output contains it (existence check, not "all"); a miss against an output that was **truncated at the assert cap** reports evidence-unavailable instead of a proven absence — the substring could lie past the cut. **Covers what the run dispatches** (`Agent`/`Task`, including `Agent(subagent_type:"fork")`), **not a `context: fork` skill** invoked through the `Skill` tool: that skill's own answer is never a dispatch — it comes back as the `Skill` tool result, which agent 2.1.284 builds as `Skill "<name>" completed (forked execution).`, a `Result:` line, then the answer. Assert on it with `tool_result_matches` anchored on that prefix, e.g. `tool_result_matches: '^Skill "[^"]*" completed \(forked execution\)[\s\S]*<pattern>'` (`[\s\S]*` because `.` stops at a newline; the match is case-insensitive, has no multiline flag so `^` is the start of the result, and sees the first 10 KB of each result). This covers a **foreground** fork only: a backgrounded fork's result is the line `Skill "<name>" launched (forked execution, running in the background).`, which carries no answer |
57
57
  | `dispatch_count_max: <N>` | at most N sub-agents dispatched — your author-chosen budget under Cowork's agent-side fan-out cap (concurrent 20 / per-session 200, inherited by the harness); records only, does not itself enforce — see gotcha 12 in `scenario-schema.md` |
58
- | `skill_triggered: <regex>` | a skill matching the regex (invoked id, e.g. `"plugin:skill"`) was invoked via the `Skill` tool — evidence-unavailable (not a normal fail) if the agent's init tools have no `Skill` tool |
59
- | `no_skill_triggered: <regex>` | no invoked skill id matched — the negative-control / description-collision catcher; evidence-unavailable (never a vacuous pass) if invocation data is absent or the `Skill` tool is unobservable |
58
+ | `skill_triggered: <regex>` | a skill matching the regex (invoked id, e.g. `"plugin:skill"`) was invoked — via the `Skill` tool, or by a prompt starting `/<skill> …` / `/<plugin>:<skill> …` (the agent expands that itself with no `Skill` call; recorded as `slashInvokedSkills`) — evidence-unavailable (not a normal fail) if neither matched and the agent's init tools have no `Skill` tool, or the leading `/name` can't be resolved (ambiguous bare name, no skill inventory) |
59
+ | `no_skill_triggered: <regex>` | no invoked skill id matched, counting a slash-command invocation as well as a `Skill` call — the negative-control / description-collision catcher; evidence-unavailable (never a vacuous pass) if invocation data is absent, the `Skill` tool is unobservable, or the prompt's leading `/name` can't be resolved |
60
60
  | `skill_available: <regex>` | a staged skill's id matched the regex (offered, not necessarily invoked — see `skill_triggered` for invocation) — content-class: the id list comes from the agent's init `skills` listing, so it replays from the frozen init event (id-only; the `whenToUse` enrichment is live-disk and thus absent on replay, but the id is what's matched); evidence-unavailable only if `RunResult.context.availableSkills` is absent entirely (an older cassette recorded before the available-skills listing was captured) |
61
61
  | `connector_available: <regex>` | an MCP server/connector's name matched the regex (available, not necessarily used) — evidence-unavailable if `RunResult.context.mcpServers` is absent |
62
62
  | `tool_available: <regex>` | a tool in the init manifest matched the regex (available, not necessarily called — see `tool_called` for invocation) — evidence-unavailable if `RunResult.context.tools` is absent. The `mcp__skills__*`/`mcp__plugins__*` discovery tools are modeled (as `alwaysLoad`) on `container`/`hostloop`/`cowork` — a miss there is a real absence; `microvm`/`protocol` declare no such server, so a miss on those two tiers means "not modeled at this tier", not "provably unavailable" |
@@ -107,7 +107,7 @@ same set live from the schema.
107
107
  | `egress_allowed: <host>` | the host was allowed through |
108
108
  | `no_mcp_error: true` | no MCP round-trip failed (`RunResult.mcpErrors` is empty — no unhandled server, no handler throw) — live-only: MCP round-trips are harness-computed, not in the SDK stdout stream, so evidence-unavailable on replay (never a vacuous pass). **Only `true` is valid** |
109
109
  | `max_peak_rss_bytes: <N>` | peak sampled RSS of the agent sandbox ≤ N bytes (`RunResult.resources.peakRssBytes`) — live-only: replay never spawns a sandbox to sample, so evidence-unavailable on replay/protocol (never a vacuous pass); also evidence-unavailable when sampling captured no RSS value |
110
- | `semantic_matches: {rubric: [...], min_pass?, judge_model?, include_subagent_text?, evidence_files?}` | a pinned LLM judge grades each fixed `rubric` claim against the run's answer — the **union of the agent's final result text (`RunResult.finalMessage`), the transcript, and the final on-disk content of any files the agent authored during the run** — so a claim about content the skill led the agent to *write to a file* grades as reliably as one about inlined prose. **"The transcript" is narrower than it reads: top-level `assistant_text` ONLY.** It excludes every `tool_use`/`tool_result` and **all sub-agent-originated text** (including fork-scoped `Skill`/`Agent(fork)` dispatches — the harness attributes their *tool* calls to the main agent, but not their text). **A `context: fork` skill's own answer is not graded either**, even with `include_subagent_text: true`: it is not a dispatch (so it has no `subagents[]` entry to fold in), and it reaches the main agent as the `Skill` tool result, which the judged document excludes. The judge sees it only if the main agent restates it; to check the fork's answer directly, use `tool_result_matches` (see `subagent_output_contains`). ⚠️ **Consequence: a rubric claim about whether a tool was called can NEVER grade true** — the evidence is not in the judged document. Such a claim looks reasonable and silently caps your pass rate; assert tool use with `tool_called` / `present_files_called` / `subagent_dispatched` / `hook_blocked` instead. Sub-agent text is captured in `RunResult.subagents[].reasoning` and reaches the judge only via opt-in `include_subagent_text: true` (`kind:"text"` turns only — sub-agent *thinking* arrives empty with `redacted:true`, so it would pad the document with blanks) (authored-file evidence is captured on every live sandbox tier including **microvm** — its session tree is snapshotted from the VM into the run dir). When the authored-file evidence backing the judged document is **incomplete** — a file dropped at the capture-size cap, unreadable at read-back, or (on `--resume`) the scratchpad walk skipped — the assert fails evidence-unavailable rather than trusting a judge grade over a partial document; this is separate from the malformed-grade `judgeInvalid` path below. The assert passes iff ≥ `min_pass` claims pass (default: all — avoid for a gating scenario). Results align by claim index and are recorded per-claim in `RunResult.assertions[].semanticClaims` (`[{index, claim, pass}]`, so a consumer can diff the per-claim profile across runs); a rep whose grade can't be parsed (after one retry) is marked `RunResult.assertions[].judgeInvalid` and **never silently dropped** — it is excluded from the pass denominator, and the guard against a misleading score from that exclusion is the gate's minimum-valid-rep floor (`MIN_VALID` ≥ 4) plus this visibility, not a claim that denominator-shrinking inflation is impossible. Within a rep, a grade that's still unparseable after the retry **fails that assert outright** (evidence-unavailable, not a vacuous pass) — a persistently-flaky judge reds the run rather than silently passing. `judge_model` pins the grader (default when neither it nor `COWORK_HARNESS_JUDGE_MODEL` is set: `claude-opus-4-8`; a dated id keeps a before/after comparison reproducible). Live-only: the judge is a live model call, so evidence-unavailable / skipped-loud on replay (never a vacuous pass) **`evidence_files: [globs]` scopes which authored files are graded** — reach for it the moment a run authors more than a couple of files. The capture budget (64 KiB total by default) is spent prefix-major then alphabetically, so a pipeline that stages intermediates (`outputs/_work/*.json`) exhausts it before reaching its own deliverable and the verdict is refused evidence-unavailable over files no rubric mentions. Scoping also makes the capture spend the budget on the named files FIRST and exempts them from the per-file cap. Paths are `<root>/<rel>` (`outputs/report.md`, never a bare `report.md`; session-root writes are `scratchpad/<rel>`); globs are `*`/`?`/`**`, not regex. A glob matching nothing FAILS and the message lists every authored path — read it rather than guessing. Still too big? Raise `$COWORK_HARNESS_AUTHORED_TOTAL_BYTES`. The typed reason is on `RunResult.assertions[].semanticEvidence` — check `.reason` instead of parsing the message |
110
+ | `semantic_matches: {rubric: [...], min_pass?, judge_model?, include_subagent_text?, evidence_files?}` | a pinned LLM judge grades each fixed `rubric` claim against the run's answer — the **union of the agent's final result text (`RunResult.finalMessage`), the transcript, and the final on-disk content of any files the agent authored during the run** — so a claim about content the skill led the agent to *write to a file* grades as reliably as one about inlined prose. **"The transcript" is narrower than it reads: top-level `assistant_text` ONLY.** It excludes every `tool_use`/`tool_result` and **all sub-agent-originated text** (including fork-scoped `Skill`/`Agent(fork)` dispatches — the harness attributes their *tool* calls to the main agent, but not their text). **A `context: fork` skill's own answer is not graded either**, even with `include_subagent_text: true`: it is not a dispatch (so it has no `subagents[]` entry to fold in), and it reaches the main agent as the `Skill` tool result, which the judged document excludes. The judge sees it only if the main agent restates it; to check the fork's answer directly, use `tool_result_matches` (see `subagent_output_contains`). ⚠️ **Consequence: a rubric claim about whether a tool was called can NEVER grade true** — the evidence is not in the judged document. Such a claim looks reasonable and silently caps your pass rate; assert tool use with `tool_called` / `present_files_called` / `subagent_dispatched` / `hook_blocked` instead. Sub-agent text is captured in `RunResult.subagents[].reasoning` and reaches the judge only via opt-in `include_subagent_text: true` (`kind:"text"` turns only — sub-agent *thinking* arrives empty with `redacted:true`, so it would pad the document with blanks) (authored-file evidence is captured on every live sandbox tier including **microvm** — its session tree is snapshotted from the VM into the run dir). When the authored-file evidence backing the judged document is **incomplete** — a file dropped at the capture-size cap, unreadable at read-back, or (on `--resume`) the scratchpad walk skipped — the assert fails evidence-unavailable rather than trusting a judge grade over a partial document; this is separate from the malformed-grade `judgeInvalid` path below. The assert passes iff ≥ `min_pass` claims pass (default: all — avoid for a gating scenario). Results align by claim index and are recorded per-claim in `RunResult.assertions[].semanticClaims` (`[{index, claim, pass, rationale?}]`, so a consumer can diff the per-claim profile across runs; `rationale` is the judge's one-sentence reason, printed under each failed claim in the failure footer: untrusted model text that can quote the judged document, whose content never affects `pass`, absent when the judge gave none (a reply whose shape is broken, such as unparseable JSON, a malformed `{"results": …}` group beside a valid grade, or a partial restatement that contradicts it, is retried once and then marked `judgeInvalid`), and comparable only between runs that share `judgePromptHash`); a rep whose grade can't be parsed (after one retry) is marked `RunResult.assertions[].judgeInvalid` and **never silently dropped** — it is excluded from the pass denominator, and the guard against a misleading score from that exclusion is the gate's minimum-valid-rep floor (`MIN_VALID` ≥ 4) plus this visibility, not a claim that denominator-shrinking inflation is impossible. Within a rep, a grade that's still unparseable after the retry **fails that assert outright** (evidence-unavailable, not a vacuous pass) — a persistently-flaky judge reds the run rather than silently passing. `judge_model` pins the grader (default when neither it nor `COWORK_HARNESS_JUDGE_MODEL` is set: `claude-opus-4-8`; a dated id keeps a before/after comparison reproducible — or let `eval` compare two skill versions per claim, with the judge pinned). Each graded assert records the judge's provenance: `RunResult.assertions[].judgeModel` (the resolved model), `judgeCostUsd` (judge spend over both attempts, reported beside `cost.usd` and never inside it; absent when unpriced) and `judgePromptHash` (the grading-prompt template — compare only runs that share it). Live-only: the judge is a live model call, so evidence-unavailable / skipped-loud on replay (never a vacuous pass) **`evidence_files: [globs]` scopes which authored files are graded** — reach for it the moment a run authors more than a couple of files. The capture budget (64 KiB total by default) is spent prefix-major then alphabetically, so a pipeline that stages intermediates (`outputs/_work/*.json`) exhausts it before reaching its own deliverable and the verdict is refused evidence-unavailable over files no rubric mentions. Scoping also makes the capture spend the budget on the named files FIRST and exempts them from the per-file cap. Paths are `<root>/<rel>` (`outputs/report.md`, never a bare `report.md`; session-root writes are `scratchpad/<rel>`); globs are `*`/`?`/`**`, not regex. A glob matching nothing FAILS and the message lists every authored path — read it rather than guessing. Still too big? Raise `$COWORK_HARNESS_AUTHORED_TOTAL_BYTES`. The typed reason is on `RunResult.assertions[].semanticEvidence` — check `.reason` instead of parsing the message |
111
111
  | `artifact_json: {artifact, path, …}` | assert a JSON artifact's contents — `equals`/`gt`/`in`/`exists`/`absent`/`is_null` over a dotted `path` (`in` = membership in a list, for a stochastic/LLM value; `absent` ≠ `is_null`; an unresolved intermediate fails loud) |
112
112
  | `computer_links_resolve: true` | every `computer://` link in the model-visible transcript resolves to an artifact that exists in the run's collected outputs/mounts — a dangling link fails, naming which target was checked (a live host path, the collected work tree, or the replay manifest). **Requires ≥1 link** (zero links fails — use `computer_links_resolve_if_present` for the presence-free variant). **Only `true` is valid** (`false` is rejected by the schema) **Sees top-level `assistant_text` only — it excludes every `tool_use`/`tool_result`**, so a `computer://` link that appeared only inside a tool call or its result is invisible to it. |
113
113
  | `computer_links_resolve_if_present: true` | like `computer_links_resolve` but passes vacuously when the transcript has zero `computer://` links — the presence-free variant. **Only `true` is valid** **Sees top-level `assistant_text` only — it excludes every `tool_use`/`tool_result`**, so a `computer://` link that appeared only inside a tool call or its result is invisible to it. |
@@ -1,6 +1,6 @@
1
1
  # Assertions guide
2
2
 
3
- Tracks `cowork-harness 4.1.1` (baseline `desktop-2.9939.4`). Read it when choosing assertion keys: the two orthogonal axes and the goal → key map. The full catalog is `assertion-catalog.md`.
3
+ Tracks `cowork-harness 4.2.0` (baseline `desktop-2.16120.0`). Read it when choosing assertion keys: the two orthogonal axes and the goal → key map. The full catalog is `assertion-catalog.md`.
4
4
 
5
5
  ### Assertions: two orthogonal axes
6
6
 
@@ -43,7 +43,7 @@ them by what you're trying to prove:
43
43
  | a skill actually **ran** (or must NOT) | `skill_triggered: <regex>`, `no_skill_triggered: <regex>` |
44
44
  | a tool ran **inside** a skill's scope | `skill_tool_used: {skill, tool}` |
45
45
  | a sub-agent did the work | `subagent_output_contains: {contains}`, `subagent_dispatched: <regex>`, `dispatch_count_max: <N>` |
46
- | a `context: fork` skill answered correctly | `tool_result_matches: '^Skill "[^"]*" completed \(forked execution\)[\s\S]*<pattern>'` — its answer is the `Skill` tool result, not a sub-agent output, so `subagent_output_contains` and `semantic_matches` never see it (foreground fork only — a backgrounded fork's result carries no answer) |
46
+ | a `context: fork` skill answered correctly | `tool_result_matches: '^Skill "[^"]*" completed \(forked execution\)[\s\S]*<pattern>'` — its answer is the `Skill` tool result, not a sub-agent output, so `subagent_output_contains` and `semantic_matches` never see it (foreground fork only — a backgrounded fork's result carries no answer). Only when the MODEL invokes the skill: a `/<skill> …` prompt runs the fork with no `Skill` call and no such result — use `skill_triggered` + `transcript_matches` there |
47
47
  | a pre-existing input wasn't mutated (incl. `uploads/**`) | `input_unmodified: <glob>` or `[<glob>, …]` (live/verify-run) |
48
48
  | no authored interactive artifact silently loses its Submit under Cowork | `no_lost_write_back: true` (**live-only**; static Tier A over the run's authored `.html`/`.py`/`.js`; per-scenario gate for the same class `analyze-skill` scans) |
49
49
  | a resource ceiling held | `max_peak_rss_bytes: <N>` (**live-only**) |
@@ -1,6 +1,6 @@
1
1
  # Authoring a scenario
2
2
 
3
- Tracks `cowork-harness 4.1.1` (baseline `desktop-2.9939.4`). Read it when composing a `scenarios/*.yaml`: session vs scenario, discovery, the fidelity tier, the answer path, `web_fetch`, and scaffold + lint.
3
+ Tracks `cowork-harness 4.2.0` (baseline `desktop-2.16120.0`). Read it when composing a `scenarios/*.yaml`: session vs scenario, discovery, the fidelity tier, the answer path, `web_fetch`, and scaffold + lint.
4
4
 
5
5
  ## Part I — AUTHOR a scenario
6
6
 
@@ -251,6 +251,7 @@ cowork-harness lint scenarios/*.yaml
251
251
  | `reference-access-contradiction` | ERROR | one reference under both `reference_read` and `no_observed_reference_access` |
252
252
  | `regex-double-quoted` | WARN | a double-quoted regex with an unescaped backslash (YAML strips it) |
253
253
  | `replay-noop` | WARN | every assertion is live-only or a verdict modifier, so a replay gate verifies nothing |
254
+ | `slash-prompt-forked-result-anchor` | WARN | a `prompt:` starting with `/<skill>` plus a `tool_result_*` anchored on `forked execution` — a slash-invoked skill makes no `Skill` call, so that tool result never exists; assert `skill_triggered` instead |
254
255
  | `tool-called-always-passes` | INFO | `tool_called` with `count: {min: 0}` and no `max` — it asserts nothing |
255
256
  | `tool-input-regex-redactable` | WARN | a `tool_not_called` input literal the redaction policy rewrites in the committed cassette (or a policy pattern it cannot check offline) |
256
257
  | `tool-input-shell-tier` | INFO | the object form with `tool: Bash` and a `command` on `hostloop` / `cowork`, where shell runs as `mcp__workspace__bash` — list both |
@@ -1,6 +1,6 @@
1
1
  # CI recipe — replay vs live lanes
2
2
 
3
- Self-contained reference. Tracks `cowork-harness 4.1.1` (baseline `desktop-2.9939.4`).
3
+ Self-contained reference. Tracks `cowork-harness 4.2.0` (baseline `desktop-2.16120.0`).
4
4
 
5
5
  **Fastest path: the packaged Action.** One step gets you `replay`/`lint`/`verify-cassettes` plus a PR
6
6
  job-summary reporter (verdict table, staleness findings, cost/turns when available):
@@ -17,7 +17,7 @@ job-summary reporter (verdict table, staleness findings, cost/turns when availab
17
17
  CLI major reaches your workflow the moment it is promoted even though your `uses:` ref never changed — so a
18
18
  copy-pasted recipe that omits the input takes a major bump with no say in it. `^4` holds the major, needs no
19
19
  patch number to remember, and only wants a human decision at the next major. Pin an exact version
20
- (e.g. `version: "4.1.1"`) instead when you want byte-reproducible CI.
20
+ (e.g. `version: "4.2.0"`) instead when you want byte-reproducible CI.
21
21
 
22
22
  Reach for the manual multi-step form below only when you need per-step control the Action's inputs don't
23
23
  cover (a custom flag combination, a different runner matrix per step, or `lint`/`verify-cassettes` gated
@@ -82,7 +82,7 @@ sha256-*checked* but not hard-blocking on mismatch — it's advisory for an inte
82
82
  GitHub-hosted runners, no token/Docker/agent:
83
83
 
84
84
  ```yaml
85
- - run: npm i -g "cowork-harness@^4.1.1"
85
+ - run: npm i -g "cowork-harness@^4.2.0"
86
86
  - run: cowork-harness lint scenarios/*.yaml --strict --min-severity WARN
87
87
  # no silent false-greens. WITHOUT --strict this
88
88
  # step cannot fail on a WARN-class rule (e.g.
@@ -395,7 +395,7 @@ jobs:
395
395
  with: { node-version: '24' }
396
396
  - uses: actions/setup-python@v5
397
397
  with: { python-version: '3.x' } # python3 only — PyYAML is bundled with the linter
398
- - run: npm i -g "cowork-harness@^4.1.1"
398
+ - run: npm i -g "cowork-harness@^4.2.0"
399
399
  - run: cowork-harness lint scenarios/*.yaml # no-silent-false-green (needs python3; PyYAML bundled)
400
400
  - run: cowork-harness verify-cassettes cassettes/ --output-format json # privacy + staleness gate
401
401
  - run: cowork-harness replay cassettes/ --output-format json # token-free content/structure
@@ -424,7 +424,7 @@ jobs:
424
424
  echo "live=true" >> "$GITHUB_OUTPUT"
425
425
  fi
426
426
  - if: steps.guard.outputs.live == 'true'
427
- run: npm i -g "cowork-harness@^4.1.1"
427
+ run: npm i -g "cowork-harness@^4.2.0"
428
428
  - if: steps.guard.outputs.live == 'true'
429
429
  run: cowork-harness run scenarios/ --output-format json
430
430
  env:
@@ -1,6 +1,6 @@
1
1
  # Critique — the facts a plugin install can't otherwise reach
2
2
 
3
- Tracks `cowork-harness 4.1.1` (baseline `desktop-2.9939.4`). This is **not** a trim of the full
3
+ Tracks `cowork-harness 4.2.0` (baseline `desktop-2.16120.0`). This is **not** a trim of the full
4
4
  [`docs/critique.md`](https://github.com/yaniv-golan/cowork-harness/blob/main/docs/critique.md) (repo-only —
5
5
  flags, cost, reproduction discipline, known limitations all live there). This file covers exactly what a
6
6
  plugin install cannot otherwise discover: the run-dir artifact a harvester actually reads, the report's
@@ -1,6 +1,6 @@
1
1
  # Debugging a run
2
2
 
3
- Tracks `cowork-harness 4.1.1` (baseline `desktop-2.9939.4`). Read it when a run misbehaved or a green looks wrong: triage, the observability output, and `chat`.
3
+ Tracks `cowork-harness 4.2.0` (baseline `desktop-2.16120.0`). Read it when a run misbehaved or a green looks wrong: triage, the observability output, and `chat`.
4
4
 
5
5
  ## Part III — Debug
6
6
 
@@ -0,0 +1,79 @@
1
+ # `eval` — paired before/after comparison of a skill edit (EXPERIMENTAL)
2
+
3
+ Tracks `cowork-harness 4.2.0` (baseline `desktop-2.16120.0`). The full guide is
4
+ [docs/eval.md](https://github.com/yaniv-golan/cowork-harness/blob/main/docs/eval.md); this is the part you
5
+ need while running it.
6
+
7
+ ```bash
8
+ cowork-harness eval <scenario.yaml | dir/> --arm before=git:HEAD:plugins/my-skill --arm after=./plugins/my-skill \
9
+ --model <concrete id> --judge-model <concrete id> [--holdout <scenario.yaml>]... [--fail-on possible|confirmed]
10
+ cowork-harness eval report <eval-dir> # rebuild the report from the eval dir, no spend
11
+ ```
12
+
13
+ ## What it does
14
+
15
+ - Two arms: a plugin directory, or `git:<ref>:<path>` (read from the commit, `<path>` relative to the repo
16
+ root). The FIRST arm is the baseline; a drop is the second arm passing less often.
17
+ - Only the session's single `plugins.local_plugins` entry is substituted. Each arm is copied once, before
18
+ the first run, and every rep mounts the copy.
19
+ - The plugin is mounted at the `local_plugins` path (`mnt/.local-plugins/marketplaces/<marketplace>/<plugin>`),
20
+ not the `remote_plugins` path a UI-installed plugin has (`mnt/.remote-plugins/plugin_<id>`). A skill that
21
+ locates its own files at runtime sees a different path under each, so the comparison holds for the
22
+ `local_plugins` layout only.
23
+ - scenarios × 2 × `--reps` live runs (10 per scenario at the default `--reps 5`), interleaved, plus one judge
24
+ call per `semantic_matches` assert per run. The start-up line prints the job count. No budget flag.
25
+ - In a scenario directory, YAML with no `prompt:` (a session file) is skipped.
26
+ - Run an A/A first (`--allow-identical-arms`, the same source twice) to see your scenarios' noise.
27
+
28
+ ## Reading the labels
29
+
30
+ | Label | Meaning |
31
+ |---|---|
32
+ | `confirmed drop/rise` | significant after the correction (bh q = 0.10, or holm) |
33
+ | `possible drop/rise` | p ≤ `--alpha` (0.05), not confirmed |
34
+ | `no detectable change` | with the smallest change this n could have detected (MDD) |
35
+ | `underpowered` | no outcome at these sizes could reach `--alpha` — NOT "no change" |
36
+ | `insufficient` | too few valid reps in an arm (4 of 5 needed by default) |
37
+
38
+ A drop is a signal to investigate, not proof: open the run dirs the report links for that row. In order,
39
+ first match wins: an infrastructure failure is excluded and reported — including a rep where no model
40
+ answered (the agent's `Not logged in` / `Authentication required` reply, rule `auth`; a usage or
41
+ spend limit as its final message, even on a nonzero exit after spend, rule `usage_limit`; or only
42
+ `<synthetic>` models at $0, rule `no_model_answered`). An agent-caused failure (timeout, max turns,
43
+ unanswered question, crash) then fails every row of its rep, even with its pin unknown. Only after that
44
+ are a pin the agent did not honour (false, or unknown on a rep that completed) and a snapshot that changed
45
+ excluded and reported. A loud UNCLASSIFIED count means a termination the classifier does not know — read
46
+ those runs. Per scenario: if EVERY rep of both arms errored, or every rep of one arm is infrastructure,
47
+ that scenario compared nothing — its rows are `insufficient` and the eval exits 1. If one arm's every rep
48
+ is the agent's own failure and the other arm ran, the reps are scored (a real drop) — exit 0 unless
49
+ `--fail-on`. Either way the header
50
+ names the arm, scenario, dominant error and a matching hint (`Every rep of arm <label> in <scenario> errored — …`).
51
+
52
+ ## Refused before any run (exit 2)
53
+
54
+ - a model or judge that is an alias (`opus`, `best`) rather than a concrete id;
55
+ - an eval dir inside any git work tree (the snapshots would mount empty);
56
+ - identical arms (unless `--allow-identical-arms`);
57
+ - an arm that contains the eval's own scenario or session files (by location, copy or symlink), an
58
+ `evals.json`, or a symlink resolving outside it;
59
+ - a scenario input a run would refuse (a missing path, a `tool_not_called` the tier can never violate);
60
+ - `--fail-on confirmed` when no row could reach `confirmed` at this `--reps`;
61
+ - no usable agent credential for a scenario's tier — the same check as `doctor --tier <tier>`'s `token` row,
62
+ with its fix. A Keychain login or a `.credentials.json` in the config dir, without an env/.env token,
63
+ passes only at `protocol`;
64
+ - a session whose plugin is declared only under `plugins.remote_plugins` (the session must declare exactly
65
+ one `local_plugins` entry). Workaround: eval a copy of the session that declares the same directory under
66
+ `local_plugins`, and check the `remote_plugins` path handling with an ordinary `run`.
67
+
68
+ ## Exit codes
69
+
70
+ `0` completed — no drop fails the eval unless you pass `--fail-on`. `1` a drop at the `--fail-on` level,
71
+ every row `insufficient`, a scenario that compared nothing, or the judge model differed across reps (an A/A run under `--fail-on possible`
72
+ can exit 1 on noise). `2` usage or a refusal. `3` an arm snapshot could not be copied or staged.
73
+
74
+ ## Files
75
+
76
+ `<eval-dir>/` (default `~/.cowork-harness/evals/<eval-id>/`): `manifest.json`, `arms/`, `runs.jsonl`,
77
+ `report.json` (every rep with its bucket), `report.md`. The runs are ordinary run dirs labelled
78
+ `eval:<eval-id>:<arm>`; a bare `prune` keeps 5 per scenario and says which evals it trimmed — `eval report`
79
+ still works, but the evidence links then dangle.
@@ -1,6 +1,6 @@
1
1
  # Fidelity tiers & answer paths
2
2
 
3
- Self-contained reference. Tracks `cowork-harness 4.1.1` (baseline `desktop-2.9939.4`).
3
+ Self-contained reference. Tracks `cowork-harness 4.2.0` (baseline `desktop-2.16120.0`).
4
4
 
5
5
  > **This page vs. the repo docs.** This is the **offline snapshot** that ships inside the installed
6
6
  > plugin — it is self-contained on purpose. The repo carries four other fidelity views, each answering a
@@ -1,6 +1,6 @@
1
1
  # Gotchas
2
2
 
3
- Tracks `cowork-harness 4.1.1` (baseline `desktop-2.9939.4`). The full "✓ passed ≠ correct" landmine catalog.
3
+ Tracks `cowork-harness 4.2.0` (baseline `desktop-2.16120.0`). The full "✓ passed ≠ correct" landmine catalog.
4
4
 
5
5
  ## Gotchas — the "✓ passed ≠ correct" landmines
6
6
 
@@ -235,6 +235,17 @@ authorable). Reach for this list when debugging a run's behavior, that one while
235
235
  fail your gate. *Fix:* nothing, unless the scenario asserts `tool_available` on
236
236
  `mcp__skills__*`/`mcp__plugins__*`; then re-record. It stays silent at `microvm`/`protocol`, where
237
237
  re-recording would never produce those tools anyway.
238
+ A sibling **`agent-version:` note** means the agent version the cassette's own `system/init` event
239
+ reports differs from the one the baseline its `fingerprint.baseline` names pins for that tier: the
240
+ `agentVersion` at `container`/`microvm`, the native agent in `agentBinary.nativeStagedPath` at
241
+ `hostloop`. It never appears at `protocol`, which runs the unpinned `claude` on your `PATH`. *Why:* one
242
+ of a fingerprint re-stamped by hand across an agent bump, a recording made under
243
+ `COWORK_HARNESS_ALLOW_AGENT_FALLBACK=1`, an explicit binary override (`COWORK_AGENT_BINARY`, or
244
+ `COWORK_HOST_AGENT_BINARY` at `hostloop`), or at `hostloop` the default patch-bump substitution of the
245
+ native agent; the note lists that tier's causes and does not pick one. It is non-gating too:
246
+ `verify-cassettes` puts it in the result's `notes[]`, and `replay` prints one
247
+ `::notice:: [replay] <file> — … [agent-version]` line per cassette on stderr (also under
248
+ `--output-format json`). *Fix:* re-record against the pinned agent.
238
249
 
239
250
  24. **Never name the file-delivery tool in a `SKILL.md`.** *Why:* Cowork has **two**, one per product
240
251
  lane, and an agent only sees the one for the surface it is on. The desktop-local sandbox this harness
@@ -271,7 +282,9 @@ authorable). Reach for this list when debugging a run's behavior, that one while
271
282
  `run` the same word additionally means *your assertions held*; on `skill --repeat N`, `PASS — N/N`
272
283
  means N runs cleared the guards — it says nothing about which model served them, whether the skill
273
284
  was invoked, or whether they were the ablated arm. *Fix:* read the three fields the record already
274
- carries before drawing any conclusion — `skillsInvoked` / `skillActivity` (was it invoked at all),
285
+ carries before drawing any conclusion — `skillsInvoked` / `skillActivity` (was it invoked at all;
286
+ a `/<skill> …` prompt runs the skill with NO `Skill` call, so read `slashInvokedSkills` too — and
287
+ `models` is then just `["<synthetic>"]`, with the real model only in `modelUsage`),
275
288
  `models` (which model), `ablated` + `context.availableSkills` (which arm). An answer that reads
276
289
  exactly like skill output is not evidence: the skill's own source is mounted where the model can
277
290
  read it — in production too — so on a self-referential prompt it may read `SKILL.md` and answer
@@ -1,6 +1,6 @@
1
1
  # Measurement
2
2
 
3
- Tracks `cowork-harness 4.1.1` (baseline `desktop-2.9939.4`). Read it before comparing runs: `--repeat`, `--ablate-skill`, and the hygiene that keeps a batch valid.
3
+ Tracks `cowork-harness 4.2.0` (baseline `desktop-2.16120.0`). Read it before comparing runs: `--repeat`, `--ablate-skill`, and the hygiene that keeps a batch valid.
4
4
 
5
5
  ### Measure — before/after, with/without (`--repeat`, `--ablate-skill`)
6
6
 
@@ -25,6 +25,11 @@ Every ablated run is stamped `ablated: true` in `result.json` and carries `ablat
25
25
  What the harness gives you here is the run execution and the control arm — designing the comparison
26
26
  (scrubbing giveaways, shuffling, judging blind, unblinding only after grading) is still yours.
27
27
 
28
+ **"Did my edit change it?"** → `cowork-harness eval` (EXPERIMENTAL): the version before your edit and the
29
+ one after, interleaved, with the agent and judge models pinned, compared per assertion and per rubric
30
+ claim with an exact test. It is a regression signal to investigate, not proof — see
31
+ [`eval.md`](eval.md) and Recipe 5 step 6 in [`task-recipes.md`](task-recipes.md).
32
+
28
33
  ### Tool timing — what `toolDurations` measures
29
34
 
30
35
  `result.json`'s `toolDurations` and `trace <run> --view tool-durations` report, per tool, the **wall gap
@@ -1,6 +1,6 @@
1
1
  # Run, record and lock
2
2
 
3
- Tracks `cowork-harness 4.1.1` (baseline `desktop-2.9939.4`). Read it when running a scenario, recording or placing a cassette, reading verdict signals, checking a background run, or choosing CI lanes.
3
+ Tracks `cowork-harness 4.2.0` (baseline `desktop-2.16120.0`). Read it when running a scenario, recording or placing a cassette, reading verdict signals, checking a background run, or choosing CI lanes.
4
4
 
5
5
  ## Part II — RUN, RECORD & LOCK
6
6
 
@@ -193,7 +193,7 @@ Recognize these before "fixing" a non-bug:
193
193
  `markitdown`/`magika` (`ml_extract`), `cv2` (`cv`), `camelot`/`tabula` (`pdf_tables`), or `wand`
194
194
  (`magick`) can trip this even though real Cowork **ships** those (per the rootfs manifest captured at
195
195
  Desktop `2.9939.2` — `baselines/provisioning/rootfs-provisioning.json`, which is the dated evidence
196
- behind that sentence; 1 baseline has shipped since without a re-capture). The message says so ("likely a FALSE
196
+ behind that sentence; 2 baselines have shipped since without a re-capture). The message says so ("likely a FALSE
197
197
  NEGATIVE (real Cowork ships them)"). Fix: rebuild full parity (`--build-arg COWORK_FULL_PARITY=1`, point
198
198
  `COWORK_AGENT_IMAGE` at it), or — if the skill's fallback is genuinely equivalent — assert
199
199
  `allow_missing_capability: true`. (Two sources: a skill *observed using* an omitted family, live lane;
@@ -1,7 +1,7 @@
1
1
  # Scenario & session schema, replay class, web_fetch, authoring gotchas
2
2
 
3
- Self-contained reference for authoring `cowork-harness` scenarios. Tracks `cowork-harness 4.1.1`
4
- (baseline `desktop-2.9939.4`). If your checkout is newer, prefer the live [`docs/scenario.md`](https://github.com/yaniv-golan/cowork-harness/blob/main/docs/scenario.md),
3
+ Self-contained reference for authoring `cowork-harness` scenarios. Tracks `cowork-harness 4.2.0`
4
+ (baseline `desktop-2.16120.0`). If your checkout is newer, prefer the live [`docs/scenario.md`](https://github.com/yaniv-golan/cowork-harness/blob/main/docs/scenario.md),
5
5
  [`docs/session.md`](https://github.com/yaniv-golan/cowork-harness/blob/main/docs/session.md), and `SPEC.md`.
6
6
 
7
7
  **Minimal scenario** — `prompt` and `fidelity` are required:
@@ -2,7 +2,7 @@
2
2
 
3
3
  Each recipe composes facts that live scattered across SKILL.md and the other references into one
4
4
  decision path. Every one answers a question a real fleet owner had to work out the hard way.
5
- Tracks `cowork-harness 4.1.1` (baseline `desktop-2.9939.4`), same as SKILL.md's front-matter. Recipe 2's `resolved-tier`/`unverifiable-tier` staleness classes and
5
+ Tracks `cowork-harness 4.2.0` (baseline `desktop-2.16120.0`), same as SKILL.md's front-matter. Recipe 2's `resolved-tier`/`unverifiable-tier` staleness classes and
6
6
  Recipe 3's `init-redact` shipped in 0.24.0 and are part of the current feature set — no version gate
7
7
  needed if your CLI meets SKILL.md's version floor.
8
8
 
@@ -198,7 +198,9 @@ degrade the advice. It is real work to calibrate; these steps are the traps that
198
198
  correct claim can pass one rep and miss the next. A claim's baseline is its pass *rate* (3/3, 2/3), read
199
199
  from `RunResult.assertions[].semanticClaims`. Do **not** chase single-run all-pass — set `min_pass` to
200
200
  the reliably-hit core for a green verdict, and treat the per-claim rates as the real signal. (N=1
201
- routinely mislabels a stable 0/3 as "intermittent" and vice-versa.)
201
+ routinely mislabels a stable 0/3 as "intermittent" and vice-versa.) `eval` (step 6) defaults to 5 reps
202
+ per arm and refuses fewer than 4 without `--allow-underpowered`: below that, no exact test can flag
203
+ even a total collapse.
202
204
  5. **Check discrimination — does the skill actually help?** Run one rep with the skill NOT installed and
203
205
  compare. `--ablate-skill` is the flag for it: it empties every skill/plugin discovery source for
204
206
  **that one invocation**, so the agent answers from its own priors, and stamps the result
@@ -207,14 +209,26 @@ degrade the advice. It is real work to calibrate; these steps are the traps that
207
209
  A/B — and the rollup labels it `PASS [ABLATED — control arm]` so you cannot bank it as one. A not-invoked rep is not a control:
208
210
  outside `--ablate-skill` it can still read the source (see step 3). If the answer still scores high without the skill, that claim is
209
211
  answerable from priors and tests the model, not your skill — strengthen it (a skill-specific fact) or
210
- drop it. Everything past "run both arms" — scrubbing giveaways, shuffling, judging blind, unblinding
211
- after grading — is yours to build; the harness supplies the runs and the control.
212
- 6. **Gate a change on the profile diff.** Capture the per-claim profile before your edit (the baseline),
213
- make the edit, re-capture, and compare per claim. A claim that DROPPED (e.g. 3/3 → 0/3) is a
214
- **regression signal to investigate**, not proof your edit caused it: at a small number of reps one
215
- observation can move by chance. Re-run that claim and read the reps' transcripts before attributing
216
- it. A claim already at 0/3 (a known gap) cannot regress. That turns "did my SKILL.md refactor quietly
217
- make the advice worse?" into a checkable signal.
212
+ drop it. The harness supplies the runs, this with/without control and, for a before/after of two
213
+ versions, the paired `eval` of step 6; scrubbing giveaways and judging blind stay with you. Before you trust any
214
+ comparison, measure your scenarios' own noise: `eval` with the SAME source as both arms
215
+ (`--allow-identical-arms`) shows how far the rates move when nothing changed.
216
+ 6. **Gate a change with `eval` — a paired comparison of the two versions.** Hand-diffing two profiles
217
+ captured at different times mixes your edit with everything else that moved in between. `eval` runs
218
+ both versions of the plugin in one interleaved schedule, holds the agent and judge models fixed, and
219
+ compares every assertion and every rubric claim with an exact test:
220
+ ```bash
221
+ cowork-harness eval evals/scenarios/ --arm before=git:HEAD:plugins/my-skill --arm after=./plugins/my-skill \
222
+ --model <concrete id> --judge-model <concrete id> --holdout evals/scenarios/untouched-question.yaml
223
+ ```
224
+ The first `--arm` is the baseline. Each arm is snapshotted before the first run (a `git:` arm is read
225
+ from the commit — freeze a recoverable source this way rather than trusting the working tree to stay
226
+ put). A row labelled `possible drop` or `confirmed drop` is a **regression signal to investigate**,
227
+ not proof your edit caused it: open the run dirs the report links for that row, read the transcripts,
228
+ and re-run if the evidence is thin. A claim already at 0% in both arms cannot regress, and one at 100%
229
+ in both cannot show an improvement — the report counts both. Keep `--holdout` scenarios you did not
230
+ tune against; a scenario you shaped the skill to is weak evidence. Details, labels and exit codes:
231
+ [docs/eval.md](https://github.com/yaniv-golan/cowork-harness/blob/main/docs/eval.md).
218
232
 
219
233
  **Lane note:** `semantic_matches` is **live-only** (the judge is a live model call), so these scenarios
220
234
  run on the `run` lane, never token-free `replay` — the linter's "all assertions live-only" warning is
@@ -800,6 +800,58 @@ def _lint_prompt_slash(doc, path):
800
800
  return findings
801
801
 
802
802
 
803
+ _TOOL_RESULT_KEYS = ("tool_result_contains", "tool_result_not_contains", "tool_result_matches", "tool_result_not_matches")
804
+ # `forked execution` as a literal or as a regex spells the gap: a space, an escaped space, `\s`, `.`, or
805
+ # `[\s\S]`, optionally quantified (`\s+`, `.*`).
806
+ _FORKED_EXECUTION_RE = re.compile(r"forked(?: |\\ |\\s|\.|\[\\s\\S\])[+*?]?execution", re.IGNORECASE)
807
+
808
+
809
+ def _lint_slash_prompt_forked_anchor(doc, items, path):
810
+ """W: a `/<skill>` prompt paired with a tool_result_* anchored on `forked execution`.
811
+
812
+ `Skill "<name>" completed (forked execution).` is the `Skill` TOOL RESULT a `context: fork` skill
813
+ returns when the MODEL invokes it. A prompt whose first character is `/` naming a staged skill is
814
+ expanded by the agent binary itself: no `Skill` tool_use is emitted and the fork runs directly, so that
815
+ result text never exists. A positive tool_result_* anchored on it then fails on a working skill, and a
816
+ negative one passes vacuously. Measured on real hostloop runs of one fork skill (agent 2.1.284): the
817
+ bare and plugin-qualified slash prompts both ran the skill with no `Skill` call; a plain prompt got one.
818
+
819
+ Registration is not checkable statically, so the rule fires on any command-shaped leading token and the
820
+ message says "if". It is deliberately narrow: only the fork-result anchor, since `skill_triggered` /
821
+ `no_skill_triggered` already count a slash-invoked skill. First character only, matching the harness's
822
+ own slash detector (a prompt with leading whitespace is not expanded).
823
+ """
824
+ prompt = doc.get("prompt")
825
+ if not isinstance(prompt, str):
826
+ return []
827
+ m = re.match(r"/(\S+)", prompt)
828
+ if not m:
829
+ return []
830
+ name = m.group(1)
831
+ if not _SLASH_CMD_NAME_RE.match(name) or name.lower() in _SLASH_PATH_WORDS:
832
+ return []
833
+ findings = []
834
+ for key in _TOOL_RESULT_KEYS:
835
+ for v in _assert_values(items, key):
836
+ if not isinstance(v, str) or not _FORKED_EXECUTION_RE.search(v):
837
+ continue
838
+ findings.append(
839
+ Finding(
840
+ "WARN",
841
+ "slash-prompt-forked-result-anchor",
842
+ f"`prompt:` starts with `/{name}` and `{key}` anchors on `forked execution`. If `/{name}` is a "
843
+ "staged skill, the agent expands the command itself — no `Skill` tool call, so the "
844
+ "`completed (forked execution)` tool result never exists: a positive check fails on a working "
845
+ "skill and a negative one passes vacuously.",
846
+ f"Assert the invocation with `skill_triggered: '{name.split(':')[-1]}'` (it counts a slash-invoked "
847
+ "skill) and the answer with `transcript_matches`; or drop the slash if "
848
+ "the scenario means to test the model invoking the skill through the `Skill` tool.",
849
+ path,
850
+ )
851
+ )
852
+ return findings
853
+
854
+
803
855
  # --- the object form of tool_called / tool_not_called, and transcript_* values shaped like a command ---
804
856
 
805
857
  # `transcript_*` reads top-level assistant prose ONLY — never a tool_use — so a value shaped like a shell
@@ -1237,6 +1289,8 @@ def lint_doc(doc, path, raw_lines, cassette_records=None):
1237
1289
  # linter stays offline — the message carries the gate fact instead of reading a baseline).
1238
1290
  findings.extend(_lint_tool_call_object_form(items, fidelity, path))
1239
1291
  findings.extend(_lint_transcript_command_shaped(items, path))
1292
+ # W: a `/<skill>` prompt never produces the `Skill` fork tool result — see _lint_slash_prompt_forked_anchor.
1293
+ findings.extend(_lint_slash_prompt_forked_anchor(doc, items, path))
1240
1294
  if "transcript_no_host_path" in assert_keys:
1241
1295
  if fidelity in ("hostloop", "protocol"):
1242
1296
  findings.append(
@@ -1955,6 +2009,7 @@ LINT_RULES = {
1955
2009
  "reference-access-contradiction": "ERROR",
1956
2010
  "regex-double-quoted": "WARN",
1957
2011
  "replay-noop": "WARN",
2012
+ "slash-prompt-forked-result-anchor": "WARN",
1958
2013
  "tool-called-always-passes": "INFO",
1959
2014
  "tool-input-regex-redactable": "WARN",
1960
2015
  "tool-input-shell-tier": "INFO",
@@ -3065,7 +3120,7 @@ def _lint_skill_corpus_size(md_path):
3065
3120
  #
3066
3121
  # Body cap only: the re-attach estimate counts UTF-16 units and the rule counts UTF-8 BYTES, which are never
3067
3122
  # fewer, so it errs early. The reference cap has no such guarantee — it rests on the measured ratio above.
3068
- _SKILL_SIZE_CAPS_VERIFIED = "binary-verified against agent 2.1.281 (VM ELF and native, both read)"
3123
+ _SKILL_SIZE_CAPS_VERIFIED = "binary-verified against agent 2.1.284 (VM ELF and native, both read)"
3069
3124
  _SKILL_BODY_REATTACH_CAP = 19_000
3070
3125
  _SKILL_BODY_NOTICE_RATIO = 0.8
3071
3126
  _SKILL_REFERENCE_READ_CAP = 60_000