cowork-harness 4.2.1 → 4.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (203) hide show
  1. package/.claude/skills/cowork-harness/SKILL.md +16 -8
  2. package/.claude/skills/cowork-harness/references/assertion-catalog.md +32 -28
  3. package/.claude/skills/cowork-harness/references/assertions-guide.md +52 -6
  4. package/.claude/skills/cowork-harness/references/authoring.md +28 -8
  5. package/.claude/skills/cowork-harness/references/ci-recipe.md +22 -15
  6. package/.claude/skills/cowork-harness/references/critique.md +1 -1
  7. package/.claude/skills/cowork-harness/references/debugging.md +20 -5
  8. package/.claude/skills/cowork-harness/references/eval.md +11 -3
  9. package/.claude/skills/cowork-harness/references/fidelity-and-answers.md +16 -4
  10. package/.claude/skills/cowork-harness/references/gotchas.md +47 -24
  11. package/.claude/skills/cowork-harness/references/hillclimb-recipe.md +208 -0
  12. package/.claude/skills/cowork-harness/references/hillclimb.md +450 -0
  13. package/.claude/skills/cowork-harness/references/measurement.md +1 -1
  14. package/.claude/skills/cowork-harness/references/run-record-replay.md +49 -19
  15. package/.claude/skills/cowork-harness/references/scenario-schema.md +37 -13
  16. package/.claude/skills/cowork-harness/references/semantic-judging.md +9 -0
  17. package/.claude/skills/cowork-harness/references/task-recipes.md +27 -10
  18. package/.claude/skills/cowork-harness/scripts/assertion-keys.json +100 -1
  19. package/.claude/skills/cowork-harness/scripts/scenario.py +766 -89
  20. package/AGENTS.md +6 -1
  21. package/CHANGELOG.md +1362 -0
  22. package/DESIGN.md +11 -6
  23. package/README.md +32 -14
  24. package/RELEASING.md +152 -10
  25. package/SPEC.md +258 -28
  26. package/baselines/desktop-2.16120.0.json +2 -2
  27. package/baselines/desktop-2.19675.0.json +1174 -0
  28. package/baselines/prompts/cowork-system-prompt-fingerprints.json +9 -0
  29. package/dist/agent/session.js +3 -2
  30. package/dist/assert.js +1028 -216
  31. package/dist/baseline.js +440 -64
  32. package/dist/boundary-paths.js +6 -2
  33. package/dist/cli.js +198 -75
  34. package/dist/critique/command.js +85 -39
  35. package/dist/critique/evaluator.js +2 -2
  36. package/dist/critique/evidence.js +11 -3
  37. package/dist/critique/scrub-artifacts.js +92 -0
  38. package/dist/critique/skill-invocation.js +19 -8
  39. package/dist/decide/decider.js +40 -11
  40. package/dist/decide/llm-transport.js +205 -18
  41. package/dist/decide/pairwise-judge.js +205 -0
  42. package/dist/decide/semantic-judge.js +4 -1
  43. package/dist/effort-env.js +13 -0
  44. package/dist/egress/sidecar.js +45 -4
  45. package/dist/eval/classify.js +94 -27
  46. package/dist/eval/command.js +361 -132
  47. package/dist/eval/pins.js +5 -4
  48. package/dist/eval/plan-history.js +266 -0
  49. package/dist/eval/plan.js +336 -0
  50. package/dist/eval/planner.js +457 -0
  51. package/dist/eval/report.js +102 -28
  52. package/dist/eval/runs.js +52 -3
  53. package/dist/eval/usage.js +27 -5
  54. package/dist/fixture/cli.js +60 -0
  55. package/dist/fixture/export.js +314 -0
  56. package/dist/fixture/usage.js +10 -0
  57. package/dist/fixture/workspace.js +412 -0
  58. package/dist/hillclimb/answer-key.js +47 -0
  59. package/dist/hillclimb/args.js +150 -0
  60. package/dist/hillclimb/cases.js +106 -0
  61. package/dist/hillclimb/check.js +244 -0
  62. package/dist/hillclimb/cli.js +535 -0
  63. package/dist/hillclimb/command.js +167 -0
  64. package/dist/hillclimb/cost.js +291 -0
  65. package/dist/hillclimb/flow.js +286 -0
  66. package/dist/hillclimb/freeze-ref.js +227 -0
  67. package/dist/hillclimb/fs.js +347 -0
  68. package/dist/hillclimb/gate.js +139 -0
  69. package/dist/hillclimb/grade-keys.js +263 -0
  70. package/dist/hillclimb/ids.js +40 -0
  71. package/dist/hillclimb/job.js +107 -0
  72. package/dist/hillclimb/judge-rollup.js +45 -0
  73. package/dist/hillclimb/metric-keys.js +129 -0
  74. package/dist/hillclimb/outputs.js +173 -0
  75. package/dist/hillclimb/pairwise.js +117 -0
  76. package/dist/hillclimb/present.js +23 -0
  77. package/dist/hillclimb/regrade.js +1648 -0
  78. package/dist/hillclimb/result-event.js +46 -0
  79. package/dist/hillclimb/rows.js +410 -0
  80. package/dist/hillclimb/run-command.js +442 -0
  81. package/dist/hillclimb/runner.js +739 -0
  82. package/dist/hillclimb/schema-check.js +113 -4
  83. package/dist/hillclimb/served-model.js +96 -0
  84. package/dist/hillclimb/skill.js +228 -0
  85. package/dist/hillclimb/snapshot.js +109 -0
  86. package/dist/hillclimb/state-template.js +134 -0
  87. package/dist/hillclimb/trace.js +432 -0
  88. package/dist/hillclimb/usage.js +118 -0
  89. package/dist/hostloop/plugin-path-rewrite.js +130 -0
  90. package/dist/hostloop/workspace-handler.js +4 -3
  91. package/dist/io.js +62 -1
  92. package/dist/loop-decision.js +47 -0
  93. package/dist/metrics.js +144 -0
  94. package/dist/refs/cli-usage.js +17 -0
  95. package/dist/refs/cli.js +172 -0
  96. package/dist/refs/compose.js +142 -0
  97. package/dist/refs/preflight.js +80 -0
  98. package/dist/refs/store.js +346 -0
  99. package/dist/run/artifacts.js +14 -0
  100. package/dist/run/budget-status.js +61 -0
  101. package/dist/run/budget.js +130 -22
  102. package/dist/run/cassette.js +866 -52
  103. package/dist/run/chat-result.js +9 -2
  104. package/dist/run/chat.js +9 -5
  105. package/dist/run/command-globals.js +5 -0
  106. package/dist/run/doctor.js +96 -14
  107. package/dist/run/envelope.js +18 -5
  108. package/dist/run/execute.js +388 -76
  109. package/dist/run/gate-provenance.js +1 -0
  110. package/dist/run/hook-events.js +38 -1
  111. package/dist/run/host-path-tokens.js +18 -0
  112. package/dist/run/input-host-paths.js +94 -12
  113. package/dist/run/input-request.js +118 -0
  114. package/dist/run/inspect-view.js +20 -2
  115. package/dist/run/lint-load.js +63 -0
  116. package/dist/run/model-provenance.js +1 -1
  117. package/dist/run/outputs-delete-tier.js +16 -0
  118. package/dist/run/pairwise-prepass.js +241 -0
  119. package/dist/run/pre-run-manifest.js +2 -2
  120. package/dist/run/regrade-usage.js +14 -0
  121. package/dist/run/regrade.js +1074 -0
  122. package/dist/run/renderer.js +7 -5
  123. package/dist/run/run-dir-identity.js +165 -0
  124. package/dist/run/run-index.js +6 -3
  125. package/dist/run/run-labels.js +76 -0
  126. package/dist/run/run-status.js +1 -1
  127. package/dist/run/run.js +47 -9
  128. package/dist/run/runs-gc.js +482 -61
  129. package/dist/run/scaffold.js +20 -1
  130. package/dist/run/scenario-tool.js +1 -1
  131. package/dist/run/skill-metadata.js +21 -1
  132. package/dist/run/subagent-reasoning.js +3 -2
  133. package/dist/run/trace-view.js +14 -5
  134. package/dist/run/turn-events.js +1 -1
  135. package/dist/run/verdict.js +61 -18
  136. package/dist/run/verify-context.js +59 -8
  137. package/dist/runtime/argv.js +19 -4
  138. package/dist/runtime/hostloop-stage.js +4 -0
  139. package/dist/runtime/hostloop.js +140 -13
  140. package/dist/runtime/protocol.js +51 -11
  141. package/dist/runtime/stage.js +6 -0
  142. package/dist/scan.js +9 -0
  143. package/dist/scrub-set.js +196 -0
  144. package/dist/secrets.js +5 -1
  145. package/dist/session.js +133 -15
  146. package/dist/skill-id.js +16 -0
  147. package/dist/sync/cowork-sync.js +270 -15
  148. package/dist/tool-call-assert.js +5 -2
  149. package/dist/types.js +307 -27
  150. package/docs/README.md +3 -1
  151. package/docs/boundary.md +5 -3
  152. package/docs/cassette.md +79 -16
  153. package/docs/ci.md +3 -3
  154. package/docs/cli.md +435 -28
  155. package/docs/companion-skill.md +2 -2
  156. package/docs/critique.md +54 -8
  157. package/docs/debugging.md +7 -0
  158. package/docs/discovery.md +1 -1
  159. package/docs/eval.md +145 -9
  160. package/docs/fidelity-gaps.md +255 -61
  161. package/docs/gotchas.md +19 -4
  162. package/docs/hillclimb.md +708 -0
  163. package/docs/invariants.md +1 -0
  164. package/docs/maintenance.md +32 -11
  165. package/docs/plugin-root.md +70 -46
  166. package/docs/scenario.md +279 -54
  167. package/docs/session.md +17 -7
  168. package/docs/stats.md +1 -1
  169. package/docs/subagents.md +14 -9
  170. package/examples/README.md +3 -1
  171. package/examples/hillclimb/README.md +142 -0
  172. package/examples/hillclimb/data/orders.csv +25 -0
  173. package/examples/hillclimb/plugin/.claude-plugin/plugin.json +1 -0
  174. package/examples/hillclimb/plugin/skills/sales-brief/SKILL.md +15 -0
  175. package/examples/hillclimb/plugin/skills/sales-brief/scripts/profile.py +143 -0
  176. package/examples/hillclimb/scenarios/control-regions.yaml +18 -0
  177. package/examples/hillclimb/scenarios/data-quality.yaml +27 -0
  178. package/examples/hillclimb/scenarios/manager-brief.yaml +27 -0
  179. package/examples/hillclimb/scenarios/totals.yaml +22 -0
  180. package/examples/hillclimb/sessions/sales-brief.yaml +8 -0
  181. package/examples/replays/README.md +4 -2
  182. package/examples/replays/example-multiselect-gate.cassette.json +51 -52
  183. package/examples/replays/example-pdf-skill-ci-selftest-failing.yaml +1 -1
  184. package/examples/replays/example-pdf-skill.cassette.json +104 -92
  185. package/examples/replays/hostloop-computer-links.cassette.json +60 -60
  186. package/examples/scenarios/example-pdf-skill.yaml +1 -1
  187. package/examples/scenarios/hostloop-computer-links.yaml +1 -1
  188. package/examples/skills/csv-fx-normalize/skills/csv-fx-normalize/SKILL.md +4 -3
  189. package/examples/skills/csv-metrics/skills/csv-metrics/SKILL.md +4 -3
  190. package/llms.txt +9 -4
  191. package/package.json +2 -1
  192. package/python/test_scenario_lint.py +231 -1
  193. package/schema/cassette.v14.json +346 -0
  194. package/schema/critique-report.json +7 -1
  195. package/schema/regrade.json +275 -0
  196. package/schema/run-result.json +97 -9
  197. package/schema/scenario.schema.json +453 -19
  198. package/schema/schedule-cost.json +50 -0
  199. package/schema/verify-cassettes.json +1 -1
  200. package/scripts/check-surface.ts +113 -0
  201. package/scripts/check-versions.ts +2 -1
  202. package/scripts/gen-schema.ts +31 -1
  203. package/scripts/release-preflight.ts +81 -1
@@ -1,10 +1,10 @@
1
1
  ---
2
2
  name: cowork-harness
3
- description: Test or debug a Claude Code skill/plugin under Claude Cowork's runtime — sandboxed agent, default-deny egress, the can_use_tool permission/question protocol — using the cowork-harness CLI. Use when validating or regression-testing a skill, authoring or debugging a scenario YAML (prompt + scripted answers + assert:), choosing a fidelity tier, scripting AskUserQuestion / tool-permission answers, asserting artifacts, egress, or sub-agent dispatch, measuring how long each tool call took (toolDurations / trace), or debugging a failed run or verdict from its result.json or transcript. Especially when a harness run no-ops an assertion, fails on an unanswered gate, false-greens, a steered answer never reaches the model, or a web_fetch is unexpectedly denied or gated. Also when iterating or hardening a skill across fixes, or grounding a skill's self-critique against its own run evidence — including a document-analysis skill (cap table, deck, financial model, transcript) that needs an uploaded file attached to be critiqued at all. Also for comparing two versions of a skill before merging an edit — did the change make its answers worse? (`eval`: paired, interleaved A/B of two plugin versions, pinned models). NOT for generic unit testing (pytest/vitest of your own scripts) or non-Cowork CI. Covers the skill / run / chat / record / replay / trace / decide / assertions / scaffold / critique / stats / eval commands and the session-vs-scenario split.
3
+ description: Test or debug a Claude Code skill/plugin under Claude Cowork's runtime — sandboxed agent, default-deny egress, the can_use_tool permission/question protocol — using the cowork-harness CLI. Use when validating or regression-testing a skill, authoring or debugging a scenario YAML (prompt + scripted answers + assert:), choosing a fidelity tier, scripting AskUserQuestion / tool-permission answers, asserting artifacts, egress, or sub-agent dispatch, measuring how long each tool call took (toolDurations / trace), or debugging a failed run or verdict from its result.json or transcript. Especially when a harness run no-ops an assertion, fails on an unanswered gate, false-greens, a steered answer never reaches the model, or a web_fetch is unexpectedly denied or gated. Also when iterating or hardening a skill across fixes, or grounding a skill's self-critique against its own run evidence — including a document-analysis skill (cap table, deck, financial model, transcript) that needs an uploaded file attached to be critiqued at all. Also for comparing two versions of a skill before merging an edit — did it make answers worse? (`eval`: paired A/B, pinned models) — or improving one round by round (`hillclimb`, the `/claude-api hillclimb` runner). NOT for generic unit testing (pytest/vitest of your own scripts) or non-Cowork CI. Covers the skill / run / chat / record / replay / trace / decide / assertions / scaffold / critique / stats / eval / hillclimb commands and the session-vs-scenario split.
4
4
  metadata:
5
5
  author: cowork-harness
6
- version: 4.2.1
7
- tracks-harness: cowork-harness 4.2.1 (baseline desktop-2.16120.0)
6
+ version: 4.4.0
7
+ tracks-harness: cowork-harness 4.4.0 (baseline desktop-2.19675.0)
8
8
  ---
9
9
 
10
10
  # cowork-harness
@@ -26,8 +26,8 @@ allowlist). This skill exists mostly to keep you out of those traps — the *Inv
26
26
  full landmine catalog in [`references/gotchas.md`](references/gotchas.md) are the highest-value part.
27
27
  Read them.
28
28
 
29
- > **Version note:** the facts and `file:line` pointers here track `cowork-harness 4.2.1` (baseline
30
- > `desktop-2.16120.0`). If your checkout is newer, prefer the live `--help` and — in a repo checkout —
29
+ > **Version note:** the facts and `file:line` pointers here track `cowork-harness 4.4.0` (baseline
30
+ > `desktop-2.19675.0`). If your checkout is newer, prefer the live `--help` and — in a repo checkout —
31
31
  > `SPEC.md` / `docs/*.md` over this snapshot, and re-run the bundled linter.
32
32
 
33
33
  ## Preflight — make sure the harness can actually run
@@ -43,7 +43,7 @@ Before the first command, confirm the CLI is reachable and **fail loud** (never
43
43
 
44
44
  - **One-shot check.** Run `cowork-harness doctor [--tier <tier>]` first — a read-only prerequisite check that inspects Docker, the staged agent, the token, and the baseline in one pass. The bullets below explain each thing it checks (and how to fix it).
45
45
  - **Replay-only? Skip `doctor`.** Replaying committed cassettes needs no Docker, no staged agent, and no token — and every tier's `doctor` validates the auth token (the live tiers also Docker + the staged agent), so a ✗ there is expected, not a blocker. Go straight to `cowork-harness replay <cassette>`.
46
- - **CLI on PATH, recent enough?** Run `cowork-harness --version` — this skill needs **≥ 4.2.1**. If it's missing or older, prefix every command with the version floor `npx "cowork-harness@^4.2.1" <cmd>` (Node ≥ 22), or install once with `npm i -g "cowork-harness@^4.2.1"`. **Pin `@^4.2.1`, never `@latest`** — `@latest` can silently fetch an older CLI and the new commands fail as "unknown command", whereas the floor **fails loud** if no compatible version is published.
46
+ - **CLI on PATH, recent enough?** Run `cowork-harness --version` — this skill needs **≥ 4.4.0**. If it's missing or older, prefix every command with the version floor `npx "cowork-harness@^4.4.0" <cmd>` (Node ≥ 22), or install once with `npm i -g "cowork-harness@^4.4.0"`. **Pin `@^4.4.0`, never `@latest`** — `@latest` can silently fetch an older CLI and the new commands fail as "unknown command", whereas the floor **fails loud** if no compatible version is published.
47
47
 
48
48
  This skill documents the CURRENT surface, not release history. If `cowork-harness --version` is
49
49
  OLDER than the floor, the per-release record of what you are missing is [CHANGELOG.md](https://github.com/yaniv-golan/cowork-harness/blob/main/CHANGELOG.md)
@@ -79,7 +79,12 @@ CI-grade scenario, and the post-hoc debug loop; the rest are narrower tools that
79
79
  correct answers after you edit it?) → author `semantic_matches` scenarios, then compare the version
80
80
  before your edit with the one after using `cowork-harness eval` (EXPERIMENTAL, live: 10 runs per
81
81
  scenario at the defaults). See **Recipe 5** step 6 in `references/task-recipes.md` (validity,
82
- discrimination — the traps) and [`references/eval.md`](references/eval.md).
82
+ discrimination — the traps) and [`references/eval.md`](references/eval.md). `eval` compares two versions you
83
+ already have; `hillclimb` runs each round of the loop that produces them; `critique` finds what is wrong; `skill`/`run`
84
+ checks it works.
85
+ - **Improve a skill round by round** with Claude Code's `/claude-api hillclimb` → `cowork-harness hillclimb run`
86
+ is the loop's runner. Follow [`references/hillclimb-recipe.md`](references/hillclimb-recipe.md) (the procedure,
87
+ step by step); every flag, refusal and exit code is in [`references/hillclimb.md`](references/hillclimb.md).
83
88
  - **"What is WRONG with this skill?"** (a graded critique, not a pass/fail) → `cowork-harness critique
84
89
  <folder> --prompt "<probe>"`. Up to four model workloads (zero with `--corpus-only`; pass 2 is skipped with no self-report) and 10–20 minutes; budget from
85
90
  `report.costUsd.totalUsd`. Reach for it when you want **findings**. **For "what does this skill
@@ -99,7 +104,7 @@ CI-grade scenario, and the post-hoc debug loop; the rest are narrower tools that
99
104
 
100
105
  Full command set: `skill · run · chat · record · replay · verify-cassettes · rehash · prune · migrate-run-dir · lint ·
101
106
  lint-skill · analyze-skill · probe-dispatch ·
102
- verify-run · trace · inspect · diff · critique · eval · eval report · stats · decide · gates · answer · scaffold · assertions --list · sync ·
107
+ verify-run · regrade · fixture · ref <freeze|verify> · trace · inspect · diff · critique · eval · eval report · hillclimb <run|check|state-template|freeze-ref|regrade> · stats · decide · gates · answer · scaffold · assertions --list · sync ·
103
108
  list · boundary-check · status · vm <init|status|delete|prune> · doctor · init-redact`. Always check `cowork-harness <cmd> --help`.
104
109
 
105
110
  ## Invariants — how a green run lies
@@ -153,9 +158,12 @@ behind each, is [`references/gotchas.md`](references/gotchas.md).
153
158
  | [`references/gotchas.md`](references/gotchas.md) | the full "✓ passed ≠ correct" landmine catalog |
154
159
  | [`references/task-recipes.md`](references/task-recipes.md) | start here for "how do I X": evolve `assert:`, audit tier drift, redaction, budgets, answer quality |
155
160
  | [`references/assertion-catalog.md`](references/assertion-catalog.md) | every `assert:` key's semantics, the verdict-signal table |
161
+ | [`references/semantic-judging.md`](references/semantic-judging.md) | `semantic_matches` in full: what the judge reads, fork results, refusal reasons, provenance |
156
162
  | [`references/scenario-schema.md`](references/scenario-schema.md) | every YAML field, which keys survive `replay`, the `web_fetch` model |
157
163
  | [`references/fidelity-and-answers.md`](references/fidelity-and-answers.md) | tier semantics, answer paths, the determinism contract |
158
164
  | [`references/ci-recipe.md`](references/ci-recipe.md) | the GitHub Action, replay-vs-live lanes, the four-stage pipeline |
159
165
  | [`references/critique.md`](references/critique.md) | `critique` report and evidence-package shapes |
160
166
  | [`references/eval.md`](references/eval.md) | `eval`: paired before/after of two plugin versions — labels, refusals, exit codes, files |
167
+ | [`references/hillclimb.md`](references/hillclimb.md) | `hillclimb run` / `check` / `state-template` / `freeze-ref` / `regrade` for a `/claude-api hillclimb` loop — pass `--flow` to each, the harness gate, resume, pairwise references, re-grading, refusals, exit codes |
168
+ | [`references/hillclimb-recipe.md`](references/hillclimb-recipe.md) | Recipe 7: the `/claude-api hillclimb` loop with the harness as its runner, step by step — the runner command, Step 0.5 checks, metrics, spend, rounds, stalls |
161
169
  | `scripts/scenario.py` | `scaffold`, `lint`, `lint-skill`, `resolve-agent-types <plugin-dir>` (validates a pinned `subagent_type` against `plugin.json` + `agents/*.md`) |
@@ -1,6 +1,6 @@
1
1
  # Assertion catalog
2
2
 
3
- Tracks `cowork-harness 4.2.1` (baseline `desktop-2.16120.0`). Every `assert:` key with its semantics, and the
3
+ Tracks `cowork-harness 4.4.0` (baseline `desktop-2.19675.0`). Every `assert:` key with its semantics, and the
4
4
  verdict-signal table. Which keys survive `replay` is in [`scenario-schema.md`](./scenario-schema.md#replay-class);
5
5
  the scenario and session YAML fields are there too.
6
6
 
@@ -24,20 +24,20 @@ same set live from the schema.
24
24
  | `transcript_not_contains: <str>` | it does not **Sees top-level `assistant_text` only — it excludes every `tool_use`/`tool_result`**, so text emitted inside a tool call (a gate question, an option label or description) can never match; use `question_context`/`question_asked` or `tool_result_contains` for those. |
25
25
  | `transcript_matches: <regex>` | the transcript matches (case-insensitive) — for stochastic prose **Sees top-level `assistant_text` only — it excludes every `tool_use`/`tool_result`**, so text emitted inside a tool call (a gate question, an option label or description) can never match; use `question_context`/`question_asked` or `tool_result_contains` for those. |
26
26
  | `transcript_not_matches: <regex>` | it does not match (e.g. no leaked stack trace) **Sees top-level `assistant_text` only — it excludes every `tool_use`/`tool_result`**, so text emitted inside a tool call (a gate question, an option label or description) can never match; use `question_context`/`question_asked` or `tool_result_contains` for those. |
27
- | `file_exists: <path>` | the path exists under the run's `work/` (anchored at `mnt/`, e.g. `outputs/x.md`). For a user-facing deliverable prefer `user_visible_artifact` — with a connected folder the file lands in `mnt/<folder>` (= `{{workspaceFolder}}`), not `mnt/outputs`, so `file_exists: outputs/x.md` misses it |
28
- | `user_visible_artifact: <path>` | exists **and** under a user-visible root (`outputs/` + each connected folder's mount name) — the right primitive for a workspace deliverable when a folder is connected |
29
- | `no_delete_in_outputs: true` | no delete op touched `mnt/outputs` — **only `true` is valid**; `false` is rejected. **Omitting it does NOT allow deletes** — a detected delete still fails via the `outputs_delete` verdict signal, which fires *because* the key was not authored; use `allow_outputs_delete: true` to accept an intended one. Detection is a post-run bash-command scan plus a per-turn filesystem diff of `outputs/`, not mount enforcement, so a green means none was *detected*. Fails when the diff proves the delete, when a delete in command/call position has an `outputs/` path as its own operand, or when the diff could not verify; a hit resting only on the detector's inference (e.g. a Python variable named `rm`) with a clean diff passes, and the `outputs_delete_unconfirmed` warn is still raised in the run output — see that code for the classes of real delete that land there |
30
- | `no_delete_in_mounts: true` | no delete op touched ANY delete-denied mount — `outputs` plus every `rw` connected folder — except those waived by `allow_delete_in`. Production denies `unlink`/`rmdir` on every such mount, so `no_delete_in_outputs` covers only part of the real rule. **only `true` is valid** |
27
+ | `file_exists: <path>` | the path exists under the run's `work/` (anchored at `mnt/`, e.g. `outputs/x.md`). For a user-facing deliverable prefer `user_visible_artifact` — with a connected folder the file lands in `mnt/<folder>` (= `{{workspaceFolder}}`), not `mnt/outputs`, so `file_exists: outputs/x.md` misses it. Object form `{path, authored}`: `authored: true` also requires that THIS run created or rewrote the file (an untouched pre-run file — e.g. from a `workspace_fixture` — fails; no pre-run manifest ⇒ evidence-unavailable); `authored: false` states that inheriting it is fine. On a `workspace_fixture` path one of the two is REQUIRED (refused at load otherwise) |
28
+ | `user_visible_artifact: <path>` | exists **and** under a user-visible root (`outputs/` + each connected folder's mount name) — the right primitive for a workspace deliverable when a folder is connected. Takes the same `{path, authored}` object form as `file_exists` (and `artifact_text`/`artifact_json` take an `authored` field) |
29
+ | `no_delete_in_outputs: true` | no delete op touched `mnt/outputs` — **only `true` is valid**; `false` is rejected. Checks on **every** baseline. Omitting it: on a baseline recording outputs `rw` (Desktop before 2.16120.0) a detected delete still fails via the `outputs_delete` verdict signal, which fires *because* the key was not authored (`allow_outputs_delete: true` accepts an intended one); on `rwd` (2.16120.0+, including `latest`, where Cowork allows deletes in outputs) nothing checks outputs deletes, so author this key to keep the check. Detection is a post-run bash-command scan plus a per-turn filesystem diff of `outputs/`, not mount enforcement, so a green means none was *detected*. Fails when the diff proves the delete, when a delete in command/call position has an `outputs/` path as its own operand, or when the diff could not verify; a hit resting only on the detector's inference (e.g. a Python variable named `rm`) with a clean diff passes, and the `outputs_delete_unconfirmed` warn is still raised in the run output — see that code for the classes of real delete that land there |
30
+ | `no_delete_in_mounts: true` | no delete op touched `outputs` or any `rw` connected folder, except mounts waived by `allow_delete_in`. Covers `outputs` on **every** baseline, including those where Cowork allows an outputs delete (unless outputs is waived, authoring it arms the outputs check, so a filesystem-proven delete fails as the `outputs_delete` signal, which `allow_outputs_delete` waives). Production denies `unlink`/`rmdir` on a `rw` connected folder until approval, so `no_delete_in_outputs` covers only part of the rule. **only `true` is valid** |
31
31
  | `no_unexpected_files: [<glob>, …]` | every **newly created** file under a user-visible root matches ≥1 glob (workRoot-relative paths; `**` = whole path segment for any depth — use `outputs/handoff/**` for per-run subdirs); `[]` = no new files; **new-files-only** — overwriting a pre-existing file in place is invisible (use content-level producer stamping); live/verify-run without a pre-run manifest ⇒ evidence-unavailable hard-fail (live runs capture the baseline only when this key is asserted; recordings always capture); an **incomplete post-run filesystem walk** (an unreadable subtree — a permission/I-O error) also fails evidence-unavailable rather than reporting "no strays" over a partial tree — distinct from the missing-manifest case (a `--resume` run), which fails for a different reason; captured on every live sandbox tier including microvm (its outputs are snapshotted from the VM into the run dir); replay needs `cassette.preRunPaths` (≥0.24 recordings) — cassettes without it **exclude** the key with a loud warning. ⚠️ **Not a stand-in for "file X must not exist":** there is no negative-existence key, and inverting this one is a different claim — it is new-files-only (a pre-existing file is invisible to it) and needs a pre-run manifest. Assert the content-level consequence instead (`tool_not_called`, or the absent `artifact_json`) rather than widening an allowlist for every incidental lock/temp file |
32
- | `input_unmodified: <glob>` or `[<glob>, …]` | a single glob or a list; every **pre-existing** file (incl. uploaded files under `uploads/**`) whose workRoot-relative path matches ≥1 glob keeps an unchanged content hash after the run — the in-place-mutation companion to `no_unexpected_files`'s new-files check (`[]` is rejected by the schema — list at least one glob); a glob that matches **no** pre-run path fails loud (a typo or renamed mount would otherwise verify zero files and pass vacuously); a matched file that was deleted counts as a content change (fails); live/verify-run without a pre-run hash manifest ⇒ evidence-unavailable hard-fail (a `--resume` run); captured on every live sandbox tier including microvm; replay needs `cassette.preRunHashes` — cassettes without it **exclude** the key with a loud warning; on replay it compares against the manifest's recorded `sha256`, never a re-hash of the materialized tree |
33
- | `self_heal_ran: <bool>` | a plugin-root self-heal script was (not) invoked |
32
+ | `input_unmodified: <glob> \| [<glob>, …]` | a single glob or a list; every **pre-existing** file (incl. uploaded files under `uploads/**`) whose workRoot-relative path matches ≥1 glob keeps an unchanged content hash after the run — the in-place-mutation companion to `no_unexpected_files`'s new-files check (`[]` is rejected by the schema — list at least one glob); a glob that matches **no** pre-run path fails loud (a typo or renamed mount would otherwise verify zero files and pass vacuously); a matched file that was deleted counts as a content change (fails); live/verify-run without a pre-run hash manifest ⇒ evidence-unavailable hard-fail (a `--resume` run); captured on every live sandbox tier including microvm; replay needs `cassette.preRunHashes` — cassettes without it **exclude** the key with a loud warning; on replay it compares against the manifest's recorded `sha256`, never a re-hash of the materialized tree |
33
+ | `self_heal_ran: <bool>` | a bash command the model wrote did (not) name a plugin under `/sessions/<id>/mnt/.local-plugins/` or `/sessions/<id>/mnt/.remote-plugins/` — the plugin-root self-heal path. It reads the command as the model wrote it, so a host plugin path that `hostloop`'s bash tool rewrote to the VM mount does not count |
34
34
  | `file_absent: <path>` | the named path does **not** exist under the work root after the run — the direct negative-existence key. Do NOT invert `no_unexpected_files` for this: that is an allowlist over NEW files only, and it needs a pre-run manifest. **LIVE/verify-run only** (a cassette records no walk health, so absence is unprovable on replay); evidence-unavailable on `lane: remote` / `preRunOrigin: remote-unavailable` |
35
- | `artifact_text: {artifact, contains?, not_contains?, matches?, not_matches?}` | assert over a delivered artifact's TEXT body — `artifact_json`'s companion for non-JSON files, and how you prove an internal name did not leak into a file the user receives. Literal path (no glob), so one entry per delivered surface. Manifest-class; body-less / symlinked / over-cap targets fail evidence-unavailable, and a non-UTF-8 body fails the NEGATIVE matchers rather than passing against bytes it never read |
35
+ | `artifact_text: {artifact, contains?: [..], not_contains?: [..], matches?, not_matches?}` | assert over a delivered artifact's TEXT body — `artifact_json`'s companion for non-JSON files, and how you prove an internal name did not leak into a file the user receives. Literal path (no glob), so one entry per delivered surface. Manifest-class; body-less / symlinked / over-cap targets fail evidence-unavailable, and a non-UTF-8 body fails the NEGATIVE matchers rather than passing against bytes it never read |
36
36
  | `no_lost_write_back: true` | fails if the run authored an interactive HTML artifact (or a `.py`/`.js` generator of one) whose **relative** Submit/POST write-back is lost under Cowork (served from Cowork's own origin → resolves non-ok, a "Saved!" is silently false). Runs the shipped **static Tier A** analyzer over the files the run authored (diffed vs the pre-run manifest). A lost write-back on an **added** agent-authored source (`outputs/`, scratchpad) **fails**; a **pre-existing** file the skill only modified on a read-write mount is **advisory**; `-suspect` findings surface but pass. **Only `true` is valid**. **Live/verify-run only** — skipped-loud on replay; runs on every live sandbox tier including **microvm** (its outputs are snapshotted from the VM into the run dir); could-not-verify (fail-closed) on a `--resume` scratchpad or an unanalyzable candidate |
37
37
  | `tool_called: <glob>` | a tool the agent ran matched this **glob** — `*` = any run, `?` = one char, exact when literal, anchored + case-sensitive. Exact name (`Write`) matches only that tool; `mcp__workspace__*` matches any workspace tool. GLOB, not regex (`.` is literal) — an empty glob, or one containing a regex/brace-expansion metacharacter (`.*`, `.+`, `\|`, `()`, `[]`, `+`, `^`, `$`, `{}`, `\d`/`\w`/`\s`/`\b`), is **rejected at load** (a hard schema error, not a runtime warning) — it would match no real tool name and pass a `_not_`/`_absent` assert vacuously. Applies whether the glob comes from an authored scenario or a recorded cassette's frozen assert. `cowork-harness lint` reports it too (ERROR `scenario-invalid` — the wrapper runs the same loader); the bundled `scenario.py lint` run directly does NOT |
38
38
  | `tool_not_called: <glob>` | NO tool the agent ran matched this glob (`mcp__*` = "no MCP tool ran"). Same glob semantics as `tool_called`, including the empty/regex-ish rejection |
39
- | `tool_called: {tool, input?, input_any?, result?, scope?, subagent_type?, count?}` | **object form** — asserts what a call CARRIED, where it RAN and what its paired result SAID, which no name glob and no `transcript_*` key can see (`transcript_*` reads top-level prose only, so it passes when the agent merely *says* it ran a command). `tool`: a glob or a list of globs (any-of; list both shells, `[Bash, mcp__workspace__bash]`, for a claim that must hold at hostloop too). `input: {<field>: <regex>}`: every named top-level input field must match (case-insensitive, unanchored; a missing field is no match; a non-string value is matched as JSON). `input_any: <regex>`: some top-level field matches. `result: {matches?, not_matches?, is_error?}`: predicates on the call's paired `tool_result` (by `toolUseId`, 10 KB of text). `scope`: `main` (default — the main agent, including a `Skill`'s or `Agent(fork)`'s children, the set the string form reads), `subagent` (the parent is a sub-agent dispatch this run recorded, at any depth), or `any` (also calls whose parent is not a recorded dispatch, e.g. a `Skill` invoked inside a sub-agent). `subagent_type` (with `scope: subagent`): regex over the IMMEDIATE parent dispatch's type or description. `count: {min?, max?}` (default `min: 1`). **Fails closed:** an unpaired call never satisfies `result`; a truncated result or input that cannot settle a predicate, or a result.json without `toolCalls`, reports *evidence unavailable*, never a pass or a plain "not called". A red lists the calls it considered and names matches in another scope ("1 matching call in scope subagent — set `scope: any`"). `{tool: X}` alone is exactly `tool_called: X`. A cassette using this form stamps **v13**. |
40
- | `tool_not_called: {tool, input?, input_any?, result?, scope?, subagent_type?}` | **object form** — no call in scope satisfies every predicate (no `count`). Same fields and fail-closed rules as above: a candidate whose result is unpaired or truncated, or whose input was truncated, is *evidence unavailable*, never a pass. Refused at load when EVERY listed tool is one the tier does not serve. ⚠️ **Redaction hazard — this is the dangerous direction.** A committed cassette is redacted, which rewrites the recorded inputs: an `input` regex naming a literal the policy rewrites (a home path, an email) cannot see its target on replay. On replay, any candidate call whose field (or paired result, for `result.matches`) carries a redaction token is *evidence unavailable*, whatever the regex — the evaluator cannot know what the token replaced — so a negative check never passes over rewritten bytes. `record` warns — naming the redacted call and the ways out (narrow `scope`/`tool`, the string form, or live-only) — and refuses the write (the exact guard), and `lint` flags a regex naming a redactable literal (`tool-input-regex-redactable`) — a heuristic that reads `.cowork-redact.json` from the current and scenario directories only, not the cassette's. Match a part of the input redaction leaves alone (the verb and flags, a workspace-relative path), or keep the check on a live gate. Note also that the default `scope: main` does not see a sub-agent's call — use `scope: any` for "nothing anywhere ran X". |
39
+ | `tool_called: {tool: <glob> \| [..], input?, input_any?, result?, scope?, subagent_type?, count?}` | **object form** — asserts what a call CARRIED, where it RAN and what its paired result SAID, which no name glob and no `transcript_*` key can see (`transcript_*` reads top-level prose only, so it passes when the agent merely *says* it ran a command). `tool`: a glob or a list of globs (any-of; list both shells, `[Bash, mcp__workspace__bash]`, for a claim that must hold at hostloop too). `input: {<field>: <regex>}`: every named top-level input field must match (case-insensitive, unanchored; a missing field is no match; a non-string value is matched as JSON). `input_any: <regex>`: some top-level field matches. `result: {matches?, not_matches?, is_error?}`: predicates on the call's paired `tool_result` (by `toolUseId`, 10,240 characters of text — 32,768 for a top-level `Skill` result). `scope`: `main` (default — the main agent, including a `Skill`'s or `Agent(fork)`'s children, the set the string form reads), `subagent` (the parent is a sub-agent dispatch this run recorded, at any depth), or `any` (also calls whose parent is not a recorded dispatch, e.g. a `Skill` invoked inside a sub-agent). `subagent_type` (with `scope: subagent`): regex over the IMMEDIATE parent dispatch's type or description. `count: {min?, max?}` (default `min: 1`). **Fails closed:** an unpaired call never satisfies `result`; a truncated result or input that cannot settle a predicate, or a result.json without `toolCalls`, reports *evidence unavailable*, never a pass or a plain "not called". A red lists the calls it considered and names matches in another scope ("1 matching call in scope subagent — set `scope: any`"). `{tool: X}` alone is exactly `tool_called: X`. A cassette using this form stamps **v13**. |
40
+ | `tool_not_called: {tool: <glob> \| [..], input?, input_any?, result?, scope?, subagent_type?}` | **object form** — no call in scope satisfies every predicate (no `count`). Same fields and fail-closed rules as above: a candidate whose result is unpaired or truncated, or whose input was truncated, is *evidence unavailable*, never a pass. Refused at load when EVERY listed tool is one the tier does not serve. ⚠️ **Redaction hazard — this is the dangerous direction.** A committed cassette is redacted, which rewrites the recorded inputs: an `input` regex naming a literal the policy rewrites (a home path, an email) cannot see its target on replay. On replay, any candidate call whose field (or paired result, for `result.matches`) carries a redaction token is *evidence unavailable*, whatever the regex — the evaluator cannot know what the token replaced — so a negative check never passes over rewritten bytes. `record` warns — naming the redacted call and the ways out (narrow `scope`/`tool`, the string form, or live-only) — and refuses the write (the exact guard), and `lint` flags a regex naming a redactable literal (`tool-input-regex-redactable`) — a heuristic that reads `.cowork-redact.json` from the current and scenario directories only, not the cassette's. Match a part of the input redaction leaves alone (the verb and flags, a workspace-relative path), or keep the check on a live gate. Note also that the default `scope: main` does not see a sub-agent's call — use `scope: any` for "nothing anywhere ran X". |
41
41
  | **legacy tool names** | The agent binary canonicalizes legacy spellings (`Task`→`Agent`, `KillShell`/`KillBash`→`TaskStop`, …) while the spawn tool list still declares the LEGACY one — so the init inventory shows `Task` and every call is emitted as `Agent`. `tool_called`/`tool_not_called`/`subagent_tool_used`/`subagent_tool_absent` match EITHER spelling, globs included. Before this, `tool_called: "Task"` could never pass and `tool_not_called: "Task"` always did. |
42
42
  | **tier vacuity — now REFUSED** | `tool_not_called` **and `subagent_tool_absent`** naming a tool the tier does not serve is rejected at scenario load (`Bash`/`WebFetch`/`NotebookEdit` at `hostloop`; `mcp__workspace__bash` at `container`/`microvm`), because it could never be violated. `scenario.py lint` WARNs on the same set before any run. The table is CLOSED to tools the harness itself removes or registers — `--tools` gates only the built-in set while every tier separately passes `--mcp-config`, so a session-MCP tool name is offered without appearing in any tool list and is never refused. Globs are never refused. `subagent_tool_absent` is covered because it reads the tools sub-agents actually USED, not a per-dispatch declared list. Under-approximates on purpose: `REPL` at hostloop is vacuous too and is not caught, which is the right side to err on for a hard refusal. |
43
43
  | `reference_read: <regex>` | a skill `references/`/`scripts/` file whose path matches this **regex** was ACCESSED — main agent or sub-agents, via `Read`, `Grep`/`Glob` (`path` input), or a `Bash`/`mcp__workspace__bash` command naming the path. Regex is **unanchored + case-insensitive** (the shared helper every regex key uses). **Under-approximates by design:** the path must be rooted in the mounted plugin, so a `cd` into the skill dir then a bare `cat references/x.md`, a heredoc body, and a `$VAR`-built path are invisible. Fails **evidence unavailable** when the run recorded no observable tool stream. Replay-capable (cassettes freeze whole tool inputs) |
@@ -53,9 +53,9 @@ same set live from the schema.
53
53
  | `subagent_dispatch_healthy: {type?, delivered?, path?, path_suffix?, no_vm_paths?}` | **`fidelity: hostloop` only** — composite: selects dispatch(es) via `type` (same matching as `subagent_dispatched`; omit to require every dispatch) and, for EACH selected dispatch, checks it (not just any sub-agent) delivered a paired non-error write (`delivered`, default true — narrowed by `path`/`path_suffix`, same exact-vs-suffix precedence as `subagent_file_write`) and made no `/sessions` VM-path attempt (`no_vm_paths`, default true) — both scoped to that dispatch's OWN `parentToolUseId`, the per-dispatch correlation `subagent_file_write` (which matches ANY sub-agent write) cannot express; a `type` that matches no dispatch FAILS; content-class (`RunResult.fileToolAttempts` + `RunResult.toolResults`); any non-hostloop tier FAILS "cannot verify" |
54
54
  | `subagent_dispatched: <regex>` | a sub-agent whose `dispatchAgentType`, binary-*resolved* `resolvedAgentType`, **or dispatch description** matches |
55
55
  | `subagent_declared_but_unused: <Tool>` | a sub-agent declared the tool but never used **that** tool (even if it used others) |
56
- | `subagent_output_contains: {match?, contains}` | a dispatched sub-agent's own output contains the substring `contains` — `match` (optional regex over `dispatchAgentType`/`resolvedAgentType`/`description`) narrows to specific dispatch(es); omitted, checks whether ANY dispatch's output contains it (existence check, not "all"); a miss against an output that was **truncated at the assert cap** reports evidence-unavailable instead of a proven absence — the substring could lie past the cut. **Covers what the run dispatches** (`Agent`/`Task`, including `Agent(subagent_type:"fork")`), **not a `context: fork` skill** invoked through the `Skill` tool: that skill's own answer is never a dispatch — it comes back as the `Skill` tool result, which agent 2.1.284 builds as `Skill "<name>" completed (forked execution).`, a `Result:` line, then the answer. Assert on it with `tool_result_matches` anchored on that prefix, e.g. `tool_result_matches: '^Skill "[^"]*" completed \(forked execution\)[\s\S]*<pattern>'` (`[\s\S]*` because `.` stops at a newline; the match is case-insensitive, has no multiline flag so `^` is the start of the result, and sees the first 10 KB of each result). This covers a **foreground** fork only: a backgrounded fork's result is the line `Skill "<name>" launched (forked execution, running in the background).`, which carries no answer |
56
+ | `subagent_output_contains: {match?, contains}` | a dispatched sub-agent's own output contains the substring `contains` — `match` (optional regex over `dispatchAgentType`/`resolvedAgentType`/`description`) narrows to specific dispatch(es); omitted, checks whether ANY dispatch's output contains it (existence check, not "all"); a miss against an output that was **truncated at the assert cap** reports evidence-unavailable instead of a proven absence — the substring could lie past the cut. **Covers what the run dispatches** (`Agent`/`Task`, including `Agent(subagent_type:"fork")`), **not a `context: fork` skill** invoked through the `Skill` tool: that skill's own answer is never a dispatch — it comes back as the `Skill` tool result, which agent 2.1.284 builds as `Skill "<name>" completed (forked execution).`, a `Result:` line, then the answer. Assert on it with `tool_result_matches` anchored on that prefix, e.g. `tool_result_matches: '^Skill "[^"]*" completed \(forked execution\)[\s\S]*<pattern>'` (`[\s\S]*` because `.` stops at a newline; the match is case-insensitive, has no multiline flag so `^` is the start of the result, and sees the first 10,240 characters of each result — 32,768 for a top-level `Skill` result). This covers a **foreground** fork only: a backgrounded fork's result is the line `Skill "<name>" launched (forked execution, running in the background).`, which carries no answer |
57
57
  | `dispatch_count_max: <N>` | at most N sub-agents dispatched — your author-chosen budget under Cowork's agent-side fan-out cap (concurrent 20 / per-session 200, inherited by the harness); records only, does not itself enforce — see gotcha 12 in `scenario-schema.md` |
58
- | `skill_triggered: <regex>` | a skill matching the regex (invoked id, e.g. `"plugin:skill"`) was invoked — via the `Skill` tool, or by a prompt starting `/<skill> …` / `/<plugin>:<skill> …` (the agent expands that itself with no `Skill` call; recorded as `slashInvokedSkills`) — evidence-unavailable (not a normal fail) if neither matched and the agent's init tools have no `Skill` tool, or the leading `/name` can't be resolved (ambiguous bare name, no skill inventory) |
58
+ | `skill_triggered: <regex>` | a skill matching the regex (invoked id, e.g. `"plugin:skill"`) was invoked — via the `Skill` tool, or by a prompt starting `/<skill> …` / `/<plugin>:<skill> …` (the agent expands that itself with no `Skill` call; recorded as `slashInvokedSkills`) — evidence-unavailable (not a normal fail) if neither matched and the agent's init tools have no `Skill` tool, or the leading `/name` can't be resolved (ambiguous bare name, no skill inventory). A green here means the AGENT expanded the slash; real Cowork's app resolves a typed slash first and has refused a bare name that differs from its plugin's name (Desktop 2.19675.0, 2026-10-03, 4 runs), a refusal no run can see — pick the skill from the menu or name it like its plugin (`/<plugin>:<skill>` was not measured with a single copy installed) |
59
59
  | `no_skill_triggered: <regex>` | no invoked skill id matched, counting a slash-command invocation as well as a `Skill` call — the negative-control / description-collision catcher; evidence-unavailable (never a vacuous pass) if invocation data is absent, the `Skill` tool is unobservable, or the prompt's leading `/name` can't be resolved |
60
60
  | `skill_available: <regex>` | a staged skill's id matched the regex (offered, not necessarily invoked — see `skill_triggered` for invocation) — content-class: the id list comes from the agent's init `skills` listing, so it replays from the frozen init event (id-only; the `whenToUse` enrichment is live-disk and thus absent on replay, but the id is what's matched); evidence-unavailable only if `RunResult.context.availableSkills` is absent entirely (an older cassette recorded before the available-skills listing was captured) |
61
61
  | `connector_available: <regex>` | an MCP server/connector's name matched the regex (available, not necessarily used) — evidence-unavailable if `RunResult.context.mcpServers` is absent |
@@ -79,11 +79,12 @@ same set live from the schema.
79
79
  | `all_tasks_completed: true` | every task in `RunResult.tasks[]` reached status `"completed"` — **requires ≥1 task** (a zero-task run fails; assert `task_count_min` for presence); evidence-unavailable if tasks telemetry is absent |
80
80
  | `task_count_min: <N>` | at least N tasks were created (`RunResult.tasks.length >= N`) — the presence companion for task assertions |
81
81
  | `task_status: {match, status}` | a task whose `subject` OR `id` matches the regex `match` reached `status` — evidence-unavailable if tasks telemetry is absent; also fails **malformed** when a TaskCreate result was unparseable (corrupt task telemetry), mirroring the guard `all_tasks_completed`/`task_count_min` already had |
82
- | `no_scratchpad_leak: true` | every file presented via `present_files` that was in the scratchpad was successfully promoted to `mnt/outputs` (none left behind) — vacuously passes if nothing was presented (pair with a presence check to require a delivery); content-class: both the `present_files` tool_use and its own tool_result live in the ordinary events stream, so `RunResult.presentedFiles` re-derives on replay at container, where the agent's cwd IS the session root the live lane measures from; at hostloop the live lane measures from the session root while a re-drive has only the recorded cwd (`mnt/outputs` before Desktop 2.7032.0, `/private/var/empty` from it), so the promoted/leaked booleans there are not equivalent — immaterial to this key, which evaluates at container only (meaningfully replay-checkable, same as `skill_triggered`); evidence-unavailable if `presentedFiles` telemetry is absent (an older run predating the feature). **Container-only on the merits**: hostloop serves `present_files` but never promotes (its handler passes a validated path through), so there is no scratch→outputs copy that could leak. On microvm/protocol a scratchpad-delivered file is neither promoted to `mnt/outputs` nor detected there (a skill that delivers via write-to-cwd→`present_files` will false-red `user_visible_artifact` on those tiers; use `container`, or write directly to `outputs/`). Hostloop is not one of them: at hostloop a delivered file under the outputs dir is visible there immediately, so `user_visible_artifact` passes (before Desktop 2.7032.0 a write-to-cwd landed there; from it the agent runs at `/var/empty` and a relative write is refused). **The tool name is lane-specific:** `present_files` is the desktop-local lane's tool (the one this harness emulates); remote Cowork delivers via the agent-native `SendUserFile` instead, so a skill should describe the delivery outcome rather than naming either tool — this key asserts the harness-side delivery record either way. **Only `true` is valid** |
82
+ | `no_scratchpad_leak: true` | every file presented via `present_files` that was in the scratchpad was successfully promoted to `mnt/outputs` (none left behind) — vacuously passes if nothing was presented (pair with a presence check to require a delivery); content-class: both the `present_files` tool_use and its own tool_result live in the ordinary events stream, so `RunResult.presentedFiles` re-derives on replay at container, where the agent's cwd IS the session root the live lane measures from; at hostloop the live lane measures from the session root while a re-drive has only the recorded cwd (`mnt/outputs` before Desktop 2.7032.0, an empty macOS system dir from it), so the promoted/leaked booleans there are not equivalent — immaterial to this key, which evaluates at container only (meaningfully replay-checkable, same as `skill_triggered`); evidence-unavailable if `presentedFiles` telemetry is absent (an older run predating the feature). **Container-only on the merits**: hostloop serves `present_files` but never promotes (its handler passes a validated path through), so there is no scratch→outputs copy that could leak. On microvm/protocol a scratchpad-delivered file is neither promoted to `mnt/outputs` nor detected there (a skill that delivers via write-to-cwd→`present_files` will false-red `user_visible_artifact` on those tiers; use `container`, or write directly to `outputs/`). Hostloop is not one of them: at hostloop a delivered file under the outputs dir is visible there immediately, so `user_visible_artifact` passes (before Desktop 2.7032.0 a write-to-cwd landed there; from it the agent runs at `/var/empty` and a relative write is refused). **The tool name is lane-specific:** `present_files` is the desktop-local lane's tool (the one this harness emulates); remote Cowork delivers via the agent-native `SendUserFile` instead, so a skill should describe the delivery outcome rather than naming either tool — this key asserts the harness-side delivery record either way. **Only `true` is valid** |
83
83
  | `present_files_called: true` | at least one file was actually delivered via the `present_files` tool (`RunResult.presentFilesCalls > 0` — the count of invocations that carried a well-formed `file_path`). Presence is read at the **invocation**, deliberately not off the classified `presentedFiles` list: that list drops any path it cannot resolve, and a host-path redaction policy makes that certain at `hostloop`, where a real host path redacts to `[REDACTED:…]/mnt/outputs/f`. So this key is safe to assert in a scenario you intend to `record` with redaction active — `record` refuses a cassette whose verdict redaction changed, and an invocation count cannot be changed by it. A run that called the tool but whose every call carried an unusable path reports **cannot verify**, never a claim of non-delivery. The presence companion to `no_scratchpad_leak` (which passes vacuously when nothing was presented). Pair them to require a delivery **and** require it not to leak. Content-class (re-derives identically on replay). **`fidelity: container` or `hostloop`** — the harness serves `present_files` at both (hostloop via a handler mirroring production's own host-loop branch: validate the path, pass it through, no promotion). `protocol` and `microvm` report cannot-verify. See the `no_scratchpad_leak` row, which stays container-only for a different reason; and see the lane note there — the tool name differs on remote Cowork. **Only `true` is valid** |
84
84
  | `question_asked: <regex>` | the agent asked an AskUserQuestion whose **question text** matches (`question`, falling back to `header`). Text only — for the options it offered, use `question_options` ⚠️ **Model-composed text, reworded run to run** — pin a producer-authored constant, not model prose. |
85
- | `question_options: {when_question?, equals?, contains?, order?}` | the option SET and ORDER a gate offered, **by label** — descriptions are not compared (see `question_context`). `when_question` selects the sub-question by the same label `question_asked` matches (omit only if exactly one fired; ambiguity FAILS). Exactly one of `equals` (complete set) / `contains` (subset); `order: exact` is the default because a re-ordered list is the defect this exists for. Captured at ask time, so a gate that was shown then denied/stalled still counts; unreadable evidence fails evidence-unavailable ⚠️ **Model-composed text, reworded run to run** — pin a producer-authored constant, not model prose. |
85
+ | `question_options: {when_question?, equals?: [..], contains?: [..], order?}` | the option SET and ORDER a gate offered, **by label** — descriptions are not compared (see `question_context`). `when_question` selects the sub-question by the same label `question_asked` matches (omit only if exactly one fired; ambiguity FAILS). Exactly one of `equals` (complete set) / `contains` (subset); `order: exact` is the default because a re-ordered list is the defect this exists for. Captured at ask time, so a gate that was shown then denied/stalled still counts; unreadable evidence fails evidence-unavailable ⚠️ **Model-composed text, reworded run to run** — pin a producer-authored constant, not model prose. |
86
86
  | `question_context: {when_question?, matches}` | a regex over **everything the gate showed the user** — question label + every option label + every option **description**. Use it when the sentence you need to prove reached the founder may land in any of those fields: `question_asked` sees only the question text, `question_options` compares only labels, so a phrase delivered in an option's `description` is invisible to both. `when_question` narrows; omitting it searches every gate (NOT ambiguous here — this key asks whether the text was shown, not which gate offered which set). Ask-time payload only, never a `tool_result` (a producer that also writes the phrase to its own gate-state file would otherwise false-green it). Zero gates FAILS ⚠️ **Model-composed text, reworded run to run** — pin a producer-authored constant, not model prose. |
87
+ | `question_option_count: {matches, exactly? \| min? \| max?, when_question?, case_sensitive?}` | counts option **labels** matching a regex on **every** selected sub-question (e.g. one `No changes — ` option per gate); zero asked FAILS. `case_sensitive` covers the whole pattern; single-quote regexes. Details: [docs/scenario.md](https://github.com/yaniv-golan/cowork-harness/blob/main/docs/scenario.md) ⚠️ **Model-composed text, reworded run to run** — pin a producer-authored constant. |
87
88
  | `questions_count_max: <N>` | at most N **sub-questions** asked — a bundled `AskUserQuestion` with K sub-questions counts as K, not 1; `trace --view questions`'s footer total uses the same definition |
88
89
  | `gate_answers_delivered: true` | every answered gate's answer reached the model (observed `tool_result`; unobserved = fail); **zero gates fired passes vacuously** — pair with `gate_answer_count_min: >= 1` to also require a gate, or drop this key and declare `questions_count_max: 0` in a scenario that expects none |
89
90
  | `gate_answers_delivered: false` | asserts at least one answered gate's answer was **confirmed not delivered** (an observed delivery failure); an unobserved/null delivery does **not** satisfy this — for negative-path delivery tests. Requires a gate, so **mutually exclusive** with `questions_count_max: 0` (refused by `run`/`skill`/`record`) |
@@ -92,22 +93,24 @@ same set live from the schema.
92
93
  | `no_hook_blocked: true` | no tool was hook-blocked during the run (distinguishes a real tool crash from an intentional hook block) — evidence-unavailable if hook telemetry is absent. Replay: needs a `controlOut` cassette. **Only `true` is valid** |
93
94
  | `hook_event_fired: <HookEvent>` | a **command hook** for this event (a plugin's `hooks/hooks.json` or manifest hook — `Stop`, `SessionStart`, `PostToolUse`, …) ran: a `hook_response` system frame with that `hook_event` was recorded (`RunResult.contextEvents`). Any outcome counts. The harness passes `--include-hook-events` whenever a staged plugin declares hooks — that is what puts events other than SessionStart/Setup on the stream — so a recording made without it reports "never fired". Content-class, grades on replay. Recorded end-to-end for `Stop` ([stop-hook-probe.scenario.yaml](https://github.com/yaniv-golan/cowork-harness/blob/main/examples/probes/stop-hook-probe.scenario.yaml)); the other names match the same frame but have not each been recorded |
94
95
  | `hook_event_blocked: <HookEvent>` | that command hook **blocked** at least once — a `hook_response` frame for the event carried `exit_code: 2`. Fails naming the exit codes seen when it fired without blocking (a frame with no `exit_code` is reported as such, never counted); fails "never fired" otherwise; cannot-verify when the run has no context events. Content-class |
96
+ | `hook_output_contains: {event, stream?, text \| matches}` / `hook_output_not_contains: {…}` | a command hook **printed** (or never printed) a text on `stdout` / `stderr` / either (`stream`, default `any`) — some (or no) `hook_response` frame for `event` carries `text` (literal, case-sensitive) or `matches` (regex, case-insensitive, no multiline flag — `^`/`$` anchor the whole stream). For a hook that fails open on stderr while exiting 0. No frame for the event FAILS both. On a redacted stream, a literal miss and any `matches` result are evidence-unavailable for either key, while a literal hit outside a token counts; a miss is also evidence-unavailable on a missing field, a hook that started without a response, or output the agent truncated. Frames carry no plugin id (a second plugin, or at `protocol` without a sealed config dir a host-installed one, can answer). Content-class. Recorded end-to-end for `Stop` |
95
97
  | `vm_path_denied: true` | **`fidelity: hostloop` only** — at least one recorded path denial (`RunResult.pathDenials`, any source) targeted a `/sessions` VM path — evidence-unavailable if path-denial telemetry is absent. Replay: needs a `controlOut` cassette. Any other tier FAILS "cannot verify". **Only `true` is valid** |
96
98
  | `path_denied: {tool?, path_matches?, source?, agent_scope?}` | **`fidelity: hostloop` only** — a path denial matching ALL given matchers (`tool` glob, `path_matches` regex, `source` ∈ pretooluse/can_use_tool/permission_denied, `agent_scope` ∈ main/subagent/any) was recorded. Replay: needs a `controlOut` cassette. Any other tier FAILS "cannot verify" |
97
99
  | `no_path_denied: true` | **`fidelity: hostloop` only** — NO path denial was recorded at all (the channel is already path-scoped, unlike `no_hook_blocked`'s indiscriminate reject). Replay: needs a `controlOut` cassette. Any other tier FAILS "cannot verify". **Only `true` is valid** |
98
100
  | `allow_permissive_auto_allow: true` | verdict modifier — suppresses the default-fail when the run recorded a cowork-parity permissive auto-allow; for tests that deliberately assert Cowork's permissive behavior |
99
101
  | `allow_missing_capability: true` | verdict modifier — suppresses the default-fail when the (partial "core") agent image omits a capability the skill used but real Cowork ships (OCR/LibreOffice/markitdown/opencv/PDF-tables). Assert only when the skill's fallback is genuinely equivalent; otherwise rebuild full parity (`--build-arg COWORK_FULL_PARITY=1`). Also opts out of the `requires_capabilities` declared-need check. Live tiers only |
100
- | `allow_l0_host_config_contamination: true` | verdict modifier — accept a contaminated L0 environment: suppresses the default-fail when `protocol` runs against your REAL config dir, where your installed plugins/skills/auto-memory/MCP servers are visible and may answer instead of the thing under test. Live tiers only |
101
- | `allow_stall: true` | verdict modifier — suppresses the `stalled` default-fail when a run ends on a question having done no productive tool work after its last gate (the agent asked for input and stopped — incl. re-asking in plain text after answering an `AskUserQuestion`); assert only when ending on a question is intended, else script the answer (`answer:` / `--answer` / a decider). On an open-ended `skill` / `probe-dispatch` run (no `assert:` block) pass `--allow-stall` instead. A helpful closing offer ("want me to run this through a structured pass?") also fails `stalled` — read the final message before believing it |
102
+ | `allow_l0_host_config_contamination: true` | verdict modifier — accept a contaminated L0 environment: suppresses the default-fail when `protocol` runs against your REAL config dir, where your installed plugins/skills/MCP servers are visible and may answer instead of the thing under test. Live tiers only |
103
+ | `allow_stall: true` | verdict modifier — suppresses the `stalled` default-fail when a run ends on a question or (after an `AskUserQuestion` gate) a closing request for input having done no productive tool work after its last gate (the agent asked for input and stopped — incl. re-asking in plain text after answering an `AskUserQuestion`); assert only when ending on a question is intended, else script the answer (`answer:` / `--answer` / a decider). On an open-ended `skill` / `probe-dispatch` run (no `assert:` block) pass `--allow-stall` instead. A helpful closing offer ("want me to run this through a structured pass?") also fails `stalled` — read the final message before believing it |
102
104
  | `allow_undelivered_deliverables: true` | verdict modifier — suppresses the `undelivered_deliverables` WARN. Working in the scratchpad is Cowork's designed pattern, so a skill that legitimately leaves intermediates, caches or downloaded inputs behind can say so instead of carrying permanent noise. The signal is warn-only and never fails a run on its own; reach for this when the scratch activity is intentional, not to silence a real delivery gap |
103
- | `allow_outputs_delete: true` | verdict modifier — accepts a detected outputs delete instead of failing the run, for a skill whose deletion is intended. Omitting `no_delete_in_outputs` does **not** permit deletes (a detected delete fails via the `outputs_delete` signal precisely because the key was not authored), so this is the way to accept one. **Mutually exclusive** with `no_delete_in_outputs`. Silences `outputs_delete`, `outputs_delete_unconfirmed` and `outputs_diff_unavailable`. Waives the harness's post-hoc detection; it does not model Cowork's `allow_cowork_file_delete` approval handshake |
105
+ | `allow_outputs_delete: true` | verdict modifier — accepts a detected outputs delete instead of failing the run, for a skill whose deletion is intended. Has an effect only on a baseline recording outputs `rw` (Desktop before 2.16120.0), where omitting `no_delete_in_outputs` does **not** permit deletes; on `rwd` (2.16120.0+) an outputs delete does not fail by default, so it is an accepted no-op (no warning), unless `no_delete_in_mounts` arms the outputs check, where it waives that check's signal as on `rw`. **Mutually exclusive** with `no_delete_in_outputs`. Silences `outputs_delete`, `outputs_delete_unconfirmed` and `outputs_diff_unavailable`. Waives the harness's post-hoc detection; it does not model Cowork's `allow_cowork_file_delete` approval handshake |
104
106
  | `allow_delete_in: [<mount>…]` | verdict modifier — accepts detected deletes in the named mounts (the per-mount analogue of `allow_outputs_delete`, mirroring production's `fileDeleteApprovedMounts`). Waives the verdict only; detection still runs and the hits stay in `result.json`. Listing `outputs` conflicts with `no_delete_in_outputs` |
105
- | `transcript_no_host_path: true` | no host path (`/Users/`, `/opt/cowork/`, `/home/`, `/root/`, and the macOS `/private/var/`, `/private/tmp/`, `/var/folders/`, `/Volumes/` roots — also inside a `file://` or `computer://` link) leaked into model-visible text (a path that came verbatim from the scenario's input files or prompt is exempt; `scan.hostPathsFromInputs` counts such paths) — **incompatible with `hostloop` AND `protocol`**: hostloop's native file tools legitimately expose real host paths, and protocol (L0) runs the agent's file tools on the real host cwd with no sealed filesystem, so this fails BY DESIGN on both (the harness warns at run start if asserted anyway); use `container`/`microvm` for this check |
107
+ | `transcript_no_host_path: true` | no host path (a path under a host home or system root: `Users`, `home`, `root`, the Cowork install dir `opt/cowork`, and the macOS `private/var`, `private/tmp`, `var/folders` and `Volumes` roots, each written here without its leading slash — also inside a `file://` or `computer://` link) leaked into model-visible text (a path that came verbatim from the scenario's input files, prompt, or declared plugins' or local skills' files is exempt; `scan.hostPathsFromInputs` counts such paths) — **incompatible with `hostloop` AND `protocol`**: hostloop's native file tools legitimately expose real host paths, and protocol (L0) runs the agent's file tools on the real host cwd with no sealed filesystem, so this fails BY DESIGN on both (the harness warns at run start if asserted anyway); use `container`/`microvm` for this check |
106
108
  | `egress_denied: <host>` | the host was blocked by the egress proxy |
107
109
  | `egress_allowed: <host>` | the host was allowed through |
108
110
  | `no_mcp_error: true` | no MCP round-trip failed (`RunResult.mcpErrors` is empty — no unhandled server, no handler throw) — live-only: MCP round-trips are harness-computed, not in the SDK stdout stream, so evidence-unavailable on replay (never a vacuous pass). **Only `true` is valid** |
109
111
  | `max_peak_rss_bytes: <N>` | peak sampled RSS of the agent sandbox ≤ N bytes (`RunResult.resources.peakRssBytes`) — live-only: replay never spawns a sandbox to sample, so evidence-unavailable on replay/protocol (never a vacuous pass); also evidence-unavailable when sampling captured no RSS value |
110
- | `semantic_matches: {rubric: [...], min_pass?, judge_model?, include_subagent_text?, evidence_files?}` | a pinned LLM judge grades each fixed `rubric` claim against the run's answer — the **union of the agent's final result text (`RunResult.finalMessage`), the transcript, and the final on-disk content of any files the agent authored during the run** — so a claim about content the skill led the agent to *write to a file* grades as reliably as one about inlined prose. **"The transcript" is narrower than it reads: top-level `assistant_text` ONLY.** It excludes every `tool_use`/`tool_result` and **all sub-agent-originated text** (including fork-scoped `Skill`/`Agent(fork)` dispatches — the harness attributes their *tool* calls to the main agent, but not their text). **A `context: fork` skill's own answer is not graded either**, even with `include_subagent_text: true`: it is not a dispatch (so it has no `subagents[]` entry to fold in), and it reaches the main agent as the `Skill` tool result, which the judged document excludes. The judge sees it only if the main agent restates it; to check the fork's answer directly, use `tool_result_matches` (see `subagent_output_contains`). ⚠️ **Consequence: a rubric claim about whether a tool was called can NEVER grade true** — the evidence is not in the judged document. Such a claim looks reasonable and silently caps your pass rate; assert tool use with `tool_called` / `present_files_called` / `subagent_dispatched` / `hook_blocked` instead. Sub-agent text is captured in `RunResult.subagents[].reasoning` and reaches the judge only via opt-in `include_subagent_text: true` (`kind:"text"` turns only — sub-agent *thinking* arrives empty with `redacted:true`, so it would pad the document with blanks) (authored-file evidence is captured on every live sandbox tier including **microvm** — its session tree is snapshotted from the VM into the run dir). When the authored-file evidence backing the judged document is **incomplete** — a file dropped at the capture-size cap, unreadable at read-back, or (on `--resume`) the scratchpad walk skipped — the assert fails evidence-unavailable rather than trusting a judge grade over a partial document; this is separate from the malformed-grade `judgeInvalid` path below. The assert passes iff ≥ `min_pass` claims pass (default: all — avoid for a gating scenario). Results align by claim index and are recorded per-claim in `RunResult.assertions[].semanticClaims` (`[{index, claim, pass, rationale?}]`, so a consumer can diff the per-claim profile across runs; `rationale` is the judge's one-sentence reason, printed under each failed claim in the failure footer: untrusted model text that can quote the judged document, whose content never affects `pass`, absent when the judge gave none (a reply whose shape is broken, such as unparseable JSON, a malformed `{"results": …}` group beside a valid grade, or a partial restatement that contradicts it, is retried once and then marked `judgeInvalid`), and comparable only between runs that share `judgePromptHash`); a rep whose grade can't be parsed (after one retry) is marked `RunResult.assertions[].judgeInvalid` and **never silently dropped** — it is excluded from the pass denominator, and the guard against a misleading score from that exclusion is the gate's minimum-valid-rep floor (`MIN_VALID` ≥ 4) plus this visibility, not a claim that denominator-shrinking inflation is impossible. Within a rep, a grade that's still unparseable after the retry **fails that assert outright** (evidence-unavailable, not a vacuous pass) — a persistently-flaky judge reds the run rather than silently passing. `judge_model` pins the grader (default when neither it nor `COWORK_HARNESS_JUDGE_MODEL` is set: `claude-opus-4-8`; a dated id keeps a before/after comparison reproducible — or let `eval` compare two skill versions per claim, with the judge pinned). Each graded assert records the judge's provenance: `RunResult.assertions[].judgeModel` (the resolved model), `judgeCostUsd` (judge spend over both attempts, reported beside `cost.usd` and never inside it; absent when unpriced) and `judgePromptHash` (the grading-prompt template — compare only runs that share it). Live-only: the judge is a live model call, so evidence-unavailable / skipped-loud on replay (never a vacuous pass) **`evidence_files: [globs]` scopes which authored files are graded** — reach for it the moment a run authors more than a couple of files. The capture budget (64 KiB total by default) is spent prefix-major then alphabetically, so a pipeline that stages intermediates (`outputs/_work/*.json`) exhausts it before reaching its own deliverable and the verdict is refused evidence-unavailable over files no rubric mentions. Scoping also makes the capture spend the budget on the named files FIRST and exempts them from the per-file cap. Paths are `<root>/<rel>` (`outputs/report.md`, never a bare `report.md`; session-root writes are `scratchpad/<rel>`); globs are `*`/`?`/`**`, not regex. A glob matching nothing FAILS and the message lists every authored path — read it rather than guessing. Still too big? Raise `$COWORK_HARNESS_AUTHORED_TOTAL_BYTES`. The typed reason is on `RunResult.assertions[].semanticEvidence` — check `.reason` instead of parsing the message |
112
+ | `semantic_matches: {rubric: [...], min_pass?, judge_model?, include_subagent_text?, include_fork_results?, evidence_files?: [..]}` | a pinned LLM judge grades each fixed `rubric` claim against the run's answer: the final result text, the transcript and the files the agent authored. The transcript is top-level `assistant_text` only — it excludes every `tool_use`/`tool_result` and sub-agent text unless opted in. Full semantics (evidence scope, fork results, refusals, judge provenance, cost): [semantic-judging.md](semantic-judging.md). |
113
+ | `semantic_pairwise: {refs?: [..], rubric?: [..], pass_if?, order?, judge_model?, include_subagent_text?, include_fork_results?, evidence_files?: [..]}` | a pinned LLM judge compares this run's judged document with a **frozen reference** — the same document an earlier run (usually the baseline) produced, written once by `ref freeze` and never regenerated — and answers win, tie, loss or `both_bad`. **The judged document is the one `semantic_matches` builds** (final message + transcript + authored files, same `evidence_files` / `include_subagent_text` / `include_fork_results` options), so the transcript **excludes every `tool_use`/`tool_result`** and a criterion about whether a tool was called cannot be judged from it. `refs` are reference stores relative to the scenario file; the run is judged against every one (under `hillclimb run`, against the flow's own references instead — only the baseline's gates `pass_if`, a later variant's is a metric recorded with `gate: false`), and `pass_if` (default `not_worse`) must hold against each gating one: `win`, `not_worse` (win or tie), or `any` (graded at all — a metric only). `both_bad` fails `win` and `not_worse`. Which output the judge sees first is a seeded coin per run, assert and reference (`order: both` judges both orders: a win in one and a loss in the other is position bias and scores as a tie; any other disagreement keeps the worse outcome; each order's own outcome is recorded as `pairwise[].orders.candidate_first` / `.ref_first`, absent for a single-order grade, and a judge that favours whichever output it sees first shows across runs as `candidate_first` winning more often than `ref_first`). The judge never sees the words "reference" or "baseline". A reference must have been frozen with the same evidence options and for the same prompt; a missing, damaged, differently-scoped or other-prompt one **refuses the run before it spends anything**, as does a reference store inside any mounted source (the agent could read its own answer key). Unavailable evidence refuses with `semantic_matches`' typed reasons. Per-reference outcomes land in `assertions[].pairwise`. **LIVE-ONLY** — skipped on replay. |
111
114
  | `artifact_json: {artifact, path, …}` | assert a JSON artifact's contents — `equals`/`gt`/`in`/`exists`/`absent`/`is_null` over a dotted `path` (`in` = membership in a list, for a stochastic/LLM value; `absent` ≠ `is_null`; an unresolved intermediate fails loud) |
112
115
  | `computer_links_resolve: true` | every `computer://` link in the model-visible transcript resolves to an artifact that exists in the run's collected outputs/mounts — a dangling link fails, naming which target was checked (a live host path, the collected work tree, or the replay manifest). **Requires ≥1 link** (zero links fails — use `computer_links_resolve_if_present` for the presence-free variant). **Only `true` is valid** (`false` is rejected by the schema) **Sees top-level `assistant_text` only — it excludes every `tool_use`/`tool_result`**, so a `computer://` link that appeared only inside a tool call or its result is invisible to it. |
113
116
  | `computer_links_resolve_if_present: true` | like `computer_links_resolve` but passes vacuously when the transcript has zero `computer://` links — the presence-free variant. **Only `true` is valid** **Sees top-level `assistant_text` only — it excludes every `tool_use`/`tool_result`**, so a `computer://` link that appeared only inside a tool call or its result is invisible to it. |
@@ -126,7 +129,7 @@ dotted path.
126
129
 
127
130
  **VerdictSignals in `result.verdict.signals`:** `computeVerdict` pushes signals into `result.verdict.signals`; eleven
128
131
  are **fail**-severity (they flip the run's pass/exit code even though `result.result` itself stays
129
- `"success"`) and eleven are **warn**-severity (informational, never flip pass/fail). All twenty-two signal
132
+ `"success"`) and twelve are **warn**-severity (informational, never flip pass/fail). All twenty-three signal
130
133
  codes (`VerdictSignal["code"]` in `src/run/verdict.ts`):
131
134
 
132
135
  | Code | Severity | Meaning |
@@ -136,24 +139,25 @@ codes (`VerdictSignal["code"]` in `src/run/verdict.ts`):
136
139
  | `usage_limit` | fail | Usage/quota limit hit (not a skill failure) — retry after the limit resets. Emitted when `RunResult.resultErrorKind === "usage_limit"` |
137
140
  | `transport_error` | fail | The connection dropped mid/after-run |
138
141
  | `permissive_auto_allow` | fail | A cowork-parity auto-allow real Cowork would block (opt out: `allow_permissive_auto_allow`) |
139
- | `outputs_delete` | fail | An unauthorized delete touched `mnt/outputs`, confirmed: the per-turn filesystem diff proves it, a delete in command/call position has an `outputs/` path as its own operand, or the diff could not verify the turn. Authoring `no_delete_in_outputs` moves it into that assertion; `allow_outputs_delete` waives it |
140
- | `outputs_delete_unconfirmed` | warn | A delete-shaped command near `mnt/outputs` nothing confirms: no output present at turn start was deleted and no flagged delete has an outputs path as its own operand (e.g. a Python variable named `rm` next to an outputs path, quoted prose, a `sed`/`grep` pattern). A *statement* is a fragment split on newline/`;`/`&&`/`\|\|`, quote-blind, so some real deletes also land here — a loop body whose operand is the loop variable (`for f in …; do rm "$f"; done`), a `cd` then a relative path, chained variables (`A=…; B=$A/x; rm "$B"`), a Python path held in a variable set on another line (`p = …` then `os.remove(p)`, or `for p in …:` then `p.unlink()`), wrappers with flag combinations the classifier does not model (`sudo -Hu user rm`, `git -C dir rm`), and calls outside the modelled set such as Node's `fs.promises.rm(…)` — read the command. Raised even when `no_delete_in_outputs` is authored (that assertion passes on it). Kept false positives that still fail `outputs_delete`: quoted text in which a delete command with an outputs operand follows a shell separator, subshell or keyword — the classifier does not track quotes (`echo 'note; rm mnt/outputs/x'`, `echo "a & rm …/outputs/x"`), and a heredoc that *writes* a script rather than running it (`cat <<EOF > clean.sh` with an `rm …/outputs/x` line). Waive: `allow_outputs_delete` |
141
- | `outputs_diff_unavailable` | warn | The per-turn filesystem diff of `outputs/` could not verify this turn and the text scan flagged nothing — a delete made without a bash command would have gone undetected. With a text hit the turn fails `outputs_delete` instead |
142
+ | `outputs_delete` | fail | An unauthorized delete touched `mnt/outputs` on a baseline that records outputs `rw` (not raised on `rwd`, Desktop 2.16120.0+, unless `no_delete_in_outputs` or `no_delete_in_mounts` (outputs not waived) is authored — with `no_delete_in_outputs` that assertion fails instead; with `no_delete_in_mounts` this signal itself fires, and its message names that key), confirmed: the per-turn filesystem diff proves it, a delete in command/call position has an `outputs/` path as its own operand, or the diff could not verify the turn. Authoring `no_delete_in_outputs` moves it into that assertion; `allow_outputs_delete` waives it |
143
+ | `outputs_delete_unconfirmed` | warn | A delete-shaped command near `mnt/outputs` nothing confirms: no output present at turn start was deleted and no flagged delete has an outputs path as its own operand (e.g. a Python variable named `rm` next to an outputs path, quoted prose, a `sed`/`grep` pattern). A *statement* is a fragment split on newline/`;`/`&&`/`\|\|`, quote-blind, so some real deletes also land here — a loop body whose operand is the loop variable (`for f in …; do rm "$f"; done`), a `cd` then a relative path, chained variables (`A=…; B=$A/x; rm "$B"`), a Python path held in a variable set on another line (`p = …` then `os.remove(p)`, or `for p in …:` then `p.unlink()`), wrappers with flag combinations the classifier does not model (`sudo -Hu user rm`, `git -C dir rm`), and calls outside the modelled set such as Node's `fs.promises.rm(…)` — read the command. Raised even when `no_delete_in_outputs` is authored (that assertion passes on it); not raised on an `rwd` baseline (Desktop 2.16120.0+) unless that key, or `no_delete_in_mounts` with outputs not waived, is authored. Kept false positives that still fail `outputs_delete`: quoted text in which a delete command with an outputs operand follows a shell separator, subshell or keyword — the classifier does not track quotes (`echo 'note; rm mnt/outputs/x'`, `echo "a & rm …/outputs/x"`), and a heredoc that *writes* a script rather than running it (`cat <<EOF > clean.sh` with an `rm …/outputs/x` line). Waive: `allow_outputs_delete` |
144
+ | `outputs_diff_unavailable` | warn | The per-turn filesystem diff of `outputs/` could not verify this turn and the text scan flagged nothing — a delete made without a bash command would have gone undetected. With a text hit the turn fails `outputs_delete` instead. Like `outputs_delete_unconfirmed`, not raised on an `rwd` baseline unless `no_delete_in_outputs` or `no_delete_in_mounts` (outputs not waived) is authored |
142
145
  | `mount_delete` | warn | A delete touched a delete-denied mount other than `outputs` (a `rw` connected folder). Production denies `unlink`/`rmdir` there until per-mount approval, so the run diverged. Warn, not fail: the harness detects post-hoc what production enforces. Author `no_delete_in_mounts` to hard-fail, or `allow_delete_in` to waive |
143
- | `host_path_leak` | fail | A host path leaked into model-visible text (opt out: author `transcript_no_host_path`). A path copied verbatim from the scenario's own uploads, connected folders or prompt is not a leak |
144
- | `l0_host_config_contamination` | fail | `protocol` ran against the operator's REAL config dir, so host-installed plugins/skills/memory/MCP may have answered instead of the thing under test — the run did not necessarily measure it (opt out: `allow_l0_host_config_contamination`). Name predates the meaning: delivery at L0 works, contamination is what this reports |
146
+ | `host_path_leak` | fail | A host path leaked into model-visible text (opt out: author `transcript_no_host_path`). A path copied verbatim from the scenario's own uploads, connected folders, prompt, or declared plugins' or local skills' files is not a leak |
147
+ | `l0_host_config_contamination` | fail | `protocol` ran against the operator's REAL config dir, so host-installed plugins/skills/MCP servers may have answered instead of the thing under test — the run did not necessarily measure it (opt out: `allow_l0_host_config_contamination`). Name predates the meaning: delivery at L0 works, contamination is what this reports |
145
148
  | `missing_capability` | fail | A `requires_capabilities` need was unmet, or the skill used a capability the image omits (opt out: `allow_missing_capability`, or `skill --allow-missing-capability` on an open-ended run) |
146
149
  | `infra_error` | fail | A supervising process died mid-run (VM/egress sidecar) — the run's evidence is contaminated, not author-suppressible |
147
- | `stalled` | fail | The run ended on an unanswered question, or on a trailing-`?` final turn with no tool work after the last gate (opt out: `allow_stall`) |
150
+ | `stalled` | fail | The run ended asking for input with no tool work after the last gate. "Asking" = the final turn's closing sentence ends in `?`; or, once an `AskUserQuestion` gate has fired, it ends in `?` after trailing bold/quotes/`)`/emoji, is a `?` followed only by a `For example: …`/parenthetical aside, or is a request that says the input comes back to the agent (`Please share/provide/upload… so I can…`/`…here`, `Let me know which… you'd like me to use`, `Once you share… I'll…`, `I need X to proceed`, `Once I have the file, I'll…` after a sentence asking for it). Polite closers and hand-offs (`Let me know if…`, `Feel free…`, `with your`/`to your`, `before sending`, `If you…`) and a closing code block or blockquote never count. Hand-offs to a named third party (`with the team`, `to the founders`, `with the CFO`) never count either; `here` is a cue only after share/paste/upload/drop/reply/type/send or as the last word. English-only. Opt out with `allow_stall` (on `run`, `replay` and `eval`); without it, under `eval` a stalled rep is `errored_agent` and fails every row |
148
151
  | `non_deterministic` | warn | The run was LLM/external/human-decided — not reproducible |
149
152
  | `model_fallback` | warn | The agent fell back off the requested model mid-run (SDK `model_fallback` event); a `model_not_found`/`model_blocked` trigger repeats every run until the pin changes |
150
153
  | `prompt_asset_missing` | warn | The run proceeded with a missing prompt asset (`COWORK_HARNESS_ALLOW_MISSING_PROMPT=1`); fidelity is degraded |
151
154
  | `scan_unavailable` | warn | Post-run scan evidence unavailable (`RunResult.scan` undefined) — the host-path guard and the outputs-delete text scan did not run this run; the outputs filesystem diff still ran |
152
- | `ended_with_question` | warn | Live-lane heuristic: the final answer contains a question and the run wrote no deliverable to `outputs/` — the lenient sibling of `stalled` (covers a mid-message `?`, or tool work after the last gate that still ended asking). Opt out: `allow_stall` |
155
+ | `ended_with_question` | warn | Live-lane heuristic: the final answer contains a question (or closes on a request for input, the same test as `stalled`) and the run wrote no deliverable to `outputs/` — the lenient sibling of `stalled` (covers a mid-message `?`, or tool work after the last gate that still ended asking). Opt out: `allow_stall` |
153
156
  | `undelivered_deliverables` | warn | The skill produced file(s) OUTSIDE every user-visible root and never delivered them, so they stay invisible to the user. Fires on every run without opting in, because the scenarios that most need it are the ones whose author never considered delivery. Silent when the evidence cannot answer the question (no workspace walk, a tier that runs no scratchpad walk, absent delivery telemetry, a resumed turn, or a lane where delivery is unobservable — see `delivery_unobservable`) — never a vacuous clean. **`lane: local` only**: on remote, delivery cannot be measured at all, so that lane reports `delivery_unobservable` instead of guessing. Opt out: `allow_undelivered_deliverables` |
154
157
  | `delivery_unobservable` | warn | `lane: remote` only — the run produced file(s), but whether any reached the user CANNOT be verified: nothing is delivered by location on that lane and the harness models no remote delivery tool (production uses the agent-native `SendUserFile`). The honest counterpart to `undelivered_deliverables`, which would otherwise fire on every remote run that writes anything — a signal that always fires carries no information. Mutually exclusive with it; quiet when the run produced nothing to deliver. A harness coverage gap, not a skill defect. Opt out: `allow_undelivered_deliverables` |
158
+ | `partly_scripted_gate` | warn | A question batch (one `AskUserQuestion` with several sub-questions) that the scenario's `answers:` matched only PART of. Answers are delivered atomically, so the whole batch went to the `on_unanswered` fallback and the matched answers were NOT delivered — the fallback may contradict them. The message names the matched and unmatched sub-questions and who answered instead; `result.partlyScriptedGates` carries the full lists. Fires only on a partial match within one batch (a single-question gate no rule matched is the ordinary unanswered case). Re-derived on `replay` (from the cassette's frozen answers) and `verify-run` (from the current scenario). Fix: script every sub-question of the batch |
155
159
  | `exec_infra_error` | warn | Host-loop: one or more container `exec` calls failed for infrastructure reasons, so those tool calls returned an error to the agent. Warns rather than fails because the run's other evidence is intact — unlike `infra_error`, where a dead supervisor contaminates everything. Caveat: if *every* exec failed, the agent ran nothing and this still only warns — check `result.infraErrors` |
156
160
 
157
161
  A **fail**-severity signal does not change `result.result` (still `"success"`), but it DOES fail the
158
162
  overall run verdict and exit code — `assert result: success` alone won't catch it; check
159
- `result.verdict.signals[].severity` or the run's exit code. Only the eleven **warn** codes are truly benign.
163
+ `result.verdict.signals[].severity` or the run's exit code. Only the twelve **warn** codes are truly benign.
@@ -1,6 +1,6 @@
1
1
  # Assertions guide
2
2
 
3
- Tracks `cowork-harness 4.2.1` (baseline `desktop-2.16120.0`). Read it when choosing assertion keys: the two orthogonal axes and the goal → key map. The full catalog is `assertion-catalog.md`.
3
+ Tracks `cowork-harness 4.4.0` (baseline `desktop-2.19675.0`). Read it when choosing assertion keys: the two orthogonal axes and the goal → key map. The full catalog is `assertion-catalog.md`.
4
4
 
5
5
  ### Assertions: two orthogonal axes
6
6
 
@@ -11,7 +11,7 @@ Conflating these is the **biggest landmine**. An assertion key has two independe
11
11
  not: match prose with `transcript_matches` / `transcript_contains` (stable lexical markers only —
12
12
  not semantic content the model paraphrases, which re-records red); check structured JSON with YAML
13
13
  `artifact_json` (or the [pytest lane](https://github.com/yaniv-golan/cowork-harness/blob/main/python/README.md) for complex predicates), not via a transcript substring.
14
- To check a command that RAN (not one the agent mentioned), use the object form `tool_called: {tool, input: {command: <regex>}, scope?, result?}` — see [assertion-catalog.md](./assertion-catalog.md).
14
+ To check a command that RAN (not one the agent mentioned), use the object form `tool_called: {tool: <name>, input: {command: <regex>}, scope?, result?}` — see [assertion-catalog.md](./assertion-catalog.md).
15
15
  - **Axis B — survives `replay`?** *Independent of Axis A.* On the token-free `replay` lane, only
16
16
  **content keys** evaluate; filesystem / egress keys are skipped (live-only) — loudly, via an
17
17
  `::warning::` annotation, not a silent no-op. A key
@@ -36,25 +36,71 @@ them by what you're trying to prove:
36
36
  | the skill didn't error out of a tool | `tool_no_error: <regex>`, `max_tool_errors: <N>` |
37
37
  | it didn't waste repeated identical calls | `max_redundant_tool_calls: <N>` |
38
38
  | a deliverable reached the user | `user_visible_artifact: <path>` (+ `no_scratchpad_leak: true` if it delivers via `present_files` — **`container` only**) |
39
- | an internal name/path did **not** leak into a delivered file | `artifact_text: {artifact, not_contains}` — `artifact_json`'s companion for non-JSON bodies; literal path, no glob, so one entry per delivered surface |
39
+ | an internal name/path did **not** leak into a delivered file | `artifact_text: {artifact, not_contains: [..]}` — `artifact_json`'s companion for non-JSON bodies; literal path, no glob, so one entry per delivered surface |
40
40
  | a named path must **not** exist after the run | `file_absent: <path>` (**live/verify-run only**) — do NOT invert `no_unexpected_files`: that is an allowlist over *newly created* files and needs a pre-run manifest |
41
41
  | a to-do workflow finished | `all_tasks_completed: true`, `task_status: {match, status}` |
42
42
  | a skill / connector / tool was **offered** | `skill_available`, `connector_available`, `tool_available` (all `<regex>`) |
43
43
  | a skill actually **ran** (or must NOT) | `skill_triggered: <regex>`, `no_skill_triggered: <regex>` |
44
44
  | a tool ran **inside** a skill's scope | `skill_tool_used: {skill, tool}` |
45
45
  | a sub-agent did the work | `subagent_output_contains: {contains}`, `subagent_dispatched: <regex>`, `dispatch_count_max: <N>` |
46
- | a `context: fork` skill answered correctly | `tool_result_matches: '^Skill "[^"]*" completed \(forked execution\)[\s\S]*<pattern>'` — its answer is the `Skill` tool result, not a sub-agent output, so `subagent_output_contains` and `semantic_matches` never see it (foreground fork only — a backgrounded fork's result carries no answer). Only when the MODEL invokes the skill: a `/<skill> …` prompt runs the fork with no `Skill` call and no such result — use `skill_triggered` + `transcript_matches` there |
47
- | a pre-existing input wasn't mutated (incl. `uploads/**`) | `input_unmodified: <glob>` or `[<glob>, …]` (live/verify-run) |
46
+ | a `context: fork` skill answered correctly | `tool_result_matches: '^Skill "[^"]*" completed \(forked execution\)[\s\S]*<pattern>'` — its answer is the `Skill` tool result, not a sub-agent output, so `subagent_output_contains` never sees it and `semantic_matches` sees it only with `include_fork_results: true` (foreground fork only — a backgrounded fork's result carries no answer). Only when the MODEL invokes the skill: a `/<skill> …` prompt runs the fork with no `Skill` call and no such result — use `skill_triggered` + `transcript_matches` there |
47
+ | a pre-existing input wasn't mutated (incl. `uploads/**`) | `input_unmodified: <glob> \| [<glob>, …]` (live/verify-run) |
48
48
  | no authored interactive artifact silently loses its Submit under Cowork | `no_lost_write_back: true` (**live-only**; static Tier A over the run's authored `.html`/`.py`/`.js`; per-scenario gate for the same class `analyze-skill` scans) |
49
49
  | a resource ceiling held | `max_peak_rss_bytes: <N>` (**live-only**) |
50
- | the user was **shown** the right choices, in order | `question_options: {when_question, equals}` — the option SET/ORDER a gate offered (`question_asked` matches the text only); order is compared by default |
50
+ | the user was **shown** the right choices, in order | `question_options: {when_question, equals: [..]}` — the option SET/ORDER a gate offered (`question_asked` matches the text only); order is compared by default |
51
51
  | the user was **told something specific** at a gate | `question_context: {when_question, matches}` — a regex over the question label + option labels + option **descriptions**. Reach for this when the wording may land in an option's `description`, which `question_asked` and `question_options` cannot see |
52
+ | every gate offered exactly one (or at most N) options of a kind | `question_option_count: {matches, exactly}` — counts the option LABELS matching a regex on EVERY sub-question asked (`when_question` narrows); zero sub-questions asked fails |
52
53
  | a hook blocked / didn't block a tool | `hook_blocked: <regex>`, `no_hook_blocked: true` (replay needs a `controlOut` cassette) |
53
54
  | every MCP round-trip succeeded | `no_mcp_error: true` (**live-only**) |
54
55
  | a context compaction happened | `compaction_occurred: true` |
56
+ | THIS run wrote the file (not just that it is there) | `file_exists: {path, authored: true}` (also `user_visible_artifact`, and `authored: true` on `artifact_text`/`artifact_json`) — an untouched pre-run file fails; needs the pre-run manifest, which `authored` arms |
55
57
 
56
58
  Every one of these still obeys the two axes above — several are live-only or need a `controlOut`
57
59
  cassette on replay, so check the catalog's replay class before putting one on a PR gate.
58
60
  `cowork-harness assertions --list` prints the full, always-current key set with one-line semantics
59
61
  straight from the schema — treat it (and the catalog) as the source of truth; this map is a
60
62
  goal-oriented index into it, not a second catalog.
63
+
64
+ ## Pointwise or pairwise judging
65
+
66
+ `semantic_matches` grades fixed claims against the run alone; reach for it when the criteria are concrete and
67
+ checkable. `semantic_pairwise` asks whether the run is better than, as good as, or worse than a **frozen reference**
68
+ — an earlier run's output, frozen once with `ref freeze` — and fits when quality is easier to compare than to score
69
+ (a rewrite of an existing skill, "is v2 better than v1"). Three cautions:
70
+
71
+ - **Freeze once, never regenerate.** A reference frozen from a run changes meaning if it is replaced; freeze a NEW
72
+ store when the task or the evidence options change (a different prompt or scope is refused, not compared).
73
+ - **A per-case comparison cannot see a cross-case collapse.** If every output drifts toward one style, each can still
74
+ "beat the reference". Pair it with a structural or set-level assert when that risk matters.
75
+ - **Do not judge with the model under test.** Pin a different `judge_model`; the harness warns when they match.
76
+
77
+ #### Step-scoped scenarios — test one late step of a long pipeline
78
+
79
+ A pipeline skill (score → draft → appendix) is expensive to re-run end to end just to check its last step.
80
+ Start the run from the state the earlier steps leave behind instead:
81
+
82
+ 1. Run the pipeline once to the point you want (or stop it there) with `--keep`, then
83
+ `cowork-harness fixture export <run-dir> --out fixtures/after-scoring` — it copies the run's `outputs/`
84
+ byte-for-byte and refuses (writing nothing) a file that carries a secret or a host path. Commit the directory.
85
+ 2. Point the scenario at it and ask for the late step only:
86
+
87
+ ```yaml
88
+ fidelity: container
89
+ prompt: Draft the investor memo from the scored deck.
90
+ workspace_fixture: fixtures/after-scoring # copied into outputs/ before turn 1
91
+ assert:
92
+ - file_exists: {path: outputs/memo.md, authored: true}
93
+ - artifact_json: {artifact: outputs/scores/deck.json, path: total, exists: true, authored: false}
94
+ - semantic_matches: {rubric: ["the memo cites the deck's total score"], evidence_files: ["outputs/memo.md"]}
95
+ ```
96
+ 3. Assert on what the STEP produces. A fixture file the step never touched is pre-run, not authored, so
97
+ `semantic_matches` does not grade it (a rewritten one is graded). A `file_exists`/`user_visible_artifact`/
98
+ `artifact_text`/`artifact_json` on a fixture path is refused at load unless it says `authored: true`
99
+ (the step must write it) or `authored: false` (inheriting it is fine) — otherwise it would pass on the
100
+ fixture alone.
101
+
102
+ What it models: re-invoking the skill in the same Cowork session after it stopped mid-work or finished — the
103
+ files persist in `outputs/` and the skill resumes from them. The only difference is that the prior conversation
104
+ context does not come along; a skill that depends on it cannot be tested this way. Editing the fixture stales
105
+ the scenario's cassette (`fixture` staleness — re-record). Full rules: [Starting from a saved
106
+ workspace](https://github.com/yaniv-golan/cowork-harness/blob/main/docs/scenario.md#starting-from-a-saved-workspace-workspace_fixture).
@@ -1,6 +1,6 @@
1
1
  # Authoring a scenario
2
2
 
3
- Tracks `cowork-harness 4.2.1` (baseline `desktop-2.16120.0`). Read it when composing a `scenarios/*.yaml`: session vs scenario, discovery, the fidelity tier, the answer path, `web_fetch`, and scaffold + lint.
3
+ Tracks `cowork-harness 4.4.0` (baseline `desktop-2.19675.0`). Read it when composing a `scenarios/*.yaml`: session vs scenario, discovery, the fidelity tier, the answer path, `web_fetch`, and scaffold + lint.
4
4
 
5
5
  ## Part I — AUTHOR a scenario
6
6
 
@@ -56,8 +56,12 @@ Set the tier in the **scenario's `fidelity:` field**, not a flag — `run` rejec
56
56
  `/sessions/<id>`, folders at `/sessions/<id>/mnt/<name>`, delivery via `present_files`. Cowork's
57
57
  **remote** lane runs server-side in a cloud container with a different filesystem (`$HOME/mnt/`),
58
58
  different delivery (`/mnt/user-data/outputs/` + `SendUserFile`) and a server-authored prompt; no tier
59
- reproduces it and none can — that container is not something a local tool can stand up. Which lane a
60
- real session gets is a Cowork setting ("Only on this computer"), observed **off** on a current install.
59
+ reproduces it and none can — that container is not something a local tool can stand up. No setting
60
+ reliably decides which lane a real session gets: sessions ran in the cloud with "Only on this computer"
61
+ **on** (observed 2026-10-02), and for Pro and Max plans Anthropic
62
+ [announces](https://support.claude.com/en/articles/15520349-use-claude-cowork-on-web-desktop-and-mobile) that new
63
+ tasks run in the cloud from 2026-10-06. Check the session's own lane
64
+ ([how](https://github.com/yaniv-golan/cowork-harness/blob/main/docs/fidelity-gaps.md#which-lane-a-session-actually-ran-on)).
61
65
  So: behaviour conclusions (triggering, tool sequencing, gate handling) travel between lanes; anything
62
66
  asserting a **path, mount or delivery mechanism** is a claim about the local lane only. Declare
63
67
  `lane: remote` when the scenario is about that lane — the affected assertions then refuse to grade
@@ -235,6 +239,7 @@ cowork-harness lint scenarios/*.yaml
235
239
  | `fidelity-missing` | ERROR | no `fidelity:` (required since 4.0.0) |
236
240
  | `file-absent-contradiction` | ERROR | one path under both `file_exists` and `file_absent` |
237
241
  | `gate-needs-controlout` | INFO | gate assertions, which evaluate on replay only when the cassette has `controlOut` |
242
+ | `hook-output-control-char` | ERROR | a `hook_output_contains` / `hook_output_not_contains` `text` or `matches` holding a control character (a double-quoted YAML `\b` is a backspace) — the harness refuses it at load |
238
243
  | `host-path-assert-cowork` | WARN | `transcript_no_host_path` on `cowork` — it fails by design if the tier resolves to hostloop |
239
244
  | `host-path-assert-tier` | ERROR | `transcript_no_host_path` on `hostloop` / `protocol`, where it fails by design |
240
245
  | `lane-remote-incompatible-key` | ERROR | `present_files_called` / `no_scratchpad_leak` / `user_visible_artifact` on `lane: remote` (the runtime rejects them at load, so the tier rules are suppressed there) |
@@ -252,6 +257,7 @@ cowork-harness lint scenarios/*.yaml
252
257
  | `regex-double-quoted` | WARN | a double-quoted regex with an unescaped backslash (YAML strips it) |
253
258
  | `replay-noop` | WARN | every assertion is live-only or a verdict modifier, so a replay gate verifies nothing |
254
259
  | `slash-prompt-forked-result-anchor` | WARN | a `prompt:` starting with `/<skill>` plus a `tool_result_*` anchored on `forked execution` — a slash-invoked skill makes no `Skill` call, so that tool result never exists; assert `skill_triggered` instead |
260
+ | `slash-skill-name-differs-from-plugin` | WARN | a `prompt:` starting with a bare `/<skill>` that names a skill of a plugin the scenario's session stages, where the plugin's name differs — the agent expands it, but real Cowork's app has refused that typed form; pick it from the slash menu or name the skill like its plugin (`/<plugin>:<skill>` was not measured with a single copy installed). Names follow the agent: `.claude-plugin/plugin.json` `name`/`skills`, else the dir name; a skill's sanitized directory name. Reads the `session:` file and its `local_plugins`/`remote_plugins`; silent for an inline `session:`, for marketplace-delivered plugins, and when the files are not on this machine |
255
261
  | `tool-called-always-passes` | INFO | `tool_called` with `count: {min: 0}` and no `max` — it asserts nothing |
256
262
  | `tool-input-regex-redactable` | WARN | a `tool_not_called` input literal the redaction policy rewrites in the committed cassette (or a policy pattern it cannot check offline) |
257
263
  | `tool-input-shell-tier` | INFO | the object form with `tool: Bash` and a `command` on `hostloop` / `cowork`, where shell runs as `mcp__workspace__bash` — list both |
@@ -262,26 +268,40 @@ cowork-harness lint scenarios/*.yaml
262
268
  | `vacuous-gate-assert` | WARN | `gate_answers_delivered` with no presence companion (zero gates passes it), or inert beside `questions_count_max: 0` |
263
269
  | `scenario-invalid` | ERROR | the harness's scenario loader refuses the file (via `cowork-harness lint` only — see below) |
264
270
  | `baseline-unknown` | ERROR | `baseline:` names no baseline this CLI ships (via `cowork-harness lint` only) |
271
+ | `workspace-fixture-invalid` | ERROR | the run refuses the scenario's `workspace_fixture` directory — missing, a symlink, a hard link, agent-config paths, untracked files in git mode, empty, or over a size cap (via `cowork-harness lint` only) |
272
+ | `workspace-fixture-not-relative` | WARN | `workspace_fixture` is an absolute or `~/` path rather than one relative to the scenario file — it names a directory on one machine; when it does not exist where lint runs it is not checked (via `cowork-harness lint` only) |
273
+ | `workspace-fixture-vacuous-assert` | ERROR | a `file_exists` / `user_visible_artifact` / `artifact_text` / `artifact_json` names a file (or directory) the `workspace_fixture` provides without stating `authored:` — it would pass on the fixture alone (via `cowork-harness lint` only) |
265
274
  | `lint-loader-internal` | ERROR | the wrapper could not run its loader check on a file — a harness bug; it never falls back to a lint that skipped the loader |
266
275
 
267
276
  `scaffold` auto-upgrades the tier if you ask for egress on `protocol`, so it never emits a scenario `lint`
268
277
  would reject.
269
278
 
270
279
  **Lint the skill itself: `cowork-harness lint-skill <skill-dir>`.** It checks the skill, not a scenario:
271
- Cowork host-loop footguns (`${CLAUDE_PLUGIN_ROOT}` in a VM bash step, hook events, a misplaced
280
+ Cowork host-loop footguns (a bare `$CLAUDE_PLUGIN_ROOT` in a VM bash step, the plugin root forwarded
281
+ through bash to a host-side reader, hook events, a misplaced
272
282
  `hooks.json`, an unresolvable `subagent_type`), the evidence corpus a `critique` can package
273
283
  (`references/critique.md`), and two size caps. `skill-body-over-reattach-cap` (WARN) fires when the
274
284
  `SKILL.md` body, frontmatter excluded, passes 19,000 B — after a compaction the agent re-attaches only the
275
285
  start of an invoked skill — and `skill-body-near-reattach-cap` (INFO) from 80% of that;
276
286
  `skill-reference-over-read-cap` (WARN) fires on a `references/**.md` over 60,000 B, past which a
277
287
  whole-file Read returns a partial view. `--strict` fails on WARN, never on INFO. To accept a reviewed
278
- judgement-call finding, pass `--ignore-rule <rule>[=<glob>]` (repeatable; the glob matches the finding's
279
- file) or fence the text in `SKILL.md` with `<!-- lint-skill: ignore-start <rule>[,<rule>…]: <reason> -->`
288
+ judgement-call finding, list it in a `--suppressions <file>` JSON file (one entry per accepted site:
289
+ `rule`, `file`, the exact source line as `match`, and a required `reason`), pass `--ignore-rule
290
+ <rule>[=<glob>]` (repeatable; the glob matches the finding's file) or fence the text in `SKILL.md` with `<!-- lint-skill: ignore-start <rule>[,<rule>…]: <reason> -->`
280
291
  … `<!-- lint-skill: ignore-end -->` (outside any code fence). A suppressed finding is still printed; it
281
292
  stops gating. A provable rule (an ERROR, a misplaced `hooks.json`, a missing pinned agent) cannot be
282
- suppressed: naming it, or an unknown rule, in `--ignore-rule` is a usage error (exit 2); in a marker it is
293
+ suppressed: naming it, or an unknown rule, in `--ignore-rule` or a `--suppressions` entry is a usage error (exit 2); in a marker it is
283
294
  WARN `lint-skill-ignore-invalid`, as is any other malformed marker. An unclosed marker is WARN
284
- `lint-skill-ignore-unclosed`, and one that suppresses nothing is INFO `lint-skill-ignore-unused`.
295
+ `lint-skill-ignore-unclosed`, and a marker, `--ignore-rule` or entry that suppresses nothing is INFO
296
+ `lint-skill-ignore-unused` (WARN under `--strict-ignores`).
297
+
298
+ **A marker is an edit to `SKILL.md`, and it costs what any edit costs.** The skill hash covers the file's
299
+ content (unless the session's `staleness.hash_ignore` excludes it), so adding or moving a marker stales every cassette of that skill (a paid re-record to clear), and
300
+ the agent reads the marker text like the rest of the file, which counts toward the re-attach cap. When
301
+ either cost matters, prefer `--suppressions <file>`, kept outside the plugin: it touches neither, and each
302
+ entry accepts exactly one site, so a new copy of an accepted line still fails `--strict`. Add
303
+ `--strict-ignores` so an entry whose site is gone fails too. `--ignore-rule <rule>=<glob>` also touches
304
+ neither, but it suppresses the rule for the whole file, including any new site.
285
305
 
286
306
  **`cowork-harness lint` runs the loader: a file it calls clean is one `run`/`record` will load.** Anything
287
307
  the loader refuses — an unknown key, a wrong value type (a scalar `semantic_matches.rubric`), a bad regex,