cowork-harness 3.6.0 → 3.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (73) hide show
  1. package/.claude/skills/cowork-harness/SKILL.md +9 -7
  2. package/.claude/skills/cowork-harness/references/ci-recipe.md +25 -11
  3. package/.claude/skills/cowork-harness/references/critique.md +46 -12
  4. package/.claude/skills/cowork-harness/references/fidelity-and-answers.md +8 -1
  5. package/.claude/skills/cowork-harness/references/scenario-schema.md +5 -3
  6. package/.claude/skills/cowork-harness/references/task-recipes.md +1 -1
  7. package/.claude/skills/cowork-harness/scripts/assertion-keys.json +73 -0
  8. package/.claude/skills/cowork-harness/scripts/scenario.py +380 -36
  9. package/CHANGELOG.md +512 -0
  10. package/CODE_OF_CONDUCT.md +143 -0
  11. package/CONTRIBUTING.md +15 -4
  12. package/DESIGN.md +31 -4
  13. package/README.md +34 -7
  14. package/RELEASING.md +16 -0
  15. package/SPEC.md +11 -1
  16. package/baselines/desktop-2.7032.0.json +1049 -0
  17. package/baselines/prompts/cowork-system-prompt-fingerprints.json +7 -0
  18. package/baselines/prompts/desktop-1.46388.3/subagent-append-hl.md +4 -1
  19. package/baselines/provisioning/rootfs-provisioning.json +17 -14
  20. package/dist/agent/session.js +4 -2
  21. package/dist/assert.js +36 -0
  22. package/dist/baseline.js +14 -0
  23. package/dist/cli.js +5 -5
  24. package/dist/critique/armor.js +10 -1
  25. package/dist/critique/command.js +389 -51
  26. package/dist/critique/corpus-walk.js +80 -0
  27. package/dist/critique/evaluator.js +17 -8
  28. package/dist/critique/limitations.js +10 -1
  29. package/dist/critique/package-evidence.js +175 -121
  30. package/dist/critique/resolve-agents.js +175 -0
  31. package/dist/critique/resolve-references.js +261 -0
  32. package/dist/critique/skill-invocation.js +140 -0
  33. package/dist/hostloop/workspace-handler.js +10 -3
  34. package/dist/prompt/subagent-manifest.js +10 -3
  35. package/dist/run/analyze-skill.js +1 -1
  36. package/dist/run/cassette.js +2 -0
  37. package/dist/run/hook-events.js +10 -7
  38. package/dist/run/skill-flag-surface.js +4 -1
  39. package/dist/run/tool-name-canonicalization.js +39 -2
  40. package/dist/run/verdict.js +3 -1
  41. package/dist/runtime/argv.js +4 -0
  42. package/dist/session.js +35 -13
  43. package/dist/sync/cowork-sync.js +52 -15
  44. package/dist/types.js +9 -0
  45. package/docs/README.md +3 -1
  46. package/docs/boundary.md +11 -0
  47. package/docs/cassette.md +21 -1
  48. package/docs/ci.md +17 -7
  49. package/docs/cli.md +56 -20
  50. package/docs/companion-skill.md +8 -18
  51. package/docs/critique.md +211 -23
  52. package/docs/decisions/README.md +2 -0
  53. package/docs/fidelity-gaps.md +152 -29
  54. package/docs/gotchas.md +3 -1
  55. package/docs/maintenance.md +1 -1
  56. package/docs/protocol.md +20 -0
  57. package/docs/scenario.md +18 -2
  58. package/docs/session.md +3 -1
  59. package/docs/subagents.md +17 -1
  60. package/examples/README.md +2 -2
  61. package/examples/replays/README.md +1 -1
  62. package/examples/replays/example-multiselect-gate.cassette.json +55 -62
  63. package/examples/replays/example-pdf-skill.cassette.json +122 -102
  64. package/examples/replays/hostloop-computer-links.cassette.json +62 -64
  65. package/examples/sessions/stop-hook-probe.yaml +5 -0
  66. package/llms.txt +2 -2
  67. package/package.json +2 -1
  68. package/python/test_scenario_lint.py +233 -0
  69. package/schema/critique-report.json +41 -3
  70. package/schema/scenario.schema.json +78 -0
  71. package/scripts/capture-rootfs-manifest.ts +35 -0
  72. package/scripts/check-versions.ts +51 -0
  73. package/scripts/gen-surface.ts +14 -3
@@ -3,8 +3,8 @@ name: cowork-harness
3
3
  description: Test or debug a Claude Code skill/plugin under Claude Cowork's runtime — sandboxed agent, default-deny egress, the can_use_tool permission/question protocol — using the cowork-harness CLI. Use when validating or regression-testing a skill, authoring or debugging a scenario YAML (prompt + scripted answers + assert:), choosing a fidelity tier, scripting AskUserQuestion / tool-permission answers, or asserting artifacts, egress, or sub-agent dispatch. Especially when a harness run no-ops an assertion, fails on an unanswered gate, false-greens, a steered answer never reaches the model, or a web_fetch is unexpectedly denied or gated. Also when iterating or hardening a skill across fixes, or grounding a skill's self-critique against its own run evidence — including a document-analysis skill (cap table, deck, financial model, transcript) that needs an uploaded file attached to be critiqued at all. NOT for generic unit testing (pytest/vitest of your own scripts) or non-Cowork CI. Covers the skill / run / chat / record / replay / trace / decide / assertions / scaffold commands and the session-vs-scenario split.
4
4
  metadata:
5
5
  author: cowork-harness
6
- version: 3.6.0
7
- tracks-harness: cowork-harness 3.6.0 (baseline desktop-2.2553.1)
6
+ version: 3.8.0
7
+ tracks-harness: cowork-harness 3.8.0 (baseline desktop-2.7032.0)
8
8
  ---
9
9
 
10
10
  # cowork-harness
@@ -25,8 +25,8 @@ flagged with a loud `::warning::`, not silent — auto-answer a gate, observe an
25
25
  allowlist). This skill exists mostly to keep you out of those traps — the Gotchas section below is
26
26
  the highest-value part. Read it.
27
27
 
28
- > **Version note:** the facts and `file:line` pointers here track `cowork-harness 3.6.0` (baseline
29
- > `desktop-2.2553.1`). If your checkout is newer, prefer the live `--help` and — in a repo checkout —
28
+ > **Version note:** the facts and `file:line` pointers here track `cowork-harness 3.8.0` (baseline
29
+ > `desktop-2.7032.0`). If your checkout is newer, prefer the live `--help` and — in a repo checkout —
30
30
  > `SPEC.md` / `docs/*.md` over this snapshot, and re-run the bundled linter.
31
31
 
32
32
  ## Preflight — make sure the harness can actually run
@@ -42,7 +42,7 @@ Before the first command, confirm the CLI is reachable and **fail loud** (never
42
42
 
43
43
  - **One-shot check.** Run `cowork-harness doctor [--tier <tier>]` first — a read-only prerequisite check that inspects Docker, the staged agent, the token, and the baseline in one pass. The bullets below explain each thing it checks (and how to fix it).
44
44
  - **Replay-only? Skip `doctor`.** Replaying committed cassettes needs no Docker, no staged agent, and no token — and every tier's `doctor` validates the auth token (the live tiers also Docker + the staged agent), so a ✗ there is expected, not a blocker. Go straight to `cowork-harness replay <cassette>`.
45
- - **CLI on PATH, recent enough?** Run `cowork-harness --version` — this skill needs **≥ 3.6.0**. If it's missing or older, prefix every command with the version floor `npx "cowork-harness@^3.6.0" <cmd>` (Node ≥ 22), or install once with `npm i -g "cowork-harness@^3.6.0"`. **Pin `@^3.6.0`, never `@latest`** — `@latest` can silently fetch an older CLI and the new commands fail as "unknown command", whereas the floor **fails loud** if no compatible version is published.
45
+ - **CLI on PATH, recent enough?** Run `cowork-harness --version` — this skill needs **≥ 3.8.0**. If it's missing or older, prefix every command with the version floor `npx "cowork-harness@^3.8.0" <cmd>` (Node ≥ 22), or install once with `npm i -g "cowork-harness@^3.8.0"`. **Pin `@^3.8.0`, never `@latest`** — `@latest` can silently fetch an older CLI and the new commands fail as "unknown command", whereas the floor **fails loud** if no compatible version is published.
46
46
 
47
47
  This skill documents the CURRENT surface, not release history. If `cowork-harness --version` is
48
48
  OLDER than the floor, the per-release record of what you are missing is [CHANGELOG.md](https://github.com/yaniv-golan/cowork-harness/blob/main/CHANGELOG.md)
@@ -81,7 +81,7 @@ CI-grade scenario, and the post-hoc debug loop; the rest are narrower tools that
81
81
  correct answers after you edit it?) → author `semantic_matches` scenarios and gate on the per-claim
82
82
  profile. See **Recipe 5** in `references/task-recipes.md` (validity, N≥3, discrimination — the traps).
83
83
  - **"What is WRONG with this skill?"** (a graded critique, not a pass/fail) → `cowork-harness critique
84
- <folder> --prompt "<probe>"`. Four model workloads and 10–20 minutes; budget from
84
+ <folder> --prompt "<probe>"`. Up to four model workloads (zero with `--corpus-only`; pass 2 is skipped with no self-report) and 10–20 minutes; budget from
85
85
  `report.costUsd.totalUsd`. Reach for it when you want **findings**. **For "what does this skill
86
86
  **DO**" — routing, artifact location, narration — use `skill` instead**: no evaluator, a fraction of
87
87
  the cost, and it answers that question directly. Report and evidence-package shapes:
@@ -568,7 +568,9 @@ Recognize these before "fixing" a non-bug:
568
568
  - **`missing_capability`** — the lean `core` agent image is a deliberate partial mirror of real Cowork's
569
569
  rootfs, so a skill that used `soffice`/LibreOffice (`office_convert`), `tesseract` (`ocr`),
570
570
  `markitdown`/`magika` (`ml_extract`), `cv2` (`cv`), `camelot`/`tabula` (`pdf_tables`), or `wand`
571
- (`magick`) can trip this even though real Cowork **ships** those. The message says so ("likely a FALSE
571
+ (`magick`) can trip this even though real Cowork **ships** those (per the rootfs manifest captured at
572
+ Desktop `2.7032.0` — `baselines/provisioning/rootfs-provisioning.json`, which is the dated evidence
573
+ behind that sentence). The message says so ("likely a FALSE
572
574
  NEGATIVE (real Cowork ships them)"). Fix: rebuild full parity (`--build-arg COWORK_FULL_PARITY=1`, point
573
575
  `COWORK_AGENT_IMAGE` at it), or — if the skill's fallback is genuinely equivalent — assert
574
576
  `allow_missing_capability: true`. (Two sources: a skill *observed using* an omitted family, live lane;
@@ -1,6 +1,6 @@
1
1
  # CI recipe — replay vs live lanes
2
2
 
3
- Self-contained reference. Tracks `cowork-harness 3.6.0` (baseline `desktop-2.2553.1`).
3
+ Self-contained reference. Tracks `cowork-harness 3.8.0` (baseline `desktop-2.7032.0`).
4
4
 
5
5
  **Fastest path: the packaged Action.** One step gets you `replay`/`lint`/`verify-cassettes` plus a PR
6
6
  job-summary reporter (verdict table, staleness findings, cost/turns when available):
@@ -17,7 +17,7 @@ job-summary reporter (verdict table, staleness findings, cost/turns when availab
17
17
  CLI major reaches your workflow the moment it is promoted even though your `uses:` ref never changed — so a
18
18
  copy-pasted recipe that omits the input takes a major bump with no say in it. `^2` holds the major, needs no
19
19
  patch number to remember, and only wants a human decision at the next major. Pin an exact version
20
- (e.g. `version: "3.6.0"`) instead when you want byte-reproducible CI.
20
+ (e.g. `version: "3.8.0"`) instead when you want byte-reproducible CI.
21
21
 
22
22
  Reach for the manual multi-step form below only when you need per-step control the Action's inputs don't
23
23
  cover (a custom flag combination, a different runner matrix per step, or `lint`/`verify-cassettes` gated
@@ -36,13 +36,17 @@ jobs:
36
36
  - uses: actions/checkout@v4
37
37
  - name: Stage the agent binary (official channel, sha256-verified against the pinned baseline)
38
38
  run: |
39
- V=2.1.275 # match your scenario's pinned baseline's agentVersion
39
+ V=2.1.280 # match your scenario's pinned baseline's agentVersion
40
40
  # The release channel is NOT always the stable one. Desktop also stages release CANDIDATES,
41
- # served only from .../claude-code-releases/rc/<commit>/ — the stable path 404s for those, and
42
- # 2.1.255 is one. Take B from your pinned baseline's agentBinary.releaseBaseUrl; baselines
43
- # written before that field existed were stable-staged, so their base is the plain
44
- # https://downloads.claude.ai/claude-code-releases.
45
- B=https://downloads.claude.ai/claude-code-releases
41
+ # served from .../claude-code-releases/rc/<commit>/. For some versions the stable path 404s
42
+ # (2.1.255); for others it returns 200 and serves a DIFFERENT BUILD UNDER THE SAME VERSION
43
+ # NUMBER — measured for 2.1.280 on 2026-09-23: stable linux-arm64 233,103,352 B / 92f2b4fd…
44
+ # (commit 80abbfe7) vs RC 233,037,816 B / a1b25d70… (commit bddba3ab), and the staged binary is
45
+ # the RC one. So "the stable URL works" is NOT evidence you have the right build: always take B
46
+ # from your pinned baseline's agentBinary.releaseBaseUrl. The checksum step fails closed if you
47
+ # don't, but it cannot tell you why. Baselines written before that field existed were
48
+ # stable-staged, so their base is the plain https://downloads.claude.ai/claude-code-releases.
49
+ B=https://downloads.claude.ai/claude-code-releases/rc/bddba3abd5da53d0c540cfc76a8d18b44633d568
46
50
  # The expected digest is baselines/desktop-<ver>.json -> agentBinary.sha256. Paste it here, or
47
51
  # read it with jq if you vendor the baseline. An unverified download is an unverified agent:
48
52
  # this step FAILS rather than staging one, which is the whole point of naming it "verified".
@@ -73,7 +77,7 @@ sha256-*checked* but not hard-blocking on mismatch — it's advisory for an inte
73
77
  GitHub-hosted runners, no token/Docker/agent:
74
78
 
75
79
  ```yaml
76
- - run: npm i -g "cowork-harness@^3.6.0"
80
+ - run: npm i -g "cowork-harness@^3.8.0"
77
81
  - run: cowork-harness lint scenarios/*.yaml --strict --min-severity WARN
78
82
  # no silent false-greens. WITHOUT --strict this
79
83
  # step cannot fail on a WARN-class rule (e.g.
@@ -322,6 +326,16 @@ A typical skill repo runs four stages, fastest/cheapest first:
322
326
  literal only when your scenarios name a `fidelity:`; one still in the deprecation window prints one
323
327
  defaulted-fidelity notice per scenario.) A scenario that lints with only
324
328
  warnings can still be unloadable, so a green `lint` is not evidence the suite runs.
329
+
330
+ **If the repo pays for `critique`, gate the evidence corpus here first, for free:**
331
+
332
+ ```bash
333
+ cowork-harness critique <folder> [--skill <name>] --corpus-only --output-format json \
334
+ | jq -e '.corpus.corpusBytes <= .corpus.corpusCeiling'
335
+ ```
336
+
337
+ The `jq -e` IS the gate — `--corpus-only` exits 0 on a measurement even over the ceiling. Same
338
+ packager, same git filter as the paid run; the number is a floor (a run-time read can only add).
325
339
  3. **Scenarios (replay)** — `cowork-harness replay cassettes/` on every PR (the committed `*.cassette.json`).
326
340
  Token-free; content + structure + gate delivery.
327
341
  4. **Parity / live (nightly, self-hosted)** — `cowork-harness run scenarios/` with a token + Docker +
@@ -350,7 +364,7 @@ jobs:
350
364
  with: { node-version: '24' }
351
365
  - uses: actions/setup-python@v5
352
366
  with: { python-version: '3.x' } # python3 only — PyYAML is bundled with the linter
353
- - run: npm i -g "cowork-harness@^3.6.0"
367
+ - run: npm i -g "cowork-harness@^3.8.0"
354
368
  - run: cowork-harness lint scenarios/*.yaml # no-silent-false-green (needs python3; PyYAML bundled)
355
369
  - run: cowork-harness verify-cassettes cassettes/ --output-format json # privacy + staleness gate
356
370
  - run: cowork-harness replay cassettes/ --output-format json # token-free content/structure
@@ -379,7 +393,7 @@ jobs:
379
393
  echo "live=true" >> "$GITHUB_OUTPUT"
380
394
  fi
381
395
  - if: steps.guard.outputs.live == 'true'
382
- run: npm i -g "cowork-harness@^3.6.0"
396
+ run: npm i -g "cowork-harness@^3.8.0"
383
397
  - if: steps.guard.outputs.live == 'true'
384
398
  run: cowork-harness run scenarios/ --output-format json
385
399
  env:
@@ -1,6 +1,6 @@
1
1
  # Critique — the facts a plugin install can't otherwise reach
2
2
 
3
- Tracks `cowork-harness 3.6.0` (baseline `desktop-2.2553.1`). This is **not** a trim of the full
3
+ Tracks `cowork-harness 3.8.0` (baseline `desktop-2.7032.0`). This is **not** a trim of the full
4
4
  [`docs/critique.md`](https://github.com/yaniv-golan/cowork-harness/blob/main/docs/critique.md) (repo-only —
5
5
  flags, cost, reproduction discipline, known limitations all live there). This file covers exactly what a
6
6
  plugin install cannot otherwise discover: the run-dir artifact a harvester actually reads, the report's
@@ -22,9 +22,12 @@ finding's `evidence` excerpt must resolve verbatim against this package, or it l
22
22
 
23
23
  ## Cost across critiques — the index, not the reports
24
24
 
25
- A critique is FOUR model workloads but only TWO produce a run, so only two produce index rows; the
26
- evaluator passes produce none. Each critique therefore appends a **roll-up row** (`critiqueRole:"rollup"`)
27
- carrying `critiqueTotalUsd` — the whole four-workload spend. Its own `costUsd` is the **evaluator passes
25
+ A critique is **up to FOUR** model workloads — two graded turns and two evaluator passes — but only the
26
+ TWO graded turns produce a run, so only two produce index rows; the evaluator passes produce none. Each
27
+ critique therefore appends a **roll-up row** (`critiqueRole:"rollup"`) carrying `critiqueTotalUsd` — the
28
+ whole spend across whatever workloads actually ran. **Evaluator pass 2 is skipped entirely when no
29
+ self-report was captured** (nothing to verify), so a completed critique can be three workloads and the
30
+ roll-up covers three. Its own `costUsd` is the **evaluator passes
28
31
  only**, so `sum(costUsd)` over every row is exactly true spend with nothing double-counted or missed. The
29
32
  turn rows carry `critiqueRole:"task"` / `"reflection"`.
30
33
 
@@ -103,19 +106,48 @@ fingerprints differently).
103
106
 
104
107
  ## What the evaluator was actually shown — `evidenceBudget`
105
108
 
106
- Skill-authored content (`SKILL.md`, every `references/**` file, `agents/<skill>.md`) ships **WHOLE, not
107
- rationed** — up to a **512 KiB combined ceiling** across all three together. A breach cuts **loudly**: the
108
- named file and byte counts are reported, never silent, never refused. The **transcript** is bounded
109
+ Skill-authored content ships **WHOLE, not rationed**: `SKILL.md`, every file under the skill's own
110
+ `references/**`, every dispatchable `agents/**.md`, and — for a multi-skill plugin — the shared
111
+ plugin-root `references/` files that the skill's own text, a packaged agent body, or the graded agent's
112
+ own read of it actually points at, up to a **512 KiB combined ceiling** across all four together. A
113
+ breach cuts **loudly**: the named file and byte counts are reported, never silent, never refused. The
114
+ plugin-root class is narrow by design — packaging the WHOLE shared tree instead was measured and
115
+ rejected: on a real 6-skill plugin it pushed one skill's corpus to 107% of the ceiling and cost that
116
+ skill's own `SKILL.md` 37,295 B, and `already-covered` (below) judges by presence with no notion of
117
+ authorship, so another skill's shared docs would silently excuse a real gap — a root file the rule leaves
118
+ out is reported in `corpusOmitted` below, never dropped silently. The **transcript** is bounded
109
119
  separately at **128 KiB**, cut **head+tail with an elided middle**, so a run's setup and its conclusion
110
120
  both survive a cut instead of just one end.
111
121
 
112
122
  You do not need a paid run to find out where you stand: **`cowork-harness lint-skill <skill-dir>` sizes
113
123
  your corpus against the same ceiling**, reporting `skill-corpus-near-evidence-ceiling` (INFO) from 80%
114
- and `skill-corpus-over-evidence-ceiling` (WARN, so it fails `--strict`) past it. It counts the three
115
- classes the ceiling governs — `SKILL.md`, every file under `references/` (**any extension**: the packager
116
- applies no extension filter, so JSON schemas and rule packs count toward your total), and a plugin
117
- skill's `agents/<name>.md`. It does not apply staging's git-tracked filter, so an untracked reference
118
- inflates the figure; `corpusCuts` below stays the authority.
124
+ and `skill-corpus-over-evidence-ceiling` (WARN, so it fails `--strict`) past it. It counts only the
125
+ skill-local classes — `SKILL.md`, every file under `references/` (**any extension**: the packager
126
+ applies no extension filter, so JSON schemas and rule packs count toward your total), and every
127
+ dispatchable `agents/**.md` a plugin skill can dispatch, and every plugin-root `references/` file the
128
+ skill links — the same four classes the packager ships. The one clause it cannot mirror is the
129
+ run-dependent one (a root reference included only because the graded agent READ it), since a static lint
130
+ has no run to read, so its number can under-report there. It also does not apply staging's git-tracked filter,
131
+ so an untracked skill-local reference inflates the figure the other way; `corpusCuts`/`corpusOmitted`
132
+ below stay the authority.
133
+
134
+ **Cheaper, and the packager's own floor: `critique <folder> --corpus-only`.** NO SPEND — no session, no spawn — it runs the
135
+ same `packageEvidence` call a paid critique makes, over an empty run dir, and prints these six fields
136
+ directly (`--output-format json` for the standard payload envelope; text mode prints the one-line
137
+ percent-of-ceiling summary + packaged-file count). `--prompt` becomes optional; every other flag is still
138
+ parsed and type-checked, but a run-shaping one is ignored and named in `ignoredFlags` (a path value such as
139
+ `--upload` is only checked when a turn stages). The number is a FLOOR — a plugin-root
140
+ reference the agent READS during the graded turn is added at critique time, so a paid run's `corpusBytes`
141
+ is `>=` this. Exit 0 = measured (even over the ceiling — gate yourself on `corpusBytes <= corpusCeiling`);
142
+ exit 2 = usage error, unresolvable target, no readable SKILL.md, or a work tree with 0 tracked files (a
143
+ non-git folder is measured raw, as staging copies it). Unlike `lint-skill`'s static count, it IS the
144
+ packager: untracked files and symlinks outside the plugin are excluded by the same filter and containment
145
+ rule a critique applies, bytes are the same UTF-8-decoded measurement, and the one clause no static
146
+ instrument can see (a plugin-root reference read at run time) is stated as the floor rather than guessed
147
+ at. Known gap: a skill that is a git submodule of its plugin (or any `--skill` subdirectory with nothing
148
+ tracked under it) is REFUSED by `--corpus-only` in staging's terms, but a live critique's packager still
149
+ accepts it from the directory's own index and grades a skill the mount never delivers — pre-existing,
150
+ rare, not fixed here.
119
151
 
120
152
  The report's `evidenceBudget` object says exactly what was shown — read it instead of inferring budgets
121
153
  from `dist/` source:
@@ -124,7 +156,9 @@ from `dist/` source:
124
156
  |---|---|
125
157
  | `corpusBytes` | total skill-content bytes found, BEFORE any cut |
126
158
  | `corpusCeiling` | the 512 KiB combined ceiling |
159
+ | `corpusPackaged` | every corpus file whose CONTENT shipped into the corpus sections, by the same key `corpusCuts`/`corpusOmitted` use — so a reader can see exactly which sub-agent bodies and shared references the grade rests on. A file the ceiling zeroed is NOT listed; a partially cut one is, with its loss in `corpusCuts`; a placeholder for an unreadable file is never a corpus entry at all |
127
160
  | `corpusCuts` | per-file cut record — empty on every real skill; non-empty only once the ceiling is actually breached |
161
+ | `corpusOmitted` | plugin-root `references/` files present on the HOST under `<plugin>/references/` but NOT packaged (a raw walk, so an untracked file staging would not deliver is listed too — `alsoUntracked: true` says so, and the property is ABSENT rather than `false` when trackedness could not be evaluated), and why: `not-linked` (nothing in the skill's authored text, a packaged agent body, or the graded agent's own read points at it), `not-utf8` (fails to decode as clean UTF-8 — the skill's **own** `references/**` has no such filter), or `ambiguous-read` (the graded agent read a path that exists under both the skill's own `references/` and the plugin root's, so which tree it read cannot be attributed) |
128
162
  | `corpusExcluded` | skill files present on the host but never delivered to the agent (see below) |
129
163
  | `trimRecord` | which section the transcript trim shaved, and by how much |
130
164
  | `packageTruncated` | `true` if ANY section was cut — check this, not `corpusCuts`, for "was anything trimmed": a transcript-only cut leaves `corpusCuts` empty and would otherwise read as "nothing cut" |
@@ -1,6 +1,13 @@
1
1
  # Fidelity tiers & answer paths
2
2
 
3
- Self-contained reference. Tracks `cowork-harness 3.6.0` (baseline `desktop-2.2553.1`).
3
+ Self-contained reference. Tracks `cowork-harness 3.8.0` (baseline `desktop-2.7032.0`).
4
+
5
+ > **This page vs. the repo docs.** This is the **offline snapshot** that ships inside the installed
6
+ > plugin — it is self-contained on purpose. The repo carries four other fidelity views, each answering a
7
+ > different question: *which tier do I pick?* ([README → Fidelity tiers](https://github.com/yaniv-golan/cowork-harness/blob/main/README.md#fidelity-tiers-pick-per-scenario--per-ci-job)),
8
+ > *what does a tier enforce?* ([docs/boundary.md](https://github.com/yaniv-golan/cowork-harness/blob/main/docs/boundary.md)), *what does it NOT reproduce?*
9
+ > ([docs/fidelity-gaps.md](https://github.com/yaniv-golan/cowork-harness/blob/main/docs/fidelity-gaps.md)), and *why is it built this way?*
10
+ > ([DESIGN.md § 2](https://github.com/yaniv-golan/cowork-harness/blob/main/DESIGN.md#2-parity-matrix-per-tier)).
4
11
 
5
12
  ## Fidelity tiers (`fidelity:` in the scenario)
6
13
 
@@ -1,7 +1,7 @@
1
1
  # Scenario & session schema, assertion catalog, web_fetch, authoring gotchas
2
2
 
3
- Self-contained reference for authoring `cowork-harness` scenarios. Tracks `cowork-harness 3.6.0`
4
- (baseline `desktop-2.2553.1`). If your checkout is newer, prefer the live [`docs/scenario.md`](https://github.com/yaniv-golan/cowork-harness/blob/main/docs/scenario.md),
3
+ Self-contained reference for authoring `cowork-harness` scenarios. Tracks `cowork-harness 3.8.0`
4
+ (baseline `desktop-2.7032.0`). If your checkout is newer, prefer the live [`docs/scenario.md`](https://github.com/yaniv-golan/cowork-harness/blob/main/docs/scenario.md),
5
5
  [`docs/session.md`](https://github.com/yaniv-golan/cowork-harness/blob/main/docs/session.md), and `SPEC.md`.
6
6
 
7
7
  **Minimal scenario** — `prompt` is the only required field:
@@ -369,6 +369,8 @@ same set live from the schema.
369
369
  | `gate_answer_count_min: <N>` | at least N AskUserQuestion gates fired AND were delivered non-error — presence companion to `gate_answers_delivered`'s vacuous-pass. **`: 0` asserts nothing** and does not satisfy that pairing; `>= 1` is **mutually exclusive** with `questions_count_max: 0` (refused by `run`/`skill`/`record`) |
370
370
  | `hook_blocked: <regex>` | a PreToolUse hook blocked a tool whose name matches the regex (`RunResult.hookEvents`) — evidence-unavailable if hook telemetry is absent. Replay: needs a `controlOut` cassette (a custom hook's decision lives only there, not the recorded stream) |
371
371
  | `no_hook_blocked: true` | no tool was hook-blocked during the run (distinguishes a real tool crash from an intentional hook block) — evidence-unavailable if hook telemetry is absent. Replay: needs a `controlOut` cassette. **Only `true` is valid** |
372
+ | `hook_event_fired: <HookEvent>` | a **command hook** for this event (a plugin's `hooks/hooks.json` or manifest hook — `Stop`, `SessionStart`, `PostToolUse`, …) ran: a `hook_response` system frame with that `hook_event` was recorded (`RunResult.contextEvents`). Any outcome counts. The harness passes `--include-hook-events` whenever a staged plugin declares hooks — that is what puts events other than SessionStart/Setup on the stream — so a recording made without it reports "never fired". Content-class, grades on replay. Recorded end-to-end for `Stop` ([stop-hook-probe.scenario.yaml](https://github.com/yaniv-golan/cowork-harness/blob/main/examples/probes/stop-hook-probe.scenario.yaml)); the other names match the same frame but have not each been recorded |
373
+ | `hook_event_blocked: <HookEvent>` | that command hook **blocked** at least once — a `hook_response` frame for the event carried `exit_code: 2`. Fails naming the exit codes seen when it fired without blocking (a frame with no `exit_code` is reported as such, never counted); fails "never fired" otherwise; cannot-verify when the run has no context events. Content-class |
372
374
  | `vm_path_denied: true` | **`fidelity: hostloop` only** — at least one recorded path denial (`RunResult.pathDenials`, any source) targeted a `/sessions` VM path — evidence-unavailable if path-denial telemetry is absent. Replay: needs a `controlOut` cassette. Any other tier FAILS "cannot verify". **Only `true` is valid** |
373
375
  | `path_denied: {tool?, path_matches?, source?, agent_scope?}` | **`fidelity: hostloop` only** — a path denial matching ALL given matchers (`tool` glob, `path_matches` regex, `source` ∈ pretooluse/can_use_tool/permission_denied, `agent_scope` ∈ main/subagent/any) was recorded. Replay: needs a `controlOut` cassette. Any other tier FAILS "cannot verify" |
374
376
  | `no_path_denied: true` | **`fidelity: hostloop` only** — NO path denial was recorded at all (the channel is already path-scoped, unlike `no_hook_blocked`'s indiscriminate reject). Replay: needs a `controlOut` cassette. Any other tier FAILS "cannot verify". **Only `true` is valid** |
@@ -454,7 +456,7 @@ sourcing ≠ evaluation (replay warns when you edit one). `verify-run` is the on
454
456
  `no_vm_path_file_op`, `dispatch_count_max`,
455
457
  `skill_triggered`, `no_skill_triggered`, `skill_available`, `connector_available`, `tool_available`,
456
458
  `skill_tool_used`, `max_cost_usd`, `max_tokens`, `tool_calls_max`, `tool_no_error`,
457
- `max_tool_errors`, `max_redundant_tool_calls`, `max_turns`, `compaction_occurred`, `all_tasks_completed`, `task_status`, `task_count_min`, `no_scratchpad_leak`, `present_files_called`, `result`
459
+ `max_tool_errors`, `max_redundant_tool_calls`, `max_turns`, `compaction_occurred`, `hook_event_fired`, `hook_event_blocked`, `all_tasks_completed`, `task_status`, `task_count_min`, `no_scratchpad_leak`, `present_files_called`, `result`
458
460
  (`max_cost_usd`/`max_tokens` assert the frozen recording's spend on replay, not fresh spend). The verdict
459
461
  modifiers `allow_permissive_auto_allow` / `allow_missing_capability` / `allow_l0_host_config_contamination` /
460
462
  `allow_stall` are also kept on replay, evaluated as no-op passes.
@@ -2,7 +2,7 @@
2
2
 
3
3
  Each recipe composes facts that live scattered across SKILL.md and the other references into one
4
4
  decision path. Every one answers a question a real fleet owner had to work out the hard way.
5
- Tracks `cowork-harness 3.6.0` (baseline `desktop-2.2553.1`), same as SKILL.md's front-matter. Recipe 2's `resolved-tier`/`unverifiable-tier` staleness classes and
5
+ Tracks `cowork-harness 3.8.0` (baseline `desktop-2.7032.0`), same as SKILL.md's front-matter. Recipe 2's `resolved-tier`/`unverifiable-tier` staleness classes and
6
6
  Recipe 3's `init-redact` shipped in 0.24.0 and are part of the current feature set — no version gate
7
7
  needed if your CLI meets SKILL.md's version floor.
8
8
 
@@ -23,6 +23,8 @@
23
23
  "gate_answer_count_min",
24
24
  "gate_answers_delivered",
25
25
  "hook_blocked",
26
+ "hook_event_blocked",
27
+ "hook_event_fired",
26
28
  "input_unmodified",
27
29
  "max_cost_usd",
28
30
  "max_peak_rss_bytes",
@@ -150,6 +152,7 @@
150
152
  "liveVerifiedHookEvents": [
151
153
  "PostToolUse",
152
154
  "SessionStart",
155
+ "Stop",
153
156
  "UserPromptSubmit"
154
157
  ],
155
158
  "enums": {
@@ -165,6 +168,76 @@
165
168
  "once",
166
169
  "domain"
167
170
  ],
171
+ "assert.hook_event_blocked": [
172
+ "PreToolUse",
173
+ "PostToolUse",
174
+ "PostToolUseFailure",
175
+ "PostToolBatch",
176
+ "Notification",
177
+ "UserPromptSubmit",
178
+ "UserPromptExpansion",
179
+ "SessionStart",
180
+ "SessionEnd",
181
+ "Stop",
182
+ "StopFailure",
183
+ "SubagentStart",
184
+ "SubagentStop",
185
+ "PreCompact",
186
+ "PostCompact",
187
+ "PreModelSwitch",
188
+ "PostModelSwitch",
189
+ "PermissionRequest",
190
+ "PermissionDenied",
191
+ "Setup",
192
+ "TeammateIdle",
193
+ "TaskCreated",
194
+ "TaskCompleted",
195
+ "Elicitation",
196
+ "ElicitationResult",
197
+ "ConfigChange",
198
+ "WorktreeCreate",
199
+ "WorktreeRemove",
200
+ "InstructionsLoaded",
201
+ "CwdChanged",
202
+ "FileChanged",
203
+ "DirectoryAdded",
204
+ "MessageDisplay"
205
+ ],
206
+ "assert.hook_event_fired": [
207
+ "PreToolUse",
208
+ "PostToolUse",
209
+ "PostToolUseFailure",
210
+ "PostToolBatch",
211
+ "Notification",
212
+ "UserPromptSubmit",
213
+ "UserPromptExpansion",
214
+ "SessionStart",
215
+ "SessionEnd",
216
+ "Stop",
217
+ "StopFailure",
218
+ "SubagentStart",
219
+ "SubagentStop",
220
+ "PreCompact",
221
+ "PostCompact",
222
+ "PreModelSwitch",
223
+ "PostModelSwitch",
224
+ "PermissionRequest",
225
+ "PermissionDenied",
226
+ "Setup",
227
+ "TeammateIdle",
228
+ "TaskCreated",
229
+ "TaskCompleted",
230
+ "Elicitation",
231
+ "ElicitationResult",
232
+ "ConfigChange",
233
+ "WorktreeCreate",
234
+ "WorktreeRemove",
235
+ "InstructionsLoaded",
236
+ "CwdChanged",
237
+ "FileChanged",
238
+ "DirectoryAdded",
239
+ "MessageDisplay"
240
+ ],
168
241
  "assert.path_denied.agent_scope": [
169
242
  "main",
170
243
  "subagent",