cowork-harness 3.6.0 → 3.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude/skills/cowork-harness/SKILL.md +9 -7
- package/.claude/skills/cowork-harness/references/ci-recipe.md +25 -11
- package/.claude/skills/cowork-harness/references/critique.md +46 -12
- package/.claude/skills/cowork-harness/references/fidelity-and-answers.md +8 -1
- package/.claude/skills/cowork-harness/references/scenario-schema.md +5 -3
- package/.claude/skills/cowork-harness/references/task-recipes.md +1 -1
- package/.claude/skills/cowork-harness/scripts/assertion-keys.json +73 -0
- package/.claude/skills/cowork-harness/scripts/scenario.py +380 -36
- package/CHANGELOG.md +512 -0
- package/CODE_OF_CONDUCT.md +143 -0
- package/CONTRIBUTING.md +15 -4
- package/DESIGN.md +31 -4
- package/README.md +34 -7
- package/RELEASING.md +16 -0
- package/SPEC.md +11 -1
- package/baselines/desktop-2.7032.0.json +1049 -0
- package/baselines/prompts/cowork-system-prompt-fingerprints.json +7 -0
- package/baselines/prompts/desktop-1.46388.3/subagent-append-hl.md +4 -1
- package/baselines/provisioning/rootfs-provisioning.json +17 -14
- package/dist/agent/session.js +4 -2
- package/dist/assert.js +36 -0
- package/dist/baseline.js +14 -0
- package/dist/cli.js +5 -5
- package/dist/critique/armor.js +10 -1
- package/dist/critique/command.js +389 -51
- package/dist/critique/corpus-walk.js +80 -0
- package/dist/critique/evaluator.js +17 -8
- package/dist/critique/limitations.js +10 -1
- package/dist/critique/package-evidence.js +175 -121
- package/dist/critique/resolve-agents.js +175 -0
- package/dist/critique/resolve-references.js +261 -0
- package/dist/critique/skill-invocation.js +140 -0
- package/dist/hostloop/workspace-handler.js +10 -3
- package/dist/prompt/subagent-manifest.js +10 -3
- package/dist/run/analyze-skill.js +1 -1
- package/dist/run/cassette.js +2 -0
- package/dist/run/hook-events.js +10 -7
- package/dist/run/skill-flag-surface.js +4 -1
- package/dist/run/tool-name-canonicalization.js +39 -2
- package/dist/run/verdict.js +3 -1
- package/dist/runtime/argv.js +4 -0
- package/dist/session.js +35 -13
- package/dist/sync/cowork-sync.js +52 -15
- package/dist/types.js +9 -0
- package/docs/README.md +3 -1
- package/docs/boundary.md +11 -0
- package/docs/cassette.md +21 -1
- package/docs/ci.md +17 -7
- package/docs/cli.md +56 -20
- package/docs/companion-skill.md +8 -18
- package/docs/critique.md +211 -23
- package/docs/decisions/README.md +2 -0
- package/docs/fidelity-gaps.md +152 -29
- package/docs/gotchas.md +3 -1
- package/docs/maintenance.md +1 -1
- package/docs/protocol.md +20 -0
- package/docs/scenario.md +18 -2
- package/docs/session.md +3 -1
- package/docs/subagents.md +17 -1
- package/examples/README.md +2 -2
- package/examples/replays/README.md +1 -1
- package/examples/replays/example-multiselect-gate.cassette.json +55 -62
- package/examples/replays/example-pdf-skill.cassette.json +122 -102
- package/examples/replays/hostloop-computer-links.cassette.json +62 -64
- package/examples/sessions/stop-hook-probe.yaml +5 -0
- package/llms.txt +2 -2
- package/package.json +2 -1
- package/python/test_scenario_lint.py +233 -0
- package/schema/critique-report.json +41 -3
- package/schema/scenario.schema.json +78 -0
- package/scripts/capture-rootfs-manifest.ts +35 -0
- package/scripts/check-versions.ts +51 -0
- package/scripts/gen-surface.ts +14 -3
|
@@ -3,8 +3,8 @@ name: cowork-harness
|
|
|
3
3
|
description: Test or debug a Claude Code skill/plugin under Claude Cowork's runtime — sandboxed agent, default-deny egress, the can_use_tool permission/question protocol — using the cowork-harness CLI. Use when validating or regression-testing a skill, authoring or debugging a scenario YAML (prompt + scripted answers + assert:), choosing a fidelity tier, scripting AskUserQuestion / tool-permission answers, or asserting artifacts, egress, or sub-agent dispatch. Especially when a harness run no-ops an assertion, fails on an unanswered gate, false-greens, a steered answer never reaches the model, or a web_fetch is unexpectedly denied or gated. Also when iterating or hardening a skill across fixes, or grounding a skill's self-critique against its own run evidence — including a document-analysis skill (cap table, deck, financial model, transcript) that needs an uploaded file attached to be critiqued at all. NOT for generic unit testing (pytest/vitest of your own scripts) or non-Cowork CI. Covers the skill / run / chat / record / replay / trace / decide / assertions / scaffold commands and the session-vs-scenario split.
|
|
4
4
|
metadata:
|
|
5
5
|
author: cowork-harness
|
|
6
|
-
version: 3.
|
|
7
|
-
tracks-harness: cowork-harness 3.
|
|
6
|
+
version: 3.8.0
|
|
7
|
+
tracks-harness: cowork-harness 3.8.0 (baseline desktop-2.7032.0)
|
|
8
8
|
---
|
|
9
9
|
|
|
10
10
|
# cowork-harness
|
|
@@ -25,8 +25,8 @@ flagged with a loud `::warning::`, not silent — auto-answer a gate, observe an
|
|
|
25
25
|
allowlist). This skill exists mostly to keep you out of those traps — the Gotchas section below is
|
|
26
26
|
the highest-value part. Read it.
|
|
27
27
|
|
|
28
|
-
> **Version note:** the facts and `file:line` pointers here track `cowork-harness 3.
|
|
29
|
-
> `desktop-2.
|
|
28
|
+
> **Version note:** the facts and `file:line` pointers here track `cowork-harness 3.8.0` (baseline
|
|
29
|
+
> `desktop-2.7032.0`). If your checkout is newer, prefer the live `--help` and — in a repo checkout —
|
|
30
30
|
> `SPEC.md` / `docs/*.md` over this snapshot, and re-run the bundled linter.
|
|
31
31
|
|
|
32
32
|
## Preflight — make sure the harness can actually run
|
|
@@ -42,7 +42,7 @@ Before the first command, confirm the CLI is reachable and **fail loud** (never
|
|
|
42
42
|
|
|
43
43
|
- **One-shot check.** Run `cowork-harness doctor [--tier <tier>]` first — a read-only prerequisite check that inspects Docker, the staged agent, the token, and the baseline in one pass. The bullets below explain each thing it checks (and how to fix it).
|
|
44
44
|
- **Replay-only? Skip `doctor`.** Replaying committed cassettes needs no Docker, no staged agent, and no token — and every tier's `doctor` validates the auth token (the live tiers also Docker + the staged agent), so a ✗ there is expected, not a blocker. Go straight to `cowork-harness replay <cassette>`.
|
|
45
|
-
- **CLI on PATH, recent enough?** Run `cowork-harness --version` — this skill needs **≥ 3.
|
|
45
|
+
- **CLI on PATH, recent enough?** Run `cowork-harness --version` — this skill needs **≥ 3.8.0**. If it's missing or older, prefix every command with the version floor `npx "cowork-harness@^3.8.0" <cmd>` (Node ≥ 22), or install once with `npm i -g "cowork-harness@^3.8.0"`. **Pin `@^3.8.0`, never `@latest`** — `@latest` can silently fetch an older CLI and the new commands fail as "unknown command", whereas the floor **fails loud** if no compatible version is published.
|
|
46
46
|
|
|
47
47
|
This skill documents the CURRENT surface, not release history. If `cowork-harness --version` is
|
|
48
48
|
OLDER than the floor, the per-release record of what you are missing is [CHANGELOG.md](https://github.com/yaniv-golan/cowork-harness/blob/main/CHANGELOG.md)
|
|
@@ -81,7 +81,7 @@ CI-grade scenario, and the post-hoc debug loop; the rest are narrower tools that
|
|
|
81
81
|
correct answers after you edit it?) → author `semantic_matches` scenarios and gate on the per-claim
|
|
82
82
|
profile. See **Recipe 5** in `references/task-recipes.md` (validity, N≥3, discrimination — the traps).
|
|
83
83
|
- **"What is WRONG with this skill?"** (a graded critique, not a pass/fail) → `cowork-harness critique
|
|
84
|
-
<folder> --prompt "<probe>"`.
|
|
84
|
+
<folder> --prompt "<probe>"`. Up to four model workloads (zero with `--corpus-only`; pass 2 is skipped with no self-report) and 10–20 minutes; budget from
|
|
85
85
|
`report.costUsd.totalUsd`. Reach for it when you want **findings**. **For "what does this skill
|
|
86
86
|
**DO**" — routing, artifact location, narration — use `skill` instead**: no evaluator, a fraction of
|
|
87
87
|
the cost, and it answers that question directly. Report and evidence-package shapes:
|
|
@@ -568,7 +568,9 @@ Recognize these before "fixing" a non-bug:
|
|
|
568
568
|
- **`missing_capability`** — the lean `core` agent image is a deliberate partial mirror of real Cowork's
|
|
569
569
|
rootfs, so a skill that used `soffice`/LibreOffice (`office_convert`), `tesseract` (`ocr`),
|
|
570
570
|
`markitdown`/`magika` (`ml_extract`), `cv2` (`cv`), `camelot`/`tabula` (`pdf_tables`), or `wand`
|
|
571
|
-
(`magick`) can trip this even though real Cowork **ships** those
|
|
571
|
+
(`magick`) can trip this even though real Cowork **ships** those (per the rootfs manifest captured at
|
|
572
|
+
Desktop `2.7032.0` — `baselines/provisioning/rootfs-provisioning.json`, which is the dated evidence
|
|
573
|
+
behind that sentence). The message says so ("likely a FALSE
|
|
572
574
|
NEGATIVE (real Cowork ships them)"). Fix: rebuild full parity (`--build-arg COWORK_FULL_PARITY=1`, point
|
|
573
575
|
`COWORK_AGENT_IMAGE` at it), or — if the skill's fallback is genuinely equivalent — assert
|
|
574
576
|
`allow_missing_capability: true`. (Two sources: a skill *observed using* an omitted family, live lane;
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# CI recipe — replay vs live lanes
|
|
2
2
|
|
|
3
|
-
Self-contained reference. Tracks `cowork-harness 3.
|
|
3
|
+
Self-contained reference. Tracks `cowork-harness 3.8.0` (baseline `desktop-2.7032.0`).
|
|
4
4
|
|
|
5
5
|
**Fastest path: the packaged Action.** One step gets you `replay`/`lint`/`verify-cassettes` plus a PR
|
|
6
6
|
job-summary reporter (verdict table, staleness findings, cost/turns when available):
|
|
@@ -17,7 +17,7 @@ job-summary reporter (verdict table, staleness findings, cost/turns when availab
|
|
|
17
17
|
CLI major reaches your workflow the moment it is promoted even though your `uses:` ref never changed — so a
|
|
18
18
|
copy-pasted recipe that omits the input takes a major bump with no say in it. `^2` holds the major, needs no
|
|
19
19
|
patch number to remember, and only wants a human decision at the next major. Pin an exact version
|
|
20
|
-
(e.g. `version: "3.
|
|
20
|
+
(e.g. `version: "3.8.0"`) instead when you want byte-reproducible CI.
|
|
21
21
|
|
|
22
22
|
Reach for the manual multi-step form below only when you need per-step control the Action's inputs don't
|
|
23
23
|
cover (a custom flag combination, a different runner matrix per step, or `lint`/`verify-cassettes` gated
|
|
@@ -36,13 +36,17 @@ jobs:
|
|
|
36
36
|
- uses: actions/checkout@v4
|
|
37
37
|
- name: Stage the agent binary (official channel, sha256-verified against the pinned baseline)
|
|
38
38
|
run: |
|
|
39
|
-
V=2.1.
|
|
39
|
+
V=2.1.280 # match your scenario's pinned baseline's agentVersion
|
|
40
40
|
# The release channel is NOT always the stable one. Desktop also stages release CANDIDATES,
|
|
41
|
-
# served
|
|
42
|
-
# 2.1.255
|
|
43
|
-
#
|
|
44
|
-
#
|
|
45
|
-
B
|
|
41
|
+
# served from .../claude-code-releases/rc/<commit>/. For some versions the stable path 404s
|
|
42
|
+
# (2.1.255); for others it returns 200 and serves a DIFFERENT BUILD UNDER THE SAME VERSION
|
|
43
|
+
# NUMBER — measured for 2.1.280 on 2026-09-23: stable linux-arm64 233,103,352 B / 92f2b4fd…
|
|
44
|
+
# (commit 80abbfe7) vs RC 233,037,816 B / a1b25d70… (commit bddba3ab), and the staged binary is
|
|
45
|
+
# the RC one. So "the stable URL works" is NOT evidence you have the right build: always take B
|
|
46
|
+
# from your pinned baseline's agentBinary.releaseBaseUrl. The checksum step fails closed if you
|
|
47
|
+
# don't, but it cannot tell you why. Baselines written before that field existed were
|
|
48
|
+
# stable-staged, so their base is the plain https://downloads.claude.ai/claude-code-releases.
|
|
49
|
+
B=https://downloads.claude.ai/claude-code-releases/rc/bddba3abd5da53d0c540cfc76a8d18b44633d568
|
|
46
50
|
# The expected digest is baselines/desktop-<ver>.json -> agentBinary.sha256. Paste it here, or
|
|
47
51
|
# read it with jq if you vendor the baseline. An unverified download is an unverified agent:
|
|
48
52
|
# this step FAILS rather than staging one, which is the whole point of naming it "verified".
|
|
@@ -73,7 +77,7 @@ sha256-*checked* but not hard-blocking on mismatch — it's advisory for an inte
|
|
|
73
77
|
GitHub-hosted runners, no token/Docker/agent:
|
|
74
78
|
|
|
75
79
|
```yaml
|
|
76
|
-
- run: npm i -g "cowork-harness@^3.
|
|
80
|
+
- run: npm i -g "cowork-harness@^3.8.0"
|
|
77
81
|
- run: cowork-harness lint scenarios/*.yaml --strict --min-severity WARN
|
|
78
82
|
# no silent false-greens. WITHOUT --strict this
|
|
79
83
|
# step cannot fail on a WARN-class rule (e.g.
|
|
@@ -322,6 +326,16 @@ A typical skill repo runs four stages, fastest/cheapest first:
|
|
|
322
326
|
literal only when your scenarios name a `fidelity:`; one still in the deprecation window prints one
|
|
323
327
|
defaulted-fidelity notice per scenario.) A scenario that lints with only
|
|
324
328
|
warnings can still be unloadable, so a green `lint` is not evidence the suite runs.
|
|
329
|
+
|
|
330
|
+
**If the repo pays for `critique`, gate the evidence corpus here first, for free:**
|
|
331
|
+
|
|
332
|
+
```bash
|
|
333
|
+
cowork-harness critique <folder> [--skill <name>] --corpus-only --output-format json \
|
|
334
|
+
| jq -e '.corpus.corpusBytes <= .corpus.corpusCeiling'
|
|
335
|
+
```
|
|
336
|
+
|
|
337
|
+
The `jq -e` IS the gate — `--corpus-only` exits 0 on a measurement even over the ceiling. Same
|
|
338
|
+
packager, same git filter as the paid run; the number is a floor (a run-time read can only add).
|
|
325
339
|
3. **Scenarios (replay)** — `cowork-harness replay cassettes/` on every PR (the committed `*.cassette.json`).
|
|
326
340
|
Token-free; content + structure + gate delivery.
|
|
327
341
|
4. **Parity / live (nightly, self-hosted)** — `cowork-harness run scenarios/` with a token + Docker +
|
|
@@ -350,7 +364,7 @@ jobs:
|
|
|
350
364
|
with: { node-version: '24' }
|
|
351
365
|
- uses: actions/setup-python@v5
|
|
352
366
|
with: { python-version: '3.x' } # python3 only — PyYAML is bundled with the linter
|
|
353
|
-
- run: npm i -g "cowork-harness@^3.
|
|
367
|
+
- run: npm i -g "cowork-harness@^3.8.0"
|
|
354
368
|
- run: cowork-harness lint scenarios/*.yaml # no-silent-false-green (needs python3; PyYAML bundled)
|
|
355
369
|
- run: cowork-harness verify-cassettes cassettes/ --output-format json # privacy + staleness gate
|
|
356
370
|
- run: cowork-harness replay cassettes/ --output-format json # token-free content/structure
|
|
@@ -379,7 +393,7 @@ jobs:
|
|
|
379
393
|
echo "live=true" >> "$GITHUB_OUTPUT"
|
|
380
394
|
fi
|
|
381
395
|
- if: steps.guard.outputs.live == 'true'
|
|
382
|
-
run: npm i -g "cowork-harness@^3.
|
|
396
|
+
run: npm i -g "cowork-harness@^3.8.0"
|
|
383
397
|
- if: steps.guard.outputs.live == 'true'
|
|
384
398
|
run: cowork-harness run scenarios/ --output-format json
|
|
385
399
|
env:
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# Critique — the facts a plugin install can't otherwise reach
|
|
2
2
|
|
|
3
|
-
Tracks `cowork-harness 3.
|
|
3
|
+
Tracks `cowork-harness 3.8.0` (baseline `desktop-2.7032.0`). This is **not** a trim of the full
|
|
4
4
|
[`docs/critique.md`](https://github.com/yaniv-golan/cowork-harness/blob/main/docs/critique.md) (repo-only —
|
|
5
5
|
flags, cost, reproduction discipline, known limitations all live there). This file covers exactly what a
|
|
6
6
|
plugin install cannot otherwise discover: the run-dir artifact a harvester actually reads, the report's
|
|
@@ -22,9 +22,12 @@ finding's `evidence` excerpt must resolve verbatim against this package, or it l
|
|
|
22
22
|
|
|
23
23
|
## Cost across critiques — the index, not the reports
|
|
24
24
|
|
|
25
|
-
A critique is FOUR model workloads
|
|
26
|
-
|
|
27
|
-
|
|
25
|
+
A critique is **up to FOUR** model workloads — two graded turns and two evaluator passes — but only the
|
|
26
|
+
TWO graded turns produce a run, so only two produce index rows; the evaluator passes produce none. Each
|
|
27
|
+
critique therefore appends a **roll-up row** (`critiqueRole:"rollup"`) carrying `critiqueTotalUsd` — the
|
|
28
|
+
whole spend across whatever workloads actually ran. **Evaluator pass 2 is skipped entirely when no
|
|
29
|
+
self-report was captured** (nothing to verify), so a completed critique can be three workloads and the
|
|
30
|
+
roll-up covers three. Its own `costUsd` is the **evaluator passes
|
|
28
31
|
only**, so `sum(costUsd)` over every row is exactly true spend with nothing double-counted or missed. The
|
|
29
32
|
turn rows carry `critiqueRole:"task"` / `"reflection"`.
|
|
30
33
|
|
|
@@ -103,19 +106,48 @@ fingerprints differently).
|
|
|
103
106
|
|
|
104
107
|
## What the evaluator was actually shown — `evidenceBudget`
|
|
105
108
|
|
|
106
|
-
Skill-authored content
|
|
107
|
-
|
|
108
|
-
|
|
109
|
+
Skill-authored content ships **WHOLE, not rationed**: `SKILL.md`, every file under the skill's own
|
|
110
|
+
`references/**`, every dispatchable `agents/**.md`, and — for a multi-skill plugin — the shared
|
|
111
|
+
plugin-root `references/` files that the skill's own text, a packaged agent body, or the graded agent's
|
|
112
|
+
own read of it actually points at, up to a **512 KiB combined ceiling** across all four together. A
|
|
113
|
+
breach cuts **loudly**: the named file and byte counts are reported, never silent, never refused. The
|
|
114
|
+
plugin-root class is narrow by design — packaging the WHOLE shared tree instead was measured and
|
|
115
|
+
rejected: on a real 6-skill plugin it pushed one skill's corpus to 107% of the ceiling and cost that
|
|
116
|
+
skill's own `SKILL.md` 37,295 B, and `already-covered` (below) judges by presence with no notion of
|
|
117
|
+
authorship, so another skill's shared docs would silently excuse a real gap — a root file the rule leaves
|
|
118
|
+
out is reported in `corpusOmitted` below, never dropped silently. The **transcript** is bounded
|
|
109
119
|
separately at **128 KiB**, cut **head+tail with an elided middle**, so a run's setup and its conclusion
|
|
110
120
|
both survive a cut instead of just one end.
|
|
111
121
|
|
|
112
122
|
You do not need a paid run to find out where you stand: **`cowork-harness lint-skill <skill-dir>` sizes
|
|
113
123
|
your corpus against the same ceiling**, reporting `skill-corpus-near-evidence-ceiling` (INFO) from 80%
|
|
114
|
-
and `skill-corpus-over-evidence-ceiling` (WARN, so it fails `--strict`) past it. It counts the
|
|
115
|
-
classes
|
|
116
|
-
applies no extension filter, so JSON schemas and rule packs count toward your total), and
|
|
117
|
-
|
|
118
|
-
|
|
124
|
+
and `skill-corpus-over-evidence-ceiling` (WARN, so it fails `--strict`) past it. It counts only the
|
|
125
|
+
skill-local classes — `SKILL.md`, every file under `references/` (**any extension**: the packager
|
|
126
|
+
applies no extension filter, so JSON schemas and rule packs count toward your total), and every
|
|
127
|
+
dispatchable `agents/**.md` a plugin skill can dispatch, and every plugin-root `references/` file the
|
|
128
|
+
skill links — the same four classes the packager ships. The one clause it cannot mirror is the
|
|
129
|
+
run-dependent one (a root reference included only because the graded agent READ it), since a static lint
|
|
130
|
+
has no run to read, so its number can under-report there. It also does not apply staging's git-tracked filter,
|
|
131
|
+
so an untracked skill-local reference inflates the figure the other way; `corpusCuts`/`corpusOmitted`
|
|
132
|
+
below stay the authority.
|
|
133
|
+
|
|
134
|
+
**Cheaper, and the packager's own floor: `critique <folder> --corpus-only`.** NO SPEND — no session, no spawn — it runs the
|
|
135
|
+
same `packageEvidence` call a paid critique makes, over an empty run dir, and prints these six fields
|
|
136
|
+
directly (`--output-format json` for the standard payload envelope; text mode prints the one-line
|
|
137
|
+
percent-of-ceiling summary + packaged-file count). `--prompt` becomes optional; every other flag is still
|
|
138
|
+
parsed and type-checked, but a run-shaping one is ignored and named in `ignoredFlags` (a path value such as
|
|
139
|
+
`--upload` is only checked when a turn stages). The number is a FLOOR — a plugin-root
|
|
140
|
+
reference the agent READS during the graded turn is added at critique time, so a paid run's `corpusBytes`
|
|
141
|
+
is `>=` this. Exit 0 = measured (even over the ceiling — gate yourself on `corpusBytes <= corpusCeiling`);
|
|
142
|
+
exit 2 = usage error, unresolvable target, no readable SKILL.md, or a work tree with 0 tracked files (a
|
|
143
|
+
non-git folder is measured raw, as staging copies it). Unlike `lint-skill`'s static count, it IS the
|
|
144
|
+
packager: untracked files and symlinks outside the plugin are excluded by the same filter and containment
|
|
145
|
+
rule a critique applies, bytes are the same UTF-8-decoded measurement, and the one clause no static
|
|
146
|
+
instrument can see (a plugin-root reference read at run time) is stated as the floor rather than guessed
|
|
147
|
+
at. Known gap: a skill that is a git submodule of its plugin (or any `--skill` subdirectory with nothing
|
|
148
|
+
tracked under it) is REFUSED by `--corpus-only` in staging's terms, but a live critique's packager still
|
|
149
|
+
accepts it from the directory's own index and grades a skill the mount never delivers — pre-existing,
|
|
150
|
+
rare, not fixed here.
|
|
119
151
|
|
|
120
152
|
The report's `evidenceBudget` object says exactly what was shown — read it instead of inferring budgets
|
|
121
153
|
from `dist/` source:
|
|
@@ -124,7 +156,9 @@ from `dist/` source:
|
|
|
124
156
|
|---|---|
|
|
125
157
|
| `corpusBytes` | total skill-content bytes found, BEFORE any cut |
|
|
126
158
|
| `corpusCeiling` | the 512 KiB combined ceiling |
|
|
159
|
+
| `corpusPackaged` | every corpus file whose CONTENT shipped into the corpus sections, by the same key `corpusCuts`/`corpusOmitted` use — so a reader can see exactly which sub-agent bodies and shared references the grade rests on. A file the ceiling zeroed is NOT listed; a partially cut one is, with its loss in `corpusCuts`; a placeholder for an unreadable file is never a corpus entry at all |
|
|
127
160
|
| `corpusCuts` | per-file cut record — empty on every real skill; non-empty only once the ceiling is actually breached |
|
|
161
|
+
| `corpusOmitted` | plugin-root `references/` files present on the HOST under `<plugin>/references/` but NOT packaged (a raw walk, so an untracked file staging would not deliver is listed too — `alsoUntracked: true` says so, and the property is ABSENT rather than `false` when trackedness could not be evaluated), and why: `not-linked` (nothing in the skill's authored text, a packaged agent body, or the graded agent's own read points at it), `not-utf8` (fails to decode as clean UTF-8 — the skill's **own** `references/**` has no such filter), or `ambiguous-read` (the graded agent read a path that exists under both the skill's own `references/` and the plugin root's, so which tree it read cannot be attributed) |
|
|
128
162
|
| `corpusExcluded` | skill files present on the host but never delivered to the agent (see below) |
|
|
129
163
|
| `trimRecord` | which section the transcript trim shaved, and by how much |
|
|
130
164
|
| `packageTruncated` | `true` if ANY section was cut — check this, not `corpusCuts`, for "was anything trimmed": a transcript-only cut leaves `corpusCuts` empty and would otherwise read as "nothing cut" |
|
|
@@ -1,6 +1,13 @@
|
|
|
1
1
|
# Fidelity tiers & answer paths
|
|
2
2
|
|
|
3
|
-
Self-contained reference. Tracks `cowork-harness 3.
|
|
3
|
+
Self-contained reference. Tracks `cowork-harness 3.8.0` (baseline `desktop-2.7032.0`).
|
|
4
|
+
|
|
5
|
+
> **This page vs. the repo docs.** This is the **offline snapshot** that ships inside the installed
|
|
6
|
+
> plugin — it is self-contained on purpose. The repo carries four other fidelity views, each answering a
|
|
7
|
+
> different question: *which tier do I pick?* ([README → Fidelity tiers](https://github.com/yaniv-golan/cowork-harness/blob/main/README.md#fidelity-tiers-pick-per-scenario--per-ci-job)),
|
|
8
|
+
> *what does a tier enforce?* ([docs/boundary.md](https://github.com/yaniv-golan/cowork-harness/blob/main/docs/boundary.md)), *what does it NOT reproduce?*
|
|
9
|
+
> ([docs/fidelity-gaps.md](https://github.com/yaniv-golan/cowork-harness/blob/main/docs/fidelity-gaps.md)), and *why is it built this way?*
|
|
10
|
+
> ([DESIGN.md § 2](https://github.com/yaniv-golan/cowork-harness/blob/main/DESIGN.md#2-parity-matrix-per-tier)).
|
|
4
11
|
|
|
5
12
|
## Fidelity tiers (`fidelity:` in the scenario)
|
|
6
13
|
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# Scenario & session schema, assertion catalog, web_fetch, authoring gotchas
|
|
2
2
|
|
|
3
|
-
Self-contained reference for authoring `cowork-harness` scenarios. Tracks `cowork-harness 3.
|
|
4
|
-
(baseline `desktop-2.
|
|
3
|
+
Self-contained reference for authoring `cowork-harness` scenarios. Tracks `cowork-harness 3.8.0`
|
|
4
|
+
(baseline `desktop-2.7032.0`). If your checkout is newer, prefer the live [`docs/scenario.md`](https://github.com/yaniv-golan/cowork-harness/blob/main/docs/scenario.md),
|
|
5
5
|
[`docs/session.md`](https://github.com/yaniv-golan/cowork-harness/blob/main/docs/session.md), and `SPEC.md`.
|
|
6
6
|
|
|
7
7
|
**Minimal scenario** — `prompt` is the only required field:
|
|
@@ -369,6 +369,8 @@ same set live from the schema.
|
|
|
369
369
|
| `gate_answer_count_min: <N>` | at least N AskUserQuestion gates fired AND were delivered non-error — presence companion to `gate_answers_delivered`'s vacuous-pass. **`: 0` asserts nothing** and does not satisfy that pairing; `>= 1` is **mutually exclusive** with `questions_count_max: 0` (refused by `run`/`skill`/`record`) |
|
|
370
370
|
| `hook_blocked: <regex>` | a PreToolUse hook blocked a tool whose name matches the regex (`RunResult.hookEvents`) — evidence-unavailable if hook telemetry is absent. Replay: needs a `controlOut` cassette (a custom hook's decision lives only there, not the recorded stream) |
|
|
371
371
|
| `no_hook_blocked: true` | no tool was hook-blocked during the run (distinguishes a real tool crash from an intentional hook block) — evidence-unavailable if hook telemetry is absent. Replay: needs a `controlOut` cassette. **Only `true` is valid** |
|
|
372
|
+
| `hook_event_fired: <HookEvent>` | a **command hook** for this event (a plugin's `hooks/hooks.json` or manifest hook — `Stop`, `SessionStart`, `PostToolUse`, …) ran: a `hook_response` system frame with that `hook_event` was recorded (`RunResult.contextEvents`). Any outcome counts. The harness passes `--include-hook-events` whenever a staged plugin declares hooks — that is what puts events other than SessionStart/Setup on the stream — so a recording made without it reports "never fired". Content-class, grades on replay. Recorded end-to-end for `Stop` ([stop-hook-probe.scenario.yaml](https://github.com/yaniv-golan/cowork-harness/blob/main/examples/probes/stop-hook-probe.scenario.yaml)); the other names match the same frame but have not each been recorded |
|
|
373
|
+
| `hook_event_blocked: <HookEvent>` | that command hook **blocked** at least once — a `hook_response` frame for the event carried `exit_code: 2`. Fails naming the exit codes seen when it fired without blocking (a frame with no `exit_code` is reported as such, never counted); fails "never fired" otherwise; cannot-verify when the run has no context events. Content-class |
|
|
372
374
|
| `vm_path_denied: true` | **`fidelity: hostloop` only** — at least one recorded path denial (`RunResult.pathDenials`, any source) targeted a `/sessions` VM path — evidence-unavailable if path-denial telemetry is absent. Replay: needs a `controlOut` cassette. Any other tier FAILS "cannot verify". **Only `true` is valid** |
|
|
373
375
|
| `path_denied: {tool?, path_matches?, source?, agent_scope?}` | **`fidelity: hostloop` only** — a path denial matching ALL given matchers (`tool` glob, `path_matches` regex, `source` ∈ pretooluse/can_use_tool/permission_denied, `agent_scope` ∈ main/subagent/any) was recorded. Replay: needs a `controlOut` cassette. Any other tier FAILS "cannot verify" |
|
|
374
376
|
| `no_path_denied: true` | **`fidelity: hostloop` only** — NO path denial was recorded at all (the channel is already path-scoped, unlike `no_hook_blocked`'s indiscriminate reject). Replay: needs a `controlOut` cassette. Any other tier FAILS "cannot verify". **Only `true` is valid** |
|
|
@@ -454,7 +456,7 @@ sourcing ≠ evaluation (replay warns when you edit one). `verify-run` is the on
|
|
|
454
456
|
`no_vm_path_file_op`, `dispatch_count_max`,
|
|
455
457
|
`skill_triggered`, `no_skill_triggered`, `skill_available`, `connector_available`, `tool_available`,
|
|
456
458
|
`skill_tool_used`, `max_cost_usd`, `max_tokens`, `tool_calls_max`, `tool_no_error`,
|
|
457
|
-
`max_tool_errors`, `max_redundant_tool_calls`, `max_turns`, `compaction_occurred`, `all_tasks_completed`, `task_status`, `task_count_min`, `no_scratchpad_leak`, `present_files_called`, `result`
|
|
459
|
+
`max_tool_errors`, `max_redundant_tool_calls`, `max_turns`, `compaction_occurred`, `hook_event_fired`, `hook_event_blocked`, `all_tasks_completed`, `task_status`, `task_count_min`, `no_scratchpad_leak`, `present_files_called`, `result`
|
|
458
460
|
(`max_cost_usd`/`max_tokens` assert the frozen recording's spend on replay, not fresh spend). The verdict
|
|
459
461
|
modifiers `allow_permissive_auto_allow` / `allow_missing_capability` / `allow_l0_host_config_contamination` /
|
|
460
462
|
`allow_stall` are also kept on replay, evaluated as no-op passes.
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
Each recipe composes facts that live scattered across SKILL.md and the other references into one
|
|
4
4
|
decision path. Every one answers a question a real fleet owner had to work out the hard way.
|
|
5
|
-
Tracks `cowork-harness 3.
|
|
5
|
+
Tracks `cowork-harness 3.8.0` (baseline `desktop-2.7032.0`), same as SKILL.md's front-matter. Recipe 2's `resolved-tier`/`unverifiable-tier` staleness classes and
|
|
6
6
|
Recipe 3's `init-redact` shipped in 0.24.0 and are part of the current feature set — no version gate
|
|
7
7
|
needed if your CLI meets SKILL.md's version floor.
|
|
8
8
|
|
|
@@ -23,6 +23,8 @@
|
|
|
23
23
|
"gate_answer_count_min",
|
|
24
24
|
"gate_answers_delivered",
|
|
25
25
|
"hook_blocked",
|
|
26
|
+
"hook_event_blocked",
|
|
27
|
+
"hook_event_fired",
|
|
26
28
|
"input_unmodified",
|
|
27
29
|
"max_cost_usd",
|
|
28
30
|
"max_peak_rss_bytes",
|
|
@@ -150,6 +152,7 @@
|
|
|
150
152
|
"liveVerifiedHookEvents": [
|
|
151
153
|
"PostToolUse",
|
|
152
154
|
"SessionStart",
|
|
155
|
+
"Stop",
|
|
153
156
|
"UserPromptSubmit"
|
|
154
157
|
],
|
|
155
158
|
"enums": {
|
|
@@ -165,6 +168,76 @@
|
|
|
165
168
|
"once",
|
|
166
169
|
"domain"
|
|
167
170
|
],
|
|
171
|
+
"assert.hook_event_blocked": [
|
|
172
|
+
"PreToolUse",
|
|
173
|
+
"PostToolUse",
|
|
174
|
+
"PostToolUseFailure",
|
|
175
|
+
"PostToolBatch",
|
|
176
|
+
"Notification",
|
|
177
|
+
"UserPromptSubmit",
|
|
178
|
+
"UserPromptExpansion",
|
|
179
|
+
"SessionStart",
|
|
180
|
+
"SessionEnd",
|
|
181
|
+
"Stop",
|
|
182
|
+
"StopFailure",
|
|
183
|
+
"SubagentStart",
|
|
184
|
+
"SubagentStop",
|
|
185
|
+
"PreCompact",
|
|
186
|
+
"PostCompact",
|
|
187
|
+
"PreModelSwitch",
|
|
188
|
+
"PostModelSwitch",
|
|
189
|
+
"PermissionRequest",
|
|
190
|
+
"PermissionDenied",
|
|
191
|
+
"Setup",
|
|
192
|
+
"TeammateIdle",
|
|
193
|
+
"TaskCreated",
|
|
194
|
+
"TaskCompleted",
|
|
195
|
+
"Elicitation",
|
|
196
|
+
"ElicitationResult",
|
|
197
|
+
"ConfigChange",
|
|
198
|
+
"WorktreeCreate",
|
|
199
|
+
"WorktreeRemove",
|
|
200
|
+
"InstructionsLoaded",
|
|
201
|
+
"CwdChanged",
|
|
202
|
+
"FileChanged",
|
|
203
|
+
"DirectoryAdded",
|
|
204
|
+
"MessageDisplay"
|
|
205
|
+
],
|
|
206
|
+
"assert.hook_event_fired": [
|
|
207
|
+
"PreToolUse",
|
|
208
|
+
"PostToolUse",
|
|
209
|
+
"PostToolUseFailure",
|
|
210
|
+
"PostToolBatch",
|
|
211
|
+
"Notification",
|
|
212
|
+
"UserPromptSubmit",
|
|
213
|
+
"UserPromptExpansion",
|
|
214
|
+
"SessionStart",
|
|
215
|
+
"SessionEnd",
|
|
216
|
+
"Stop",
|
|
217
|
+
"StopFailure",
|
|
218
|
+
"SubagentStart",
|
|
219
|
+
"SubagentStop",
|
|
220
|
+
"PreCompact",
|
|
221
|
+
"PostCompact",
|
|
222
|
+
"PreModelSwitch",
|
|
223
|
+
"PostModelSwitch",
|
|
224
|
+
"PermissionRequest",
|
|
225
|
+
"PermissionDenied",
|
|
226
|
+
"Setup",
|
|
227
|
+
"TeammateIdle",
|
|
228
|
+
"TaskCreated",
|
|
229
|
+
"TaskCompleted",
|
|
230
|
+
"Elicitation",
|
|
231
|
+
"ElicitationResult",
|
|
232
|
+
"ConfigChange",
|
|
233
|
+
"WorktreeCreate",
|
|
234
|
+
"WorktreeRemove",
|
|
235
|
+
"InstructionsLoaded",
|
|
236
|
+
"CwdChanged",
|
|
237
|
+
"FileChanged",
|
|
238
|
+
"DirectoryAdded",
|
|
239
|
+
"MessageDisplay"
|
|
240
|
+
],
|
|
168
241
|
"assert.path_denied.agent_scope": [
|
|
169
242
|
"main",
|
|
170
243
|
"subagent",
|