cowork-harness 1.25.0 → 2.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude/skills/cowork-harness/SKILL.md +29 -14
- package/.claude/skills/cowork-harness/references/ci-recipe.md +66 -22
- package/.claude/skills/cowork-harness/references/critique.md +1 -1
- package/.claude/skills/cowork-harness/references/fidelity-and-answers.md +2 -2
- package/.claude/skills/cowork-harness/references/scenario-schema.md +3 -3
- package/.claude/skills/cowork-harness/references/task-recipes.md +3 -3
- package/.claude/skills/cowork-harness/scripts/scenario.py +80 -7
- package/CHANGELOG.md +637 -2
- package/CONTRIBUTING.md +12 -0
- package/DESIGN.md +10 -7
- package/README.md +28 -23
- package/RELEASING.md +28 -7
- package/SECURITY.md +10 -8
- package/SPEC.md +40 -20
- package/baselines/desktop-1.32885.1.json +1 -0
- package/baselines/desktop-1.34493.1.json +791 -0
- package/dist/answer-policy.js +42 -0
- package/dist/cli.js +31 -20
- package/dist/run/cassette.js +956 -120
- package/dist/run/jcs.js +81 -0
- package/dist/run/skill-hash.js +182 -45
- package/dist/session.js +13 -10
- package/dist/sync/baseline-diff.js +9 -0
- package/dist/sync/cowork-sync.js +172 -15
- package/dist/types.js +15 -12
- package/docs/boundary.md +2 -2
- package/docs/cassette.md +207 -43
- package/docs/decider-dir.md +16 -2
- package/docs/discovery.md +3 -1
- package/docs/fidelity-gaps.md +45 -6
- package/docs/invariants.md +2 -1
- package/docs/maintenance.md +64 -10
- package/docs/protocol.md +6 -3
- package/docs/run-status.md +2 -2
- package/docs/scenario.md +44 -11
- package/docs/session.md +14 -8
- package/docs/subagents.md +13 -6
- package/examples/README.md +9 -6
- package/examples/data/mcp.json +10 -0
- package/examples/data/project/notes.txt +1 -0
- package/examples/data/report.pdf +54 -0
- package/examples/data/sales.csv +6 -0
- package/examples/data/sales_eur.csv +6 -0
- package/examples/replays/README.md +1 -1
- package/examples/replays/example-multiselect-gate.cassette.json +49 -44
- package/examples/replays/example-pdf-skill.cassette.json +86 -85
- package/examples/replays/hostloop-computer-links.cassette.json +70 -63
- package/examples/scenarios/csv-fx-normalize.yaml +40 -0
- package/examples/scenarios/csv-metrics.yaml +41 -0
- package/examples/scenarios/example-pdf-skill.yaml +38 -0
- package/examples/scenarios/hostloop-computer-links.yaml +43 -0
- package/examples/scenarios/protocol-smoke.yaml +46 -0
- package/examples/scenarios/skill-loads.yaml +18 -0
- package/examples/scenarios/trigger-accuracy-sweep/negative-unrelated-request.yaml +18 -0
- package/examples/scenarios/trigger-accuracy-sweep/positive-clear-pdf-request.yaml +29 -0
- package/examples/sessions/csv-fx-normalize.yaml +15 -0
- package/examples/sessions/csv-metrics.yaml +15 -0
- package/examples/sessions/default.yaml +54 -0
- package/examples/sessions/hostloop-computer-links.yaml +9 -0
- package/examples/sessions/protocol-smoke.yaml +10 -0
- package/examples/sessions/skill.yaml +10 -0
- package/examples/skills/csv-fx-normalize/.claude-plugin/plugin.json +1 -0
- package/examples/skills/csv-fx-normalize/skills/csv-fx-normalize/SKILL.md +40 -0
- package/examples/skills/csv-fx-normalize/skills/csv-fx-normalize/scripts/normalize.py +118 -0
- package/examples/skills/csv-metrics/.claude-plugin/plugin.json +1 -0
- package/examples/skills/csv-metrics/skills/csv-metrics/SKILL.md +39 -0
- package/examples/skills/csv-metrics/skills/csv-metrics/scripts/metrics.py +143 -0
- package/examples/skills/my-pdf-skill/.claude-plugin/plugin.json +1 -0
- package/examples/skills/my-pdf-skill/skills/my-pdf-skill/SKILL.md +6 -0
- package/llms.txt +1 -1
- package/package.json +6 -2
- package/python/README.md +12 -0
- package/python/conftest.py +15 -3
- package/python/cowork_harness.py +44 -0
- package/python/test_lane_optin.py +117 -0
- package/python/test_scenario_lint.py +69 -0
- package/schema/cassette.v10.json +1 -1
- package/schema/cassette.v11.json +1 -1
- package/schema/cassette.v12.json +318 -0
- package/schema/cassette.v9.json +21 -19
- package/schema/run-result.json +68 -284
- package/schema/scenario.schema.json +1 -1
- package/schema/verify-cassettes.json +6 -2
- package/scripts/bump-version.ts +6 -5
- package/scripts/check-versions.ts +199 -10
- package/scripts/gen-schema.ts +13 -1
|
@@ -3,8 +3,8 @@ name: cowork-harness
|
|
|
3
3
|
description: Test or debug a Claude Code skill/plugin under Claude Cowork's runtime — sandboxed agent, default-deny egress, the can_use_tool permission/question protocol — using the cowork-harness CLI. Use when validating or regression-testing a skill, authoring or debugging a scenario YAML (prompt + scripted answers + assert:), choosing a fidelity tier, scripting AskUserQuestion / tool-permission answers, or asserting artifacts, egress, or sub-agent dispatch. Especially when a harness run no-ops an assertion, fails on an unanswered gate, false-greens, a steered answer never reaches the model, or a web_fetch is unexpectedly denied or gated. Also when iterating or hardening a skill across fixes, or grounding a skill's self-critique against its own run evidence — including a document-analysis skill (cap table, deck, financial model, transcript) that needs an uploaded file attached to be critiqued at all. NOT for generic unit testing (pytest/vitest of your own scripts) or non-Cowork CI. Covers the skill / run / chat / record / replay / trace / decide / assertions / scaffold commands and the session-vs-scenario split.
|
|
4
4
|
metadata:
|
|
5
5
|
author: cowork-harness
|
|
6
|
-
version:
|
|
7
|
-
tracks-harness: cowork-harness
|
|
6
|
+
version: 2.0.1
|
|
7
|
+
tracks-harness: cowork-harness 2.0.1 (baseline desktop-1.34493.1)
|
|
8
8
|
---
|
|
9
9
|
|
|
10
10
|
# cowork-harness
|
|
@@ -22,8 +22,8 @@ flagged with a loud `::warning::`, not silent — auto-answer a gate, observe an
|
|
|
22
22
|
allowlist). This skill exists mostly to keep you out of those traps — the Gotchas section below is
|
|
23
23
|
the highest-value part. Read it.
|
|
24
24
|
|
|
25
|
-
> **Version note:** the facts and `file:line` pointers here track `cowork-harness
|
|
26
|
-
> `desktop-1.
|
|
25
|
+
> **Version note:** the facts and `file:line` pointers here track `cowork-harness 2.0.1` (baseline
|
|
26
|
+
> `desktop-1.34493.1`). If your checkout is newer, prefer the live `--help` and — in a repo checkout —
|
|
27
27
|
> `SPEC.md` / `docs/*.md` over this snapshot, and re-run the bundled linter.
|
|
28
28
|
|
|
29
29
|
## Preflight — make sure the harness can actually run
|
|
@@ -39,7 +39,7 @@ Before the first command, confirm the CLI is reachable and **fail loud** (never
|
|
|
39
39
|
|
|
40
40
|
- **One-shot check.** Run `cowork-harness doctor [--tier <tier>]` first — a read-only prerequisite check that inspects Docker, the staged agent, the token, and the baseline in one pass. The bullets below explain each thing it checks (and how to fix it).
|
|
41
41
|
- **Replay-only? Skip `doctor`.** Replaying committed cassettes needs no Docker, no staged agent, and no token — and every tier's `doctor` validates the auth token (the live tiers also Docker + the staged agent), so a ✗ there is expected, not a blocker. Go straight to `cowork-harness replay <cassette>`.
|
|
42
|
-
- **CLI on PATH, recent enough?** Run `cowork-harness --version` — this skill needs **≥
|
|
42
|
+
- **CLI on PATH, recent enough?** Run `cowork-harness --version` — this skill needs **≥ 2.0.1**. If it's missing or older, prefix every command with the version floor `npx "cowork-harness@^2.0.1" <cmd>` (Node ≥ 22), or install once with `npm i -g "cowork-harness@^2.0.1"`. **Pin `@^2.0.1`, never `@latest`** — `@latest` can silently fetch an older CLI and the new commands fail as "unknown command", whereas the floor **fails loud** if no compatible version is published.
|
|
43
43
|
|
|
44
44
|
This skill documents the CURRENT surface, not release history. If `cowork-harness --version` is
|
|
45
45
|
OLDER than the floor, the per-release record of what you are missing is [CHANGELOG.md](https://github.com/yaniv-golan/cowork-harness/blob/main/CHANGELOG.md)
|
|
@@ -220,9 +220,14 @@ run's echoed `--answer "<q>=<choice>"` footer lines into the scenario's `answers
|
|
|
220
220
|
being unattended. Skip the transcribe step only for one-off/exploratory runs.
|
|
221
221
|
<!-- answer-channels:end -->
|
|
222
222
|
|
|
223
|
-
**
|
|
224
|
-
atomic temp+rename, the `{id, answers}` envelope, the multiSelect array shape.
|
|
225
|
-
the raw files is the single most common mistake on this channel.
|
|
223
|
+
**For a QUESTION gate, never hand-write the `req-N.json`/`resp-N.json` files.** `gates` and `answer` wrap
|
|
224
|
+
the protocol — the atomic temp+rename, the `{id, answers}` envelope, the multiSelect array shape.
|
|
225
|
+
Hand-rolling a Monitor over the raw files is the single most common mistake on this channel.
|
|
226
|
+
|
|
227
|
+
`answer` writes `{id, answers}` and nothing else, so it covers **question gates only**. The channel also
|
|
228
|
+
carries **permission**, **dialog** and **elicit** gates, whose replies need `{behavior}` / `{action}` — for
|
|
229
|
+
those, write `resp-N.json` yourself, following the `reply_with` template the gate's own `req-N.json`
|
|
230
|
+
advertises (it spells out the exact shape, e.g. `{"id":"…","behavior":"allow|deny"}`).
|
|
226
231
|
|
|
227
232
|
Exact accepted values (teach precisely): `--on-unanswered` takes `fail|prompt|first` on `skill`,
|
|
228
233
|
only `fail|first` on `run`. **`llm` is NOT an `--on-unanswered` value** — the bare flag
|
|
@@ -407,7 +412,7 @@ That choice is permanent: the cassette rewrites `scenario.session` and `scenario
|
|
|
407
412
|
its own directory** at record time, so moving the file later — a different `--out`, a `git mv`, a copy
|
|
408
413
|
into another repo — leaves those unresolvable and
|
|
409
414
|
`verify-cassettes` reports `unverifiable-skill` ("can't verify ⇒ not green", exit 3) until you
|
|
410
|
-
re-record at the new location. **`record` now says so BEFORE it spends:** a pre-flight — at the same
|
|
415
|
+
re-record at the new location — or point `replay`/`verify-cassettes` at the session with `--session <file>`, which resolves it without a re-record. Since 2.0.0 a bare `replay` FAILS on this class rather than warning. **`record` now says so BEFORE it spends:** a pre-flight — at the same
|
|
411
416
|
pre-spend point as the host-inventory refusal, and in `record --dry-run`, so the rehearsal is free —
|
|
412
417
|
warns when the cassette would be written outside the scenario's tree, or when `session:` itself lives
|
|
413
418
|
outside it (an absolute or `~` path: the mirror case, invisible to a check that only looks at where the
|
|
@@ -416,7 +421,8 @@ missing was anything saying so while you could still act. Related: recording at
|
|
|
416
421
|
(`protocol`/`hostloop`/`cowork`→hostloop) into a repo-visible path is refused outright (gotcha 25).
|
|
417
422
|
The clean answer there is `fidelity: container` (sealed, `HOME=/tmp`, nothing to leak) — **not**
|
|
418
423
|
redirecting `--out` outside the repo and moving the file in afterwards, which trades a loud refusal
|
|
419
|
-
for a
|
|
424
|
+
for a cassette that cannot verify staleness from its own location — recoverable only by passing
|
|
425
|
+
`--session <file>` on every invocation thereafter.
|
|
420
426
|
|
|
421
427
|
**Author answers WITHOUT re-paying — the cheap loop.** You don't need a fresh paid record to discover a
|
|
422
428
|
scenario's gates or their labels: `--keep` ONE run, then `cowork-harness trace <run-dir> --view questions`
|
|
@@ -917,8 +923,11 @@ repeats the assertion/replay-relevant ones alongside the schema (a scoped subset
|
|
|
917
923
|
`artifact_json` / `transcript_matches`), never just `result: success`.
|
|
918
924
|
- `on_unanswered` governs **unanswered** `AskUserQuestion` gates; the `stalled` signal covers
|
|
919
925
|
stalling *after* one is answered — two different failure modes.
|
|
920
|
-
- **Free-text aside:** a "type-it-in-notes" option
|
|
921
|
-
|
|
926
|
+
- **Free-text aside:** the scripted key for a "type-it-in-notes" option is **`answer:`** — an
|
|
927
|
+
arbitrary string delivered verbatim, bypassing label validation by author intent (Cowork
|
|
928
|
+
auto-provides an "Other" free-text path on every gate). Mutually exclusive with `choose:`; setting
|
|
929
|
+
both fails loud. What has no scripted equivalent is the `OTHER:` *directive*
|
|
930
|
+
(it works only on the LLM-decider path, not scripted `choose:`, and only on
|
|
922
931
|
**single-select** gates — a **multi-select** gate is index-only, so `OTHER:` fails loud there; on an
|
|
923
932
|
options-bearing single-select gate a bare out-of-set LLM answer also fails loud (exit 2) — see the
|
|
924
933
|
LLM-decider free-text note in `references/fidelity-and-answers.md`). An LLM decision answered via
|
|
@@ -1028,8 +1037,14 @@ repeats the assertion/replay-relevant ones alongside the schema (a scoped subset
|
|
|
1028
1037
|
25. **Two distinct host-inventory consent flags — a record-time one and a verify-time one.** `record
|
|
1029
1038
|
--allow-host-inventory-fixture` is the boolean consent to proceed recording a host-inheriting
|
|
1030
1039
|
(`protocol`/`hostloop`/`cowork`-resolving-to-hostloop) cassette into a repo-visible path — otherwise
|
|
1031
|
-
`record` refuses (freezing this machine's MCP servers/agents/account into a committed
|
|
1032
|
-
risk).
|
|
1040
|
+
`record` refuses before it spends (freezing this machine's MCP servers/agents/account into a committed
|
|
1041
|
+
fixture is the risk). That pre-spend check **warns rather than refuses when the cassette already
|
|
1042
|
+
exists** — refusing would fire on every `--rerecord-stale` pass — and it reads the tier and the
|
|
1043
|
+
destination path, never the bytes. So `record` also scans the FINISHED recording, after redaction and
|
|
1044
|
+
before the write: a `host-inventory`/`machine-inventory` finding on a repo-visible path is
|
|
1045
|
+
**quarantined** to `<runs-root>/quarantine/` with a `.findings.txt` naming what leaked, and the command
|
|
1046
|
+
fails without writing the path you asked for (the recording is not discarded — you paid for it).
|
|
1047
|
+
`verify-cassettes --allow-host-inventory <regex>` is unrelated: a per-finding suppressor for the
|
|
1033
1048
|
scanner's `host-inventory` class on an already-committed cassette. Passing one where the other command
|
|
1034
1049
|
wants it fails as an unrecognized flag — they don't interchange. Depth: `references/ci-recipe.md`.
|
|
1035
1050
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# CI recipe — replay vs live lanes
|
|
2
2
|
|
|
3
|
-
Self-contained reference. Tracks `cowork-harness
|
|
3
|
+
Self-contained reference. Tracks `cowork-harness 2.0.1` (baseline `desktop-1.34493.1`).
|
|
4
4
|
|
|
5
5
|
**Fastest path: the packaged Action.** One step gets you `replay`/`lint`/`verify-cassettes` plus a PR
|
|
6
6
|
job-summary reporter (verdict table, staleness findings, cost/turns when available):
|
|
@@ -13,7 +13,7 @@ job-summary reporter (verdict table, staleness findings, cost/turns when availab
|
|
|
13
13
|
```
|
|
14
14
|
|
|
15
15
|
The Action's `version` input defaults to `latest` — intentional so a copy-pasted recipe tracks the current
|
|
16
|
-
release; pin an exact version (e.g. `version: "
|
|
16
|
+
release; pin an exact version (e.g. `version: "2.0.1"`) for reproducible CI.
|
|
17
17
|
|
|
18
18
|
Reach for the manual multi-step form below only when you need per-step control the Action's inputs don't
|
|
19
19
|
cover (a custom flag combination, a different runner matrix per step, or `lint`/`verify-cassettes` gated
|
|
@@ -30,15 +30,19 @@ jobs:
|
|
|
30
30
|
runs-on: [self-hosted, linux, arm64] # needs Docker + this staged ELF; not a stock GitHub-hosted runner
|
|
31
31
|
steps:
|
|
32
32
|
- uses: actions/checkout@v4
|
|
33
|
-
- name: Stage the agent binary (official channel, sha256-verified
|
|
33
|
+
- name: Stage the agent binary (official channel, sha256-verified against the pinned baseline)
|
|
34
34
|
run: |
|
|
35
|
-
V=2.1.
|
|
35
|
+
V=2.1.237 # match your scenario's pinned baseline's agentVersion
|
|
36
|
+
# The expected digest is baselines/desktop-<ver>.json -> agentBinary.sha256. Paste it here, or
|
|
37
|
+
# read it with jq if you vendor the baseline. An unverified download is an unverified agent:
|
|
38
|
+
# this step FAILS rather than staging one, which is the whole point of naming it "verified".
|
|
39
|
+
EXPECTED=<paste agentBinary.sha256 for $V>
|
|
36
40
|
curl -fSL "https://downloads.claude.ai/claude-code-releases/$V/linux-arm64/claude" -o "$RUNNER_TEMP/claude-$V"
|
|
41
|
+
echo "$EXPECTED $RUNNER_TEMP/claude-$V" | sha256sum -c -
|
|
37
42
|
chmod +x "$RUNNER_TEMP/claude-$V"
|
|
38
|
-
# verify against the committed baseline's sha256 (baselines/desktop-*.json → agentBinary.sha256)
|
|
39
|
-
# before trusting it — see the "Agent-binary provenance" section of
|
|
40
|
-
# https://github.com/yaniv-golan/cowork-harness/blob/main/docs/maintenance.md
|
|
41
43
|
echo "COWORK_AGENT_BINARY=$RUNNER_TEMP/claude-$V" >> "$GITHUB_ENV"
|
|
44
|
+
# Background on the provenance chain: the "Agent-binary provenance" section of
|
|
45
|
+
# https://github.com/yaniv-golan/cowork-harness/blob/main/docs/maintenance.md
|
|
42
46
|
- uses: yaniv-golan/cowork-harness@main
|
|
43
47
|
with:
|
|
44
48
|
command: run
|
|
@@ -58,7 +62,7 @@ sha256-*checked* but not hard-blocking on mismatch — it's advisory for an inte
|
|
|
58
62
|
GitHub-hosted runners, no token/Docker/agent:
|
|
59
63
|
|
|
60
64
|
```yaml
|
|
61
|
-
- run: npm i -g "cowork-harness
|
|
65
|
+
- run: npm i -g "cowork-harness@^2.0.1"
|
|
62
66
|
- run: cowork-harness lint scenarios/*.yaml --strict --min-severity WARN
|
|
63
67
|
# no silent false-greens. WITHOUT --strict this
|
|
64
68
|
# step cannot fail on a WARN-class rule (e.g.
|
|
@@ -77,7 +81,20 @@ GitHub-hosted runners, no token/Docker/agent:
|
|
|
77
81
|
- run: cowork-harness replay cassettes/ # token-free content/structure
|
|
78
82
|
```
|
|
79
83
|
|
|
80
|
-
|
|
84
|
+
If a cassette has MOVED (a `git mv`, a repo reorg, a copy between projects), staleness becomes
|
|
85
|
+
unverifiable — `verify-cassettes` exits 3 and says the skill dirs are not resolvable. Recover with
|
|
86
|
+
`--session <file>` on either command rather than re-recording:
|
|
87
|
+
|
|
88
|
+
```yaml
|
|
89
|
+
- run: cowork-harness verify-cassettes cassettes/moved.cassette.json --session sessions/default.yaml
|
|
90
|
+
```
|
|
91
|
+
|
|
92
|
+
It takes a session (not skill dirs) so `staleness.hash_ignore` survives, refuses a directory target,
|
|
93
|
+
and echoes the dirs it resolved. Full contract:
|
|
94
|
+
[docs/cassette.md](https://github.com/yaniv-golan/cowork-harness/blob/main/docs/cassette.md).
|
|
95
|
+
|
|
96
|
+
Both lines matter: for the CONTENT-drift classes (`skill`, `shared-root`) `replay` alone **warns** and
|
|
97
|
+
exits 0, so dropping the
|
|
81
98
|
`verify-cassettes` step means a skill edit silently stops being tested. (One command instead of two:
|
|
82
99
|
`replay --fail-on-skill-drift`.)
|
|
83
100
|
|
|
@@ -210,9 +227,23 @@ dollar figures). In a skill repo these cassettes get **committed**. So:
|
|
|
210
227
|
- **Host-inheriting record refused by default — `--allow-host-inventory-fixture` is the consent.** A
|
|
211
228
|
`protocol`/`hostloop`/`cowork`-resolving-to-hostloop record into a repo-visible cassette path would
|
|
212
229
|
freeze THIS machine's MCP server names, agents, and account metadata into a committed fixture, so
|
|
213
|
-
`record` refuses
|
|
230
|
+
`record` refuses **before the paid spawn**. Pass `--allow-host-inventory-fixture` only when the
|
|
214
231
|
recording session genuinely has no personal MCP servers or plugins to leak — it is a per-record
|
|
215
232
|
boolean consent, not a pattern.
|
|
233
|
+
|
|
234
|
+
Two details that matter for a re-record loop. The pre-spend check **warns rather than refuses when the
|
|
235
|
+
cassette already exists**, deliberately: refusing there would fire on every `--rerecord-stale` pass and
|
|
236
|
+
make the escape flag reflexive. And it is a **prediction** — it reads the tier and the destination path,
|
|
237
|
+
never the resulting bytes, so it can be wrong in both directions.
|
|
238
|
+
|
|
239
|
+
So `record` also checks the **evidence**. After redaction and before the write, the finished cassette is
|
|
240
|
+
scanned; a `host-inventory` or `machine-inventory` finding on a repo-visible path is **quarantined** —
|
|
241
|
+
written to `<runs-root>/quarantine/` (honouring `--run-dir`/`COWORK_HARNESS_RUNS_DIR`) with a
|
|
242
|
+
`.findings.txt` sibling naming what leaked, and the command fails without writing the path you asked for.
|
|
243
|
+
The recording is not discarded; you paid for it. Only the machine-identity classes trigger this —
|
|
244
|
+
`email`/`currency`/`domain`/`path` are frequently legitimate scenario content, and a gate that fires on
|
|
245
|
+
those just teaches you to pass the escape flag. Outside a git repo it warns instead: nothing there
|
|
246
|
+
publishes the file by accident.
|
|
216
247
|
- **Always-on scan gate** — `verify-cassettes` flags email / currency / bare-domain / local-path /
|
|
217
248
|
machine-inventory matches it finds in the committed cassettes and **exits non-zero**, so "no leak" is
|
|
218
249
|
a gate, not discipline. Non-zero is not one thing, though: exit `1` means verification RAN and found a
|
|
@@ -222,6 +253,13 @@ dollar figures). In a skill repo these cassettes get **committed**. So:
|
|
|
222
253
|
$? -ne 0 ]` tripwire treats both the same — if you need to tell "the gate caught something" apart from
|
|
223
254
|
"the gate couldn't run", branch on the exit code (or parse `--output-format json`'s per-file
|
|
224
255
|
`findings`/`staleness` vs `unverifiable`/`version`/`error` buckets).
|
|
256
|
+
|
|
257
|
+
**Do not read `error` as "this file was never scanned".** The privacy scan needs a readable *transcript*
|
|
258
|
+
(an `events` array of strings), not a *valid* cassette — so a file that fails shape validation is still
|
|
259
|
+
scanned, and reports its findings **and** its `error`. Each result carries **`privacyScanned`**, which
|
|
260
|
+
answers that question directly. A gate that must not treat "could not verify" as "verified clean" should
|
|
261
|
+
key on `privacyScanned === false`, where `findings: []` is an absence of evidence rather than evidence of
|
|
262
|
+
absence. `--skip-privacy` also reports `false`, for the same reason.
|
|
225
263
|
Suppress synthetic / public reference names (NVCA, Cooley GO, …) with `--allow <regex>`. (Multi-word
|
|
226
264
|
proper names are NOT a default class — too noisy to gate on; add a pattern via config if your corpus
|
|
227
265
|
needs it.)
|
|
@@ -289,7 +327,7 @@ jobs:
|
|
|
289
327
|
with: { node-version: '24' }
|
|
290
328
|
- uses: actions/setup-python@v5
|
|
291
329
|
with: { python-version: '3.x' } # python3 only — PyYAML is bundled with the linter
|
|
292
|
-
- run: npm i -g "cowork-harness
|
|
330
|
+
- run: npm i -g "cowork-harness@^2.0.1"
|
|
293
331
|
- run: cowork-harness lint scenarios/*.yaml # no-silent-false-green (needs python3; PyYAML bundled)
|
|
294
332
|
- run: cowork-harness verify-cassettes cassettes/ --output-format json # privacy + staleness gate
|
|
295
333
|
- run: cowork-harness replay cassettes/ --output-format json # token-free content/structure
|
|
@@ -318,7 +356,7 @@ jobs:
|
|
|
318
356
|
echo "live=true" >> "$GITHUB_OUTPUT"
|
|
319
357
|
fi
|
|
320
358
|
- if: steps.guard.outputs.live == 'true'
|
|
321
|
-
run: npm i -g "cowork-harness
|
|
359
|
+
run: npm i -g "cowork-harness@^2.0.1"
|
|
322
360
|
- if: steps.guard.outputs.live == 'true'
|
|
323
361
|
run: cowork-harness run scenarios/ --output-format json
|
|
324
362
|
env:
|
|
@@ -340,10 +378,14 @@ sandbox).
|
|
|
340
378
|
## Reading results in CI
|
|
341
379
|
|
|
342
380
|
`--output-format json` emits a machine envelope on stdout (human output goes to stderr):
|
|
343
|
-
`{tool, version, command, ok, results[], error}` — one `RunResult` per scenario. Overall pass for a
|
|
344
|
-
scenario
|
|
345
|
-
|
|
346
|
-
|
|
381
|
+
`{tool, version, command, ok, results[], error}` — one `RunResult` per scenario. **Overall pass for a
|
|
382
|
+
scenario is `verdict.pass`** (envelope-wide: `ok`), and it is strictly stronger than
|
|
383
|
+
`result === "success" && assertions.every(pass)`: the verdict also carries ~20 signal codes that fail a run
|
|
384
|
+
with no failing assertion at all — `stalled`, `outputs_delete`, `mount_delete`, `host_path_leak`,
|
|
385
|
+
`undelivered_deliverables`, `missing_capability`, `permissive_auto_allow`, `ended_with_question`,
|
|
386
|
+
`infra_error`, and more. A parser that reimplements the shorter formula greens through every one of them.
|
|
387
|
+
Read `ok` / `verdict.pass`, or just use the exit code — a plain `cowork-harness run scenarios/` is already
|
|
388
|
+
CI-ready without parsing JSON.
|
|
347
389
|
|
|
348
390
|
**Telling *why* a run failed, without scraping stderr.** Each result carries a `verdict` whose
|
|
349
391
|
`failures[]` is the one place every failure reason is enumerated, in one shape (the same object lands
|
|
@@ -406,19 +448,21 @@ evaluation), not present-and-passing. A CI script that counts assertions will se
|
|
|
406
448
|
on replay vs live — compare by assertion identity / pass-fail, not by total count. The count of
|
|
407
449
|
skipped live-only assertions is reported on each replay result as `skippedAssertions: {full, partial}`.
|
|
408
450
|
|
|
409
|
-
## Staleness does NOT fail a replay
|
|
451
|
+
## Staleness mostly does NOT fail a replay — read it from the JSON
|
|
410
452
|
|
|
411
|
-
A plain `replay` **warns** on a
|
|
412
|
-
does **not** imply the recording is still valid.
|
|
453
|
+
A plain `replay` **warns** on a DRIFTED cassette (skill/baseline drift) but stays `ok:true` — a green
|
|
454
|
+
replay does **not** imply the recording is still valid. **Since 2.0.0 there is one exception:**
|
|
455
|
+
`unverifiable-skill` — staleness that could not be checked at all, most often a cassette that moved —
|
|
456
|
+
FAILS a bare `replay`. Recover with `--session <file>` rather than re-recording. Each replay result carries `staleness[]`, an array of
|
|
413
457
|
`{class, message}`, so a token-free gate can act on it without `ok` being the whole story:
|
|
414
458
|
|
|
415
459
|
| `class` | meaning | concern |
|
|
416
460
|
|---|---|---|
|
|
417
461
|
| `baseline` | platform baseline moved since record | low (format-compatible) |
|
|
418
462
|
| `skill` / `shared-root` | the skill source the assertions validate drifted | **high** (assertions may validate dead code) |
|
|
419
|
-
| `format` |
|
|
463
|
+
| `format` | the git/raw file-set mode or agent-scope differs from the recording | re-record under the same setting (waivable) |
|
|
420
464
|
| `unverifiable-baseline` | the latest baseline couldn't be loaded | couldn't verify (env, not skill) |
|
|
421
|
-
| `unverifiable-skill` | skill dirs unresolvable
|
|
465
|
+
| `unverifiable-skill` | skill dirs unresolvable, **or** the cassette predates the hash-format epoch (v12) so its digest is not comparable | couldn't verify the skill. **Fails a bare `replay`.** For the epoch case try `rehash` first — it migrates without a re-record where it can prove the content unchanged (`rehash <file> --session <s.yaml>` if the cassette moved) |
|
|
422
466
|
| `resolved-tier` | a `fidelity: cowork` cassette's recorded `effectiveFidelity` no longer matches what the baseline resolves to today (the host-loop gate flipped) | **high** (the recording exercises the wrong tier) |
|
|
423
467
|
| `unverifiable-tier` | tier check couldn't run for a `fidelity: cowork` cassette (no recorded `effectiveFidelity`, or its pinned baseline failed to load) | couldn't verify the tier — re-record |
|
|
424
468
|
|
|
@@ -433,7 +477,7 @@ fail it either way.)
|
|
|
433
477
|
To gate in CI, pick the severity you want:
|
|
434
478
|
|
|
435
479
|
- `replay --strict` — fail (exit 1) on **any** staleness class.
|
|
436
|
-
- `replay --fail-on-skill-drift` — fail
|
|
480
|
+
- `replay --fail-on-skill-drift` — fail on the skill-source DRIFT classes (`skill` / `shared-root`); `unverifiable-skill` needs no flag, it fails the default verdict since 2.0.0;
|
|
437
481
|
baseline / format / `unverifiable-baseline` stay non-failing warnings.
|
|
438
482
|
Note `--allow-failing` waives this gate wholesale, including the copy `--assert-from` turns on for you:
|
|
439
483
|
`replay --assert-from … --write --allow-failing` will persist an assert block validated against a
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# Critique — the facts a plugin install can't otherwise reach
|
|
2
2
|
|
|
3
|
-
Tracks `cowork-harness
|
|
3
|
+
Tracks `cowork-harness 2.0.1` (baseline `desktop-1.34493.1`). This is **not** a trim of the full
|
|
4
4
|
[`docs/critique.md`](https://github.com/yaniv-golan/cowork-harness/blob/main/docs/critique.md) (repo-only —
|
|
5
5
|
flags, cost, reproduction discipline, known limitations all live there). This file covers exactly what a
|
|
6
6
|
plugin install cannot otherwise discover: the run-dir artifact a harvester actually reads, the report's
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# Fidelity tiers & answer paths
|
|
2
2
|
|
|
3
|
-
Self-contained reference. Tracks `cowork-harness
|
|
3
|
+
Self-contained reference. Tracks `cowork-harness 2.0.1` (baseline `desktop-1.34493.1`).
|
|
4
4
|
|
|
5
5
|
## Fidelity tiers (`fidelity:` in the scenario)
|
|
6
6
|
|
|
@@ -210,7 +210,7 @@ up often enough to spell out:
|
|
|
210
210
|
- **A green `replay` proves "same as when recorded," not "correct today."** `replay` never touches
|
|
211
211
|
a filesystem or network — it re-evaluates assertions from the frozen cassette. A fixed set of
|
|
212
212
|
keys is live-only and **skipped outright** on replay (absent from `assertions[]`, not vacuously
|
|
213
|
-
passed): `file_absent`, `no_delete_in_outputs`, `self_heal_ran`, `transcript_no_host_path`, `egress_denied`,
|
|
213
|
+
passed): `file_absent`, `no_delete_in_outputs`, `no_delete_in_mounts`, `self_heal_ran`, `transcript_no_host_path`, `egress_denied`,
|
|
214
214
|
`egress_allowed`, `no_mcp_error`, `max_peak_rss_bytes`, `semantic_matches`, `no_lost_write_back`, and `expect_denied`.
|
|
215
215
|
Everything else that *is* evaluated is checked against the **recording**, not fresh behavior — a
|
|
216
216
|
green replay says the skill produced these events when it was recorded, not that it still does
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# Scenario & session schema, assertion catalog, web_fetch, full gotchas
|
|
2
2
|
|
|
3
|
-
Self-contained reference for authoring `cowork-harness` scenarios. Tracks `cowork-harness
|
|
4
|
-
(baseline `desktop-1.
|
|
3
|
+
Self-contained reference for authoring `cowork-harness` scenarios. Tracks `cowork-harness 2.0.1`
|
|
4
|
+
(baseline `desktop-1.34493.1`). If your checkout is newer, prefer the live [`docs/scenario.md`](https://github.com/yaniv-golan/cowork-harness/blob/main/docs/scenario.md),
|
|
5
5
|
[`docs/session.md`](https://github.com/yaniv-golan/cowork-harness/blob/main/docs/session.md), and `SPEC.md`.
|
|
6
6
|
|
|
7
7
|
**Minimal scenario** — `prompt` is the only required field:
|
|
@@ -107,7 +107,7 @@ form a relocatable bundle. `~` expands to home.
|
|
|
107
107
|
> references relative to **its own** directory at record time (`scenario.session` and
|
|
108
108
|
> `scenarioSource`), so moving it afterwards — a different `--out`, a
|
|
109
109
|
> `git mv`, a copy into another repo — leaves them unresolvable and `verify-cassettes` reports
|
|
110
|
-
> `unverifiable-skill` ("can't verify ⇒ not green", exit 3) until you re-record at the new location
|
|
110
|
+
> `unverifiable-skill` ("can't verify ⇒ not green", exit 3) until you re-record at the new location, or pass `--session <file>`.
|
|
111
111
|
> Decide where a cassette will live *before* you record it.
|
|
112
112
|
|
|
113
113
|
## Session YAML
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
Each recipe composes facts that live scattered across SKILL.md and the other references into one
|
|
4
4
|
decision path. Every one answers a question a real fleet owner had to work out the hard way.
|
|
5
|
-
Tracks `cowork-harness
|
|
5
|
+
Tracks `cowork-harness 2.0.1` (baseline `desktop-1.34493.1`), same as SKILL.md's front-matter. Recipe 2's `resolved-tier`/`unverifiable-tier` staleness classes and
|
|
6
6
|
Recipe 3's `init-redact` shipped in 0.24.0 and are part of the current feature set — no version gate
|
|
7
7
|
needed if your CLI meets SKILL.md's version floor.
|
|
8
8
|
|
|
@@ -49,11 +49,11 @@ what production would do. Two layers of defense:
|
|
|
49
49
|
|
|
50
50
|
### Cassette anatomy (what you're looking at when you open one)
|
|
51
51
|
|
|
52
|
-
Top-level fields of a `*.cassette.json` (schema [`schema/cassette.
|
|
52
|
+
Top-level fields of a `*.cassette.json` (schema [`schema/cassette.v12.json`](https://github.com/yaniv-golan/cowork-harness/blob/main/schema/cassette.v12.json)):
|
|
53
53
|
|
|
54
54
|
| Field | What it is |
|
|
55
55
|
|---|---|
|
|
56
|
-
| `$schema`, `generator`, `cassetteVersion` | Provenance: schema URL, producing tool, format version — the MINIMUM a reader needs for this scenario, not the recorder's version (current max:
|
|
56
|
+
| `$schema`, `generator`, `cassetteVersion` | Provenance: schema URL, producing tool, format version — the MINIMUM a reader needs for this scenario, not the recorder's version (current max: 12 — the hash-format epoch floors every stamp there, so a fresh recording stamps 12 whatever its `lane:`) |
|
|
57
57
|
| `scenario` | The embedded scenario snapshot at record time |
|
|
58
58
|
| `events` | The recorded agent event stream (the replay source) |
|
|
59
59
|
| `controlOut` | Driver→agent control responses — presence unlocks gate asserts on replay |
|
|
@@ -409,6 +409,67 @@ def _is_positional_choose(choose):
|
|
|
409
409
|
return any(isinstance(v, str) and (v == "first" or v.isdigit()) for v in vals)
|
|
410
410
|
|
|
411
411
|
|
|
412
|
+
# Single-segment absolute paths that legitimately appear in prompt prose. A `/word` from this set is a
|
|
413
|
+
# path, not a slash command, so it never raises the not-leading warning below.
|
|
414
|
+
_SLASH_PATH_WORDS = frozenset(
|
|
415
|
+
{"outputs", "mnt", "tmp", "home", "users", "var", "etc", "usr", "bin", "dev", "opt", "workspace", "root", "srv"}
|
|
416
|
+
)
|
|
417
|
+
# A `/`-prefixed token at a word start (start-of-string or whitespace), captured WHOLE up to the next
|
|
418
|
+
# space. An opening bracket or quote also counts as a word start, so "(/deck-review)" is seen; `and/or`,
|
|
419
|
+
# `8/22` and `https://x` still cannot match, because their slash follows a letter, digit or colon. The
|
|
420
|
+
# token is then classified in Python rather than by a lookahead: an earlier lookahead-based pattern
|
|
421
|
+
# BACKTRACKED, matching `/mn` inside `/mnt/uploads` because a shorter prefix satisfied the lookahead.
|
|
422
|
+
_SLASH_TOKEN_RE = re.compile(r"""(?:^|[\s(\[{"'])/(\S+)""")
|
|
423
|
+
# What a slash command may look like once trailing sentence punctuation is stripped: a bare name, or a
|
|
424
|
+
# plugin-qualified `plugin:skill`. Anchored, so any residual `/` or `.` disqualifies it as a path/filename.
|
|
425
|
+
_SLASH_CMD_NAME_RE = re.compile(r"^[A-Za-z][A-Za-z0-9_-]*(?::[A-Za-z][A-Za-z0-9_-]*)?$")
|
|
426
|
+
|
|
427
|
+
|
|
428
|
+
def _lint_prompt_slash(doc, path):
|
|
429
|
+
"""W: `prompt:` names a slash command somewhere other than position 0.
|
|
430
|
+
|
|
431
|
+
The agent binary resolves a slash command only when the TRIMMED prompt starts with `/` — its parser
|
|
432
|
+
trims, then requires `startsWith("/")` (verified against agent 2.1.239 in the harness's own spawn
|
|
433
|
+
shape: `-p --input-format stream-json --output-format stream-json --setting-sources user`). A slash
|
|
434
|
+
named mid-sentence is never expanded. It reaches the model as ordinary prose, and the model may then
|
|
435
|
+
reach for the `Skill` tool on its own — the model-invocation path, i.e. exactly the unreliable
|
|
436
|
+
auto-trigger a slash is normally used to bypass. The scenario still runs and can still pass, so the
|
|
437
|
+
failure mode is a scenario that silently tests something other than what it reads as.
|
|
438
|
+
|
|
439
|
+
Deliberately silent when the prompt DOES start with `/`: that is the working case. Registration is not
|
|
440
|
+
checkable statically (it depends on how the skill is staged), so an unresolvable leading name is left
|
|
441
|
+
to the run itself, where it shows up as `Unknown command: /x` with `num_turns: 0`.
|
|
442
|
+
"""
|
|
443
|
+
findings = []
|
|
444
|
+
prompt = doc.get("prompt")
|
|
445
|
+
if not isinstance(prompt, str) or prompt.lstrip().startswith("/"):
|
|
446
|
+
return findings
|
|
447
|
+
seen = []
|
|
448
|
+
for m in _SLASH_TOKEN_RE.finditer(prompt):
|
|
449
|
+
# Trailing sentence punctuation is not part of the name, so "use /deck-review." still counts.
|
|
450
|
+
name = m.group(1).rstrip(".,;:!?)]}\"'")
|
|
451
|
+
if not _SLASH_CMD_NAME_RE.match(name):
|
|
452
|
+
continue # a path (`/mnt/uploads`), a filename (`/deck.pdf`), or not command-shaped
|
|
453
|
+
if name.lower() in _SLASH_PATH_WORDS or name in seen:
|
|
454
|
+
continue
|
|
455
|
+
seen.append(name)
|
|
456
|
+
for name in seen:
|
|
457
|
+
findings.append(
|
|
458
|
+
Finding(
|
|
459
|
+
"WARN",
|
|
460
|
+
"prompt-slash-not-leading",
|
|
461
|
+
f"`prompt:` names `/{name}` but does not START with it. A slash command is expanded only "
|
|
462
|
+
"when the trimmed prompt begins with `/`, so here it reaches the model as ordinary prose "
|
|
463
|
+
"and the skill is NOT preloaded — the model may or may not reach for it on its own, which "
|
|
464
|
+
"is the auto-trigger path a slash is normally used to bypass.",
|
|
465
|
+
f'Put the command first — `prompt: "/{name} <args>"` — or drop the slash if the scenario '
|
|
466
|
+
"means to test auto-triggering from a natural request.",
|
|
467
|
+
path,
|
|
468
|
+
)
|
|
469
|
+
)
|
|
470
|
+
return findings
|
|
471
|
+
|
|
472
|
+
|
|
412
473
|
def lint_doc(doc, path, raw_lines):
|
|
413
474
|
findings = []
|
|
414
475
|
if not isinstance(doc, dict):
|
|
@@ -456,6 +517,9 @@ def lint_doc(doc, path, raw_lines):
|
|
|
456
517
|
)
|
|
457
518
|
)
|
|
458
519
|
|
|
520
|
+
# W: a slash command named mid-prompt is never expanded — see _lint_prompt_slash.
|
|
521
|
+
findings.extend(_lint_prompt_slash(doc, path))
|
|
522
|
+
|
|
459
523
|
# W: unknown assertion keys inside assert items (e.g. invented file_not_empty, kind, path)
|
|
460
524
|
unknown_assert = sorted(assert_keys - ASSERT_KEYS)
|
|
461
525
|
for k in unknown_assert:
|
|
@@ -1834,11 +1898,16 @@ def build_scenario(args):
|
|
|
1834
1898
|
)
|
|
1835
1899
|
content_lines.append(" - gate_answers_delivered: true # the steered answers actually reached the model")
|
|
1836
1900
|
|
|
1837
|
-
|
|
1901
|
+
# Two buckets, not one. `file_exists`/`user_visible_artifact` are MANIFEST_KEYS above — they DO
|
|
1902
|
+
# evaluate on replay whenever the cassette carries an artifacts manifest, which `record` has
|
|
1903
|
+
# snapshotted since 0.24. Filing them under a "LIVE-only" heading taught the reader the opposite of
|
|
1904
|
+
# what this file's own taxonomy says, and of what the `manifest-needs-snapshot` INFO tells them.
|
|
1905
|
+
manifest_lines = []
|
|
1838
1906
|
for p in (args.file or []):
|
|
1839
|
-
|
|
1907
|
+
manifest_lines.append(f" - file_exists: {p}")
|
|
1840
1908
|
for p in (args.artifact or []):
|
|
1841
|
-
|
|
1909
|
+
manifest_lines.append(f" - user_visible_artifact: {p}")
|
|
1910
|
+
live_lines = []
|
|
1842
1911
|
if args.no_delete:
|
|
1843
1912
|
live_lines.append(" - no_delete_in_outputs: true")
|
|
1844
1913
|
for h in (args.egress_allowed or []):
|
|
@@ -1850,12 +1919,16 @@ def build_scenario(args):
|
|
|
1850
1919
|
L.append("assert:")
|
|
1851
1920
|
L.append(" # --- content / structure: evaluate on the token-free replay PR gate AND live ---")
|
|
1852
1921
|
L.extend(content_lines)
|
|
1922
|
+
if manifest_lines:
|
|
1923
|
+
L.append(" # --- artifacts: replay-checkable WHEN the cassette carries an artifacts manifest ---")
|
|
1924
|
+
L.extend(manifest_lines)
|
|
1853
1925
|
if live_lines:
|
|
1854
|
-
L.append(" # ---
|
|
1926
|
+
L.append(" # --- egress / deletes: LIVE-only (skipped on replay, with a loud warning) ---")
|
|
1855
1927
|
L.extend(live_lines)
|
|
1856
|
-
|
|
1857
|
-
L.append(" # TODO add
|
|
1858
|
-
L.append(" #
|
|
1928
|
+
if not manifest_lines and not live_lines:
|
|
1929
|
+
L.append(" # TODO add artifact checks (file_exists / user_visible_artifact — these replay from")
|
|
1930
|
+
L.append(" # the cassette's artifacts manifest) and filesystem/egress checks")
|
|
1931
|
+
L.append(" # (egress_denied / no_delete_in_outputs — LIVE lane only).")
|
|
1859
1932
|
|
|
1860
1933
|
if args.web_fetch:
|
|
1861
1934
|
notes.append(
|