cowork-harness 1.1.0 → 1.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude/skills/cowork-harness/SKILL.md +72 -14
- package/.claude/skills/cowork-harness/references/ci-recipe.md +5 -5
- package/.claude/skills/cowork-harness/references/fidelity-and-answers.md +13 -7
- package/.claude/skills/cowork-harness/references/scenario-schema.md +14 -10
- package/.claude/skills/cowork-harness/references/task-recipes.md +27 -0
- package/.claude/skills/cowork-harness/scripts/assertion-keys.json +3 -0
- package/.claude/skills/cowork-harness/scripts/scenario.py +5 -0
- package/CHANGELOG.md +286 -0
- package/README.md +26 -21
- package/SPEC.md +40 -8
- package/dist/agent/session.js +77 -8
- package/dist/assert.js +339 -40
- package/dist/baseline.js +36 -8
- package/dist/cli.js +129 -35
- package/dist/decide/decider.js +61 -7
- package/dist/decide/external-channel.js +19 -6
- package/dist/hostloop/canusetool-gate.js +86 -2
- package/dist/hostloop/provenance.js +7 -2
- package/dist/hostloop/workspace-handler.js +4 -2
- package/dist/run/analyze-artifact-runtime.js +309 -65
- package/dist/run/analyze-artifact.js +786 -142
- package/dist/run/analyze-skill.js +182 -37
- package/dist/run/artifacts.js +105 -22
- package/dist/run/cassette.js +74 -7
- package/dist/run/chat-result.js +7 -2
- package/dist/run/doctor.js +93 -16
- package/dist/run/execute.js +71 -16
- package/dist/run/inspect-view.js +19 -0
- package/dist/run/latest-run.js +43 -0
- package/dist/run/pre-run-manifest.js +42 -10
- package/dist/run/renderer.js +15 -0
- package/dist/run/run-index.js +16 -5
- package/dist/run/run-status.js +1 -0
- package/dist/run/run.js +49 -7
- package/dist/run/status-target.js +28 -0
- package/dist/run/trace-view.js +28 -8
- package/dist/run/verdict.js +19 -0
- package/dist/runtime/hostloop.js +23 -4
- package/dist/runtime/microvm.js +44 -7
- package/dist/scan.js +17 -14
- package/dist/sync/cowork-sync.js +10 -1
- package/dist/types.js +17 -4
- package/docs/README.md +5 -1
- package/docs/cassette.md +27 -11
- package/docs/debugging.md +34 -0
- package/docs/decisions/README.md +7 -0
- package/docs/fidelity-gaps.md +5 -0
- package/docs/gotchas.md +9 -0
- package/docs/maintenance.md +2 -0
- package/docs/run-status.md +5 -0
- package/docs/scenario.md +67 -9
- package/docs/stats.md +5 -1
- package/docs/subagents.md +35 -1
- package/examples/replays/README.md +1 -1
- package/llms.txt +1 -0
- package/package.json +1 -1
- package/python/README.md +13 -4
- package/schema/run-result.json +19 -1
- package/schema/scenario.schema.json +18 -3
- package/scripts/check-baseline-staleness.ts +147 -0
|
@@ -1,10 +1,10 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: cowork-harness
|
|
3
|
-
description: Test or debug a Claude Code skill/plugin under Claude Cowork's runtime — sandboxed agent, default-deny egress, the can_use_tool permission/question protocol — using the cowork-harness CLI. Use when validating or regression-testing a skill, authoring or debugging a scenario YAML (prompt + scripted answers + assert:), choosing a fidelity tier, scripting AskUserQuestion / tool-permission answers, or asserting artifacts, egress, or sub-agent dispatch. Especially when a harness run no-ops an assertion, fails on an unanswered gate, false-greens, a steered answer never reaches the model, or a web_fetch is unexpectedly denied or gated. NOT for generic unit testing (pytest/vitest of your own scripts) or non-Cowork CI. Covers the skill / run / chat / record / replay / trace / decide / assertions / scaffold commands and the session-vs-scenario split.
|
|
3
|
+
description: Test or debug a Claude Code skill/plugin under Claude Cowork's runtime — sandboxed agent, default-deny egress, the can_use_tool permission/question protocol — using the cowork-harness CLI. Use when validating or regression-testing a skill, authoring or debugging a scenario YAML (prompt + scripted answers + assert:), choosing a fidelity tier, scripting AskUserQuestion / tool-permission answers, or asserting artifacts, egress, or sub-agent dispatch. Especially when a harness run no-ops an assertion, fails on an unanswered gate, false-greens, a steered answer never reaches the model, or a web_fetch is unexpectedly denied or gated. Also when iterating or hardening a skill across fixes, or grounding a skill's self-critique against its own run evidence. NOT for generic unit testing (pytest/vitest of your own scripts) or non-Cowork CI. Covers the skill / run / chat / record / replay / trace / decide / assertions / scaffold commands and the session-vs-scenario split.
|
|
4
4
|
metadata:
|
|
5
5
|
author: cowork-harness
|
|
6
|
-
version: 1.
|
|
7
|
-
tracks-harness: cowork-harness 1.
|
|
6
|
+
version: 1.3.0
|
|
7
|
+
tracks-harness: cowork-harness 1.3.0 (baseline desktop-1.21459.0)
|
|
8
8
|
---
|
|
9
9
|
|
|
10
10
|
# cowork-harness
|
|
@@ -22,7 +22,7 @@ flagged with a loud `::warning::`, not silent — auto-answer a gate, observe an
|
|
|
22
22
|
allowlist). This skill exists mostly to keep you out of those traps — the Gotchas section below is
|
|
23
23
|
the highest-value part. Read it.
|
|
24
24
|
|
|
25
|
-
> **Version note:** the facts and `file:line` pointers here track `cowork-harness 1.
|
|
25
|
+
> **Version note:** the facts and `file:line` pointers here track `cowork-harness 1.3.0` (baseline
|
|
26
26
|
> `desktop-1.21459.0`). If your checkout is newer, prefer the live `--help` and — in a repo checkout —
|
|
27
27
|
> `SPEC.md` / `docs/*.md` over this snapshot, and re-run the bundled linter.
|
|
28
28
|
|
|
@@ -39,9 +39,9 @@ Before the first command, confirm the CLI is reachable and **fail loud** (never
|
|
|
39
39
|
|
|
40
40
|
- **One-shot check.** Run `cowork-harness doctor [--tier <tier>]` first — a read-only prerequisite check that inspects Docker, the staged agent, the token, and the baseline in one pass. The bullets below explain each thing it checks (and how to fix it).
|
|
41
41
|
- **Replay-only? Skip `doctor`.** Replaying committed cassettes needs no Docker, no staged agent, and no token — and every tier's `doctor` validates the auth token (the live tiers also Docker + the staged agent), so a ✗ there is expected, not a blocker. Go straight to `cowork-harness replay <cassette>`.
|
|
42
|
-
- **CLI on PATH, recent enough?** Run `cowork-harness --version` — this skill needs **≥ 1.
|
|
42
|
+
- **CLI on PATH, recent enough?** Run `cowork-harness --version` — this skill needs **≥ 1.3.0**. If it's missing or older, prefix every command with the version floor `npx "cowork-harness@>=1.3.0" <cmd>` (Node ≥ 20), or install once with `npm i -g "cowork-harness@>=1.3.0"`. **Pin `@>=1.3.0`, never `@latest`** — `@latest` can silently fetch an older CLI and the new commands fail as "unknown command", whereas the floor **fails loud** if no compatible version is published.
|
|
43
43
|
|
|
44
|
-
What the ≥ 1.
|
|
44
|
+
What the ≥ 1.3.0 floor gates, by release:
|
|
45
45
|
|
|
46
46
|
- **core set (pre-0.21.0 vintage, or mixed):** `assertions --list`, `scaffold <run-id>`, `trace --view dispatches`, `artifact_json` incl. the `in:` operator (passes when the resolved value deep-equals one of the listed members — value ∈ your list, not the reverse), `verify-cassettes` incl. the `--allow-domain`/`--allow-email`/`--allow-patterns-file` allows (`--allow-patterns-file <path>` is a FILE of patterns, one regex per line — not a path to allow, unlike `--allow <regex>`), batch `record <dir>`/`--rerecord-stale`, `record --concurrency <N>`, record-time redaction, multiSelect/`answer:`, `verify-run` answer-coverage, `record --max-artifact-bytes`, live record-time deciders, scenario `skills:` staleness scoping with `COWORK_HARNESS_AGENT_SCOPE=skill`, `chat --plugin`, and `/help` in the REPL.
|
|
47
47
|
- **0.21.0:** `verify-cassettes --allow-path` (`path` — local absolute filesystem paths — is the scanner's 4th class), and `hostloop`'s native host/VM process split with its `allow_host_writes:` consent field.
|
|
@@ -53,6 +53,8 @@ Before the first command, confirm the CLI is reachable and **fail loud** (never
|
|
|
53
53
|
- **0.33.0:** the `redacted` marker on display-omitted reasoning — `subagents[].reasoning` and the top-level `thinking[]` now carry `{text:"", redacted:true}` when the model returns a signed-but-empty thought (so "reasoned, text omitted" is distinct from "no thought"), plus the fenced `debug.thinking_display` escape hatch.
|
|
54
54
|
- **1.0.0:** first stable release — the SPEC §12 compatibility contract takes effect (covered CLI/schema/env/Action surfaces are now stable; breaking changes need a major bump). No new author-facing command; the floor simply tracks the 1.0 release.
|
|
55
55
|
- **1.1.0:** `analyze-skill` now also flags **interactive-artifact write-backs lost under Cowork** — a relative `fetch`/XHR/`sendBeacon`/`<form method=post>` in an emitted `.html` (or its `.py`/`.js` generator) that silently fails when the artifact is served from Cowork's own origin. `artifact-write-back-lost` gates under `--strict`; `artifact-write-back-suspect` is advisory; an unanalyzable candidate is a could-not-verify exit 3. An optional **`analyze-skill --runtime`** drives the artifact in a headless DOM (needs `jsdom`) to *observe* the lost write-back — enrichment only, never changes the exit code. Plus a `lint` check for a container-only assertion key (`no_scratchpad_leak`/`present_files_called`) used off the `container` tier, and the `doctor --output-format json` envelope frozen as a covered SPEC §12 surface (`schema/doctor.json`).
|
|
56
|
+
- **1.2.0:** three new assertion keys — `no_lost_write_back: true` (the write-back detector above wired as a per-scenario gate over the run's authored files, live-only) and the regex siblings `tool_result_matches`/`tool_result_not_matches` (case-insensitive per-result, for an error-signature *family* a literal substring can't express). **`microvm` outputs are now observable** — its session tree is snapshotted from the VM into the run dir, so `file_exists`/`artifact_json`/`user_visible_artifact`/`no_unexpected_files`/`input_unmodified`/`no_lost_write_back`/`semantic_matches` all work there, no longer `container`/`hostloop`-only. `status <dir>` also resolves the newest session under a `--run-dir` root; the completion footer prints a `→ result: …/result.json` pointer; the `on_unanswered=fail` error also points to `on_unanswered: llm`. `analyze-skill` hardening: a phantom `<script>` prose block no longer sinks a real verdict, a delete/remove flow claiming success classifies as lost (error) not just suspect, and write-back detection widened (optional-call `?.` spellings, member-spelled/aliased `fetch`/`sendBeacon`, axios instance/config forms).
|
|
57
|
+
- **1.3.0:** run-identity for the iterate-across-fixes loop — `skill`/`run` take `--label <tag>` (a generation tag surfaced in `result.json` `runLabel`, the run-index row, `inspect`, and `status.json`) alongside an auto-recorded `skillCommit`, layered on the **authoritative** content-exact `fingerprint.skillHash` a harvest step should pair critiques by (`inspect`/index surface a short `skillHash` prefix). `trace --full-results` captures the full input+result of **every** tool call — successful ones too, not just errors — so an external grader can ground a self-critique finding against the call it cites. `verify-run` now **warns** on skillHash drift for answer-less scenarios (was silent unless the scenario declared scripted `answers`). Plus `skill --allow-missing-capability` (the open-ended-run opt-out for a capability FALSE-NEGATIVE on the lean `core` image), a new **warn-severity `ended_with_question`** verdict signal (the agent's final answer contains a question and the run wrote no `outputs/` deliverable — the lenient sibling of the strict `stalled`), and LLM-decider `OTHER:` free-text answers now marked `[via Other free-text]` in gate provenance.
|
|
56
58
|
- **Agent binary (sandboxed live tiers — `container`/`microvm`/`hostloop`/`cowork`).** The staged Claude Code agent is **bind-mounted** from a local Claude Desktop install, or point `COWORK_AGENT_BINARY` at a `claude-code-vm/<ver>/claude` ELF. Nothing is bundled. `protocol` (L0) and `replay` need no staged agent; for the sandboxed tiers, no agent → no run; report that, don't skip silently.
|
|
57
59
|
- **Docker / Lima.** Only `--fidelity protocol` (L0) runs without them. `container` / `microvm` / `hostloop` / `cowork` need Docker (Lima for L2). If they're absent, drop to `--fidelity protocol` and **say so** — a green that never exercised the sandbox is not a sandbox pass.
|
|
58
60
|
- **Auth.** `CLAUDE_CODE_OAUTH_TOKEN` (preferred), or `ANTHROPIC_API_KEY` / `ANTHROPIC_AUTH_TOKEN`, via env or `.env`. Minting an OAuth token needs the **`claude` CLI** (`npm i -g @anthropic-ai/claude-code`, then `claude setup-token`).
|
|
@@ -221,7 +223,8 @@ them by what you're trying to prove:
|
|
|
221
223
|
| a skill actually **ran** (or must NOT) | `skill_triggered: <regex>`, `no_skill_triggered: <regex>` |
|
|
222
224
|
| a tool ran **inside** a skill's scope | `skill_tool_used: {skill, tool}` |
|
|
223
225
|
| a sub-agent did the work | `subagent_output_contains: {contains}`, `subagent_dispatched: <regex>`, `dispatch_count_max: <N>` |
|
|
224
|
-
| a pre-existing input wasn't mutated (incl. `uploads/**`) | `input_unmodified: <glob>` or `[<glob>, …]` (live/verify-run
|
|
226
|
+
| a pre-existing input wasn't mutated (incl. `uploads/**`) | `input_unmodified: <glob>` or `[<glob>, …]` (live/verify-run) |
|
|
227
|
+
| no authored interactive artifact silently loses its Submit under Cowork | `no_lost_write_back: true` (**live-only**; static Tier A over the run's authored `.html`/`.py`/`.js`; per-scenario gate for the same class `analyze-skill` scans) |
|
|
225
228
|
| a resource ceiling held | `max_peak_rss_bytes: <N>` (**live-only**) |
|
|
226
229
|
| a hook blocked / didn't block a tool | `hook_blocked: <regex>`, `no_hook_blocked: true` (replay needs a `controlOut` cassette) |
|
|
227
230
|
| every MCP round-trip succeeded | `no_mcp_error: true` (**live-only**) |
|
|
@@ -309,8 +312,10 @@ your `answers:` against that kept run, then record once. **But the kept run is a
|
|
|
309
312
|
skill's gate phrasing afterward, re-`--keep` — verify-run's answer-coverage *refuses* (exit 2, "predates the
|
|
310
313
|
current skill") rather than vouch against stale labels, but the trace/inspect path can't warn you, so re-keep
|
|
311
314
|
deliberately. (Same fail-closed family: corrupt gate evidence — unparseable `events.jsonl` lines, or fewer
|
|
312
|
-
gates than `trace.json` recorded questions —
|
|
313
|
-
|
|
315
|
+
gates than `trace.json` recorded questions — a structurally invalid `result.json`, a `command:"replay"`
|
|
316
|
+
result (a replay is a re-check of a recorded cassette, not run evidence — verify the original live run dir),
|
|
317
|
+
and a `mode:"chat"` result (chat carries no assertions or verdict by contract) also refuse rather than
|
|
318
|
+
certify.) (A token-free probe of "which gates fire" isn't possible — gates are model-decided per run.)
|
|
314
319
|
|
|
315
320
|
Run artifacts are written to `~/.cowork-harness/runs/…` by default — **outside any working tree**, so a run
|
|
316
321
|
launched from a repo root never drops sensitive skill inputs/outputs into it. Pass `--run-dir <path>` (or set
|
|
@@ -352,6 +357,15 @@ cassette — has its own recipe:
|
|
|
352
357
|
unless the scenario asserts `allow_missing_capability: true`, which downgrades it to a notice and
|
|
353
358
|
proceeds. Rebuild with `--build-arg COWORK_FULL_PARITY=1` and point `COWORK_AGENT_IMAGE` at it for those
|
|
354
359
|
skills.
|
|
360
|
+
6. **Iterate across fixes — verify before you trust, and don't cross-pair generations.** A green run is
|
|
361
|
+
not a correct run, and a skill's self-reported finding is not real until its cited evidence is found in
|
|
362
|
+
the run's own output. Ground each finding against `result.json` (`finalMessage` = the skill's own
|
|
363
|
+
answer/critique; `toolResults` = tool outputs) and the tool-call stream via
|
|
364
|
+
`cowork-harness trace <run-dir> --output-format json` — add `--full-results` so a successful call's full
|
|
365
|
+
input + result are captured, not just errored ones. When iterating, tag generations with `--label` and
|
|
366
|
+
pair a critique only with a `result.json` whose `fingerprint.skillHash` **matches** the skill that
|
|
367
|
+
produced it (`inspect`/the run-index row surface a short `skillHash` prefix; `verify-run` warns when a
|
|
368
|
+
kept run predates the current skill). See `docs/debugging.md` (repo-only) for the full loop.
|
|
355
369
|
|
|
356
370
|
#### Interpreting verdict signals
|
|
357
371
|
|
|
@@ -362,6 +376,42 @@ The run verdict may include `WARN`-severity signals in addition to pass/fail. On
|
|
|
362
376
|
the run can still green. If you see it, fix the asset path — a green with a missing asset is
|
|
363
377
|
not a valid pass.
|
|
364
378
|
|
|
379
|
+
**False negatives — signals that are tier/image artifacts, not skill defects.** Some fail-severity
|
|
380
|
+
signals read like a skill gap but are really a property of the reduced test image or the fidelity tier.
|
|
381
|
+
Recognize these before "fixing" a non-bug:
|
|
382
|
+
|
|
383
|
+
- **`missing_capability`** — the lean `core` agent image is a deliberate partial mirror of real Cowork's
|
|
384
|
+
rootfs, so a skill that used `soffice`/LibreOffice (`office_convert`), `tesseract` (`ocr`),
|
|
385
|
+
`markitdown`/`magika` (`ml_extract`), `cv2` (`cv`), `camelot`/`tabula` (`pdf_tables`), or `wand`
|
|
386
|
+
(`magick`) can trip this even though real Cowork **ships** those. The message says so ("likely a FALSE
|
|
387
|
+
NEGATIVE (real Cowork ships them)"). Fix: rebuild full parity (`--build-arg COWORK_FULL_PARITY=1`, point
|
|
388
|
+
`COWORK_AGENT_IMAGE` at it), or — if the skill's fallback is genuinely equivalent — assert
|
|
389
|
+
`allow_missing_capability: true`. (Two sources: a skill *observed using* an omitted family, live lane;
|
|
390
|
+
or a declared `requires_capabilities` the tier can't provide, both lanes — an unknown family name
|
|
391
|
+
hard-fails rather than silently passing.) **On an open-ended `skill` run** (no `assert:` block to carry
|
|
392
|
+
the modifier), pass **`--allow-missing-capability`** — the CLI equivalent of the assertion.
|
|
393
|
+
- **`ended_with_question`** (`WARN`, live lane) — a heuristic: the agent's final answer contains a
|
|
394
|
+
question and the run wrote **no deliverable to `outputs/`** — it may have ended on a request for input
|
|
395
|
+
instead of finishing. Warn-only; the fix is scripting/steering the answer (`answer:` / `--answer` / a
|
|
396
|
+
decider, or `--decider-llm --intent`), not editing the skill's prose. The strict, fail-severity sibling
|
|
397
|
+
`stalled` already catches a *trailing*-`?` final turn that did no tool work after the last gate; this
|
|
398
|
+
covers the residual (a mid-message `?`, or tool work after the last gate that still ended asking). Read
|
|
399
|
+
the final message before acting — a legitimate question-posing answer that wrote a file never fires.
|
|
400
|
+
Assert `allow_stall: true` if ending on a question is the intended terminal state.
|
|
401
|
+
- **`host_path_leak`** — skipped at **`hostloop` and `protocol`** fidelity (the agent runs on real host
|
|
402
|
+
paths there, so a host path in model-visible text is expected, not a leak); it is *armed* at
|
|
403
|
+
`container`/`microvm`, but only *fires* on an actual scanned leak with no authored
|
|
404
|
+
`transcript_no_host_path`. At `fidelity: cowork` the skip follows the **resolved** tier, so a `cowork`
|
|
405
|
+
run that lands on `container` is armed. Author `transcript_no_host_path` to enforce cleanliness where
|
|
406
|
+
it's valid.
|
|
407
|
+
- **`scan_unavailable`** (`WARN`) — emitted only on the live lane: `events.jsonl` was missing/corrupt, so
|
|
408
|
+
`RunResult.scan` is undefined and the host-path + outputs-delete guards **did not run this run**. Not a
|
|
409
|
+
pass or a defect — assert `no_delete_in_outputs` / `transcript_no_host_path` to hard-fail on it instead.
|
|
410
|
+
|
|
411
|
+
The full 14-code signal table (severity + per-signal opt-out) is in
|
|
412
|
+
[`references/scenario-schema.md`](./references/scenario-schema.md); `docs/scenario.md` (repo-only) carries
|
|
413
|
+
the fuller narrative.
|
|
414
|
+
|
|
365
415
|
### Checking whether a background run is alive
|
|
366
416
|
|
|
367
417
|
Never use `ps aux` to check on a `cowork-harness` run you launched in the background — it only sees
|
|
@@ -374,7 +424,9 @@ harness writes/updates throughout the run's lifecycle (including a crash-safety
|
|
|
374
424
|
error/`SIGTERM`, AND staleness detection for a hard `SIGKILL`/OOM-kill that no exit handler can catch —
|
|
375
425
|
either way you get `"error"`/`stale` instead of a permanently-trusted `"running"`), so liveness is
|
|
376
426
|
checkable regardless of PID namespace. The harness prints `[status] <outDir>` to stderr as soon as the
|
|
377
|
-
run starts, so capture stderr to get the exact directory
|
|
427
|
+
run starts, so capture stderr to get the exact directory — but `<dir>` also accepts the run-dir root
|
|
428
|
+
passed to `--run-dir` (a directory without its own `status.json`): it scans up to two levels down for the
|
|
429
|
+
newest session's `status.json` and reads that. `--follow` fails loud on a timeout/staleness
|
|
378
430
|
rather than hanging forever. (Fuller recipe in `docs/run-status.md` — repo-only, not in the installed
|
|
379
431
|
payload; `cowork-harness status --help` has the flags.)
|
|
380
432
|
|
|
@@ -468,7 +520,11 @@ decide which assertions from *Assertions: two orthogonal axes* are worth adding)
|
|
|
468
520
|
- **Debugging a wrong Cowork UI panel.** Each panel is reconstructed in `result.json`: **Progress** =
|
|
469
521
|
`tasks[]`, **Working folder** = `workspaceFiles[]` (classified output/mount/input, with a
|
|
470
522
|
`trace --view files` diff), **Context / Connectors** = `context` (tools / mcpServers / availableSkills),
|
|
471
|
-
**Scratch-pad → outputs** = `presentedFiles[]`. If a panel looks wrong in a run, read its field.
|
|
523
|
+
**Scratch-pad → outputs** = `presentedFiles[]`. If a panel looks wrong in a run, read its field. An
|
|
524
|
+
**absent** `workspaceFiles`/`artifacts` (a replay result, or a run whose workspace root was missing at
|
|
525
|
+
collection) is evidence **UNAVAILABLE**, not an empty run — `trace --view files` reports a loud
|
|
526
|
+
UNAVAILABLE marker (`workspaceFilesRecorded: false` in JSON, and no phantom "removed" diff rows) and
|
|
527
|
+
`inspect` prints `artifacts: UNAVAILABLE` (`artifactsRecorded: false`) instead of `artifacts (0):`.
|
|
472
528
|
|
|
473
529
|
### Debugging with `chat`
|
|
474
530
|
|
|
@@ -611,9 +667,11 @@ repeats the assertion/replay-relevant ones alongside the schema (a scoped subset
|
|
|
611
667
|
- `on_unanswered` governs **unanswered** `AskUserQuestion` gates; the `stalled` signal covers
|
|
612
668
|
stalling *after* one is answered — two different failure modes.
|
|
613
669
|
- **Free-text aside:** a "type-it-in-notes" option has **no scripted deterministic answer** today
|
|
614
|
-
(the `OTHER:` directive works only on the LLM-decider path, not scripted `choose
|
|
615
|
-
|
|
616
|
-
|
|
670
|
+
(the `OTHER:` directive works only on the LLM-decider path, not scripted `choose:`, and only on
|
|
671
|
+
**single-select** gates — a **multi-select** gate is index-only, so `OTHER:` fails loud there; on an
|
|
672
|
+
options-bearing single-select gate a bare out-of-set LLM answer also fails loud (exit 2) — see the
|
|
673
|
+
LLM-decider free-text note in `references/fidelity-and-answers.md`). An LLM decision answered via
|
|
674
|
+
`OTHER:` is marked `[via Other free-text]` in its `gateProvenance` rationale.
|
|
617
675
|
14. **A positional `choose` (`first` / index) is order-dependent.** `choose: "2"` survives label drift
|
|
618
676
|
but NOT option *re-ordering* — if the gate presents its options in a different order run-to-run, the
|
|
619
677
|
index lands on a different option (a silent re-record flake). Prefer an exact label when order is
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# CI recipe — replay vs live lanes
|
|
2
2
|
|
|
3
|
-
Self-contained reference. Tracks `cowork-harness 1.
|
|
3
|
+
Self-contained reference. Tracks `cowork-harness 1.3.0` (baseline `desktop-1.20186.1`).
|
|
4
4
|
|
|
5
5
|
**Fastest path: the packaged Action.** One step gets you `replay`/`lint`/`verify-cassettes` plus a PR
|
|
6
6
|
job-summary reporter (verdict table, staleness findings, cost/turns when available):
|
|
@@ -13,7 +13,7 @@ job-summary reporter (verdict table, staleness findings, cost/turns when availab
|
|
|
13
13
|
```
|
|
14
14
|
|
|
15
15
|
The Action's `version` input defaults to `latest` — intentional so a copy-pasted recipe tracks the current
|
|
16
|
-
release; pin an exact version (e.g. `version: "1.
|
|
16
|
+
release; pin an exact version (e.g. `version: "1.3.0"`) for reproducible CI.
|
|
17
17
|
|
|
18
18
|
Reach for the manual multi-step form below only when you need per-step control the Action's inputs don't
|
|
19
19
|
cover (a custom flag combination, a different runner matrix per step, or `lint`/`verify-cassettes` gated
|
|
@@ -57,7 +57,7 @@ sha256-*checked* but not hard-blocking on mismatch — it's advisory for an inte
|
|
|
57
57
|
GitHub-hosted runners, no token/Docker/agent:
|
|
58
58
|
|
|
59
59
|
```yaml
|
|
60
|
-
- run: npm i -g "cowork-harness@>=1.
|
|
60
|
+
- run: npm i -g "cowork-harness@>=1.3.0"
|
|
61
61
|
- run: cowork-harness lint scenarios/*.yaml # no silent false-greens
|
|
62
62
|
- run: cowork-harness verify-cassettes cassettes/ # privacy + staleness
|
|
63
63
|
- run: cowork-harness replay cassettes/ # token-free content/structure
|
|
@@ -197,7 +197,7 @@ jobs:
|
|
|
197
197
|
with: { node-version: '20' }
|
|
198
198
|
- uses: actions/setup-python@v5
|
|
199
199
|
with: { python-version: '3.x' } # python3 only — PyYAML is bundled with the linter
|
|
200
|
-
- run: npm i -g "cowork-harness@>=1.
|
|
200
|
+
- run: npm i -g "cowork-harness@>=1.3.0"
|
|
201
201
|
- run: cowork-harness lint scenarios/*.yaml # no-silent-false-green (needs python3; PyYAML bundled)
|
|
202
202
|
- run: cowork-harness verify-cassettes cassettes/ --output-format json # privacy + staleness gate
|
|
203
203
|
- run: cowork-harness replay cassettes/ --output-format json # token-free content/structure
|
|
@@ -226,7 +226,7 @@ jobs:
|
|
|
226
226
|
echo "live=true" >> "$GITHUB_OUTPUT"
|
|
227
227
|
fi
|
|
228
228
|
- if: steps.guard.outputs.live == 'true'
|
|
229
|
-
run: npm i -g "cowork-harness@>=1.
|
|
229
|
+
run: npm i -g "cowork-harness@>=1.3.0"
|
|
230
230
|
- if: steps.guard.outputs.live == 'true'
|
|
231
231
|
run: cowork-harness run scenarios/ --output-format json
|
|
232
232
|
env:
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# Fidelity tiers & answer paths
|
|
2
2
|
|
|
3
|
-
Self-contained reference. Tracks `cowork-harness 1.
|
|
3
|
+
Self-contained reference. Tracks `cowork-harness 1.3.0` (baseline `desktop-1.20186.1`).
|
|
4
4
|
|
|
5
5
|
## Fidelity tiers (`fidelity:` in the scenario)
|
|
6
6
|
|
|
@@ -130,7 +130,10 @@ hands `fn` exactly this dict.
|
|
|
130
130
|
### Determinism contract
|
|
131
131
|
|
|
132
132
|
- `fail` — the default for `run`. On an unscripted gate it hard-errors; the error names the exact
|
|
133
|
-
`--answer`/`choose` to add
|
|
133
|
+
`--answer`/`choose` to add, and also suggests `on_unanswered: llm` (in the scenario YAML) as a
|
|
134
|
+
secondary escape valve for a gate whose wording drifts run-to-run — a regex chases a moving target,
|
|
135
|
+
at the cost of non-determinism (one model call per gate). Correct, but flaky for skills whose gates
|
|
136
|
+
appear stochastically.
|
|
134
137
|
- `first` — picks option 1 and warns loudly. **Flagged `nonDeterministic`** — not a deterministic
|
|
135
138
|
substitute for scripted answers. (For a web_fetch approval gate it abstains → fail-closed.)
|
|
136
139
|
- `prompt` — asks at the TTY (`skill` only).
|
|
@@ -161,10 +164,13 @@ Caveat: `decide` only builds a **single-select** sample (set choices with `--opt
|
|
|
161
164
|
multiSelect flag), so its printed request shows `options[].label` but never `multiSelect:true` — to
|
|
162
165
|
exercise the array reply path, run a real multiSelect gate or unit-test the helper directly.
|
|
163
166
|
|
|
164
|
-
LLM-decider free-text goes via `OTHER: <value>` on an **options-bearing** gate; a bare
|
|
165
|
-
answer (no matching label, no `OTHER:`) fails loud (`UnansweredError` → exit 2) — it never
|
|
166
|
-
guesses an option.
|
|
167
|
-
|
|
167
|
+
LLM-decider free-text goes via `OTHER: <value>` on an **options-bearing single-select** gate; a bare
|
|
168
|
+
out-of-set answer (no matching label, no `OTHER:`) fails loud (`UnansweredError` → exit 2) — it never
|
|
169
|
+
stalls or guesses an option. A **multi-select** gate is **index-only**: it accepts comma-separated option
|
|
170
|
+
numbers, and `OTHER:` fails loud there (no free-text escape on that path). Open-ended (no-option) gates
|
|
171
|
+
need no `OTHER:` prefix: free text is delivered verbatim. A decision answered via any free-text path is
|
|
172
|
+
marked `[via Other free-text]` in its `gateProvenance` rationale, so a `result.json` consumer can tell it
|
|
173
|
+
from an offered-option pick. (Scripted scenarios use the separate `answer:` escape hatch.)
|
|
168
174
|
|
|
169
175
|
A gate that fails loud (`on_unanswered: fail`, the default) still **salvages a PARTIAL run**: the harness
|
|
170
176
|
writes a `result.json` (marked `partial: true`) with the artifacts the agent produced before the whiff, so
|
|
@@ -180,7 +186,7 @@ up often enough to spell out:
|
|
|
180
186
|
a filesystem or network — it re-evaluates assertions from the frozen cassette. A fixed set of
|
|
181
187
|
keys is live-only and **skipped outright** on replay (absent from `assertions[]`, not vacuously
|
|
182
188
|
passed): `no_delete_in_outputs`, `self_heal_ran`, `transcript_no_host_path`, `egress_denied`,
|
|
183
|
-
`egress_allowed`, `no_mcp_error`, `max_peak_rss_bytes`, `semantic_matches`, and `expect_denied`.
|
|
189
|
+
`egress_allowed`, `no_mcp_error`, `max_peak_rss_bytes`, `semantic_matches`, `no_lost_write_back`, and `expect_denied`.
|
|
184
190
|
Everything else that *is* evaluated is checked against the **recording**, not fresh behavior — a
|
|
185
191
|
green replay says the skill produced these events when it was recorded, not that it still does
|
|
186
192
|
(`staleness[]` flags skill/baseline drift as a hint; only a live `run` re-confirms current
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# Scenario & session schema, assertion catalog, web_fetch, full gotchas
|
|
2
2
|
|
|
3
|
-
Self-contained reference for authoring `cowork-harness` scenarios. Tracks `cowork-harness 1.
|
|
3
|
+
Self-contained reference for authoring `cowork-harness` scenarios. Tracks `cowork-harness 1.3.0`
|
|
4
4
|
(baseline `desktop-1.20186.1`). If your checkout is newer, prefer the live `docs/scenario.md`,
|
|
5
5
|
`docs/session.md`, and `SPEC.md`.
|
|
6
6
|
|
|
@@ -251,13 +251,16 @@ same set live from the schema.
|
|
|
251
251
|
| `file_exists: <path>` | the path exists under the run's `work/` (anchored at `mnt/`, e.g. `outputs/x.md`). For a user-facing deliverable prefer `user_visible_artifact` — with a connected folder the file lands in `mnt/<folder>` (= `{{workspaceFolder}}`), not `mnt/outputs`, so `file_exists: outputs/x.md` misses it |
|
|
252
252
|
| `user_visible_artifact: <path>` | exists **and** under a user-visible root (`outputs/` + each connected folder's mount name) — the right primitive for a workspace deliverable when a folder is connected |
|
|
253
253
|
| `no_delete_in_outputs: true` | no delete op touched `mnt/outputs` — **only `true` is valid**; `false` is rejected (omit to allow deletes) |
|
|
254
|
-
| `no_unexpected_files: [<glob>, …]` | every **newly created** file under a user-visible root matches ≥1 glob (workRoot-relative paths; `**` = whole path segment for any depth — use `outputs/handoff/**` for per-run subdirs); `[]` = no new files; **new-files-only** — overwriting a pre-existing file in place is invisible (use content-level producer stamping); live/verify-run without a pre-run manifest ⇒ evidence-unavailable hard-fail (live runs capture the baseline only when this key is asserted; recordings always capture); an **incomplete post-run filesystem walk** (an unreadable subtree — a permission/I-O error) also fails evidence-unavailable rather than reporting "no strays" over a partial tree — distinct from the missing-manifest
|
|
255
|
-
| `input_unmodified: <glob>` or `[<glob>, …]` | a single glob or a list; every **pre-existing** file (incl. uploaded files under `uploads/**`) whose workRoot-relative path matches ≥1 glob keeps an unchanged content hash after the run — the in-place-mutation companion to `no_unexpected_files`'s new-files check (`[]` is rejected by the schema — list at least one glob); a matched file that was deleted counts as a content change (fails); live/verify-run without a pre-run hash manifest ⇒ evidence-unavailable hard-fail;
|
|
254
|
+
| `no_unexpected_files: [<glob>, …]` | every **newly created** file under a user-visible root matches ≥1 glob (workRoot-relative paths; `**` = whole path segment for any depth — use `outputs/handoff/**` for per-run subdirs); `[]` = no new files; **new-files-only** — overwriting a pre-existing file in place is invisible (use content-level producer stamping); live/verify-run without a pre-run manifest ⇒ evidence-unavailable hard-fail (live runs capture the baseline only when this key is asserted; recordings always capture); an **incomplete post-run filesystem walk** (an unreadable subtree — a permission/I-O error) also fails evidence-unavailable rather than reporting "no strays" over a partial tree — distinct from the missing-manifest case (a `--resume` run), which fails for a different reason; captured on every live sandbox tier including microvm (its outputs are snapshotted from the VM into the run dir); replay needs `cassette.preRunPaths` (≥0.24 recordings) — cassettes without it **exclude** the key with a loud warning |
|
|
255
|
+
| `input_unmodified: <glob>` or `[<glob>, …]` | a single glob or a list; every **pre-existing** file (incl. uploaded files under `uploads/**`) whose workRoot-relative path matches ≥1 glob keeps an unchanged content hash after the run — the in-place-mutation companion to `no_unexpected_files`'s new-files check (`[]` is rejected by the schema — list at least one glob); a glob that matches **no** pre-run path fails loud (a typo or renamed mount would otherwise verify zero files and pass vacuously); a matched file that was deleted counts as a content change (fails); live/verify-run without a pre-run hash manifest ⇒ evidence-unavailable hard-fail (a `--resume` run); captured on every live sandbox tier including microvm; replay needs `cassette.preRunHashes` — cassettes without it **exclude** the key with a loud warning; on replay it compares against the manifest's recorded `sha256`, never a re-hash of the materialized tree |
|
|
256
256
|
| `self_heal_ran: <bool>` | a plugin-root self-heal script was (not) invoked |
|
|
257
|
+
| `no_lost_write_back: true` | fails if the run authored an interactive HTML artifact (or a `.py`/`.js` generator of one) whose **relative** Submit/POST write-back is lost under Cowork (served from Cowork's own origin → resolves non-ok, a "Saved!" is silently false). Runs the shipped **static Tier A** analyzer over the files the run authored (diffed vs the pre-run manifest). A lost write-back on an **added** agent-authored source (`outputs/`, scratchpad) **fails**; a **pre-existing** file the skill only modified on a read-write mount is **advisory**; `-suspect` findings surface but pass. **Only `true` is valid**. **Live/verify-run only** — skipped-loud on replay; runs on every live sandbox tier including **microvm** (its outputs are snapshotted from the VM into the run dir); could-not-verify (fail-closed) on a `--resume` scratchpad or an unanalyzable candidate |
|
|
257
258
|
| `tool_called: <glob>` | a tool the agent ran matched this **glob** — `*` = any run, `?` = one char, exact when literal, anchored + case-sensitive. Exact name (`Write`) matches only that tool; `mcp__workspace__*` matches any workspace tool. GLOB, not regex (`.` is literal) — an empty glob, or one containing a regex/brace-expansion metacharacter (`.*`, `.+`, `\|`, `()`, `[]`, `+`, `^`, `$`, `{}`, `\d`/`\w`/`\s`/`\b`), is **rejected at load** (a hard schema error, not a runtime warning) — it would match no real tool name and pass a `_not_`/`_absent` assert vacuously. Applies whether the glob comes from an authored scenario or a recorded cassette's frozen assert. The bundled `scenario.py lint` does NOT perform this check — only the harness enforces it, at actual load (`run`/`skill`/`record`) |
|
|
258
259
|
| `tool_not_called: <glob>` | NO tool the agent ran matched this glob (`mcp__*` = "no MCP tool ran"). Same glob semantics as `tool_called`, including the empty/regex-ish rejection |
|
|
259
260
|
| `tool_result_contains: <str>` | a tool result includes the literal string (content / replay-checkable — substring match) |
|
|
260
261
|
| `tool_result_not_contains: <str>` | no tool result includes the literal string (content / replay-checkable; fails loud when tool results are absent) |
|
|
262
|
+
| `tool_result_matches: <regex>` | the regex sibling of `tool_result_contains` — case-insensitive regex matches at least one tool result; use for an error-signature family, not just one literal string |
|
|
263
|
+
| `tool_result_not_matches: <regex>` | the regex sibling of `tool_result_not_contains` — same fails-loud-on-absent-evidence semantics |
|
|
261
264
|
| `subagent_tool_used: <glob>` | a sub-agent used a tool matching this glob (same `*`/`?`, anchored, case-sensitive semantics as `tool_called`, including the empty/regex-ish rejection) |
|
|
262
265
|
| `subagent_tool_absent: <glob>` | no sub-agent used a tool matching this glob (same rejection) |
|
|
263
266
|
| `no_vm_path_file_op: true` | **`fidelity: hostloop` only** — NO gated file tool attempted a `/sessions`(-prefixed) path (`RunResult.fileToolAttempts`) — content-class, replay-checkable without `controlOut`; any other tier FAILS "cannot verify" (`/sessions/...` is valid there). **Only `true` is valid** |
|
|
@@ -312,7 +315,7 @@ same set live from the schema.
|
|
|
312
315
|
| `egress_allowed: <host>` | the host was allowed through |
|
|
313
316
|
| `no_mcp_error: true` | no MCP round-trip failed (`RunResult.mcpErrors` is empty — no unhandled server, no handler throw) — live-only: MCP round-trips are harness-computed, not in the SDK stdout stream, so evidence-unavailable on replay (never a vacuous pass). **Only `true` is valid** |
|
|
314
317
|
| `max_peak_rss_bytes: <N>` | peak sampled RSS of the agent sandbox ≤ N bytes (`RunResult.resources.peakRssBytes`) — live-only: replay never spawns a sandbox to sample, so evidence-unavailable on replay/protocol (never a vacuous pass); also evidence-unavailable when sampling captured no RSS value |
|
|
315
|
-
| `semantic_matches: {rubric: [...], min_pass?, judge_model?}` | a pinned LLM judge grades each fixed `rubric` claim against the run's answer — the **union of the agent's final result text (`RunResult.finalMessage`), the transcript, and the final on-disk content of any files the agent authored during the run** — so a claim about content the skill led the agent to *write to a file* grades as reliably as one about inlined prose (authored-file evidence is
|
|
318
|
+
| `semantic_matches: {rubric: [...], min_pass?, judge_model?}` | a pinned LLM judge grades each fixed `rubric` claim against the run's answer — the **union of the agent's final result text (`RunResult.finalMessage`), the transcript, and the final on-disk content of any files the agent authored during the run** — so a claim about content the skill led the agent to *write to a file* grades as reliably as one about inlined prose (authored-file evidence is captured on every live sandbox tier including **microvm** — its session tree is snapshotted from the VM into the run dir). When the authored-file evidence backing the judged document is **incomplete** — a file dropped at the capture-size cap, unreadable at read-back, or (on `--resume`) the scratchpad walk skipped — the assert fails evidence-unavailable rather than trusting a judge grade over a partial document; this is separate from the malformed-grade `judgeInvalid` path below. The assert passes iff ≥ `min_pass` claims pass (default: all — avoid for a gating scenario). Results align by claim index and are recorded per-claim in `RunResult.assertions[].semanticClaims` (`[{index, claim, pass}]`, so a consumer can diff the per-claim profile across runs); a rep whose grade can't be parsed (after one retry) is marked `RunResult.assertions[].judgeInvalid` and **never silently dropped** — it is excluded from the pass denominator, and the guard against a misleading score from that exclusion is the gate's minimum-valid-rep floor (`MIN_VALID` ≥ 4) plus this visibility, not a claim that denominator-shrinking inflation is impossible. Within a rep, a grade that's still unparseable after the retry **fails that assert outright** (evidence-unavailable, not a vacuous pass) — a persistently-flaky judge reds the run rather than silently passing. `judge_model` pins the grader (default when neither it nor `COWORK_HARNESS_JUDGE_MODEL` is set: `claude-opus-4-8`; a dated id keeps a before/after comparison reproducible). Live-only: the judge is a live model call, so evidence-unavailable / skipped-loud on replay (never a vacuous pass) |
|
|
316
319
|
| `artifact_json: {artifact, path, …}` | assert a JSON artifact's contents — `equals`/`gt`/`in`/`exists`/`absent`/`is_null` over a dotted `path` (`in` = membership in a list, for a stochastic/LLM value; `absent` ≠ `is_null`; an unresolved intermediate fails loud) |
|
|
317
320
|
| `computer_links_resolve: true` | every `computer://` link in the model-visible transcript resolves to an artifact that exists in the run's collected outputs/mounts — a dangling link fails, naming which target was checked (a live host path, the collected work tree, or the replay manifest). **Requires ≥1 link** (zero links fails — use `computer_links_resolve_if_present` for the presence-free variant). **Only `true` is valid** (`false` is rejected by the schema) |
|
|
318
321
|
| `computer_links_resolve_if_present: true` | like `computer_links_resolve` but passes vacuously when the transcript has zero `computer://` links — the presence-free variant. **Only `true` is valid** |
|
|
@@ -331,7 +334,7 @@ dotted path.
|
|
|
331
334
|
|
|
332
335
|
**VerdictSignals in `result.verdict.signals`:** `computeVerdict` pushes signals into `result.verdict.signals`; most
|
|
333
336
|
are **fail**-severity (they flip the run's pass/exit code even though `result.result` itself stays
|
|
334
|
-
`"success"`) and only
|
|
337
|
+
`"success"`) and only four are **warn**-severity (informational, never flip pass/fail). Current signal
|
|
335
338
|
codes (`VerdictSignal["code"]` in `src/run/verdict.ts`):
|
|
336
339
|
|
|
337
340
|
| Code | Severity | Meaning |
|
|
@@ -344,16 +347,17 @@ codes (`VerdictSignal["code"]` in `src/run/verdict.ts`):
|
|
|
344
347
|
| `outputs_delete` | fail | An unauthorized delete touched `mnt/outputs` (opt out: author `no_delete_in_outputs`) |
|
|
345
348
|
| `host_path_leak` | fail | A host path leaked into model-visible text (opt out: author `transcript_no_host_path`) |
|
|
346
349
|
| `l0_plugin_divergence` | fail | L0/protocol plugin loading diverged from Cowork (opt out: `allow_l0_plugin_divergence`) |
|
|
347
|
-
| `missing_capability` | fail | A `requires_capabilities` need was unmet, or the skill used a capability the image omits (opt out: `allow_missing_capability`) |
|
|
350
|
+
| `missing_capability` | fail | A `requires_capabilities` need was unmet, or the skill used a capability the image omits (opt out: `allow_missing_capability`, or `skill --allow-missing-capability` on an open-ended run) |
|
|
348
351
|
| `infra_error` | fail | A VM/egress sidecar crashed mid-run — not author-suppressible |
|
|
349
|
-
| `stalled` | fail | The run ended on an unanswered question (opt out: `allow_stall`) |
|
|
352
|
+
| `stalled` | fail | The run ended on an unanswered question, or on a trailing-`?` final turn with no tool work after the last gate (opt out: `allow_stall`) |
|
|
350
353
|
| `non_deterministic` | warn | The run was LLM/external/human-decided — not reproducible |
|
|
351
354
|
| `prompt_asset_missing` | warn | The run proceeded with a missing prompt asset (`COWORK_HARNESS_ALLOW_MISSING_PROMPT=1`); fidelity is degraded |
|
|
352
355
|
| `scan_unavailable` | warn | Post-run scan evidence unavailable (`RunResult.scan` undefined) — the host-path and outputs-delete guards did not run this run |
|
|
356
|
+
| `ended_with_question` | warn | Live-lane heuristic: the final answer contains a question and the run wrote no deliverable to `outputs/` — the lenient sibling of `stalled` (covers a mid-message `?`, or tool work after the last gate that still ended asking). Opt out: `allow_stall` |
|
|
353
357
|
|
|
354
358
|
A **fail**-severity signal does not change `result.result` (still `"success"`), but it DOES fail the
|
|
355
359
|
overall run verdict and exit code — `assert result: success` alone won't catch it; check
|
|
356
|
-
`result.verdict.signals[].severity` or the run's exit code. Only the
|
|
360
|
+
`result.verdict.signals[].severity` or the run's exit code. Only the four **warn** codes are truly benign.
|
|
357
361
|
|
|
358
362
|
## Replay class
|
|
359
363
|
|
|
@@ -407,8 +411,8 @@ staleness `fingerprint` shows ANY skill/baseline drift, or `replay --fail-on-ski
|
|
|
407
411
|
skill-source drift; every replay result also reports it class-tagged in `staleness[]` for a JSON gate.
|
|
408
412
|
|
|
409
413
|
**Egress + other filesystem — still skipped on replay (live-only):** `no_delete_in_outputs`,
|
|
410
|
-
`self_heal_ran`, `transcript_no_host_path`, `egress_*` / `expect_denied`, `no_mcp_error`, `max_peak_rss_bytes
|
|
411
|
-
These run only on a live `run`/`record`.
|
|
414
|
+
`self_heal_ran`, `transcript_no_host_path`, `egress_*` / `expect_denied`, `no_mcp_error`, `max_peak_rss_bytes`,
|
|
415
|
+
`no_lost_write_back`. These run only on a live `run`/`record`.
|
|
412
416
|
|
|
413
417
|
**Mixed assertions on replay:** before evaluating, `replay` strips each assertion to its replay-checkable
|
|
414
418
|
keys and drops any left empty. So `{result, egress_denied}` evaluates on replay as `{result}` alone — its
|
|
@@ -163,3 +163,30 @@ degrade the advice. It is real work to calibrate; these steps are the traps that
|
|
|
163
163
|
**Lane note:** `semantic_matches` is **live-only** (the judge is a live model call), so these scenarios
|
|
164
164
|
run on the `run` lane, never token-free `replay` — the linter's "all assertions live-only" warning is
|
|
165
165
|
expected and correct here.
|
|
166
|
+
|
|
167
|
+
## Recipe 6 — Iterate a skill across fixes (ground findings, don't cross-pair generations)
|
|
168
|
+
|
|
169
|
+
Hardening a skill is a loop: run → read what it did → fix → run again. Two disciplines keep it honest.
|
|
170
|
+
|
|
171
|
+
1. **Verify before you trust.** A green run is not a correct run, and a skill's self-reported finding (a
|
|
172
|
+
self-critique appendix, "I extracted X") is not real until its cited evidence is found in the run's own
|
|
173
|
+
output. The harness emits the substrate; the grader is yours (it lives outside the harness):
|
|
174
|
+
- `result.json` → `finalMessage` (the skill's own answer/critique) + `toolResults[]` (tool outputs).
|
|
175
|
+
- `cowork-harness trace <run-dir> --output-format json` → the tool-call stream. Add `--full-results` so
|
|
176
|
+
a **successful** call's full input + result are captured (the default view slices them to ~100/120
|
|
177
|
+
chars) — this is what lets your grader confirm "the skill claims it read X and derived Y" against the
|
|
178
|
+
actual call.
|
|
179
|
+
- `cowork-harness inspect <run-dir>` → what the run produced, plus the run's `label` and `skillHash`.
|
|
180
|
+
- In-run alternative: dispatch a checker **sub-agent** (maker/checker) whose result folds into the
|
|
181
|
+
verdict.
|
|
182
|
+
2. **Don't cross-pair generations.** When you run the same skill across fixes, never pair a *pre-fix*
|
|
183
|
+
`result.json` with a *post-fix* critique. The authoritative version key is `fingerprint.skillHash` —
|
|
184
|
+
content-exact, on every live run, changes on any tracked edit. **Group/pair on it** (`inspect` and the
|
|
185
|
+
run-index row surface a short prefix). Add `--label <tag>` for a human-readable generation name
|
|
186
|
+
(skillHash is the correctness key; the label is ergonomics). `cowork-harness verify-run <run-dir>
|
|
187
|
+
<scenario.yaml>` is the native staleness guard: it **warns** when a kept run predates the current
|
|
188
|
+
skill, and with scripted `answers` **hard-fails** rather than vouch for a stale gate snapshot.
|
|
189
|
+
|
|
190
|
+
**Lane note:** the exploratory driver is `skill <dir> --decider-llm --intent "<what this run tests>"`,
|
|
191
|
+
which is flagged non-deterministic (a green here is exploration, not a scripted pass) — pin the
|
|
192
|
+
load-bearing gates with `--answer` once you know which fire.
|
|
@@ -27,6 +27,7 @@
|
|
|
27
27
|
"max_turns",
|
|
28
28
|
"no_delete_in_outputs",
|
|
29
29
|
"no_hook_blocked",
|
|
30
|
+
"no_lost_write_back",
|
|
30
31
|
"no_mcp_error",
|
|
31
32
|
"no_path_denied",
|
|
32
33
|
"no_scratchpad_leak",
|
|
@@ -60,7 +61,9 @@
|
|
|
60
61
|
"tool_no_error_if_called",
|
|
61
62
|
"tool_not_called",
|
|
62
63
|
"tool_result_contains",
|
|
64
|
+
"tool_result_matches",
|
|
63
65
|
"tool_result_not_contains",
|
|
66
|
+
"tool_result_not_matches",
|
|
64
67
|
"transcript_contains",
|
|
65
68
|
"transcript_matches",
|
|
66
69
|
"transcript_no_host_path",
|
|
@@ -62,6 +62,8 @@ CONTENT_KEYS = {
|
|
|
62
62
|
"transcript_not_matches",
|
|
63
63
|
"tool_result_contains",
|
|
64
64
|
"tool_result_not_contains",
|
|
65
|
+
"tool_result_matches",
|
|
66
|
+
"tool_result_not_matches",
|
|
65
67
|
"tool_called",
|
|
66
68
|
"tool_not_called",
|
|
67
69
|
"subagent_tool_used",
|
|
@@ -129,6 +131,7 @@ LIVE_ONLY_KEYS = {
|
|
|
129
131
|
"no_mcp_error",
|
|
130
132
|
"max_peak_rss_bytes",
|
|
131
133
|
"semantic_matches",
|
|
134
|
+
"no_lost_write_back",
|
|
132
135
|
}
|
|
133
136
|
EGRESS_KEYS = {"egress_denied", "egress_allowed"}
|
|
134
137
|
# container-only: served only at fidelity: container (present_files / the scratchpad promotion path
|
|
@@ -229,6 +232,8 @@ REGEX_KEYS = {
|
|
|
229
232
|
"subagent_dispatched",
|
|
230
233
|
"question_asked",
|
|
231
234
|
"hook_blocked",
|
|
235
|
+
"tool_result_matches",
|
|
236
|
+
"tool_result_not_matches",
|
|
232
237
|
}
|
|
233
238
|
VALID_ON_UNANSWERED = {"fail", "prompt", "first", "llm"}
|
|
234
239
|
VALID_TIERS = ("protocol", "container", "microvm", "hostloop", "cowork")
|