cowork-harness 4.1.1 → 4.2.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude/skills/cowork-harness/SKILL.md +14 -10
- package/.claude/skills/cowork-harness/references/assertion-catalog.md +4 -4
- package/.claude/skills/cowork-harness/references/assertions-guide.md +2 -2
- package/.claude/skills/cowork-harness/references/authoring.md +2 -1
- package/.claude/skills/cowork-harness/references/ci-recipe.md +5 -5
- package/.claude/skills/cowork-harness/references/critique.md +1 -1
- package/.claude/skills/cowork-harness/references/debugging.md +1 -1
- package/.claude/skills/cowork-harness/references/eval.md +79 -0
- package/.claude/skills/cowork-harness/references/fidelity-and-answers.md +1 -1
- package/.claude/skills/cowork-harness/references/gotchas.md +15 -2
- package/.claude/skills/cowork-harness/references/measurement.md +6 -1
- package/.claude/skills/cowork-harness/references/run-record-replay.md +2 -2
- package/.claude/skills/cowork-harness/references/scenario-schema.md +2 -2
- package/.claude/skills/cowork-harness/references/task-recipes.md +24 -10
- package/.claude/skills/cowork-harness/scripts/scenario.py +56 -1
- package/CHANGELOG.md +346 -0
- package/DESIGN.md +2 -2
- package/README.md +11 -5
- package/SPEC.md +12 -4
- package/baselines/desktop-2.16120.0.json +1148 -0
- package/baselines/prompts/cowork-system-prompt-fingerprints.json +10 -1
- package/dist/agent/session.js +26 -0
- package/dist/assert.js +102 -13
- package/dist/baseline.js +7 -0
- package/dist/cli.js +150 -366
- package/dist/critique/command.js +27 -10
- package/dist/critique/evaluator.js +2 -1
- package/dist/critique/skill-invocation.js +61 -1
- package/dist/decide/decider.js +25 -3
- package/dist/decide/llm-transport.js +163 -6
- package/dist/decide/semantic-judge.js +170 -37
- package/dist/decide/usage.js +52 -0
- package/dist/eval/classify.js +308 -0
- package/dist/eval/command.js +591 -0
- package/dist/eval/invocation.js +44 -0
- package/dist/eval/job-runner.js +50 -0
- package/dist/eval/manifest.js +11 -0
- package/dist/eval/pins.js +45 -0
- package/dist/eval/report.js +479 -0
- package/dist/eval/runs.js +127 -0
- package/dist/eval/schedule.js +31 -0
- package/dist/eval/snapshot.js +328 -0
- package/dist/eval/stats.js +240 -0
- package/dist/eval/usage.js +62 -0
- package/dist/hillclimb/schema-check.js +657 -0
- package/dist/run/api-retries.js +31 -0
- package/dist/run/artifacts.js +5 -4
- package/dist/run/authored-capture-opts.js +23 -0
- package/dist/run/cassette.js +149 -12
- package/dist/run/chat-result.js +9 -1
- package/dist/run/chat.js +125 -63
- package/dist/run/command-globals.js +15 -2
- package/dist/run/doctor.js +62 -45
- package/dist/run/execute.js +206 -62
- package/dist/run/model-provenance.js +50 -3
- package/dist/run/provenance.js +29 -11
- package/dist/run/renderer.js +15 -1
- package/dist/run/run-index.js +6 -0
- package/dist/run/run.js +8 -0
- package/dist/run/runs-gc.js +26 -1
- package/dist/run/verify-context.js +403 -0
- package/dist/runtime/agent-tree.js +480 -0
- package/dist/runtime/hostloop.js +6 -5
- package/dist/runtime/protocol.js +6 -1
- package/dist/sync/cowork-sync.js +266 -5
- package/dist/termination.js +76 -10
- package/dist/types.js +2 -2
- package/docs/README.md +2 -1
- package/docs/boundary.md +7 -0
- package/docs/cassette.md +13 -6
- package/docs/chat.md +7 -1
- package/docs/ci.md +25 -1
- package/docs/cli.md +37 -12
- package/docs/companion-skill.md +2 -2
- package/docs/critique.md +8 -3
- package/docs/debugging.md +12 -4
- package/docs/eval.md +244 -0
- package/docs/fidelity-gaps.md +78 -9
- package/docs/run-status.md +7 -3
- package/docs/scenario.md +6 -6
- package/docs/session.md +1 -1
- package/docs/stats.md +15 -3
- package/examples/replays/README.md +1 -1
- package/examples/replays/example-multiselect-gate.cassette.json +1 -1
- package/examples/replays/example-pdf-skill.cassette.json +1 -1
- package/examples/replays/hostloop-computer-links.cassette.json +1 -1
- package/llms.txt +3 -2
- package/package.json +1 -1
- package/python/test_scenario_lint.py +71 -0
- package/schema/run-result.json +77 -1
- package/schema/scenario.schema.json +2 -2
|
@@ -1,10 +1,10 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: cowork-harness
|
|
3
|
-
description: Test or debug a Claude Code skill/plugin under Claude Cowork's runtime — sandboxed agent, default-deny egress, the can_use_tool permission/question protocol — using the cowork-harness CLI. Use when validating or regression-testing a skill, authoring or debugging a scenario YAML (prompt + scripted answers + assert:), choosing a fidelity tier, scripting AskUserQuestion / tool-permission answers, asserting artifacts, egress, or sub-agent dispatch, measuring how long each tool call took (toolDurations / trace), or debugging a failed run or verdict from its result.json or transcript. Especially when a harness run no-ops an assertion, fails on an unanswered gate, false-greens, a steered answer never reaches the model, or a web_fetch is unexpectedly denied or gated. Also when iterating or hardening a skill across fixes, or grounding a skill's self-critique against its own run evidence — including a document-analysis skill (cap table, deck, financial model, transcript) that needs an uploaded file attached to be critiqued at all. NOT for generic unit testing (pytest/vitest of your own scripts) or non-Cowork CI. Covers the skill / run / chat / record / replay / trace / decide / assertions / scaffold commands and the session-vs-scenario split.
|
|
3
|
+
description: Test or debug a Claude Code skill/plugin under Claude Cowork's runtime — sandboxed agent, default-deny egress, the can_use_tool permission/question protocol — using the cowork-harness CLI. Use when validating or regression-testing a skill, authoring or debugging a scenario YAML (prompt + scripted answers + assert:), choosing a fidelity tier, scripting AskUserQuestion / tool-permission answers, asserting artifacts, egress, or sub-agent dispatch, measuring how long each tool call took (toolDurations / trace), or debugging a failed run or verdict from its result.json or transcript. Especially when a harness run no-ops an assertion, fails on an unanswered gate, false-greens, a steered answer never reaches the model, or a web_fetch is unexpectedly denied or gated. Also when iterating or hardening a skill across fixes, or grounding a skill's self-critique against its own run evidence — including a document-analysis skill (cap table, deck, financial model, transcript) that needs an uploaded file attached to be critiqued at all. Also for comparing two versions of a skill before merging an edit — did the change make its answers worse? (`eval`: paired, interleaved A/B of two plugin versions, pinned models). NOT for generic unit testing (pytest/vitest of your own scripts) or non-Cowork CI. Covers the skill / run / chat / record / replay / trace / decide / assertions / scaffold / critique / stats / eval commands and the session-vs-scenario split.
|
|
4
4
|
metadata:
|
|
5
5
|
author: cowork-harness
|
|
6
|
-
version: 4.
|
|
7
|
-
tracks-harness: cowork-harness 4.
|
|
6
|
+
version: 4.2.1
|
|
7
|
+
tracks-harness: cowork-harness 4.2.1 (baseline desktop-2.16120.0)
|
|
8
8
|
---
|
|
9
9
|
|
|
10
10
|
# cowork-harness
|
|
@@ -26,8 +26,8 @@ allowlist). This skill exists mostly to keep you out of those traps — the *Inv
|
|
|
26
26
|
full landmine catalog in [`references/gotchas.md`](references/gotchas.md) are the highest-value part.
|
|
27
27
|
Read them.
|
|
28
28
|
|
|
29
|
-
> **Version note:** the facts and `file:line` pointers here track `cowork-harness 4.
|
|
30
|
-
> `desktop-2.
|
|
29
|
+
> **Version note:** the facts and `file:line` pointers here track `cowork-harness 4.2.1` (baseline
|
|
30
|
+
> `desktop-2.16120.0`). If your checkout is newer, prefer the live `--help` and — in a repo checkout —
|
|
31
31
|
> `SPEC.md` / `docs/*.md` over this snapshot, and re-run the bundled linter.
|
|
32
32
|
|
|
33
33
|
## Preflight — make sure the harness can actually run
|
|
@@ -43,7 +43,7 @@ Before the first command, confirm the CLI is reachable and **fail loud** (never
|
|
|
43
43
|
|
|
44
44
|
- **One-shot check.** Run `cowork-harness doctor [--tier <tier>]` first — a read-only prerequisite check that inspects Docker, the staged agent, the token, and the baseline in one pass. The bullets below explain each thing it checks (and how to fix it).
|
|
45
45
|
- **Replay-only? Skip `doctor`.** Replaying committed cassettes needs no Docker, no staged agent, and no token — and every tier's `doctor` validates the auth token (the live tiers also Docker + the staged agent), so a ✗ there is expected, not a blocker. Go straight to `cowork-harness replay <cassette>`.
|
|
46
|
-
- **CLI on PATH, recent enough?** Run `cowork-harness --version` — this skill needs **≥ 4.
|
|
46
|
+
- **CLI on PATH, recent enough?** Run `cowork-harness --version` — this skill needs **≥ 4.2.1**. If it's missing or older, prefix every command with the version floor `npx "cowork-harness@^4.2.1" <cmd>` (Node ≥ 22), or install once with `npm i -g "cowork-harness@^4.2.1"`. **Pin `@^4.2.1`, never `@latest`** — `@latest` can silently fetch an older CLI and the new commands fail as "unknown command", whereas the floor **fails loud** if no compatible version is published.
|
|
47
47
|
|
|
48
48
|
This skill documents the CURRENT surface, not release history. If `cowork-harness --version` is
|
|
49
49
|
OLDER than the floor, the per-release record of what you are missing is [CHANGELOG.md](https://github.com/yaniv-golan/cowork-harness/blob/main/CHANGELOG.md)
|
|
@@ -76,8 +76,10 @@ CI-grade scenario, and the post-hoc debug loop; the rest are narrower tools that
|
|
|
76
76
|
grades against a different artifact, `critique-evidence-package.txt`, which none of these tools
|
|
77
77
|
surface; see `references/critique.md`.
|
|
78
78
|
- **Regression-test your skill's ANSWER quality** (not just its behavior — does its guidance still lead to
|
|
79
|
-
correct answers after you edit it?) → author `semantic_matches` scenarios
|
|
80
|
-
|
|
79
|
+
correct answers after you edit it?) → author `semantic_matches` scenarios, then compare the version
|
|
80
|
+
before your edit with the one after using `cowork-harness eval` (EXPERIMENTAL, live: 10 runs per
|
|
81
|
+
scenario at the defaults). See **Recipe 5** step 6 in `references/task-recipes.md` (validity,
|
|
82
|
+
discrimination — the traps) and [`references/eval.md`](references/eval.md).
|
|
81
83
|
- **"What is WRONG with this skill?"** (a graded critique, not a pass/fail) → `cowork-harness critique
|
|
82
84
|
<folder> --prompt "<probe>"`. Up to four model workloads (zero with `--corpus-only`; pass 2 is skipped with no self-report) and 10–20 minutes; budget from
|
|
83
85
|
`report.costUsd.totalUsd`. Reach for it when you want **findings**. **For "what does this skill
|
|
@@ -97,7 +99,7 @@ CI-grade scenario, and the post-hoc debug loop; the rest are narrower tools that
|
|
|
97
99
|
|
|
98
100
|
Full command set: `skill · run · chat · record · replay · verify-cassettes · rehash · prune · migrate-run-dir · lint ·
|
|
99
101
|
lint-skill · analyze-skill · probe-dispatch ·
|
|
100
|
-
verify-run · trace · inspect · diff · critique · stats · decide · gates · answer · scaffold · assertions --list · sync ·
|
|
102
|
+
verify-run · trace · inspect · diff · critique · eval · eval report · stats · decide · gates · answer · scaffold · assertions --list · sync ·
|
|
101
103
|
list · boundary-check · status · vm <init|status|delete|prune> · doctor · init-redact`. Always check `cowork-harness <cmd> --help`.
|
|
102
104
|
|
|
103
105
|
## Invariants — how a green run lies
|
|
@@ -107,7 +109,8 @@ behind each, is [`references/gotchas.md`](references/gotchas.md).
|
|
|
107
109
|
|
|
108
110
|
1. **`result: success` is not "the task completed".** It means the agent didn't error. Assert the
|
|
109
111
|
deliverable (`file_exists` / `artifact_json` / `transcript_matches`). A `skill`-lane `PASS` only means
|
|
110
|
-
no guard fired: read `skillsInvoked
|
|
112
|
+
no guard fired: read `skillsInvoked` (plus `slashInvokedSkills` — a `/<skill>` prompt runs the skill
|
|
113
|
+
with no `Skill` call), `models` and `ablated` before concluding anything from it.
|
|
111
114
|
2. **`replay` skips live-only keys.** Filesystem and egress keys are skipped on replay (loudly), so a
|
|
112
115
|
mixed item like `{result, egress_denied}` greens on its content half. Keep one concern per `assert:`
|
|
113
116
|
item, put live-only checks on a live gate, and run `cowork-harness lint`.
|
|
@@ -154,4 +157,5 @@ behind each, is [`references/gotchas.md`](references/gotchas.md).
|
|
|
154
157
|
| [`references/fidelity-and-answers.md`](references/fidelity-and-answers.md) | tier semantics, answer paths, the determinism contract |
|
|
155
158
|
| [`references/ci-recipe.md`](references/ci-recipe.md) | the GitHub Action, replay-vs-live lanes, the four-stage pipeline |
|
|
156
159
|
| [`references/critique.md`](references/critique.md) | `critique` report and evidence-package shapes |
|
|
160
|
+
| [`references/eval.md`](references/eval.md) | `eval`: paired before/after of two plugin versions — labels, refusals, exit codes, files |
|
|
157
161
|
| `scripts/scenario.py` | `scaffold`, `lint`, `lint-skill`, `resolve-agent-types <plugin-dir>` (validates a pinned `subagent_type` against `plugin.json` + `agents/*.md`) |
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# Assertion catalog
|
|
2
2
|
|
|
3
|
-
Tracks `cowork-harness 4.
|
|
3
|
+
Tracks `cowork-harness 4.2.1` (baseline `desktop-2.16120.0`). Every `assert:` key with its semantics, and the
|
|
4
4
|
verdict-signal table. Which keys survive `replay` is in [`scenario-schema.md`](./scenario-schema.md#replay-class);
|
|
5
5
|
the scenario and session YAML fields are there too.
|
|
6
6
|
|
|
@@ -55,8 +55,8 @@ same set live from the schema.
|
|
|
55
55
|
| `subagent_declared_but_unused: <Tool>` | a sub-agent declared the tool but never used **that** tool (even if it used others) |
|
|
56
56
|
| `subagent_output_contains: {match?, contains}` | a dispatched sub-agent's own output contains the substring `contains` — `match` (optional regex over `dispatchAgentType`/`resolvedAgentType`/`description`) narrows to specific dispatch(es); omitted, checks whether ANY dispatch's output contains it (existence check, not "all"); a miss against an output that was **truncated at the assert cap** reports evidence-unavailable instead of a proven absence — the substring could lie past the cut. **Covers what the run dispatches** (`Agent`/`Task`, including `Agent(subagent_type:"fork")`), **not a `context: fork` skill** invoked through the `Skill` tool: that skill's own answer is never a dispatch — it comes back as the `Skill` tool result, which agent 2.1.284 builds as `Skill "<name>" completed (forked execution).`, a `Result:` line, then the answer. Assert on it with `tool_result_matches` anchored on that prefix, e.g. `tool_result_matches: '^Skill "[^"]*" completed \(forked execution\)[\s\S]*<pattern>'` (`[\s\S]*` because `.` stops at a newline; the match is case-insensitive, has no multiline flag so `^` is the start of the result, and sees the first 10 KB of each result). This covers a **foreground** fork only: a backgrounded fork's result is the line `Skill "<name>" launched (forked execution, running in the background).`, which carries no answer |
|
|
57
57
|
| `dispatch_count_max: <N>` | at most N sub-agents dispatched — your author-chosen budget under Cowork's agent-side fan-out cap (concurrent 20 / per-session 200, inherited by the harness); records only, does not itself enforce — see gotcha 12 in `scenario-schema.md` |
|
|
58
|
-
| `skill_triggered: <regex>` | a skill matching the regex (invoked id, e.g. `"plugin:skill"`) was invoked via the `Skill` tool — evidence-unavailable (not a normal fail) if the agent's init tools have no `Skill` tool |
|
|
59
|
-
| `no_skill_triggered: <regex>` | no invoked skill id matched — the negative-control / description-collision catcher; evidence-unavailable (never a vacuous pass) if invocation data is absent
|
|
58
|
+
| `skill_triggered: <regex>` | a skill matching the regex (invoked id, e.g. `"plugin:skill"`) was invoked — via the `Skill` tool, or by a prompt starting `/<skill> …` / `/<plugin>:<skill> …` (the agent expands that itself with no `Skill` call; recorded as `slashInvokedSkills`) — evidence-unavailable (not a normal fail) if neither matched and the agent's init tools have no `Skill` tool, or the leading `/name` can't be resolved (ambiguous bare name, no skill inventory) |
|
|
59
|
+
| `no_skill_triggered: <regex>` | no invoked skill id matched, counting a slash-command invocation as well as a `Skill` call — the negative-control / description-collision catcher; evidence-unavailable (never a vacuous pass) if invocation data is absent, the `Skill` tool is unobservable, or the prompt's leading `/name` can't be resolved |
|
|
60
60
|
| `skill_available: <regex>` | a staged skill's id matched the regex (offered, not necessarily invoked — see `skill_triggered` for invocation) — content-class: the id list comes from the agent's init `skills` listing, so it replays from the frozen init event (id-only; the `whenToUse` enrichment is live-disk and thus absent on replay, but the id is what's matched); evidence-unavailable only if `RunResult.context.availableSkills` is absent entirely (an older cassette recorded before the available-skills listing was captured) |
|
|
61
61
|
| `connector_available: <regex>` | an MCP server/connector's name matched the regex (available, not necessarily used) — evidence-unavailable if `RunResult.context.mcpServers` is absent |
|
|
62
62
|
| `tool_available: <regex>` | a tool in the init manifest matched the regex (available, not necessarily called — see `tool_called` for invocation) — evidence-unavailable if `RunResult.context.tools` is absent. The `mcp__skills__*`/`mcp__plugins__*` discovery tools are modeled (as `alwaysLoad`) on `container`/`hostloop`/`cowork` — a miss there is a real absence; `microvm`/`protocol` declare no such server, so a miss on those two tiers means "not modeled at this tier", not "provably unavailable" |
|
|
@@ -107,7 +107,7 @@ same set live from the schema.
|
|
|
107
107
|
| `egress_allowed: <host>` | the host was allowed through |
|
|
108
108
|
| `no_mcp_error: true` | no MCP round-trip failed (`RunResult.mcpErrors` is empty — no unhandled server, no handler throw) — live-only: MCP round-trips are harness-computed, not in the SDK stdout stream, so evidence-unavailable on replay (never a vacuous pass). **Only `true` is valid** |
|
|
109
109
|
| `max_peak_rss_bytes: <N>` | peak sampled RSS of the agent sandbox ≤ N bytes (`RunResult.resources.peakRssBytes`) — live-only: replay never spawns a sandbox to sample, so evidence-unavailable on replay/protocol (never a vacuous pass); also evidence-unavailable when sampling captured no RSS value |
|
|
110
|
-
| `semantic_matches: {rubric: [...], min_pass?, judge_model?, include_subagent_text?, evidence_files?}` | a pinned LLM judge grades each fixed `rubric` claim against the run's answer — the **union of the agent's final result text (`RunResult.finalMessage`), the transcript, and the final on-disk content of any files the agent authored during the run** — so a claim about content the skill led the agent to *write to a file* grades as reliably as one about inlined prose. **"The transcript" is narrower than it reads: top-level `assistant_text` ONLY.** It excludes every `tool_use`/`tool_result` and **all sub-agent-originated text** (including fork-scoped `Skill`/`Agent(fork)` dispatches — the harness attributes their *tool* calls to the main agent, but not their text). **A `context: fork` skill's own answer is not graded either**, even with `include_subagent_text: true`: it is not a dispatch (so it has no `subagents[]` entry to fold in), and it reaches the main agent as the `Skill` tool result, which the judged document excludes. The judge sees it only if the main agent restates it; to check the fork's answer directly, use `tool_result_matches` (see `subagent_output_contains`). ⚠️ **Consequence: a rubric claim about whether a tool was called can NEVER grade true** — the evidence is not in the judged document. Such a claim looks reasonable and silently caps your pass rate; assert tool use with `tool_called` / `present_files_called` / `subagent_dispatched` / `hook_blocked` instead. Sub-agent text is captured in `RunResult.subagents[].reasoning` and reaches the judge only via opt-in `include_subagent_text: true` (`kind:"text"` turns only — sub-agent *thinking* arrives empty with `redacted:true`, so it would pad the document with blanks) (authored-file evidence is captured on every live sandbox tier including **microvm** — its session tree is snapshotted from the VM into the run dir). When the authored-file evidence backing the judged document is **incomplete** — a file dropped at the capture-size cap, unreadable at read-back, or (on `--resume`) the scratchpad walk skipped — the assert fails evidence-unavailable rather than trusting a judge grade over a partial document; this is separate from the malformed-grade `judgeInvalid` path below. The assert passes iff ≥ `min_pass` claims pass (default: all — avoid for a gating scenario). Results align by claim index and are recorded per-claim in `RunResult.assertions[].semanticClaims` (`[{index, claim, pass}]`, so a consumer can diff the per-claim profile across runs); a rep whose grade can't be parsed (after one retry) is marked `RunResult.assertions[].judgeInvalid` and **never silently dropped** — it is excluded from the pass denominator, and the guard against a misleading score from that exclusion is the gate's minimum-valid-rep floor (`MIN_VALID` ≥ 4) plus this visibility, not a claim that denominator-shrinking inflation is impossible. Within a rep, a grade that's still unparseable after the retry **fails that assert outright** (evidence-unavailable, not a vacuous pass) — a persistently-flaky judge reds the run rather than silently passing. `judge_model` pins the grader (default when neither it nor `COWORK_HARNESS_JUDGE_MODEL` is set: `claude-opus-4-8`; a dated id keeps a before/after comparison reproducible). Live-only: the judge is a live model call, so evidence-unavailable / skipped-loud on replay (never a vacuous pass) **`evidence_files: [globs]` scopes which authored files are graded** — reach for it the moment a run authors more than a couple of files. The capture budget (64 KiB total by default) is spent prefix-major then alphabetically, so a pipeline that stages intermediates (`outputs/_work/*.json`) exhausts it before reaching its own deliverable and the verdict is refused evidence-unavailable over files no rubric mentions. Scoping also makes the capture spend the budget on the named files FIRST and exempts them from the per-file cap. Paths are `<root>/<rel>` (`outputs/report.md`, never a bare `report.md`; session-root writes are `scratchpad/<rel>`); globs are `*`/`?`/`**`, not regex. A glob matching nothing FAILS and the message lists every authored path — read it rather than guessing. Still too big? Raise `$COWORK_HARNESS_AUTHORED_TOTAL_BYTES`. The typed reason is on `RunResult.assertions[].semanticEvidence` — check `.reason` instead of parsing the message |
|
|
110
|
+
| `semantic_matches: {rubric: [...], min_pass?, judge_model?, include_subagent_text?, evidence_files?}` | a pinned LLM judge grades each fixed `rubric` claim against the run's answer — the **union of the agent's final result text (`RunResult.finalMessage`), the transcript, and the final on-disk content of any files the agent authored during the run** — so a claim about content the skill led the agent to *write to a file* grades as reliably as one about inlined prose. **"The transcript" is narrower than it reads: top-level `assistant_text` ONLY.** It excludes every `tool_use`/`tool_result` and **all sub-agent-originated text** (including fork-scoped `Skill`/`Agent(fork)` dispatches — the harness attributes their *tool* calls to the main agent, but not their text). **A `context: fork` skill's own answer is not graded either**, even with `include_subagent_text: true`: it is not a dispatch (so it has no `subagents[]` entry to fold in), and it reaches the main agent as the `Skill` tool result, which the judged document excludes. The judge sees it only if the main agent restates it; to check the fork's answer directly, use `tool_result_matches` (see `subagent_output_contains`). ⚠️ **Consequence: a rubric claim about whether a tool was called can NEVER grade true** — the evidence is not in the judged document. Such a claim looks reasonable and silently caps your pass rate; assert tool use with `tool_called` / `present_files_called` / `subagent_dispatched` / `hook_blocked` instead. Sub-agent text is captured in `RunResult.subagents[].reasoning` and reaches the judge only via opt-in `include_subagent_text: true` (`kind:"text"` turns only — sub-agent *thinking* arrives empty with `redacted:true`, so it would pad the document with blanks) (authored-file evidence is captured on every live sandbox tier including **microvm** — its session tree is snapshotted from the VM into the run dir). When the authored-file evidence backing the judged document is **incomplete** — a file dropped at the capture-size cap, unreadable at read-back, or (on `--resume`) the scratchpad walk skipped — the assert fails evidence-unavailable rather than trusting a judge grade over a partial document; this is separate from the malformed-grade `judgeInvalid` path below. The assert passes iff ≥ `min_pass` claims pass (default: all — avoid for a gating scenario). Results align by claim index and are recorded per-claim in `RunResult.assertions[].semanticClaims` (`[{index, claim, pass, rationale?}]`, so a consumer can diff the per-claim profile across runs; `rationale` is the judge's one-sentence reason, printed under each failed claim in the failure footer: untrusted model text that can quote the judged document, whose content never affects `pass`, absent when the judge gave none (a reply whose shape is broken, such as unparseable JSON, a malformed `{"results": …}` group beside a valid grade, or a partial restatement that contradicts it, is retried once and then marked `judgeInvalid`), and comparable only between runs that share `judgePromptHash`); a rep whose grade can't be parsed (after one retry) is marked `RunResult.assertions[].judgeInvalid` and **never silently dropped** — it is excluded from the pass denominator, and the guard against a misleading score from that exclusion is the gate's minimum-valid-rep floor (`MIN_VALID` ≥ 4) plus this visibility, not a claim that denominator-shrinking inflation is impossible. Within a rep, a grade that's still unparseable after the retry **fails that assert outright** (evidence-unavailable, not a vacuous pass) — a persistently-flaky judge reds the run rather than silently passing. `judge_model` pins the grader (default when neither it nor `COWORK_HARNESS_JUDGE_MODEL` is set: `claude-opus-4-8`; a dated id keeps a before/after comparison reproducible — or let `eval` compare two skill versions per claim, with the judge pinned). Each graded assert records the judge's provenance: `RunResult.assertions[].judgeModel` (the resolved model), `judgeCostUsd` (judge spend over both attempts, reported beside `cost.usd` and never inside it; absent when unpriced) and `judgePromptHash` (the grading-prompt template — compare only runs that share it). Live-only: the judge is a live model call, so evidence-unavailable / skipped-loud on replay (never a vacuous pass) **`evidence_files: [globs]` scopes which authored files are graded** — reach for it the moment a run authors more than a couple of files. The capture budget (64 KiB total by default) is spent prefix-major then alphabetically, so a pipeline that stages intermediates (`outputs/_work/*.json`) exhausts it before reaching its own deliverable and the verdict is refused evidence-unavailable over files no rubric mentions. Scoping also makes the capture spend the budget on the named files FIRST and exempts them from the per-file cap. Paths are `<root>/<rel>` (`outputs/report.md`, never a bare `report.md`; session-root writes are `scratchpad/<rel>`); globs are `*`/`?`/`**`, not regex. A glob matching nothing FAILS and the message lists every authored path — read it rather than guessing. Still too big? Raise `$COWORK_HARNESS_AUTHORED_TOTAL_BYTES`. The typed reason is on `RunResult.assertions[].semanticEvidence` — check `.reason` instead of parsing the message |
|
|
111
111
|
| `artifact_json: {artifact, path, …}` | assert a JSON artifact's contents — `equals`/`gt`/`in`/`exists`/`absent`/`is_null` over a dotted `path` (`in` = membership in a list, for a stochastic/LLM value; `absent` ≠ `is_null`; an unresolved intermediate fails loud) |
|
|
112
112
|
| `computer_links_resolve: true` | every `computer://` link in the model-visible transcript resolves to an artifact that exists in the run's collected outputs/mounts — a dangling link fails, naming which target was checked (a live host path, the collected work tree, or the replay manifest). **Requires ≥1 link** (zero links fails — use `computer_links_resolve_if_present` for the presence-free variant). **Only `true` is valid** (`false` is rejected by the schema) **Sees top-level `assistant_text` only — it excludes every `tool_use`/`tool_result`**, so a `computer://` link that appeared only inside a tool call or its result is invisible to it. |
|
|
113
113
|
| `computer_links_resolve_if_present: true` | like `computer_links_resolve` but passes vacuously when the transcript has zero `computer://` links — the presence-free variant. **Only `true` is valid** **Sees top-level `assistant_text` only — it excludes every `tool_use`/`tool_result`**, so a `computer://` link that appeared only inside a tool call or its result is invisible to it. |
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# Assertions guide
|
|
2
2
|
|
|
3
|
-
Tracks `cowork-harness 4.
|
|
3
|
+
Tracks `cowork-harness 4.2.1` (baseline `desktop-2.16120.0`). Read it when choosing assertion keys: the two orthogonal axes and the goal → key map. The full catalog is `assertion-catalog.md`.
|
|
4
4
|
|
|
5
5
|
### Assertions: two orthogonal axes
|
|
6
6
|
|
|
@@ -43,7 +43,7 @@ them by what you're trying to prove:
|
|
|
43
43
|
| a skill actually **ran** (or must NOT) | `skill_triggered: <regex>`, `no_skill_triggered: <regex>` |
|
|
44
44
|
| a tool ran **inside** a skill's scope | `skill_tool_used: {skill, tool}` |
|
|
45
45
|
| a sub-agent did the work | `subagent_output_contains: {contains}`, `subagent_dispatched: <regex>`, `dispatch_count_max: <N>` |
|
|
46
|
-
| a `context: fork` skill answered correctly | `tool_result_matches: '^Skill "[^"]*" completed \(forked execution\)[\s\S]*<pattern>'` — its answer is the `Skill` tool result, not a sub-agent output, so `subagent_output_contains` and `semantic_matches` never see it (foreground fork only — a backgrounded fork's result carries no answer) |
|
|
46
|
+
| a `context: fork` skill answered correctly | `tool_result_matches: '^Skill "[^"]*" completed \(forked execution\)[\s\S]*<pattern>'` — its answer is the `Skill` tool result, not a sub-agent output, so `subagent_output_contains` and `semantic_matches` never see it (foreground fork only — a backgrounded fork's result carries no answer). Only when the MODEL invokes the skill: a `/<skill> …` prompt runs the fork with no `Skill` call and no such result — use `skill_triggered` + `transcript_matches` there |
|
|
47
47
|
| a pre-existing input wasn't mutated (incl. `uploads/**`) | `input_unmodified: <glob>` or `[<glob>, …]` (live/verify-run) |
|
|
48
48
|
| no authored interactive artifact silently loses its Submit under Cowork | `no_lost_write_back: true` (**live-only**; static Tier A over the run's authored `.html`/`.py`/`.js`; per-scenario gate for the same class `analyze-skill` scans) |
|
|
49
49
|
| a resource ceiling held | `max_peak_rss_bytes: <N>` (**live-only**) |
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# Authoring a scenario
|
|
2
2
|
|
|
3
|
-
Tracks `cowork-harness 4.
|
|
3
|
+
Tracks `cowork-harness 4.2.1` (baseline `desktop-2.16120.0`). Read it when composing a `scenarios/*.yaml`: session vs scenario, discovery, the fidelity tier, the answer path, `web_fetch`, and scaffold + lint.
|
|
4
4
|
|
|
5
5
|
## Part I — AUTHOR a scenario
|
|
6
6
|
|
|
@@ -251,6 +251,7 @@ cowork-harness lint scenarios/*.yaml
|
|
|
251
251
|
| `reference-access-contradiction` | ERROR | one reference under both `reference_read` and `no_observed_reference_access` |
|
|
252
252
|
| `regex-double-quoted` | WARN | a double-quoted regex with an unescaped backslash (YAML strips it) |
|
|
253
253
|
| `replay-noop` | WARN | every assertion is live-only or a verdict modifier, so a replay gate verifies nothing |
|
|
254
|
+
| `slash-prompt-forked-result-anchor` | WARN | a `prompt:` starting with `/<skill>` plus a `tool_result_*` anchored on `forked execution` — a slash-invoked skill makes no `Skill` call, so that tool result never exists; assert `skill_triggered` instead |
|
|
254
255
|
| `tool-called-always-passes` | INFO | `tool_called` with `count: {min: 0}` and no `max` — it asserts nothing |
|
|
255
256
|
| `tool-input-regex-redactable` | WARN | a `tool_not_called` input literal the redaction policy rewrites in the committed cassette (or a policy pattern it cannot check offline) |
|
|
256
257
|
| `tool-input-shell-tier` | INFO | the object form with `tool: Bash` and a `command` on `hostloop` / `cowork`, where shell runs as `mcp__workspace__bash` — list both |
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# CI recipe — replay vs live lanes
|
|
2
2
|
|
|
3
|
-
Self-contained reference. Tracks `cowork-harness 4.
|
|
3
|
+
Self-contained reference. Tracks `cowork-harness 4.2.1` (baseline `desktop-2.16120.0`).
|
|
4
4
|
|
|
5
5
|
**Fastest path: the packaged Action.** One step gets you `replay`/`lint`/`verify-cassettes` plus a PR
|
|
6
6
|
job-summary reporter (verdict table, staleness findings, cost/turns when available):
|
|
@@ -17,7 +17,7 @@ job-summary reporter (verdict table, staleness findings, cost/turns when availab
|
|
|
17
17
|
CLI major reaches your workflow the moment it is promoted even though your `uses:` ref never changed — so a
|
|
18
18
|
copy-pasted recipe that omits the input takes a major bump with no say in it. `^4` holds the major, needs no
|
|
19
19
|
patch number to remember, and only wants a human decision at the next major. Pin an exact version
|
|
20
|
-
(e.g. `version: "4.
|
|
20
|
+
(e.g. `version: "4.2.0"`) instead when you want byte-reproducible CI.
|
|
21
21
|
|
|
22
22
|
Reach for the manual multi-step form below only when you need per-step control the Action's inputs don't
|
|
23
23
|
cover (a custom flag combination, a different runner matrix per step, or `lint`/`verify-cassettes` gated
|
|
@@ -82,7 +82,7 @@ sha256-*checked* but not hard-blocking on mismatch — it's advisory for an inte
|
|
|
82
82
|
GitHub-hosted runners, no token/Docker/agent:
|
|
83
83
|
|
|
84
84
|
```yaml
|
|
85
|
-
- run: npm i -g "cowork-harness@^4.
|
|
85
|
+
- run: npm i -g "cowork-harness@^4.2.1"
|
|
86
86
|
- run: cowork-harness lint scenarios/*.yaml --strict --min-severity WARN
|
|
87
87
|
# no silent false-greens. WITHOUT --strict this
|
|
88
88
|
# step cannot fail on a WARN-class rule (e.g.
|
|
@@ -395,7 +395,7 @@ jobs:
|
|
|
395
395
|
with: { node-version: '24' }
|
|
396
396
|
- uses: actions/setup-python@v5
|
|
397
397
|
with: { python-version: '3.x' } # python3 only — PyYAML is bundled with the linter
|
|
398
|
-
- run: npm i -g "cowork-harness@^4.
|
|
398
|
+
- run: npm i -g "cowork-harness@^4.2.1"
|
|
399
399
|
- run: cowork-harness lint scenarios/*.yaml # no-silent-false-green (needs python3; PyYAML bundled)
|
|
400
400
|
- run: cowork-harness verify-cassettes cassettes/ --output-format json # privacy + staleness gate
|
|
401
401
|
- run: cowork-harness replay cassettes/ --output-format json # token-free content/structure
|
|
@@ -424,7 +424,7 @@ jobs:
|
|
|
424
424
|
echo "live=true" >> "$GITHUB_OUTPUT"
|
|
425
425
|
fi
|
|
426
426
|
- if: steps.guard.outputs.live == 'true'
|
|
427
|
-
run: npm i -g "cowork-harness@^4.
|
|
427
|
+
run: npm i -g "cowork-harness@^4.2.1"
|
|
428
428
|
- if: steps.guard.outputs.live == 'true'
|
|
429
429
|
run: cowork-harness run scenarios/ --output-format json
|
|
430
430
|
env:
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# Critique — the facts a plugin install can't otherwise reach
|
|
2
2
|
|
|
3
|
-
Tracks `cowork-harness 4.
|
|
3
|
+
Tracks `cowork-harness 4.2.1` (baseline `desktop-2.16120.0`). This is **not** a trim of the full
|
|
4
4
|
[`docs/critique.md`](https://github.com/yaniv-golan/cowork-harness/blob/main/docs/critique.md) (repo-only —
|
|
5
5
|
flags, cost, reproduction discipline, known limitations all live there). This file covers exactly what a
|
|
6
6
|
plugin install cannot otherwise discover: the run-dir artifact a harvester actually reads, the report's
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# Debugging a run
|
|
2
2
|
|
|
3
|
-
Tracks `cowork-harness 4.
|
|
3
|
+
Tracks `cowork-harness 4.2.1` (baseline `desktop-2.16120.0`). Read it when a run misbehaved or a green looks wrong: triage, the observability output, and `chat`.
|
|
4
4
|
|
|
5
5
|
## Part III — Debug
|
|
6
6
|
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
# `eval` — paired before/after comparison of a skill edit (EXPERIMENTAL)
|
|
2
|
+
|
|
3
|
+
Tracks `cowork-harness 4.2.1` (baseline `desktop-2.16120.0`). The full guide is
|
|
4
|
+
[docs/eval.md](https://github.com/yaniv-golan/cowork-harness/blob/main/docs/eval.md); this is the part you
|
|
5
|
+
need while running it.
|
|
6
|
+
|
|
7
|
+
```bash
|
|
8
|
+
cowork-harness eval <scenario.yaml | dir/> --arm before=git:HEAD:plugins/my-skill --arm after=./plugins/my-skill \
|
|
9
|
+
--model <concrete id> --judge-model <concrete id> [--holdout <scenario.yaml>]... [--fail-on possible|confirmed]
|
|
10
|
+
cowork-harness eval report <eval-dir> # rebuild the report from the eval dir, no spend
|
|
11
|
+
```
|
|
12
|
+
|
|
13
|
+
## What it does
|
|
14
|
+
|
|
15
|
+
- Two arms: a plugin directory, or `git:<ref>:<path>` (read from the commit, `<path>` relative to the repo
|
|
16
|
+
root). The FIRST arm is the baseline; a drop is the second arm passing less often.
|
|
17
|
+
- Only the session's single `plugins.local_plugins` entry is substituted. Each arm is copied once, before
|
|
18
|
+
the first run, and every rep mounts the copy.
|
|
19
|
+
- The plugin is mounted at the `local_plugins` path (`mnt/.local-plugins/marketplaces/<marketplace>/<plugin>`),
|
|
20
|
+
not the `remote_plugins` path a UI-installed plugin has (`mnt/.remote-plugins/plugin_<id>`). A skill that
|
|
21
|
+
locates its own files at runtime sees a different path under each, so the comparison holds for the
|
|
22
|
+
`local_plugins` layout only.
|
|
23
|
+
- scenarios × 2 × `--reps` live runs (10 per scenario at the default `--reps 5`), interleaved, plus one judge
|
|
24
|
+
call per `semantic_matches` assert per run. The start-up line prints the job count. No budget flag.
|
|
25
|
+
- In a scenario directory, YAML with no `prompt:` (a session file) is skipped.
|
|
26
|
+
- Run an A/A first (`--allow-identical-arms`, the same source twice) to see your scenarios' noise.
|
|
27
|
+
|
|
28
|
+
## Reading the labels
|
|
29
|
+
|
|
30
|
+
| Label | Meaning |
|
|
31
|
+
|---|---|
|
|
32
|
+
| `confirmed drop/rise` | significant after the correction (bh q = 0.10, or holm) |
|
|
33
|
+
| `possible drop/rise` | p ≤ `--alpha` (0.05), not confirmed |
|
|
34
|
+
| `no detectable change` | with the smallest change this n could have detected (MDD) |
|
|
35
|
+
| `underpowered` | no outcome at these sizes could reach `--alpha` — NOT "no change" |
|
|
36
|
+
| `insufficient` | too few valid reps in an arm (4 of 5 needed by default) |
|
|
37
|
+
|
|
38
|
+
A drop is a signal to investigate, not proof: open the run dirs the report links for that row. In order,
|
|
39
|
+
first match wins: an infrastructure failure is excluded and reported — including a rep where no model
|
|
40
|
+
answered (the agent's `Not logged in` / `Authentication required` reply, rule `auth`; a usage or
|
|
41
|
+
spend limit as its final message, even on a nonzero exit after spend, rule `usage_limit`; or only
|
|
42
|
+
`<synthetic>` models at $0, rule `no_model_answered`). An agent-caused failure (timeout, max turns,
|
|
43
|
+
unanswered question, crash) then fails every row of its rep, even with its pin unknown. Only after that
|
|
44
|
+
are a pin the agent did not honour (false, or unknown on a rep that completed) and a snapshot that changed
|
|
45
|
+
excluded and reported. A loud UNCLASSIFIED count means a termination the classifier does not know — read
|
|
46
|
+
those runs. Per scenario: if EVERY rep of both arms errored, or every rep of one arm is infrastructure,
|
|
47
|
+
that scenario compared nothing — its rows are `insufficient` and the eval exits 1. If one arm's every rep
|
|
48
|
+
is the agent's own failure and the other arm ran, the reps are scored (a real drop) — exit 0 unless
|
|
49
|
+
`--fail-on`. Either way the header
|
|
50
|
+
names the arm, scenario, dominant error and a matching hint (`Every rep of arm <label> in <scenario> errored — …`).
|
|
51
|
+
|
|
52
|
+
## Refused before any run (exit 2)
|
|
53
|
+
|
|
54
|
+
- a model or judge that is an alias (`opus`, `best`) rather than a concrete id;
|
|
55
|
+
- an eval dir inside any git work tree (the snapshots would mount empty);
|
|
56
|
+
- identical arms (unless `--allow-identical-arms`);
|
|
57
|
+
- an arm that contains the eval's own scenario or session files (by location, copy or symlink), an
|
|
58
|
+
`evals.json`, or a symlink resolving outside it;
|
|
59
|
+
- a scenario input a run would refuse (a missing path, a `tool_not_called` the tier can never violate);
|
|
60
|
+
- `--fail-on confirmed` when no row could reach `confirmed` at this `--reps`;
|
|
61
|
+
- no usable agent credential for a scenario's tier — the same check as `doctor --tier <tier>`'s `token` row,
|
|
62
|
+
with its fix. A Keychain login or a `.credentials.json` in the config dir, without an env/.env token,
|
|
63
|
+
passes only at `protocol`;
|
|
64
|
+
- a session whose plugin is declared only under `plugins.remote_plugins` (the session must declare exactly
|
|
65
|
+
one `local_plugins` entry). Workaround: eval a copy of the session that declares the same directory under
|
|
66
|
+
`local_plugins`, and check the `remote_plugins` path handling with an ordinary `run`.
|
|
67
|
+
|
|
68
|
+
## Exit codes
|
|
69
|
+
|
|
70
|
+
`0` completed — no drop fails the eval unless you pass `--fail-on`. `1` a drop at the `--fail-on` level,
|
|
71
|
+
every row `insufficient`, a scenario that compared nothing, or the judge model differed across reps (an A/A run under `--fail-on possible`
|
|
72
|
+
can exit 1 on noise). `2` usage or a refusal. `3` an arm snapshot could not be copied or staged.
|
|
73
|
+
|
|
74
|
+
## Files
|
|
75
|
+
|
|
76
|
+
`<eval-dir>/` (default `~/.cowork-harness/evals/<eval-id>/`): `manifest.json`, `arms/`, `runs.jsonl`,
|
|
77
|
+
`report.json` (every rep with its bucket), `report.md`. The runs are ordinary run dirs labelled
|
|
78
|
+
`eval:<eval-id>:<arm>`; a bare `prune` keeps 5 per scenario and says which evals it trimmed — `eval report`
|
|
79
|
+
still works, but the evidence links then dangle.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# Fidelity tiers & answer paths
|
|
2
2
|
|
|
3
|
-
Self-contained reference. Tracks `cowork-harness 4.
|
|
3
|
+
Self-contained reference. Tracks `cowork-harness 4.2.1` (baseline `desktop-2.16120.0`).
|
|
4
4
|
|
|
5
5
|
> **This page vs. the repo docs.** This is the **offline snapshot** that ships inside the installed
|
|
6
6
|
> plugin — it is self-contained on purpose. The repo carries four other fidelity views, each answering a
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# Gotchas
|
|
2
2
|
|
|
3
|
-
Tracks `cowork-harness 4.
|
|
3
|
+
Tracks `cowork-harness 4.2.1` (baseline `desktop-2.16120.0`). The full "✓ passed ≠ correct" landmine catalog.
|
|
4
4
|
|
|
5
5
|
## Gotchas — the "✓ passed ≠ correct" landmines
|
|
6
6
|
|
|
@@ -235,6 +235,17 @@ authorable). Reach for this list when debugging a run's behavior, that one while
|
|
|
235
235
|
fail your gate. *Fix:* nothing, unless the scenario asserts `tool_available` on
|
|
236
236
|
`mcp__skills__*`/`mcp__plugins__*`; then re-record. It stays silent at `microvm`/`protocol`, where
|
|
237
237
|
re-recording would never produce those tools anyway.
|
|
238
|
+
A sibling **`agent-version:` note** means the agent version the cassette's own `system/init` event
|
|
239
|
+
reports differs from the one the baseline its `fingerprint.baseline` names pins for that tier: the
|
|
240
|
+
`agentVersion` at `container`/`microvm`, the native agent in `agentBinary.nativeStagedPath` at
|
|
241
|
+
`hostloop`. It never appears at `protocol`, which runs the unpinned `claude` on your `PATH`. *Why:* one
|
|
242
|
+
of a fingerprint re-stamped by hand across an agent bump, a recording made under
|
|
243
|
+
`COWORK_HARNESS_ALLOW_AGENT_FALLBACK=1`, an explicit binary override (`COWORK_AGENT_BINARY`, or
|
|
244
|
+
`COWORK_HOST_AGENT_BINARY` at `hostloop`), or at `hostloop` the default patch-bump substitution of the
|
|
245
|
+
native agent; the note lists that tier's causes and does not pick one. It is non-gating too:
|
|
246
|
+
`verify-cassettes` puts it in the result's `notes[]`, and `replay` prints one
|
|
247
|
+
`::notice:: [replay] <file> — … [agent-version]` line per cassette on stderr (also under
|
|
248
|
+
`--output-format json`). *Fix:* re-record against the pinned agent.
|
|
238
249
|
|
|
239
250
|
24. **Never name the file-delivery tool in a `SKILL.md`.** *Why:* Cowork has **two**, one per product
|
|
240
251
|
lane, and an agent only sees the one for the surface it is on. The desktop-local sandbox this harness
|
|
@@ -271,7 +282,9 @@ authorable). Reach for this list when debugging a run's behavior, that one while
|
|
|
271
282
|
`run` the same word additionally means *your assertions held*; on `skill --repeat N`, `PASS — N/N`
|
|
272
283
|
means N runs cleared the guards — it says nothing about which model served them, whether the skill
|
|
273
284
|
was invoked, or whether they were the ablated arm. *Fix:* read the three fields the record already
|
|
274
|
-
carries before drawing any conclusion — `skillsInvoked` / `skillActivity` (was it invoked at all
|
|
285
|
+
carries before drawing any conclusion — `skillsInvoked` / `skillActivity` (was it invoked at all;
|
|
286
|
+
a `/<skill> …` prompt runs the skill with NO `Skill` call, so read `slashInvokedSkills` too — and
|
|
287
|
+
`models` is then just `["<synthetic>"]`, with the real model only in `modelUsage`),
|
|
275
288
|
`models` (which model), `ablated` + `context.availableSkills` (which arm). An answer that reads
|
|
276
289
|
exactly like skill output is not evidence: the skill's own source is mounted where the model can
|
|
277
290
|
read it — in production too — so on a self-referential prompt it may read `SKILL.md` and answer
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# Measurement
|
|
2
2
|
|
|
3
|
-
Tracks `cowork-harness 4.
|
|
3
|
+
Tracks `cowork-harness 4.2.1` (baseline `desktop-2.16120.0`). Read it before comparing runs: `--repeat`, `--ablate-skill`, and the hygiene that keeps a batch valid.
|
|
4
4
|
|
|
5
5
|
### Measure — before/after, with/without (`--repeat`, `--ablate-skill`)
|
|
6
6
|
|
|
@@ -25,6 +25,11 @@ Every ablated run is stamped `ablated: true` in `result.json` and carries `ablat
|
|
|
25
25
|
What the harness gives you here is the run execution and the control arm — designing the comparison
|
|
26
26
|
(scrubbing giveaways, shuffling, judging blind, unblinding only after grading) is still yours.
|
|
27
27
|
|
|
28
|
+
**"Did my edit change it?"** → `cowork-harness eval` (EXPERIMENTAL): the version before your edit and the
|
|
29
|
+
one after, interleaved, with the agent and judge models pinned, compared per assertion and per rubric
|
|
30
|
+
claim with an exact test. It is a regression signal to investigate, not proof — see
|
|
31
|
+
[`eval.md`](eval.md) and Recipe 5 step 6 in [`task-recipes.md`](task-recipes.md).
|
|
32
|
+
|
|
28
33
|
### Tool timing — what `toolDurations` measures
|
|
29
34
|
|
|
30
35
|
`result.json`'s `toolDurations` and `trace <run> --view tool-durations` report, per tool, the **wall gap
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# Run, record and lock
|
|
2
2
|
|
|
3
|
-
Tracks `cowork-harness 4.
|
|
3
|
+
Tracks `cowork-harness 4.2.1` (baseline `desktop-2.16120.0`). Read it when running a scenario, recording or placing a cassette, reading verdict signals, checking a background run, or choosing CI lanes.
|
|
4
4
|
|
|
5
5
|
## Part II — RUN, RECORD & LOCK
|
|
6
6
|
|
|
@@ -193,7 +193,7 @@ Recognize these before "fixing" a non-bug:
|
|
|
193
193
|
`markitdown`/`magika` (`ml_extract`), `cv2` (`cv`), `camelot`/`tabula` (`pdf_tables`), or `wand`
|
|
194
194
|
(`magick`) can trip this even though real Cowork **ships** those (per the rootfs manifest captured at
|
|
195
195
|
Desktop `2.9939.2` — `baselines/provisioning/rootfs-provisioning.json`, which is the dated evidence
|
|
196
|
-
behind that sentence;
|
|
196
|
+
behind that sentence; 2 baselines have shipped since without a re-capture). The message says so ("likely a FALSE
|
|
197
197
|
NEGATIVE (real Cowork ships them)"). Fix: rebuild full parity (`--build-arg COWORK_FULL_PARITY=1`, point
|
|
198
198
|
`COWORK_AGENT_IMAGE` at it), or — if the skill's fallback is genuinely equivalent — assert
|
|
199
199
|
`allow_missing_capability: true`. (Two sources: a skill *observed using* an omitted family, live lane;
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# Scenario & session schema, replay class, web_fetch, authoring gotchas
|
|
2
2
|
|
|
3
|
-
Self-contained reference for authoring `cowork-harness` scenarios. Tracks `cowork-harness 4.
|
|
4
|
-
(baseline `desktop-2.
|
|
3
|
+
Self-contained reference for authoring `cowork-harness` scenarios. Tracks `cowork-harness 4.2.1`
|
|
4
|
+
(baseline `desktop-2.16120.0`). If your checkout is newer, prefer the live [`docs/scenario.md`](https://github.com/yaniv-golan/cowork-harness/blob/main/docs/scenario.md),
|
|
5
5
|
[`docs/session.md`](https://github.com/yaniv-golan/cowork-harness/blob/main/docs/session.md), and `SPEC.md`.
|
|
6
6
|
|
|
7
7
|
**Minimal scenario** — `prompt` and `fidelity` are required:
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
Each recipe composes facts that live scattered across SKILL.md and the other references into one
|
|
4
4
|
decision path. Every one answers a question a real fleet owner had to work out the hard way.
|
|
5
|
-
Tracks `cowork-harness 4.
|
|
5
|
+
Tracks `cowork-harness 4.2.1` (baseline `desktop-2.16120.0`), same as SKILL.md's front-matter. Recipe 2's `resolved-tier`/`unverifiable-tier` staleness classes and
|
|
6
6
|
Recipe 3's `init-redact` shipped in 0.24.0 and are part of the current feature set — no version gate
|
|
7
7
|
needed if your CLI meets SKILL.md's version floor.
|
|
8
8
|
|
|
@@ -198,7 +198,9 @@ degrade the advice. It is real work to calibrate; these steps are the traps that
|
|
|
198
198
|
correct claim can pass one rep and miss the next. A claim's baseline is its pass *rate* (3/3, 2/3), read
|
|
199
199
|
from `RunResult.assertions[].semanticClaims`. Do **not** chase single-run all-pass — set `min_pass` to
|
|
200
200
|
the reliably-hit core for a green verdict, and treat the per-claim rates as the real signal. (N=1
|
|
201
|
-
routinely mislabels a stable 0/3 as "intermittent" and vice-versa.)
|
|
201
|
+
routinely mislabels a stable 0/3 as "intermittent" and vice-versa.) `eval` (step 6) defaults to 5 reps
|
|
202
|
+
per arm and refuses fewer than 4 without `--allow-underpowered`: below that, no exact test can flag
|
|
203
|
+
even a total collapse.
|
|
202
204
|
5. **Check discrimination — does the skill actually help?** Run one rep with the skill NOT installed and
|
|
203
205
|
compare. `--ablate-skill` is the flag for it: it empties every skill/plugin discovery source for
|
|
204
206
|
**that one invocation**, so the agent answers from its own priors, and stamps the result
|
|
@@ -207,14 +209,26 @@ degrade the advice. It is real work to calibrate; these steps are the traps that
|
|
|
207
209
|
A/B — and the rollup labels it `PASS [ABLATED — control arm]` so you cannot bank it as one. A not-invoked rep is not a control:
|
|
208
210
|
outside `--ablate-skill` it can still read the source (see step 3). If the answer still scores high without the skill, that claim is
|
|
209
211
|
answerable from priors and tests the model, not your skill — strengthen it (a skill-specific fact) or
|
|
210
|
-
drop it.
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
|
|
212
|
+
drop it. The harness supplies the runs, this with/without control and, for a before/after of two
|
|
213
|
+
versions, the paired `eval` of step 6; scrubbing giveaways and judging blind stay with you. Before you trust any
|
|
214
|
+
comparison, measure your scenarios' own noise: `eval` with the SAME source as both arms
|
|
215
|
+
(`--allow-identical-arms`) shows how far the rates move when nothing changed.
|
|
216
|
+
6. **Gate a change with `eval` — a paired comparison of the two versions.** Hand-diffing two profiles
|
|
217
|
+
captured at different times mixes your edit with everything else that moved in between. `eval` runs
|
|
218
|
+
both versions of the plugin in one interleaved schedule, holds the agent and judge models fixed, and
|
|
219
|
+
compares every assertion and every rubric claim with an exact test:
|
|
220
|
+
```bash
|
|
221
|
+
cowork-harness eval evals/scenarios/ --arm before=git:HEAD:plugins/my-skill --arm after=./plugins/my-skill \
|
|
222
|
+
--model <concrete id> --judge-model <concrete id> --holdout evals/scenarios/untouched-question.yaml
|
|
223
|
+
```
|
|
224
|
+
The first `--arm` is the baseline. Each arm is snapshotted before the first run (a `git:` arm is read
|
|
225
|
+
from the commit — freeze a recoverable source this way rather than trusting the working tree to stay
|
|
226
|
+
put). A row labelled `possible drop` or `confirmed drop` is a **regression signal to investigate**,
|
|
227
|
+
not proof your edit caused it: open the run dirs the report links for that row, read the transcripts,
|
|
228
|
+
and re-run if the evidence is thin. A claim already at 0% in both arms cannot regress, and one at 100%
|
|
229
|
+
in both cannot show an improvement — the report counts both. Keep `--holdout` scenarios you did not
|
|
230
|
+
tune against; a scenario you shaped the skill to is weak evidence. Details, labels and exit codes:
|
|
231
|
+
[docs/eval.md](https://github.com/yaniv-golan/cowork-harness/blob/main/docs/eval.md).
|
|
218
232
|
|
|
219
233
|
**Lane note:** `semantic_matches` is **live-only** (the judge is a live model call), so these scenarios
|
|
220
234
|
run on the `run` lane, never token-free `replay` — the linter's "all assertions live-only" warning is
|
|
@@ -800,6 +800,58 @@ def _lint_prompt_slash(doc, path):
|
|
|
800
800
|
return findings
|
|
801
801
|
|
|
802
802
|
|
|
803
|
+
_TOOL_RESULT_KEYS = ("tool_result_contains", "tool_result_not_contains", "tool_result_matches", "tool_result_not_matches")
|
|
804
|
+
# `forked execution` as a literal or as a regex spells the gap: a space, an escaped space, `\s`, `.`, or
|
|
805
|
+
# `[\s\S]`, optionally quantified (`\s+`, `.*`).
|
|
806
|
+
_FORKED_EXECUTION_RE = re.compile(r"forked(?: |\\ |\\s|\.|\[\\s\\S\])[+*?]?execution", re.IGNORECASE)
|
|
807
|
+
|
|
808
|
+
|
|
809
|
+
def _lint_slash_prompt_forked_anchor(doc, items, path):
|
|
810
|
+
"""W: a `/<skill>` prompt paired with a tool_result_* anchored on `forked execution`.
|
|
811
|
+
|
|
812
|
+
`Skill "<name>" completed (forked execution).` is the `Skill` TOOL RESULT a `context: fork` skill
|
|
813
|
+
returns when the MODEL invokes it. A prompt whose first character is `/` naming a staged skill is
|
|
814
|
+
expanded by the agent binary itself: no `Skill` tool_use is emitted and the fork runs directly, so that
|
|
815
|
+
result text never exists. A positive tool_result_* anchored on it then fails on a working skill, and a
|
|
816
|
+
negative one passes vacuously. Measured on real hostloop runs of one fork skill (agent 2.1.284): the
|
|
817
|
+
bare and plugin-qualified slash prompts both ran the skill with no `Skill` call; a plain prompt got one.
|
|
818
|
+
|
|
819
|
+
Registration is not checkable statically, so the rule fires on any command-shaped leading token and the
|
|
820
|
+
message says "if". It is deliberately narrow: only the fork-result anchor, since `skill_triggered` /
|
|
821
|
+
`no_skill_triggered` already count a slash-invoked skill. First character only, matching the harness's
|
|
822
|
+
own slash detector (a prompt with leading whitespace is not expanded).
|
|
823
|
+
"""
|
|
824
|
+
prompt = doc.get("prompt")
|
|
825
|
+
if not isinstance(prompt, str):
|
|
826
|
+
return []
|
|
827
|
+
m = re.match(r"/(\S+)", prompt)
|
|
828
|
+
if not m:
|
|
829
|
+
return []
|
|
830
|
+
name = m.group(1)
|
|
831
|
+
if not _SLASH_CMD_NAME_RE.match(name) or name.lower() in _SLASH_PATH_WORDS:
|
|
832
|
+
return []
|
|
833
|
+
findings = []
|
|
834
|
+
for key in _TOOL_RESULT_KEYS:
|
|
835
|
+
for v in _assert_values(items, key):
|
|
836
|
+
if not isinstance(v, str) or not _FORKED_EXECUTION_RE.search(v):
|
|
837
|
+
continue
|
|
838
|
+
findings.append(
|
|
839
|
+
Finding(
|
|
840
|
+
"WARN",
|
|
841
|
+
"slash-prompt-forked-result-anchor",
|
|
842
|
+
f"`prompt:` starts with `/{name}` and `{key}` anchors on `forked execution`. If `/{name}` is a "
|
|
843
|
+
"staged skill, the agent expands the command itself — no `Skill` tool call, so the "
|
|
844
|
+
"`completed (forked execution)` tool result never exists: a positive check fails on a working "
|
|
845
|
+
"skill and a negative one passes vacuously.",
|
|
846
|
+
f"Assert the invocation with `skill_triggered: '{name.split(':')[-1]}'` (it counts a slash-invoked "
|
|
847
|
+
"skill) and the answer with `transcript_matches`; or drop the slash if "
|
|
848
|
+
"the scenario means to test the model invoking the skill through the `Skill` tool.",
|
|
849
|
+
path,
|
|
850
|
+
)
|
|
851
|
+
)
|
|
852
|
+
return findings
|
|
853
|
+
|
|
854
|
+
|
|
803
855
|
# --- the object form of tool_called / tool_not_called, and transcript_* values shaped like a command ---
|
|
804
856
|
|
|
805
857
|
# `transcript_*` reads top-level assistant prose ONLY — never a tool_use — so a value shaped like a shell
|
|
@@ -1237,6 +1289,8 @@ def lint_doc(doc, path, raw_lines, cassette_records=None):
|
|
|
1237
1289
|
# linter stays offline — the message carries the gate fact instead of reading a baseline).
|
|
1238
1290
|
findings.extend(_lint_tool_call_object_form(items, fidelity, path))
|
|
1239
1291
|
findings.extend(_lint_transcript_command_shaped(items, path))
|
|
1292
|
+
# W: a `/<skill>` prompt never produces the `Skill` fork tool result — see _lint_slash_prompt_forked_anchor.
|
|
1293
|
+
findings.extend(_lint_slash_prompt_forked_anchor(doc, items, path))
|
|
1240
1294
|
if "transcript_no_host_path" in assert_keys:
|
|
1241
1295
|
if fidelity in ("hostloop", "protocol"):
|
|
1242
1296
|
findings.append(
|
|
@@ -1955,6 +2009,7 @@ LINT_RULES = {
|
|
|
1955
2009
|
"reference-access-contradiction": "ERROR",
|
|
1956
2010
|
"regex-double-quoted": "WARN",
|
|
1957
2011
|
"replay-noop": "WARN",
|
|
2012
|
+
"slash-prompt-forked-result-anchor": "WARN",
|
|
1958
2013
|
"tool-called-always-passes": "INFO",
|
|
1959
2014
|
"tool-input-regex-redactable": "WARN",
|
|
1960
2015
|
"tool-input-shell-tier": "INFO",
|
|
@@ -3065,7 +3120,7 @@ def _lint_skill_corpus_size(md_path):
|
|
|
3065
3120
|
#
|
|
3066
3121
|
# Body cap only: the re-attach estimate counts UTF-16 units and the rule counts UTF-8 BYTES, which are never
|
|
3067
3122
|
# fewer, so it errs early. The reference cap has no such guarantee — it rests on the measured ratio above.
|
|
3068
|
-
_SKILL_SIZE_CAPS_VERIFIED = "binary-verified against agent 2.1.
|
|
3123
|
+
_SKILL_SIZE_CAPS_VERIFIED = "binary-verified against agent 2.1.284 (VM ELF and native, both read)"
|
|
3069
3124
|
_SKILL_BODY_REATTACH_CAP = 19_000
|
|
3070
3125
|
_SKILL_BODY_NOTICE_RATIO = 0.8
|
|
3071
3126
|
_SKILL_REFERENCE_READ_CAP = 60_000
|