cowork-harness 1.13.0 → 1.13.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude/skills/cowork-harness/SKILL.md +14 -5
- package/.claude/skills/cowork-harness/references/ci-recipe.md +6 -6
- package/.claude/skills/cowork-harness/references/critique.md +23 -1
- package/.claude/skills/cowork-harness/references/fidelity-and-answers.md +4 -2
- package/.claude/skills/cowork-harness/references/scenario-schema.md +1 -1
- package/.claude/skills/cowork-harness/references/task-recipes.md +2 -2
- package/.claude/skills/cowork-harness/scripts/scenario.py +70 -0
- package/CHANGELOG.md +82 -0
- package/README.md +6 -6
- package/dist/cli.js +9 -0
- package/dist/run/envelope.js +63 -3
- package/docs/critique.md +8 -1
- package/docs/debugging.md +6 -0
- package/examples/replays/README.md +1 -1
- package/package.json +1 -1
- package/scripts/check-skill-doc-links.ts +58 -25
|
@@ -3,8 +3,8 @@ name: cowork-harness
|
|
|
3
3
|
description: Test or debug a Claude Code skill/plugin under Claude Cowork's runtime — sandboxed agent, default-deny egress, the can_use_tool permission/question protocol — using the cowork-harness CLI. Use when validating or regression-testing a skill, authoring or debugging a scenario YAML (prompt + scripted answers + assert:), choosing a fidelity tier, scripting AskUserQuestion / tool-permission answers, or asserting artifacts, egress, or sub-agent dispatch. Especially when a harness run no-ops an assertion, fails on an unanswered gate, false-greens, a steered answer never reaches the model, or a web_fetch is unexpectedly denied or gated. Also when iterating or hardening a skill across fixes, or grounding a skill's self-critique against its own run evidence — including a document-analysis skill (cap table, deck, financial model, transcript) that needs an uploaded file attached to be critiqued at all. NOT for generic unit testing (pytest/vitest of your own scripts) or non-Cowork CI. Covers the skill / run / chat / record / replay / trace / decide / assertions / scaffold commands and the session-vs-scenario split.
|
|
4
4
|
metadata:
|
|
5
5
|
author: cowork-harness
|
|
6
|
-
version: 1.13.
|
|
7
|
-
tracks-harness: cowork-harness 1.13.
|
|
6
|
+
version: 1.13.2
|
|
7
|
+
tracks-harness: cowork-harness 1.13.2 (baseline desktop-1.24012.9)
|
|
8
8
|
---
|
|
9
9
|
|
|
10
10
|
# cowork-harness
|
|
@@ -22,7 +22,7 @@ flagged with a loud `::warning::`, not silent — auto-answer a gate, observe an
|
|
|
22
22
|
allowlist). This skill exists mostly to keep you out of those traps — the Gotchas section below is
|
|
23
23
|
the highest-value part. Read it.
|
|
24
24
|
|
|
25
|
-
> **Version note:** the facts and `file:line` pointers here track `cowork-harness 1.13.
|
|
25
|
+
> **Version note:** the facts and `file:line` pointers here track `cowork-harness 1.13.2` (baseline
|
|
26
26
|
> `desktop-1.24012.9`). If your checkout is newer, prefer the live `--help` and — in a repo checkout —
|
|
27
27
|
> `SPEC.md` / `docs/*.md` over this snapshot, and re-run the bundled linter.
|
|
28
28
|
|
|
@@ -39,7 +39,7 @@ Before the first command, confirm the CLI is reachable and **fail loud** (never
|
|
|
39
39
|
|
|
40
40
|
- **One-shot check.** Run `cowork-harness doctor [--tier <tier>]` first — a read-only prerequisite check that inspects Docker, the staged agent, the token, and the baseline in one pass. The bullets below explain each thing it checks (and how to fix it).
|
|
41
41
|
- **Replay-only? Skip `doctor`.** Replaying committed cassettes needs no Docker, no staged agent, and no token — and every tier's `doctor` validates the auth token (the live tiers also Docker + the staged agent), so a ✗ there is expected, not a blocker. Go straight to `cowork-harness replay <cassette>`.
|
|
42
|
-
- **CLI on PATH, recent enough?** Run `cowork-harness --version` — this skill needs **≥ 1.13.
|
|
42
|
+
- **CLI on PATH, recent enough?** Run `cowork-harness --version` — this skill needs **≥ 1.13.2**. If it's missing or older, prefix every command with the version floor `npx "cowork-harness@>=1.13.2" <cmd>` (Node ≥ 20), or install once with `npm i -g "cowork-harness@>=1.13.2"`. **Pin `@>=1.13.2`, never `@latest`** — `@latest` can silently fetch an older CLI and the new commands fail as "unknown command", whereas the floor **fails loud** if no compatible version is published.
|
|
43
43
|
|
|
44
44
|
This skill documents the CURRENT surface, not release history. If `cowork-harness --version` is
|
|
45
45
|
OLDER than the floor, the per-release record of what you are missing is [CHANGELOG.md](https://github.com/yaniv-golan/cowork-harness/blob/main/CHANGELOG.md)
|
|
@@ -63,6 +63,12 @@ reproducible regression (Part II), and **debug** a run that misbehaved or greene
|
|
|
63
63
|
- **Regression-test your skill's ANSWER quality** (not just its behavior — does its guidance still lead to
|
|
64
64
|
correct answers after you edit it?) → author `semantic_matches` scenarios and gate on the per-claim
|
|
65
65
|
profile. See **Recipe 5** in `references/task-recipes.md` (validity, N≥3, discrimination — the traps).
|
|
66
|
+
- **"What is WRONG with this skill?"** (a graded critique, not a pass/fail) → `cowork-harness critique
|
|
67
|
+
<folder> --prompt "<probe>"`. Four model workloads and 10–20 minutes; budget from
|
|
68
|
+
`report.costUsd.totalUsd`. Reach for it when you want **findings**. **For "what does this skill
|
|
69
|
+
**DO**" — routing, artifact location, narration — use `skill` instead**: no evaluator, a fraction of
|
|
70
|
+
the cost, and it answers that question directly. Report and evidence-package shapes:
|
|
71
|
+
`references/critique.md`.
|
|
66
72
|
- **A run failed — or greened and you don't trust it** (the debugging loop) → don't re-run and hope.
|
|
67
73
|
The run already wrote its evidence to a **kept run dir** (`~/.cowork-harness/runs/…`; `--keep` prints
|
|
68
74
|
the path, `trace <run-id>` finds it). **Localize the failure post-hoc** from that evidence:
|
|
@@ -71,6 +77,9 @@ reproducible regression (Part II), and **debug** a run that misbehaved or greene
|
|
|
71
77
|
the loop 0.32.0's observability is built for; the *Triage* and *Inspecting a run's observability
|
|
72
78
|
output* sections in **Part III — Debug** are the detail (the fuller human-facing map lives in
|
|
73
79
|
[`docs/debugging.md`](https://github.com/yaniv-golan/cowork-harness/blob/main/docs/debugging.md) — repo-only, not shipped with the installed skill).
|
|
80
|
+
**"Evidence" here means the RUN's own record** — events, trace, transcript. `critique`'s evaluator
|
|
81
|
+
grades against a different artifact, `critique-evidence-package.txt`, which none of these tools
|
|
82
|
+
surface; see `references/critique.md`.
|
|
74
83
|
- **Multi-turn / interactive reproduction** → `cowork-harness chat` (interactive; gates answered at the
|
|
75
84
|
TTY, **not** an asserted test — see *Debugging with `chat`* in **Part III — Debug**).
|
|
76
85
|
|
|
@@ -515,7 +524,7 @@ decide which assertions from *Assertions: two orthogonal axes* are worth adding)
|
|
|
515
524
|
`outputTruncated` on a matched tool result). Three separately-shaped rollups, easy to conflate in a
|
|
516
525
|
`jq` recipe: `toolCounts` is a flat `{tool: number}` call-count map, `toolErrors` is
|
|
517
526
|
`{tool: {calls, errors}}`, and `toolDurations` is `{tool: {calls, totalMs, maxMs}}`. (Full per-field
|
|
518
|
-
semantics: the README's "Observability fields" section — repo-only; `schema/run-result.json` is the
|
|
527
|
+
semantics: the README's "Observability fields" section — repo-only; [`schema/run-result.json`](https://github.com/yaniv-golan/cowork-harness/blob/main/schema/run-result.json) is the
|
|
519
528
|
machine source.)
|
|
520
529
|
- **Opaque failure?** A failed run also records **`errorSource`** (where the failure originated) and
|
|
521
530
|
**`stderrLogPath`** (the captured agent stderr) — read those and `trace <run-dir>` *before* re-running;
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# CI recipe — replay vs live lanes
|
|
2
2
|
|
|
3
|
-
Self-contained reference. Tracks `cowork-harness 1.13.
|
|
3
|
+
Self-contained reference. Tracks `cowork-harness 1.13.2` (baseline `desktop-1.24012.9`).
|
|
4
4
|
|
|
5
5
|
**Fastest path: the packaged Action.** One step gets you `replay`/`lint`/`verify-cassettes` plus a PR
|
|
6
6
|
job-summary reporter (verdict table, staleness findings, cost/turns when available):
|
|
@@ -13,7 +13,7 @@ job-summary reporter (verdict table, staleness findings, cost/turns when availab
|
|
|
13
13
|
```
|
|
14
14
|
|
|
15
15
|
The Action's `version` input defaults to `latest` — intentional so a copy-pasted recipe tracks the current
|
|
16
|
-
release; pin an exact version (e.g. `version: "1.13.
|
|
16
|
+
release; pin an exact version (e.g. `version: "1.13.2"`) for reproducible CI.
|
|
17
17
|
|
|
18
18
|
Reach for the manual multi-step form below only when you need per-step control the Action's inputs don't
|
|
19
19
|
cover (a custom flag combination, a different runner matrix per step, or `lint`/`verify-cassettes` gated
|
|
@@ -58,7 +58,7 @@ sha256-*checked* but not hard-blocking on mismatch — it's advisory for an inte
|
|
|
58
58
|
GitHub-hosted runners, no token/Docker/agent:
|
|
59
59
|
|
|
60
60
|
```yaml
|
|
61
|
-
- run: npm i -g "cowork-harness@>=1.13.
|
|
61
|
+
- run: npm i -g "cowork-harness@>=1.13.2"
|
|
62
62
|
- run: cowork-harness lint scenarios/*.yaml # no silent false-greens
|
|
63
63
|
- run: cowork-harness verify-cassettes cassettes/ # privacy + staleness
|
|
64
64
|
- run: cowork-harness replay cassettes/ # token-free content/structure
|
|
@@ -222,7 +222,7 @@ jobs:
|
|
|
222
222
|
with: { node-version: '20' }
|
|
223
223
|
- uses: actions/setup-python@v5
|
|
224
224
|
with: { python-version: '3.x' } # python3 only — PyYAML is bundled with the linter
|
|
225
|
-
- run: npm i -g "cowork-harness@>=1.13.
|
|
225
|
+
- run: npm i -g "cowork-harness@>=1.13.2"
|
|
226
226
|
- run: cowork-harness lint scenarios/*.yaml # no-silent-false-green (needs python3; PyYAML bundled)
|
|
227
227
|
- run: cowork-harness verify-cassettes cassettes/ --output-format json # privacy + staleness gate
|
|
228
228
|
- run: cowork-harness replay cassettes/ --output-format json # token-free content/structure
|
|
@@ -251,7 +251,7 @@ jobs:
|
|
|
251
251
|
echo "live=true" >> "$GITHUB_OUTPUT"
|
|
252
252
|
fi
|
|
253
253
|
- if: steps.guard.outputs.live == 'true'
|
|
254
|
-
run: npm i -g "cowork-harness@>=1.13.
|
|
254
|
+
run: npm i -g "cowork-harness@>=1.13.2"
|
|
255
255
|
- if: steps.guard.outputs.live == 'true'
|
|
256
256
|
run: cowork-harness run scenarios/ --output-format json
|
|
257
257
|
env:
|
|
@@ -280,7 +280,7 @@ JSON.
|
|
|
280
280
|
|
|
281
281
|
`verify-cassettes` emits its **own** envelope (`{command, ok, coverage, results[]}` with per-file
|
|
282
282
|
`findings`/`staleness`/`unverifiable`/`notes`/`version`/`error`), published as
|
|
283
|
-
`schema/verify-cassettes.json` in the npm package. `ok:false` doesn't say *why* — read the buckets, or
|
|
283
|
+
`schema/verify-cassettes.json` in the npm package. `ok:false` doesn't say *why* — read the buckets, or <!-- npm-only-ok -->
|
|
284
284
|
the exit code (`1` = `findings`/`staleness`/`scenarioDrift` populated, a real problem verified & found;
|
|
285
285
|
`3` = `unverifiable`/`version`/`error` populated, verification could not complete; a real finding wins
|
|
286
286
|
`1` if both are present). Both envelope schemas are covered 1.0 contract surfaces (SPEC §12) — parse the
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# Critique — the facts a plugin install can't otherwise reach
|
|
2
2
|
|
|
3
|
-
Tracks `cowork-harness 1.13.
|
|
3
|
+
Tracks `cowork-harness 1.13.2` (baseline `desktop-1.24012.9`). This is **not** a trim of the full
|
|
4
4
|
[`docs/critique.md`](https://github.com/yaniv-golan/cowork-harness/blob/main/docs/critique.md) (repo-only —
|
|
5
5
|
flags, cost, reproduction discipline, known limitations all live there). This file covers exactly what a
|
|
6
6
|
plugin install cannot otherwise discover: the run-dir artifact a harvester actually reads, the report's
|
|
@@ -51,6 +51,14 @@ named file and byte counts are reported, never silent, never refused. The **tran
|
|
|
51
51
|
separately at **128 KiB**, cut **head+tail with an elided middle**, so a run's setup and its conclusion
|
|
52
52
|
both survive a cut instead of just one end.
|
|
53
53
|
|
|
54
|
+
You do not need a paid run to find out where you stand: **`cowork-harness lint-skill <skill-dir>` sizes
|
|
55
|
+
your corpus against the same ceiling**, reporting `skill-corpus-near-evidence-ceiling` (INFO) from 80%
|
|
56
|
+
and `skill-corpus-over-evidence-ceiling` (WARN, so it fails `--strict`) past it. It counts the three
|
|
57
|
+
classes the ceiling governs — `SKILL.md`, every file under `references/` (**any extension**: the packager
|
|
58
|
+
applies no extension filter, so JSON schemas and rule packs count toward your total), and a plugin
|
|
59
|
+
skill's `agents/<name>.md`. It does not apply staging's git-tracked filter, so an untracked reference
|
|
60
|
+
inflates the figure; `corpusCuts` below stays the authority.
|
|
61
|
+
|
|
54
62
|
The report's `evidenceBudget` object says exactly what was shown — read it instead of inferring budgets
|
|
55
63
|
from `dist/` source:
|
|
56
64
|
|
|
@@ -70,3 +78,17 @@ file under the skill folder was never in the agent's mount, so grading against i
|
|
|
70
78
|
false `already-covered` verdict. It is named in `corpusExcluded` instead, and an untracked `SKILL.md`
|
|
71
79
|
specifically reports `skillMdStatus: "untracked"`, forcing the mechanical `already-covered` →
|
|
72
80
|
`not-adjudicable` downgrade. `git add` it (or commit before critiquing) if it should count as evidence.
|
|
81
|
+
|
|
82
|
+
## `referencesRead` is main-agent-only — `noSkillFilesRead` is not
|
|
83
|
+
|
|
84
|
+
`result.json`'s top-level `referencesRead` lists **main-agent Reads only**. A dispatcher-style skill does
|
|
85
|
+
its reading one level down, and those Reads live under `subagents[].referencesRead` — so an empty
|
|
86
|
+
top-level list on a sub-agent-heavy run is not evidence the material went unread. The critique report's
|
|
87
|
+
**`noSkillFilesRead`** unions both, which is why it is the signal to read.
|
|
88
|
+
|
|
89
|
+
It is stated **observationally** and must be rendered that way: the predicate matches `references/` and
|
|
90
|
+
`scripts/` only — never `assets/`, never `SKILL.md` (delivered whole, never Read as a file) — and keys on
|
|
91
|
+
the `Read` **tool**. A skill that reached its material with `Grep`, or kept it under `assets/`, reports
|
|
92
|
+
`true` having demonstrably done the work. `undefined` has **two** causes and neither means "nothing was read": a degraded turn-1 result
|
|
93
|
+
(genuinely unknown), or a skill that ships no `references/` and no `scripts/` at all — there is nothing
|
|
94
|
+
for the signal to be about, so emitting it would be noise about material that does not exist.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# Fidelity tiers & answer paths
|
|
2
2
|
|
|
3
|
-
Self-contained reference. Tracks `cowork-harness 1.13.
|
|
3
|
+
Self-contained reference. Tracks `cowork-harness 1.13.2` (baseline `desktop-1.24012.9`).
|
|
4
4
|
|
|
5
5
|
## Fidelity tiers (`fidelity:` in the scenario)
|
|
6
6
|
|
|
@@ -171,10 +171,12 @@ catch a non-matching regex or a wire-protocol bug before twelve minutes of live
|
|
|
171
171
|
cowork-harness decide \
|
|
172
172
|
--question "Which output format do you want?" \
|
|
173
173
|
--option Markdown --option PDF \
|
|
174
|
-
--answer-policy
|
|
174
|
+
--answer-policy ./answer-policy.yaml
|
|
175
175
|
# ✓ rule matched: "Which output format do you want?" → "Markdown"
|
|
176
176
|
```
|
|
177
177
|
|
|
178
|
+
A ready-made sample to copy: [`examples/answer-policies/demo.yaml`](https://github.com/yaniv-golan/cowork-harness/blob/main/examples/answer-policies/demo.yaml).
|
|
179
|
+
|
|
178
180
|
It works with `--answer`/`--answer-policy` (reports which rule matched, or exits non-zero if none),
|
|
179
181
|
`--decider-cmd` (shows the exact request + answer), or `--decider-llm` (a live model answers).
|
|
180
182
|
Caveat: `decide` only builds a **single-select** sample (set choices with `--option`; there is no
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# Scenario & session schema, assertion catalog, web_fetch, full gotchas
|
|
2
2
|
|
|
3
|
-
Self-contained reference for authoring `cowork-harness` scenarios. Tracks `cowork-harness 1.13.
|
|
3
|
+
Self-contained reference for authoring `cowork-harness` scenarios. Tracks `cowork-harness 1.13.2`
|
|
4
4
|
(baseline `desktop-1.24012.9`). If your checkout is newer, prefer the live [`docs/scenario.md`](https://github.com/yaniv-golan/cowork-harness/blob/main/docs/scenario.md),
|
|
5
5
|
[`docs/session.md`](https://github.com/yaniv-golan/cowork-harness/blob/main/docs/session.md), and `SPEC.md`.
|
|
6
6
|
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
Each recipe composes facts that live scattered across SKILL.md and the other references into one
|
|
4
4
|
decision path. Every one answers a question a real fleet owner had to work out the hard way.
|
|
5
|
-
Tracks `cowork-harness 1.13.
|
|
5
|
+
Tracks `cowork-harness 1.13.2` (baseline `desktop-1.24012.9`), same as SKILL.md's front-matter. Recipe 2's `resolved-tier`/`unverifiable-tier` staleness classes and
|
|
6
6
|
Recipe 3's `init-redact` shipped in 0.24.0 and are part of the current feature set — no version gate
|
|
7
7
|
needed if your CLI meets SKILL.md's version floor.
|
|
8
8
|
|
|
@@ -48,7 +48,7 @@ what production would do. Two layers of defense:
|
|
|
48
48
|
|
|
49
49
|
### Cassette anatomy (what you're looking at when you open one)
|
|
50
50
|
|
|
51
|
-
Top-level fields of a `*.cassette.json` (schema `schema/cassette.v10.json`):
|
|
51
|
+
Top-level fields of a `*.cassette.json` (schema [`schema/cassette.v10.json`](https://github.com/yaniv-golan/cowork-harness/blob/main/schema/cassette.v10.json)):
|
|
52
52
|
|
|
53
53
|
| Field | What it is |
|
|
54
54
|
|---|---|
|
|
@@ -1197,6 +1197,75 @@ def _resolve_skill_targets(arg):
|
|
|
1197
1197
|
return None, []
|
|
1198
1198
|
|
|
1199
1199
|
|
|
1200
|
+
# Mirrors SKILL_CORPUS_CEILING in the packager (src/critique/package-evidence.ts). A cross-language
|
|
1201
|
+
# test pins the two together, so change both or the test fails.
|
|
1202
|
+
_EVIDENCE_CORPUS_CEILING = 512 * 1024
|
|
1203
|
+
# Warn well before the valve: a corpus this close is one reference file away from being cut mid-grade.
|
|
1204
|
+
_EVIDENCE_CORPUS_NOTICE_RATIO = 0.8
|
|
1205
|
+
|
|
1206
|
+
|
|
1207
|
+
def _lint_skill_corpus_size(md_path):
|
|
1208
|
+
"""Total skill-authored bytes against the critique evidence ceiling.
|
|
1209
|
+
|
|
1210
|
+
Counts the SAME THREE CLASSES the ceiling governs: SKILL.md, every file under references/ (any
|
|
1211
|
+
extension -- the packager applies no extension filter, so JSON schemas and rule packs count), and,
|
|
1212
|
+
for a skill inside a multi-skill plugin, the invoked skill's <root>/agents/<name>.md.
|
|
1213
|
+
|
|
1214
|
+
Omitting the agents md is not a rounding error: a plugin whose SKILL.md + references sit in the INFO
|
|
1215
|
+
band while the agents md carries the corpus past the ceiling reported INFO and PASSED --strict on
|
|
1216
|
+
content the packager would cut. A proximity check that greens a corpus destined to be cut is worse
|
|
1217
|
+
than no check.
|
|
1218
|
+
|
|
1219
|
+
Still approximate in ONE direction only, and it now over- rather than under-counts: the packager
|
|
1220
|
+
applies staging's git-tracked filter, so an untracked reference inflates this figure. That errs
|
|
1221
|
+
toward warning early. The report's corpusCuts stays the authority."""
|
|
1222
|
+
skill_dir = Path(md_path).parent
|
|
1223
|
+
total = 0
|
|
1224
|
+
files = [Path(md_path)]
|
|
1225
|
+
refs = skill_dir / "references"
|
|
1226
|
+
if refs.is_dir():
|
|
1227
|
+
files.extend(p for p in sorted(refs.rglob("*")) if p.is_file())
|
|
1228
|
+
# Multi-skill plugin layout: skillDir is <root>/skills/<name> and the invoked skill's sub-agent
|
|
1229
|
+
# system prompt is <root>/agents/<name>.md -- the same resolution the critique command performs.
|
|
1230
|
+
# A standalone skill (no `skills/` parent) has no agents md and is unaffected.
|
|
1231
|
+
if skill_dir.parent.name == "skills":
|
|
1232
|
+
agents_md = skill_dir.parent.parent / "agents" / f"{skill_dir.name}.md"
|
|
1233
|
+
if agents_md.is_file():
|
|
1234
|
+
files.append(agents_md)
|
|
1235
|
+
for p in files:
|
|
1236
|
+
try:
|
|
1237
|
+
total += p.stat().st_size
|
|
1238
|
+
except OSError:
|
|
1239
|
+
continue # unreadable file: same posture as the packager's per-file degrade
|
|
1240
|
+
pct = total * 100.0 / _EVIDENCE_CORPUS_CEILING
|
|
1241
|
+
if total > _EVIDENCE_CORPUS_CEILING:
|
|
1242
|
+
return [
|
|
1243
|
+
Finding(
|
|
1244
|
+
"WARN",
|
|
1245
|
+
"skill-corpus-over-evidence-ceiling",
|
|
1246
|
+
f"skill content is {total:,} B ({pct:.0f}% of the {_EVIDENCE_CORPUS_CEILING:,} B critique "
|
|
1247
|
+
f"evidence ceiling) — a critique will cut it before grading.",
|
|
1248
|
+
"Split or trim the largest references/ files. This counts SKILL.md + references/** + "
|
|
1249
|
+
"agents/<skill>.md, the same three classes the ceiling governs; it does not apply "
|
|
1250
|
+
"staging's git-tracked filter, so an untracked reference inflates it. The critique "
|
|
1251
|
+
"report's corpusCuts names exactly which files lose bytes.",
|
|
1252
|
+
str(skill_dir),
|
|
1253
|
+
)
|
|
1254
|
+
]
|
|
1255
|
+
if total >= _EVIDENCE_CORPUS_CEILING * _EVIDENCE_CORPUS_NOTICE_RATIO:
|
|
1256
|
+
return [
|
|
1257
|
+
Finding(
|
|
1258
|
+
"INFO",
|
|
1259
|
+
"skill-corpus-near-evidence-ceiling",
|
|
1260
|
+
f"skill content is {total:,} B ({pct:.0f}% of the {_EVIDENCE_CORPUS_CEILING:,} B critique "
|
|
1261
|
+
f"evidence ceiling).",
|
|
1262
|
+
"No action needed yet; adding a large reference file would push a critique into cutting content.",
|
|
1263
|
+
str(skill_dir),
|
|
1264
|
+
)
|
|
1265
|
+
]
|
|
1266
|
+
return []
|
|
1267
|
+
|
|
1268
|
+
|
|
1200
1269
|
def cmd_lint_skill(args):
|
|
1201
1270
|
all_findings = []
|
|
1202
1271
|
n_files = 0
|
|
@@ -1218,6 +1287,7 @@ def cmd_lint_skill(args):
|
|
|
1218
1287
|
md_lines = Path(md).read_text(encoding="utf-8").splitlines()
|
|
1219
1288
|
all_findings.extend(_lint_skill_text(md, md_lines))
|
|
1220
1289
|
all_findings.extend(_lint_subagent_types(md, md_lines))
|
|
1290
|
+
all_findings.extend(_lint_skill_corpus_size(md))
|
|
1221
1291
|
for hp in hooks:
|
|
1222
1292
|
n_files += 1
|
|
1223
1293
|
all_findings.extend(
|
package/CHANGELOG.md
CHANGED
|
@@ -6,6 +6,88 @@ All notable changes to this project are documented here. The format is based on
|
|
|
6
6
|
|
|
7
7
|
## [Unreleased]
|
|
8
8
|
|
|
9
|
+
## [1.13.2] — 2026-07-28
|
|
10
|
+
|
|
11
|
+
### Fixed
|
|
12
|
+
|
|
13
|
+
- **`lint-skill`'s corpus check sized a smaller set than the ceiling it warns about, and could pass
|
|
14
|
+
`--strict` on a corpus a critique would cut.** The evidence ceiling governs `SKILL.md` + `references/**`
|
|
15
|
+
+ `agents/<skill>.md` **combined**; the check summed only the first two. A multi-skill plugin whose
|
|
16
|
+
SKILL.md and references sat in the INFO band while its `agents/<name>.md` carried the total past the
|
|
17
|
+
ceiling reported INFO and **exited 0 under `--strict`** — a green gate on content that was already
|
|
18
|
+
destined to be cut. The check now counts all three classes, resolving `<root>/agents/<name>.md` the same
|
|
19
|
+
way `critique --skill <name>` does; a standalone skill (no `skills/` parent) is unaffected.
|
|
20
|
+
|
|
21
|
+
Two clarifications that follow, because both were understated: every file under `references/` counts
|
|
22
|
+
regardless of extension — the packager applies **no** extension filter, so JSON schemas and rule packs
|
|
23
|
+
are part of your corpus — and the remaining approximation now errs in one direction only. The check
|
|
24
|
+
skips staging's git-tracked filter, so an untracked reference inflates the figure and warns early
|
|
25
|
+
rather than late. `corpusCuts` in the report remains the authority.
|
|
26
|
+
|
|
27
|
+
Reported by a consumer, who caught it by measuring their own plugin against the new check on upgrade.
|
|
28
|
+
|
|
29
|
+
## [1.13.1] — 2026-07-28
|
|
30
|
+
|
|
31
|
+
### Fixed
|
|
32
|
+
|
|
33
|
+
- **A misplaced global flag written as `--dotenv=<path>` / `--run-dir=<path>` keeps its exact-fix hint.**
|
|
34
|
+
The pre-dispatch guard matches the spaced form by exact token — deliberately, so a per-command value
|
|
35
|
+
like `--answer "--dotenv=x=foo"` is never hijacked — which left the equals form to surface as a bare
|
|
36
|
+
`unknown flag`, dropping the one thing the caller needed: where to put it instead. The hint is derived
|
|
37
|
+
where usage errors are rendered, and `run` additionally rejects a stray global ahead of its scenario
|
|
38
|
+
path — that ordering previously absorbed the flag as the target and reported the user's *correct* path
|
|
39
|
+
as the unexpected argument. The trigger is anchored on the two message shapes that mean "parsed as an
|
|
40
|
+
unrecognized flag": a broader match fires on errors about a *correctly-placed* leading flag
|
|
41
|
+
(`--dotenv file not found`, `--dotenv requires a path`) and on a flag name inside a quoted value,
|
|
42
|
+
telling the user to move a flag that is already in the right place.
|
|
43
|
+
|
|
44
|
+
Coverage is most of the CLI, not all of it. A command that never reaches that rendering point
|
|
45
|
+
(`critique`, `chat`, `prune`, `migrate-run-dir`) or whose usage message is shaped differently (`vm`, a
|
|
46
|
+
bare `assertions`, and `lint`/`lint-skill`, which forward the token to Python argparse) keeps its own
|
|
47
|
+
bare `unknown flag` message.
|
|
48
|
+
|
|
49
|
+
- **The shipped-skill pointer guard covers `schema/` and `examples/`, not `docs/` alone.** A plugin
|
|
50
|
+
install materializes only `.claude/skills/<name>/**`, so a bare pointer to any other repo directory
|
|
51
|
+
dangles the same way — including the one naming the machine-readable schema a consumer needs to read
|
|
52
|
+
field semantics. Four live pointers rewritten to permalinks or made runnable. `scripts/`, `cassettes/`
|
|
53
|
+
and `src/` stay out of scope: the first resolves inside the payload, the second names the reader's own
|
|
54
|
+
directory in a recipe, the third is provenance. A line whose sentence already qualifies a pointer as
|
|
55
|
+
npm-only opts out with an explicit `<!-- npm-only-ok -->` marker.
|
|
56
|
+
|
|
57
|
+
- **Pointer-guard and hint boundary cases.** The guard matched a path segment that was only the tail of
|
|
58
|
+
a longer word (`mydocs/x.md`), reported a wrong target for an extension outside its set (`docs/x.mdx`
|
|
59
|
+
as `docs/x.md`), missed an upper-case extension, and rejected any absolute URL that was not a GitHub
|
|
60
|
+
*blob* permalink — so an equally-resolvable `raw.githubusercontent` / GitLab / GitHub-`tree` link was
|
|
61
|
+
reported as a dead relative pointer, the checker refusing its own prescribed fix written another way.
|
|
62
|
+
The opt-out marker's whole-line scope is a deliberate trade-off and is now pinned by a test. Separately,
|
|
63
|
+
`.` is no longer a token terminator in the misplaced-flag match: a typo'd filename (`--dotenv.yaml`) was
|
|
64
|
+
read as the flag and answered with "move it before the subcommand".
|
|
65
|
+
|
|
66
|
+
### Added
|
|
67
|
+
|
|
68
|
+
- **`lint-skill` flags a skill corpus at or over the 512 KiB critique evidence ceiling** —
|
|
69
|
+
`skill-corpus-near-evidence-ceiling` (INFO, ≥ 80%) and `skill-corpus-over-evidence-ceiling` (WARN), so
|
|
70
|
+
the fact is free and static instead of costing a paid critique. The figure approximates the packager's
|
|
71
|
+
corpus — it excludes `agents/<skill>.md` and does not apply the tracked-set filter — which the finding
|
|
72
|
+
text states; `corpusCuts` in the report remains the authority. **CI note:** the over-ceiling finding is a
|
|
73
|
+
WARN, so a `lint-skill --strict` job that was green can now fail — which is the point, since that corpus
|
|
74
|
+
is one a critique would cut before grading.
|
|
75
|
+
|
|
76
|
+
### Documentation
|
|
77
|
+
|
|
78
|
+
- **`critique` is in the skill's orientation router, with the routing rule.** "What does this skill do"
|
|
79
|
+
is a `skill` question; "what is wrong with this skill" is a `critique` question, and the second costs
|
|
80
|
+
four model workloads. The router listed five entry points and `critique` was not among them.
|
|
81
|
+
Reported by a consumer.
|
|
82
|
+
|
|
83
|
+
- **`referencesRead` is main-agent-only, stated inside the plugin payload.** Sub-agent Reads live under
|
|
84
|
+
`subagents[].referencesRead`, so an empty top-level list on a dispatcher-style skill is not evidence
|
|
85
|
+
the material went unread; the critique report's `noSkillFilesRead` unions both. Reported by a consumer.
|
|
86
|
+
|
|
87
|
+
- **The run's own evidence and the critique evidence package are distinguished at the router and on the
|
|
88
|
+
debugging page**, not only mid-document — the first place a reader meets the word, rather than the
|
|
89
|
+
third. Reported by a consumer.
|
|
90
|
+
|
|
9
91
|
## [1.13.0] — 2026-07-28
|
|
10
92
|
|
|
11
93
|
**Upgrade notes.**
|
package/README.md
CHANGED
|
@@ -91,7 +91,7 @@ node dist/cli.js replay examples/replays/example-pdf-skill.cassette.json
|
|
|
91
91
|
|
|
92
92
|
> **Installed globally instead?** Once linked/installed, the same command is `cowork-harness replay
|
|
93
93
|
> <cassette>` — but the relative path above only resolves from a source checkout's `examples/replays/`.
|
|
94
|
-
> From a global install (`npm i -g "cowork-harness@>=1.13.
|
|
94
|
+
> From a global install (`npm i -g "cowork-harness@>=1.13.2"`), point at the package root instead:
|
|
95
95
|
> `cowork-harness replay "$(npm root -g)/cowork-harness/examples/replays/example-pdf-skill.cassette.json"`
|
|
96
96
|
> (or copy the cassette into your own project and pass that path).
|
|
97
97
|
|
|
@@ -101,7 +101,7 @@ Live `run`/`skill` need the prerequisites in the next section — note the `prot
|
|
|
101
101
|
> - **Replay only (zero setup):** `cowork-harness replay <cassette>` — no token, no Docker, no agent. The command above.
|
|
102
102
|
> - **`protocol` (real model, no Docker):** needs only the auth token (item 3 below).
|
|
103
103
|
> - **Live `container` / `microvm` / `hostloop` / `cowork`:** needs Docker (or Lima for `microvm`), a staged agent, and the token — run `cowork-harness doctor` first.
|
|
104
|
-
> - **Invocation:** from a source checkout, `node dist/cli.js <cmd>` (or `npm link` to get the `cowork-harness` command); from a global install, `cowork-harness <cmd>`; the companion skill falls back to `npx "cowork-harness@>=1.13.
|
|
104
|
+
> - **Invocation:** from a source checkout, `node dist/cli.js <cmd>` (or `npm link` to get the `cowork-harness` command); from a global install, `cowork-harness <cmd>`; the companion skill falls back to `npx "cowork-harness@>=1.13.2"`.
|
|
105
105
|
|
|
106
106
|
Two more worked examples worth knowing about: `examples/scenarios/protocol-smoke.yaml` (zero-Docker smoke
|
|
107
107
|
test) and `examples/scenarios/skill-loads.yaml` (container-tier acceptance check) — see
|
|
@@ -126,7 +126,7 @@ claude plugin marketplace add yaniv-golan/cowork-harness
|
|
|
126
126
|
claude plugin install cowork-harness@cowork-harness
|
|
127
127
|
```
|
|
128
128
|
|
|
129
|
-
The skill **self-bootstraps the CLI**: if `cowork-harness` isn't on your PATH it falls back to `npx "cowork-harness@>=1.13.
|
|
129
|
+
The skill **self-bootstraps the CLI**: if `cowork-harness` isn't on your PATH it falls back to `npx "cowork-harness@>=1.13.2"` (a version floor that fails loud rather than silently fetching a too-old CLI; Node ≥ 20). Tiers above `protocol` still need Docker/Lima and a Claude Desktop agent binary — see the prerequisites below.
|
|
130
130
|
|
|
131
131
|
It also follows the open [Agent Skills](https://agentskills.io) spec, so it installs cross-editor (Cursor, Codex, OpenCode, …) via [`npx skills`](https://github.com/vercel-labs/skills) (Vercel Labs' CLI implementation of that spec):
|
|
132
132
|
|
|
@@ -147,7 +147,7 @@ A global install is enough for CI `lint`, reading the teaching skill, and replay
|
|
|
147
147
|
To `run` the worked examples live or copy them as a starting point, use a source checkout. (The marketplace
|
|
148
148
|
skill install itself only pulls `.claude/skills/cowork-harness/` — SKILL.md + `references/` + `scenario.py`/
|
|
149
149
|
assertion keys, per `.claude-plugin/marketplace.json`'s `source` — not the rest of this table; the full set
|
|
150
|
-
above becomes available once the skill's first command self-bootstraps `npx "cowork-harness@>=1.13.
|
|
150
|
+
above becomes available once the skill's first command self-bootstraps `npx "cowork-harness@>=1.13.2"` — see
|
|
151
151
|
[above](#drive-it-from-claude-code-companion-skill) — which pulls the same npm package as the global-install row.)
|
|
152
152
|
|
|
153
153
|
### Prerequisites for anything above `protocol` fidelity
|
|
@@ -411,7 +411,7 @@ Skill testing is the headline use, but the tool is a general harness over the Co
|
|
|
411
411
|
| `scaffold <run-id>` | Turn a kept run into a starter scenario YAML (gates→answers, artifacts→`file_exists`) | authoring a scenario from a real run instead of guessing |
|
|
412
412
|
| `python3 …/scenario.py scaffold --name <name> --skill <dir>` | Generate a starter scenario skeleton from scratch (the `…` is `.claude/skills/cowork-harness/scripts/`) | starting a new scenario when you have no prior run |
|
|
413
413
|
| `lint <scenario.yaml \| dir/>…` | Check scenarios for silent false-greens — assertions placed on the wrong CI lane, mixed content/live keys, missing `controlOut`-required keys (files or a directory of `*.yaml`/`*.yml`; bundled `scenario.py`; needs python3 — PyYAML is bundled); `--json` emits findings as machine-readable JSON instead of the text report; `--min-severity ERROR|WARN|INFO` drops findings below a floor **before** both the report and the exit computation (so `--strict --min-severity ERROR` behaves as a plain lint), which is the way to mute the unconditional INFO advisories in CI | before committing a new scenario or after changing assertions |
|
|
414
|
-
| `lint-skill <SKILL.md \| skill-dir/>…` | Lint a skill body (and any sibling `hooks.json`) for two Cowork host-loop footguns — a `${CLAUDE_PLUGIN_ROOT}` path used in an in-VM bash context, and a hook command that exports an env var or writes into `/tmp` for the in-VM agent — plus static resolution of any pinned `subagent_type` against the enclosing plugin's `agents/*.md` (bundled `scenario.py`; needs python3); the two footguns are WARN-only, and of the three `subagent_type` outcomes, an in-plugin-prefixed agent missing from the enclosing plugin's fully-enumerated `agents/*.md` (`subagent-type-not-found-in-plugin`) is a **provable typo and is WARN too**, while a cross-plugin (`subagent-type-unresolvable`) or unknown-bare (`subagent-type-unknown`) value stays INFO (no built-in agent-type registry to disprove it against) — pass `--strict` (the CI-recommended invocation; plain `lint-skill` is advisory-only) to fail on any WARN, or `--json` for machine-readable findings | authoring or reviewing a skill before a paid Cowork host-loop run exposes the footgun, or before a pinned `subagent_type` typo breaks a `Task` dispatch |
|
|
414
|
+
| `lint-skill <SKILL.md \| skill-dir/>…` | Lint a skill body (and any sibling `hooks.json`) for two Cowork host-loop footguns — a `${CLAUDE_PLUGIN_ROOT}` path used in an in-VM bash context, and a hook command that exports an env var or writes into `/tmp` for the in-VM agent — plus static resolution of any pinned `subagent_type` against the enclosing plugin's `agents/*.md` (bundled `scenario.py`; needs python3); the two footguns are WARN-only, and of the three `subagent_type` outcomes, an in-plugin-prefixed agent missing from the enclosing plugin's fully-enumerated `agents/*.md` (`subagent-type-not-found-in-plugin`) is a **provable typo and is WARN too**, while a cross-plugin (`subagent-type-unresolvable`) or unknown-bare (`subagent-type-unknown`) value stays INFO (no built-in agent-type registry to disprove it against); it also sizes the skill's own content — `SKILL.md` + every file under `references/` (any extension) + a plugin skill's `agents/<name>.md`, the same three classes the ceiling governs — against the 512 KiB `critique` evidence ceiling — `skill-corpus-near-evidence-ceiling` at ≥ 80% is INFO, and `skill-corpus-over-evidence-ceiling` past it is **WARN**, so a corpus a critique would cut fails `--strict` — pass `--strict` (the CI-recommended invocation; plain `lint-skill` is advisory-only) to fail on any WARN, or `--json` for machine-readable findings | authoring or reviewing a skill before a paid Cowork host-loop run exposes the footgun, or before a pinned `subagent_type` typo breaks a `Task` dispatch |
|
|
415
415
|
| `python3 …/scenario.py resolve-agent-types <plugin-dir> [--json]` | Token-free: print a plugin's valid `<plugin>:<agent>` subagent types, resolved from `.claude-plugin/plugin.json` + `agents/*.md` frontmatter (filename-stem fallback when an agent file has no `name:`); the `…` is `.claude/skills/cowork-harness/scripts/` | "does `founder-skills:deck-review` resolve within this plugin?" without a live dispatch |
|
|
416
416
|
| `analyze-skill <SKILL.md \| skill-dir/ \| glob>…` | Token-free ADVISORY scan: flags a `/sessions/...` path handed to a file tool or used as a dispatch/sub-agent output path — that path class is DENIED on host-loop. Accepts multiple positionals (files, dirs, or a simple `*`/`**` glob), walked recursively across a plugin's `SKILL.md`/`agents/`/`references/`/`commands/`. Findings print but exit 0 by default; `--strict` (the CI-recommended invocation) fails on any unsuppressed finding. Three per-file ignore markers (`ignore-next-line`, `ignore-start`/`ignore-end`, file-wide `ignore`) and a `--output-format json` payload are supported. **It also flags interactive-artifact write-backs lost under Cowork**: it statically analyzes `.html`/`.js`/`.ts`/`.py` sources under the target for a relative `fetch`/XHR/`sendBeacon`/`<form method=post>` write-back that silently fails when the artifact is served from Cowork's own origin (`artifact-write-back-lost` gates under `--strict`; `artifact-write-back-suspect` is advisory; a candidate that can't be parsed is a could-not-verify exit `3`). `--runtime` adds an optional headless-DOM confirmation (needs `jsdom`) that *observes* the lost write-back. Full reference — ignore-marker syntax, directory-walk rules, JSON shape, and the artifact detector: [docs/subagents.md](./docs/subagents.md#static-path-fidelity-check-analyze-skill) | catching the exact "skill hands `/sessions/...` to a file tool" defect — or an artifact whose Submit silently fails under Cowork — statically, before paying for a live run to discover it |
|
|
417
417
|
| `probe-dispatch <skill-dir> "<prompt>"` | Cheap single-dispatch mechanics probe: a THIN wrapper over `skill` (fidelity `container`/`microvm`/`hostloop`, default `hostloop`) that scopes a prompt to trigger ONE `Task` dispatch, then prints just that dispatch's `{resolvedAgentType, pathDenials, delivered}` — no new data model, a pure projection of the same `RunResult` `skill` already produces; `--expect-write <suffix>` narrows `delivered`; "one dispatch" is prompt-scoped, not enforced (`--output-format json` for machine consumption); also inherits the common decider/answer flags — `--decider-cmd`, `--decider-dir`, `--on-unanswered`, `--ablate-skill` | "did THIS dispatch resolve to the type I expect, avoid a path denial, and actually deliver its write?" without hand-writing a scenario or reading `trace --view dispatches` |
|
|
@@ -694,7 +694,7 @@ jobs:
|
|
|
694
694
|
anthropic-api-key: ${{ secrets.ANTHROPIC_API_KEY }}
|
|
695
695
|
```
|
|
696
696
|
|
|
697
|
-
Every run writes a Markdown verdict table (scenario, pass/fail, signals, cost/turns when available, staleness findings, and the replay-skipped-assertions honesty line) to the job summary. Inputs: `command`, `path` (required), `version` (npm dist-tag/version, default `latest` — intentional so recipes track the current release; pin an exact version for reproducible CI. The companion skill's `cowork-harness@>=1.13.
|
|
697
|
+
Every run writes a Markdown verdict table (scenario, pass/fail, signals, cost/turns when available, staleness findings, and the replay-skipped-assertions honesty line) to the job summary. Inputs: `command`, `path` (required), `version` (npm dist-tag/version, default `latest` — intentional so recipes track the current release; pin an exact version for reproducible CI. The companion skill's `cowork-harness@>=1.13.2` floor guidance applies to ad-hoc CLI installs, not this input), `strict` (applies to `replay` (staleness findings), `lint`/`lint-skill` (WARN/INFO), and `analyze-skill` (any advisory finding); IGNORED — not forwarded — for `verify-cassettes`/`run`, which don't accept the flag), `fail-on-skill-drift` (**`replay`-only** — never forwarded to the analyzers), `extra-args`, `summary` (default `true`), `anthropic-api-key` (live lane only). See [`action.yml`](./action.yml) for the full input reference.
|
|
698
698
|
|
|
699
699
|
The provided [GitHub Actions workflow](.github/workflows/ci.yml) runs a **seven-stage pipeline**. The **build** + **test** stages are the token-free gate you can copy into your skill repo; the `action-self-test`, `python`, `boundary`, `scenarios`, and `parity-drift` stages are this repo's own fidelity self-tests and are not directly portable (they build the harness's Docker image and run harness-specific e2e scenarios — see [`ci-recipe.md`](./.claude/skills/cowork-harness/references/ci-recipe.md) for the skill-repo template):
|
|
700
700
|
|
package/dist/cli.js
CHANGED
|
@@ -1264,6 +1264,15 @@ async function cmdRun(rawArgs) {
|
|
|
1264
1264
|
const args = rawRest.filter((a) => a !== "--keep");
|
|
1265
1265
|
if (keepRequested)
|
|
1266
1266
|
log("note: `run` always keeps runs (under the runs root); --keep is a no-op here.");
|
|
1267
|
+
// A leftover global-only flag is NOT a positional. `takeCommonFlags` has already consumed every real
|
|
1268
|
+
// flag VALUE, so a `--dotenv=`/`--run-dir=` token surviving to here is provably a misplaced flag rather
|
|
1269
|
+
// than someone's argument. Letting it fall through made `args[0]` the FLAG and the user's real scenario
|
|
1270
|
+
// path the "unexpected argument(s)" — blaming the one token that was correct, and in the no-other-arg
|
|
1271
|
+
// case reporting `scenario path not found: --dotenv=…`. Emitting the unknown-flag SHAPE routes this
|
|
1272
|
+
// through fail()'s misplaced-global derivation instead of restating that sentence here.
|
|
1273
|
+
const strayGlobal = args.find((a) => /^--(dotenv|run-dir)(=|$)/.test(a));
|
|
1274
|
+
if (strayGlobal)
|
|
1275
|
+
fail("run", "usage", `unknown flag: ${strayGlobal}`, undefined, flags.output === "json");
|
|
1267
1276
|
const target = args[0];
|
|
1268
1277
|
if (!target)
|
|
1269
1278
|
fail("run", "usage", "usage: run <scenario.yaml | dir/>", undefined, flags.output === "json");
|
package/dist/run/envelope.js
CHANGED
|
@@ -135,6 +135,63 @@ export function isJsonOutput(args) {
|
|
|
135
135
|
}
|
|
136
136
|
return process.env.COWORK_HARNESS_OUTPUT_FORMAT === "json";
|
|
137
137
|
}
|
|
138
|
+
/** Flags the CLI honors ONLY in leading position, before the subcommand. */
|
|
139
|
+
const GLOBAL_ONLY_FLAGS = ["--dotenv", "--run-dir"];
|
|
140
|
+
/** The ONLY message shapes for which "you put a global flag after the subcommand" is a correct
|
|
141
|
+
* diagnosis. Anchored deliberately narrowly: the same usage exit also serves errors about a
|
|
142
|
+
* CORRECTLY-PLACED leading global (`--dotenv file not found: …`, `--dotenv requires a path …`) and
|
|
143
|
+
* errors where the flag name appears inside a QUOTED VALUE (`… got a flag-looking token
|
|
144
|
+
* "--dotenv=/x"`). Matching on "the message mentions --dotenv" fires on all of those and tells the user
|
|
145
|
+
* to move a flag that is already in the right place. */
|
|
146
|
+
const UNRECOGNIZED_FLAG_SHAPES = [/^unknown flag:/, /^unexpected argument\(s\):/];
|
|
147
|
+
/** Recover the "put it before the subcommand" guidance for a misplaced global flag written in the
|
|
148
|
+
* `--flag=value` form.
|
|
149
|
+
*
|
|
150
|
+
* The pre-dispatch guard in cli.ts matches the SPACED form by EXACT token and deliberately does not
|
|
151
|
+
* match `--dotenv=<path>` — that exact-token rule is what stops a legitimate per-command VALUE
|
|
152
|
+
* (`skill … --answer "--dotenv=x=foo"`) from being hijacked, and it must stay. The equals form therefore
|
|
153
|
+
* reaches each command's own parser and surfaces as a bare `unknown flag: --dotenv`, dropping the one
|
|
154
|
+
* thing the user needs: where to put it instead. Deriving it here also covers `run`'s
|
|
155
|
+
* `unexpected argument(s):` path, which a flag-level fix never reaches.
|
|
156
|
+
*
|
|
157
|
+
* "It is a usage error, therefore the flag is misplaced" is FALSE, and the narrowness below is the
|
|
158
|
+
* whole safety argument. The same exit serves errors about a CORRECTLY-PLACED leading global
|
|
159
|
+
* (`--dotenv file not found: …`) and errors naming the flag inside a quoted VALUE (`… got a
|
|
160
|
+
* flag-looking token "--dotenv=/x"`). The SHAPE allowlist is what excludes both: neither wording
|
|
161
|
+
* begins with `unknown flag:` or `unexpected argument(s):`. The program-name skip below is
|
|
162
|
+
* defense-in-depth on top of that — every leading-position error also carries `cowork-harness` as its
|
|
163
|
+
* command — and removing it today changes no observable behaviour. Keep it: it is the guard that still
|
|
164
|
+
* holds if a future message is reworded into one of the two shapes.
|
|
165
|
+
*
|
|
166
|
+
* `critique` needs no exemption here: it never calls `fail()` (it writes to stderr and exits directly),
|
|
167
|
+
* so its legitimate per-command `--dotenv` cannot reach this function. Its exemption lives in cli.ts's
|
|
168
|
+
* pre-dispatch guard and nowhere else — do not add a dead branch for it here.
|
|
169
|
+
*
|
|
170
|
+
* Coverage is most of the CLI, not all of it, and the gaps are structural rather than oversights: a
|
|
171
|
+
* command that never reaches `fail()` (`critique`, `chat`, `prune`, `migrate-run-dir` all log and exit
|
|
172
|
+
* directly) or whose usage message is shaped differently (`vm` and a bare `assertions` emit a usage
|
|
173
|
+
* block; `lint`/`lint-skill` forward the token to Python argparse) keeps its own bare message. Those
|
|
174
|
+
* are worth closing at their own call sites — not by widening the allowlist here, which is what keeps
|
|
175
|
+
* the false positives out. */
|
|
176
|
+
function misplacedGlobalHint(command, message) {
|
|
177
|
+
// Leading-position global-flag errors are emitted with the PROGRAM NAME as `command`, never a
|
|
178
|
+
// subcommand, and are about a flag that is already correctly placed. Never hint there — and the
|
|
179
|
+
// example string would read `cowork-harness --dotenv <path> cowork-harness …` if we did.
|
|
180
|
+
if (command === "cowork-harness")
|
|
181
|
+
return undefined;
|
|
182
|
+
if (!UNRECOGNIZED_FLAG_SHAPES.some((re) => re.test(message)))
|
|
183
|
+
return undefined;
|
|
184
|
+
for (const flag of GLOBAL_ONLY_FLAGS) {
|
|
185
|
+
// Token-boundary match: the flag bare at end-of-token, or introducing an `=value` form. `.` is NOT a
|
|
186
|
+
// terminator — with it, a typo'd filename (`--dotenv.yaml`) was read as the flag and answered with
|
|
187
|
+
// "move it before the subcommand". No real message ends a flag token with a period; the two shapes
|
|
188
|
+
// this runs against terminate it with `=`, whitespace, or end-of-string.
|
|
189
|
+
if (!new RegExp(`(^|[\\s:])${flag}(=|$|[\\s,])`).test(message))
|
|
190
|
+
continue;
|
|
191
|
+
return `${flag} is a GLOBAL flag and must come BEFORE the subcommand (e.g. \`cowork-harness ${flag} <path> ${command} …\`)`;
|
|
192
|
+
}
|
|
193
|
+
return undefined;
|
|
194
|
+
}
|
|
138
195
|
/** The single error exit used by every command + the top-level catch, in both `cli.ts` and `doctor.ts`.
|
|
139
196
|
* boundary → exit 3, every other category → exit 2, UNLESS `exitCode` overrides it — SPEC.md's exit-code
|
|
140
197
|
* contract names two exceptions that exit `1` instead of the general `2`: `sync` hard-failures (missing
|
|
@@ -142,12 +199,15 @@ export function isJsonOutput(args) {
|
|
|
142
199
|
* failure reading a prior run's output (SPEC.md:428-436). Every EXISTING call site omits `exitCode` and
|
|
143
200
|
* keeps its current behavior exactly. */
|
|
144
201
|
export function fail(command, category, message, hint, json, exitCode) {
|
|
202
|
+
// A caller-supplied hint always wins. The pre-dispatch guard's own message is not one of the two
|
|
203
|
+
// recognized shapes, so it can never pick up a duplicate copy of its own sentence.
|
|
204
|
+
const effective = hint ?? (category === "usage" ? misplacedGlobalHint(command, message) : undefined);
|
|
145
205
|
if (json)
|
|
146
|
-
out(jsonError(command, category, message,
|
|
206
|
+
out(jsonError(command, category, message, effective));
|
|
147
207
|
else {
|
|
148
208
|
log(message);
|
|
149
|
-
if (
|
|
150
|
-
log(
|
|
209
|
+
if (effective)
|
|
210
|
+
log(effective);
|
|
151
211
|
}
|
|
152
212
|
process.exit(exitCode ?? (category === "boundary" ? 3 : 2));
|
|
153
213
|
}
|
package/docs/critique.md
CHANGED
|
@@ -243,7 +243,14 @@ malformed evaluator items (the surviving findings are then not necessarily the c
|
|
|
243
243
|
agents md), `corpusCuts` (per-file — empty on every real skill; only non-empty once the ceiling is
|
|
244
244
|
actually breached), `corpusExcluded` (files present on the host but never delivered to the agent by
|
|
245
245
|
staging — untracked, with git-mode on), and `trimRecord` (any section the overall belt-and-suspenders cap
|
|
246
|
-
shaved).
|
|
246
|
+
shaved). `cowork-harness lint-skill <skill-dir>` answers the same proximity question **without a paid
|
|
247
|
+
run** — `skill-corpus-near-evidence-ceiling` (INFO) from 80%, `skill-corpus-over-evidence-ceiling` (WARN,
|
|
248
|
+
so it fails `--strict`) past it. It counts the same three classes the ceiling governs: `SKILL.md`, every
|
|
249
|
+
file under `references/` (**any extension** — the packager applies no extension filter, so JSON schemas
|
|
250
|
+
and rule packs count), and a plugin skill's `agents/<name>.md`. It does not apply staging's git-tracked
|
|
251
|
+
filter, so an untracked reference inflates the figure — it errs toward warning early, and `corpusCuts`
|
|
252
|
+
stays the authority.
|
|
253
|
+
On a normal skill this is one reassuring line; the other fields only grow teeth on a genuinely
|
|
247
254
|
oversized skill or an untracked-file mistake.
|
|
248
255
|
|
|
249
256
|
**A corpus-ceiling breach has a second, sharper consequence than the not-adjudicable steer: DROPPED
|
package/docs/debugging.md
CHANGED
|
@@ -4,6 +4,12 @@ When a run does the wrong thing — or greens when you don't trust it — this i
|
|
|
4
4
|
the right tool; the authoritative reference for each command's flags is its `--help` and the
|
|
5
5
|
[README → Commands at a glance](../README.md#commands-at-a-glance).
|
|
6
6
|
|
|
7
|
+
> **Scope.** "Evidence" on this page means the RUN's own record — events, trace, transcript; what
|
|
8
|
+
> `trace` / `inspect` / `diff` / `verify-run` / `replay --explain` read. `critique`'s evaluator grades
|
|
9
|
+
> against a separate, narrower record: `critique-evidence-package.txt`, written at the run-dir root
|
|
10
|
+
> whenever the evaluator ran. None of the five tools below surface it — see
|
|
11
|
+
> [`docs/critique.md`](./critique.md#run-dir-artifacts).
|
|
12
|
+
|
|
7
13
|
<!-- BEGIN triage-canonical -->
|
|
8
14
|
Two situations need different tools — figure out which one you're in first, then reach for the tool
|
|
9
15
|
instead of re-running and hoping. The run already wrote its evidence to a kept run dir (`--keep` prints
|
|
@@ -16,7 +16,7 @@ DOES exercise a real gate exchange, see `example-multiselect-gate.cassette.json`
|
|
|
16
16
|
|
|
17
17
|
Run it with:
|
|
18
18
|
|
|
19
|
-
> Assumes the `cowork-harness` CLI is available — from a source checkout run `npm ci && npm run build && npm link` first, or `npm i -g "cowork-harness@>=1.13.
|
|
19
|
+
> Assumes the `cowork-harness` CLI is available — from a source checkout run `npm ci && npm run build && npm link` first, or `npm i -g "cowork-harness@>=1.13.2"`. (`replay` itself needs nothing else — no token, no Docker.)
|
|
20
20
|
|
|
21
21
|
```sh
|
|
22
22
|
cowork-harness replay examples/replays/example-pdf-skill.cassette.json
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "cowork-harness",
|
|
3
|
-
"version": "1.13.
|
|
3
|
+
"version": "1.13.2",
|
|
4
4
|
"description": "Scriptable, CI-friendly harness for Claude Cowork's runtime contract for testing skills across scenarios \u2014 same agent, mounts, egress allowlist, permission protocol, and sandbox limitations.",
|
|
5
5
|
"license": "MIT",
|
|
6
6
|
"type": "module",
|
|
@@ -1,16 +1,18 @@
|
|
|
1
|
-
// Guards against a
|
|
2
|
-
//
|
|
3
|
-
//
|
|
4
|
-
//
|
|
5
|
-
//
|
|
6
|
-
//
|
|
7
|
-
//
|
|
1
|
+
// Guards against a pointer to a repo directory the shipped skill payload (.claude/skills/**) cannot
|
|
2
|
+
// reach. A Claude Code plugin cache materializes ONLY `.claude/skills/<name>/**`, so a bare relative
|
|
3
|
+
// reference like `docs/critique.md`, `schema/run-result.json` or `examples/answer-policies/demo.yaml`
|
|
4
|
+
// is a dead pointer for every plugin-install user, even where it resolves fine for an npm/tarball
|
|
5
|
+
// install. This cost a real consumer hours (they needed `critique-evidence-package.txt`'s name,
|
|
6
|
+
// documented only in docs/critique.md) and shipped unfixed across two releases before this check
|
|
7
|
+
// existed. See DEAD_ROOTS below for which roots are in scope and, just as importantly, which are not.
|
|
8
8
|
//
|
|
9
9
|
// npx tsx scripts/check-skill-doc-links.ts
|
|
10
10
|
//
|
|
11
|
-
// Fix for a violation: rewrite the bare `docs/x.md` into a
|
|
12
|
-
// (https://github.com/<owner>/<repo>/blob/main
|
|
13
|
-
// path — see the many examples already in .claude/skills/cowork-harness/SKILL.md and
|
|
11
|
+
// Fix for a violation: rewrite the bare path (`docs/x.md`, `schema/x.json`, `examples/x.yaml`) into a
|
|
12
|
+
// GitHub blob permalink (https://github.com/<owner>/<repo>/blob/main/<path>) so it resolves regardless
|
|
13
|
+
// of install path — see the many examples already in .claude/skills/cowork-harness/SKILL.md and
|
|
14
|
+
// references/. Where the surrounding sentence ALREADY qualifies the pointer as npm-only, append the
|
|
15
|
+
// OPT_OUT_MARKER to that line instead.
|
|
14
16
|
//
|
|
15
17
|
// Scope: git-TRACKED files under .claude/skills/** only. This is deliberately the same boundary a
|
|
16
18
|
// plugin marketplace install and the harness's own `local_plugins` staging use — an untracked file
|
|
@@ -24,21 +26,50 @@ import { fileURLToPath, pathToFileURL } from "node:url";
|
|
|
24
26
|
|
|
25
27
|
const REPO_ROOT = join(dirname(fileURLToPath(import.meta.url)), "..");
|
|
26
28
|
|
|
27
|
-
//
|
|
29
|
+
// Repo roots that exist in the REPO but never in a PLUGIN cache, so a bare relative pointer to one is
|
|
30
|
+
// dead for every plugin-install consumer. Deliberately NOT exhaustive:
|
|
31
|
+
// scripts/ — the payload ships its own .claude/skills/<name>/scripts/, so the path resolves
|
|
32
|
+
// cassettes/ — in a recipe this names the READER's directory, not ours
|
|
33
|
+
// src/, test/ — provenance citations ("the rule lives here"), not "go open this"
|
|
34
|
+
// baselines/ — its payload occurrences are a glob and a bare directory mention; no filename pattern matches either
|
|
35
|
+
// Nested segments are included: a reference into a SUBDIRECTORY is just as dead as a flat one, and a
|
|
36
|
+
// pattern that only matched the flat form would let the next one through.
|
|
37
|
+
const DEAD_ROOTS = ["docs", "schema", "examples"] as const;
|
|
38
|
+
// LEFT boundary `(?<![\w.-])`: without it `mydocs/x.md` and `xschema/y.json` matched their tails and were
|
|
39
|
+
// reported as violations of paths nobody wrote. `/` is deliberately NOT excluded, so a genuine relative
|
|
40
|
+
// pointer (`./docs/x.md`) is still caught — the residual cost is that an absolute system path containing
|
|
41
|
+
// one of these segments would flag, which does not occur in a skill payload and has the marker as an out.
|
|
42
|
+
// RIGHT boundary `(?![\w])`: without it `docs/x.mdx` matched as `docs/x.md`, so the error told the author
|
|
43
|
+
// to permalink a file that does not exist. Case-insensitive so `schema/x.JSON` cannot slip through.
|
|
44
|
+
const DEAD_PATH_BODY = `(?<![\\w.-])(?:${DEAD_ROOTS.join("|")})/(?:[\\w.-]+/)*[\\w.-]+\\.(?:md|json|ya?ml)(?![\\w])`;
|
|
45
|
+
|
|
46
|
+
// A full GitHub blob permalink to one of those paths, e.g.
|
|
28
47
|
// `https://github.com/yaniv-golan/cowork-harness/blob/main/docs/critique.md` — this resolves
|
|
29
48
|
// regardless of install path, so any occurrence of it is exempt from the bare-reference check below.
|
|
30
|
-
|
|
49
|
+
// ANY absolute URL, not just a github blob permalink. Scoping the exemption to one URL shape meant an
|
|
50
|
+
// equally-resolvable link — a raw.githubusercontent URL, a GitLab `/-/blob/`, a github `/tree/` — was
|
|
51
|
+
// reported as a dead relative pointer, i.e. the checker rejected its own prescribed fix written another
|
|
52
|
+
// way. An absolute URL resolves for every install path, which is the only property this guard cares about.
|
|
53
|
+
const ANY_URL_RE = /https?:\/\/\S+/g;
|
|
31
54
|
|
|
32
55
|
// A markdown link `[text](permalink)` whose target is one of the permalinks above — stripped as a
|
|
33
|
-
// whole unit FIRST, because the link text itself is often the same bare
|
|
56
|
+
// whole unit FIRST, because the link text itself is often the same bare path (e.g.
|
|
34
57
|
// `` [`docs/critique.md`](https://…/docs/critique.md) ``) and must not separately trip the bare check.
|
|
35
|
-
|
|
58
|
+
// `[text](url)` — stripped as a whole unit FIRST, because the link TEXT is often the same bare path
|
|
59
|
+
// (`` [`docs/critique.md`](https://…) ``); stripping only the URL would leave that text to trip the bare
|
|
60
|
+
// check. A link whose target is RELATIVE is deliberately not stripped: `[docs/x.md](./docs/x.md)` is dead
|
|
61
|
+
// for a plugin install in both halves.
|
|
62
|
+
const MD_LINK_TO_URL_RE = /\[[^\]\n]*\]\(https?:\/\/[^\s)]+\)/g;
|
|
36
63
|
|
|
37
|
-
// A bare/relative reference to
|
|
64
|
+
// A bare/relative reference to one of those paths, anywhere it isn't already covered by a permalink —
|
|
38
65
|
// this is the dead pointer for a plugin install.
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
66
|
+
const BARE_DEAD_RE = new RegExp(DEAD_PATH_BODY, "gi");
|
|
67
|
+
|
|
68
|
+
// Explicit, greppable per-line opt-out for a pointer the SAME sentence already qualifies as npm-only
|
|
69
|
+
// (e.g. "published as `schema/verify-cassettes.json` in the npm package"). Mirrors lint-skill's
|
|
70
|
+
// `ignore-next-line` convention: an escape hatch visible in the source, never a silent heuristic over
|
|
71
|
+
// surrounding prose. Line-scoped by design — a file-wide opt-out would hide the next real one.
|
|
72
|
+
const OPT_OUT_MARKER = "<!-- npm-only-ok -->";
|
|
42
73
|
|
|
43
74
|
export interface Violation {
|
|
44
75
|
file: string;
|
|
@@ -55,10 +86,11 @@ export function findViolations(files: Array<{ path: string; content: string }>):
|
|
|
55
86
|
for (const { path, content } of files) {
|
|
56
87
|
const lines = content.split("\n");
|
|
57
88
|
lines.forEach((lineText, i) => {
|
|
89
|
+
if (lineText.includes(OPT_OUT_MARKER)) return;
|
|
58
90
|
// Strip whole markdown-link-to-permalink constructs, then any remaining bare permalink (the
|
|
59
91
|
// YAML/Python-comment case, where the URL appears with no surrounding [text](...) brackets).
|
|
60
|
-
const residual = lineText.replace(
|
|
61
|
-
const matches = residual.match(
|
|
92
|
+
const residual = lineText.replace(MD_LINK_TO_URL_RE, "").replace(ANY_URL_RE, "");
|
|
93
|
+
const matches = residual.match(BARE_DEAD_RE);
|
|
62
94
|
if (matches) {
|
|
63
95
|
for (const target of matches) violations.push({ file: path, line: i + 1, target });
|
|
64
96
|
}
|
|
@@ -82,9 +114,10 @@ export function checkSkillDocLinks(repoRoot: string = REPO_ROOT): { ok: boolean;
|
|
|
82
114
|
const violations = findViolations(files);
|
|
83
115
|
const errors = violations.map(
|
|
84
116
|
(v) =>
|
|
85
|
-
`${v.file}:${v.line}: bare reference to "${v.target}" — dead for a plugin install
|
|
86
|
-
`only
|
|
87
|
-
`(https://github.com/yaniv-golan/cowork-harness/blob/main/${v.target})
|
|
117
|
+
`${v.file}:${v.line}: bare reference to "${v.target}" — dead for a plugin install, which ` +
|
|
118
|
+
`materializes only .claude/skills/<name>/**; rewrite to a GitHub blob permalink ` +
|
|
119
|
+
`(https://github.com/yaniv-golan/cowork-harness/blob/main/${v.target}), or append ` +
|
|
120
|
+
`"${OPT_OUT_MARKER}" to the line if its own sentence already qualifies the pointer as npm-only`,
|
|
88
121
|
);
|
|
89
122
|
return { ok: violations.length === 0, errors, violations };
|
|
90
123
|
}
|
|
@@ -92,10 +125,10 @@ export function checkSkillDocLinks(repoRoot: string = REPO_ROOT): { ok: boolean;
|
|
|
92
125
|
function main(): void {
|
|
93
126
|
const { ok, errors, violations } = checkSkillDocLinks();
|
|
94
127
|
if (ok) {
|
|
95
|
-
process.stdout.write(
|
|
128
|
+
process.stdout.write(`✓ no dangling ${DEAD_ROOTS.join("/")}/ references under .claude/skills/**\n`);
|
|
96
129
|
return;
|
|
97
130
|
}
|
|
98
|
-
process.stdout.write(`found ${violations.length} dangling
|
|
131
|
+
process.stdout.write(`found ${violations.length} dangling repo-path reference(s):\n`);
|
|
99
132
|
for (const e of errors) process.stderr.write(`::error::${e}\n`);
|
|
100
133
|
process.exitCode = 1;
|
|
101
134
|
}
|