@dzhechkov/harness-cli 0.5.0 → 0.5.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/.dz-manifest.json CHANGED
@@ -9,7 +9,7 @@
9
9
  },
10
10
  {
11
11
  "path": "README.md",
12
- "sha256": "27e2ddfd256e5c89a1c51a2520b52cc4b947dd7295f5730bde0a2c735186fb8f"
12
+ "sha256": "6d4e868f0fef3d86f83c6316f4bb760fd9228ef1008d3bbb31916f6c8890b944"
13
13
  },
14
14
  {
15
15
  "path": "coverage/coverage-final.json",
@@ -37,23 +37,23 @@
37
37
  },
38
38
  {
39
39
  "path": "dist/cli.d.ts",
40
- "sha256": "56db11f99b505b2ffbc314aa70b4bcd17ec741e14ca99a3662d0ca9f4afcc1cc"
40
+ "sha256": "3893831e365016b5bd118f7f0ce2e8b106cfd0400689b9e190a5f80aa61b79f5"
41
41
  },
42
42
  {
43
43
  "path": "dist/cli.d.ts.map",
44
- "sha256": "19170be889713fffd051dfde1044074125f66bb4623e415c355f812a47721436"
44
+ "sha256": "514bbdd5b14a1257693418c64bf2fb6c13d78beccd307f0f3a9c66461a42a62f"
45
45
  },
46
46
  {
47
47
  "path": "dist/cli.js",
48
- "sha256": "c80fe1740887b68a3521ee6d22e09303785238c6e27dd0c80fb76289927e73a4"
48
+ "sha256": "09277bf170699a591ca9d2c8f6ac3b2727ec41cff3a75e56058724bd1e4686e1"
49
49
  },
50
50
  {
51
51
  "path": "dist/cli.js.map",
52
- "sha256": "1ef2072181086e6b14c30005417f3edba91db5c3f2ae82201236ea493385d6a8"
52
+ "sha256": "e2a46281a86931f342cf726ae0fd0aeb25eef7ce94edbd07abce68635f547e78"
53
53
  },
54
54
  {
55
55
  "path": "dist/core-compat.d.ts",
56
- "sha256": "dedfc859e163f229af0445d6439172a4b66c150c0942bce3ae1bf0304dbb3290"
56
+ "sha256": "868dd0e0ec12ea9df5f4df148518f5c59226ecc1de04fa67d5455f4932a7819b"
57
57
  },
58
58
  {
59
59
  "path": "dist/core-compat.d.ts.map",
@@ -61,7 +61,7 @@
61
61
  },
62
62
  {
63
63
  "path": "dist/core-compat.js",
64
- "sha256": "57930e6918c2a5d90826c74924a62de9a1ffab53d8196ff8f98a59c89baf23be"
64
+ "sha256": "973e9bd7c0951e9f74bd3e8b38a0cc81076d6ea8393fa3e52a428b92ebc967c8"
65
65
  },
66
66
  {
67
67
  "path": "dist/core-compat.js.map",
@@ -69,19 +69,19 @@
69
69
  },
70
70
  {
71
71
  "path": "dist/index.d.ts",
72
- "sha256": "aaac0134a074fd06e9b5e09daa952ef6353510e481885da51a6e4f2e92beba35"
72
+ "sha256": "a25d0a97a30fa5abc35473ed4ca1bc32bc5a8bcb8a45c57d7feeb03cb6058e02"
73
73
  },
74
74
  {
75
75
  "path": "dist/index.d.ts.map",
76
- "sha256": "459c3381a9fbcd3378b623f9fe103a8d3337e4025a2f546b6d40e4d9dd4e0f70"
76
+ "sha256": "7c554d109fb0aabcf9a2923b0d9ac84723652abdec81202b2cc0a726d38c84bd"
77
77
  },
78
78
  {
79
79
  "path": "dist/index.js",
80
- "sha256": "15ff1e6b84537b91b73b6a7e2d1c45f1120b25af9d1139df8da8ad094e99cf0e"
80
+ "sha256": "5b7992e09b843a01c9bf7a05b5db4475ca9abbf893ead79f7bad73f0f92c4249"
81
81
  },
82
82
  {
83
83
  "path": "dist/index.js.map",
84
- "sha256": "2f61a3be1af4ab9f7bbcbacb912ccbd89e3b73414f9fd24766598bfbdc1f8655"
84
+ "sha256": "54c8133a172c2ca81c3f5a32e701386812a621198758423fcc7f0bf3db8c1ff7"
85
85
  },
86
86
  {
87
87
  "path": "keys/README.md",
@@ -89,7 +89,7 @@
89
89
  },
90
90
  {
91
91
  "path": "package.json",
92
- "sha256": "be73e142cc0dcd05cf774c73827572b1e55e503c8ab84f4a396fa785b271980b"
92
+ "sha256": "503d43d8ab6e7fb04c28f96270eb00da9782e4ec0570772f2f701aef33037c25"
93
93
  },
94
94
  {
95
95
  "path": "src/bin.ts",
@@ -97,15 +97,15 @@
97
97
  },
98
98
  {
99
99
  "path": "src/cli.ts",
100
- "sha256": "c875709ccafe3493342776b28c8aa7b34e1fb800e2fccedcb499e6490e6ad1c0"
100
+ "sha256": "747d3a279349208ce8f5c6121bbf8ea039ebafaa8251c520ba3092a7aa80013f"
101
101
  },
102
102
  {
103
103
  "path": "src/core-compat.ts",
104
- "sha256": "3638272a91166a047c3edc14626c0011e91b2de862a197d1b8e424470e7a0575"
104
+ "sha256": "53747612facd41ade5c291949b854d354e694b733e1f294cb3e521193760ae26"
105
105
  },
106
106
  {
107
107
  "path": "src/index.ts",
108
- "sha256": "395b27b317e575b94a1049cdb284db185e0ec382fbae1e588c78ff2398b2563f"
108
+ "sha256": "76c2dc1598e72d11df59319cc64fb2aa39b8c3af4eb3077f66598f4f2a3323c0"
109
109
  },
110
110
  {
111
111
  "path": "test/agents-sync.test.ts",
@@ -113,11 +113,11 @@
113
113
  },
114
114
  {
115
115
  "path": "test/cli.test.ts",
116
- "sha256": "83d8c6a9d7e8d46b57bb2e5da47c0345e8a512bdcc2e7cc8c59e163ade41de10"
116
+ "sha256": "d9adcce84b509e00213bda59a0948eca90d83bdd2fdbbc86aee5ce11d9176940"
117
117
  },
118
118
  {
119
119
  "path": "test/command-count.test.ts",
120
- "sha256": "9958e10a8adb80ee64baad0d91fb4c9a3b6e8801c6379f7ba5f8f5ab6c1f1716"
120
+ "sha256": "f744cfe02805f567312a196805ebc49486c3e2a8e3f64de1700be276481f0cc1"
121
121
  },
122
122
  {
123
123
  "path": "test/core-compat-guard.test.ts",
@@ -129,11 +129,11 @@
129
129
  },
130
130
  {
131
131
  "path": "test/core-import-floor.test.ts",
132
- "sha256": "84e16e0f31608570f72575038ae246d9ff002671ade0e9f9fa6009dd862dfc03"
132
+ "sha256": "948ce5017d7b512f7f32c817f8d1f5cbc5ede7d7c3f2788456ec5d24b8983f14"
133
133
  },
134
134
  {
135
135
  "path": "test/discrimination-check-cli.test.ts",
136
- "sha256": "cd7507c8ec1f621fea6f05596541235d09efbd215540b56b012cf465e06b6ba0"
136
+ "sha256": "d0ec5796bc526d45cdad9041ea7357afc065917b347cf5421ec7d2696106b08c"
137
137
  },
138
138
  {
139
139
  "path": "test/epoch-replay-cli.test.ts",
@@ -255,6 +255,22 @@
255
255
  "path": "test/mutation-registry.json",
256
256
  "sha256": "3e6f9bb105154121988da41568ff85573099235857ba25e76c4264c166b414ae"
257
257
  },
258
+ {
259
+ "path": "test/parallel-worktree-smoke.test.ts",
260
+ "sha256": "0522b12ccdf3db79215c4c42162ef970f4cba0c0af96208426ca0323c367d0ee"
261
+ },
262
+ {
263
+ "path": "test/plan-completeness-gate.test.ts",
264
+ "sha256": "be010af797aeae754a3dd3e9f6f25d6f01066d4bf2037a01322554453d943ee3"
265
+ },
266
+ {
267
+ "path": "test/qe-bridge-cli.test.ts",
268
+ "sha256": "336ba7657c55f5b958a0ebed5617f3bc3b1956017b9389bbe7c38137889f8f1b"
269
+ },
270
+ {
271
+ "path": "test/qe-bridge-manifest-drift.test.ts",
272
+ "sha256": "b7b4d9e0c1dcee32e972aab97273b04bdde9634433e7c22345793a677dfff300"
273
+ },
258
274
  {
259
275
  "path": "test/skills-verify-static-advisories.test.ts",
260
276
  "sha256": "b1140c20ec04d1c8ae67f97a42dbcece1e8e63c98a5de2c087f0359d27c25f17"
@@ -283,6 +299,18 @@
283
299
  "path": "test/workflow-legacy-shim.test.ts",
284
300
  "sha256": "47d35ec36aa4793ee5d61debf1d345e7d236ad21db27f052aa3e7add03852205"
285
301
  },
302
+ {
303
+ "path": "test/workflow-run-cli.test.ts",
304
+ "sha256": "ac0468d8e97795602e8b8298f278356a273309674483854cd136b99ddf6e39d4"
305
+ },
306
+ {
307
+ "path": "test/workflow-run-r2.test.ts",
308
+ "sha256": "d7f308ac9538b7e9d7dc8dcd0ce7cf8127f26b5aea1f44e9d44ac1500ad733fb"
309
+ },
310
+ {
311
+ "path": "test/workflow-run-read-plane.test.ts",
312
+ "sha256": "c56111e578500d690190f9fdb4ec44182a892c7a6c2c289896509c1229d7b06d"
313
+ },
286
314
  {
287
315
  "path": "test/workflow-trace-cli-surface.test.ts",
288
316
  "sha256": "28df4b51fff1dc80bb38671d56f548d859949d7a9e2f00e7045e732efc7c2e3d"
@@ -297,5 +325,5 @@
297
325
  }
298
326
  ]
299
327
  },
300
- "signature": "jyIwJnH5LH3+Wrq8W19HGqMpqt8e0srMnfZH/atV2H3ddnlrYzMe0f7bvyigDpdLzVApxiq+Xna5z4psVILMDg=="
328
+ "signature": "/iL9GLKryxZjC2MS2xcwig3BkbaMG+FLHOW5yhYVOI6Oe8MOEFCqE0d+a/xIWgGfKBy6UMH3Alzf25qYmPldAA=="
301
329
  }
package/README.md CHANGED
@@ -414,7 +414,7 @@ dz bundle --select news-digest,goap-research-ed25519 --out ./dist
414
414
  dz init --target claude-code --select design-thinking
415
415
 
416
416
  # Curated set by topic (recommended):
417
- dz setup --target claude-code --preset meta # 18 development skills + self-learning
417
+ dz setup --target claude-code --preset meta # 20 development skills + self-learning
418
418
 
419
419
  # Full toolkit with orchestrated pipeline:
420
420
  npx @dzhechkov/keysarium init # 7-phase research + commands + memory
@@ -739,6 +739,82 @@ ever enter a work order.
739
739
  **When to use:** after `dz compounding` reports the replay as READY; before claiming that recall
740
740
  "works"; and any time you want the claim re-checked as the corpus grows.
741
741
 
742
+ ### Обратный мост QE: Claude-ревьюер из Codex-сессии — `dz qe-bridge`
743
+
744
+ The cross-family rule ("the family that writes the code must not review it") was enforceable in one
745
+ direction only. When **Codex hosts** the run there is no Claude agent plane to dispatch from — and
746
+ `dz reqe`'s brief admits it: for a claude review family it prints `null` where the codex branch
747
+ prints a ready command. `dz qe-bridge` is that missing vehicle: a plain-shell command that probes a
748
+ Claude model, sends a Step-8-shaped brief over SCOPED extracts, and PARSES the verdict.
749
+
750
+ ```bash
751
+ # MEASURED 2026-08-19 on this repo — reproducer: the exact command below, reviewing a real shipped feature
752
+ $ dz qe-bridge --family claude --slug wave1-scorer-negation --coder-family codex --model opus
753
+ dz qe-bridge: GRADE C from claude/opus — 7 finding(s) in 343s
754
+ report: features/wave1-scorer-negation/08b_reqe_report.md
755
+ signoff: features/wave1-scorer-negation/.fa-state/qe-bridge/signoff-2026-08-19T18-48-45-545Z.json
756
+ settle: dz reqe --slug wave1-scorer-negation --done --report features/wave1-scorer-negation/08b_reqe_report.md
757
+ the bridge REPORTS (any grade exits 0); gating stays with dz reqe and the host pipeline.
758
+
759
+ # the same command with a binary that cannot answer — a failed call, and NO report to settle with
760
+ $ DZ_QE_BRIDGE_CLAUDE_BIN=/bin/false dz qe-bridge --family claude --slug wave1-scorer-negation --coder-family codex
761
+ dz qe-bridge: FAILED — probe-failed
762
+ no candidate model answered the liveness probe — opus: exit 1, no `OK` in 0 chars of stdout; sonnet: exit 1, …; haiku: exit 1, …
763
+ record: features/wave1-scorer-negation/.fa-state/qe-bridge/failed-2026-08-19T18-48-52-931Z.json
764
+ no report was written — an unparseable or absent review is never a passing one.
765
+
766
+ # with a debt on record, the report settles it through the untouched fail-closed path
767
+ $ dz reqe --slug add-x --done --report features/add-x/08b_reqe_report.md
768
+ dz reqe: debt settled: re-QE grade C (report …) — settlement appended to features/add-x/08_qe_report.md
769
+ ```
770
+
771
+ **When to use:** you are hosting a run outside Claude Code (Codex, CI, a plain terminal), you have
772
+ just written code, and the independent reviewer must be the OTHER family. Also: whenever `dz reqe`
773
+ lists a debt whose coder family is `openai`.
774
+
775
+ **The reviewer runs isolated.** Both calls (probe and review) run from an EMPTY temporary directory
776
+ with `--safe-mode --strict-mcp-config --tools '' --no-session-persistence`, and the verdict is read
777
+ from the `--output-format json` **result envelope**. Why: customization output lands on the same
778
+ stdout — MEASURED on this machine, a session-start plugin prints a banner ahead of the model's
779
+ answer — so a crafted hook could otherwise print a complete grade-A signoff and a stream parser
780
+ would believe it (reproducer: `features/qe-bridge-claude/07_code_changes/mutants/c1-forgery-repro.mjs`).
781
+ Residue, stated: `--safe-mode` leaves ADMIN-MANAGED policy settings in force, and no flag proves
782
+ which binary answered.
783
+
784
+ **What makes the grade valid.** Three channels must EXIST and AGREE, each read **LAST-anchored**,
785
+ and the marker must be the FINAL content of the answer:
786
+ the terminal `QE-BRIDGE-SIGNOFF grade=<A-F> findings=<n>` line, the last fenced `qe-bridge-signoff`
787
+ JSON block, and the report's own line-anchored `GRADE:` line. Repo content flows into the prompt and
788
+ comes back quoted, so a planted earlier verdict must lose — and it does (there is a test whose
789
+ fixture plants `grade=A` early and requires the genuine trailing `grade=D` to win). Extracts are
790
+ DEFANGED on the way in, so quoted content can never mint a verdict. Empty, gradeless, marker-only or
791
+ self-contradicting output is a **named failure** — one of 17 closed reasons (`envelope-unparseable`,
792
+ `marker-not-terminal`, `findings-count-mismatch`, `grade-mismatch`, `ambiguous-grade`,
793
+ `audit-write-failed`, `report-write-failed`, … ; closed BOTH ways — every one is produced by a real
794
+ run in the suite and leaves a record) — with an audit record under
795
+ `features/<slug>/.fa-state/qe-bridge/` and the raw stdout beside it, never `findings: []`. Finding
796
+ numbers are the reviewer's: a missing, non-positive or duplicated `n` fails the call instead of being
797
+ renumbered, and a marker whose `findings=<n>` disagrees with the block is `findings-count-mismatch`.
798
+
799
+ **The record is auditable, not just a conclusion.** Every run writes a `runId`, the resolved
800
+ executable plus `binOverride` (true whenever `DZ_QE_BRIDGE_CLAUDE_BIN` was used — the documented TEST
801
+ SEAM; there is no `--claude-bin` flag), the prompt sha256, the byte offsets at which each channel was
802
+ found, the `requestedOut` path and `reportWritten: true|false` — so "no report was written" is a
803
+ stated fact rather than an inference from an absent file. Records and reports are written `0600` in a
804
+ `0700` directory, through `O_EXCL`, with realpath containment that refuses a symlinked parent — and
805
+ the state directory itself is contained the same way, before anything is created in it. The audit
806
+ trail is written BEFORE the report and corrected after it, so `reportWritten` can only ever
807
+ understate; if the trail cannot be written at all, the run FAILS (`audit-write-failed`) rather than
808
+ shipping a verdict nobody can re-derive.
809
+
810
+ **Exit codes:** `0` a signoff was parsed (ANY grade — a grade F still exits 0: the bridge reports, it
811
+ does not gate), `1` a named failure, `2` a usage error. **Honest limits:** it proves the call was
812
+ procedurally sound (a live model was probed, a scoped brief was sent, a self-consistent verdict came
813
+ back); it cannot prove which model authored the text, and it cannot classify your secrets — the
814
+ extracts you scope are what leaves the machine. RU: мост в обратную сторону — из Codex-сессии
815
+ позвать независимого Claude-ревьюера и получить РАЗОБРАННЫЙ вердикт; пустой или безоценочный ответ —
816
+ это названная ошибка, а не «чисто».
817
+
742
818
  ### Пересмотр после аварийного само-ревью — `dz reqe`
743
819
 
744
820
  The feature-adr pipeline's cross-model guard says *the model that writes code must not review it*.
@@ -1161,7 +1237,7 @@ Each pack is an npm package — click through for the **full per-skill documenta
1161
1237
  | [@dzhechkov/skills-qe](https://www.npmjs.com/package/@dzhechkov/skills-qe) | 20 | Quality engineering — test-gen, coverage, chaos, defect intelligence, QCSD swarms |
1162
1238
  | [@dzhechkov/skills-reasoning](https://www.npmjs.com/package/@dzhechkov/skills-reasoning) | 4 | Generic reasoning & code-quality — investigate (root-cause), solid (SOLID/TDD), karpathy-guidelines, agents-md-creator |
1163
1239
  | [@dzhechkov/skills-ecc](https://www.npmjs.com/package/@dzhechkov/skills-ecc) | 20 | Claude-Code engineering craft — agent architecture, autonomous loops, framework patterns |
1164
- | [@dzhechkov/skills-meta](https://www.npmjs.com/package/@dzhechkov/skills-meta) | 19 | Dev-process meta skills — explore, feature-adr, design-thinking, audit, skill-advisor, loop-plan-author |
1240
+ | [@dzhechkov/skills-meta](https://www.npmjs.com/package/@dzhechkov/skills-meta) | 20 | Dev-process meta skills — explore, feature-adr, design-thinking, audit, skill-advisor, loop-plan-author, decision-mockups (vendored mirror of `@dzhechkov/skills-decision-mockups`) |
1165
1241
  | [@dzhechkov/skills-academic](https://www.npmjs.com/package/@dzhechkov/skills-academic) | 5 | Thesis-defense toolkit — dissertation review, questions, doc-check, defense eval |
1166
1242
  | [@dzhechkov/skills-news](https://www.npmjs.com/package/@dzhechkov/skills-news) | 3 | *dz-original* — news digests (`news-digest`) + delta watches (`news-monitor`) + bundled `goap-research-ed25519` verified-research backend (mandatory) |
1167
1243
  | [@dzhechkov/skills-idea2prd](https://www.npmjs.com/package/@dzhechkov/skills-idea2prd) | 1 | *dz-original* — `idea2prd-manual`: idea/problem → PRD+ADR+DDD+C4+Pseudocode+Tests+Completion (9 checkpoints); bundles the analyst trio as a sources.json-tracked vendor ([ADR-0001](https://github.com/djd1m/dz-harness-hub/blob/main/docs/adr/0001-skill-canonicalization-and-dependency-model.md)) |
@@ -1178,7 +1254,7 @@ Each pack is an npm package — click through for the **full per-skill documenta
1178
1254
 
1179
1255
  | Preset | Skills | Description |
1180
1256
  |--------|--------|-------------|
1181
- | `meta` | 18 | Development process (explore, goap-research, problem-solver, design-thinking, feature-adr, knowledge-extractor, understand-anything-bridge, agentshield-scan, adversarial-verifier, skill-advisor, audit) |
1257
+ | `meta` | 20 | Development process (explore, goap-research, problem-solver, design-thinking, feature-adr, knowledge-extractor, understand-anything-bridge, agentshield-scan, adversarial-verifier, skill-advisor, audit, loop-plan-author, decision-mockups) |
1182
1258
  | `qe-engineer` | 20 | Quality engineering (test-gen, coverage, chaos, defect, ...) |
1183
1259
  | `bto` | 1 | Build-Benchmark-Test-Optimize pipeline |
1184
1260
  | `health` | 8 | Medical AI (diagnostics, drugs, labs, clinical decisions) |
@@ -1208,6 +1284,7 @@ Get the whole set with `dz init --target claude-code --preset meta`, or pick one
1208
1284
  | `reflection-loop` | Standalone critique → revise cycle (≤3 rounds) for code, text, architecture or research | `/reflection-loop` · "critique this" / "review and improve" |
1209
1285
  | `structured-reasoning` | Picks the reasoning strategy (Tree-of-Thought / CoT / compression) and checks the conclusion follows | "reason about…" / "explore options" / "compare approaches" |
1210
1286
  | `skill-crystallizer` | Auto-creates skills from execution traces, combines skills, and repairs broken ones | "create skill from this" / "combine skills" / "fix skill" |
1287
+ | `decision-mockups` | Owner-facing DECISION PAGE — plain-language write-up + browser-frame before/after mockups + clickable option forks + a copy-answers export you paste back into the chat (a one-option fork is deleted as fake, and a deterministic G0–G14 gate blocks the page if it is not) | "объясни понятным языком" / "что сделано, польза, риски, из чего выбираем" / "покажи владельцу развилки и собери решения" / "оформи артефактом" |
1211
1288
 
1212
1289
  ### Standalone Packages (install via npx, no dz CLI needed)
1213
1290
 
@@ -1228,7 +1305,7 @@ Get the whole set with `dz init --target claude-code --preset meta`, or pick one
1228
1305
 
1229
1306
  > **A skill and its npx toolkit are not duplicates — they're a graduation.** Several skills (e.g. `feature-adr`, `design-thinking`) exist BOTH as a skill inside a `dz` preset AND as a standalone `npx` package. The preset's SKILL.md is **fully functional on its own** (the whole methodology — modules + references — travels with it, and it auto-activates by description), and it's the only way to compile that capability to the **non-Claude platforms** (Codex/OpenCode/Hermes/OpenClaude) via `dz`. The npx package adds **project-level runtime governance** around the same skill: a slash command, governance rules, a context shard, and (for feature-adr) reward-learning + `/harvest`. So: pick the **skill/preset** for a working capability across platforms; pick the **npx toolkit** when you want it as a governed, command-driven fixture of one project.
1230
1307
 
1231
- ## Design custom Workflow loops (`workflow` · `workflow-lint` · `workflow-trace`)
1308
+ ## Design custom Workflow loops (`workflow` · `workflow run` · `workflow-lint` · `workflow-trace`)
1232
1309
 
1233
1310
  Custom loops used to be born by copy-pasting a 1470-line workflow script; nothing deterministic
1234
1311
  checked the copy. The loop-designer meta-factory replaces that: a versioned typed plan
@@ -1266,15 +1343,36 @@ rules honestly report `inconclusive` there, never a silent green). `dz workflow
1266
1343
  self-checks the shared-subsystem blob registry (checkpoints, model-resolver, trace, …) that the
1267
1344
  generator injects verbatim — edit the canonical TS in harness-core, regenerate, never the copies.
1268
1345
 
1269
- **Scope, stated plainly: `dz` AUTHORS, GATES and READS loops — it never RUNS one.** There is no
1270
- `dz workflow run`, and step 5 above is not a `dz` command: execution belongs to the Claude Code
1271
- host's `Workflow({scriptPath})` runtime, which owns the agent dispatch the generated script calls
1272
- into. So every claim on this page is about the plan, the generated script, the lint verdict, and the
1273
- run's own trace file not about runtime behaviour `dz` could observe itself. `dz workflow-trace`
1274
- reads what a HOST run already wrote (`trace.jsonl`, seq-ordered by the loop's own counter); with no
1275
- host run there is nothing to read, and it says so rather than inventing a timeline. Portability
1276
- follows from the same boundary: on a non-Claude-Code target the authoring and lint verbs work
1277
- unchanged, and only execution is absent.
1346
+ **Scope, stated plainly (and NARROWED since `dz workflow run` shipped): `dz` AUTHORS, GATES, READS
1347
+ and now RUNS loops but it never runs a RENDERED SCRIPT.** Step 5 above is still not a `dz`
1348
+ command: executing the generated script belongs to the Claude Code host's `Workflow({scriptPath})`
1349
+ runtime, which owns the agent dispatch that script calls into. What `dz workflow run` executes is
1350
+ the PLAN (see "Run a plan WITHOUT the Claude host" below) a second, independent enactor that
1351
+ dispatches to `codex exec` / `claude -p` and writes the SAME trace shape. So a claim on this page is
1352
+ about the plan, the generated script, the lint verdict, and a run's own trace file the two hosts
1353
+ are compared through the same reader, never assumed equivalent. `dz workflow-trace` reads what
1354
+ EITHER host wrote (`trace.jsonl`, seq-ordered by the loop's own counter); with no run there is
1355
+ nothing to read, and it says so rather than inventing a timeline.
1356
+
1357
+ **Who writes the trace, and how far the cross-host claim actually reaches (MEASURED 2026-08-20).**
1358
+ The two enactors do not attest their runs the same way, and the difference is load-bearing:
1359
+
1360
+ | Enactor | Who appends `trace.jsonl` | Evidentiary weight |
1361
+ |---|---|---|
1362
+ | `dz workflow run` (Codex **or** Claude family) | the `dz` process itself, `appendFileSync` in `cli.ts` | **instrument-written** |
1363
+ | the rendered script under the Claude host's `Workflow({scriptPath})` | an AGENT the script asks to run the flush command (`loop-render.ts`) | **agent-attested** |
1364
+
1365
+ The host runtime's own records cannot substitute for the second row. `journal.jsonl` carries four
1366
+ fields (`type`, `key`, `agentId`, `result`) — no `seq`, no `ts`, so it can order nothing; the
1367
+ per-agent `agent-*.jsonl` transcripts do carry `timestamp` and `uuid`/`parentUuid`, so they can
1368
+ order AGENT runs — but a join, a gate redo and a typed pause are steps of the loop, not agents, and
1369
+ leave no record there at all.
1370
+
1371
+ Consequently the cross-host structural equivalence proved by the committed fixture (`pkg-audit-1`)
1372
+ covers a bounded fanout, an all-activated join, a dep chain and a gate. It does **not** cover the
1373
+ gate redo route, the typed terminal route, the typed pause or the file deliverable: those four are
1374
+ what `discrimination.plan.json` adds, and no capture of them exists yet. Read every equivalence
1375
+ statement here as scoped to the first list.
1278
1376
 
1279
1377
  ### Build a loop for YOUR scenario — the end-to-end use case (a real one)
1280
1378
 
@@ -1441,6 +1539,97 @@ edit loop converges. What it cannot do is run the result: the generated script c
1441
1539
  `agent()`/`parallel()` sandbox, which only the Claude Code Workflow runtime provides. A green lint
1442
1540
  from Codex + a run under Claude Code is a legitimate two-agent split.
1443
1541
 
1542
+ ### Run a plan WITHOUT the Claude host (`dz workflow run`)
1543
+
1544
+ Everything above renders a plan into a script that only Claude Code's `Workflow({scriptPath})`
1545
+ runtime can execute. `dz workflow run` is the other half: it **interprets the plan itself**, from a
1546
+ plain shell, dispatching each step to `codex exec` or an isolated `claude -p`.
1547
+
1548
+ **When to use which** — one sentence each:
1549
+
1550
+ | | use it when |
1551
+ |---|---|
1552
+ | `workflow render` + `Workflow({scriptPath})` | you are already inside Claude Code and want the loop to run in that session, with its agents and its context |
1553
+ | `dz workflow run` | you are in a shell, in CI, or on a box with no Claude Code session — and you want the same plan enacted with a trace the same reader can read |
1554
+
1555
+ It interprets the PLAN, never the rendered script (a rendered script is a Claude-host artifact; a
1556
+ second enactor reading it would be reading someone else's implementation). Same `loop-plan/1`, same
1557
+ gate grammar, same join policies, same failure classes — those decisions live in ONE module both
1558
+ enactors consume, not in two lookalike copies.
1559
+
1560
+ ```bash
1561
+ dz workflow run audit.plan.json --run-id pkg-audit-2 --coder-family claude
1562
+ # → dz workflow run: completed (pkg-audit-2) — trace at .dz/loop-trace/pkg-audit-2/trace.jsonl
1563
+ # → {"schema":"wf-run-result/1","runId":"pkg-audit-2","status":"completed","exitCode":0}
1564
+
1565
+ # the run wrote into the addressing the reader already uses, so nothing new is needed to read it:
1566
+ dz workflow-trace --run pkg-audit-2 --invariants audit.plan.json
1567
+ # → INVARIANT PASS seq-monotonic: seq unique and contiguous 1..12 (12 events)
1568
+ # → INVARIANT PASS dispatch-settle-pairing: every dispatch has exactly one settle
1569
+ # → INVARIANT PASS join-coverage:fan: every dispatched branch settled; …
1570
+ ```
1571
+
1572
+ **Exit codes — `run` and `workflow-lint` have DIFFERENT tables. Both, side by side:**
1573
+
1574
+ | | 0 | 1 | 2 | 3 | 75 |
1575
+ |---|---|---|---|---|---|
1576
+ | `dz workflow run` | completed | failed (named reason) | usage / invalid plan | — | **typed pause** |
1577
+ | `dz workflow-lint` | clean | findings | — | inconclusive | — |
1578
+
1579
+ `75` is `EX_TEMPFAIL` ("try again later"), and it is deliberately **not** `3`: `3` collides with
1580
+ lint's inconclusive and reads ignorable, while a pause strands work that is genuinely resumable.
1581
+ On a pause the **last stdout line** is a `wf-pause-envelope/1` JSON object; a FAILURE emits none —
1582
+ so a wrapper distinguishes the two from stdout and the exit code alone, without parsing prose.
1583
+
1584
+ **Pause and resume** (a `kind: 'pause'` step, or the budget ceiling):
1585
+
1586
+ ```bash
1587
+ dz workflow run release.plan.json --run-id rel-7
1588
+ # → dz workflow run: PAUSED (AWAITING_APPROVAL) — resume with: dz workflow run release.plan.json --resume rel-7 --arg approve=<value>
1589
+ # → {"schema":"wf-pause-envelope/1","runId":"rel-7","exitCode":75,"pauseState":"AWAITING_APPROVAL", …}
1590
+ echo $? # 75
1591
+
1592
+ dz workflow run release.plan.json --resume rel-7 --arg approve=yes
1593
+ # → dz workflow run: completed (rel-7) — …
1594
+ ```
1595
+
1596
+ A resume never re-spends work: the cursor comes from checkpoint lines plus artifact probes, and a
1597
+ STALE-INPUT mismatch (plan digest, exec fingerprint, or the run-args hash) refuses to resume at all —
1598
+ there is no override, because the checkpoints describe a different run. Extending a ceiling is the
1599
+ one thing that is not an identity change: `--budget-extra` and `--wall-clock-extra` are recorded and
1600
+ capped, never silent.
1601
+
1602
+ **Budget.** Every boundary reserves its worst case BEFORE it dispatches, so a region that will not
1603
+ fit pauses in front of the region rather than halfway through it. `budget.jsonl` gets one
1604
+ `wf-budget-1` row per dispatch (plus probe rows, which never decrement the ceiling).
1605
+
1606
+ **Cross-model safety.** A step marked `x-role: "qe"` that resolves to the same family as
1607
+ `--coder-family` is REFUSED: the family that wrote the code may not review it. `--allow-same-family-qe`
1608
+ proceeds, and writes a real re-QE debt that `dz reqe` surfaces — a suspension you can see, not a
1609
+ comment nobody reads.
1610
+
1611
+ #### Honest limits — three divergences from the Claude host, named rather than discovered
1612
+
1613
+ 1. **dz-side settle events carry no `wallTime`.** The Claude host stamps it shell-side during its
1614
+ flush; the shared emitter accepts none, and stamping it afterwards would mean editing lines that
1615
+ were already validated — exactly what the buffer discipline forbids. Per-dispatch wall clock
1616
+ lives in `budget.jsonl` instead. `wallTime` was always diagnostic-only (never an operand of an
1617
+ invariant), so no verdict changes.
1618
+ 2. **The runner checkpoints every top-level stage unconditionally**, whatever `plan.checkpointing`
1619
+ says. Its resume cursor is BUILT from those lines, so making them optional would make resume
1620
+ optional. `plan.checkpointing` remains what it always was: the Claude-host opt-in.
1621
+ 3. **A gate `terminal:` route ends the run by plan design, and leaves the trace INCOMPLETE.** This is
1622
+ parity with the rendered script, whose top-level terminal `return` skips the epilogue that writes
1623
+ `run.closed`. The run exits 0 (the plan declared this ending; the Workflow host completes too) and
1624
+ the ledger row names the route — but `dz workflow-trace` will report the trace as incomplete and
1625
+ downgrade window-truncated invariants to `inconclusive`. That is correct: nothing proves the
1626
+ un-run steps would have passed.
1627
+
1628
+ **Deferred, and said so:** budget rows do not appear in the `dz workflow-trace` timeline yet. The
1629
+ condition for adding them was zero reader change for Claude-host runs, and it is not met — a
1630
+ Claude-host run has no `budget.jsonl`, so the timeline would grow a section that is empty for half
1631
+ its inputs. `budget.jsonl` is readable by eye and by the recommender in the meantime.
1632
+
1444
1633
  ### Move a run's telemetry to another machine (`workflow-trace export` / `import`)
1445
1634
 
1446
1635
  A run leaves traces on the machine that produced it. `export` puts one run's telemetry into a single
@@ -1511,9 +1700,9 @@ only at their documented scopes (plan + step), and `fanouts[].registry` items mu
1511
1700
  ItemKey domain the trace plane enforces — `trace.emit` can never decide whether a valid plan runs.
1512
1701
  Deferred options are on the loop-designer roadmap.
1513
1702
 
1514
- ## All Commands (68)
1703
+ ## All Commands (69)
1515
1704
 
1516
- *(67 MEASURED — reproducer: `grep -c "^ case '" src/cli.ts`, the dispatch cases.)*
1705
+ *(69 MEASURED — reproducer: `grep -c "^ case '" src/cli.ts`, the dispatch cases.)*
1517
1706
 
1518
1707
  ```
1519
1708
  dz setup --target <name> [--preset <name>] [--select id,id,...] [--skills-dir <dir>] [--memory agentdb] [--no-memory] [--no-hooks] [--install-driver] [--force]
@@ -1579,6 +1768,7 @@ dz epoch-replay --judge <filled-work-order.json> [--out <file>] [--json] # bli
1579
1768
  dz epoch-replay --score <judgments.json> --work-order <file> [--slice <name>] [--json] # un-blind against the VERIFIED pre-registered assignment; ONE paired binomial over DECISIVE pairs (ties excluded, reported) → SUPPORTED only when the lift interval (2p−1) lies entirely above zero; FALSIFIED only on harm or a passed non-superiority test (lift upper bound below the margin PRE-REGISTERED in the work order, default 0.05, at 10+ decisive pairs); else INCONCLUSIVE (min 5 decisive pairs). Refuses a forged work order, a --margin flag, or duplicate judgement ids; the verdict is data, not an exit code
1580
1769
  dz score --slug <feature> [--project <dir>] [--json] # process scorecard for ONE feature-adr run, from its artifacts: ADR confirmation, discrimination proof, cross-model QE grade, live verification, README-first, learning loop, amendments — DESCRIPTIVE-ONLY (a low score exits 0); evidence lines are shown so the reader judges the heuristics
1581
1770
  dz reqe [--slug <feature> [--done --report <f>]] [--project <dir>] [--json] # the re-QE debt ledger: a usage-switched feature-adr run whose Step-8 QE ran on the coder's OWN family (cross-model guard suspended, FR-2.9) records a debt; list debts (also surfaced by dz usage), print the cross-family review brief, settle FAIL-CLOSED against an existing GRADED report (the run's own 08_qe_report.md — even hard-linked — can never settle its own debt); settlement lands in 08_qe_report.md, evidence rotates to reqe-settled.json
1771
+ dz qe-bridge --family claude --slug <feature> [--coder-family codex|claude] [--model <id>] [--files a,b] [--out <f>] [--timeout <s>] [--allow-same-family] [--json] # the REVERSE QE bridge (Codex-hosted → Claude reviewer): the reviewer runs ISOLATED (an empty temp cwd + --safe-mode --strict-mcp-config --tools '' --no-session-persistence, so no CLAUDE.md/skills/plugins/hooks/MCP load) and its verdict is read from the --output-format json RESULT ENVELOPE, so text a customization printed onto the same stdout can never become a signoff. Probes the model first; sends SCOPED extracts under a loud 200k-char ceiling; the grade must agree across three LAST-anchored channels AND the marker must be the final content — empty/gradeless/mismatched/miscounted output is a named failure with an audit record under features/<slug>/.fa-state/qe-bridge/, never a clean review. exit 0 signoff parsed (ANY grade — it reports, it does not gate) / 1 named failure / 2 usage. DZ_QE_BRIDGE_CLAUDE_BIN is a TEST SEAM (recorded as binOverride:true)
1582
1772
  dz backlog <sub> add "<idea>" | list | show <id> | goals [--validate] | roulette [--seed n] [--commit <id>] | ship <id…> | drop <id…> | reopen <id…> | enrich <id> | jira <id> | harmonize [--apply] # brain-backed idea backlog: capture an idea → semantic dedup against past ideas/features via the REUSED agentdb vector engine (two-signal: bounded-excerpt cosine DUPLICATE≥0.92 corroborated by shared subject vocabulary — a register-only 0.94 is demoted to RELATED, a length-only re-capture is caught as a subset duplicate; absorbed texts kept in absorbed.jsonl) + GoalMap alignment ("map+compass") → weighted seeded roulette picks one to work on → enrich STAGES an idea2prd hand-off → jira writes an auditable outbox via a configurable MCP adapter seam (jira-mcp|copilot-mcp|none). No 2nd vector store; without agentdb it degrades to exact-text dedup (honest)
1583
1773
  dz sign --init --out <path> | --pack <dir> --key <path> # --init: generate the Ed25519 keypair (private OUTSIDE the repo, prints the public key for keys/dz.pub); else sign a pack's manifest + CycloneDX SBOM
1584
1774
  dz sbom --pack <dir> [--out <file>] # emit the CycloneDX 1.5 SBOM for a pack standalone (file-level bill of materials); print to stdout or write to a file
@@ -1592,10 +1782,11 @@ dz auto-canonicalize --source <github-url> --pack <skills-pack>
1592
1782
  dz sync-upstream [--package <dir>] [--list] [--all]
1593
1783
  dz drift-check [--all] [--json] [--project <dir>] # CI gate: exit 1 on NEW shared-skill drift (baseline: .dz/drift-allowlist.json; --all incl .claude dogfood)
1594
1784
  dz agents-sync [--check] [--json] [--project <dir>] # sync anchored bearing rules into the root AGENTS.md policy fence; exit 0 sync / 1 drift / 3 inconclusive
1595
- dz hooks-sync --target codex [--check] [--remove] [--json] # install + ARM the dz veto/recall hooks in $CODEX_HOME/hooks.json; exit 0 armed+trusted / 1 not armed / 3 inconclusive
1785
+ dz hooks-sync --target codex [--check] [--verify|--no-verify] [--project <dir>] [--remove] [--json] # install + ARM the dz veto/recall hooks in $CODEX_HOME/hooks.json and PROVE they fire with a live veto probe; exit 0 armed+trusted+verified / 1 not armed / 3 inconclusive (incl. --no-verify)
1596
1786
  dz sync-canonical <skill> [--check] [--from <dir>] [--auto] [--project <dir>] # heal every copy from skills-meta/<skill> or --from; no canonical + --check = compare copies to each other (exit 1 on drift); no canonical + write = refuse unless --auto (LOUD, picks most-complete copy); --check writes nothing
1597
1787
  dz scout [--topics <list>] [--since <date>] [--deep] [--output <file>] [--diff] [--report]
1598
1788
  dz workflow init --name <n> [--pattern pipeline|barrier|fanout|gate] [--o <plan.json>] | validate <plan.json> [--json] | render <plan.json> --o <script.js> [--check] [--force] | blobs [--check] # loop-plan/1 authoring (the ADR-005 templates are retired)
1789
+ dz workflow run <plan.json> [--run-id <id>] [--resume <runId>] [--arg k=v]... [--coder-family codex|claude] [--default-family codex|claude] [--budget <n>] [--max-wall-clock <s>] [--stage-timeout <s>] [--budget-extra <n>] [--wall-clock-extra <s>] [--run-dir <dir>] [--allow-same-family-qe] [--json] # INTERPRET the plan without the Claude host; exit 0/1/2/75 (75 = typed pause)
1599
1790
  dz workflow-lint <script.js> [--plan <plan.json>] [--require-plan|--legacy] [--json] # 18-rule deterministic gate; exit 0/1/3 — inconclusive is never a pass
1600
1791
  dz workflow-trace <runDir|--slug <s>|--run <id>> [--invariants <plan.json>] [--html <out.html>] [--json] # timeline + SEQ invariant runner over the loop's own trace.jsonl
1601
1792
  dz workflow-trace export <run> --o <file> [--include-pairs --yes] [--strict] # one run's telemetry as ONE movable file
@@ -2249,18 +2440,36 @@ install rewrites nothing and the hook keeps its trust.
2249
2440
 
2250
2441
  ```console
2251
2442
  $ dz hooks-sync --target codex
2252
- dz hooks-sync: codex hooks installed and ARMED (trust: trusted) — ready
2443
+ dz hooks-sync: codex hooks installed and ARMED (trust: trusted) — VERIFIED by a live veto probe — ready
2253
2444
 
2254
- $ dz hooks-sync --target codex --check
2255
- dz hooks-sync: codex hooks installed and ARMED (trust: trusted) — ready
2445
+ $ dz hooks-sync --target codex --check # read-only, and it re-proves the guard fires
2446
+ dz hooks-sync: codex hooks installed and ARMED (trust: trusted) — VERIFIED by a live veto probe — ready
2447
+
2448
+ $ dz hooks-sync --target codex --no-verify # skips the probe — and can never say "ready"
2449
+ dz hooks-sync: installed+trusted, NOT verified — ARMED = NO (trust: trusted, executable: true, verify: not verified (no live probe ran))
2256
2450
 
2257
2451
  $ dz hooks-sync --target codex --remove
2258
2452
  dz hooks-sync: removed 2 managed entr(ies) from /root/.codex/hooks.json
2259
2453
  ```
2260
2454
 
2261
- Exit codes are **0** for armed **and** trusted, **1** for not-armed / drift / a refusal, and **3**
2262
- when the answer is inconclusive (including "no `codex` binary on PATH", where dz writes **nothing**).
2263
- `--check` writes nothing and is **silent** in a home that never opted in.
2455
+ **"ready" means a command was actually blocked, in this run.** By default `dz hooks-sync` runs a
2456
+ LIVE, nonce-scoped veto probe through `codex exec` in a hermetic workspace: it asks Codex to run one
2457
+ forbidden command and requires BOTH halves of the evidence dz's `DZ-VETO:` marker in the
2458
+ transcript AND the absence of the command's side effect. `--dangerously-bypass-hook-trust` is never
2459
+ passed, because a bypassed run proves the helper body works and nothing about the installed state.
2460
+ Anything else — a silent transcript, a dead invocation, a timeout, a version mismatch — is
2461
+ **inconclusive**, never ready. `--no-verify` skips the probe and is reported as unverified;
2462
+ `--project <dir>` runs the probe in a project that has already opted into `"shellVeto": "block"`
2463
+ instead of the hermetic workspace.
2464
+
2465
+ Exit codes are **0** for armed **and** trusted **and** verified, **1** for not-armed / drift / a
2466
+ refusal, and **3** when the answer is inconclusive (including "no `codex` binary on PATH", where dz
2467
+ writes **nothing**, and `--no-verify`, where nothing was measured). `--check` writes nothing and is
2468
+ **silent** in a home that never opted in.
2469
+
2470
+ **`dz setup --target codex` and `dz init --target codex` deliver these hooks too**, verify them the
2471
+ same way, and report a failure without aborting the rest of the command. Pass `--no-hooks` for
2472
+ skills only.
2264
2473
 
2265
2474
  **What the two hooks do.**
2266
2475
 
@@ -2574,7 +2783,7 @@ dz pretrain # detects stack, recommends pres
2574
2783
  dz recommend "work on this Node.js API" # suggests skills + toolkits
2575
2784
 
2576
2785
  # 2. Install skills (choose your level)
2577
- dz setup --target claude-code --preset meta --memory agentdb # 18 skills (includes feature-adr)
2786
+ dz setup --target claude-code --preset meta --memory agentdb # 20 skills (includes feature-adr)
2578
2787
  dz setup --target claude-code --preset qe-engineer # + 20 QE skills
2579
2788
 
2580
2789
  # Want the full feature-adr toolkit with /feature-adr command + governance?
@@ -3748,9 +3957,18 @@ npx @dzhechkov/p-replicator init
3748
3957
 
3749
3958
  ## Status
3750
3959
 
3751
- `v0.4.8` — staged (not yet published). Also available as [Claude Plugin](#claude-plugin). Part of [DZ Harness Hub](https://github.com/djd1m/dz-harness-hub).
3960
+ `v0.5.1` — **published 2026-08-20**. Ships `dz workflow run`, the portable plan enactor: it INTERPRETS a `loop-plan/1` plan instead of executing a rendered script, dispatching to `codex exec` or an isolated `claude -p`, exit 0/1/2/75 (75 = a typed pause whose last stdout line is a `wf-pause-envelope/1`). Requires `@dzhechkov/harness-core >= 0.5.1` (the compat guard refuses below it by name). See "Who writes the trace" above for the stated scope of the cross-host equivalence claim — it is narrower than "the two hosts agree".
3961
+
3962
+ `v0.5.0` — published. Also available as [Claude Plugin](#claude-plugin). Part of [DZ Harness Hub](https://github.com/djd1m/dz-harness-hub).
3963
+
3964
+ New in 0.5.0 (feature `qe-bridge-claude`, cross-runtime leg 3/4): `dz qe-bridge --family claude`
3965
+ runs an INDEPENDENT Claude reviewer from any host — a Codex session included — and lands a PARSED
3966
+ signoff whose grade must agree across three LAST-anchored channels; an empty or gradeless answer is
3967
+ a named failure with a forensic record, never a clean review. `withNamedLockSync` generalises the
3968
+ store lock and now guards the `$CODEX_HOME/hooks.json` read-merge-write, so two dz processes can no
3969
+ longer lose each other's hook entries.
3752
3970
 
3753
- New in 0.4.8 (feature `crossrt-1-agents-md`): `dz agents-sync` ports the fixed registry of bearing
3971
+ Also in 0.4.8 (feature `crossrt-1-agents-md`): `dz agents-sync` ports the fixed registry of bearing
3754
3972
  rules into an early root-`AGENTS.md` fence, `--check` exposes source drift to CI, and both surfaces
3755
3973
  report the measured Codex project-doc byte budget. A live cold-start probe, not file presence,
3756
3974
  remains the runtime acceptance gate.