@try-works/dsh-recursive-mode 0.3.0 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (120) hide show
  1. package/README.md +959 -0
  2. package/lib/client.js +9 -2
  3. package/lib/closeout-report.d.ts +113 -0
  4. package/lib/closeout-standards.d.ts +35 -0
  5. package/lib/closeout.d.ts +12 -0
  6. package/lib/commands.d.ts +1 -1
  7. package/lib/config.d.ts +202 -0
  8. package/lib/delegation.d.ts +123 -3
  9. package/lib/enforcement.d.ts +90 -1
  10. package/lib/errors.d.ts +168 -0
  11. package/lib/git-context.d.ts +17 -0
  12. package/lib/guard-log.d.ts +39 -0
  13. package/lib/handoff.d.ts +29 -0
  14. package/lib/hooks.d.ts +103 -0
  15. package/lib/identity.d.ts +61 -0
  16. package/lib/index.d.ts +33 -12
  17. package/lib/index.js +10017 -3969
  18. package/lib/job-log.d.ts +34 -0
  19. package/lib/jobs-runner.d.ts +105 -0
  20. package/lib/json-safe.d.ts +33 -0
  21. package/lib/lock.d.ts +42 -0
  22. package/lib/memory-feedback.d.ts +52 -0
  23. package/lib/memory-select.d.ts +78 -0
  24. package/lib/memory.d.ts +137 -0
  25. package/lib/model-inventory.d.ts +106 -0
  26. package/lib/phase-graph.d.ts +111 -0
  27. package/lib/phase-rules.d.ts +67 -8
  28. package/lib/plan-gate.d.ts +68 -0
  29. package/lib/policy-globs.d.ts +222 -0
  30. package/lib/policy-write.d.ts +42 -0
  31. package/lib/policy.d.ts +39 -0
  32. package/lib/recursive_ask.tool.d.ts +88 -0
  33. package/lib/recursive_closeout.tool.d.ts +1 -1
  34. package/lib/recursive_delegate.tool.d.ts +22 -0
  35. package/lib/recursive_preview.tool.d.ts +48 -0
  36. package/lib/recursive_review.tool.d.ts +28 -0
  37. package/lib/result-cap.d.ts +70 -0
  38. package/lib/review-round.d.ts +82 -0
  39. package/lib/review.d.ts +9 -0
  40. package/lib/role-route.d.ts +122 -0
  41. package/lib/router.d.ts +90 -5
  42. package/lib/runtime.d.ts +252 -12
  43. package/lib/settlement.d.ts +132 -0
  44. package/lib/skills-phase.d.ts +71 -0
  45. package/lib/skills.d.ts +70 -0
  46. package/lib/status.d.ts +53 -1
  47. package/lib/teams-loop.d.ts +91 -2
  48. package/lib/training.d.ts +211 -0
  49. package/lib/ts-lint.d.ts +15 -0
  50. package/lib/types.d.ts +48 -0
  51. package/lib/workflow-audit.d.ts +207 -0
  52. package/package.json +31 -31
  53. package/preset/recursive.patch.yml +312 -0
  54. package/scripts/e2e-run.mjs +51 -0
  55. package/scripts/link-dsh.mjs +233 -0
  56. package/scripts/live/fake-llm.mjs +150 -0
  57. package/scripts/live-session-plugin.mjs +179 -0
  58. package/scripts/live-session-stock.mjs +106 -0
  59. package/scripts/live-session.mjs +139 -0
  60. package/skills/recursive-mode/SKILL.md +66 -0
  61. package/src/client/derive.ts +18 -2
  62. package/src/closeout-report.ts +274 -0
  63. package/src/closeout-standards.ts +102 -0
  64. package/src/closeout.ts +39 -2
  65. package/src/commands.ts +116 -4
  66. package/src/config.ts +113 -0
  67. package/src/delegation.ts +336 -18
  68. package/src/enforcement.ts +262 -72
  69. package/src/errors.ts +197 -0
  70. package/src/git-context.ts +33 -2
  71. package/src/guard-log.ts +134 -0
  72. package/src/handoff.ts +62 -0
  73. package/src/hooks.ts +316 -0
  74. package/src/identity.ts +230 -0
  75. package/src/index.ts +394 -20
  76. package/src/job-log.ts +112 -0
  77. package/src/jobs-runner.ts +222 -0
  78. package/src/json-safe.ts +75 -0
  79. package/src/lock.ts +153 -16
  80. package/src/memory-feedback.ts +185 -0
  81. package/src/memory-select.ts +187 -0
  82. package/src/memory.ts +309 -0
  83. package/src/model-inventory.ts +196 -0
  84. package/src/phase-graph.ts +191 -0
  85. package/src/phase-rules.ts +236 -0
  86. package/src/plan-gate.ts +111 -0
  87. package/src/policy-globs.ts +636 -0
  88. package/src/policy-write.ts +210 -0
  89. package/src/policy.ts +70 -5
  90. package/src/recursive_ask.tool.ts +276 -0
  91. package/src/recursive_audit_team.tool.ts +7 -3
  92. package/src/recursive_closeout.tool.ts +36 -35
  93. package/src/recursive_delegate.tool.ts +194 -0
  94. package/src/recursive_init.tool.ts +4 -3
  95. package/src/recursive_lint.tool.ts +81 -6
  96. package/src/recursive_lock.tool.ts +21 -4
  97. package/src/recursive_phase.tool.ts +3 -2
  98. package/src/recursive_preview.tool.ts +142 -0
  99. package/src/recursive_review.tool.ts +190 -0
  100. package/src/recursive_scratch.tool.ts +5 -4
  101. package/src/recursive_status.tool.ts +3 -2
  102. package/src/recursive_worktree.tool.ts +6 -5
  103. package/src/result-cap.ts +130 -0
  104. package/src/review-round.ts +335 -0
  105. package/src/review.ts +17 -3
  106. package/src/role-route.ts +230 -0
  107. package/src/router.ts +128 -2
  108. package/src/runtime.ts +968 -39
  109. package/src/settlement.ts +355 -0
  110. package/src/skills-phase.ts +143 -0
  111. package/src/skills.ts +151 -0
  112. package/src/snapshot.ts +39 -8
  113. package/src/status.ts +209 -4
  114. package/src/teams-loop.ts +223 -9
  115. package/src/training.ts +565 -0
  116. package/src/ts-lint.ts +38 -4
  117. package/src/types.ts +51 -0
  118. package/src/workflow-audit.ts +288 -0
  119. package/scripts/install-preset.cmd +0 -7
  120. package/scripts/install-preset.js +0 -101
package/README.md ADDED
@@ -0,0 +1,959 @@
1
+ # dsh-recursive-mode
2
+
3
+ **A 12-phase, evidence-gated workflow for coding agents — implemented as a DeepSeek Harness plugin.**
4
+
5
+ `@try-works/dsh-recursive-mode` turns "an agent wrote some code" into a **run** with a shape: requirements before
6
+ plans, plans before code, independent review before a lock, tests before a claim, and a receipt for every phase
7
+ that says who locked it and on what evidence. The discipline lives in the harness, not in a prompt, so it cannot
8
+ be skipped by an agent that is in a hurry.
9
+
10
+ - **12 tools** on the agent surface, one slash command, a workspace control plane, and a memory plane that
11
+ learns from what actually got used.
12
+ - **Zero runtime dependencies** beyond the harness itself — everything is a structural seam.
13
+ - **862 tests across 91 files**, three parity specs against the reference implementation, a live-session
14
+ harness, and a fresh-clone check that runs unattended.
15
+
16
+ > **Honest status, up front.** One capability in this repository is **implemented but NOT VERIFIED live**: the
17
+ > continuable review's repair leg (a child reviewer that receives a repair instruction for the same child). Its
18
+ > remaining blocker is in the host, not here, and is documented in [DSH-LIMITATIONS.md](DSH-LIMITATIONS.md)
19
+ > entries 9 and 10. Everything else described below was verified against a running harness. Where a claim in this
20
+ > document rests on a specific measurement, the measurement is named.
21
+
22
+ ---
23
+
24
+ ## Contents
25
+
26
+ 1. [Background: why this exists](#1-background-why-this-exists)
27
+ 2. [Purpose and design principles](#2-purpose-and-design-principles)
28
+ 3. [Architecture](#3-architecture)
29
+ 4. [Capabilities](#4-capabilities)
30
+ 5. [How a run works, phase by phase](#5-how-a-run-works-phase-by-phase)
31
+ 6. [How it works: the control plane](#6-how-it-works-the-control-plane)
32
+ 7. [How it works: the guard](#7-how-it-works-the-guard)
33
+ 8. [How it works: delegation and review](#8-how-it-works-delegation-and-review)
34
+ 9. [How it works: the memory plane](#9-how-it-works-the-memory-plane)
35
+ 10. [How it works: training](#10-how-it-works-training)
36
+ 11. [How it works: closeout as a linter](#11-how-it-works-closeout-as-a-linter)
37
+ 12. [How it works: mounting and the preset](#12-how-it-works-mounting-and-the-preset)
38
+ 13. [Getting started](#13-getting-started)
39
+ 14. [Verification: what is proven, and how](#14-verification-what-is-proven-and-how)
40
+ 15. [File map and further reading](#15-file-map-and-further-reading)
41
+
42
+ ---
43
+
44
+ ## 1. Background: why this exists
45
+
46
+ An agent with a shell and a test runner can produce working code. What it cannot reliably produce on its own is
47
+ **the record of how it knows**. Ask it afterwards why a decision was made, what it checked, what it could not
48
+ check, and which of its claims were verified, and you get a reconstruction — prose written after the fact, in
49
+ whatever shape the model happened to prefer that day.
50
+
51
+ **recursive-mode** is the answer to that: a workflow that makes the *evidence* a first-class artifact, produced
52
+ **before** the lock rather than after the question.
53
+
54
+ Three ideas do the work:
55
+
56
+ **Phases are ordered and monotonic.** Twelve artifacts, from `00-requirements.md` to `08-memory-impact.md`. Each
57
+ one locks, and a lock is one-way: an earlier phase cannot be rewritten after a later one exists. A run therefore
58
+ has a **history**, not a current state.
59
+
60
+ **Every claim has a container.** Phase artifacts have required sections — checked by a linter, not by a model's
61
+ judgement. A test summary has an "Execution Mode" and "Commands Executed (Exact)". A code review has an "Audit
62
+ Verdict". If the container is empty, the phase does not lock.
63
+
64
+ **Verification is independent.** The reviewer is not the author. Where a subagent is available, the review runs
65
+ on a child with its own context; where it is not, the fallback is **named** rather than hidden, so a self-audit
66
+ is never mistaken for an independent one.
67
+
68
+ The name is literal: the workflow applies to its own development. This plugin is built under itself — its own
69
+ control plane lives in `.recursive/` in this repository, and its own tracker
70
+ ([STRENGTHENING-PLAN.md](STRENGTHENING-PLAN.md)) records every defect it has found in its own behaviour.
71
+
72
+ ---
73
+
74
+ ## 2. Purpose and design principles
75
+
76
+ **Purpose.** Give a coding agent a workflow it can be *held to*: ordered phases, evidence gates, independent
77
+ review, a durable record — enforceable by the host on every tool call, and inspectable by a human afterwards
78
+ without asking the agent anything.
79
+
80
+ Five principles, each of which cost real defects to learn:
81
+
82
+ | Principle | What it means in code | Why |
83
+ |---|---|---|
84
+ | **The linter owns the standard** | Required sections live in `phase-rules.ts` and are enforced by `ts-lint.ts` (2175 lines) | A standard a model can restate is a standard a model can drift from |
85
+ | **The guard is a gate, not advice** | `tools/pre-execute` refuses writes to locked artifacts | A rule that is only in the prompt is a rule that holds until it is inconvenient |
86
+ | **Fail closed** | An unreadable verdict is a REVISE, never an APPROVE | The failure mode of "assume good" is an approval nobody gave |
87
+ | **Fail loud, and name the cause** | Every fallback reports *which* condition fired | Nine separate defects in this codebase were surfaces that said less than the code knew |
88
+ | **Structural seams, no hard dependencies** | Every host service is an optional interface resolved at composition | The plugin must load on a host that mounts none of them |
89
+
90
+ The fourth principle is the one this project learned the hard way. A representative example, from
91
+ [DSH-LIMITATIONS.md](DSH-LIMITATIONS.md): a fallback message that read *"no continuable seam or no live parent"*
92
+ covered **three** distinct conditions, so a live failure could not be diagnosed from its own report — three
93
+ rounds of experimentation went into a sentence that could have named the cause outright.
94
+
95
+ ---
96
+
97
+ ## 3. Architecture
98
+
99
+ ```mermaid
100
+ flowchart TB
101
+ subgraph host["DeepSeek Harness"]
102
+ LOADER["Loader / profiles"]
103
+ TOOLS["Tool runtime"]
104
+ SESSION["Session + event log"]
105
+ FS["fs observation"]
106
+ AGENT["Agent loop"]
107
+ SUB["ctx.subagents<br/>(optional)"]
108
+ WF["ctx.workflow<br/>(optional)"]
109
+ JOBS["ctx.jobs<br/>(optional)"]
110
+ end
111
+
112
+ subgraph plugin["@try-works/dsh-recursive-mode"]
113
+ IDX["index.ts<br/>composition + hooks"]
114
+ RT["runtime.ts<br/>RecursiveRuntime (1625 lines)"]
115
+ GUARD["enforcement.ts + policy-globs.ts<br/>the gate"]
116
+ LINT["ts-lint.ts + phase-rules.ts<br/>the standard"]
117
+ LOCK["lock.ts<br/>monotonic locks + receipts"]
118
+ DELEG["delegation.ts + router.ts<br/>who reviews"]
119
+ MEM["memory*.ts<br/>what gets injected"]
120
+ TOOLSET["12 recursive_* tools"]
121
+ end
122
+
123
+ subgraph disk["Workspace control plane (.recursive/)"]
124
+ RUNS["run/&lt;id&gt;/ artifacts"]
125
+ LOCKS["locks/ + receipts"]
126
+ MEMD["memory/"]
127
+ CFG["config/ + policy"]
128
+ end
129
+
130
+ LOADER -->|"mounts patch + preset"| IDX
131
+ IDX --> RT
132
+ IDX --> TOOLSET
133
+ TOOLSET --> RT
134
+ RT --> GUARD
135
+ RT --> LINT
136
+ RT --> LOCK
137
+ RT --> DELEG
138
+ RT --> MEM
139
+
140
+ TOOLS -.->|"tools/pre-execute"| GUARD
141
+ FS -.->|"fs/observed"| GUARD
142
+ SESSION -.->|"session/event"| DELEG
143
+ AGENT -.->|"agent/pre-step"| RT
144
+ SUB -.-> DELEG
145
+ WF -.-> RT
146
+ JOBS -.-> RT
147
+
148
+ LOCK --> LOCKS
149
+ LINT --> RUNS
150
+ MEM --> MEMD
151
+ GUARD --> CFG
152
+
153
+ classDef opt stroke-dasharray: 5 5
154
+ class SUB,WF,JOBS opt
155
+ ```
156
+
157
+ **Reading the diagram.** The four dashed lines into the plugin are the harness's own extension points — the
158
+ plugin does not poll and does not own a loop; it is *called* at four moments (a tool about to run, a file
159
+ observed, a session event committed, a step about to start). The three dashed services on the right are
160
+ **optional**: the plugin resolves each at composition and degrades with a *named* reason when one is absent. The
161
+ control plane on disk is the only state it trusts across restarts.
162
+
163
+ ---
164
+
165
+ ## 4. Capabilities
166
+
167
+ ### 4.1 The twelve tools
168
+
169
+ | Tool | What it does |
170
+ |---|---|
171
+ | `recursive_status` | Where the run is: phases, lock states, gates, next action |
172
+ | `recursive_init` | Scaffold a run and a workspace (templates, profile, memory skeleton) |
173
+ | `recursive_lock` | Lock a phase — refuses if the artifact does not meet the standard |
174
+ | `recursive_lint` | Lint a run or an artifact against the phase standard, with remediation text |
175
+ | `recursive_closeout` | Report what a phase artifact is missing; **writes a receipt, never the artifact** |
176
+ | `recursive_phase` | Read the phase graph: what is required, what is next, what is blocked |
177
+ | `recursive_worktree` | Create/promote a git worktree for a run, so work is isolated |
178
+ | `recursive_scratch` | The child's scratch space: durable, per-run working notes |
179
+ | `recursive_review` | **Independent review** of the phase artifact, with a repair path |
180
+ | `recursive_audit_team` | Fan a phase out across roles (audit) |
181
+ | `recursive_ask` | Ask the workspace a question, with the control plane as context |
182
+ | `recursive_preview` | Preview what a tool would do, without doing it |
183
+
184
+ ### 4.2 The command surface
185
+
186
+ One slash command, `/recursive <verb>`, usable with **no agent loop** — the verbs run the workspace's own code:
187
+
188
+ ```
189
+ /recursive status | spec | init | lock | qa | closeout | addendum | review | scratch | worktree | memory
190
+ /recursive bootstrap | list | help (global verbs)
191
+ ```
192
+
193
+ `memory` is the newest and is a good example of the design: it runs the *same* scorer the agent's context
194
+ injection uses, and prints the **score components** for every candidate — so "why did the agent get this
195
+ context?" is answerable without starting an agent:
196
+
197
+ ```
198
+ $ /recursive memory lock chain ordering --phase 04
199
+ ```
200
+
201
+ ### 4.3 Beyond the tools
202
+
203
+ - **A guard** that refuses writes to locked artifacts, including through the shell (`policy-globs.ts`, 636 lines
204
+ of pattern analysis).
205
+ - **A memory plane** with relevance scoring, phase applicability, and a feedback loop that learns from what was
206
+ actually injected.
207
+ - **A delegation layer** that routes a review to a native subagent, an external CLI, a self-audit, or a local
208
+ controller — and **always reports which**.
209
+ - **A closeout linter** over the whole run (00–08), with per-phase and run-level receipts.
210
+ - **A client half** (`platform: web`) so the run's state is visible in the harness UI.
211
+ - **A training path** that improves this plugin's prompts and its injected context from its own recorded
212
+ outcomes — extraction gated on phase 08 having actually locked, and on evidence being sufficient. See
213
+ [How it works: training](#10-how-it-works-training).
214
+
215
+ ---
216
+
217
+ ## 5. How a run works, phase by phase
218
+
219
+ ```mermaid
220
+ flowchart LR
221
+ R["00 requirements<br/>+ 00 worktree"] --> A["01 as-is"]
222
+ A --> RC["01.5 root cause"]
223
+ RC --> P["02 to-be plan"]
224
+ P --> I["03 implementation<br/>summary"]
225
+ I --> CR["03.5 code review"]
226
+ CR --> T["04 test summary"]
227
+ T --> Q["05 manual QA"]
228
+ Q --> D["06 decisions"]
229
+ D --> S["07 state"]
230
+ S --> M["08 memory impact"]
231
+
232
+ L(["each phase:<br/>write → lint → lock<br/>+ receipt"])
233
+ R -.-> L
234
+ CR -.->|"independent review<br/>+ repair"| CR
235
+
236
+ classDef gate fill:#eef,stroke:#446
237
+ class L gate
238
+ ```
239
+
240
+ Every arrow is enforced: a later artifact may not exist while an earlier one is unlocked, and the guard refuses
241
+ writes to a locked one. The phase rules are **profile-aware** — a `feature` run and an `audit` run ask for
242
+ different sections — and the required sections for each are read from `getArtifactRequiredSections`, so the
243
+ linter and the guard can never disagree about the standard.
244
+
245
+ **Phase 08 is where the workflow learns.** Locking `08-memory-impact.md` is the evidence a training pass waits
246
+ for, so the run's own record of *what it taught us* becomes input to the memory plane rather than a closing
247
+ formality — see [How it works: training](#10-how-it-works-training).
248
+
249
+ **A lock is a receipt, not just a marker.** When a phase locks, a receipt records the artifact's content hash,
250
+ the gate status, and the evidence that let it through. This is why a run can be audited months later: the
251
+ receipts are the *reasoning trail*, and they are files, not log lines.
252
+
253
+ **Phase artifacts are never rewritten by the tooling.** Where the phase document lacks a section, the closeout
254
+ says so and the **agent** adds it — the plugin's job is to check, not to author. (This was a design decision made
255
+ twice: an early version overwrote DRAFT artifacts, and a probe confirmed the agent's content was gone. See
256
+ [STRENGTHENING-PLAN.md](STRENGTHENING-PLAN.md), FU-8/FU-12.)
257
+
258
+ ---
259
+
260
+ ## 6. How it works: the control plane
261
+
262
+ ```
263
+ .recursive/ ← one control plane per workspace
264
+ ├── AGENTS.md, RECURSIVE.md, STATE.md, DECISIONS.md ← the agent-facing contract
265
+ ├── config/
266
+ │ ├── recursive-router.json ← which provider serves which role
267
+ │ └── policy (globs, enforcement) ← what the guard refuses
268
+ ├── memory/
269
+ │ ├── MEMORY.md ← the router file
270
+ │ ├── domains/ patterns/ incidents/ episodes/ archive/ skills/
271
+ │ └── .feedback.json ← what was applied vs contradicted
272
+ ├── memory-injections.json ← what context each phase was given
273
+ ├── run/<run-id>/
274
+ │ ├── 00-requirements.md … 08-memory-impact.md ← the artifacts
275
+ │ ├── 0N-*.receipt.json ← per-phase lock receipts
276
+ │ ├── closeout.receipt.json ← the whole-run receipt
277
+ │ ├── evidence/review-bundles/ ← what the reviewer was given
278
+ │ └── subagents/
279
+ │ ├── <delegation>/child-<id>/{brief.md,reply.md}
280
+ │ ├── *-action.md ← what the child claims it did
281
+ │ └── settlements.jsonl ← how each child round ended
282
+ └── locks/<stem>.lock ← the monotonic lock itself
283
+ ```
284
+
285
+ Two conventions matter more than they look:
286
+
287
+ **The control plane is found from the agent, not from the process.** `resolveControlPlaneRoot(agent, registry,
288
+ repoRoot)` resolves the workspace the *session* is in. A plugin that used `process.cwd()` would silently lint the
289
+ wrong repository when the host checkout and the workspace differ.
290
+
291
+ **The memory plane is read at `<controlPlaneRoot>/memory/<kind>/`.** This was measured, not assumed: a probe
292
+ showed `<dir>/memory/domains` is selected, `<dir>/.recursive/memory/...` is not, and using `.recursive` as the
293
+ root is selected again. A behaviour test that seeds the wrong path fails *exactly like* a broken implementation,
294
+ which is why the distinction is written down here.
295
+
296
+ ---
297
+
298
+ ## 7. How it works: the guard
299
+
300
+ The guard is the part an agent cannot talk its way past. It runs **before** a tool executes and on every
301
+ observed file write.
302
+
303
+ ```mermaid
304
+ sequenceDiagram
305
+ participant A as Agent
306
+ participant T as Tool runtime
307
+ participant G as Guard (enforcement.ts)
308
+ participant P as Policy globs
309
+ participant D as Control plane
310
+ participant L as Lock
311
+
312
+ A->>T: tool call (write/edit/shell)
313
+ T->>G: tools/pre-execute (payload, next)
314
+ G->>P: does this path match a protected pattern?
315
+ P->>G: yes → which rule
316
+ G->>D: resolve workspace root (from the AGENT, not cwd)
317
+ G->>L: is the target artifact LOCKED?
318
+ alt locked
319
+ L-->>G: LOCKED + receipt
320
+ G-->>T: refuse, with the rule and the receipt named
321
+ T-->>A: the call does not happen
322
+ else writable
323
+ G->>T: next()
324
+ T-->>A: the call runs
325
+ end
326
+
327
+ Note over G,D: fs/observed double-checks writes that bypass the tool path
328
+ ```
329
+
330
+ **Why two hooks.** `tools/pre-execute` covers everything that goes through the tool runtime; `fs/observed` catches
331
+ the rest. A shell command that redirects into a locked artifact is the case that motivates the second one.
332
+
333
+ **Why the refusal names the rule.** A refusal that says only "denied" teaches an agent to try a different route.
334
+ A refusal that names the pattern and the lock receipt ends the attempt, and gives a human the same information.
335
+
336
+ **Enforcement is staged, not binary.** `enforcement.ts` resolves a configuration that can be advisory or strict,
337
+ per rule — because a workflow that can only be on or off gets turned off.
338
+
339
+ ---
340
+
341
+ ## 8. How it works: delegation and review
342
+
343
+ The review is the phase where the workflow earns its name: **the reviewer is a different agent**. Getting the
344
+ review to the right provider, and knowing afterwards which one actually ran, is the whole of this subsystem.
345
+
346
+ ```mermaid
347
+ flowchart TB
348
+ REQ["recursive_review(phase)"] --> POL["router.ts<br/>resolveRole(role, policy, providers)"]
349
+ POL --> T1{"native provider<br/>registered?"}
350
+ T1 -->|yes| NAT["tier: native<br/>in-process subagent"]
351
+ T1 -->|no| T2{"external CLI<br/>configured + installed?"}
352
+ T2 -->|yes| EXT["tier: external-cli"]
353
+ T2 -->|no| T3{"policy fallback"}
354
+ T3 -->|local-controller| LC["tier: local-controller"]
355
+ T3 -->|self-audit| SA["tier: self-audit<br/>the agent reviews itself — NAMED, not hidden"]
356
+
357
+ NAT --> CONT["delegateContinuable()<br/>child + brief + reply contract"]
358
+ EXT --> CONT
359
+ CONT --> SETTLE["settlement.jsonl"]
360
+ SETTLE -->|APPROVE| LOCKIT["phase may lock"]
361
+ SETTLE -->|REVISE| REPAIR["repair instruction to the SAME child"]
362
+ REPAIR --> CONT
363
+ SA --> LOCKIT
364
+
365
+ classDef tier fill:#efe,stroke:#484
366
+ class NAT,EXT,LC,SA tier
367
+ ```
368
+
369
+ **Router.** `resolveRole` tries providers named `[role, 'spawn', 'fork', 'dsh-sdk']` in the provider map and
370
+ returns the **native** tier for the first one present; only then does it consider an external CLI and finally the
371
+ policy fallback. The provider map is built from the attached subagents service — and if that service cannot
372
+ enumerate its providers, the plugin reports the names it *did* see rather than pretending.
373
+
374
+ **The child gets a contract, in files.** Before any service call, the plugin writes
375
+ `subagents/<delegation>/child-<id>/brief.md`, which names the delegation, the artifact, the anti-patterns to look
376
+ for, and — critically — the **`Reply file:`** path the child must write and cite. The reply is a file because a
377
+ file survives a restart, a compaction, or a crash.
378
+
379
+ **One child per review, kept across rounds.** A REVISE is delivered as a follow-up to the **same** child, so the
380
+ reviewer retains its working set: the repair is checked by the agent that found the problem, not by a fresh one
381
+ with no memory of it.
382
+
383
+ **The settlement is the truth.** A round ends when a settlement lands in `subagents/settlements.jsonl` (captured
384
+ from `session/event`, with a cheap shape test first, because that event fires for *every* committed event). The
385
+ observer **waits** — bounded, polling, with a timeout — and reports "no settlement yet" only after actually
386
+ waiting. The park remains as the fallback, so an unobserved round is never an approval.
387
+
388
+ **Action records are written for every delegation**, and they carry `Execution Mode` and `Status`. After a
389
+ failed delegation, they now also carry **`Failure:`** — because `Status: failed` on its own cannot distinguish a
390
+ child that ran and failed from a child that never started. That distinction cost three rounds of investigation
391
+ before the field existed.
392
+
393
+ ---
394
+
395
+ ### The delegation map, phase by phase
396
+
397
+ §8 above describes *how* a delegation happens. This is *where* it can happen, and what each phase must then
398
+ prove. Three facts shape the picture:
399
+
400
+ 1. **Delegation is per-phase, and so is accountability.** A phase that used a subagent must record what the
401
+ main agent did with the result — the linter enforces this in strict profiles, and it is the reason a
402
+ subagent's claim is never taken on trust.
403
+ 2. **Not every phase can be delegated into.** `AUDITED_PHASE_FILES` names the nine phases whose artifacts carry
404
+ the audit contract: `01-as-is`, `01.5-root-cause`, `02-to-be-plan`, `03-implementation-summary`,
405
+ `03.5-code-review`, `04-test-summary`, `06-decisions-update`, `07-state-update`, `08-memory-impact`.
406
+ `00-requirements`, `00-worktree` and `05-manual-qa` are not in it.
407
+ 3. **`recursive_ask` is not a subagent tool.** It carries the workflow's three **human** gates —
408
+ `ASK_GATE_IDS = ['tdd-mode', 'qa-signoff', 'gate-block']` — as structured decisions rather than prose, so the
409
+ answer is validated and citeable.
410
+
411
+ ```mermaid
412
+ flowchart TB
413
+ subgraph main["the main agent, per phase"]
414
+ direction TB
415
+ P00["00 requirements · 00 worktree<br/><i>not audited</i>"]
416
+ P01["01 as-is · 01.5 root cause<br/><i>audited</i>"]
417
+ P02["02 to-be plan<br/><i>audited</i>"]
418
+ P03["03 implementation summary<br/><i>audited</i>"]
419
+ P035["03.5 code review<br/><i>audited</i>"]
420
+ P04["04 test summary<br/><i>audited</i>"]
421
+ P05["05 manual QA<br/><i>not audited</i>"]
422
+ P06["06 decisions · 07 state<br/><i>audited</i>"]
423
+ P08["08 memory impact<br/><i>audited</i>"]
424
+ end
425
+
426
+ P00 --> P01 --> P02 --> P03 --> P035 --> P04 --> P05 --> P06 --> P08
427
+
428
+ ASK["recursive_ask<br/>HUMAN gates"]
429
+ P00 -.->|"tdd-mode"| ASK
430
+ P05 -.->|"qa-signoff"| ASK
431
+ P02 -.->|"gate-block, when policy says ask"| ASK
432
+
433
+ REV["recursive_review(role)<br/>code-reviewer · auditor · reviewer · memory-auditor"]
434
+ P035 ==>|"the review phase"| REV
435
+ P08 -.->|"memory-auditor role"| REV
436
+
437
+ TEAM["recursive_audit_team<br/>one phase per ROLE, one item per reviewer"]
438
+ P01 -.-> TEAM
439
+ P02 -.-> TEAM
440
+ P03 -.-> TEAM
441
+ P035 -.-> TEAM
442
+ P04 -.-> TEAM
443
+ P08 -.-> TEAM
444
+
445
+ subgraph contract["what EVERY audited phase must then record in its own artifact"]
446
+ direction LR
447
+ C1["Reviewed<br/>Action Records<br/><i>in-run only</i>"]
448
+ C2["Main-Agent<br/>Verification<br/>Performed"]
449
+ C3["Acceptance decision<br/>accepted · partially<br/>accepted · rejected"]
450
+ C4["Refresh<br/>Handling"]
451
+ C5["Repair Performed<br/>After Verification"]
452
+ end
453
+
454
+ P01 --> contract
455
+ P02 --> contract
456
+ P03 --> contract
457
+ P035 --> contract
458
+ P04 --> contract
459
+ P06 --> contract
460
+ P08 --> contract
461
+
462
+ subgraph train["08 also closes the learning loop"]
463
+ T1["phase 08 locks"] --> T2["trainingGate"] --> T3["extract · group · write shard<br/>→ memory/MEMORY.md"]
464
+ end
465
+ P08 --> T1
466
+
467
+ classDef audited fill:#efe,stroke:#484
468
+ classDef notaudited fill:#f5f5f5,stroke:#999
469
+ classDef human fill:#eef,stroke:#446
470
+ class P01,P02,P03,P035,P04,P06,P08 audited
471
+ class P00,P05 notaudited
472
+ class ASK human
473
+ ```
474
+
475
+ **Reading it.** The heavy arrow marks the phase the workflow **depends** on delegation for: **03.5** is the
476
+ review phase, and `recursive_review` routes it by role (default `code-reviewer`). The dotted arrows into
477
+ `recursive_audit_team` are the *optional* fan-out — one phase per role, one item per reviewer, so the engine's
478
+ progress reads as a per-phase per-role review rather than an undifferentiated pile. The dotted arrows into
479
+ `recursive_ask` are **human** decisions, not subagents. Every audited phase feeds the same five-part contract, and
480
+ **08** additionally triggers the training pass from §10.
481
+
482
+ > ### ⚠ This map shows where delegation is *expected*, not where it is *permitted*
483
+ >
484
+ > **The main agent can delegate in any phase that has an artifact.** Nothing in this plugin gates delegation by
485
+ > phase — a grep for a `03.5`-only restriction finds none. `recursive_review` resolves the phase as
486
+ > *"the one a review is actually about, because that is the artifact whose lock the review gates"*, defaults it to
487
+ > the run's **current** phase, and rejects a phase for exactly one reason: **`phase <n> has no artifact to review
488
+ > yet`**. Pass any role, any phase that has an artifact, and it will run.
489
+ >
490
+ > What *is* phase-dependent is **accountability**, not capability:
491
+ >
492
+ > | | Any phase with an artifact | The nine `AUDITED_PHASE_FILES` |
493
+ > |---|---|---|
494
+ > | `recursive_review` works | ✅ | ✅ |
495
+ > | `recursive_audit_team` fan-out | ✅ | ✅ |
496
+ > | the artifact must record `Subagent Contribution Verification` | **only in strict profiles, and only for audited phases** | ✅ required |
497
+ >
498
+ > So `00-requirements`, `00-worktree` and `05-manual-qa` **can** be reviewed — they are simply not in the set
499
+ > whose artifacts must carry the five-part verification. And the plugin says so to the agent directly: each
500
+ > phase's skill carries `audited: yes — this phase needs a delegated audit` or `audited: no`
501
+ > (`skills-phase.ts`), so the expectation is stated per phase rather than implied.
502
+ >
503
+ > **Two delegation paths exist, and they are different animals.** The workflow's own — `recursive_review`,
504
+ > `recursive_audit_team` — is phase-aware and writes evidence **into the run** (brief, reply, action record,
505
+ > settlement), which is what makes the per-phase contract checkable. The harness's generic subagent and team
506
+ > tools are always available and know nothing about phases; work delegated through those leaves no run-scoped
507
+ > record unless the agent writes one. **Use the workflow's path when the phase's artifact must prove something.**
508
+
509
+
510
+ ### What each phase owes after delegating
511
+
512
+ The contract is enforced, not advisory: in a strict profile (`recursive-mode-audit-v2`, `recursive-mode-audit-v1`)
513
+ every artifact in `AUDITED_PHASE_FILES` is checked for `## Subagent Contribution Verification`, and the section is
514
+ rejected unless it records all five parts:
515
+
516
+ | Must record | Enforced because | Failure message |
517
+ |---|---|---|
518
+ | **Reviewed Action Records** | a delegation claim must cite the run's own `*-action.md` files | `Subagent Contribution Verification must record Reviewed Action Records` |
519
+ | **Main-Agent Verification Performed** | *the main agent*, not the child, is accountable for what the phase asserts | `… must record Main-Agent Verification Performed` |
520
+ | **Acceptance decision** | `accepted` / `partially accepted` / `rejected` — a decision, not an impression | `… must record an Acceptance Decision` |
521
+ | **Refresh Handling** | work redone after the review must be declared | `… must record Refresh Handling` (`n/a` and `none` are not meaningful values) |
522
+ | **Repair Performed After Verification** | a found defect must be shown repaired | `… must record Repair Performed After Verification` |
523
+
524
+ Two further rules close the obvious loopholes: an action record cited from **outside the run** is rejected
525
+ (`may only reference action records in this run`), and the section may not be empty. Together they mean a phase
526
+ cannot claim "a subagent did this" without naming the record, stating that the main agent checked it, and saying
527
+ what it decided.
528
+
529
+ **Why this is the answer to "can I just delegate this phase?"** — you can delegate the *work*, and the phase
530
+ artifact still has to carry the *judgement*. The workflow's position is that delegation moves effort, never
531
+ accountability.
532
+
533
+
534
+
535
+ ## 9. How it works: the memory plane
536
+
537
+ Memory here is not a vector store; it is **markdown files with a scorer in front of them**, and the scorer's
538
+ inputs are visible.
539
+
540
+ ```mermaid
541
+ flowchart LR
542
+ subgraph sources["memory/&lt;kind&gt;/*.md"]
543
+ D1["domains/"]
544
+ D2["patterns/"]
545
+ D3["incidents/"]
546
+ D4["episodes/"]
547
+ end
548
+
549
+ Q["query:<br/>the run's own<br/>00-requirements.md<br/>(first 4000 chars)"] --> SEL
550
+ F["changed paths<br/>for this phase"] --> SEL
551
+ P["current phase"] --> SEL
552
+ FB["feedback book<br/>.feedback.json"] --> SEL
553
+
554
+ D1 --> IDX["loadMemoryIndex()"] --> SEL["selectMemory()<br/>score + rank + render"]
555
+ D2 --> IDX
556
+ D3 --> IDX
557
+ D4 --> IDX
558
+
559
+ SEL --> OUT["top N shards →<br/>the review bundle"]
560
+ SEL --> REC["recordInjection()<br/>memory-injections.json"]
561
+ REC --> SET["settleInjections() at closeout<br/>over the LOCKED phases only"]
562
+ SET --> FB
563
+
564
+ classDef input fill:#eef,stroke:#446
565
+ class Q,F,P,FB input
566
+ ```
567
+
568
+ **Scoring is additive and inspectable.** A shard scores for matching the query, for overlapping the changed
569
+ paths (`MEMORY_PATH_MATCH_WEIGHT = 3`), for applying to the current phase (`MEMORY_PHASE_MATCH_WEIGHT = 2`), and
570
+ for its feedback history (`feedbackBonus`, clamped to ±1). `explainMemorySelection` returns the **components** per
571
+ shard, which is what `/recursive memory` prints — no agent loop, no guessing.
572
+
573
+ **The explainer and production share the scorer.** A spec asserts that both produce the **same order** for every
574
+ option set — including a feedback case, a phase case, and a key that matches nothing. If they ever disagree, the
575
+ test fails immediately rather than after someone notices the context is wrong.
576
+
577
+ **The feedback loop closes at closeout.** Injections are recorded where selection happens
578
+ (`memory-injections.json`), and settled at closeout — but **only over phases that actually LOCKED**, because a
579
+ shard that was injected into work that never locked taught nobody anything. `settleInjections(root, runDir,
580
+ PHASE_SEQUENCE.filter(file => getLockStatus(join(runDir, file)) === 'LOCKED'))` is the whole rule, in one line.
581
+
582
+ **Feedback is a nudge, never a verdict.** `feedbackBonus` moves a shard by at most ±1, which is deliberately
583
+ less than a path or phase match: history should break ties, not overrule relevance.
584
+
585
+ ### The behaviour test that makes the loader matter
586
+
587
+ The only acceptance test that proves the *loader* (rather than the scorer) is this: run the same workflow twice
588
+ in the FU-1 harness, changing **one** shard, and assert the model was told something different. It is
589
+ intentionally end-to-end, because a unit test on `selectMemory` cannot prove the selected shards reach a prompt.
590
+ It failed twice before it passed, and the cause was the **seeding path**, not the wording — which is exactly the
591
+ kind of false conclusion a unit test would have let stand.
592
+
593
+ ---
594
+
595
+ ## 10. How it works: training
596
+
597
+ The workflow closes on `08-memory-impact.md` — the phase whose whole subject is *what did this run teach us?*
598
+ **Training is what makes that phase more than a document:** when phase 08 locks, the plugin can extract the
599
+ run's learnings, group them, and write them into the memory plane as shards that future runs will be scored
600
+ against.
601
+
602
+ ```mermaid
603
+ flowchart TB
604
+ P8["08-memory-impact.md locks"] --> COUNT["countPhase8LockedRuns(root)<br/>how many runs have locked it?"]
605
+ COUNT --> GATE{"trainingGate(lockedRuns)"}
606
+ GATE -->|"too few"| INSUF["INSUFFICIENT_EVIDENCE<br/>refuse, and say why"]
607
+ GATE -->|"no extractor configured"| NOEX["EXTRACTOR_UNAVAILABLE<br/>refuse, and say why"]
608
+ GATE -->|"OK"| TRIG["runPhase8Trigger()"]
609
+
610
+ TRIG --> CHOOSE{"which extractor?"}
611
+ CHOOSE -->|cmd| CMD["RECURSIVE_TRAINING_EXTRACTOR_CMD<br/>spawnSync, stdio: 'ignore'"]
612
+ CHOOSE -->|file| FILE["RECURSIVE_TRAINING_RESPONSE_FILE<br/>the answer is a FILE"]
613
+ CMD --> ITEMS["parseExtractorItems(payload)"]
614
+ FILE --> ITEMS
615
+
616
+ ITEMS --> INF["inferSubsystem(item)"]
617
+ INF --> GROUP["groupLearnings(items, isWinner)<br/>winners vs losers per subsystem"]
618
+ GROUP --> SHARD["renderGroupShard() / renderTaskTypeShard()"]
619
+ SHARD --> WRITE["write the shard into memory/"]
620
+ WRITE --> REG["updateMemoryRegistry()<br/>registryLine() → MEMORY.md"]
621
+ REG --> NEXT(["the memory plane's next run scores it"])
622
+
623
+ classDef refuse fill:#fee,stroke:#a44
624
+ class INSUF,NOEX refuse
625
+ ```
626
+
627
+ ### Why an external extractor, and why a file
628
+
629
+ The plugin ships **no LLM client**. Extraction is delegated to a command the operator configures
630
+ (`RECURSIVE_TRAINING_EXTRACTOR_CMD`) or to a **response file** (`RECURSIVE_TRAINING_RESPONSE_FILE`) — and the
631
+ file is not a convenience:
632
+
633
+ > This harness's sandbox denies a child process the **piped stdio** a capture needs, so a spawn that read the
634
+ > extractor's stdout fails with **EPERM in the environment it runs in**. The extractor therefore **writes its
635
+ > answer to a path**, and the plugin spawns with `stdio: 'ignore'`.
636
+
637
+ Two consequences are visible in the code and worth knowing if you work on it:
638
+
639
+ - **The spawn is injected, not performed, inside the module.** `spawnExtractorRunner` takes the runner as an
640
+ argument and `runExtractor` is asserted with a fake, because a module that spawned directly would be
641
+ untestable under the same sandbox. The real spawn lives at the caller (`runtime.ts`).
642
+ - **"Configured" is not "working".** An e2e run walked into exactly that footgun: the env var was set, the spawn
643
+ was wired, and the trigger still reported no extractor. That is why the unavailable case is a **named code**
644
+ (`EXTRACTOR_UNAVAILABLE`) rather than an empty result.
645
+
646
+ ### What the gate is for
647
+
648
+ `trainingGate(lockedRuns)` refuses below a threshold, with the code `INSUFFICIENT_EVIDENCE`. This is the same
649
+ principle as the rest of the plugin applied to *learning*: **a shard inferred from two runs is a rumour**, and a
650
+ memory plane that fills with rumours is worse than one that stays empty, because the scorer cannot tell the
651
+ difference. Phase 08 is the unit of evidence — a run has to have *reached* the memory-impact phase, and locked
652
+ it, before it counts.
653
+
654
+ ### What comes out
655
+
656
+ Items are attributed to a **subsystem** (`inferSubsystem`) and grouped by **outcome** — `groupLearnings(items,
657
+ isWinner)` separates what won from what lost — then rendered as shards and registered:
658
+
659
+ | Output | Where it goes |
660
+ |---|---|
661
+ | A grouped shard per subsystem / task type | `renderGroupShard`, `renderTaskTypeShard` → `memory/` |
662
+ | A registry line pointing at the new shard | `registryLine` + `updateMemoryRegistry` → `memory/MEMORY.md` |
663
+
664
+ Which closes the loop the [memory plane](#9-how-it-works-the-memory-plane) opened: training **writes** shards,
665
+ `selectMemory` **scores** them, `recordInjection` remembers what was used, `settleInjections` settles it at
666
+ closeout, and `feedbackBonus` nudges the next selection. The plugin's prompts and its context are meant to
667
+ improve from its own recorded outcomes rather than from someone's recollection of them.
668
+
669
+ **Design and status:** [TRAINING.md](TRAINING.md) is the design — including the phases P1–P5, each with a live
670
+ acceptance, and the measured comparison with the reference implementation it deliberately does **not** copy.
671
+ Its non-goals are as load-bearing as its goals: this is not a fine-tuning pipeline, and it does not ship a model
672
+ client.
673
+
674
+ ---
675
+
676
+ ## 11. How it works: closeout as a linter
677
+
678
+ Closeout is not a writer. It is a **linter over the run**, plus a receipt:
679
+
680
+ ```mermaid
681
+ flowchart TB
682
+ START["recursive_closeout or /recursive closeout"] --> READ["read the phase artifact<br/>+ its required sections<br/>(profile-aware)"]
683
+ READ --> GATE{"artifact exists?<br/>sections present?<br/>gate fields filled?"}
684
+ GATE -->|missing| FIND["findings, each naming<br/>the section and<br/>copy-paste remediation"]
685
+ GATE -->|complete| OK["fits the standard"]
686
+ READ --> PREREQ["advisory: earlier phases<br/>not yet LOCKED"]
687
+ FIND --> RECEIPT
688
+ OK --> RECEIPT
689
+ PREREQ --> RECEIPT
690
+ RECEIPT["write OWN receipt(s)<br/>locks/&lt;stem&gt;.closeout.receipt.json<br/>and at 08: run/&lt;id&gt;/closeout.receipt.json"]
691
+ RECEIPT --> NEVER["⚠ the phase artifact is<br/>NEVER modified"]
692
+
693
+ classDef never fill:#fee,stroke:#a44
694
+ class NEVER never
695
+ ```
696
+
697
+ **Why a linter and not a fixer.** The agent is the author; the tool is the check. An earlier version of this
698
+ plugin *did* scaffold missing sections, and a probe showed it overwriting the agent's own content — so a phase
699
+ that the agent had written correctly could be silently replaced by a stub. The closeout now reports, and the
700
+ agent fills the gap. A **receipt** is written, because a check that leaves no trace is a check nobody can audit.
701
+
702
+ **The receipt covers the whole run.** `runCloseoutReport` walks `RUN_ARTIFACT_SEQUENCE` (00–08) and reports each
703
+ artifact's state — which is why a run can be summarised without reading nine files.
704
+
705
+ **The command surface agrees with the tool.** `/recursive closeout <run> --phase 04` runs the same report and
706
+ prints findings and advisories. It used to return *"closeout scaffolded for …"* while reading nothing and
707
+ checking nothing — a command that reported success for work it never did. That defect is why this document
708
+ describes the **linter** behaviour as a contract rather than a detail.
709
+
710
+ ---
711
+
712
+ ## 12. How it works: mounting and the preset
713
+
714
+ The plugin ships **two halves that mount in different planes**, and knowing which is which answers most
715
+ "why isn't this working?" questions:
716
+
717
+ | Half | Artifact | Plane | What it carries |
718
+ |---|---|---|---|
719
+ | **Bundle** | `cordis.patch.yml` (declared as `dsh.bundle.patch`) | **profile** | one enabled row so the host's ClientModuleRegistry can discover the UI half |
720
+ | **Agent preset** | `preset/recursive/agent.cordis.yml` + `preset/recursive/preset.yml` | **agent plane** | **the entire server surface**: the `RecursiveRuntime` service, the twelve tools, `/recursive`, the policy prompt section |
721
+
722
+ ```mermaid
723
+ flowchart TB
724
+ subgraph profilePlane["PROFILE plane (installed once)"]
725
+ PKG["@try-works/dsh-recursive-mode<br/>profiles/&lt;profile&gt;/node_modules/…/lib/index.js"]
726
+ SHELL["cordis.patch.yml<br/>insert: id recursive<br/>config: shellOnly: true"]
727
+ NOOP["apply() is a NO-OP on the server root:<br/>no tools, no command, no projection"]
728
+ CLIENT["client discovery:<br/>/plugins/&lt;id&gt;/client.js"]
729
+ PKG --> SHELL --> NOOP
730
+ SHELL --> CLIENT
731
+ end
732
+
733
+ INSTALL["preset/recursive.patch.yml<br/>insert: id preset-recursive<br/>name: @deepseek-ai/dsh-agent-preset"]
734
+ HOME["the agent-preset registry, in process<br/>NOTHING is written to the DSH home"]
735
+ INSTALL -->|"the bundle patch declares it"| HOME
736
+ PKG -.->|"the preset's server row names the PACKAGE,<br/>resolved from the profile node_modules,<br/>so no absolute path is written anywhere"| HOME
737
+
738
+ subgraph agentPlane["AGENT plane (each session that selects it)"]
739
+ SEL["session selects the recursive preset<br/>registered by the bundle row"]
740
+ STD["the standard coding agent surface<br/>+ tool-presentation mode: both"]
741
+ REALM["group recursive-realm<br/>isolate: true"]
742
+ SURF["RecursiveRuntime + 12 tools<br/>+ /recursive + recursive:policy"]
743
+ SEL --> STD
744
+ SEL --> REALM --> SURF
745
+ end
746
+
747
+ HOME --> SEL
748
+
749
+ classDef shell fill:#eef,stroke:#446
750
+ class SHELL,CLIENT shell
751
+ classDef isolated fill:#efe,stroke:#484
752
+ class REALM,SURF isolated
753
+ ```
754
+
755
+ ### Installing the preset
756
+
757
+ **There is nothing to install.** A preset is **declared by a plugin row**, not discovered from a directory, so
758
+ enabling the bundle and letting it mount *is* the installation. No script, no post-install step, and nothing
759
+ written into the DSH home.
760
+
761
+ > **⚠ THIS SECTION USED TO DESCRIBE AN INSTALLER SCRIPT, AND IT WAS WRONG.** It documented
762
+ > `node scripts/install-preset.js --profile web`, a destination of `~/.dsh/.agent-presets/recursive/`, and the
763
+ > claim that **`agentPreset.list` then returns `recursive`**. A screenshot of Settings → Agent presets showed an
764
+ > **empty CUSTOM section** while the file sat exactly where that script put it, which falsified all three claims
765
+ > at once. The script had no bug: it wrote a correct file to a path nothing reads. It is now deleted, and the
766
+ > correction is recorded here rather than quietly edited away, because a document asserting a behaviour that does
767
+ > not occur is the defect this section used to be.
768
+
769
+ How the declaration works is in [§12.1](#121-how-the-preset-is-declared-and-how-it-used-to-be-wrong) below: the
770
+ bundle patch ships a `- insert:` row naming `@deepseek-ai/dsh-agent-preset`, whose config **is** the preset
771
+ definition, and `dsh.bundle.patch` lists it as an **array** — which is what a package that ships a preset does.
772
+
773
+ ### 12.1 How the preset is declared (and how it used to be, wrongly)
774
+
775
+ A preset is **declared by a plugin row**, not discovered from a directory. `preset/recursive.patch.yml` ships the
776
+ declaration, and the bundle's manifest lists it:
777
+
778
+ ```yaml
779
+ - insert:
780
+ - id: preset-recursive
781
+ name: '@deepseek-ai/dsh-agent-preset'
782
+ config:
783
+ id: recursive
784
+ order: 25
785
+ plugins: [ …the composition, verbatim… ]
786
+ ```
787
+
788
+ `@deepseek-ai/dsh-agent-preset` is the row type whose only work is `ctx.agentPresets.register(this.config)`, and it
789
+ sets `EntryGroup.key`, documented in the harness as *"preserve child expressions until their own plugins
790
+ activate"* — which is why the two `!!js` platform switches stay expressions rather than being evaluated at
791
+ packaging time. **`dsh.bundle.patch` is an ARRAY** for a package that ships a preset: the harness `web-app` bundle
792
+ lists its main patch plus one patch per built-in preset the same way, and every plugin that ships *no* preset uses
793
+ a plain string.
794
+
795
+ **`preset/recursive/agent.cordis.yml` remains the source composition**, and a parity spec proves the patch agrees
796
+ with it line for line — one definition with two consumers proven equal, rather than two copies free to drift.
797
+
798
+ **What this replaced, and why the old way could not work.** `scripts/install-preset.js` wrote
799
+ `~/.dsh/.agent-presets/recursive/agent.cordis.yml` through a temp file and a rename, and printed success. **That
800
+ directory is not a discovery path.** The registry's documentation says its parameter is *"Parsed configuration
801
+ supplied by the declaring plugin"*, and the settings UI renders whatever `remote.agentPresets.list()` returns — so a
802
+ correct file on disk produced an empty CUSTOM section, and the installer's success message was true about the file
803
+ and false about the effect.
804
+
805
+ ### Why the realm is not optional
806
+
807
+ `agent.cordis.yml` is an **agent-plane composition**, and its own header states the constraint in the strongest
808
+ terms available:
809
+
810
+ > A service row here **MUST** sit inside a group carrying an `isolate` realm. Without one it publishes into the
811
+ > root realm, where it is process-global — another preset publishing the same name collides, and a host reader
812
+ > would resolve one preset's instance for every session; `dsh-agent-presets` **rejects that at mount**.
813
+
814
+ So the surface lives inside `- id: recursive-realm` / `name: cordis:group` / `group: true` / `isolate: true` —
815
+ one private `RecursiveRuntime` per session, apart from every other preset's. Note the header's precision about
816
+ why a shared **label** would not do: `provide()` throws on a second registration under the same realm symbol, and
817
+ labels **join realms rather than pooling instances**.
818
+
819
+ ### What the preset deliberately does not own
820
+
821
+ The host composition keeps the registries themselves, the sandbox and approval stack, persistence, and the model
822
+ route. A preset that mounted those would claim authority it should not have — and would collide with the profile
823
+ that legitimately owns them.
824
+
825
+ **Why the split.** Client discovery requires an **enabled bare-name entry** in the loader's profile
826
+ (`entry.fiber !== undefined && !entry.disabled`), and preset-mounted subtrees are absent from
827
+ `ctx.loader.entries()`. So the bundle contributes a single enabled row whose `apply()` deliberately does nothing
828
+ on the server — it exists so the UI half can be discovered. The **full** server surface belongs to the preset,
829
+ where it is isolated per session: installing the plugin does not impose the workflow on every agent in the
830
+ profile, and opting in does not require editing the host profile by hand.
831
+
832
+ **Optionality everywhere.** `ctx.get('goals')`, `ctx.get('jobs')`, `ctx.get('subagents')`, `ctx.get('workflow')`
833
+ are all resolved at composition and may be `undefined`. Two lessons are encoded in that:
834
+
835
+ - **`ctx.provide` alone does not reach `ctx.get`** — the value must also be `ctx.set`. The failure is silent:
836
+ the consumer reports its own "not mounted" path, which reads like a configuration choice.
837
+ - **A one-shot `ctx.get` at apply time can miss a service mounted by a later layer.** The subagents seam is
838
+ therefore attached through `ctx.inject('subagents', …)`, the harness's own pattern, so the seam arrives
839
+ whenever the service does. A live run proved the cost of relying on the one-shot read.
840
+
841
+ ---
842
+
843
+ ## 13. Getting started
844
+
845
+ **Requirements.** A DeepSeek Harness checkout (the plugin's peer dependencies are the harness packages) and
846
+ Node with pnpm.
847
+
848
+ ```bash
849
+ # 1. install (links the harness packages into this checkout)
850
+ pnpm install
851
+ pnpm link:dsh
852
+
853
+ # 2. gate the tree
854
+ pnpm typecheck && pnpm test && pnpm build
855
+
856
+ # 3. end-to-end: the plugin under a scripted harness
857
+ pnpm e2e
858
+
859
+ # 4. live session against a real CLI (needs a built harness)
860
+ node scripts/live-session-plugin.mjs
861
+ ```
862
+
863
+ **Mounting it.** The package declares `dsh.bundle.patch`, so a profile can install it as a bundle; the server
864
+ surface arrives when a session opts into the `recursive` preset. See
865
+ [`cordis.patch.yml`](cordis.patch.yml) for the shell row and its reasoning.
866
+
867
+ **First run.**
868
+
869
+ ```
870
+ /recursive bootstrap # scaffold a workspace control plane
871
+ /recursive init <run-id> # create a run (templates + memory skeleton)
872
+ # … the agent writes 00-requirements.md …
873
+ /recursive lint <run-id>
874
+ /recursive lock <run-id> --phase 00
875
+ /recursive status <run-id>
876
+ ```
877
+
878
+ **Inspect what the agent was told:**
879
+
880
+ ```
881
+ /recursive memory <query> --phase 04 # score components per shard
882
+ ```
883
+
884
+ ---
885
+
886
+ ## 14. Verification: what is proven, and how
887
+
888
+ | Claim | How it is established |
889
+ |---|---|
890
+ | The workflow's own rules hold | `pnpm test` — **862 tests, 91 files**, including three **parity specs** (lock, status, lint) against the reference implementation |
891
+ | The tree is sound | `pnpm typecheck` (0 errors), `pnpm build` (0) |
892
+ | The docs match the code | `docs-contract.spec.ts` — 15 tests asserting documented paths, tools, and sections exist |
893
+ | The plugin runs under a real harness | `pnpm e2e` — the FU-1 harness, **10/10**, including the behaviour test where removing one shard changes what the agent is told |
894
+ | A fresh checkout works unattended | `git clone` → `pnpm install` → **862/862**, verified in the closing sweep |
895
+ | The plugin runs in a real CLI session | `scripts/live-session-plugin.mjs` — headless CLI, temp HOME, scripted LLM; the run reaches the review tool and the control plane is written |
896
+ | Memory actually reaches the model | the P5 harness test, end to end |
897
+ | The guard refuses locked writes | enforcement specs plus the live probe that found the overwrite defect |
898
+ | The workflow engine is reachable | a live `workflow` tool call returned `{ parallel: [42, "phase-hooks-ran"], phasesRan: true }` with `0 agents` |
899
+
900
+ **Not proven, and stated as such:**
901
+
902
+ > **The continuable review's repair leg has never completed in a live session.** The child is created, the brief
903
+ > and reply contract are written correctly, and the delegation reaches the *native* tier — but the round never
904
+ > produces a settlement, and the original error is destroyed inside the host before this plugin can observe it.
905
+ > See [DSH-LIMITATIONS.md](DSH-LIMITATIONS.md) entries 9 and 10 for the mechanism and the two one-line fixes
906
+ > that would unblock it. It is recorded as NOT VERIFIED in [STRENGTHENING-PLAN.md](STRENGTHENING-PLAN.md) rather
907
+ > than described as working.
908
+
909
+ **How this project verifies its own claims.** Every gate runs **before** a commit, never after. Two changes in
910
+ this session were reverted rather than patched onward when they turned the suite red: a settlement-observer
911
+ change (its fallout was four specs asserting an immediate park) and a failure-reason field (13 specs pinning the
912
+ action-record shape). Both landed later, correctly, once the specs were brought along. The rule is simple: **an
913
+ item counts as done when a live run demonstrates the behaviour it claims — otherwise it is reported as
914
+ unverified.**
915
+
916
+ ---
917
+
918
+ ## 15. File map and further reading
919
+
920
+ **Source, by responsibility** (`src/`, ~16k lines):
921
+
922
+ | Area | Files |
923
+ |---|---|
924
+ | Composition & runtime | `index.ts`, `runtime.ts`, `config.ts`, `types.ts`, `workspace.ts`, `client.ts` |
925
+ | The standard | `phase-rules.ts`, `ts-lint.ts`, `closeout-standards.ts`, `closeout.ts`, `closeout-report.ts`, `phase-graph.ts` |
926
+ | Locks & receipts | `lock.ts`, `run.ts`, `snapshot.ts` |
927
+ | The guard | `enforcement.ts`, `policy-globs.ts`, `fs-intent.ts`, `guard-log.ts`, `errors.ts` |
928
+ | Delegation & review | `delegation.ts`, `router.ts`, `role-route.ts`, `review.ts`, `review-round.ts`, `settlement.ts`, `handoff.ts`, `teams-loop.ts`, `live-route.ts` |
929
+ | Memory | `memory.ts`, `memory-select.ts`, `memory-feedback.ts` |
930
+ | Surfaces | `recursive_*.tool.ts` (12), `commands.ts`, `hooks.ts`, `status.ts` |
931
+ | Infrastructure | `jobs-runner.ts`, `job-log.ts`, `goals-projection.ts`, `lifecycle.ts`, `bootstrap.ts`, `init-templates.ts`, `worktree.ts`, `git-context.ts`, `scratch.ts`, `skills.ts`, `skills-phase.ts`, `training.ts`, `workflow-audit.ts`, `plan-gate.ts`, `result-cap.ts`, `json-safe.ts`, `identity.ts`, `policy.ts` |
932
+
933
+ **Documents:**
934
+
935
+ | Document | What it is |
936
+ |---|---|
937
+ | [WORKFLOW.md](WORKFLOW.md) | The workflow as the agent sees it — phases, gates, profiles |
938
+ | [STRENGTHENING-PLAN.md](STRENGTHENING-PLAN.md) | The live tracker: every task (T\*) and follow-up (FU-\*) with evidence |
939
+ | [DSH-LIMITATIONS.md](DSH-LIMITATIONS.md) | Ten host limitations found while building this, each with its measurement |
940
+ | [TRAINING.md](TRAINING.md) | The training path: improving this plugin's prompts from its own outcomes |
941
+ | [IMPLEMENTATION-NOTES.md](IMPLEMENTATION-NOTES.md) | Notes from building it |
942
+ | [PROPOSAL.md](PROPOSAL.md), [CHANGELOG.md](CHANGELOG.md) | Origin and history |
943
+ | [ADR-event-log-and-durable-queue.md](ADR-event-log-and-durable-queue.md) | Decision record: the event log and durable queue |
944
+ | [HANDOFF-review-and-opine.md](HANDOFF-review-and-opine.md) | Handoff notes for the review path |
945
+
946
+ **Layout:** `preset/` (the agent preset), `skills/` (agent-facing skill), `references/` (templates for the
947
+ pointer files a workspace gets), `scripts/` (harness, e2e, live-session, installer), `tests/` (91 spec files).
948
+
949
+ ---
950
+
951
+ ## The one-paragraph version
952
+
953
+ **dsh-recursive-mode makes an agent's process auditable by making it a file system.** Twelve phases with required
954
+ sections, monotonic locks with receipts, a guard that refuses writes to locked artifacts, an independent review
955
+ routed to whichever provider is actually available, a memory plane whose scoring is printed on demand, and a
956
+ closeout that lints the whole run and writes a receipt instead of touching the agent's work. It is verified by
957
+ 862 tests, three parity specs against the reference implementation, a live-session harness, and an unattended
958
+ fresh-clone check — and it reports the one capability it has not managed to prove live, with the host-side
959
+ reason, rather than describing it as working.