@try-works/dsh-recursive-mode 0.6.0 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/lib/enforcement.d.ts +6 -1
- package/lib/index.js +597 -198
- package/lib/memory-feedback.d.ts +83 -0
- package/lib/memory.d.ts +15 -2
- package/lib/recursive_phase.tool.d.ts +16 -0
- package/package.json +1 -1
- package/scripts/check-workflow-map-escapes.mjs +103 -0
- package/scripts/check-workflow-map.mjs +423 -0
- package/scripts/gen-workflow-map.mjs +2812 -0
- package/src/enforcement.ts +6 -1
- package/src/memory-feedback.ts +148 -3
- package/src/memory.ts +30 -4
- package/src/policy-globs.ts +111 -0
- package/src/policy.ts +17 -0
- package/src/recursive_phase.tool.ts +24 -1
- package/src/runtime.ts +1975 -1945
package/src/runtime.ts
CHANGED
|
@@ -1,1948 +1,1978 @@
|
|
|
1
|
-
import { Service, type Context } from '@deepseek-ai/cordis'
|
|
2
|
-
import { existsSync, mkdirSync, readFileSync, readdirSync, writeFileSync } from 'node:fs'
|
|
3
|
-
import { dirname, join } from 'node:path'
|
|
4
|
-
import { lintRun } from './ts-lint.ts'
|
|
5
|
-
import { requirementsContent, worktreeContent, laterPhaseContent, detectGitContext, RUN_SCAFFOLD_DIRS, type GitContext } from './init-templates.ts'
|
|
6
|
-
import { foldRun, getMdFieldValue, pendingWork, resolveRunDir } from './status.ts'
|
|
7
|
-
import {
|
|
8
|
-
getLockStatus,
|
|
9
|
-
PHASE_SEQUENCE,
|
|
10
|
-
getNextLegalPhase,
|
|
11
|
-
getPrerequisiteBlockers,
|
|
12
|
-
getStaleDownstreamPhases,
|
|
13
|
-
invalidateReceipt,
|
|
14
|
-
lockHashFromContent,
|
|
15
|
-
validateReceiptChain,
|
|
16
|
-
writeReceipt,
|
|
17
|
-
type ReceiptChainResult,
|
|
18
|
-
} from './lock.ts'
|
|
19
|
-
import type { PendingWorkItem, RecursiveStatusResult } from './types.ts'
|
|
20
|
-
import { findOperation, countOperations, operationId, recordOperation } from './identity.ts'
|
|
21
|
-
import { createHookRegistry, type HookRegistry } from './hooks.ts'
|
|
22
|
-
import { runTracked, abortReason, type JobsRegistryLike } from './jobs-runner.ts'
|
|
23
|
-
import { resolveSubagentTarget } from './role-route.ts'
|
|
24
|
-
import { checkModelChoice, describeInventory, type LlmInventoryLike } from './model-inventory.ts'
|
|
25
|
-
import { recordJobRun } from './job-log.ts'
|
|
26
|
-
import { toolError } from './errors.ts'
|
|
27
|
-
import { readGuardDecisions, type GuardDecisionRecord } from './guard-log.ts'
|
|
28
|
-
import { resolveControlPlaneRoot, type WorkspaceRegistryLike } from './workspace.ts'
|
|
29
|
-
import { phaseRulesFor, type PhaseRules } from './phase-rules.ts'
|
|
30
|
-
import { closeoutReport, writeCloseoutReceipt } from './closeout-report.ts'
|
|
31
|
-
import { readScratch, writeScratch, appendScratch, type ScratchTarget } from './scratch.ts'
|
|
32
|
-
import { buildReviewBundle, type ReviewBundleInput } from './review.ts'
|
|
33
|
-
import { readMemoryEntries, retrieveMemory, renderMemorySection, selectMemory } from './memory.ts'
|
|
34
|
-
import { readFeedback, recordInjection, settleInjections } from './memory-feedback.ts'
|
|
35
|
-
import { runPhase8Trigger, resolveExtractor, spawnExtractorRunner, phase8MemoryLockRefusal } from './training.ts'
|
|
36
|
-
import { buildAskQuestion, GATE_DEFAULT_ARTIFACT, pendingGateFor } from './recursive_ask.tool.ts'
|
|
37
|
-
import { contractDigest } from './policy.ts'
|
|
38
|
-
import type { WorkflowEngineLike } from './workflow-audit.ts'
|
|
39
|
-
import { createHandoff, createChildBrief, replyPath, childScratchPath, buildDelegationPrompt, type HandoffInput, type ChildBriefInput } from './handoff.ts'
|
|
40
|
-
import { loadRouterPolicy, routerPolicyPath, resolveRole, capabilityProbe, delegationDecisionBasis, type RouterPolicy, type RouterPolicyOverrides, type SubagentProviderLike, type RouteDecision, type CapabilityProbe } from './router.ts'
|
|
41
|
-
import { delegate, delegateContinuable, drainContinuableChildren, remainingDepthFor, validateReferences, referencesFromResult, writeActionRecord, evaluateDelegationResult, reviewOutputSchema, defaultReviewToolFilter, type SubagentsRuntimeLike, type SubagentStartRequestLike, type SubagentResultLike, type Reference, type ActionRecordInput, type ContinuableDelegationLike, type SubagentParentHandle } from './delegation.ts'
|
|
42
|
-
import { validateTransition, coupleGateBlockToGoal, type PhaseTransitionIntent, type RecursivePhaseState, type GateCheckResult } from './lifecycle.ts'
|
|
43
|
-
import { resolveEnforcementConfig, DEFAULT_ENFORCEMENT, evaluateToolGuard, detectTamper, type EnforcementConfig, type ToolGuardDecision, type ToolExecLike } from './enforcement.ts'
|
|
44
|
-
import type { Session } from '@deepseek-ai/dsh-session'
|
|
45
|
-
import { renderRecursivePolicy, type PolicyContext } from './policy.ts'
|
|
46
|
-
import { snapshotWorkspace } from './snapshot.ts'
|
|
47
|
-
import { createLinkedWorktree, promoteBranch, listWorktrees, defaultWorktreeBranch, type CreateWorktreeResult, type PromoteBranchResult } from './worktree.ts'
|
|
48
|
-
import { changedPaths, gitFacts } from './git-context.ts'
|
|
49
|
-
import { syncRunGoal, blockRunGoal, resumeRunGoal, type GoalServiceLike, type SyncResult } from './goals-projection.ts'
|
|
50
|
-
import {
|
|
51
|
-
RUN_START_APPROVE,
|
|
52
|
-
RUN_START_ARTIFACT,
|
|
53
|
-
RUN_START_GATE_ID,
|
|
54
|
-
RUN_START_MARKER,
|
|
55
|
-
RUN_START_NOT_APPROVED,
|
|
56
|
-
readRunStartApproval,
|
|
57
|
-
runStartArtifactPath,
|
|
58
|
-
} from './run-start.ts'
|
|
59
|
-
import { auditToPass, renderTaskHistory, type TeamRuntimeLike, type AuditToPassResult, type TeamCallerHandle, type TeamTaskViewLike, type AuditRoundOutcome } from './teams-loop.ts'
|
|
60
|
-
import type { ContinuableChildId, ContinuableMessageId } from './delegation.ts'
|
|
61
|
-
import { runChildIds } from './settlement.ts'
|
|
62
|
-
|
|
63
|
-
declare module '@deepseek-ai/cordis' {
|
|
64
|
-
interface Context {
|
|
65
|
-
recursive: RecursiveRuntime
|
|
66
|
-
}
|
|
67
|
-
}
|
|
68
|
-
|
|
69
|
-
export interface LockArtifactResult {
|
|
70
|
-
artifact: string
|
|
71
|
-
runId: string
|
|
72
|
-
status: string
|
|
73
|
-
lockedAt: string | null
|
|
74
|
-
lockHash: string | null
|
|
75
|
-
receipt?: unknown
|
|
76
|
-
blockers: string[]
|
|
77
|
-
}
|
|
78
|
-
|
|
79
|
-
export interface LintArtifactResult {
|
|
80
|
-
artifact: string
|
|
81
|
-
runId: string
|
|
82
|
-
errors: string[]
|
|
83
|
-
warnings: string[]
|
|
84
|
-
passed: boolean
|
|
85
|
-
}
|
|
86
|
-
|
|
87
|
-
/**
|
|
88
|
-
* PHASE 0 — the structural seam for the host's human-question channel (`ctx.userQuestions`).
|
|
89
|
-
*
|
|
90
|
-
* Declared here as a minimal seam for the same reason as every other harness touchpoint in this plugin:
|
|
91
|
-
* the live `UserQuestionService` satisfies it structurally, so the plugin never imports the host package,
|
|
92
|
-
* and a test can drive the run-start gate with a fake that behaves like the real one. The error case is
|
|
93
|
-
* part of the contract, not an afterthought: the real `ask()` REJECTS (NO_PROVIDER / CALLER_NOT_LIVE /
|
|
94
|
-
* ASK_ABORTED) instead of resolving with something that could be mistaken for an answer, which is what
|
|
95
|
-
* lets `recursive_ask` fail closed rather than invent an approval.
|
|
96
|
-
*/
|
|
97
|
-
export interface UserQuestionsLike {
|
|
98
|
-
ask(request: {
|
|
99
|
-
questions: Array<{ id: string; header?: string; question: string; options?: Array<{ label: string; description?: string }> }>
|
|
100
|
-
agent?: unknown
|
|
101
|
-
signal?: AbortSignal
|
|
102
|
-
/** Links the card to the tool call that asked, the way plan-mode's exit does. */
|
|
103
|
-
wait?: { callId?: unknown }
|
|
104
|
-
}): Promise<{ answers: Array<{ id: string; selected: string[]; custom?: string }> }>
|
|
105
|
-
}
|
|
106
|
-
|
|
107
|
-
/**
|
|
108
|
-
* T15 (G): the folded status PLUS the rolling guard-decision evidence. Declared
|
|
109
|
-
* as an intersection rather than by editing RecursiveStatusResult/foldRun — the
|
|
110
|
-
* fold's own shape is parity-asserted and stays exactly as it was.
|
|
111
|
-
*/
|
|
112
|
-
export type RecursiveStatusWithGuardDecisions = RecursiveStatusResult & {
|
|
113
|
-
guardDecisions?: GuardDecisionRecord[]
|
|
114
|
-
/**
|
|
115
|
-
* T18: unresolved in-flight work, derived from the run directory on every call.
|
|
116
|
-
* Always present (empty when nothing is in flight) so consumers need no null
|
|
117
|
-
* check, and non-empty explains a `RM4403` lock refusal.
|
|
118
|
-
*/
|
|
119
|
-
pendingWork?: PendingWorkItem[]
|
|
120
|
-
/**
|
|
121
|
-
* T32: the receipt-chain verdict, derived read-only on every call. Always present
|
|
122
|
-
* so a caller can read `ok` without a null check; non-empty `breaks` names the
|
|
123
|
-
* first broken link. This is what makes a spliced or edited chain VISIBLE rather
|
|
124
|
-
* than merely detectable in a test.
|
|
125
|
-
*/
|
|
126
|
-
receiptChain?: ReceiptChainResult
|
|
127
|
-
/**
|
|
128
|
-
* T22: the local identifier of the policy section's STABLE prefix.
|
|
129
|
-
*
|
|
130
|
-
* Surfaced so "did the contract change under me?" is answerable from the status alone — the same
|
|
131
|
-
* digest the prompt carries, so a reader can compare them without re-rendering anything. It is an
|
|
132
|
-
* IDENTIFIER, not a cache directive: whether any provider caches the prefix is provider-side and
|
|
133
|
-
* unverified, which is why the item's "largest cost lever" label was withdrawn.
|
|
134
|
-
*/
|
|
135
|
-
contractDigest?: string
|
|
136
|
-
}
|
|
137
|
-
|
|
138
|
-
/** How many recent decisions to read from the log before scoping to one run. */
|
|
139
|
-
const GUARD_DECISION_READ_LIMIT = 200
|
|
140
|
-
|
|
141
|
-
/**
|
|
142
|
-
* T28: the delegating agent's own delegation depth, read structurally rather than by
|
|
143
|
-
* importing the harness's `delegationDepthOf`. The plugin already models every
|
|
144
|
-
* harness touchpoint as a minimal structural seam, and this keeps that convention —
|
|
145
|
-
* an absent depth means top level, which is the safe reading for a budget.
|
|
146
|
-
*/
|
|
147
|
-
function parentDepthOf(parent: unknown): number {
|
|
148
|
-
const depth = (parent as { options?: { subagentDepth?: unknown } } | null | undefined)?.options?.subagentDepth
|
|
149
|
-
return typeof depth === 'number' && Number.isSafeInteger(depth) && depth > 0 ? depth : 0
|
|
150
|
-
}
|
|
151
|
-
|
|
152
|
-
/** How many of that run's decisions the status surface carries. */
|
|
153
|
-
const GUARD_DECISION_SURFACE_LIMIT = 20
|
|
154
|
-
|
|
155
|
-
const ARTIFACT_STUB = {
|
|
156
|
-
'00-requirements.md': ['Run: ', 'Phase: 0', 'Status: DRAFT', 'Workflow version: recursive-mode-audit-v2', 'Inputs: none', 'Outputs: none', 'Scope note: '],
|
|
157
|
-
'00-worktree.md': ['Run: ', 'Phase: 0 (Worktree)', 'Status: DRAFT', 'Workflow version: recursive-mode-audit-v2', 'Inputs: none', 'Outputs: none', 'Scope note: '],
|
|
158
|
-
'01-as-is.md': ['Run: ', 'Phase: 1 (AS-IS)', 'Status: DRAFT', 'Workflow version: recursive-mode-audit-v2', 'Inputs: none', 'Outputs: none', 'Scope note: '],
|
|
159
|
-
'02-to-be-plan.md': ['Run: ', 'Phase: 2 (TO-BE Plan)', 'Status: DRAFT', 'Workflow version: recursive-mode-audit-v2', 'Inputs: none', 'Outputs: none', 'Scope note: '],
|
|
160
|
-
'03-implementation-summary.md': ['Run: ', 'Phase: 3 (Implementation)', 'Status: DRAFT', 'Workflow version: recursive-mode-audit-v2', 'Inputs: none', 'Outputs: none', 'Scope note: '],
|
|
161
|
-
'04-test-summary.md': ['Run: ', 'Phase: 4 (Test Summary)', 'Status: DRAFT', 'Workflow version: recursive-mode-audit-v2', 'Inputs: none', 'Outputs: none', 'Scope note: '],
|
|
162
|
-
'05-manual-qa.md': ['Run: ', 'Phase: 5 (Manual QA)', 'Status: DRAFT', 'Workflow version: recursive-mode-audit-v2', 'Inputs: none', 'Outputs: none', 'Scope note: '],
|
|
163
|
-
'06-decisions-update.md': ['Run: ', 'Phase: 6 (Decisions Update)', 'Status: DRAFT', 'Workflow version: recursive-mode-audit-v2', 'Inputs: none', 'Outputs: none', 'Scope note: '],
|
|
164
|
-
'07-state-update.md': ['Run: ', 'Phase: 7 (State Update)', 'Status: DRAFT', 'Workflow version: recursive-mode-audit-v2', 'Inputs: none', 'Outputs: none', 'Scope note: '],
|
|
165
|
-
'08-memory-impact.md': ['Run: ', 'Phase: 8 (Memory Impact)', 'Status: DRAFT', 'Workflow version: recursive-mode-audit-v2', 'Inputs: none', 'Outputs: none', 'Scope note: '],
|
|
166
|
-
}
|
|
167
|
-
|
|
168
|
-
export class RecursiveRuntime extends Service {
|
|
169
|
-
/** Recursive-mode runtime service. Owns run-state reads + lock/init/lint operations. */
|
|
170
|
-
|
|
171
|
-
constructor(ctx: Context, config: { repoRoot?: string; workspaceRegistry?: WorkspaceRegistryLike; goals?: GoalServiceLike | null; jobs?: JobsRegistryLike | null; subagents?: SubagentsRuntimeLike | null; workflow?: WorkflowEngineLike | null } = {}) {
|
|
172
|
-
super(ctx, 'recursive')
|
|
173
|
-
this.repoRoot = config.repoRoot ?? process.cwd()
|
|
174
|
-
this.workspaceRegistry = config.workspaceRegistry ?? null
|
|
175
|
-
this.goalsService = config.goals ?? null
|
|
176
|
-
// T10: the native jobs registry is OPTIONAL. Absent, long operations run inline and say
|
|
177
|
-
// so; see `runTracked` for why that is better than refusing to work without a board.
|
|
178
|
-
this.jobs = config.jobs ?? null
|
|
179
|
-
// T39: the subagents seam, resolved at the COMPOSITION like the other optional services.
|
|
180
|
-
this.subagentsSeam = config.subagents ?? null
|
|
181
|
-
// T2: the workflow engine, likewise.
|
|
182
|
-
this.workflow = config.workflow ?? null
|
|
183
|
-
}
|
|
184
|
-
|
|
185
|
-
/**
|
|
186
|
-
* T10: the native jobs registry, when the composition mounts one. */
|
|
187
|
-
private readonly jobs: JobsRegistryLike | null
|
|
188
|
-
|
|
189
|
-
/**
|
|
190
|
-
* PHASE 0 — attach the goals service after construction.
|
|
191
|
-
*
|
|
192
|
-
* The composition resolves `goals` with ONE `ctx.get` at apply time and passes it to the constructor,
|
|
193
|
-
* which is fine for a service that is already mounted. This seam exists for the two cases that pattern
|
|
194
|
-
* cannot cover: a composition that mounts `goals` later (the same late-attach reason `attachSubagents`
|
|
195
|
-
* and `attachLlmInventory` exist), and a test that needs the REAL runtime wired to a structural fake —
|
|
196
|
-
* a fake passed through the plugin's Config is dropped, because the Config schema is the settings
|
|
197
|
-
* namespace and strips keys it does not declare.
|
|
198
|
-
*/
|
|
199
|
-
attachGoals(service: GoalServiceLike | null): void {
|
|
200
|
-
this.goalsService = service
|
|
201
|
-
}
|
|
202
|
-
|
|
203
|
-
/**
|
|
204
|
-
* T23 — write a gate's answer into an artifact as a marker line.
|
|
205
|
-
*
|
|
206
|
-
* ⚠ REPLACED IN PLACE when the artifact already carries that gate's marker: two `TDD Mode:` lines
|
|
207
|
-
* would leave two answers to one question and make "what was decided?" depend on which a reader
|
|
208
|
-
* found first. The write is confined to the run directory, and an artifact that does not exist is
|
|
209
|
-
* CREATED — a decision recorded nowhere is not recorded.
|
|
210
|
-
*/
|
|
211
|
-
recordAskAnswer(root: string, runId: string, artifact: string, marker: string): { path: string; replaced: boolean } {
|
|
212
|
-
const dir = join(root, '.recursive', 'run', runId)
|
|
213
|
-
const path = join(dir, artifact)
|
|
214
|
-
const label = marker.slice(2).split(':')[0].trim()
|
|
215
|
-
let content = ''
|
|
216
|
-
try {
|
|
217
|
-
content = readFileSync(path, 'utf8')
|
|
218
|
-
} catch {
|
|
219
|
-
// A missing artifact is created below, so the decision still has somewhere to live.
|
|
220
|
-
}
|
|
221
|
-
const lines = content === '' ? [] : content.replace(/\n$/, '').split('\n')
|
|
222
|
-
const at = lines.findIndex((line) => line.startsWith('- ' + label + ':'))
|
|
223
|
-
const replaced = at >= 0
|
|
224
|
-
if (replaced) lines[at] = marker
|
|
225
|
-
else lines.push(marker)
|
|
226
|
-
mkdirSync(dir, { recursive: true })
|
|
227
|
-
writeFileSync(path, lines.join('\n') + '\n', 'utf8')
|
|
228
|
-
return { path, replaced }
|
|
229
|
-
}
|
|
230
|
-
|
|
231
|
-
/**
|
|
232
|
-
* T2: the workflow engine, when the composition mounts one.
|
|
233
|
-
*
|
|
234
|
-
* OPTIONAL like every other seam here — without it an audit fan-out reports that it could not be
|
|
235
|
-
* orchestrated rather than pretending a fan-out happened. The engine's `workflow/*` events are
|
|
236
|
-
* observe-only, so this is used to START a run and await its result, never to drive one.
|
|
237
|
-
*/
|
|
238
|
-
private readonly workflow: WorkflowEngineLike | null
|
|
239
|
-
|
|
240
|
-
/**
|
|
241
|
-
* T39: the subagents seam the composition mounted, used when a caller does not pass one.
|
|
242
|
-
*
|
|
243
|
-
* ⚠ WHY THIS EXISTS, measured rather than assumed: `recursive_review.tool.ts` — the ONLY
|
|
244
|
-
* production caller of `delegateReview` — passes **no `subagents`** at its call site, and
|
|
245
|
-
* `delegateReview` reports *"no ctx.subagents runtime available (self-audit fallback)"* when
|
|
246
|
-
* the input lacks one. So on a composition that HAS the service, the review tool's rounds
|
|
247
|
-
* never reached a child at all, and the fallback message blamed a missing runtime that was
|
|
248
|
-
* in fact mounted. Resolving the seam here fixes the wiring without asking every call site to
|
|
249
|
-
* remember, while an explicit `input.subagents` still wins for a test or a narrower caller.
|
|
250
|
-
*/
|
|
251
|
-
private subagentsSeam: SubagentsRuntimeLike | null
|
|
252
|
-
/** FU-19: the host's provider/model inventory, or null when no llm service is mounted. */
|
|
253
|
-
private llmInventory: LlmInventoryLike | null = null
|
|
254
|
-
|
|
255
|
-
/**
|
|
256
|
-
* ⚠ FU-9 — ATTACH THE SEAM WHEN THE SERVICE APPEARS, not only when this plugin happens to apply.
|
|
257
|
-
*
|
|
258
|
-
* The composition resolved the seam with a ONE-SHOT `ctx.get('subagents')` at apply time, and a live run
|
|
259
|
-
* showed what that costs: the review fell back to self-audit, the action record said
|
|
260
|
-
* `Execution Mode: self-audit (continuable)` and `Status: failed`, and **no child was ever started** — while
|
|
261
|
-
* the child DIRECTORY existed all along, because the plugin writes its own brief before calling any service.
|
|
262
|
-
* I read the directory and built a host-limitation story on top of it; the record said otherwise.
|
|
263
|
-
*
|
|
264
|
-
* If the subagents service is mounted by a later loader layer, a one-shot get returns undefined and nothing
|
|
265
|
-
* re-resolves it. `ctx.inject(['subagents'], …)` is the harness's own pattern for exactly this, and calling
|
|
266
|
-
* this method from there makes the seam arrive whenever it arrives. Idempotent: the last attach wins, which
|
|
267
|
-
* is what a re-apply after a reload wants.
|
|
268
|
-
*/
|
|
269
|
-
attachSubagents(seam: SubagentsRuntimeLike | null): void {
|
|
270
|
-
this.subagentsSeam = seam
|
|
271
|
-
}
|
|
272
|
-
|
|
273
|
-
/**
|
|
274
|
-
* ⚠ FU-19 — THE LLM INVENTORY SEAM, resolved the same late-attaching way the subagents seam is and for the same
|
|
275
|
-
* measured reason: a one-shot `ctx.get` at apply time misses a service mounted by a later layer.
|
|
276
|
-
*
|
|
277
|
-
* Null is a legitimate value and it is NOT treated as "no models exist" — `describeInventory(null)` reports a
|
|
278
|
-
* named unavailability, and `checkModelChoice` turns that into the `unverified` verdict. A missing inventory
|
|
279
|
-
* therefore never silently approves a model and never silently replaces one.
|
|
280
|
-
*/
|
|
281
|
-
attachLlmInventory(seam: LlmInventoryLike | null): void {
|
|
282
|
-
this.llmInventory = seam
|
|
283
|
-
}
|
|
284
|
-
|
|
285
|
-
/** What the composition attached, for a caller that needs to report or assert it. */
|
|
286
|
-
attachedSubagents(): SubagentsRuntimeLike | null {
|
|
287
|
-
return this.subagentsSeam
|
|
288
|
-
}
|
|
289
|
-
|
|
290
|
-
/**
|
|
291
|
-
* ⚠ FU-9 — THE ROUTER'S PROVIDER MAP, BUILT FROM THE SEAM THAT IS ALREADY ATTACHED.
|
|
292
|
-
*
|
|
293
|
-
* `router.ts` returns the NATIVE tier for the first of `[role, 'spawn', 'fork', 'dsh-sdk']` present in this
|
|
294
|
-
* map. Handed `{}` it tried the external CLI route (null in the default policy) and fell to the policy
|
|
295
|
-
* fallback — self-audit — with a message naming neither. The router already preferred native; nobody ever
|
|
296
|
-
* gave it a name. A live review self-audited for five rounds because of it.
|
|
297
|
-
*
|
|
298
|
-
* `SubagentProviderLike` is only a DESCRIPTOR (`{ name, capabilities? }`), so a provider is registered by
|
|
299
|
-
* ASKING the service for it rather than by wrapping it. A service that cannot enumerate yields an empty map
|
|
300
|
-
* and the previous behaviour, which is the correct degradation rather than a guess about the shape.
|
|
301
|
-
*/
|
|
302
|
-
private providerMapFromSeam(): Record<string, SubagentProviderLike> {
|
|
303
|
-
const seam = this.subagentsSeam
|
|
304
|
-
this.lastProviderNames = []
|
|
305
|
-
if (seam === null) return {}
|
|
306
|
-
const map: Record<string, SubagentProviderLike> = {}
|
|
307
|
-
// ⚠ ASK THE SERVICE WHAT IT HAS, FIRST. The previous version only probed three invented names and
|
|
308
|
-
// registered whatever came back — and the live record then said `provider spawn`, so the router faithfully
|
|
309
|
-
// returned a name the host does not serve and `start('spawn', …)` produced NOTHING. A name this plugin made
|
|
310
|
-
// up is not a provider the host knows, and the difference is a child that runs versus a silent no-op.
|
|
311
|
-
try {
|
|
312
|
-
const listed = seam.list?.()
|
|
313
|
-
if (Array.isArray(listed)) {
|
|
314
|
-
for (const entry of listed) {
|
|
315
|
-
if (typeof entry === 'string' && entry !== '') map[entry] = { name: entry }
|
|
316
|
-
else if (entry !== null && typeof entry === 'object' && typeof (entry as { name?: unknown }).name === 'string') {
|
|
317
|
-
const named = entry as { name: string } & SubagentProviderLike
|
|
318
|
-
map[named.name] = named
|
|
319
|
-
}
|
|
320
|
-
}
|
|
321
|
-
}
|
|
322
|
-
} catch {
|
|
323
|
-
// A service that cannot enumerate is not an error: the probes below are the fallback, not the plan.
|
|
324
|
-
}
|
|
325
|
-
// The probes stay as a FALLBACK for a service that exposes getProvider but no list. Whatever they return is
|
|
326
|
-
// registered under the name ASKED FOR, which is only sound because the name came from a real lookup.
|
|
327
|
-
for (const name of ['spawn', 'fork', 'dsh-sdk']) {
|
|
328
|
-
if (map[name] !== undefined) continue
|
|
329
|
-
try {
|
|
330
|
-
const found = seam.getProvider?.(name)
|
|
331
|
-
if (found !== undefined && found !== null) map[name] = found as SubagentProviderLike
|
|
332
|
-
} catch {
|
|
333
|
-
// An unavailable name is not an error: the next candidate still gets its turn.
|
|
334
|
-
}
|
|
335
|
-
}
|
|
336
|
-
this.lastProviderNames = Object.keys(map)
|
|
337
|
-
return map
|
|
338
|
-
}
|
|
339
|
-
|
|
340
|
-
/** The provider names the last `providerMapFromSeam` call registered, for the record and for assertions. */
|
|
341
|
-
private lastProviderNames: string[] = []
|
|
342
|
-
|
|
343
|
-
/** What the router could choose from, so a failure can say whether the name it used was ever on offer. */
|
|
344
|
-
knownProviderNames(): string[] {
|
|
345
|
-
return this.lastProviderNames
|
|
346
|
-
}
|
|
347
|
-
|
|
348
|
-
private readonly repoRoot: string
|
|
349
|
-
private readonly workspaceRegistry: WorkspaceRegistryLike | null
|
|
350
|
-
private goalsService: GoalServiceLike | null
|
|
351
|
-
/**
|
|
352
|
-
* T27 — the hook registry, EXPOSED so a sibling plugin can participate in a run
|
|
353
|
-
* without patching this one:
|
|
354
|
-
*
|
|
355
|
-
* ctx.recursive.hooks.register('pre_trigger', { name: 'my-check', priority: 10, run })
|
|
356
|
-
*
|
|
357
|
-
* That is the whole point of the item: the plugin's own enforcement will be
|
|
358
|
-
* re-expressed as built-in hooks on this same registry, so a sibling and a built-in
|
|
359
|
-
* are peers — same ordering rules, same failure policy, same audit trail — rather
|
|
360
|
-
* than one being privileged code and the other a guest.
|
|
361
|
-
*
|
|
362
|
-
* Public and created eagerly: a registry that has to be "got" before it can be used
|
|
363
|
-
* is a registry whose ordering depends on when someone remembered to fetch it.
|
|
364
|
-
*/
|
|
365
|
-
readonly hooks: HookRegistry = createHookRegistry()
|
|
366
|
-
|
|
367
|
-
/**
|
|
368
|
-
* T7 — router overrides from the settings namespace. Kept beside the config rather than
|
|
369
|
-
* merged into the file so the workspace's `recursive-router.json` stays the declarative
|
|
370
|
-
* source: `loadRouterPolicy` reads the file and lays these on top.
|
|
371
|
-
*/
|
|
372
|
-
private _routerOverrides: RouterPolicyOverrides | undefined = undefined
|
|
373
|
-
|
|
374
|
-
/** T7: set (or clear) the router overrides. Called from `apply` on every plugin load. */
|
|
375
|
-
setRouterOverrides(overrides: RouterPolicyOverrides | undefined): void {
|
|
376
|
-
this._routerOverrides = overrides
|
|
377
|
-
}
|
|
378
|
-
|
|
379
|
-
private _enforcementConfig: EnforcementConfig | null = null
|
|
380
|
-
|
|
381
|
-
/**
|
|
382
|
-
* T3 (agentTeams task loop): run the audit→repair→re-audit state machine on
|
|
383
|
-
* ONE durable team task. The `teams` seam (live `ctx.agentTeams`) is injected
|
|
384
|
-
* per-call so the loop stays unit-testable; `runAuditRound` is the caller's
|
|
385
|
-
* round executor (live usage wires T4's continuable delegation). Locking the
|
|
386
|
-
* phase artifact is `lockPhase` — the loop NEVER locks before an APPROVE.
|
|
387
|
-
*/
|
|
388
|
-
async auditToPass(input: {
|
|
389
|
-
teams: TeamRuntimeLike
|
|
390
|
-
caller: TeamCallerHandle
|
|
391
|
-
root: string
|
|
392
|
-
runId: string
|
|
393
|
-
phase: string
|
|
394
|
-
artifact: string
|
|
395
|
-
agent?: { session?: { header?: { cwd?: string } } } | null
|
|
396
|
-
runAuditRound: (round: number, task: TeamTaskViewLike) => Promise<AuditRoundOutcome>
|
|
397
|
-
blockedBy?: readonly string[]
|
|
398
|
-
writeScopes?: readonly string[]
|
|
399
|
-
reviewerName?: string
|
|
400
|
-
maxRounds?: number
|
|
401
|
-
waitTimeoutMs?: number
|
|
402
|
-
}): Promise<AuditToPassResult & { history?: string; lock?: LockArtifactResult }> {
|
|
403
|
-
const { teams, caller, runId, phase, artifact, runAuditRound, agent } = input
|
|
404
|
-
let lockResult: LockArtifactResult | undefined
|
|
405
|
-
const lockPhase = async () => { lockResult = await this.lockArtifact(runId, artifact, false, agent) }
|
|
406
|
-
const result = await auditToPass({
|
|
407
|
-
teams,
|
|
408
|
-
caller,
|
|
409
|
-
runId,
|
|
410
|
-
phase,
|
|
411
|
-
blockedBy: input.blockedBy,
|
|
412
|
-
writeScopes: input.writeScopes,
|
|
413
|
-
reviewerName: input.reviewerName,
|
|
414
|
-
maxRounds: input.maxRounds,
|
|
415
|
-
waitTimeoutMs: input.waitTimeoutMs,
|
|
416
|
-
runAuditRound,
|
|
417
|
-
lockPhase,
|
|
418
|
-
})
|
|
419
|
-
const report: AuditToPassResult & { history?: string; lock?: LockArtifactResult } = {
|
|
420
|
-
...result,
|
|
421
|
-
history: renderTaskHistory(result.taskView, result.rounds),
|
|
422
|
-
}
|
|
423
|
-
if (lockResult !== undefined) report.lock = lockResult
|
|
424
|
-
return report
|
|
425
|
-
}
|
|
426
|
-
|
|
427
|
-
/**
|
|
428
|
-
* PHASE 0 — read a run's start approval from its own Phase 0 artifact.
|
|
429
|
-
*
|
|
430
|
-
* The approval is a DURABLE line in `.recursive/run/<runId>/00-requirements.md`, not a value held in
|
|
431
|
-
* memory, for the reason every other gate here is durable: a decision that only exists in a session
|
|
432
|
-
* cannot be cited, and cannot survive the session it was made in. Read-only; asking changes nothing.
|
|
433
|
-
*/
|
|
434
|
-
readRunStartApproval(root: string, runId: string): { approved: boolean; artifact: string; reason: string } {
|
|
435
|
-
return readRunStartApproval(root, runId)
|
|
436
|
-
}
|
|
437
|
-
|
|
438
|
-
/**
|
|
439
|
-
* PHASE 0 — the harness's blocking human-question channel (`ctx.userQuestions`), when this composition
|
|
440
|
-
* mounts one.
|
|
441
|
-
*
|
|
442
|
-
* ⚠ WHY THE PLUGIN REACHES FOR THIS AT ALL. The other three gates answer through a question card and a
|
|
443
|
-
* relayed label, which is fine for a decision the workflow acts on later. STARTING A RUN is different:
|
|
444
|
-
* the first human turn is the only place the harness can say "arming this goal means autonomous rounds"
|
|
445
|
-
* BEFORE arming it. This channel is the same one plan-mode's exit uses; `ask()` resolves only with a
|
|
446
|
-
* real answer from a real person, and it THROWS when there is no answerer or no live root agent. So the
|
|
447
|
-
* absence of this service cannot be papered over: `recursive_ask` refuses the run-start gate and names
|
|
448
|
-
* the missing channel (RM5502).
|
|
449
|
-
*/
|
|
450
|
-
private userQuestions: UserQuestionsLike | null = null
|
|
451
|
-
|
|
452
|
-
/** Late-bind the human-question channel when the composition mounts it. */
|
|
453
|
-
attachUserQuestions(service: UserQuestionsLike | null): void {
|
|
454
|
-
this.userQuestions = service
|
|
455
|
-
}
|
|
456
|
-
|
|
457
|
-
/** The human-question channel this composition mounted, or null. */
|
|
458
|
-
get userQuestionsChannel(): UserQuestionsLike | null {
|
|
459
|
-
return this.userQuestions
|
|
460
|
-
}
|
|
461
|
-
|
|
462
|
-
/**
|
|
463
|
-
* T1 (goals projection): project the run into the native goals service so it is
|
|
464
|
-
* a first-class durable, resumable, blockable object. Best-effort — the run's
|
|
465
|
-
* filesystem state is the source of truth; a goal is the durable projection.
|
|
466
|
-
*
|
|
467
|
-
* ⚠ PHASE 0 — AND IT ARMS NOTHING UNLESS THE RUN WAS STARTED. `create` returns an ARMED goal, and an
|
|
468
|
-
* armed goal is the harness driving autonomous rounds, so this is the one place where "project the
|
|
469
|
-
* state" can quietly equal "start the run". The approval is therefore REQUIRED from the caller and
|
|
470
|
-
* has no default here: a caller that has not resolved the run's approval cannot arm a goal by
|
|
471
|
-
* forgetting to pass one, and `syncRunGoal` refuses every branch that would create one without it.
|
|
472
|
-
*
|
|
473
|
-
* Unapproved is the EXPECTED state for a scaffolded run, so the refusal comes back as a plain
|
|
474
|
-
* `{ ok: false }` carrying {@link RUN_START_NOT_APPROVED}: callers must treat that as normal work,
|
|
475
|
-
* never as a warning (see `armRunGoalIfApproved`, the one caller that arms).
|
|
476
|
-
*/
|
|
477
|
-
projectRunToGoal(agent: { session?: { header?: { cwd?: string } } } | null | undefined, runId: string, state: Parameters<typeof syncRunGoal>[3] = 'active', approved = false): SyncResult {
|
|
478
|
-
if (!agent) return { ok: false, reason: 'no agent' }
|
|
479
|
-
return syncRunGoal(this.goalsService, agent, runId, state, approved)
|
|
480
|
-
}
|
|
481
|
-
|
|
482
|
-
/**
|
|
483
|
-
* PHASE 0 — the approved-run arm step: read the run's approval from `root` and project the goal only
|
|
484
|
-
* if it is there. This is the phase-progress path (`syncRunGoal` reached on ordinary work), so the
|
|
485
|
-
* unapproved case is deliberately silent: `{ ok: false, reason: RUN_START_NOT_APPROVED }` with no
|
|
486
|
-
* goal, no write and no throw. Read on EVERY call rather than cached, because the approval can arrive
|
|
487
|
-
* mid-session and a cached "not yet" would leave an approved run unable to arm until a plugin reload.
|
|
488
|
-
*/
|
|
489
|
-
armRunGoalIfApproved(agent: { session?: { header?: { cwd?: string } } } | null | undefined, root: string, runId: string, state: Parameters<typeof syncRunGoal>[3] = 'active'): SyncResult {
|
|
490
|
-
if (!agent) return { ok: false, reason: 'no agent' }
|
|
491
|
-
const approval = this.readRunStartApproval(root, runId)
|
|
492
|
-
return syncRunGoal(this.goalsService, agent, runId, state, approval.approved)
|
|
493
|
-
}
|
|
494
|
-
|
|
495
|
-
/** T1: block the run's goal on a gate-block (durable + UI-visible). */
|
|
496
|
-
blockRunToGoal(agent: { session?: { header?: { cwd?: string } } } | null | undefined, runId: string, reason: { code: string; message: string }): SyncResult {
|
|
497
|
-
if (!agent) return { ok: false, reason: 'no agent' }
|
|
498
|
-
return blockRunGoal(this.goalsService, agent, runId, reason)
|
|
499
|
-
}
|
|
500
|
-
|
|
501
|
-
/**
|
|
502
|
-
* T1: re-arm the run's goal on a reopen (blocked/paused -> active). Never starts an unstarted run —
|
|
503
|
-
* REOPEN IS NOT A BACK DOOR TO STARTING A RUN. `approved` is required for the same reason as in
|
|
504
|
-
* `projectRunToGoal`: the phase-0 gate cannot be defaulted open. An approved run's approval outlives
|
|
505
|
-
* a reopen because it is a durable line in the run's own Phase 0 artifact, not a held value.
|
|
506
|
-
*/
|
|
507
|
-
resumeRunToGoal(agent: { session?: { header?: { cwd?: string } } } | null | undefined, runId: string, approved = false): SyncResult {
|
|
508
|
-
if (!agent) return { ok: false, reason: 'no agent' }
|
|
509
|
-
return resumeRunGoal(this.goalsService, agent, runId, approved)
|
|
510
|
-
}
|
|
511
|
-
|
|
512
|
-
/**
|
|
513
|
-
* Workspace-scoped control-plane root (R1 binding invariant).
|
|
514
|
-
* Resolves the session agent's canonical cwd -> workspace path via the
|
|
515
|
-
* registry; NEVER scans list(). Returns null when unavailable (defer).
|
|
516
|
-
*/
|
|
517
|
-
async resolveWorkspaceRoot(agent?: { session?: { header?: { cwd?: string } } } | null): Promise<string | null> {
|
|
518
|
-
return resolveControlPlaneRoot(agent, this.workspaceRegistry)
|
|
519
|
-
}
|
|
520
|
-
|
|
521
|
-
/**
|
|
522
|
-
* Run-scoped closeout receipt scaffold (R2), rooted under the given
|
|
523
|
-
* workspace root. Refuses runIds outside the root (never crosses workspaces).
|
|
524
|
-
*/
|
|
525
|
-
async closeoutRun(root: string, runId: string, phase: string, agent?: { session?: { header?: { cwd?: string } } } | null) {
|
|
526
|
-
const runDir = join(root, '.recursive', 'run', runId)
|
|
527
|
-
const runRoot = join(root, '.recursive', 'run')
|
|
528
|
-
// workspace-scoping guard: the run must be under this root
|
|
529
|
-
if (!runDir.startsWith(runRoot) || !existsSync(runDir)) {
|
|
530
|
-
return { error: 'Run not found in current workspace: ' + runId }
|
|
531
|
-
}
|
|
532
|
-
// Declared OUTSIDE the try so the failure path can report it too: a refused closeout still tells the
|
|
533
|
-
// caller whether its children were released.
|
|
534
|
-
let drain: { children: number; drained: boolean; reason?: string } | null = null
|
|
535
|
-
try {
|
|
536
|
-
// T30 — THE RUN-CLOSE TRIGGER, at the RE-RUN of closeout phase 08 and nowhere else.
|
|
537
|
-
//
|
|
538
|
-
// ⚠ DETECTED FROM THE RECEIPT THAT ALREADY EXISTS, not from a counter kept beside it: if phase 08
|
|
539
|
-
// has a receipt BEFORE this closeout runs, then it has been closed out before and this is the
|
|
540
|
-
// re-run the parent asks for. A first lock would otherwise train the run on itself.
|
|
541
|
-
//
|
|
542
|
-
// ⚠ THE SEAMS ARE THE REAL FILESYSTEM HERE — this IS the production call site, so the write and
|
|
543
|
-
// read seams that the tests inject are bound to the actual run root, and the registry read goes
|
|
544
|
-
// to `memory/MEMORY.md` under the same root.
|
|
545
|
-
const isPhase8 = phase.startsWith('08') || phase === '08-memory-impact.md'
|
|
546
|
-
const rerun = isPhase8 && existsSync(join(runDir, 'locks', '08-memory-impact.receipt.json'))
|
|
547
|
-
// ⚠ FU-3 — THE DRAIN IS COMPUTED **BEFORE** THE STUB WRITE, and the first version had it after:
|
|
548
|
-
// `closeoutPhase` now THROWS when the artifact is already LOCKED (the FU-8 guard), so a drain placed
|
|
549
|
-
// after that call never ran on exactly the runs that reach closeout twice — the ones that HAVE
|
|
550
|
-
// children to release. Draining is a run-close action and does not depend on scaffolding a stub.
|
|
551
|
-
const children = isPhase8 ? runChildIds(runDir) : []
|
|
552
|
-
drain = isPhase8
|
|
553
|
-
? this.subagentsSeam === null
|
|
554
|
-
? { children: children.length, drained: false, reason: 'no subagents runtime is mounted, so the children could not be drained' }
|
|
555
|
-
: await drainContinuableChildren(this.subagentsSeam, agent as never, children)
|
|
556
|
-
.then(() => ({ children: children.length, drained: true }))
|
|
557
|
-
.catch((err: unknown) => ({ children: children.length, drained: false, reason: err instanceof Error ? err.message : String(err) }))
|
|
558
|
-
: null
|
|
559
|
-
// ⚠ FU-12 — THE CLOSEOUT NO LONGER WRITES TO THE PHASE DOCUMENT.
|
|
560
|
-
// It READS the artifact, reports what the standard requires and what is missing, and records that
|
|
561
|
-
// examination as a receipt of its own under locks/<stem>.closeout.receipt.json — never over the doc.
|
|
562
|
-
const result = closeoutReport(runDir, phase)
|
|
563
|
-
// P3b: settle this run's injections against its own outcome. Only phases that LOCKED are evidence, and
|
|
564
|
-
// a run settles once, at closeout.
|
|
565
|
-
settleInjections(root, runDir, PHASE_SEQUENCE.filter((file) => getLockStatus(join(runDir, file)) === 'LOCKED'))
|
|
566
|
-
writeCloseoutReceipt(runDir, phase)
|
|
567
|
-
const training = isPhase8
|
|
568
|
-
? runPhase8Trigger(root, runId, {
|
|
569
|
-
rerun,
|
|
570
|
-
// The extractor is resolved but NOT spawned here: this plugin never embeds one, and an
|
|
571
|
-
// unset command must surface as exit 2 rather than as a silent success.
|
|
572
|
-
extractorAvailable: resolveExtractor(process.env) !== null,
|
|
573
|
-
// FU-5: the PRODUCTION runner — spawn the command with `stdio: 'ignore'` and read the file it
|
|
574
|
-
// was asked to write. `stdio: 'ignore'` is deliberate: this sandbox denies a child the piped
|
|
575
|
-
// stdio a capture needs, and the response file is the parent's own interface anyway.
|
|
576
|
-
runner: spawnExtractorRunner({ cwd: root, responseFile: join(runDir, 'training-response.json') }),
|
|
577
|
-
write: (relativePath, content) => {
|
|
1
|
+
import { Service, type Context } from '@deepseek-ai/cordis'
|
|
2
|
+
import { existsSync, mkdirSync, readFileSync, readdirSync, writeFileSync } from 'node:fs'
|
|
3
|
+
import { dirname, join } from 'node:path'
|
|
4
|
+
import { lintRun } from './ts-lint.ts'
|
|
5
|
+
import { requirementsContent, worktreeContent, laterPhaseContent, detectGitContext, RUN_SCAFFOLD_DIRS, type GitContext } from './init-templates.ts'
|
|
6
|
+
import { foldRun, getMdFieldValue, pendingWork, resolveRunDir } from './status.ts'
|
|
7
|
+
import {
|
|
8
|
+
getLockStatus,
|
|
9
|
+
PHASE_SEQUENCE,
|
|
10
|
+
getNextLegalPhase,
|
|
11
|
+
getPrerequisiteBlockers,
|
|
12
|
+
getStaleDownstreamPhases,
|
|
13
|
+
invalidateReceipt,
|
|
14
|
+
lockHashFromContent,
|
|
15
|
+
validateReceiptChain,
|
|
16
|
+
writeReceipt,
|
|
17
|
+
type ReceiptChainResult,
|
|
18
|
+
} from './lock.ts'
|
|
19
|
+
import type { PendingWorkItem, RecursiveStatusResult } from './types.ts'
|
|
20
|
+
import { findOperation, countOperations, operationId, recordOperation } from './identity.ts'
|
|
21
|
+
import { createHookRegistry, type HookRegistry } from './hooks.ts'
|
|
22
|
+
import { runTracked, abortReason, type JobsRegistryLike } from './jobs-runner.ts'
|
|
23
|
+
import { resolveSubagentTarget } from './role-route.ts'
|
|
24
|
+
import { checkModelChoice, describeInventory, type LlmInventoryLike } from './model-inventory.ts'
|
|
25
|
+
import { recordJobRun } from './job-log.ts'
|
|
26
|
+
import { toolError } from './errors.ts'
|
|
27
|
+
import { readGuardDecisions, type GuardDecisionRecord } from './guard-log.ts'
|
|
28
|
+
import { resolveControlPlaneRoot, type WorkspaceRegistryLike } from './workspace.ts'
|
|
29
|
+
import { phaseRulesFor, type PhaseRules } from './phase-rules.ts'
|
|
30
|
+
import { closeoutReport, writeCloseoutReceipt } from './closeout-report.ts'
|
|
31
|
+
import { readScratch, writeScratch, appendScratch, type ScratchTarget } from './scratch.ts'
|
|
32
|
+
import { buildReviewBundle, type ReviewBundleInput } from './review.ts'
|
|
33
|
+
import { readMemoryEntries, retrieveMemory, renderMemorySection, selectMemory } from './memory.ts'
|
|
34
|
+
import { readFeedback, recordInjection, recordMemoryRead, settleInjections } from './memory-feedback.ts'
|
|
35
|
+
import { runPhase8Trigger, resolveExtractor, spawnExtractorRunner, phase8MemoryLockRefusal } from './training.ts'
|
|
36
|
+
import { buildAskQuestion, GATE_DEFAULT_ARTIFACT, pendingGateFor } from './recursive_ask.tool.ts'
|
|
37
|
+
import { contractDigest } from './policy.ts'
|
|
38
|
+
import type { WorkflowEngineLike } from './workflow-audit.ts'
|
|
39
|
+
import { createHandoff, createChildBrief, replyPath, childScratchPath, buildDelegationPrompt, type HandoffInput, type ChildBriefInput } from './handoff.ts'
|
|
40
|
+
import { loadRouterPolicy, routerPolicyPath, resolveRole, capabilityProbe, delegationDecisionBasis, type RouterPolicy, type RouterPolicyOverrides, type SubagentProviderLike, type RouteDecision, type CapabilityProbe } from './router.ts'
|
|
41
|
+
import { delegate, delegateContinuable, drainContinuableChildren, remainingDepthFor, validateReferences, referencesFromResult, writeActionRecord, evaluateDelegationResult, reviewOutputSchema, defaultReviewToolFilter, type SubagentsRuntimeLike, type SubagentStartRequestLike, type SubagentResultLike, type Reference, type ActionRecordInput, type ContinuableDelegationLike, type SubagentParentHandle } from './delegation.ts'
|
|
42
|
+
import { validateTransition, coupleGateBlockToGoal, type PhaseTransitionIntent, type RecursivePhaseState, type GateCheckResult } from './lifecycle.ts'
|
|
43
|
+
import { resolveEnforcementConfig, DEFAULT_ENFORCEMENT, evaluateToolGuard, detectTamper, type EnforcementConfig, type ToolGuardDecision, type ToolExecLike } from './enforcement.ts'
|
|
44
|
+
import type { Session } from '@deepseek-ai/dsh-session'
|
|
45
|
+
import { renderRecursivePolicy, type PolicyContext } from './policy.ts'
|
|
46
|
+
import { snapshotWorkspace } from './snapshot.ts'
|
|
47
|
+
import { createLinkedWorktree, promoteBranch, listWorktrees, defaultWorktreeBranch, type CreateWorktreeResult, type PromoteBranchResult } from './worktree.ts'
|
|
48
|
+
import { changedPaths, gitFacts } from './git-context.ts'
|
|
49
|
+
import { syncRunGoal, blockRunGoal, resumeRunGoal, type GoalServiceLike, type SyncResult } from './goals-projection.ts'
|
|
50
|
+
import {
|
|
51
|
+
RUN_START_APPROVE,
|
|
52
|
+
RUN_START_ARTIFACT,
|
|
53
|
+
RUN_START_GATE_ID,
|
|
54
|
+
RUN_START_MARKER,
|
|
55
|
+
RUN_START_NOT_APPROVED,
|
|
56
|
+
readRunStartApproval,
|
|
57
|
+
runStartArtifactPath,
|
|
58
|
+
} from './run-start.ts'
|
|
59
|
+
import { auditToPass, renderTaskHistory, type TeamRuntimeLike, type AuditToPassResult, type TeamCallerHandle, type TeamTaskViewLike, type AuditRoundOutcome } from './teams-loop.ts'
|
|
60
|
+
import type { ContinuableChildId, ContinuableMessageId } from './delegation.ts'
|
|
61
|
+
import { runChildIds } from './settlement.ts'
|
|
62
|
+
|
|
63
|
+
declare module '@deepseek-ai/cordis' {
|
|
64
|
+
interface Context {
|
|
65
|
+
recursive: RecursiveRuntime
|
|
66
|
+
}
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
export interface LockArtifactResult {
|
|
70
|
+
artifact: string
|
|
71
|
+
runId: string
|
|
72
|
+
status: string
|
|
73
|
+
lockedAt: string | null
|
|
74
|
+
lockHash: string | null
|
|
75
|
+
receipt?: unknown
|
|
76
|
+
blockers: string[]
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
export interface LintArtifactResult {
|
|
80
|
+
artifact: string
|
|
81
|
+
runId: string
|
|
82
|
+
errors: string[]
|
|
83
|
+
warnings: string[]
|
|
84
|
+
passed: boolean
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
/**
|
|
88
|
+
* PHASE 0 — the structural seam for the host's human-question channel (`ctx.userQuestions`).
|
|
89
|
+
*
|
|
90
|
+
* Declared here as a minimal seam for the same reason as every other harness touchpoint in this plugin:
|
|
91
|
+
* the live `UserQuestionService` satisfies it structurally, so the plugin never imports the host package,
|
|
92
|
+
* and a test can drive the run-start gate with a fake that behaves like the real one. The error case is
|
|
93
|
+
* part of the contract, not an afterthought: the real `ask()` REJECTS (NO_PROVIDER / CALLER_NOT_LIVE /
|
|
94
|
+
* ASK_ABORTED) instead of resolving with something that could be mistaken for an answer, which is what
|
|
95
|
+
* lets `recursive_ask` fail closed rather than invent an approval.
|
|
96
|
+
*/
|
|
97
|
+
export interface UserQuestionsLike {
|
|
98
|
+
ask(request: {
|
|
99
|
+
questions: Array<{ id: string; header?: string; question: string; options?: Array<{ label: string; description?: string }> }>
|
|
100
|
+
agent?: unknown
|
|
101
|
+
signal?: AbortSignal
|
|
102
|
+
/** Links the card to the tool call that asked, the way plan-mode's exit does. */
|
|
103
|
+
wait?: { callId?: unknown }
|
|
104
|
+
}): Promise<{ answers: Array<{ id: string; selected: string[]; custom?: string }> }>
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
/**
|
|
108
|
+
* T15 (G): the folded status PLUS the rolling guard-decision evidence. Declared
|
|
109
|
+
* as an intersection rather than by editing RecursiveStatusResult/foldRun — the
|
|
110
|
+
* fold's own shape is parity-asserted and stays exactly as it was.
|
|
111
|
+
*/
|
|
112
|
+
export type RecursiveStatusWithGuardDecisions = RecursiveStatusResult & {
|
|
113
|
+
guardDecisions?: GuardDecisionRecord[]
|
|
114
|
+
/**
|
|
115
|
+
* T18: unresolved in-flight work, derived from the run directory on every call.
|
|
116
|
+
* Always present (empty when nothing is in flight) so consumers need no null
|
|
117
|
+
* check, and non-empty explains a `RM4403` lock refusal.
|
|
118
|
+
*/
|
|
119
|
+
pendingWork?: PendingWorkItem[]
|
|
120
|
+
/**
|
|
121
|
+
* T32: the receipt-chain verdict, derived read-only on every call. Always present
|
|
122
|
+
* so a caller can read `ok` without a null check; non-empty `breaks` names the
|
|
123
|
+
* first broken link. This is what makes a spliced or edited chain VISIBLE rather
|
|
124
|
+
* than merely detectable in a test.
|
|
125
|
+
*/
|
|
126
|
+
receiptChain?: ReceiptChainResult
|
|
127
|
+
/**
|
|
128
|
+
* T22: the local identifier of the policy section's STABLE prefix.
|
|
129
|
+
*
|
|
130
|
+
* Surfaced so "did the contract change under me?" is answerable from the status alone — the same
|
|
131
|
+
* digest the prompt carries, so a reader can compare them without re-rendering anything. It is an
|
|
132
|
+
* IDENTIFIER, not a cache directive: whether any provider caches the prefix is provider-side and
|
|
133
|
+
* unverified, which is why the item's "largest cost lever" label was withdrawn.
|
|
134
|
+
*/
|
|
135
|
+
contractDigest?: string
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
/** How many recent decisions to read from the log before scoping to one run. */
|
|
139
|
+
const GUARD_DECISION_READ_LIMIT = 200
|
|
140
|
+
|
|
141
|
+
/**
|
|
142
|
+
* T28: the delegating agent's own delegation depth, read structurally rather than by
|
|
143
|
+
* importing the harness's `delegationDepthOf`. The plugin already models every
|
|
144
|
+
* harness touchpoint as a minimal structural seam, and this keeps that convention —
|
|
145
|
+
* an absent depth means top level, which is the safe reading for a budget.
|
|
146
|
+
*/
|
|
147
|
+
function parentDepthOf(parent: unknown): number {
|
|
148
|
+
const depth = (parent as { options?: { subagentDepth?: unknown } } | null | undefined)?.options?.subagentDepth
|
|
149
|
+
return typeof depth === 'number' && Number.isSafeInteger(depth) && depth > 0 ? depth : 0
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
/** How many of that run's decisions the status surface carries. */
|
|
153
|
+
const GUARD_DECISION_SURFACE_LIMIT = 20
|
|
154
|
+
|
|
155
|
+
const ARTIFACT_STUB = {
|
|
156
|
+
'00-requirements.md': ['Run: ', 'Phase: 0', 'Status: DRAFT', 'Workflow version: recursive-mode-audit-v2', 'Inputs: none', 'Outputs: none', 'Scope note: '],
|
|
157
|
+
'00-worktree.md': ['Run: ', 'Phase: 0 (Worktree)', 'Status: DRAFT', 'Workflow version: recursive-mode-audit-v2', 'Inputs: none', 'Outputs: none', 'Scope note: '],
|
|
158
|
+
'01-as-is.md': ['Run: ', 'Phase: 1 (AS-IS)', 'Status: DRAFT', 'Workflow version: recursive-mode-audit-v2', 'Inputs: none', 'Outputs: none', 'Scope note: '],
|
|
159
|
+
'02-to-be-plan.md': ['Run: ', 'Phase: 2 (TO-BE Plan)', 'Status: DRAFT', 'Workflow version: recursive-mode-audit-v2', 'Inputs: none', 'Outputs: none', 'Scope note: '],
|
|
160
|
+
'03-implementation-summary.md': ['Run: ', 'Phase: 3 (Implementation)', 'Status: DRAFT', 'Workflow version: recursive-mode-audit-v2', 'Inputs: none', 'Outputs: none', 'Scope note: '],
|
|
161
|
+
'04-test-summary.md': ['Run: ', 'Phase: 4 (Test Summary)', 'Status: DRAFT', 'Workflow version: recursive-mode-audit-v2', 'Inputs: none', 'Outputs: none', 'Scope note: '],
|
|
162
|
+
'05-manual-qa.md': ['Run: ', 'Phase: 5 (Manual QA)', 'Status: DRAFT', 'Workflow version: recursive-mode-audit-v2', 'Inputs: none', 'Outputs: none', 'Scope note: '],
|
|
163
|
+
'06-decisions-update.md': ['Run: ', 'Phase: 6 (Decisions Update)', 'Status: DRAFT', 'Workflow version: recursive-mode-audit-v2', 'Inputs: none', 'Outputs: none', 'Scope note: '],
|
|
164
|
+
'07-state-update.md': ['Run: ', 'Phase: 7 (State Update)', 'Status: DRAFT', 'Workflow version: recursive-mode-audit-v2', 'Inputs: none', 'Outputs: none', 'Scope note: '],
|
|
165
|
+
'08-memory-impact.md': ['Run: ', 'Phase: 8 (Memory Impact)', 'Status: DRAFT', 'Workflow version: recursive-mode-audit-v2', 'Inputs: none', 'Outputs: none', 'Scope note: '],
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
export class RecursiveRuntime extends Service {
|
|
169
|
+
/** Recursive-mode runtime service. Owns run-state reads + lock/init/lint operations. */
|
|
170
|
+
|
|
171
|
+
constructor(ctx: Context, config: { repoRoot?: string; workspaceRegistry?: WorkspaceRegistryLike; goals?: GoalServiceLike | null; jobs?: JobsRegistryLike | null; subagents?: SubagentsRuntimeLike | null; workflow?: WorkflowEngineLike | null } = {}) {
|
|
172
|
+
super(ctx, 'recursive')
|
|
173
|
+
this.repoRoot = config.repoRoot ?? process.cwd()
|
|
174
|
+
this.workspaceRegistry = config.workspaceRegistry ?? null
|
|
175
|
+
this.goalsService = config.goals ?? null
|
|
176
|
+
// T10: the native jobs registry is OPTIONAL. Absent, long operations run inline and say
|
|
177
|
+
// so; see `runTracked` for why that is better than refusing to work without a board.
|
|
178
|
+
this.jobs = config.jobs ?? null
|
|
179
|
+
// T39: the subagents seam, resolved at the COMPOSITION like the other optional services.
|
|
180
|
+
this.subagentsSeam = config.subagents ?? null
|
|
181
|
+
// T2: the workflow engine, likewise.
|
|
182
|
+
this.workflow = config.workflow ?? null
|
|
183
|
+
}
|
|
184
|
+
|
|
185
|
+
/**
|
|
186
|
+
* T10: the native jobs registry, when the composition mounts one. */
|
|
187
|
+
private readonly jobs: JobsRegistryLike | null
|
|
188
|
+
|
|
189
|
+
/**
|
|
190
|
+
* PHASE 0 — attach the goals service after construction.
|
|
191
|
+
*
|
|
192
|
+
* The composition resolves `goals` with ONE `ctx.get` at apply time and passes it to the constructor,
|
|
193
|
+
* which is fine for a service that is already mounted. This seam exists for the two cases that pattern
|
|
194
|
+
* cannot cover: a composition that mounts `goals` later (the same late-attach reason `attachSubagents`
|
|
195
|
+
* and `attachLlmInventory` exist), and a test that needs the REAL runtime wired to a structural fake —
|
|
196
|
+
* a fake passed through the plugin's Config is dropped, because the Config schema is the settings
|
|
197
|
+
* namespace and strips keys it does not declare.
|
|
198
|
+
*/
|
|
199
|
+
attachGoals(service: GoalServiceLike | null): void {
|
|
200
|
+
this.goalsService = service
|
|
201
|
+
}
|
|
202
|
+
|
|
203
|
+
/**
|
|
204
|
+
* T23 — write a gate's answer into an artifact as a marker line.
|
|
205
|
+
*
|
|
206
|
+
* ⚠ REPLACED IN PLACE when the artifact already carries that gate's marker: two `TDD Mode:` lines
|
|
207
|
+
* would leave two answers to one question and make "what was decided?" depend on which a reader
|
|
208
|
+
* found first. The write is confined to the run directory, and an artifact that does not exist is
|
|
209
|
+
* CREATED — a decision recorded nowhere is not recorded.
|
|
210
|
+
*/
|
|
211
|
+
recordAskAnswer(root: string, runId: string, artifact: string, marker: string): { path: string; replaced: boolean } {
|
|
212
|
+
const dir = join(root, '.recursive', 'run', runId)
|
|
213
|
+
const path = join(dir, artifact)
|
|
214
|
+
const label = marker.slice(2).split(':')[0].trim()
|
|
215
|
+
let content = ''
|
|
216
|
+
try {
|
|
217
|
+
content = readFileSync(path, 'utf8')
|
|
218
|
+
} catch {
|
|
219
|
+
// A missing artifact is created below, so the decision still has somewhere to live.
|
|
220
|
+
}
|
|
221
|
+
const lines = content === '' ? [] : content.replace(/\n$/, '').split('\n')
|
|
222
|
+
const at = lines.findIndex((line) => line.startsWith('- ' + label + ':'))
|
|
223
|
+
const replaced = at >= 0
|
|
224
|
+
if (replaced) lines[at] = marker
|
|
225
|
+
else lines.push(marker)
|
|
226
|
+
mkdirSync(dir, { recursive: true })
|
|
227
|
+
writeFileSync(path, lines.join('\n') + '\n', 'utf8')
|
|
228
|
+
return { path, replaced }
|
|
229
|
+
}
|
|
230
|
+
|
|
231
|
+
/**
|
|
232
|
+
* T2: the workflow engine, when the composition mounts one.
|
|
233
|
+
*
|
|
234
|
+
* OPTIONAL like every other seam here — without it an audit fan-out reports that it could not be
|
|
235
|
+
* orchestrated rather than pretending a fan-out happened. The engine's `workflow/*` events are
|
|
236
|
+
* observe-only, so this is used to START a run and await its result, never to drive one.
|
|
237
|
+
*/
|
|
238
|
+
private readonly workflow: WorkflowEngineLike | null
|
|
239
|
+
|
|
240
|
+
/**
|
|
241
|
+
* T39: the subagents seam the composition mounted, used when a caller does not pass one.
|
|
242
|
+
*
|
|
243
|
+
* ⚠ WHY THIS EXISTS, measured rather than assumed: `recursive_review.tool.ts` — the ONLY
|
|
244
|
+
* production caller of `delegateReview` — passes **no `subagents`** at its call site, and
|
|
245
|
+
* `delegateReview` reports *"no ctx.subagents runtime available (self-audit fallback)"* when
|
|
246
|
+
* the input lacks one. So on a composition that HAS the service, the review tool's rounds
|
|
247
|
+
* never reached a child at all, and the fallback message blamed a missing runtime that was
|
|
248
|
+
* in fact mounted. Resolving the seam here fixes the wiring without asking every call site to
|
|
249
|
+
* remember, while an explicit `input.subagents` still wins for a test or a narrower caller.
|
|
250
|
+
*/
|
|
251
|
+
private subagentsSeam: SubagentsRuntimeLike | null
|
|
252
|
+
/** FU-19: the host's provider/model inventory, or null when no llm service is mounted. */
|
|
253
|
+
private llmInventory: LlmInventoryLike | null = null
|
|
254
|
+
|
|
255
|
+
/**
|
|
256
|
+
* ⚠ FU-9 — ATTACH THE SEAM WHEN THE SERVICE APPEARS, not only when this plugin happens to apply.
|
|
257
|
+
*
|
|
258
|
+
* The composition resolved the seam with a ONE-SHOT `ctx.get('subagents')` at apply time, and a live run
|
|
259
|
+
* showed what that costs: the review fell back to self-audit, the action record said
|
|
260
|
+
* `Execution Mode: self-audit (continuable)` and `Status: failed`, and **no child was ever started** — while
|
|
261
|
+
* the child DIRECTORY existed all along, because the plugin writes its own brief before calling any service.
|
|
262
|
+
* I read the directory and built a host-limitation story on top of it; the record said otherwise.
|
|
263
|
+
*
|
|
264
|
+
* If the subagents service is mounted by a later loader layer, a one-shot get returns undefined and nothing
|
|
265
|
+
* re-resolves it. `ctx.inject(['subagents'], …)` is the harness's own pattern for exactly this, and calling
|
|
266
|
+
* this method from there makes the seam arrive whenever it arrives. Idempotent: the last attach wins, which
|
|
267
|
+
* is what a re-apply after a reload wants.
|
|
268
|
+
*/
|
|
269
|
+
attachSubagents(seam: SubagentsRuntimeLike | null): void {
|
|
270
|
+
this.subagentsSeam = seam
|
|
271
|
+
}
|
|
272
|
+
|
|
273
|
+
/**
|
|
274
|
+
* ⚠ FU-19 — THE LLM INVENTORY SEAM, resolved the same late-attaching way the subagents seam is and for the same
|
|
275
|
+
* measured reason: a one-shot `ctx.get` at apply time misses a service mounted by a later layer.
|
|
276
|
+
*
|
|
277
|
+
* Null is a legitimate value and it is NOT treated as "no models exist" — `describeInventory(null)` reports a
|
|
278
|
+
* named unavailability, and `checkModelChoice` turns that into the `unverified` verdict. A missing inventory
|
|
279
|
+
* therefore never silently approves a model and never silently replaces one.
|
|
280
|
+
*/
|
|
281
|
+
attachLlmInventory(seam: LlmInventoryLike | null): void {
|
|
282
|
+
this.llmInventory = seam
|
|
283
|
+
}
|
|
284
|
+
|
|
285
|
+
/** What the composition attached, for a caller that needs to report or assert it. */
|
|
286
|
+
attachedSubagents(): SubagentsRuntimeLike | null {
|
|
287
|
+
return this.subagentsSeam
|
|
288
|
+
}
|
|
289
|
+
|
|
290
|
+
/**
|
|
291
|
+
* ⚠ FU-9 — THE ROUTER'S PROVIDER MAP, BUILT FROM THE SEAM THAT IS ALREADY ATTACHED.
|
|
292
|
+
*
|
|
293
|
+
* `router.ts` returns the NATIVE tier for the first of `[role, 'spawn', 'fork', 'dsh-sdk']` present in this
|
|
294
|
+
* map. Handed `{}` it tried the external CLI route (null in the default policy) and fell to the policy
|
|
295
|
+
* fallback — self-audit — with a message naming neither. The router already preferred native; nobody ever
|
|
296
|
+
* gave it a name. A live review self-audited for five rounds because of it.
|
|
297
|
+
*
|
|
298
|
+
* `SubagentProviderLike` is only a DESCRIPTOR (`{ name, capabilities? }`), so a provider is registered by
|
|
299
|
+
* ASKING the service for it rather than by wrapping it. A service that cannot enumerate yields an empty map
|
|
300
|
+
* and the previous behaviour, which is the correct degradation rather than a guess about the shape.
|
|
301
|
+
*/
|
|
302
|
+
private providerMapFromSeam(): Record<string, SubagentProviderLike> {
|
|
303
|
+
const seam = this.subagentsSeam
|
|
304
|
+
this.lastProviderNames = []
|
|
305
|
+
if (seam === null) return {}
|
|
306
|
+
const map: Record<string, SubagentProviderLike> = {}
|
|
307
|
+
// ⚠ ASK THE SERVICE WHAT IT HAS, FIRST. The previous version only probed three invented names and
|
|
308
|
+
// registered whatever came back — and the live record then said `provider spawn`, so the router faithfully
|
|
309
|
+
// returned a name the host does not serve and `start('spawn', …)` produced NOTHING. A name this plugin made
|
|
310
|
+
// up is not a provider the host knows, and the difference is a child that runs versus a silent no-op.
|
|
311
|
+
try {
|
|
312
|
+
const listed = seam.list?.()
|
|
313
|
+
if (Array.isArray(listed)) {
|
|
314
|
+
for (const entry of listed) {
|
|
315
|
+
if (typeof entry === 'string' && entry !== '') map[entry] = { name: entry }
|
|
316
|
+
else if (entry !== null && typeof entry === 'object' && typeof (entry as { name?: unknown }).name === 'string') {
|
|
317
|
+
const named = entry as { name: string } & SubagentProviderLike
|
|
318
|
+
map[named.name] = named
|
|
319
|
+
}
|
|
320
|
+
}
|
|
321
|
+
}
|
|
322
|
+
} catch {
|
|
323
|
+
// A service that cannot enumerate is not an error: the probes below are the fallback, not the plan.
|
|
324
|
+
}
|
|
325
|
+
// The probes stay as a FALLBACK for a service that exposes getProvider but no list. Whatever they return is
|
|
326
|
+
// registered under the name ASKED FOR, which is only sound because the name came from a real lookup.
|
|
327
|
+
for (const name of ['spawn', 'fork', 'dsh-sdk']) {
|
|
328
|
+
if (map[name] !== undefined) continue
|
|
329
|
+
try {
|
|
330
|
+
const found = seam.getProvider?.(name)
|
|
331
|
+
if (found !== undefined && found !== null) map[name] = found as SubagentProviderLike
|
|
332
|
+
} catch {
|
|
333
|
+
// An unavailable name is not an error: the next candidate still gets its turn.
|
|
334
|
+
}
|
|
335
|
+
}
|
|
336
|
+
this.lastProviderNames = Object.keys(map)
|
|
337
|
+
return map
|
|
338
|
+
}
|
|
339
|
+
|
|
340
|
+
/** The provider names the last `providerMapFromSeam` call registered, for the record and for assertions. */
|
|
341
|
+
private lastProviderNames: string[] = []
|
|
342
|
+
|
|
343
|
+
/** What the router could choose from, so a failure can say whether the name it used was ever on offer. */
|
|
344
|
+
knownProviderNames(): string[] {
|
|
345
|
+
return this.lastProviderNames
|
|
346
|
+
}
|
|
347
|
+
|
|
348
|
+
private readonly repoRoot: string
|
|
349
|
+
private readonly workspaceRegistry: WorkspaceRegistryLike | null
|
|
350
|
+
private goalsService: GoalServiceLike | null
|
|
351
|
+
/**
|
|
352
|
+
* T27 — the hook registry, EXPOSED so a sibling plugin can participate in a run
|
|
353
|
+
* without patching this one:
|
|
354
|
+
*
|
|
355
|
+
* ctx.recursive.hooks.register('pre_trigger', { name: 'my-check', priority: 10, run })
|
|
356
|
+
*
|
|
357
|
+
* That is the whole point of the item: the plugin's own enforcement will be
|
|
358
|
+
* re-expressed as built-in hooks on this same registry, so a sibling and a built-in
|
|
359
|
+
* are peers — same ordering rules, same failure policy, same audit trail — rather
|
|
360
|
+
* than one being privileged code and the other a guest.
|
|
361
|
+
*
|
|
362
|
+
* Public and created eagerly: a registry that has to be "got" before it can be used
|
|
363
|
+
* is a registry whose ordering depends on when someone remembered to fetch it.
|
|
364
|
+
*/
|
|
365
|
+
readonly hooks: HookRegistry = createHookRegistry()
|
|
366
|
+
|
|
367
|
+
/**
|
|
368
|
+
* T7 — router overrides from the settings namespace. Kept beside the config rather than
|
|
369
|
+
* merged into the file so the workspace's `recursive-router.json` stays the declarative
|
|
370
|
+
* source: `loadRouterPolicy` reads the file and lays these on top.
|
|
371
|
+
*/
|
|
372
|
+
private _routerOverrides: RouterPolicyOverrides | undefined = undefined
|
|
373
|
+
|
|
374
|
+
/** T7: set (or clear) the router overrides. Called from `apply` on every plugin load. */
|
|
375
|
+
setRouterOverrides(overrides: RouterPolicyOverrides | undefined): void {
|
|
376
|
+
this._routerOverrides = overrides
|
|
377
|
+
}
|
|
378
|
+
|
|
379
|
+
private _enforcementConfig: EnforcementConfig | null = null
|
|
380
|
+
|
|
381
|
+
/**
|
|
382
|
+
* T3 (agentTeams task loop): run the audit→repair→re-audit state machine on
|
|
383
|
+
* ONE durable team task. The `teams` seam (live `ctx.agentTeams`) is injected
|
|
384
|
+
* per-call so the loop stays unit-testable; `runAuditRound` is the caller's
|
|
385
|
+
* round executor (live usage wires T4's continuable delegation). Locking the
|
|
386
|
+
* phase artifact is `lockPhase` — the loop NEVER locks before an APPROVE.
|
|
387
|
+
*/
|
|
388
|
+
async auditToPass(input: {
|
|
389
|
+
teams: TeamRuntimeLike
|
|
390
|
+
caller: TeamCallerHandle
|
|
391
|
+
root: string
|
|
392
|
+
runId: string
|
|
393
|
+
phase: string
|
|
394
|
+
artifact: string
|
|
395
|
+
agent?: { session?: { header?: { cwd?: string } } } | null
|
|
396
|
+
runAuditRound: (round: number, task: TeamTaskViewLike) => Promise<AuditRoundOutcome>
|
|
397
|
+
blockedBy?: readonly string[]
|
|
398
|
+
writeScopes?: readonly string[]
|
|
399
|
+
reviewerName?: string
|
|
400
|
+
maxRounds?: number
|
|
401
|
+
waitTimeoutMs?: number
|
|
402
|
+
}): Promise<AuditToPassResult & { history?: string; lock?: LockArtifactResult }> {
|
|
403
|
+
const { teams, caller, runId, phase, artifact, runAuditRound, agent } = input
|
|
404
|
+
let lockResult: LockArtifactResult | undefined
|
|
405
|
+
const lockPhase = async () => { lockResult = await this.lockArtifact(runId, artifact, false, agent) }
|
|
406
|
+
const result = await auditToPass({
|
|
407
|
+
teams,
|
|
408
|
+
caller,
|
|
409
|
+
runId,
|
|
410
|
+
phase,
|
|
411
|
+
blockedBy: input.blockedBy,
|
|
412
|
+
writeScopes: input.writeScopes,
|
|
413
|
+
reviewerName: input.reviewerName,
|
|
414
|
+
maxRounds: input.maxRounds,
|
|
415
|
+
waitTimeoutMs: input.waitTimeoutMs,
|
|
416
|
+
runAuditRound,
|
|
417
|
+
lockPhase,
|
|
418
|
+
})
|
|
419
|
+
const report: AuditToPassResult & { history?: string; lock?: LockArtifactResult } = {
|
|
420
|
+
...result,
|
|
421
|
+
history: renderTaskHistory(result.taskView, result.rounds),
|
|
422
|
+
}
|
|
423
|
+
if (lockResult !== undefined) report.lock = lockResult
|
|
424
|
+
return report
|
|
425
|
+
}
|
|
426
|
+
|
|
427
|
+
/**
|
|
428
|
+
* PHASE 0 — read a run's start approval from its own Phase 0 artifact.
|
|
429
|
+
*
|
|
430
|
+
* The approval is a DURABLE line in `.recursive/run/<runId>/00-requirements.md`, not a value held in
|
|
431
|
+
* memory, for the reason every other gate here is durable: a decision that only exists in a session
|
|
432
|
+
* cannot be cited, and cannot survive the session it was made in. Read-only; asking changes nothing.
|
|
433
|
+
*/
|
|
434
|
+
readRunStartApproval(root: string, runId: string): { approved: boolean; artifact: string; reason: string } {
|
|
435
|
+
return readRunStartApproval(root, runId)
|
|
436
|
+
}
|
|
437
|
+
|
|
438
|
+
/**
|
|
439
|
+
* PHASE 0 — the harness's blocking human-question channel (`ctx.userQuestions`), when this composition
|
|
440
|
+
* mounts one.
|
|
441
|
+
*
|
|
442
|
+
* ⚠ WHY THE PLUGIN REACHES FOR THIS AT ALL. The other three gates answer through a question card and a
|
|
443
|
+
* relayed label, which is fine for a decision the workflow acts on later. STARTING A RUN is different:
|
|
444
|
+
* the first human turn is the only place the harness can say "arming this goal means autonomous rounds"
|
|
445
|
+
* BEFORE arming it. This channel is the same one plan-mode's exit uses; `ask()` resolves only with a
|
|
446
|
+
* real answer from a real person, and it THROWS when there is no answerer or no live root agent. So the
|
|
447
|
+
* absence of this service cannot be papered over: `recursive_ask` refuses the run-start gate and names
|
|
448
|
+
* the missing channel (RM5502).
|
|
449
|
+
*/
|
|
450
|
+
private userQuestions: UserQuestionsLike | null = null
|
|
451
|
+
|
|
452
|
+
/** Late-bind the human-question channel when the composition mounts it. */
|
|
453
|
+
attachUserQuestions(service: UserQuestionsLike | null): void {
|
|
454
|
+
this.userQuestions = service
|
|
455
|
+
}
|
|
456
|
+
|
|
457
|
+
/** The human-question channel this composition mounted, or null. */
|
|
458
|
+
get userQuestionsChannel(): UserQuestionsLike | null {
|
|
459
|
+
return this.userQuestions
|
|
460
|
+
}
|
|
461
|
+
|
|
462
|
+
/**
|
|
463
|
+
* T1 (goals projection): project the run into the native goals service so it is
|
|
464
|
+
* a first-class durable, resumable, blockable object. Best-effort — the run's
|
|
465
|
+
* filesystem state is the source of truth; a goal is the durable projection.
|
|
466
|
+
*
|
|
467
|
+
* ⚠ PHASE 0 — AND IT ARMS NOTHING UNLESS THE RUN WAS STARTED. `create` returns an ARMED goal, and an
|
|
468
|
+
* armed goal is the harness driving autonomous rounds, so this is the one place where "project the
|
|
469
|
+
* state" can quietly equal "start the run". The approval is therefore REQUIRED from the caller and
|
|
470
|
+
* has no default here: a caller that has not resolved the run's approval cannot arm a goal by
|
|
471
|
+
* forgetting to pass one, and `syncRunGoal` refuses every branch that would create one without it.
|
|
472
|
+
*
|
|
473
|
+
* Unapproved is the EXPECTED state for a scaffolded run, so the refusal comes back as a plain
|
|
474
|
+
* `{ ok: false }` carrying {@link RUN_START_NOT_APPROVED}: callers must treat that as normal work,
|
|
475
|
+
* never as a warning (see `armRunGoalIfApproved`, the one caller that arms).
|
|
476
|
+
*/
|
|
477
|
+
projectRunToGoal(agent: { session?: { header?: { cwd?: string } } } | null | undefined, runId: string, state: Parameters<typeof syncRunGoal>[3] = 'active', approved = false): SyncResult {
|
|
478
|
+
if (!agent) return { ok: false, reason: 'no agent' }
|
|
479
|
+
return syncRunGoal(this.goalsService, agent, runId, state, approved)
|
|
480
|
+
}
|
|
481
|
+
|
|
482
|
+
/**
|
|
483
|
+
* PHASE 0 — the approved-run arm step: read the run's approval from `root` and project the goal only
|
|
484
|
+
* if it is there. This is the phase-progress path (`syncRunGoal` reached on ordinary work), so the
|
|
485
|
+
* unapproved case is deliberately silent: `{ ok: false, reason: RUN_START_NOT_APPROVED }` with no
|
|
486
|
+
* goal, no write and no throw. Read on EVERY call rather than cached, because the approval can arrive
|
|
487
|
+
* mid-session and a cached "not yet" would leave an approved run unable to arm until a plugin reload.
|
|
488
|
+
*/
|
|
489
|
+
armRunGoalIfApproved(agent: { session?: { header?: { cwd?: string } } } | null | undefined, root: string, runId: string, state: Parameters<typeof syncRunGoal>[3] = 'active'): SyncResult {
|
|
490
|
+
if (!agent) return { ok: false, reason: 'no agent' }
|
|
491
|
+
const approval = this.readRunStartApproval(root, runId)
|
|
492
|
+
return syncRunGoal(this.goalsService, agent, runId, state, approval.approved)
|
|
493
|
+
}
|
|
494
|
+
|
|
495
|
+
/** T1: block the run's goal on a gate-block (durable + UI-visible). */
|
|
496
|
+
blockRunToGoal(agent: { session?: { header?: { cwd?: string } } } | null | undefined, runId: string, reason: { code: string; message: string }): SyncResult {
|
|
497
|
+
if (!agent) return { ok: false, reason: 'no agent' }
|
|
498
|
+
return blockRunGoal(this.goalsService, agent, runId, reason)
|
|
499
|
+
}
|
|
500
|
+
|
|
501
|
+
/**
|
|
502
|
+
* T1: re-arm the run's goal on a reopen (blocked/paused -> active). Never starts an unstarted run —
|
|
503
|
+
* REOPEN IS NOT A BACK DOOR TO STARTING A RUN. `approved` is required for the same reason as in
|
|
504
|
+
* `projectRunToGoal`: the phase-0 gate cannot be defaulted open. An approved run's approval outlives
|
|
505
|
+
* a reopen because it is a durable line in the run's own Phase 0 artifact, not a held value.
|
|
506
|
+
*/
|
|
507
|
+
resumeRunToGoal(agent: { session?: { header?: { cwd?: string } } } | null | undefined, runId: string, approved = false): SyncResult {
|
|
508
|
+
if (!agent) return { ok: false, reason: 'no agent' }
|
|
509
|
+
return resumeRunGoal(this.goalsService, agent, runId, approved)
|
|
510
|
+
}
|
|
511
|
+
|
|
512
|
+
/**
|
|
513
|
+
* Workspace-scoped control-plane root (R1 binding invariant).
|
|
514
|
+
* Resolves the session agent's canonical cwd -> workspace path via the
|
|
515
|
+
* registry; NEVER scans list(). Returns null when unavailable (defer).
|
|
516
|
+
*/
|
|
517
|
+
async resolveWorkspaceRoot(agent?: { session?: { header?: { cwd?: string } } } | null): Promise<string | null> {
|
|
518
|
+
return resolveControlPlaneRoot(agent, this.workspaceRegistry)
|
|
519
|
+
}
|
|
520
|
+
|
|
521
|
+
/**
|
|
522
|
+
* Run-scoped closeout receipt scaffold (R2), rooted under the given
|
|
523
|
+
* workspace root. Refuses runIds outside the root (never crosses workspaces).
|
|
524
|
+
*/
|
|
525
|
+
async closeoutRun(root: string, runId: string, phase: string, agent?: { session?: { header?: { cwd?: string } } } | null) {
|
|
526
|
+
const runDir = join(root, '.recursive', 'run', runId)
|
|
527
|
+
const runRoot = join(root, '.recursive', 'run')
|
|
528
|
+
// workspace-scoping guard: the run must be under this root
|
|
529
|
+
if (!runDir.startsWith(runRoot) || !existsSync(runDir)) {
|
|
530
|
+
return { error: 'Run not found in current workspace: ' + runId }
|
|
531
|
+
}
|
|
532
|
+
// Declared OUTSIDE the try so the failure path can report it too: a refused closeout still tells the
|
|
533
|
+
// caller whether its children were released.
|
|
534
|
+
let drain: { children: number; drained: boolean; reason?: string } | null = null
|
|
535
|
+
try {
|
|
536
|
+
// T30 — THE RUN-CLOSE TRIGGER, at the RE-RUN of closeout phase 08 and nowhere else.
|
|
537
|
+
//
|
|
538
|
+
// ⚠ DETECTED FROM THE RECEIPT THAT ALREADY EXISTS, not from a counter kept beside it: if phase 08
|
|
539
|
+
// has a receipt BEFORE this closeout runs, then it has been closed out before and this is the
|
|
540
|
+
// re-run the parent asks for. A first lock would otherwise train the run on itself.
|
|
541
|
+
//
|
|
542
|
+
// ⚠ THE SEAMS ARE THE REAL FILESYSTEM HERE — this IS the production call site, so the write and
|
|
543
|
+
// read seams that the tests inject are bound to the actual run root, and the registry read goes
|
|
544
|
+
// to `memory/MEMORY.md` under the same root.
|
|
545
|
+
const isPhase8 = phase.startsWith('08') || phase === '08-memory-impact.md'
|
|
546
|
+
const rerun = isPhase8 && existsSync(join(runDir, 'locks', '08-memory-impact.receipt.json'))
|
|
547
|
+
// ⚠ FU-3 — THE DRAIN IS COMPUTED **BEFORE** THE STUB WRITE, and the first version had it after:
|
|
548
|
+
// `closeoutPhase` now THROWS when the artifact is already LOCKED (the FU-8 guard), so a drain placed
|
|
549
|
+
// after that call never ran on exactly the runs that reach closeout twice — the ones that HAVE
|
|
550
|
+
// children to release. Draining is a run-close action and does not depend on scaffolding a stub.
|
|
551
|
+
const children = isPhase8 ? runChildIds(runDir) : []
|
|
552
|
+
drain = isPhase8
|
|
553
|
+
? this.subagentsSeam === null
|
|
554
|
+
? { children: children.length, drained: false, reason: 'no subagents runtime is mounted, so the children could not be drained' }
|
|
555
|
+
: await drainContinuableChildren(this.subagentsSeam, agent as never, children)
|
|
556
|
+
.then(() => ({ children: children.length, drained: true }))
|
|
557
|
+
.catch((err: unknown) => ({ children: children.length, drained: false, reason: err instanceof Error ? err.message : String(err) }))
|
|
558
|
+
: null
|
|
559
|
+
// ⚠ FU-12 — THE CLOSEOUT NO LONGER WRITES TO THE PHASE DOCUMENT.
|
|
560
|
+
// It READS the artifact, reports what the standard requires and what is missing, and records that
|
|
561
|
+
// examination as a receipt of its own under locks/<stem>.closeout.receipt.json — never over the doc.
|
|
562
|
+
const result = closeoutReport(runDir, phase)
|
|
563
|
+
// P3b: settle this run's injections against its own outcome. Only phases that LOCKED are evidence, and
|
|
564
|
+
// a run settles once, at closeout.
|
|
565
|
+
settleInjections(root, runDir, PHASE_SEQUENCE.filter((file) => getLockStatus(join(runDir, file)) === 'LOCKED'))
|
|
566
|
+
writeCloseoutReceipt(runDir, phase)
|
|
567
|
+
const training = isPhase8
|
|
568
|
+
? runPhase8Trigger(root, runId, {
|
|
569
|
+
rerun,
|
|
570
|
+
// The extractor is resolved but NOT spawned here: this plugin never embeds one, and an
|
|
571
|
+
// unset command must surface as exit 2 rather than as a silent success.
|
|
572
|
+
extractorAvailable: resolveExtractor(process.env) !== null,
|
|
573
|
+
// FU-5: the PRODUCTION runner — spawn the command with `stdio: 'ignore'` and read the file it
|
|
574
|
+
// was asked to write. `stdio: 'ignore'` is deliberate: this sandbox denies a child the piped
|
|
575
|
+
// stdio a capture needs, and the response file is the parent's own interface anyway.
|
|
576
|
+
runner: spawnExtractorRunner({ cwd: root, responseFile: join(runDir, 'training-response.json') }),
|
|
577
|
+
write: (relativePath, content) => {
|
|
578
578
|
// T40 - WRITER/READER AGREEMENT. The shards this trigger writes are memory-plane docs, so they
|
|
579
579
|
// belong under .recursive/memory/ where the plane lint, the registry and the phase-8 artifact all
|
|
580
580
|
// look - the seam previously joined them onto the workspace root, landing them OUTSIDE the plane.
|
|
581
|
-
const target = join(root, '.recursive', relativePath)
|
|
582
|
-
mkdirSync(dirname(target), { recursive: true })
|
|
583
|
-
writeFileSync(target, content, 'utf8')
|
|
584
|
-
return relativePath
|
|
585
|
-
},
|
|
586
|
-
readText: (relativePath) => {
|
|
587
|
-
try {
|
|
588
|
-
return readFileSync(join(root, '.recursive', relativePath), 'utf8')
|
|
589
|
-
} catch {
|
|
590
|
-
return null
|
|
591
|
-
}
|
|
592
|
-
},
|
|
593
|
-
})
|
|
594
|
-
: null
|
|
595
|
-
// ⚠ THE SECOND drain ASSIGNMENT WAS REMOVED (FU-12): it re-drained on the success path and
|
|
596
|
-
// overwrote the value the throw path depends on. The assignment above is the one that matters.
|
|
597
|
-
|
|
598
|
-
// ⚠ THE DRAIN HAPPENS BEFORE THE STUB WRITE, and the first version got this wrong: the FU-8 guard
|
|
599
|
-
// THROWS when 08 is already LOCKED, so a drain placed after it never ran on exactly the runs that
|
|
600
|
-
// reach closeout twice — leaking every child of a completed run. Draining is a RUN-CLOSE action and
|
|
601
|
-
// does not depend on scaffolding a stub, so it happens first and is reported on BOTH paths.
|
|
602
|
-
return { closeoutPhase: phase, runId, ...result, ...(training === null ? {} : { training }), ...(drain === null ? {} : { drain }) }
|
|
603
|
-
} catch (err) {
|
|
604
|
-
// The drain result is carried into the failure path too: a refused closeout still tells the caller
|
|
605
|
-
// whether its children were released.
|
|
606
|
-
return { error: (err as Error).message, ...(typeof drain !== 'undefined' && drain !== null ? { drain } : {}) }
|
|
607
|
-
}
|
|
608
|
-
}
|
|
609
|
-
|
|
610
|
-
/**
|
|
611
|
-
* Run-scoped scratchpad access (R5), rooted under the given workspace root.
|
|
612
|
-
*/
|
|
613
|
-
scratchRun(root: string, runId: string, action: string, target: ScratchTarget, content?: string) {
|
|
614
|
-
const runDir = join(root, '.recursive', 'run', runId)
|
|
615
|
-
const runRoot = join(root, '.recursive', 'run')
|
|
616
|
-
if (!runDir.startsWith(runRoot) || !existsSync(runDir)) {
|
|
617
|
-
return { error: 'Run not found in current workspace: ' + runId }
|
|
618
|
-
}
|
|
619
|
-
try {
|
|
620
|
-
if (action === 'read') {
|
|
621
|
-
return { runId, target, action, content: readScratch(runDir, target), path: join(runDir, 'scratch', 'scratch.' + target) }
|
|
622
|
-
}
|
|
623
|
-
if (action === 'write') {
|
|
624
|
-
const path = writeScratch(runDir, target, content ?? '')
|
|
625
|
-
return { runId, target, action, path }
|
|
626
|
-
}
|
|
627
|
-
if (action === 'append') {
|
|
628
|
-
const path = appendScratch(runDir, target, content ?? '')
|
|
629
|
-
return { runId, target, action, path }
|
|
630
|
-
}
|
|
631
|
-
return { error: 'Unsupported scratch action: ' + action + ' (expected read|write|append)' }
|
|
632
|
-
} catch (err) {
|
|
633
|
-
return { error: (err as Error).message }
|
|
634
|
-
}
|
|
635
|
-
}
|
|
636
|
-
|
|
637
|
-
/**
|
|
638
|
-
* Phase B (native delegation): build a review bundle (R1) + file-backed
|
|
639
|
-
* handoff docs (R2), resolve the role via the router policy (R3), and call
|
|
640
|
-
* ctx.subagents with the full request (R4). Workspace-scoped: every path
|
|
641
|
-
* resolves under the session's control-plane root.
|
|
642
|
-
*
|
|
643
|
-
* DELEGATION IS ALWAYS CONTINUABLE (T35). The default mode is `continuable`:
|
|
644
|
-
* an explicit `mode: 'one-shot'` is the ONLY way to give up the repair path,
|
|
645
|
-
* and that path only exists for a caller that genuinely discards the result.
|
|
646
|
-
*
|
|
647
|
-
* WHY THIS IS THE DEFAULT. A one-shot child is NOT resumable — the harness
|
|
648
|
-
* rejects a resume with "subagent cannot be resumed" — so choosing one-shot
|
|
649
|
-
* forfeits the ability to send a failed review back to the agent that did the
|
|
650
|
-
* work. A continuable child has ONE durable Session across activations, so a
|
|
651
|
-
* REVISE reaches the SAME child with its working context intact instead of
|
|
652
|
-
* spawning a fresh one that must re-read the whole handoff to rediscover what
|
|
653
|
-
* it already knew. Since verification that cannot be followed by repair is
|
|
654
|
-
* just a complaint, the repair path is the default rather than an option.
|
|
655
|
-
*
|
|
656
|
-
* `mode: 'continuable'` (T4) runs the audit→repair→re-audit loop on ONE
|
|
657
|
-
* durable continuable child (startContinuable → followup with the repair
|
|
658
|
-
* instruction → settle) and drains the child on closeout. It requires an
|
|
659
|
-
* `awaitRoundResult` observer (the parent-side settlement seam) AND the exact
|
|
660
|
-
* live `parent` Agent (continuable followup is object-identity authority);
|
|
661
|
-
* when either is absent it falls back to one-shot `delegate()` with a flag —
|
|
662
|
-
* never silently. One-shot `start()` is never called on the continuable path.
|
|
663
|
-
*/
|
|
664
|
-
async delegateReview(input: {
|
|
665
|
-
root: string
|
|
666
|
-
runId: string
|
|
667
|
-
phase: string
|
|
668
|
-
role: string
|
|
669
|
-
delegationId: string
|
|
670
|
-
childId: string
|
|
671
|
-
artifactPath: string
|
|
672
|
-
upstreamArtifacts: string[]
|
|
673
|
-
auditQuestions: string[]
|
|
674
|
-
requiredOutput: string
|
|
675
|
-
codeRefs?: string[]
|
|
676
|
-
changedFiles?: string[]
|
|
677
|
-
diffBasis?: ReviewBundleInput['diffBasis']
|
|
678
|
-
policyPath?: string
|
|
679
|
-
providers?: Record<string, SubagentProviderLike>
|
|
680
|
-
subagents?: SubagentsRuntimeLike
|
|
681
|
-
maxDepth?: number
|
|
682
|
-
toolFilter?: unknown
|
|
683
|
-
/**
|
|
684
|
-
* ⚠ FU-17 — THE BRIEF SLICE, WHEN THE CALLER OWNS IT.
|
|
685
|
-
*
|
|
686
|
-
* A review's slice is written here because a reviewer's briefing is review-shaped by definition. A WORK
|
|
687
|
-
* delegation needs the opposite kind of briefing — what to produce and the standard it will be linted
|
|
688
|
-
* against — and that is computed by `buildWorkSlice` from the phase rules. This seam lets a work caller pass
|
|
689
|
-
* it in without this method growing a second, drifting copy of the phase standard.
|
|
690
|
-
*
|
|
691
|
-
* ADDITIVE BY CONSTRUCTION: absent, the review slice below is built exactly as it always was, which is what
|
|
692
|
-
* keeps the review path's behaviour provable rather than merely claimed.
|
|
693
|
-
*/
|
|
694
|
-
slice?: string
|
|
695
|
-
/**
|
|
696
|
-
* ⚠ FU-17 — whether this delegation is WORK or a REVIEW. It changes two things and nothing else: the slice
|
|
697
|
-
* (when `slice` is passed) and the `Purpose` line of the action record, so a reader of the run can tell a
|
|
698
|
-
* child that produced something from a child that judged something.
|
|
699
|
-
*/
|
|
700
|
-
kind?: 'review' | 'work'
|
|
701
|
-
/**
|
|
702
|
-
* T35: which child lifecycle to use. DEFAULT `continuable` — a one-shot
|
|
703
|
-
* child cannot be resumed, so one-shot forfeits the repair path and must be
|
|
704
|
-
* requested explicitly by a caller that will discard the result.
|
|
705
|
-
*/
|
|
706
|
-
/**
|
|
707
|
-
* ⚠ FU-18 — IS THIS ROUND A CONTINUATION OF AN OPEN DELEGATION RATHER THAN A FRESH ONE?
|
|
708
|
-
*
|
|
709
|
-
* Set by a caller that is resuming the same child to deliver FEEDBACK. It is the difference between "run this
|
|
710
|
-
* operation again", which the T19 guard exists to refuse, and "carry on with the operation that is already
|
|
711
|
-
* open", which the guard must not refuse — see the guard below for why the obvious alternative is worse.
|
|
712
|
-
*/
|
|
713
|
-
continuing?: boolean
|
|
714
|
-
/**
|
|
715
|
-
* ⚠ FU-19 — A PER-CALL CHOICE, which is what "change them on demand" means. Both win over every configured
|
|
716
|
-
* level, and both are optional: absent means "resolve the ladder", NOT "clear".
|
|
717
|
-
*/
|
|
718
|
-
providerOverride?: string | null
|
|
719
|
-
modelOverride?: string | null
|
|
720
|
-
mode?: 'one-shot' | 'continuable'
|
|
721
|
-
awaitRoundResult?: (childId: ContinuableChildId, messageId: ContinuableMessageId) => Promise<SubagentResultLike | null>
|
|
722
|
-
/**
|
|
723
|
-
* T39: interrupt the LIVE child when this delegation's job is killed — the one thing T10's
|
|
724
|
-
* synchronous call sites cannot do, and the reason a delegation's kill can be genuinely
|
|
725
|
-
* pre-emptive. Optional: without it a kill still parks the round and cannot reach the
|
|
726
|
-
* child, which is stated rather than implied.
|
|
727
|
-
*/
|
|
728
|
-
interrupt?: (childId: string, reason: string) => void
|
|
729
|
-
maxRounds?: number
|
|
730
|
-
/** T4: the exact live direct-parent Agent (object-identity authority). */
|
|
731
|
-
parent?: SubagentParentHandle
|
|
732
|
-
}) {
|
|
733
|
-
const policy = loadRouterPolicy(input.policyPath ?? routerPolicyPath(input.root), this._routerOverrides)
|
|
734
|
-
// ⚠ FU-9 — THE ROUTER GETS THE SAME SEAM FALLBACK THE DELEGATION ONE LINE BELOW ALREADY HAD, and not
|
|
735
|
-
// getting it is why a live review self-audited for five rounds of investigation. `resolveRole` tries
|
|
736
|
-
// `[role, 'spawn', 'fork', 'dsh-sdk']` against this map and returns the NATIVE tier for the first name it
|
|
737
|
-
// finds; handed `{}` it fell through to an external CLI the policy leaves null and then to the policy
|
|
738
|
-
// fallback, and its message named none of that. The router already preferred native — it was never given a
|
|
739
|
-
// name to prefer. `SubagentProviderLike` is only a descriptor (`{ name, capabilities? }`), so a provider is
|
|
740
|
-
// registered by ASKING the service for it; a service that cannot enumerate yields an empty map and the old
|
|
741
|
-
// behaviour, which is the correct degradation rather than a guess.
|
|
742
|
-
const providers = input.providers ?? this.providerMapFromSeam()
|
|
743
|
-
// T39: the seam the caller passed, else the one the COMPOSITION mounted. See the field's
|
|
744
|
-
// comment: without this the review tool's rounds silently self-audited on a host that had
|
|
745
|
-
// the service all along.
|
|
746
|
-
const subagents: SubagentsRuntimeLike | undefined = input.subagents ?? this.subagentsSeam ?? undefined
|
|
747
|
-
const decision = resolveRole(input.role, policy, providers)
|
|
748
|
-
const probe = capabilityProbe({ providers, role: input.role, policy })
|
|
749
|
-
|
|
750
|
-
// R1 bundle + R2 handoff/brief/prompt (file-backed context-in contract).
|
|
751
|
-
// T14 — PRIOR-RUN MEMORY, retrieved by relevance to THIS phase/artifact/role and written into
|
|
752
|
-
// the bundle. Two facts decided the shape: there is NO native memory service (measured), so the
|
|
753
|
-
// store is the plugin's own `.recursive/memory/` layer; and the prompt references the bundle by
|
|
754
|
-
// PATH, so the content must go INTO the bundle rather than beside the prompt.
|
|
755
|
-
//
|
|
756
|
-
// `memoryRefs` is filled as well as the content, and it is NOT a duplicate: it was a DEAD SLOT —
|
|
757
|
-
// rendered by the bundle and set by nobody — so the memory section a reviewer was meant to see
|
|
758
|
-
// never existed. Refs give traceability; the content is what a reviewer can actually cite.
|
|
759
|
-
const memoryEntries = retrieveMemory(
|
|
760
|
-
readMemoryEntries(
|
|
761
|
-
(path) => { try { return readFileSync(path, 'utf8') } catch { return null } },
|
|
762
|
-
(kind) => {
|
|
763
|
-
try {
|
|
764
|
-
return readdirSync(join(input.root, '.recursive', 'memory', kind))
|
|
765
|
-
.filter((name) => name.endsWith('.md'))
|
|
766
|
-
.map((name) => join(input.root, '.recursive', 'memory', kind, name))
|
|
767
|
-
} catch {
|
|
768
|
-
// An absent or unreadable memory directory is a missing advantage, not a failed review.
|
|
769
|
-
return []
|
|
770
|
-
}
|
|
771
|
-
},
|
|
772
|
-
),
|
|
773
|
-
[input.phase, input.role, ...input.auditQuestions].join(' '),
|
|
774
|
-
)
|
|
775
|
-
const bundle = buildReviewBundle({
|
|
776
|
-
root: input.root,
|
|
777
|
-
runId: input.runId,
|
|
778
|
-
phase: input.phase,
|
|
779
|
-
role: input.role,
|
|
780
|
-
artifactPath: input.artifactPath,
|
|
781
|
-
upstreamArtifacts: input.upstreamArtifacts,
|
|
782
|
-
auditQuestions: input.auditQuestions,
|
|
783
|
-
requiredOutput: input.requiredOutput,
|
|
784
|
-
codeRefs: input.codeRefs,
|
|
785
|
-
changedFiles: input.changedFiles,
|
|
786
|
-
diffBasis: input.diffBasis,
|
|
787
|
-
...(memoryEntries.length === 0 ? {} : {
|
|
788
|
-
memory: renderMemorySection(memoryEntries),
|
|
789
|
-
memoryRefs: [...new Set(memoryEntries.map((entry) => entry.source))],
|
|
790
|
-
}),
|
|
791
|
-
})
|
|
792
|
-
const handoffPath = createHandoff({
|
|
793
|
-
root: input.root,
|
|
794
|
-
runId: input.runId,
|
|
795
|
-
delegationId: input.delegationId,
|
|
796
|
-
role: input.role,
|
|
797
|
-
objective: input.requiredOutput,
|
|
798
|
-
runDocRefs: input.upstreamArtifacts,
|
|
799
|
-
codeRefs: input.codeRefs ?? [],
|
|
800
|
-
auditQuestions: input.auditQuestions,
|
|
801
|
-
requiredOutput: input.requiredOutput,
|
|
802
|
-
decisionBasis: decision.reason,
|
|
803
|
-
constraints: [
|
|
804
|
-
'Workspace-scoped: never read another workspace\'s .recursive/ tree.',
|
|
805
|
-
'Optionality: if the probe fails, fall back to self-audit — never weaken the audit.',
|
|
806
|
-
'Fail loud: capability mismatches reject; do not silently degrade.',
|
|
807
|
-
],
|
|
808
|
-
})
|
|
809
|
-
const briefPath = createChildBrief({
|
|
810
|
-
root: input.root,
|
|
811
|
-
runId: input.runId,
|
|
812
|
-
delegationId: input.delegationId,
|
|
813
|
-
childId: input.childId,
|
|
814
|
-
// ⚠ FU-17 — `input.slice` wins when a work caller supplies one; otherwise this is byte-for-byte the review
|
|
815
|
-
// slice it has always been. A work brief cannot be built here without a second copy of the phase standard.
|
|
816
|
-
slice: input.slice ?? 'Perform the delegated ' + input.role + ' for run ' + input.runId + ' (' + input.phase + ') and write your submission to reply.md.',
|
|
817
|
-
})
|
|
818
|
-
const prompt = buildDelegationPrompt({
|
|
819
|
-
root: input.root,
|
|
820
|
-
runId: input.runId,
|
|
821
|
-
delegationId: input.delegationId,
|
|
822
|
-
childId: input.childId,
|
|
823
|
-
handoffPath,
|
|
824
|
-
briefPath,
|
|
825
|
-
})
|
|
826
|
-
|
|
827
|
-
// R4: plugin-driven delegation with the full request shape. `parent` is the
|
|
828
|
-
// exact live direct-parent Agent (object-identity authority in the live
|
|
829
|
-
// subagent service); absent it, the live start() rejects the request.
|
|
830
|
-
// T28: the depth budget replaces the hardcoded `?? 2`. The parent's own depth is
|
|
831
|
-
// read structurally (the plugin's seam style) and SUBTRACTED, so the configured
|
|
832
|
-
// ceiling bounds the whole recursion rather than being re-granted at every level
|
|
833
|
-
// — a "depth 2" budget that resets per level bounds nothing.
|
|
834
|
-
// T9: notes about routing decisions that could NOT be applied, so a caller is told rather
|
|
835
|
-
// than left to infer it from a model that silently did not take effect.
|
|
836
|
-
const routingNotes: string[] = []
|
|
837
|
-
const request: SubagentStartRequestLike = {
|
|
838
|
-
prompt: [{ type: 'text', text: prompt }],
|
|
839
|
-
label: input.delegationId + '/' + input.childId,
|
|
840
|
-
outputSchema: reviewOutputSchema(),
|
|
841
|
-
toolFilter: input.toolFilter ?? defaultReviewToolFilter(),
|
|
842
|
-
maxDepth: remainingDepthFor(this.enforcementConfig.budgets, parentDepthOf(input.parent), input.maxDepth),
|
|
843
|
-
}
|
|
844
|
-
if (input.parent !== undefined) request.parent = input.parent
|
|
845
|
-
|
|
846
|
-
// ⚠ FU-19 — THE PROVIDER/MODEL LADDER, APPLIED WHERE THE PROVIDER SAYS IT CAN BE.
|
|
847
|
-
//
|
|
848
|
-
// The ladder (per-call → phase → role → general default) is resolved by `resolveSubagentTarget`, which also
|
|
849
|
-
// reports WHICH LEVEL chose each value — so "why did this child run on that model" is answerable from the
|
|
850
|
-
// decision rather than by reading this code. `modelForRole` was the single-level version of this and is gone.
|
|
851
|
-
//
|
|
852
|
-
// ⚠ THE CAPABILITY GATE STAYS AUTHORITATIVE, unchanged: the harness REJECTS a start that sends `agentOptions`
|
|
853
|
-
// to a provider without the `agentOptions` capability, so gating on the model alone would BREAK delegations on
|
|
854
|
-
// providers that do not accept overrides — a routing feature that takes down the delegation it was meant to
|
|
855
|
-
// improve. A choice that cannot be honoured is REPORTED, never silently dropped.
|
|
856
|
-
const target = resolveSubagentTarget({
|
|
857
|
-
role: input.role,
|
|
858
|
-
phase: input.phase,
|
|
859
|
-
policy,
|
|
860
|
-
...(input.providerOverride === undefined && input.modelOverride === undefined
|
|
861
|
-
? {}
|
|
862
|
-
: { override: { provider: input.providerOverride ?? null, model: input.modelOverride ?? null } }),
|
|
863
|
-
ladderProvider: decision.provider ?? null,
|
|
864
|
-
})
|
|
865
|
-
if (target.provider !== null && target.provider !== decision.provider) {
|
|
866
|
-
// The user named a provider. The router still says WHICH TIER it resolved, but the name used to create the
|
|
867
|
-
// child is the user's — and the note states both rather than leaving two answers in the record.
|
|
868
|
-
routingNotes.push(
|
|
869
|
-
'provider ' + target.provider + ' chosen from ' + target.chosen.provider
|
|
870
|
-
+ ' (the router tier resolved as ' + decision.tier + ')',
|
|
871
|
-
)
|
|
872
|
-
}
|
|
873
|
-
if (target.model !== null) {
|
|
874
|
-
const chosenProvider = target.provider ?? decision.provider ?? ''
|
|
875
|
-
const capable = providers[chosenProvider]?.capabilities?.agentOptions === true
|
|
876
|
-
// ⚠ FU-19 — THE CHOICE IS CHECKED AGAINST WHAT DSH ACTUALLY HAS, and the three-state verdict decides what
|
|
877
|
-
// happens next. `missing` means the inventory ANSWERED and does not have this model: it is NOT applied, and
|
|
878
|
-
// crucially NO OTHER MODEL IS SUBSTITUTED — the child inherits the session default and the note says so.
|
|
879
|
-
// `unverified` (no inventory to ask) still applies the model, because refusing on the strength of a service
|
|
880
|
-
// that merely was not mounted would refuse something the user asked for on no evidence.
|
|
881
|
-
const check = checkModelChoice(await describeInventory(this.llmInventory), target.model)
|
|
882
|
-
if (check.verdict === 'missing') {
|
|
883
|
-
routingNotes.push(
|
|
884
|
-
'model ' + target.model + ' was chosen (from ' + target.chosen.model + '), but ' + check.reason
|
|
885
|
-
+ ' — it was NOT applied and no other model was substituted, so the child inherits the session default',
|
|
886
|
-
)
|
|
887
|
-
} else if (!capable) {
|
|
888
|
-
routingNotes.push(
|
|
889
|
-
'model ' + target.model + ' was chosen (from ' + target.chosen.model + '), but provider '
|
|
890
|
-
+ (chosenProvider || '(none)') + ' does not declare the agentOptions capability, so the model was NOT applied',
|
|
891
|
-
)
|
|
892
|
-
} else {
|
|
893
|
-
request.agentOptions = { model: target.model }
|
|
894
|
-
routingNotes.push('model ' + target.model + ' applied (from ' + target.chosen.model + '); inventory says: ' + check.reason)
|
|
895
|
-
}
|
|
896
|
-
}
|
|
897
|
-
|
|
898
|
-
// T19 — DETERMINISTIC OPERATION IDENTITY for the delegation itself. The id
|
|
899
|
-
// covers the inputs PLUS the artifact's BODY, and the body is what keeps a
|
|
900
|
-
// re-review of a REPAIRED artifact a genuinely different operation rather than a
|
|
901
|
-
// recognised repeat. `childId` is deliberately NOT part of it: a fresh round
|
|
902
|
-
// allocates one at random, so including it would make the id meaningless.
|
|
903
|
-
//
|
|
904
|
-
// The body — not the `LockHash` — for the reason T37 found the hard way: a lock
|
|
905
|
-
// hash covers the wall-clock `LockedAt`, so it changes between two locks of
|
|
906
|
-
// byte-identical content and would make this identity fire only half the time.
|
|
907
|
-
const artifactBody = (() => {
|
|
908
|
-
try {
|
|
909
|
-
return lockHashFromContent(
|
|
910
|
-
readFileSync(input.artifactPath, 'utf8').replace(/^[ \t]*Status:.*$/m, '').replace(/^[ \t]*LockedAt:.*\n?/m, ''),
|
|
911
|
-
)
|
|
912
|
-
} catch {
|
|
913
|
-
return 'absent'
|
|
914
|
-
}
|
|
915
|
-
})()
|
|
916
|
-
const operation = operationId({
|
|
917
|
-
act: 'delegate-review',
|
|
918
|
-
input: {
|
|
919
|
-
runId: input.runId,
|
|
920
|
-
phase: input.phase,
|
|
921
|
-
role: input.role,
|
|
922
|
-
delegationId: input.delegationId,
|
|
923
|
-
body: artifactBody,
|
|
924
|
-
},
|
|
925
|
-
})
|
|
926
|
-
|
|
927
|
-
/**
|
|
928
|
-
* T39 — the PRODUCTION provider of the interrupt seam.
|
|
929
|
-
*
|
|
930
|
-
* The delegation's job kill must reach the LIVE child, and the child is interrupted
|
|
931
|
-
* through the SAME `subagents.interrupt` seam the continuable lifecycle already uses —
|
|
932
|
-
* never a second stop path, so a board kill and a lifecycle stop cannot drift apart. The
|
|
933
|
-
* authority is `{ kind: 'ancestor', agent: parent }`: the parent Agent is the object-
|
|
934
|
-
* identity authority the seam expects, exactly as `followup` uses it.
|
|
935
|
-
*
|
|
936
|
-
* An explicit `input.interrupt` still WINS, so a caller with a better authority (a user-
|
|
937
|
-
* initiated stop, say) can supply one. With no seam and no parent there is nothing to
|
|
938
|
-
* interrupt, and this quietly does nothing rather than throwing: the kill already parked
|
|
939
|
-
* the round, and a kill that cannot reach a child must not become an error in its place.
|
|
940
|
-
*/
|
|
941
|
-
const interruptChild = input.interrupt ?? ((childId: string, reason: string) => {
|
|
942
|
-
const seam = subagents
|
|
943
|
-
if (seam?.interrupt === undefined || input.parent === undefined) return
|
|
944
|
-
try {
|
|
945
|
-
seam.interrupt(childId as ContinuableChildId, { kind: 'ancestor', agent: input.parent })
|
|
946
|
-
} catch {
|
|
947
|
-
// Best-effort, like the call site that uses it.
|
|
948
|
-
}
|
|
949
|
-
})
|
|
950
|
-
|
|
951
|
-
let result: SubagentResultLike | null = null
|
|
952
|
-
let error: string | null = null
|
|
953
|
-
let continuable: ContinuableDelegationLike | null = null
|
|
954
|
-
// T36: a parked round is NOT an error — it means the child has not settled yet.
|
|
955
|
-
// Kept apart from `error` so a turn-shaped caller can resume instead of treating
|
|
956
|
-
// an ordinary wait as a failure of the delegation.
|
|
957
|
-
let parked = false
|
|
958
|
-
let parkedReason: string | null = null
|
|
959
|
-
// The run directory the operation index lives under. Computed once, and the
|
|
960
|
-
// index is skipped entirely when there is no root: `join('', …)` would otherwise
|
|
961
|
-
// resolve a RELATIVE path and could read or write an unrelated directory.
|
|
962
|
-
const operationsDir = input.root ? join(input.root, '.recursive', 'run', input.runId) : ''
|
|
963
|
-
|
|
964
|
-
// T19 — a RECOGNISED repeat is not re-executed. Only an operation that already
|
|
965
|
-
// COMPLETED and was ACCEPTED blocks a retry: a failed or refused attempt must
|
|
966
|
-
// stay retryable, and a repaired artifact is a different operation entirely
|
|
967
|
-
// because its body is part of the id. Without this the same review could spawn a
|
|
968
|
-
// second reviewer for an artifact that has not changed.
|
|
969
|
-
const prior = operationsDir === '' ? null : findOperation(operationsDir, operation)
|
|
970
|
-
// ⚠ FU-18 — A CONTINUATION IS NOT A REPEAT, and the distinction has to live HERE rather than in the operation
|
|
971
|
-
// id. The tempting fix is to fold the round into the id so a feedback round becomes a different operation,
|
|
972
|
-
// and it is the wrong one: T28's children budget below counts DISTINCT OPERATIONS and its own comment says a
|
|
973
|
-
// resumed turn "re-enters here with the same id and must not be counted as another child". Making a
|
|
974
|
-
// continuation a new identity would fix this refusal and silently miscount the budget instead.
|
|
975
|
-
//
|
|
976
|
-
// The collision it resolves: the T19 guard refuses an operation already accepted against an UNCHANGED
|
|
977
|
-
// artifact, which is right in general — re-reviewing an artifact nothing has touched since it passed wastes a
|
|
978
|
-
// child. But when the main agent sends FEEDBACK, the artifact has not changed YET: the child has not repaired
|
|
979
|
-
// it. So the guard refused exactly the call that starts the repair. A continuation is the operation still
|
|
980
|
-
// running, so it is exempt; a genuinely fresh delegation against an unchanged artifact is still refused.
|
|
981
|
-
if (prior?.outcome === 'accepted' && input.continuing !== true) {
|
|
982
|
-
throw new Error(
|
|
983
|
-
'review of ' + input.phase + ' is a recognised repeat of an operation already accepted (operation ' +
|
|
984
|
-
operation + '); the artifact has not changed since it passed',
|
|
985
|
-
)
|
|
986
|
-
}
|
|
987
|
-
|
|
988
|
-
// T28 — THE CHILDREN BUDGET, counted FROM the index rather than tracked beside it,
|
|
989
|
-
// so the bound cannot drift from the operations it bounds. Distinct operations,
|
|
990
|
-
// not records: a resumed turn re-enters here with the same id and must not be
|
|
991
|
-
// counted as another child.
|
|
992
|
-
if (operationsDir !== '') {
|
|
993
|
-
const started = countOperations(operationsDir, 'delegate-review', input.phase)
|
|
994
|
-
if (started >= this.enforcementConfig.budgets.maxChildrenPerPhase) {
|
|
995
|
-
throw new Error(
|
|
996
|
-
'children budget reached: phase ' + input.phase + ' has already started ' + started +
|
|
997
|
-
' delegation(s), and the configured cap is ' + this.enforcementConfig.budgets.maxChildrenPerPhase,
|
|
998
|
-
)
|
|
999
|
-
}
|
|
1000
|
-
}
|
|
1001
|
-
|
|
1002
|
-
if (decision.tier === 'native' || decision.tier === 'external-cli') {
|
|
1003
|
-
if (!subagents) {
|
|
1004
|
-
error = 'no ctx.subagents runtime available (self-audit fallback)'
|
|
1005
|
-
} else if (input.mode !== 'one-shot') {
|
|
1006
|
-
// T35: continuable BY DEFAULT — only an explicit 'one-shot' opts out of
|
|
1007
|
-
// the repair path, because a one-shot child cannot be resumed.
|
|
1008
|
-
continuable = await delegateContinuable({
|
|
1009
|
-
subagents,
|
|
1010
|
-
provider: (target.provider ?? decision.provider ?? '') as string,
|
|
1011
|
-
label: input.delegationId + '/' + input.childId,
|
|
1012
|
-
prompt,
|
|
1013
|
-
childId: input.childId,
|
|
1014
|
-
maxDepth: remainingDepthFor(this.enforcementConfig.budgets, parentDepthOf(input.parent), input.maxDepth),
|
|
1015
|
-
toolFilter: input.toolFilter ?? defaultReviewToolFilter(),
|
|
1016
|
-
maxRounds: input.maxRounds ?? 3,
|
|
1017
|
-
// T9: the model the role resolved to, when the provider accepts overrides. It must be
|
|
1018
|
-
// forwarded THROUGH the continuable delegation — that seam builds the start request
|
|
1019
|
-
// itself, so setting it on `request` alone would never reach the provider.
|
|
1020
|
-
...(request.agentOptions === undefined ? {} : { agentOptions: request.agentOptions }),
|
|
1021
|
-
// T39: the ROUND AWAIT is the hangable part of a delegation — the child may work for
|
|
1022
|
-
// minutes — so it is what the board should be able to see as a running job.
|
|
1023
|
-
//
|
|
1024
|
-
// ⚠ SAFETY BY CONSTRUCTION, which is why this wrapper is safe to add at all: the
|
|
1025
|
-
// seam's contract is `SubagentResultLike | null`, and `delegateContinuable` reads a
|
|
1026
|
-
// null as PARKED. So every non-completed job outcome returns `null` — never a throw
|
|
1027
|
-
// and never an invented result — which means a killed or failed round parks exactly
|
|
1028
|
-
// as an unobserved round does. This wrapper therefore CANNOT convert a parked round
|
|
1029
|
-
// into a settlement, and the property holds without a special case.
|
|
1030
|
-
//
|
|
1031
|
-
// ⚠ MEASURED, and it corrects T39's own plan: `delegateReview` has NO interrupt
|
|
1032
|
-
// seam — its input carries none, and this module has no interrupt path (the audit
|
|
1033
|
-
// loop has one only because IT receives the teams runtime). So the job's cancel
|
|
1034
|
-
// cannot yet pre-empt the child; wiring that needs the seam added here first. The
|
|
1035
|
-
// job still gives the board visibility, and the signal is honoured at the boundary.
|
|
1036
|
-
awaitRoundResult: input.awaitRoundResult === undefined ? undefined : async (childId, messageId) => {
|
|
1037
|
-
const awaited = await runTracked(this.jobs, {
|
|
1038
|
-
kind: 'delegation',
|
|
1039
|
-
label: 'delegation round ' + input.runId + ' ' + input.phase,
|
|
1040
|
-
run: async ({ report, signal }) => {
|
|
1041
|
-
report('awaiting the review round')
|
|
1042
|
-
|
|
1043
|
-
// T39 (part 2) — THE PRE-EMPTIVE KILL. A job kill aborts the signal, and this
|
|
1044
|
-
// listener turns that abort into an INTERRUPT of the LIVE CHILD, so the kill
|
|
1045
|
-
// stops the WORK rather than merely stopping our wait for it. That is the
|
|
1046
|
-
// difference between this call site and T10's synchronous ones, where no such
|
|
1047
|
-
// thing is possible.
|
|
1048
|
-
//
|
|
1049
|
-
// The seam is OPTIONAL and the no-seam case stays honest: without a provider
|
|
1050
|
-
// the kill still parks the round and never throws — it simply cannot reach the
|
|
1051
|
-
// child, and nothing here pretends otherwise.
|
|
1052
|
-
const onAbort = () => {
|
|
1053
|
-
try {
|
|
1054
|
-
interruptChild(input.childId, abortReason(signal))
|
|
1055
|
-
} catch {
|
|
1056
|
-
// Best-effort by design: an interrupt that throws must not replace the
|
|
1057
|
-
// kill's own outcome with an unrelated error.
|
|
1058
|
-
}
|
|
1059
|
-
}
|
|
1060
|
-
if (signal?.aborted === true) onAbort()
|
|
1061
|
-
else signal?.addEventListener('abort', onAbort, { once: true })
|
|
1062
|
-
|
|
1063
|
-
try {
|
|
1064
|
-
// ⚠ FU-9 — THE `?.` IS THE FIX FOR THE WHOLE INVESTIGATION. `signal` is typed as an
|
|
1065
|
-
// `AbortSignal` but a caller can reach here with nothing, and `signal.aborted` on undefined
|
|
1066
|
-
// throws `Cannot read properties of undefined (reading 'aborted')` — which the delegation's
|
|
1067
|
-
// catch recorded as its reason, the driver misread as "no continuable repair path", and which
|
|
1068
|
-
// therefore meant THE CHILD NEVER STARTED. Every artifact this investigation chased — no child
|
|
1069
|
-
// session, no reply, no settlement, four brief-only child directories — was downstream of this
|
|
1070
|
-
// one unguarded property read. An absent signal means "nobody can cancel this", not "crash".
|
|
1071
|
-
if (signal?.aborted === true) throw new Error('the round was cancelled before it was awaited')
|
|
1072
|
-
return await input.awaitRoundResult!(childId, messageId)
|
|
1073
|
-
} finally {
|
|
1074
|
-
signal?.removeEventListener('abort', onAbort)
|
|
1075
|
-
}
|
|
1076
|
-
},
|
|
1077
|
-
})
|
|
1078
|
-
if (awaited.status !== 'completed') return null
|
|
1079
|
-
// T39: record the round where the run id is known — same pattern as the lint, and
|
|
1080
|
-
// for the same reason (an event would carry the job, not the run).
|
|
1081
|
-
recordJobRun(input.root, input.runId, {
|
|
1082
|
-
kind: 'delegation',
|
|
1083
|
-
label: 'delegation round ' + input.runId + ' ' + input.phase,
|
|
1084
|
-
result: awaited,
|
|
1085
|
-
})
|
|
1086
|
-
return awaited.value ?? null
|
|
1087
|
-
},
|
|
1088
|
-
parent: input.parent,
|
|
1089
|
-
})
|
|
1090
|
-
if (continuable.parked === true) {
|
|
1091
|
-
// T36: no settlement has landed for this round yet. The child is still
|
|
1092
|
-
// working (or a repair was just sent), so the caller resumes on a later
|
|
1093
|
-
// turn with the SAME child id. `accepted` stays false, and no `result` is
|
|
1094
|
-
// fabricated — an unobserved round is never an approval.
|
|
1095
|
-
parked = true
|
|
1096
|
-
parkedReason = continuable.reason ?? null
|
|
1097
|
-
} else if (continuable.fellBackToOneShot) {
|
|
1098
|
-
// The seam has no continuable capability — keep the one-shot result.
|
|
1099
|
-
result = continuable.rounds[0]?.result ?? null
|
|
1100
|
-
if (!result) error = 'continuable fallback produced no result'
|
|
1101
|
-
} else if (continuable.ok && continuable.rounds.length > 0) {
|
|
1102
|
-
result = continuable.rounds[continuable.rounds.length - 1].result ?? null
|
|
1103
|
-
if (!result) error = 'continuable child produced no final result'
|
|
1104
|
-
} else {
|
|
1105
|
-
// ⚠ A ROUND THAT SETTLED IS A RESULT, EVEN WHEN IT WAS NOT ACCEPTED — and this branch used to
|
|
1106
|
-
// discard it. The condition above requires `continuable.ok`, so a child that REPORTED and was
|
|
1107
|
-
// refused (`success: false`, or a non-completed stop reason) fell through to here: `result` stayed
|
|
1108
|
-
// null, the action record said "NO SETTLEMENT arrived within the wait", and the child's own stop
|
|
1109
|
-
// reason — the one fact that explains the refusal — was dropped on the floor. It is the same defect
|
|
1110
|
-
// as the parked one, one branch over: an absence asserted where the code had evidence. Keeping the
|
|
1111
|
-
// result is what lets the record say "the delegation returned without acceptance; stop reason error"
|
|
1112
|
-
// instead of blaming a wait that ended perfectly well.
|
|
1113
|
-
result = continuable.rounds[continuable.rounds.length - 1]?.result ?? null
|
|
1114
|
-
if (result === null) error = continuable.reason ?? 'continuable delegation failed'
|
|
1115
|
-
}
|
|
1116
|
-
} else {
|
|
1117
|
-
try {
|
|
1118
|
-
result = await delegate({
|
|
1119
|
-
subagents,
|
|
1120
|
-
provider: (target.provider ?? decision.provider ?? '') as string,
|
|
1121
|
-
request,
|
|
1122
|
-
})
|
|
1123
|
-
} catch (err) {
|
|
1124
|
-
error = (err as Error).message
|
|
1125
|
-
}
|
|
1126
|
-
}
|
|
1127
|
-
} else {
|
|
1128
|
-
error = 'delegation resolved to ' + decision.tier + ' (' + decision.reason + ')'
|
|
1129
|
-
}
|
|
1130
|
-
|
|
1131
|
-
const rawEvaluation = result ? evaluateDelegationResult(result) : { accepted: false, reason: error ?? 'no result' }
|
|
1132
|
-
|
|
1133
|
-
// T8 — THE WIRING BUG THE ITEM'S RESCOPE NAMED. `validateReferences` existed, was
|
|
1134
|
-
// exported, was even reachable through this runtime — and was called by NOTHING on the
|
|
1135
|
-
// delegate path. The module header had always claimed the opposite ("validates the
|
|
1136
|
-
// child's references before writing an action record"), so a reviewer could cite files
|
|
1137
|
-
// that do not exist, or paths that escape the root, and the review was recorded as a
|
|
1138
|
-
// PASS. A claim nobody checks is worse than no claim, because it reads as evidence.
|
|
1139
|
-
//
|
|
1140
|
-
// Placed BEFORE the operation record and the action record, which is what makes it
|
|
1141
|
-
// matter: a review with unverifiable references is never indexed as `accepted`, so the
|
|
1142
|
-
// T19 repeat guard treats it as retryable instead of freezing a bad review in place.
|
|
1143
|
-
//
|
|
1144
|
-
// A delegation that claims NOTHING is not checked — refusing an empty list would be a
|
|
1145
|
-
// policy change (is a reference-free review invalid?) rather than a wiring fix, and it
|
|
1146
|
-
// is recorded here as a deliberate boundary rather than an oversight.
|
|
1147
|
-
const claims = result ? referencesFromResult(result) : []
|
|
1148
|
-
const referenceCheck = claims.length === 0 ? null : validateReferences(input.root, claims)
|
|
1149
|
-
const evaluation = referenceCheck === null || referenceCheck.ok
|
|
1150
|
-
? rawEvaluation
|
|
1151
|
-
: { accepted: false, reason: 'review references failed: ' + referenceCheck.failures.join('; ') }
|
|
1152
|
-
|
|
1153
|
-
// T19 — persist the attempt so a RESTART can match it. Only an accepted review is
|
|
1154
|
-
// recorded as `accepted`, which is what makes the recognition above block a
|
|
1155
|
-
// genuine repeat while leaving every failure path retryable. Best-effort: a
|
|
1156
|
-
// failed index write must never change the outcome of a review that already ran.
|
|
1157
|
-
if (operationsDir !== '') {
|
|
1158
|
-
recordOperation(operationsDir, {
|
|
1159
|
-
id: operation,
|
|
1160
|
-
act: 'delegate-review',
|
|
1161
|
-
at: new Date().toISOString().replace(/\.\d{3}Z$/, 'Z'),
|
|
1162
|
-
// ⚠ A PARK IS NOT A REFUSAL, and `unaccepted` said it was. This line used to be
|
|
1163
|
-
// `accepted ? 'accepted' : 'unaccepted'`, so a round that had merely not settled yet was indexed
|
|
1164
|
-
// exactly like a delegation that was evaluated and refused — while the round it really was (still in
|
|
1165
|
-
// flight, resume the same child) was nowhere in the run's own operation log. The defect the action
|
|
1166
|
-
// record had, the log had too. The new value is honest on both readings that matter: it is not
|
|
1167
|
-
// `accepted`, so every retry gate still treats the operation as unfinished and retryable — which is
|
|
1168
|
-
// what a parked round is — and it no longer claims the delegation was judged and rejected.
|
|
1169
|
-
outcome: parked ? 'parked' : (evaluation.accepted ? 'accepted' : 'unaccepted'),
|
|
1170
|
-
phase: input.phase,
|
|
1171
|
-
})
|
|
1172
|
-
}
|
|
1173
|
-
|
|
1174
|
-
// R6: record the attempt as an action record (accepted only if evaluation passes).
|
|
1175
|
-
const actionRecordPath = writeActionRecord({
|
|
1176
|
-
root: input.root,
|
|
1177
|
-
runId: input.runId,
|
|
1178
|
-
subagentId: input.childId,
|
|
1179
|
-
phase: input.phase,
|
|
1180
|
-
// ⚠ FU-17 — the kind is stated in the record. `Status` says accepted, failed or parked; nothing said whether
|
|
1181
|
-
// the child PRODUCED the phase's work or JUDGED it, and a reader of a run could not tell the two apart.
|
|
1182
|
-
purpose: input.role + (input.kind === 'work' ? ' (work)' : '') + ' for run ' + input.runId,
|
|
1183
|
-
executionMode: decision.tier + (input.mode !== 'one-shot' ? ' (continuable)' : ''),
|
|
1184
|
-
artifactPath: input.artifactPath,
|
|
1185
|
-
upstreamArtifacts: input.upstreamArtifacts,
|
|
1186
|
-
reviewBundle: bundle.repoRelativePath,
|
|
1187
|
-
diffBasis: input.diffBasis?.normalizedDiffCommand,
|
|
1188
|
-
codeRefs: input.codeRefs,
|
|
1189
|
-
auditQuestions: input.auditQuestions,
|
|
1190
|
-
findings: evaluation.accepted && result?.structured ? [(result.structured as { verdict?: string })?.verdict ?? 'accepted'] : undefined,
|
|
1191
|
-
success: evaluation.accepted,
|
|
1192
|
-
stopReason: result?.stopReason,
|
|
1193
|
-
// ⚠ A PARKED ROUND IS RECORDED AS PARKED — the whole defect in one field. `success: evaluation.accepted`
|
|
1194
|
-
// is false for a park (correct: nothing was accepted), and `writeActionRecord` reads this flag to state
|
|
1195
|
-
// the third state instead of collapsing it into `failed`.
|
|
1196
|
-
...(parked ? { parked: true } : {}),
|
|
1197
|
-
// ⚠ FU-9 — AND SAY WHICH KIND OF FAILURE, because the record previously could not. `result == null` means
|
|
1198
|
-
// the provider never produced anything at all (never started, or returned nothing) — which is what the
|
|
1199
|
-
// live record's `Stop Reason: n/a` was quietly telling me — while a present result that failed to be
|
|
1200
|
-
// accepted means a child DID run and its work was refused. Different problems, identical artifacts.
|
|
1201
|
-
//
|
|
1202
|
-
// ⚠ AND `parked` IS BRANCHED FIRST. A parked round produced NO result at all — that IS what parking
|
|
1203
|
-
// means — so without this branch first it fell into the `result == null` text below: "the continuable
|
|
1204
|
-
// start was made and NO SETTLEMENT arrived within the wait (the child never reported, or never ran)".
|
|
1205
|
-
// That is the exact false conclusion this fix exists for, and it was reached whatever the record's
|
|
1206
|
-
// Status said. Order is therefore load-bearing here.
|
|
1207
|
-
failure: evaluation.accepted
|
|
1208
|
-
? undefined
|
|
1209
|
-
: parked
|
|
1210
|
-
// ⚠ WHAT IS KNOWN, AND ONLY WHAT IS KNOWN, WITH THE ID THE READER NEEDS TO ACT.
|
|
1211
|
-
//
|
|
1212
|
-
// The text this replaces said "the child never reported, or never ran" — a CONCLUSION drawn from an
|
|
1213
|
-
// ABSENCE, and it was false: a live child went on to complete three review rounds and reply eighteen
|
|
1214
|
-
// minutes later, while the main agent read `Status: failed`, concluded the child was dead, and
|
|
1215
|
-
// obtained its review by other means. So the parked message asserts nothing about the child's state
|
|
1216
|
-
// beyond "no settlement had landed when the wait ended", keeps "may still be working" as the
|
|
1217
|
-
// possibility it is, and NAMES the childId plus the exact next step, because advice to resume is
|
|
1218
|
-
// unactionable without the id. The identity diagnostics stay, because they are what makes a
|
|
1219
|
-
// misconfigured provider readable — but they are diagnostics, not the reason.
|
|
1220
|
-
? 'no settlement had landed when the wait ended, so this round is PARKED, not failed: nothing was'
|
|
1221
|
-
+ ' accepted and nothing was refused, and the child may still be working. The next step is to RESUME'
|
|
1222
|
-
+ ' this round, not to re-dispatch it or replace the child: call `recursive_review` again on a later'
|
|
1223
|
-
+ ' turn with childId ' + String(continuable?.childId ?? input.childId) + ' (the child the round was'
|
|
1224
|
-
+ ' started for, which stays resumable). Diagnostics: tier ' + decision.tier
|
|
1225
|
-
+ ', provider ' + (decision.provider ?? 'none chosen')
|
|
1226
|
-
+ ', names on offer [' + (this.lastProviderNames.join(', ') || 'none') + ']'
|
|
1227
|
-
// ⚠ FU-9 — THE PARENT IDENTITY, because the host refuses a prompt when it cannot resolve the
|
|
1228
|
-
// parent session as a live Agent (`subagent/parent-unavailable`, index.ts L429-436), and the tool
|
|
1229
|
-
// builds this handle with a CAST (`exec.agent as unknown as SubagentParentHandle`). A cast is not
|
|
1230
|
-
// a contract: if the id here is not the one the host looks up, the refusal is real and the
|
|
1231
|
-
// classifier's crash has been hiding it. Printing it here costs nothing and settles the question.
|
|
1232
|
-
+ '; parent id ' + ((input.parent as { id?: string } | undefined)?.id ?? 'none')
|
|
1233
|
-
+ ', parent session keys [' + (input.parent === undefined ? 'no parent' : Object.keys(input.parent as object).join(', ')) + ']'
|
|
1234
|
-
: result == null
|
|
1235
|
-
// ⚠ THE TWO STATES ARE NOT THE SAME AND THE MESSAGE USED TO CONFLATE THEM. The one-shot path cannot
|
|
1236
|
-
// resolve to nothing — the host's `start` returns a run or throws (assertCapabilities, expectProvider)
|
|
1237
|
-
// — so a null result on the CONTINUABLE path means the opposite of what I first wrote: the start WAS
|
|
1238
|
-
// made and NO SETTLEMENT ARRIVED within the wait. That distinction cost me two rounds of looking at
|
|
1239
|
-
// provider names, so the record now states which path a run took and what it was waiting for.
|
|
1240
|
-
// (A park is handled above and never reaches this branch; this one is a continuable round that
|
|
1241
|
-
// produced neither a result nor the park signal, which IS a failure to report.)
|
|
1242
|
-
? (input.mode !== 'one-shot'
|
|
1243
|
-
? 'the continuable start was made and NO SETTLEMENT arrived within the wait (the child never reported,'
|
|
1244
|
-
+ ' or never ran); tier ' + decision.tier + ', provider ' + (decision.provider ?? 'none chosen')
|
|
1245
|
-
+ ', names on offer [' + (this.lastProviderNames.join(', ') || 'none') + ']'
|
|
1246
|
-
// ⚠ FU-9 — THE PARENT IDENTITY, because the host refuses a prompt when it cannot resolve the
|
|
1247
|
-
// parent session as a live Agent (`subagent/parent-unavailable`, index.ts L429-436), and the tool
|
|
1248
|
-
// builds this handle with a CAST (`exec.agent as unknown as SubagentParentHandle`). A cast is not
|
|
1249
|
-
// a contract: if the id here is not the one the host looks up, the refusal is real and the
|
|
1250
|
-
// classifier's crash has been hiding it. Printing it here costs nothing and settles the question.
|
|
1251
|
-
+ '; parent id ' + ((input.parent as { id?: string } | undefined)?.id ?? 'none')
|
|
1252
|
-
+ ', parent session keys [' + (input.parent === undefined ? 'no parent' : Object.keys(input.parent as object).join(', ')) + ']'
|
|
1253
|
-
: 'no delegate result was produced by the one-shot path; tier ' + decision.tier
|
|
1254
|
-
+ ', provider ' + (decision.provider ?? 'none chosen'))
|
|
1255
|
-
: 'the delegation returned without acceptance; stop reason ' + (result.stopReason ?? 'none reported')
|
|
1256
|
-
+ (result.success === false ? ' (the child itself reported success:false)' : ''),
|
|
1257
|
-
})
|
|
1258
|
-
|
|
1259
|
-
// T35: report the mode that ACTUALLY ran, not the one that was asked for. A
|
|
1260
|
-
// missing continuable capability is a real loss — the review can no longer be
|
|
1261
|
-
// sent back to the child that did the work — so it is NAMED rather than
|
|
1262
|
-
// hidden behind a generic success. `continuable-unavailable` is the honest
|
|
1263
|
-
// label for that case: the delegation still happened, the repair path did not.
|
|
1264
|
-
const delegationMode: 'continuable' | 'one-shot' | 'continuable-unavailable' | 'none'
|
|
1265
|
-
= continuable !== null
|
|
1266
|
-
? (continuable.fellBackToOneShot ? 'continuable-unavailable' : 'continuable')
|
|
1267
|
-
: result !== null ? 'one-shot' : 'none'
|
|
1268
|
-
|
|
1269
|
-
// T4: the durable child id is reported for the caller (a tool/closeout that
|
|
1270
|
-
// holds the live parent Agent may drain it explicitly); the HOST owns the
|
|
1271
|
-
// teardown drain (drainContinuableDescendants) at session close — this loop
|
|
1272
|
-
// never forces a drain with a wrong authority credential (childId ≠ parent).
|
|
1273
|
-
return {
|
|
1274
|
-
decision,
|
|
1275
|
-
probe,
|
|
1276
|
-
bundle,
|
|
1277
|
-
handoffPath,
|
|
1278
|
-
briefPath,
|
|
1279
|
-
replyPath: replyPath({ root: input.root, runId: input.runId, delegationId: input.delegationId, childId: input.childId }),
|
|
1280
|
-
childScratchPath: childScratchPath({ root: input.root, runId: input.runId, childId: input.childId }),
|
|
1281
|
-
prompt,
|
|
1282
|
-
request,
|
|
1283
|
-
result,
|
|
1284
|
-
evaluation,
|
|
1285
|
-
/** T9: routing decisions that could NOT be applied, so a caller is told rather than left to infer. */
|
|
1286
|
-
routingNotes,
|
|
1287
|
-
actionRecordPath,
|
|
1288
|
-
error,
|
|
1289
|
-
/** T35: which child lifecycle actually carried this delegation. */
|
|
1290
|
-
delegationMode,
|
|
1291
|
-
/** T19: the deterministic id of this review, persisted so a restart can match it. */
|
|
1292
|
-
operationId: operation,
|
|
1293
|
-
/**
|
|
1294
|
-
* T36: true when the round has NOT settled yet, so the caller resumes with
|
|
1295
|
-
* `continuable.childId` on a later turn. `parkedReason` carries the loop's own
|
|
1296
|
-
* sentence ("the child is still working") without it being an `error`.
|
|
1297
|
-
*/
|
|
1298
|
-
parked,
|
|
1299
|
-
parkedReason,
|
|
1300
|
-
continuable: continuable ? { rounds: continuable.rounds, childId: continuable.childId, fellBackToOneShot: continuable.fellBackToOneShot, parked: continuable.parked === true,
|
|
1301
|
-
// ⚠ FU-9 — `ok` AND `reason` TRAVEL WITH IT. The adapter that builds the review driver's view read only
|
|
1302
|
-
// `rounds`, `childId`, `fellBackToOneShot` and `parked`, so every failure branch's REASON — the whole
|
|
1303
|
-
// point of the field — was discarded one layer above the message that needed it, and `ok: false` was
|
|
1304
|
-
// replaced by a hardcoded `ok: true`. A driver cannot report which branch fired if the branch's own
|
|
1305
|
-
// name never reaches it.
|
|
1306
|
-
ok: continuable.ok, reason: continuable.reason } : null,
|
|
1307
|
-
}
|
|
1308
|
-
}
|
|
1309
|
-
|
|
1310
|
-
/** R6: validate a child's claimed references against actual files. */
|
|
1311
|
-
validateReferences(root: string, references: Reference[]) {
|
|
1312
|
-
return validateReferences(root, references)
|
|
1313
|
-
}
|
|
1314
|
-
|
|
1315
|
-
/** R7: probe availability for a role and render the decision basis prose. */
|
|
1316
|
-
probeDelegation(root: string, role: string, providers: Record<string, SubagentProviderLike> = {}) {
|
|
1317
|
-
const policy = loadRouterPolicy(routerPolicyPath(root), this._routerOverrides)
|
|
1318
|
-
const decision = resolveRole(role, policy, providers)
|
|
1319
|
-
const probe = capabilityProbe({ providers, role, policy })
|
|
1320
|
-
return {
|
|
1321
|
-
decision,
|
|
1322
|
-
probe,
|
|
1323
|
-
basis: delegationDecisionBasis({ role, available: probe.available, provider: probe.provider, fallback: 'self-audit' }),
|
|
1324
|
-
}
|
|
1325
|
-
}
|
|
1326
|
-
|
|
1327
|
-
/**
|
|
1328
|
-
* B3: per-call workspace root resolution. The control-plane root is the
|
|
1329
|
-
* session's cwd (or registry-canonicalized), NEVER process.cwd() — the host
|
|
1330
|
-
* checkout is not the run's workspace. Reads resolve under that root only.
|
|
1331
|
-
*/
|
|
1332
|
-
async resolveRootFor(agent?: { session?: { header?: { cwd?: string } } } | null): Promise<string | null> {
|
|
1333
|
-
return resolveControlPlaneRoot(agent, this.workspaceRegistry, this.repoRoot)
|
|
1334
|
-
}
|
|
1335
|
-
|
|
1336
|
-
/**
|
|
1337
|
-
* SP2 R1 route adapter: resolve the control-plane root for the live route.
|
|
1338
|
-
* sessionId PRIMARY — the host looks up the attached session header cwd and
|
|
1339
|
-
* resolves the root from THAT; the client-passed cwd is a fallback hint only
|
|
1340
|
-
* (hydration / headless callers). Two sessions in one workspace collapse to
|
|
1341
|
-
* one root; a subdir cwd resolves up to the workspace root.
|
|
1342
|
-
*/
|
|
1343
|
-
async resolveRootForRoute(sessionId: string | undefined, cwd: string, sessionsStore?: { get?: (id: string) => { header?: { cwd?: string } } | undefined } | null): Promise<string | null> {
|
|
1344
|
-
// 1) Attached session header (authoritative when present).
|
|
1345
|
-
if (sessionId && sessionsStore?.get) {
|
|
1346
|
-
const attached = sessionsStore.get(sessionId)
|
|
1347
|
-
const headerCwd = attached?.header?.cwd
|
|
1348
|
-
if (headerCwd) {
|
|
1349
|
-
const root = await resolveControlPlaneRoot({ session: { header: { cwd: headerCwd } } }, this.workspaceRegistry, this.repoRoot)
|
|
1350
|
-
if (root) return root
|
|
1351
|
-
}
|
|
1352
|
-
}
|
|
1353
|
-
// 2) Client cwd fallback (hydration / headless) — still registry-canonicalized.
|
|
1354
|
-
if (cwd) {
|
|
1355
|
-
const root = await resolveControlPlaneRoot({ session: { header: { cwd } } }, this.workspaceRegistry, this.repoRoot)
|
|
1356
|
-
if (root) return root
|
|
1357
|
-
}
|
|
1358
|
-
return null
|
|
1359
|
-
}
|
|
1360
|
-
|
|
1361
|
-
async status(runId?: string, agent?: { session?: { header?: { cwd?: string } } } | null): Promise<RecursiveStatusWithGuardDecisions | null> {
|
|
1362
|
-
const root = await this.resolveRootFor(agent)
|
|
1363
|
-
if (!root) return null
|
|
1364
|
-
const resolved = resolveRunDir(root, runId)
|
|
1365
|
-
if (!resolved) return null
|
|
1366
|
-
// T15 (G): surface the rolling guard-decision evidence BESIDE the fold, spread
|
|
1367
|
-
// over it — foldRun's own output shape is parity-asserted (status.parity) and
|
|
1368
|
-
// must not change. Scoped to the resolved run so the field answers "what did
|
|
1369
|
-
// the guard decide about THIS run", newest first.
|
|
1370
|
-
const guardDecisions = readGuardDecisions(root, GUARD_DECISION_READ_LIMIT)
|
|
1371
|
-
.filter((d) => d.runId === resolved.runId)
|
|
1372
|
-
.slice(0, GUARD_DECISION_SURFACE_LIMIT)
|
|
1373
|
-
// T18: the pending set rides beside the fold for the same reason — foldRun's
|
|
1374
|
-
// shape is parity-asserted. Always present (empty when nothing is in flight)
|
|
1375
|
-
// so a consumer needs no null dance, and DERIVED on every call rather than
|
|
1376
|
-
// stored, so it cannot go stale.
|
|
1377
|
-
//
|
|
1378
|
-
// T32: the receipt-chain verdict rides here too. A chain that has been edited or
|
|
1379
|
-
// spliced must be VISIBLE on the status a caller actually reads, not only inside
|
|
1380
|
-
// a test — that was the whole finding: the mechanism was written and read by
|
|
1381
|
-
// nothing. Read-only, so asking cannot change the answer.
|
|
1382
|
-
return {
|
|
1383
|
-
...foldRun(resolved.runDir, resolved.runId),
|
|
1384
|
-
guardDecisions,
|
|
1385
|
-
pendingWork: pendingWork(resolved.runDir),
|
|
1386
|
-
receiptChain: validateReceiptChain(resolved.runDir, resolved.runId),
|
|
1387
|
-
// T22: the same identifier the prompt's stable prefix carries, so a reader can compare status
|
|
1388
|
-
// against prompt without re-rendering the section.
|
|
1389
|
-
contractDigest: contractDigest(this.enforcementConfig),
|
|
1390
|
-
}
|
|
1391
|
-
}
|
|
1392
|
-
|
|
1393
|
-
/**
|
|
1394
|
-
* LIVE BUG 6 refined: structured phase rules for the CURRENT phase. Resolves
|
|
1395
|
-
* the workspace root (same as status/lock), finds the latest run (or the
|
|
1396
|
-
* given runId), advances via getNextLegalPhase, and returns the phase's lint
|
|
1397
|
-
* rules + instructions. Returns null when no active phase exists. This is the
|
|
1398
|
-
* canonical data source for the recursive_phase tool.
|
|
1399
|
-
*/
|
|
1400
|
-
async phaseRules(runId?: string, agent?: { session?: { header?: { cwd?: string } } } | null, files?: readonly string[]): Promise<(PhaseRules & { runId: string; phase: string; memory: string; memoryReason: string }) | null> {
|
|
1401
|
-
const root = await this.resolveRootFor(agent)
|
|
1402
|
-
if (!root) return null
|
|
1403
|
-
const resolved = resolveRunDir(root, runId)
|
|
1404
|
-
if (!resolved) return null
|
|
1405
|
-
const phase = getNextLegalPhase(resolved.runDir)
|
|
1406
|
-
if (!phase) return null
|
|
1407
|
-
// T29 — MEMORY AT RUN ENTRY, ON THIS CALL AND NOWHERE ELSE.
|
|
1408
|
-
//
|
|
1409
|
-
// ⚠ THE ONCE-GATE IS THIS FUNCTION, not a mechanism added beside it: `recursive_phase` calls it
|
|
1410
|
-
// once per phase entry, so riding the injection on its EXISTING return is what keeps it
|
|
1411
|
-
// once-per-run instead of once-per-turn. A second dedupe would be a second thing to get wrong.
|
|
1412
|
-
//
|
|
1413
|
-
// ⚠ AND NOTHING RELEVANT INJECTS NOTHING — `memory` is the empty string, with the reason saying
|
|
1414
|
-
// why, rather than a section that fabricates relevance the plane does not have.
|
|
1415
|
-
const requirements = (() => {
|
|
1416
|
-
try {
|
|
1417
|
-
return readFileSync(join(resolved.runDir, '00-requirements.md'), 'utf8')
|
|
1418
|
-
} catch {
|
|
1419
|
-
// A run with no requirements yet has a weaker query, not an error: the paths still count.
|
|
1420
|
-
return ''
|
|
1421
|
-
}
|
|
1422
|
-
})()
|
|
1423
|
-
// `phase` IS the artifact file name (`getNextLegalPhase` returns one of PHASE_SEQUENCE), so the gate
|
|
1424
|
-
// lookup needs no mapping — and the artifact's own text is what says whether it is still owed.
|
|
1425
|
-
const pending = pendingGateFor(phase, (() => {
|
|
1426
|
-
try {
|
|
1427
|
-
return readFileSync(join(resolved.runDir, phase), 'utf8')
|
|
1428
|
-
} catch {
|
|
1429
|
-
return null
|
|
1430
|
-
}
|
|
1431
|
-
})())
|
|
1432
|
-
const selection = selectMemory(root, {
|
|
1433
|
-
query: requirements.slice(0, 4000),
|
|
1434
|
-
// FU-4: the run's OWN changed paths, computed when the caller supplies none — so T29's path
|
|
1435
|
-
// weighting is fed by a real run rather than only by tests. `[]` from a non-git root is fine: the
|
|
1436
|
-
// query still ranks, and a memory hint must never be why a phase call fails.
|
|
1437
|
-
files: files ?? changedPaths(root),
|
|
1438
|
-
// P2: the phase in play, so an entry declaring it applies here outranks general guidance.
|
|
1439
|
-
// P3b: the counters, read ONCE here and handed to the ranking — the book is evidence about retrieval,
|
|
1440
|
-
// and where it lives is the caller's business, not the ranker's.
|
|
1441
|
-
feedback: readFeedback(root),
|
|
1442
|
-
phase,
|
|
1443
|
-
})
|
|
1444
|
-
// P3b: record what this phase was shown, so the loop has evidence to settle at closeout.
|
|
1445
|
-
recordInjection(resolved.runDir, selection.shards.map((shard) => ({
|
|
1446
|
-
source: shard.entry.source,
|
|
1447
|
-
title: shard.entry.title,
|
|
1448
|
-
score: shard.score,
|
|
1449
|
-
})), phase)
|
|
1450
|
-
|
|
1451
|
-
|
|
1452
|
-
|
|
1453
|
-
|
|
1454
|
-
|
|
1455
|
-
|
|
1456
|
-
|
|
1457
|
-
|
|
1458
|
-
|
|
1459
|
-
|
|
1460
|
-
|
|
1461
|
-
}
|
|
1462
|
-
|
|
1463
|
-
|
|
1464
|
-
|
|
1465
|
-
|
|
1466
|
-
|
|
1467
|
-
|
|
1468
|
-
|
|
1469
|
-
|
|
1470
|
-
|
|
1471
|
-
|
|
1472
|
-
|
|
1473
|
-
|
|
1474
|
-
|
|
1475
|
-
|
|
1476
|
-
|
|
1477
|
-
|
|
1478
|
-
|
|
1479
|
-
|
|
1480
|
-
|
|
1481
|
-
|
|
1482
|
-
|
|
1483
|
-
|
|
1484
|
-
|
|
1485
|
-
|
|
1486
|
-
|
|
1487
|
-
|
|
1488
|
-
|
|
1489
|
-
|
|
1490
|
-
|
|
1491
|
-
|
|
1492
|
-
|
|
1493
|
-
|
|
1494
|
-
|
|
1495
|
-
|
|
1496
|
-
|
|
1497
|
-
|
|
1498
|
-
|
|
1499
|
-
|
|
1500
|
-
|
|
1501
|
-
|
|
1502
|
-
|
|
1503
|
-
|
|
1504
|
-
|
|
1505
|
-
|
|
1506
|
-
//
|
|
1507
|
-
|
|
1508
|
-
|
|
1509
|
-
|
|
1510
|
-
|
|
1511
|
-
|
|
1512
|
-
|
|
1513
|
-
|
|
1514
|
-
|
|
1515
|
-
|
|
1516
|
-
|
|
1517
|
-
|
|
1518
|
-
|
|
1519
|
-
|
|
1520
|
-
if (
|
|
1521
|
-
|
|
1522
|
-
|
|
1523
|
-
|
|
1524
|
-
|
|
1525
|
-
|
|
1526
|
-
|
|
1527
|
-
|
|
1528
|
-
|
|
1529
|
-
|
|
1530
|
-
const
|
|
1531
|
-
|
|
1532
|
-
|
|
1533
|
-
|
|
1534
|
-
for (const
|
|
1535
|
-
const
|
|
1536
|
-
if (existsSync(
|
|
1537
|
-
|
|
1538
|
-
created.push(
|
|
1539
|
-
}
|
|
1540
|
-
|
|
1541
|
-
//
|
|
1542
|
-
|
|
1543
|
-
|
|
1544
|
-
|
|
1545
|
-
|
|
1546
|
-
|
|
1547
|
-
'
|
|
1548
|
-
'
|
|
1549
|
-
|
|
1550
|
-
|
|
1551
|
-
|
|
1552
|
-
|
|
1553
|
-
|
|
1554
|
-
|
|
1555
|
-
|
|
1556
|
-
|
|
1557
|
-
|
|
1558
|
-
|
|
1559
|
-
|
|
1560
|
-
|
|
1561
|
-
|
|
1562
|
-
|
|
1563
|
-
|
|
1564
|
-
|
|
1565
|
-
|
|
1566
|
-
|
|
1567
|
-
|
|
1568
|
-
|
|
1569
|
-
|
|
1570
|
-
|
|
1571
|
-
|
|
1572
|
-
|
|
1573
|
-
|
|
1574
|
-
|
|
1575
|
-
|
|
1576
|
-
|
|
1577
|
-
|
|
1578
|
-
|
|
1579
|
-
|
|
1580
|
-
|
|
1581
|
-
|
|
1582
|
-
|
|
1583
|
-
|
|
1584
|
-
|
|
1585
|
-
|
|
1586
|
-
|
|
1587
|
-
|
|
1588
|
-
|
|
1589
|
-
|
|
1590
|
-
|
|
1591
|
-
|
|
1592
|
-
|
|
1593
|
-
|
|
1594
|
-
|
|
1595
|
-
|
|
1596
|
-
|
|
1597
|
-
|
|
1598
|
-
|
|
1599
|
-
|
|
1600
|
-
|
|
1601
|
-
}
|
|
1602
|
-
|
|
1603
|
-
|
|
1604
|
-
|
|
1605
|
-
|
|
1606
|
-
|
|
1607
|
-
|
|
1608
|
-
|
|
1609
|
-
|
|
1610
|
-
|
|
1611
|
-
|
|
1612
|
-
|
|
1613
|
-
|
|
1614
|
-
|
|
1615
|
-
|
|
1616
|
-
|
|
1617
|
-
|
|
1618
|
-
|
|
1619
|
-
|
|
1620
|
-
|
|
1621
|
-
*
|
|
1622
|
-
*
|
|
1623
|
-
|
|
1624
|
-
|
|
1625
|
-
|
|
1626
|
-
|
|
1627
|
-
|
|
1628
|
-
|
|
1629
|
-
|
|
1630
|
-
|
|
1631
|
-
|
|
1632
|
-
|
|
1633
|
-
}
|
|
1634
|
-
}
|
|
1635
|
-
|
|
1636
|
-
/**
|
|
1637
|
-
*
|
|
1638
|
-
*
|
|
1639
|
-
*/
|
|
1640
|
-
|
|
1641
|
-
const
|
|
1642
|
-
|
|
1643
|
-
|
|
1644
|
-
|
|
1645
|
-
|
|
1646
|
-
|
|
1647
|
-
|
|
1648
|
-
|
|
1649
|
-
|
|
1650
|
-
|
|
1651
|
-
|
|
1652
|
-
|
|
1653
|
-
|
|
1654
|
-
|
|
1655
|
-
|
|
1656
|
-
|
|
1657
|
-
|
|
1658
|
-
|
|
1659
|
-
|
|
1660
|
-
|
|
1661
|
-
|
|
1662
|
-
|
|
1663
|
-
|
|
1664
|
-
|
|
1665
|
-
|
|
1666
|
-
|
|
1667
|
-
|
|
1668
|
-
|
|
1669
|
-
|
|
1670
|
-
|
|
1671
|
-
|
|
1672
|
-
|
|
1673
|
-
|
|
1674
|
-
|
|
1675
|
-
|
|
1676
|
-
|
|
1677
|
-
//
|
|
1678
|
-
//
|
|
1679
|
-
//
|
|
1680
|
-
//
|
|
1681
|
-
//
|
|
1682
|
-
//
|
|
1683
|
-
|
|
1684
|
-
|
|
1685
|
-
|
|
1686
|
-
|
|
1687
|
-
//
|
|
1688
|
-
//
|
|
1689
|
-
//
|
|
1690
|
-
//
|
|
1691
|
-
//
|
|
1692
|
-
//
|
|
1693
|
-
//
|
|
1694
|
-
|
|
1695
|
-
|
|
1696
|
-
|
|
1697
|
-
|
|
1698
|
-
)
|
|
1699
|
-
|
|
1700
|
-
|
|
1701
|
-
|
|
1702
|
-
|
|
1703
|
-
|
|
1704
|
-
|
|
1705
|
-
|
|
1706
|
-
|
|
1707
|
-
|
|
1708
|
-
|
|
1709
|
-
//
|
|
1710
|
-
|
|
1711
|
-
|
|
1712
|
-
|
|
1713
|
-
|
|
1714
|
-
|
|
1715
|
-
|
|
1716
|
-
|
|
1717
|
-
|
|
1718
|
-
|
|
1719
|
-
|
|
1720
|
-
|
|
1721
|
-
|
|
1722
|
-
|
|
1723
|
-
|
|
1724
|
-
|
|
1725
|
-
|
|
1726
|
-
|
|
1727
|
-
|
|
1728
|
-
|
|
1729
|
-
|
|
1730
|
-
|
|
1731
|
-
|
|
1732
|
-
|
|
1733
|
-
|
|
1734
|
-
|
|
1735
|
-
|
|
1736
|
-
|
|
1737
|
-
|
|
1738
|
-
|
|
1739
|
-
//
|
|
1740
|
-
|
|
1741
|
-
|
|
1742
|
-
|
|
1743
|
-
|
|
1744
|
-
|
|
1745
|
-
|
|
1746
|
-
|
|
1747
|
-
|
|
1748
|
-
|
|
1749
|
-
|
|
1750
|
-
|
|
1751
|
-
|
|
1752
|
-
|
|
1753
|
-
if (
|
|
1754
|
-
|
|
1755
|
-
|
|
1756
|
-
|
|
1757
|
-
|
|
1758
|
-
|
|
1759
|
-
|
|
1760
|
-
|
|
1761
|
-
|
|
1762
|
-
|
|
1763
|
-
|
|
1764
|
-
|
|
1765
|
-
|
|
1766
|
-
|
|
1767
|
-
//
|
|
1768
|
-
//
|
|
1769
|
-
|
|
1770
|
-
|
|
1771
|
-
|
|
1772
|
-
|
|
1773
|
-
//
|
|
1774
|
-
//
|
|
1775
|
-
//
|
|
1776
|
-
//
|
|
1777
|
-
|
|
1778
|
-
|
|
1779
|
-
|
|
1780
|
-
|
|
1781
|
-
|
|
1782
|
-
|
|
1783
|
-
|
|
1784
|
-
|
|
1785
|
-
|
|
1786
|
-
|
|
1787
|
-
|
|
1788
|
-
|
|
1789
|
-
|
|
1790
|
-
|
|
1791
|
-
|
|
1792
|
-
|
|
1793
|
-
|
|
1794
|
-
|
|
1795
|
-
|
|
1796
|
-
|
|
1797
|
-
|
|
1798
|
-
|
|
1799
|
-
|
|
1800
|
-
const
|
|
1801
|
-
|
|
1802
|
-
|
|
1803
|
-
|
|
1804
|
-
|
|
1805
|
-
|
|
1806
|
-
|
|
1807
|
-
|
|
1808
|
-
|
|
1809
|
-
|
|
1810
|
-
|
|
1811
|
-
|
|
1812
|
-
|
|
1813
|
-
|
|
1814
|
-
|
|
1815
|
-
|
|
1816
|
-
|
|
1817
|
-
|
|
1818
|
-
|
|
1819
|
-
|
|
1820
|
-
|
|
1821
|
-
|
|
1822
|
-
|
|
1823
|
-
|
|
1824
|
-
|
|
1825
|
-
|
|
1826
|
-
|
|
1827
|
-
|
|
1828
|
-
|
|
1829
|
-
|
|
1830
|
-
|
|
1831
|
-
|
|
1832
|
-
|
|
1833
|
-
|
|
1834
|
-
|
|
1835
|
-
|
|
1836
|
-
|
|
1837
|
-
|
|
1838
|
-
|
|
1839
|
-
|
|
1840
|
-
|
|
1841
|
-
|
|
1842
|
-
|
|
1843
|
-
|
|
1844
|
-
|
|
1845
|
-
|
|
1846
|
-
|
|
1847
|
-
|
|
1848
|
-
|
|
1849
|
-
|
|
1850
|
-
|
|
1851
|
-
|
|
1852
|
-
|
|
1853
|
-
|
|
1854
|
-
|
|
1855
|
-
|
|
1856
|
-
if (status
|
|
1857
|
-
|
|
1858
|
-
|
|
1859
|
-
|
|
1860
|
-
|
|
1861
|
-
|
|
1862
|
-
|
|
1863
|
-
|
|
1864
|
-
|
|
1865
|
-
|
|
1866
|
-
|
|
1867
|
-
|
|
1868
|
-
|
|
1869
|
-
|
|
1870
|
-
|
|
1871
|
-
|
|
1872
|
-
|
|
1873
|
-
|
|
1874
|
-
|
|
1875
|
-
|
|
1876
|
-
|
|
1877
|
-
|
|
1878
|
-
|
|
1879
|
-
|
|
1880
|
-
|
|
1881
|
-
|
|
1882
|
-
|
|
1883
|
-
|
|
1884
|
-
|
|
1885
|
-
|
|
1886
|
-
|
|
1887
|
-
|
|
1888
|
-
|
|
1889
|
-
|
|
1890
|
-
|
|
1891
|
-
|
|
1892
|
-
|
|
1893
|
-
|
|
1894
|
-
|
|
1895
|
-
|
|
1896
|
-
|
|
1897
|
-
|
|
1898
|
-
|
|
1899
|
-
|
|
1900
|
-
|
|
1901
|
-
|
|
1902
|
-
|
|
1903
|
-
|
|
1904
|
-
|
|
1905
|
-
|
|
1906
|
-
|
|
1907
|
-
|
|
1908
|
-
|
|
1909
|
-
|
|
1910
|
-
|
|
1911
|
-
|
|
1912
|
-
|
|
1913
|
-
|
|
1914
|
-
|
|
1915
|
-
}
|
|
1916
|
-
|
|
1917
|
-
|
|
1918
|
-
|
|
1919
|
-
|
|
1920
|
-
|
|
1921
|
-
|
|
1922
|
-
|
|
1923
|
-
|
|
1924
|
-
|
|
1925
|
-
|
|
1926
|
-
|
|
1927
|
-
|
|
1928
|
-
|
|
1929
|
-
|
|
1930
|
-
const
|
|
1931
|
-
const
|
|
1932
|
-
|
|
1933
|
-
|
|
1934
|
-
|
|
1935
|
-
|
|
1936
|
-
|
|
1937
|
-
|
|
1938
|
-
|
|
1939
|
-
|
|
1940
|
-
|
|
1941
|
-
|
|
1942
|
-
|
|
1943
|
-
|
|
1944
|
-
|
|
1945
|
-
|
|
1946
|
-
|
|
1947
|
-
|
|
1948
|
-
|
|
581
|
+
const target = join(root, '.recursive', relativePath)
|
|
582
|
+
mkdirSync(dirname(target), { recursive: true })
|
|
583
|
+
writeFileSync(target, content, 'utf8')
|
|
584
|
+
return relativePath
|
|
585
|
+
},
|
|
586
|
+
readText: (relativePath) => {
|
|
587
|
+
try {
|
|
588
|
+
return readFileSync(join(root, '.recursive', relativePath), 'utf8')
|
|
589
|
+
} catch {
|
|
590
|
+
return null
|
|
591
|
+
}
|
|
592
|
+
},
|
|
593
|
+
})
|
|
594
|
+
: null
|
|
595
|
+
// ⚠ THE SECOND drain ASSIGNMENT WAS REMOVED (FU-12): it re-drained on the success path and
|
|
596
|
+
// overwrote the value the throw path depends on. The assignment above is the one that matters.
|
|
597
|
+
|
|
598
|
+
// ⚠ THE DRAIN HAPPENS BEFORE THE STUB WRITE, and the first version got this wrong: the FU-8 guard
|
|
599
|
+
// THROWS when 08 is already LOCKED, so a drain placed after it never ran on exactly the runs that
|
|
600
|
+
// reach closeout twice — leaking every child of a completed run. Draining is a RUN-CLOSE action and
|
|
601
|
+
// does not depend on scaffolding a stub, so it happens first and is reported on BOTH paths.
|
|
602
|
+
return { closeoutPhase: phase, runId, ...result, ...(training === null ? {} : { training }), ...(drain === null ? {} : { drain }) }
|
|
603
|
+
} catch (err) {
|
|
604
|
+
// The drain result is carried into the failure path too: a refused closeout still tells the caller
|
|
605
|
+
// whether its children were released.
|
|
606
|
+
return { error: (err as Error).message, ...(typeof drain !== 'undefined' && drain !== null ? { drain } : {}) }
|
|
607
|
+
}
|
|
608
|
+
}
|
|
609
|
+
|
|
610
|
+
/**
|
|
611
|
+
* Run-scoped scratchpad access (R5), rooted under the given workspace root.
|
|
612
|
+
*/
|
|
613
|
+
scratchRun(root: string, runId: string, action: string, target: ScratchTarget, content?: string) {
|
|
614
|
+
const runDir = join(root, '.recursive', 'run', runId)
|
|
615
|
+
const runRoot = join(root, '.recursive', 'run')
|
|
616
|
+
if (!runDir.startsWith(runRoot) || !existsSync(runDir)) {
|
|
617
|
+
return { error: 'Run not found in current workspace: ' + runId }
|
|
618
|
+
}
|
|
619
|
+
try {
|
|
620
|
+
if (action === 'read') {
|
|
621
|
+
return { runId, target, action, content: readScratch(runDir, target), path: join(runDir, 'scratch', 'scratch.' + target) }
|
|
622
|
+
}
|
|
623
|
+
if (action === 'write') {
|
|
624
|
+
const path = writeScratch(runDir, target, content ?? '')
|
|
625
|
+
return { runId, target, action, path }
|
|
626
|
+
}
|
|
627
|
+
if (action === 'append') {
|
|
628
|
+
const path = appendScratch(runDir, target, content ?? '')
|
|
629
|
+
return { runId, target, action, path }
|
|
630
|
+
}
|
|
631
|
+
return { error: 'Unsupported scratch action: ' + action + ' (expected read|write|append)' }
|
|
632
|
+
} catch (err) {
|
|
633
|
+
return { error: (err as Error).message }
|
|
634
|
+
}
|
|
635
|
+
}
|
|
636
|
+
|
|
637
|
+
/**
|
|
638
|
+
* Phase B (native delegation): build a review bundle (R1) + file-backed
|
|
639
|
+
* handoff docs (R2), resolve the role via the router policy (R3), and call
|
|
640
|
+
* ctx.subagents with the full request (R4). Workspace-scoped: every path
|
|
641
|
+
* resolves under the session's control-plane root.
|
|
642
|
+
*
|
|
643
|
+
* DELEGATION IS ALWAYS CONTINUABLE (T35). The default mode is `continuable`:
|
|
644
|
+
* an explicit `mode: 'one-shot'` is the ONLY way to give up the repair path,
|
|
645
|
+
* and that path only exists for a caller that genuinely discards the result.
|
|
646
|
+
*
|
|
647
|
+
* WHY THIS IS THE DEFAULT. A one-shot child is NOT resumable — the harness
|
|
648
|
+
* rejects a resume with "subagent cannot be resumed" — so choosing one-shot
|
|
649
|
+
* forfeits the ability to send a failed review back to the agent that did the
|
|
650
|
+
* work. A continuable child has ONE durable Session across activations, so a
|
|
651
|
+
* REVISE reaches the SAME child with its working context intact instead of
|
|
652
|
+
* spawning a fresh one that must re-read the whole handoff to rediscover what
|
|
653
|
+
* it already knew. Since verification that cannot be followed by repair is
|
|
654
|
+
* just a complaint, the repair path is the default rather than an option.
|
|
655
|
+
*
|
|
656
|
+
* `mode: 'continuable'` (T4) runs the audit→repair→re-audit loop on ONE
|
|
657
|
+
* durable continuable child (startContinuable → followup with the repair
|
|
658
|
+
* instruction → settle) and drains the child on closeout. It requires an
|
|
659
|
+
* `awaitRoundResult` observer (the parent-side settlement seam) AND the exact
|
|
660
|
+
* live `parent` Agent (continuable followup is object-identity authority);
|
|
661
|
+
* when either is absent it falls back to one-shot `delegate()` with a flag —
|
|
662
|
+
* never silently. One-shot `start()` is never called on the continuable path.
|
|
663
|
+
*/
|
|
664
|
+
async delegateReview(input: {
|
|
665
|
+
root: string
|
|
666
|
+
runId: string
|
|
667
|
+
phase: string
|
|
668
|
+
role: string
|
|
669
|
+
delegationId: string
|
|
670
|
+
childId: string
|
|
671
|
+
artifactPath: string
|
|
672
|
+
upstreamArtifacts: string[]
|
|
673
|
+
auditQuestions: string[]
|
|
674
|
+
requiredOutput: string
|
|
675
|
+
codeRefs?: string[]
|
|
676
|
+
changedFiles?: string[]
|
|
677
|
+
diffBasis?: ReviewBundleInput['diffBasis']
|
|
678
|
+
policyPath?: string
|
|
679
|
+
providers?: Record<string, SubagentProviderLike>
|
|
680
|
+
subagents?: SubagentsRuntimeLike
|
|
681
|
+
maxDepth?: number
|
|
682
|
+
toolFilter?: unknown
|
|
683
|
+
/**
|
|
684
|
+
* ⚠ FU-17 — THE BRIEF SLICE, WHEN THE CALLER OWNS IT.
|
|
685
|
+
*
|
|
686
|
+
* A review's slice is written here because a reviewer's briefing is review-shaped by definition. A WORK
|
|
687
|
+
* delegation needs the opposite kind of briefing — what to produce and the standard it will be linted
|
|
688
|
+
* against — and that is computed by `buildWorkSlice` from the phase rules. This seam lets a work caller pass
|
|
689
|
+
* it in without this method growing a second, drifting copy of the phase standard.
|
|
690
|
+
*
|
|
691
|
+
* ADDITIVE BY CONSTRUCTION: absent, the review slice below is built exactly as it always was, which is what
|
|
692
|
+
* keeps the review path's behaviour provable rather than merely claimed.
|
|
693
|
+
*/
|
|
694
|
+
slice?: string
|
|
695
|
+
/**
|
|
696
|
+
* ⚠ FU-17 — whether this delegation is WORK or a REVIEW. It changes two things and nothing else: the slice
|
|
697
|
+
* (when `slice` is passed) and the `Purpose` line of the action record, so a reader of the run can tell a
|
|
698
|
+
* child that produced something from a child that judged something.
|
|
699
|
+
*/
|
|
700
|
+
kind?: 'review' | 'work'
|
|
701
|
+
/**
|
|
702
|
+
* T35: which child lifecycle to use. DEFAULT `continuable` — a one-shot
|
|
703
|
+
* child cannot be resumed, so one-shot forfeits the repair path and must be
|
|
704
|
+
* requested explicitly by a caller that will discard the result.
|
|
705
|
+
*/
|
|
706
|
+
/**
|
|
707
|
+
* ⚠ FU-18 — IS THIS ROUND A CONTINUATION OF AN OPEN DELEGATION RATHER THAN A FRESH ONE?
|
|
708
|
+
*
|
|
709
|
+
* Set by a caller that is resuming the same child to deliver FEEDBACK. It is the difference between "run this
|
|
710
|
+
* operation again", which the T19 guard exists to refuse, and "carry on with the operation that is already
|
|
711
|
+
* open", which the guard must not refuse — see the guard below for why the obvious alternative is worse.
|
|
712
|
+
*/
|
|
713
|
+
continuing?: boolean
|
|
714
|
+
/**
|
|
715
|
+
* ⚠ FU-19 — A PER-CALL CHOICE, which is what "change them on demand" means. Both win over every configured
|
|
716
|
+
* level, and both are optional: absent means "resolve the ladder", NOT "clear".
|
|
717
|
+
*/
|
|
718
|
+
providerOverride?: string | null
|
|
719
|
+
modelOverride?: string | null
|
|
720
|
+
mode?: 'one-shot' | 'continuable'
|
|
721
|
+
awaitRoundResult?: (childId: ContinuableChildId, messageId: ContinuableMessageId) => Promise<SubagentResultLike | null>
|
|
722
|
+
/**
|
|
723
|
+
* T39: interrupt the LIVE child when this delegation's job is killed — the one thing T10's
|
|
724
|
+
* synchronous call sites cannot do, and the reason a delegation's kill can be genuinely
|
|
725
|
+
* pre-emptive. Optional: without it a kill still parks the round and cannot reach the
|
|
726
|
+
* child, which is stated rather than implied.
|
|
727
|
+
*/
|
|
728
|
+
interrupt?: (childId: string, reason: string) => void
|
|
729
|
+
maxRounds?: number
|
|
730
|
+
/** T4: the exact live direct-parent Agent (object-identity authority). */
|
|
731
|
+
parent?: SubagentParentHandle
|
|
732
|
+
}) {
|
|
733
|
+
const policy = loadRouterPolicy(input.policyPath ?? routerPolicyPath(input.root), this._routerOverrides)
|
|
734
|
+
// ⚠ FU-9 — THE ROUTER GETS THE SAME SEAM FALLBACK THE DELEGATION ONE LINE BELOW ALREADY HAD, and not
|
|
735
|
+
// getting it is why a live review self-audited for five rounds of investigation. `resolveRole` tries
|
|
736
|
+
// `[role, 'spawn', 'fork', 'dsh-sdk']` against this map and returns the NATIVE tier for the first name it
|
|
737
|
+
// finds; handed `{}` it fell through to an external CLI the policy leaves null and then to the policy
|
|
738
|
+
// fallback, and its message named none of that. The router already preferred native — it was never given a
|
|
739
|
+
// name to prefer. `SubagentProviderLike` is only a descriptor (`{ name, capabilities? }`), so a provider is
|
|
740
|
+
// registered by ASKING the service for it; a service that cannot enumerate yields an empty map and the old
|
|
741
|
+
// behaviour, which is the correct degradation rather than a guess.
|
|
742
|
+
const providers = input.providers ?? this.providerMapFromSeam()
|
|
743
|
+
// T39: the seam the caller passed, else the one the COMPOSITION mounted. See the field's
|
|
744
|
+
// comment: without this the review tool's rounds silently self-audited on a host that had
|
|
745
|
+
// the service all along.
|
|
746
|
+
const subagents: SubagentsRuntimeLike | undefined = input.subagents ?? this.subagentsSeam ?? undefined
|
|
747
|
+
const decision = resolveRole(input.role, policy, providers)
|
|
748
|
+
const probe = capabilityProbe({ providers, role: input.role, policy })
|
|
749
|
+
|
|
750
|
+
// R1 bundle + R2 handoff/brief/prompt (file-backed context-in contract).
|
|
751
|
+
// T14 — PRIOR-RUN MEMORY, retrieved by relevance to THIS phase/artifact/role and written into
|
|
752
|
+
// the bundle. Two facts decided the shape: there is NO native memory service (measured), so the
|
|
753
|
+
// store is the plugin's own `.recursive/memory/` layer; and the prompt references the bundle by
|
|
754
|
+
// PATH, so the content must go INTO the bundle rather than beside the prompt.
|
|
755
|
+
//
|
|
756
|
+
// `memoryRefs` is filled as well as the content, and it is NOT a duplicate: it was a DEAD SLOT —
|
|
757
|
+
// rendered by the bundle and set by nobody — so the memory section a reviewer was meant to see
|
|
758
|
+
// never existed. Refs give traceability; the content is what a reviewer can actually cite.
|
|
759
|
+
const memoryEntries = retrieveMemory(
|
|
760
|
+
readMemoryEntries(
|
|
761
|
+
(path) => { try { return readFileSync(path, 'utf8') } catch { return null } },
|
|
762
|
+
(kind) => {
|
|
763
|
+
try {
|
|
764
|
+
return readdirSync(join(input.root, '.recursive', 'memory', kind))
|
|
765
|
+
.filter((name) => name.endsWith('.md'))
|
|
766
|
+
.map((name) => join(input.root, '.recursive', 'memory', kind, name))
|
|
767
|
+
} catch {
|
|
768
|
+
// An absent or unreadable memory directory is a missing advantage, not a failed review.
|
|
769
|
+
return []
|
|
770
|
+
}
|
|
771
|
+
},
|
|
772
|
+
),
|
|
773
|
+
[input.phase, input.role, ...input.auditQuestions].join(' '),
|
|
774
|
+
)
|
|
775
|
+
const bundle = buildReviewBundle({
|
|
776
|
+
root: input.root,
|
|
777
|
+
runId: input.runId,
|
|
778
|
+
phase: input.phase,
|
|
779
|
+
role: input.role,
|
|
780
|
+
artifactPath: input.artifactPath,
|
|
781
|
+
upstreamArtifacts: input.upstreamArtifacts,
|
|
782
|
+
auditQuestions: input.auditQuestions,
|
|
783
|
+
requiredOutput: input.requiredOutput,
|
|
784
|
+
codeRefs: input.codeRefs,
|
|
785
|
+
changedFiles: input.changedFiles,
|
|
786
|
+
diffBasis: input.diffBasis,
|
|
787
|
+
...(memoryEntries.length === 0 ? {} : {
|
|
788
|
+
memory: renderMemorySection(memoryEntries),
|
|
789
|
+
memoryRefs: [...new Set(memoryEntries.map((entry) => entry.source))],
|
|
790
|
+
}),
|
|
791
|
+
})
|
|
792
|
+
const handoffPath = createHandoff({
|
|
793
|
+
root: input.root,
|
|
794
|
+
runId: input.runId,
|
|
795
|
+
delegationId: input.delegationId,
|
|
796
|
+
role: input.role,
|
|
797
|
+
objective: input.requiredOutput,
|
|
798
|
+
runDocRefs: input.upstreamArtifacts,
|
|
799
|
+
codeRefs: input.codeRefs ?? [],
|
|
800
|
+
auditQuestions: input.auditQuestions,
|
|
801
|
+
requiredOutput: input.requiredOutput,
|
|
802
|
+
decisionBasis: decision.reason,
|
|
803
|
+
constraints: [
|
|
804
|
+
'Workspace-scoped: never read another workspace\'s .recursive/ tree.',
|
|
805
|
+
'Optionality: if the probe fails, fall back to self-audit — never weaken the audit.',
|
|
806
|
+
'Fail loud: capability mismatches reject; do not silently degrade.',
|
|
807
|
+
],
|
|
808
|
+
})
|
|
809
|
+
const briefPath = createChildBrief({
|
|
810
|
+
root: input.root,
|
|
811
|
+
runId: input.runId,
|
|
812
|
+
delegationId: input.delegationId,
|
|
813
|
+
childId: input.childId,
|
|
814
|
+
// ⚠ FU-17 — `input.slice` wins when a work caller supplies one; otherwise this is byte-for-byte the review
|
|
815
|
+
// slice it has always been. A work brief cannot be built here without a second copy of the phase standard.
|
|
816
|
+
slice: input.slice ?? 'Perform the delegated ' + input.role + ' for run ' + input.runId + ' (' + input.phase + ') and write your submission to reply.md.',
|
|
817
|
+
})
|
|
818
|
+
const prompt = buildDelegationPrompt({
|
|
819
|
+
root: input.root,
|
|
820
|
+
runId: input.runId,
|
|
821
|
+
delegationId: input.delegationId,
|
|
822
|
+
childId: input.childId,
|
|
823
|
+
handoffPath,
|
|
824
|
+
briefPath,
|
|
825
|
+
})
|
|
826
|
+
|
|
827
|
+
// R4: plugin-driven delegation with the full request shape. `parent` is the
|
|
828
|
+
// exact live direct-parent Agent (object-identity authority in the live
|
|
829
|
+
// subagent service); absent it, the live start() rejects the request.
|
|
830
|
+
// T28: the depth budget replaces the hardcoded `?? 2`. The parent's own depth is
|
|
831
|
+
// read structurally (the plugin's seam style) and SUBTRACTED, so the configured
|
|
832
|
+
// ceiling bounds the whole recursion rather than being re-granted at every level
|
|
833
|
+
// — a "depth 2" budget that resets per level bounds nothing.
|
|
834
|
+
// T9: notes about routing decisions that could NOT be applied, so a caller is told rather
|
|
835
|
+
// than left to infer it from a model that silently did not take effect.
|
|
836
|
+
const routingNotes: string[] = []
|
|
837
|
+
const request: SubagentStartRequestLike = {
|
|
838
|
+
prompt: [{ type: 'text', text: prompt }],
|
|
839
|
+
label: input.delegationId + '/' + input.childId,
|
|
840
|
+
outputSchema: reviewOutputSchema(),
|
|
841
|
+
toolFilter: input.toolFilter ?? defaultReviewToolFilter(),
|
|
842
|
+
maxDepth: remainingDepthFor(this.enforcementConfig.budgets, parentDepthOf(input.parent), input.maxDepth),
|
|
843
|
+
}
|
|
844
|
+
if (input.parent !== undefined) request.parent = input.parent
|
|
845
|
+
|
|
846
|
+
// ⚠ FU-19 — THE PROVIDER/MODEL LADDER, APPLIED WHERE THE PROVIDER SAYS IT CAN BE.
|
|
847
|
+
//
|
|
848
|
+
// The ladder (per-call → phase → role → general default) is resolved by `resolveSubagentTarget`, which also
|
|
849
|
+
// reports WHICH LEVEL chose each value — so "why did this child run on that model" is answerable from the
|
|
850
|
+
// decision rather than by reading this code. `modelForRole` was the single-level version of this and is gone.
|
|
851
|
+
//
|
|
852
|
+
// ⚠ THE CAPABILITY GATE STAYS AUTHORITATIVE, unchanged: the harness REJECTS a start that sends `agentOptions`
|
|
853
|
+
// to a provider without the `agentOptions` capability, so gating on the model alone would BREAK delegations on
|
|
854
|
+
// providers that do not accept overrides — a routing feature that takes down the delegation it was meant to
|
|
855
|
+
// improve. A choice that cannot be honoured is REPORTED, never silently dropped.
|
|
856
|
+
const target = resolveSubagentTarget({
|
|
857
|
+
role: input.role,
|
|
858
|
+
phase: input.phase,
|
|
859
|
+
policy,
|
|
860
|
+
...(input.providerOverride === undefined && input.modelOverride === undefined
|
|
861
|
+
? {}
|
|
862
|
+
: { override: { provider: input.providerOverride ?? null, model: input.modelOverride ?? null } }),
|
|
863
|
+
ladderProvider: decision.provider ?? null,
|
|
864
|
+
})
|
|
865
|
+
if (target.provider !== null && target.provider !== decision.provider) {
|
|
866
|
+
// The user named a provider. The router still says WHICH TIER it resolved, but the name used to create the
|
|
867
|
+
// child is the user's — and the note states both rather than leaving two answers in the record.
|
|
868
|
+
routingNotes.push(
|
|
869
|
+
'provider ' + target.provider + ' chosen from ' + target.chosen.provider
|
|
870
|
+
+ ' (the router tier resolved as ' + decision.tier + ')',
|
|
871
|
+
)
|
|
872
|
+
}
|
|
873
|
+
if (target.model !== null) {
|
|
874
|
+
const chosenProvider = target.provider ?? decision.provider ?? ''
|
|
875
|
+
const capable = providers[chosenProvider]?.capabilities?.agentOptions === true
|
|
876
|
+
// ⚠ FU-19 — THE CHOICE IS CHECKED AGAINST WHAT DSH ACTUALLY HAS, and the three-state verdict decides what
|
|
877
|
+
// happens next. `missing` means the inventory ANSWERED and does not have this model: it is NOT applied, and
|
|
878
|
+
// crucially NO OTHER MODEL IS SUBSTITUTED — the child inherits the session default and the note says so.
|
|
879
|
+
// `unverified` (no inventory to ask) still applies the model, because refusing on the strength of a service
|
|
880
|
+
// that merely was not mounted would refuse something the user asked for on no evidence.
|
|
881
|
+
const check = checkModelChoice(await describeInventory(this.llmInventory), target.model)
|
|
882
|
+
if (check.verdict === 'missing') {
|
|
883
|
+
routingNotes.push(
|
|
884
|
+
'model ' + target.model + ' was chosen (from ' + target.chosen.model + '), but ' + check.reason
|
|
885
|
+
+ ' — it was NOT applied and no other model was substituted, so the child inherits the session default',
|
|
886
|
+
)
|
|
887
|
+
} else if (!capable) {
|
|
888
|
+
routingNotes.push(
|
|
889
|
+
'model ' + target.model + ' was chosen (from ' + target.chosen.model + '), but provider '
|
|
890
|
+
+ (chosenProvider || '(none)') + ' does not declare the agentOptions capability, so the model was NOT applied',
|
|
891
|
+
)
|
|
892
|
+
} else {
|
|
893
|
+
request.agentOptions = { model: target.model }
|
|
894
|
+
routingNotes.push('model ' + target.model + ' applied (from ' + target.chosen.model + '); inventory says: ' + check.reason)
|
|
895
|
+
}
|
|
896
|
+
}
|
|
897
|
+
|
|
898
|
+
// T19 — DETERMINISTIC OPERATION IDENTITY for the delegation itself. The id
|
|
899
|
+
// covers the inputs PLUS the artifact's BODY, and the body is what keeps a
|
|
900
|
+
// re-review of a REPAIRED artifact a genuinely different operation rather than a
|
|
901
|
+
// recognised repeat. `childId` is deliberately NOT part of it: a fresh round
|
|
902
|
+
// allocates one at random, so including it would make the id meaningless.
|
|
903
|
+
//
|
|
904
|
+
// The body — not the `LockHash` — for the reason T37 found the hard way: a lock
|
|
905
|
+
// hash covers the wall-clock `LockedAt`, so it changes between two locks of
|
|
906
|
+
// byte-identical content and would make this identity fire only half the time.
|
|
907
|
+
const artifactBody = (() => {
|
|
908
|
+
try {
|
|
909
|
+
return lockHashFromContent(
|
|
910
|
+
readFileSync(input.artifactPath, 'utf8').replace(/^[ \t]*Status:.*$/m, '').replace(/^[ \t]*LockedAt:.*\n?/m, ''),
|
|
911
|
+
)
|
|
912
|
+
} catch {
|
|
913
|
+
return 'absent'
|
|
914
|
+
}
|
|
915
|
+
})()
|
|
916
|
+
const operation = operationId({
|
|
917
|
+
act: 'delegate-review',
|
|
918
|
+
input: {
|
|
919
|
+
runId: input.runId,
|
|
920
|
+
phase: input.phase,
|
|
921
|
+
role: input.role,
|
|
922
|
+
delegationId: input.delegationId,
|
|
923
|
+
body: artifactBody,
|
|
924
|
+
},
|
|
925
|
+
})
|
|
926
|
+
|
|
927
|
+
/**
|
|
928
|
+
* T39 — the PRODUCTION provider of the interrupt seam.
|
|
929
|
+
*
|
|
930
|
+
* The delegation's job kill must reach the LIVE child, and the child is interrupted
|
|
931
|
+
* through the SAME `subagents.interrupt` seam the continuable lifecycle already uses —
|
|
932
|
+
* never a second stop path, so a board kill and a lifecycle stop cannot drift apart. The
|
|
933
|
+
* authority is `{ kind: 'ancestor', agent: parent }`: the parent Agent is the object-
|
|
934
|
+
* identity authority the seam expects, exactly as `followup` uses it.
|
|
935
|
+
*
|
|
936
|
+
* An explicit `input.interrupt` still WINS, so a caller with a better authority (a user-
|
|
937
|
+
* initiated stop, say) can supply one. With no seam and no parent there is nothing to
|
|
938
|
+
* interrupt, and this quietly does nothing rather than throwing: the kill already parked
|
|
939
|
+
* the round, and a kill that cannot reach a child must not become an error in its place.
|
|
940
|
+
*/
|
|
941
|
+
const interruptChild = input.interrupt ?? ((childId: string, reason: string) => {
|
|
942
|
+
const seam = subagents
|
|
943
|
+
if (seam?.interrupt === undefined || input.parent === undefined) return
|
|
944
|
+
try {
|
|
945
|
+
seam.interrupt(childId as ContinuableChildId, { kind: 'ancestor', agent: input.parent })
|
|
946
|
+
} catch {
|
|
947
|
+
// Best-effort, like the call site that uses it.
|
|
948
|
+
}
|
|
949
|
+
})
|
|
950
|
+
|
|
951
|
+
let result: SubagentResultLike | null = null
|
|
952
|
+
let error: string | null = null
|
|
953
|
+
let continuable: ContinuableDelegationLike | null = null
|
|
954
|
+
// T36: a parked round is NOT an error — it means the child has not settled yet.
|
|
955
|
+
// Kept apart from `error` so a turn-shaped caller can resume instead of treating
|
|
956
|
+
// an ordinary wait as a failure of the delegation.
|
|
957
|
+
let parked = false
|
|
958
|
+
let parkedReason: string | null = null
|
|
959
|
+
// The run directory the operation index lives under. Computed once, and the
|
|
960
|
+
// index is skipped entirely when there is no root: `join('', …)` would otherwise
|
|
961
|
+
// resolve a RELATIVE path and could read or write an unrelated directory.
|
|
962
|
+
const operationsDir = input.root ? join(input.root, '.recursive', 'run', input.runId) : ''
|
|
963
|
+
|
|
964
|
+
// T19 — a RECOGNISED repeat is not re-executed. Only an operation that already
|
|
965
|
+
// COMPLETED and was ACCEPTED blocks a retry: a failed or refused attempt must
|
|
966
|
+
// stay retryable, and a repaired artifact is a different operation entirely
|
|
967
|
+
// because its body is part of the id. Without this the same review could spawn a
|
|
968
|
+
// second reviewer for an artifact that has not changed.
|
|
969
|
+
const prior = operationsDir === '' ? null : findOperation(operationsDir, operation)
|
|
970
|
+
// ⚠ FU-18 — A CONTINUATION IS NOT A REPEAT, and the distinction has to live HERE rather than in the operation
|
|
971
|
+
// id. The tempting fix is to fold the round into the id so a feedback round becomes a different operation,
|
|
972
|
+
// and it is the wrong one: T28's children budget below counts DISTINCT OPERATIONS and its own comment says a
|
|
973
|
+
// resumed turn "re-enters here with the same id and must not be counted as another child". Making a
|
|
974
|
+
// continuation a new identity would fix this refusal and silently miscount the budget instead.
|
|
975
|
+
//
|
|
976
|
+
// The collision it resolves: the T19 guard refuses an operation already accepted against an UNCHANGED
|
|
977
|
+
// artifact, which is right in general — re-reviewing an artifact nothing has touched since it passed wastes a
|
|
978
|
+
// child. But when the main agent sends FEEDBACK, the artifact has not changed YET: the child has not repaired
|
|
979
|
+
// it. So the guard refused exactly the call that starts the repair. A continuation is the operation still
|
|
980
|
+
// running, so it is exempt; a genuinely fresh delegation against an unchanged artifact is still refused.
|
|
981
|
+
if (prior?.outcome === 'accepted' && input.continuing !== true) {
|
|
982
|
+
throw new Error(
|
|
983
|
+
'review of ' + input.phase + ' is a recognised repeat of an operation already accepted (operation ' +
|
|
984
|
+
operation + '); the artifact has not changed since it passed',
|
|
985
|
+
)
|
|
986
|
+
}
|
|
987
|
+
|
|
988
|
+
// T28 — THE CHILDREN BUDGET, counted FROM the index rather than tracked beside it,
|
|
989
|
+
// so the bound cannot drift from the operations it bounds. Distinct operations,
|
|
990
|
+
// not records: a resumed turn re-enters here with the same id and must not be
|
|
991
|
+
// counted as another child.
|
|
992
|
+
if (operationsDir !== '') {
|
|
993
|
+
const started = countOperations(operationsDir, 'delegate-review', input.phase)
|
|
994
|
+
if (started >= this.enforcementConfig.budgets.maxChildrenPerPhase) {
|
|
995
|
+
throw new Error(
|
|
996
|
+
'children budget reached: phase ' + input.phase + ' has already started ' + started +
|
|
997
|
+
' delegation(s), and the configured cap is ' + this.enforcementConfig.budgets.maxChildrenPerPhase,
|
|
998
|
+
)
|
|
999
|
+
}
|
|
1000
|
+
}
|
|
1001
|
+
|
|
1002
|
+
if (decision.tier === 'native' || decision.tier === 'external-cli') {
|
|
1003
|
+
if (!subagents) {
|
|
1004
|
+
error = 'no ctx.subagents runtime available (self-audit fallback)'
|
|
1005
|
+
} else if (input.mode !== 'one-shot') {
|
|
1006
|
+
// T35: continuable BY DEFAULT — only an explicit 'one-shot' opts out of
|
|
1007
|
+
// the repair path, because a one-shot child cannot be resumed.
|
|
1008
|
+
continuable = await delegateContinuable({
|
|
1009
|
+
subagents,
|
|
1010
|
+
provider: (target.provider ?? decision.provider ?? '') as string,
|
|
1011
|
+
label: input.delegationId + '/' + input.childId,
|
|
1012
|
+
prompt,
|
|
1013
|
+
childId: input.childId,
|
|
1014
|
+
maxDepth: remainingDepthFor(this.enforcementConfig.budgets, parentDepthOf(input.parent), input.maxDepth),
|
|
1015
|
+
toolFilter: input.toolFilter ?? defaultReviewToolFilter(),
|
|
1016
|
+
maxRounds: input.maxRounds ?? 3,
|
|
1017
|
+
// T9: the model the role resolved to, when the provider accepts overrides. It must be
|
|
1018
|
+
// forwarded THROUGH the continuable delegation — that seam builds the start request
|
|
1019
|
+
// itself, so setting it on `request` alone would never reach the provider.
|
|
1020
|
+
...(request.agentOptions === undefined ? {} : { agentOptions: request.agentOptions }),
|
|
1021
|
+
// T39: the ROUND AWAIT is the hangable part of a delegation — the child may work for
|
|
1022
|
+
// minutes — so it is what the board should be able to see as a running job.
|
|
1023
|
+
//
|
|
1024
|
+
// ⚠ SAFETY BY CONSTRUCTION, which is why this wrapper is safe to add at all: the
|
|
1025
|
+
// seam's contract is `SubagentResultLike | null`, and `delegateContinuable` reads a
|
|
1026
|
+
// null as PARKED. So every non-completed job outcome returns `null` — never a throw
|
|
1027
|
+
// and never an invented result — which means a killed or failed round parks exactly
|
|
1028
|
+
// as an unobserved round does. This wrapper therefore CANNOT convert a parked round
|
|
1029
|
+
// into a settlement, and the property holds without a special case.
|
|
1030
|
+
//
|
|
1031
|
+
// ⚠ MEASURED, and it corrects T39's own plan: `delegateReview` has NO interrupt
|
|
1032
|
+
// seam — its input carries none, and this module has no interrupt path (the audit
|
|
1033
|
+
// loop has one only because IT receives the teams runtime). So the job's cancel
|
|
1034
|
+
// cannot yet pre-empt the child; wiring that needs the seam added here first. The
|
|
1035
|
+
// job still gives the board visibility, and the signal is honoured at the boundary.
|
|
1036
|
+
awaitRoundResult: input.awaitRoundResult === undefined ? undefined : async (childId, messageId) => {
|
|
1037
|
+
const awaited = await runTracked(this.jobs, {
|
|
1038
|
+
kind: 'delegation',
|
|
1039
|
+
label: 'delegation round ' + input.runId + ' ' + input.phase,
|
|
1040
|
+
run: async ({ report, signal }) => {
|
|
1041
|
+
report('awaiting the review round')
|
|
1042
|
+
|
|
1043
|
+
// T39 (part 2) — THE PRE-EMPTIVE KILL. A job kill aborts the signal, and this
|
|
1044
|
+
// listener turns that abort into an INTERRUPT of the LIVE CHILD, so the kill
|
|
1045
|
+
// stops the WORK rather than merely stopping our wait for it. That is the
|
|
1046
|
+
// difference between this call site and T10's synchronous ones, where no such
|
|
1047
|
+
// thing is possible.
|
|
1048
|
+
//
|
|
1049
|
+
// The seam is OPTIONAL and the no-seam case stays honest: without a provider
|
|
1050
|
+
// the kill still parks the round and never throws — it simply cannot reach the
|
|
1051
|
+
// child, and nothing here pretends otherwise.
|
|
1052
|
+
const onAbort = () => {
|
|
1053
|
+
try {
|
|
1054
|
+
interruptChild(input.childId, abortReason(signal))
|
|
1055
|
+
} catch {
|
|
1056
|
+
// Best-effort by design: an interrupt that throws must not replace the
|
|
1057
|
+
// kill's own outcome with an unrelated error.
|
|
1058
|
+
}
|
|
1059
|
+
}
|
|
1060
|
+
if (signal?.aborted === true) onAbort()
|
|
1061
|
+
else signal?.addEventListener('abort', onAbort, { once: true })
|
|
1062
|
+
|
|
1063
|
+
try {
|
|
1064
|
+
// ⚠ FU-9 — THE `?.` IS THE FIX FOR THE WHOLE INVESTIGATION. `signal` is typed as an
|
|
1065
|
+
// `AbortSignal` but a caller can reach here with nothing, and `signal.aborted` on undefined
|
|
1066
|
+
// throws `Cannot read properties of undefined (reading 'aborted')` — which the delegation's
|
|
1067
|
+
// catch recorded as its reason, the driver misread as "no continuable repair path", and which
|
|
1068
|
+
// therefore meant THE CHILD NEVER STARTED. Every artifact this investigation chased — no child
|
|
1069
|
+
// session, no reply, no settlement, four brief-only child directories — was downstream of this
|
|
1070
|
+
// one unguarded property read. An absent signal means "nobody can cancel this", not "crash".
|
|
1071
|
+
if (signal?.aborted === true) throw new Error('the round was cancelled before it was awaited')
|
|
1072
|
+
return await input.awaitRoundResult!(childId, messageId)
|
|
1073
|
+
} finally {
|
|
1074
|
+
signal?.removeEventListener('abort', onAbort)
|
|
1075
|
+
}
|
|
1076
|
+
},
|
|
1077
|
+
})
|
|
1078
|
+
if (awaited.status !== 'completed') return null
|
|
1079
|
+
// T39: record the round where the run id is known — same pattern as the lint, and
|
|
1080
|
+
// for the same reason (an event would carry the job, not the run).
|
|
1081
|
+
recordJobRun(input.root, input.runId, {
|
|
1082
|
+
kind: 'delegation',
|
|
1083
|
+
label: 'delegation round ' + input.runId + ' ' + input.phase,
|
|
1084
|
+
result: awaited,
|
|
1085
|
+
})
|
|
1086
|
+
return awaited.value ?? null
|
|
1087
|
+
},
|
|
1088
|
+
parent: input.parent,
|
|
1089
|
+
})
|
|
1090
|
+
if (continuable.parked === true) {
|
|
1091
|
+
// T36: no settlement has landed for this round yet. The child is still
|
|
1092
|
+
// working (or a repair was just sent), so the caller resumes on a later
|
|
1093
|
+
// turn with the SAME child id. `accepted` stays false, and no `result` is
|
|
1094
|
+
// fabricated — an unobserved round is never an approval.
|
|
1095
|
+
parked = true
|
|
1096
|
+
parkedReason = continuable.reason ?? null
|
|
1097
|
+
} else if (continuable.fellBackToOneShot) {
|
|
1098
|
+
// The seam has no continuable capability — keep the one-shot result.
|
|
1099
|
+
result = continuable.rounds[0]?.result ?? null
|
|
1100
|
+
if (!result) error = 'continuable fallback produced no result'
|
|
1101
|
+
} else if (continuable.ok && continuable.rounds.length > 0) {
|
|
1102
|
+
result = continuable.rounds[continuable.rounds.length - 1].result ?? null
|
|
1103
|
+
if (!result) error = 'continuable child produced no final result'
|
|
1104
|
+
} else {
|
|
1105
|
+
// ⚠ A ROUND THAT SETTLED IS A RESULT, EVEN WHEN IT WAS NOT ACCEPTED — and this branch used to
|
|
1106
|
+
// discard it. The condition above requires `continuable.ok`, so a child that REPORTED and was
|
|
1107
|
+
// refused (`success: false`, or a non-completed stop reason) fell through to here: `result` stayed
|
|
1108
|
+
// null, the action record said "NO SETTLEMENT arrived within the wait", and the child's own stop
|
|
1109
|
+
// reason — the one fact that explains the refusal — was dropped on the floor. It is the same defect
|
|
1110
|
+
// as the parked one, one branch over: an absence asserted where the code had evidence. Keeping the
|
|
1111
|
+
// result is what lets the record say "the delegation returned without acceptance; stop reason error"
|
|
1112
|
+
// instead of blaming a wait that ended perfectly well.
|
|
1113
|
+
result = continuable.rounds[continuable.rounds.length - 1]?.result ?? null
|
|
1114
|
+
if (result === null) error = continuable.reason ?? 'continuable delegation failed'
|
|
1115
|
+
}
|
|
1116
|
+
} else {
|
|
1117
|
+
try {
|
|
1118
|
+
result = await delegate({
|
|
1119
|
+
subagents,
|
|
1120
|
+
provider: (target.provider ?? decision.provider ?? '') as string,
|
|
1121
|
+
request,
|
|
1122
|
+
})
|
|
1123
|
+
} catch (err) {
|
|
1124
|
+
error = (err as Error).message
|
|
1125
|
+
}
|
|
1126
|
+
}
|
|
1127
|
+
} else {
|
|
1128
|
+
error = 'delegation resolved to ' + decision.tier + ' (' + decision.reason + ')'
|
|
1129
|
+
}
|
|
1130
|
+
|
|
1131
|
+
const rawEvaluation = result ? evaluateDelegationResult(result) : { accepted: false, reason: error ?? 'no result' }
|
|
1132
|
+
|
|
1133
|
+
// T8 — THE WIRING BUG THE ITEM'S RESCOPE NAMED. `validateReferences` existed, was
|
|
1134
|
+
// exported, was even reachable through this runtime — and was called by NOTHING on the
|
|
1135
|
+
// delegate path. The module header had always claimed the opposite ("validates the
|
|
1136
|
+
// child's references before writing an action record"), so a reviewer could cite files
|
|
1137
|
+
// that do not exist, or paths that escape the root, and the review was recorded as a
|
|
1138
|
+
// PASS. A claim nobody checks is worse than no claim, because it reads as evidence.
|
|
1139
|
+
//
|
|
1140
|
+
// Placed BEFORE the operation record and the action record, which is what makes it
|
|
1141
|
+
// matter: a review with unverifiable references is never indexed as `accepted`, so the
|
|
1142
|
+
// T19 repeat guard treats it as retryable instead of freezing a bad review in place.
|
|
1143
|
+
//
|
|
1144
|
+
// A delegation that claims NOTHING is not checked — refusing an empty list would be a
|
|
1145
|
+
// policy change (is a reference-free review invalid?) rather than a wiring fix, and it
|
|
1146
|
+
// is recorded here as a deliberate boundary rather than an oversight.
|
|
1147
|
+
const claims = result ? referencesFromResult(result) : []
|
|
1148
|
+
const referenceCheck = claims.length === 0 ? null : validateReferences(input.root, claims)
|
|
1149
|
+
const evaluation = referenceCheck === null || referenceCheck.ok
|
|
1150
|
+
? rawEvaluation
|
|
1151
|
+
: { accepted: false, reason: 'review references failed: ' + referenceCheck.failures.join('; ') }
|
|
1152
|
+
|
|
1153
|
+
// T19 — persist the attempt so a RESTART can match it. Only an accepted review is
|
|
1154
|
+
// recorded as `accepted`, which is what makes the recognition above block a
|
|
1155
|
+
// genuine repeat while leaving every failure path retryable. Best-effort: a
|
|
1156
|
+
// failed index write must never change the outcome of a review that already ran.
|
|
1157
|
+
if (operationsDir !== '') {
|
|
1158
|
+
recordOperation(operationsDir, {
|
|
1159
|
+
id: operation,
|
|
1160
|
+
act: 'delegate-review',
|
|
1161
|
+
at: new Date().toISOString().replace(/\.\d{3}Z$/, 'Z'),
|
|
1162
|
+
// ⚠ A PARK IS NOT A REFUSAL, and `unaccepted` said it was. This line used to be
|
|
1163
|
+
// `accepted ? 'accepted' : 'unaccepted'`, so a round that had merely not settled yet was indexed
|
|
1164
|
+
// exactly like a delegation that was evaluated and refused — while the round it really was (still in
|
|
1165
|
+
// flight, resume the same child) was nowhere in the run's own operation log. The defect the action
|
|
1166
|
+
// record had, the log had too. The new value is honest on both readings that matter: it is not
|
|
1167
|
+
// `accepted`, so every retry gate still treats the operation as unfinished and retryable — which is
|
|
1168
|
+
// what a parked round is — and it no longer claims the delegation was judged and rejected.
|
|
1169
|
+
outcome: parked ? 'parked' : (evaluation.accepted ? 'accepted' : 'unaccepted'),
|
|
1170
|
+
phase: input.phase,
|
|
1171
|
+
})
|
|
1172
|
+
}
|
|
1173
|
+
|
|
1174
|
+
// R6: record the attempt as an action record (accepted only if evaluation passes).
|
|
1175
|
+
const actionRecordPath = writeActionRecord({
|
|
1176
|
+
root: input.root,
|
|
1177
|
+
runId: input.runId,
|
|
1178
|
+
subagentId: input.childId,
|
|
1179
|
+
phase: input.phase,
|
|
1180
|
+
// ⚠ FU-17 — the kind is stated in the record. `Status` says accepted, failed or parked; nothing said whether
|
|
1181
|
+
// the child PRODUCED the phase's work or JUDGED it, and a reader of a run could not tell the two apart.
|
|
1182
|
+
purpose: input.role + (input.kind === 'work' ? ' (work)' : '') + ' for run ' + input.runId,
|
|
1183
|
+
executionMode: decision.tier + (input.mode !== 'one-shot' ? ' (continuable)' : ''),
|
|
1184
|
+
artifactPath: input.artifactPath,
|
|
1185
|
+
upstreamArtifacts: input.upstreamArtifacts,
|
|
1186
|
+
reviewBundle: bundle.repoRelativePath,
|
|
1187
|
+
diffBasis: input.diffBasis?.normalizedDiffCommand,
|
|
1188
|
+
codeRefs: input.codeRefs,
|
|
1189
|
+
auditQuestions: input.auditQuestions,
|
|
1190
|
+
findings: evaluation.accepted && result?.structured ? [(result.structured as { verdict?: string })?.verdict ?? 'accepted'] : undefined,
|
|
1191
|
+
success: evaluation.accepted,
|
|
1192
|
+
stopReason: result?.stopReason,
|
|
1193
|
+
// ⚠ A PARKED ROUND IS RECORDED AS PARKED — the whole defect in one field. `success: evaluation.accepted`
|
|
1194
|
+
// is false for a park (correct: nothing was accepted), and `writeActionRecord` reads this flag to state
|
|
1195
|
+
// the third state instead of collapsing it into `failed`.
|
|
1196
|
+
...(parked ? { parked: true } : {}),
|
|
1197
|
+
// ⚠ FU-9 — AND SAY WHICH KIND OF FAILURE, because the record previously could not. `result == null` means
|
|
1198
|
+
// the provider never produced anything at all (never started, or returned nothing) — which is what the
|
|
1199
|
+
// live record's `Stop Reason: n/a` was quietly telling me — while a present result that failed to be
|
|
1200
|
+
// accepted means a child DID run and its work was refused. Different problems, identical artifacts.
|
|
1201
|
+
//
|
|
1202
|
+
// ⚠ AND `parked` IS BRANCHED FIRST. A parked round produced NO result at all — that IS what parking
|
|
1203
|
+
// means — so without this branch first it fell into the `result == null` text below: "the continuable
|
|
1204
|
+
// start was made and NO SETTLEMENT arrived within the wait (the child never reported, or never ran)".
|
|
1205
|
+
// That is the exact false conclusion this fix exists for, and it was reached whatever the record's
|
|
1206
|
+
// Status said. Order is therefore load-bearing here.
|
|
1207
|
+
failure: evaluation.accepted
|
|
1208
|
+
? undefined
|
|
1209
|
+
: parked
|
|
1210
|
+
// ⚠ WHAT IS KNOWN, AND ONLY WHAT IS KNOWN, WITH THE ID THE READER NEEDS TO ACT.
|
|
1211
|
+
//
|
|
1212
|
+
// The text this replaces said "the child never reported, or never ran" — a CONCLUSION drawn from an
|
|
1213
|
+
// ABSENCE, and it was false: a live child went on to complete three review rounds and reply eighteen
|
|
1214
|
+
// minutes later, while the main agent read `Status: failed`, concluded the child was dead, and
|
|
1215
|
+
// obtained its review by other means. So the parked message asserts nothing about the child's state
|
|
1216
|
+
// beyond "no settlement had landed when the wait ended", keeps "may still be working" as the
|
|
1217
|
+
// possibility it is, and NAMES the childId plus the exact next step, because advice to resume is
|
|
1218
|
+
// unactionable without the id. The identity diagnostics stay, because they are what makes a
|
|
1219
|
+
// misconfigured provider readable — but they are diagnostics, not the reason.
|
|
1220
|
+
? 'no settlement had landed when the wait ended, so this round is PARKED, not failed: nothing was'
|
|
1221
|
+
+ ' accepted and nothing was refused, and the child may still be working. The next step is to RESUME'
|
|
1222
|
+
+ ' this round, not to re-dispatch it or replace the child: call `recursive_review` again on a later'
|
|
1223
|
+
+ ' turn with childId ' + String(continuable?.childId ?? input.childId) + ' (the child the round was'
|
|
1224
|
+
+ ' started for, which stays resumable). Diagnostics: tier ' + decision.tier
|
|
1225
|
+
+ ', provider ' + (decision.provider ?? 'none chosen')
|
|
1226
|
+
+ ', names on offer [' + (this.lastProviderNames.join(', ') || 'none') + ']'
|
|
1227
|
+
// ⚠ FU-9 — THE PARENT IDENTITY, because the host refuses a prompt when it cannot resolve the
|
|
1228
|
+
// parent session as a live Agent (`subagent/parent-unavailable`, index.ts L429-436), and the tool
|
|
1229
|
+
// builds this handle with a CAST (`exec.agent as unknown as SubagentParentHandle`). A cast is not
|
|
1230
|
+
// a contract: if the id here is not the one the host looks up, the refusal is real and the
|
|
1231
|
+
// classifier's crash has been hiding it. Printing it here costs nothing and settles the question.
|
|
1232
|
+
+ '; parent id ' + ((input.parent as { id?: string } | undefined)?.id ?? 'none')
|
|
1233
|
+
+ ', parent session keys [' + (input.parent === undefined ? 'no parent' : Object.keys(input.parent as object).join(', ')) + ']'
|
|
1234
|
+
: result == null
|
|
1235
|
+
// ⚠ THE TWO STATES ARE NOT THE SAME AND THE MESSAGE USED TO CONFLATE THEM. The one-shot path cannot
|
|
1236
|
+
// resolve to nothing — the host's `start` returns a run or throws (assertCapabilities, expectProvider)
|
|
1237
|
+
// — so a null result on the CONTINUABLE path means the opposite of what I first wrote: the start WAS
|
|
1238
|
+
// made and NO SETTLEMENT ARRIVED within the wait. That distinction cost me two rounds of looking at
|
|
1239
|
+
// provider names, so the record now states which path a run took and what it was waiting for.
|
|
1240
|
+
// (A park is handled above and never reaches this branch; this one is a continuable round that
|
|
1241
|
+
// produced neither a result nor the park signal, which IS a failure to report.)
|
|
1242
|
+
? (input.mode !== 'one-shot'
|
|
1243
|
+
? 'the continuable start was made and NO SETTLEMENT arrived within the wait (the child never reported,'
|
|
1244
|
+
+ ' or never ran); tier ' + decision.tier + ', provider ' + (decision.provider ?? 'none chosen')
|
|
1245
|
+
+ ', names on offer [' + (this.lastProviderNames.join(', ') || 'none') + ']'
|
|
1246
|
+
// ⚠ FU-9 — THE PARENT IDENTITY, because the host refuses a prompt when it cannot resolve the
|
|
1247
|
+
// parent session as a live Agent (`subagent/parent-unavailable`, index.ts L429-436), and the tool
|
|
1248
|
+
// builds this handle with a CAST (`exec.agent as unknown as SubagentParentHandle`). A cast is not
|
|
1249
|
+
// a contract: if the id here is not the one the host looks up, the refusal is real and the
|
|
1250
|
+
// classifier's crash has been hiding it. Printing it here costs nothing and settles the question.
|
|
1251
|
+
+ '; parent id ' + ((input.parent as { id?: string } | undefined)?.id ?? 'none')
|
|
1252
|
+
+ ', parent session keys [' + (input.parent === undefined ? 'no parent' : Object.keys(input.parent as object).join(', ')) + ']'
|
|
1253
|
+
: 'no delegate result was produced by the one-shot path; tier ' + decision.tier
|
|
1254
|
+
+ ', provider ' + (decision.provider ?? 'none chosen'))
|
|
1255
|
+
: 'the delegation returned without acceptance; stop reason ' + (result.stopReason ?? 'none reported')
|
|
1256
|
+
+ (result.success === false ? ' (the child itself reported success:false)' : ''),
|
|
1257
|
+
})
|
|
1258
|
+
|
|
1259
|
+
// T35: report the mode that ACTUALLY ran, not the one that was asked for. A
|
|
1260
|
+
// missing continuable capability is a real loss — the review can no longer be
|
|
1261
|
+
// sent back to the child that did the work — so it is NAMED rather than
|
|
1262
|
+
// hidden behind a generic success. `continuable-unavailable` is the honest
|
|
1263
|
+
// label for that case: the delegation still happened, the repair path did not.
|
|
1264
|
+
const delegationMode: 'continuable' | 'one-shot' | 'continuable-unavailable' | 'none'
|
|
1265
|
+
= continuable !== null
|
|
1266
|
+
? (continuable.fellBackToOneShot ? 'continuable-unavailable' : 'continuable')
|
|
1267
|
+
: result !== null ? 'one-shot' : 'none'
|
|
1268
|
+
|
|
1269
|
+
// T4: the durable child id is reported for the caller (a tool/closeout that
|
|
1270
|
+
// holds the live parent Agent may drain it explicitly); the HOST owns the
|
|
1271
|
+
// teardown drain (drainContinuableDescendants) at session close — this loop
|
|
1272
|
+
// never forces a drain with a wrong authority credential (childId ≠ parent).
|
|
1273
|
+
return {
|
|
1274
|
+
decision,
|
|
1275
|
+
probe,
|
|
1276
|
+
bundle,
|
|
1277
|
+
handoffPath,
|
|
1278
|
+
briefPath,
|
|
1279
|
+
replyPath: replyPath({ root: input.root, runId: input.runId, delegationId: input.delegationId, childId: input.childId }),
|
|
1280
|
+
childScratchPath: childScratchPath({ root: input.root, runId: input.runId, childId: input.childId }),
|
|
1281
|
+
prompt,
|
|
1282
|
+
request,
|
|
1283
|
+
result,
|
|
1284
|
+
evaluation,
|
|
1285
|
+
/** T9: routing decisions that could NOT be applied, so a caller is told rather than left to infer. */
|
|
1286
|
+
routingNotes,
|
|
1287
|
+
actionRecordPath,
|
|
1288
|
+
error,
|
|
1289
|
+
/** T35: which child lifecycle actually carried this delegation. */
|
|
1290
|
+
delegationMode,
|
|
1291
|
+
/** T19: the deterministic id of this review, persisted so a restart can match it. */
|
|
1292
|
+
operationId: operation,
|
|
1293
|
+
/**
|
|
1294
|
+
* T36: true when the round has NOT settled yet, so the caller resumes with
|
|
1295
|
+
* `continuable.childId` on a later turn. `parkedReason` carries the loop's own
|
|
1296
|
+
* sentence ("the child is still working") without it being an `error`.
|
|
1297
|
+
*/
|
|
1298
|
+
parked,
|
|
1299
|
+
parkedReason,
|
|
1300
|
+
continuable: continuable ? { rounds: continuable.rounds, childId: continuable.childId, fellBackToOneShot: continuable.fellBackToOneShot, parked: continuable.parked === true,
|
|
1301
|
+
// ⚠ FU-9 — `ok` AND `reason` TRAVEL WITH IT. The adapter that builds the review driver's view read only
|
|
1302
|
+
// `rounds`, `childId`, `fellBackToOneShot` and `parked`, so every failure branch's REASON — the whole
|
|
1303
|
+
// point of the field — was discarded one layer above the message that needed it, and `ok: false` was
|
|
1304
|
+
// replaced by a hardcoded `ok: true`. A driver cannot report which branch fired if the branch's own
|
|
1305
|
+
// name never reaches it.
|
|
1306
|
+
ok: continuable.ok, reason: continuable.reason } : null,
|
|
1307
|
+
}
|
|
1308
|
+
}
|
|
1309
|
+
|
|
1310
|
+
/** R6: validate a child's claimed references against actual files. */
|
|
1311
|
+
validateReferences(root: string, references: Reference[]) {
|
|
1312
|
+
return validateReferences(root, references)
|
|
1313
|
+
}
|
|
1314
|
+
|
|
1315
|
+
/** R7: probe availability for a role and render the decision basis prose. */
|
|
1316
|
+
probeDelegation(root: string, role: string, providers: Record<string, SubagentProviderLike> = {}) {
|
|
1317
|
+
const policy = loadRouterPolicy(routerPolicyPath(root), this._routerOverrides)
|
|
1318
|
+
const decision = resolveRole(role, policy, providers)
|
|
1319
|
+
const probe = capabilityProbe({ providers, role, policy })
|
|
1320
|
+
return {
|
|
1321
|
+
decision,
|
|
1322
|
+
probe,
|
|
1323
|
+
basis: delegationDecisionBasis({ role, available: probe.available, provider: probe.provider, fallback: 'self-audit' }),
|
|
1324
|
+
}
|
|
1325
|
+
}
|
|
1326
|
+
|
|
1327
|
+
/**
|
|
1328
|
+
* B3: per-call workspace root resolution. The control-plane root is the
|
|
1329
|
+
* session's cwd (or registry-canonicalized), NEVER process.cwd() — the host
|
|
1330
|
+
* checkout is not the run's workspace. Reads resolve under that root only.
|
|
1331
|
+
*/
|
|
1332
|
+
async resolveRootFor(agent?: { session?: { header?: { cwd?: string } } } | null): Promise<string | null> {
|
|
1333
|
+
return resolveControlPlaneRoot(agent, this.workspaceRegistry, this.repoRoot)
|
|
1334
|
+
}
|
|
1335
|
+
|
|
1336
|
+
/**
|
|
1337
|
+
* SP2 R1 route adapter: resolve the control-plane root for the live route.
|
|
1338
|
+
* sessionId PRIMARY — the host looks up the attached session header cwd and
|
|
1339
|
+
* resolves the root from THAT; the client-passed cwd is a fallback hint only
|
|
1340
|
+
* (hydration / headless callers). Two sessions in one workspace collapse to
|
|
1341
|
+
* one root; a subdir cwd resolves up to the workspace root.
|
|
1342
|
+
*/
|
|
1343
|
+
async resolveRootForRoute(sessionId: string | undefined, cwd: string, sessionsStore?: { get?: (id: string) => { header?: { cwd?: string } } | undefined } | null): Promise<string | null> {
|
|
1344
|
+
// 1) Attached session header (authoritative when present).
|
|
1345
|
+
if (sessionId && sessionsStore?.get) {
|
|
1346
|
+
const attached = sessionsStore.get(sessionId)
|
|
1347
|
+
const headerCwd = attached?.header?.cwd
|
|
1348
|
+
if (headerCwd) {
|
|
1349
|
+
const root = await resolveControlPlaneRoot({ session: { header: { cwd: headerCwd } } }, this.workspaceRegistry, this.repoRoot)
|
|
1350
|
+
if (root) return root
|
|
1351
|
+
}
|
|
1352
|
+
}
|
|
1353
|
+
// 2) Client cwd fallback (hydration / headless) — still registry-canonicalized.
|
|
1354
|
+
if (cwd) {
|
|
1355
|
+
const root = await resolveControlPlaneRoot({ session: { header: { cwd } } }, this.workspaceRegistry, this.repoRoot)
|
|
1356
|
+
if (root) return root
|
|
1357
|
+
}
|
|
1358
|
+
return null
|
|
1359
|
+
}
|
|
1360
|
+
|
|
1361
|
+
async status(runId?: string, agent?: { session?: { header?: { cwd?: string } } } | null): Promise<RecursiveStatusWithGuardDecisions | null> {
|
|
1362
|
+
const root = await this.resolveRootFor(agent)
|
|
1363
|
+
if (!root) return null
|
|
1364
|
+
const resolved = resolveRunDir(root, runId)
|
|
1365
|
+
if (!resolved) return null
|
|
1366
|
+
// T15 (G): surface the rolling guard-decision evidence BESIDE the fold, spread
|
|
1367
|
+
// over it — foldRun's own output shape is parity-asserted (status.parity) and
|
|
1368
|
+
// must not change. Scoped to the resolved run so the field answers "what did
|
|
1369
|
+
// the guard decide about THIS run", newest first.
|
|
1370
|
+
const guardDecisions = readGuardDecisions(root, GUARD_DECISION_READ_LIMIT)
|
|
1371
|
+
.filter((d) => d.runId === resolved.runId)
|
|
1372
|
+
.slice(0, GUARD_DECISION_SURFACE_LIMIT)
|
|
1373
|
+
// T18: the pending set rides beside the fold for the same reason — foldRun's
|
|
1374
|
+
// shape is parity-asserted. Always present (empty when nothing is in flight)
|
|
1375
|
+
// so a consumer needs no null dance, and DERIVED on every call rather than
|
|
1376
|
+
// stored, so it cannot go stale.
|
|
1377
|
+
//
|
|
1378
|
+
// T32: the receipt-chain verdict rides here too. A chain that has been edited or
|
|
1379
|
+
// spliced must be VISIBLE on the status a caller actually reads, not only inside
|
|
1380
|
+
// a test — that was the whole finding: the mechanism was written and read by
|
|
1381
|
+
// nothing. Read-only, so asking cannot change the answer.
|
|
1382
|
+
return {
|
|
1383
|
+
...foldRun(resolved.runDir, resolved.runId),
|
|
1384
|
+
guardDecisions,
|
|
1385
|
+
pendingWork: pendingWork(resolved.runDir),
|
|
1386
|
+
receiptChain: validateReceiptChain(resolved.runDir, resolved.runId),
|
|
1387
|
+
// T22: the same identifier the prompt's stable prefix carries, so a reader can compare status
|
|
1388
|
+
// against prompt without re-rendering the section.
|
|
1389
|
+
contractDigest: contractDigest(this.enforcementConfig),
|
|
1390
|
+
}
|
|
1391
|
+
}
|
|
1392
|
+
|
|
1393
|
+
/**
|
|
1394
|
+
* LIVE BUG 6 refined: structured phase rules for the CURRENT phase. Resolves
|
|
1395
|
+
* the workspace root (same as status/lock), finds the latest run (or the
|
|
1396
|
+
* given runId), advances via getNextLegalPhase, and returns the phase's lint
|
|
1397
|
+
* rules + instructions. Returns null when no active phase exists. This is the
|
|
1398
|
+
* canonical data source for the recursive_phase tool.
|
|
1399
|
+
*/
|
|
1400
|
+
async phaseRules(runId?: string, agent?: { session?: { header?: { cwd?: string } } } | null, files?: readonly string[]): Promise<(PhaseRules & { runId: string; phase: string; memory: string; memoryReason: string }) | null> {
|
|
1401
|
+
const root = await this.resolveRootFor(agent)
|
|
1402
|
+
if (!root) return null
|
|
1403
|
+
const resolved = resolveRunDir(root, runId)
|
|
1404
|
+
if (!resolved) return null
|
|
1405
|
+
const phase = getNextLegalPhase(resolved.runDir)
|
|
1406
|
+
if (!phase) return null
|
|
1407
|
+
// T29 — MEMORY AT RUN ENTRY, ON THIS CALL AND NOWHERE ELSE.
|
|
1408
|
+
//
|
|
1409
|
+
// ⚠ THE ONCE-GATE IS THIS FUNCTION, not a mechanism added beside it: `recursive_phase` calls it
|
|
1410
|
+
// once per phase entry, so riding the injection on its EXISTING return is what keeps it
|
|
1411
|
+
// once-per-run instead of once-per-turn. A second dedupe would be a second thing to get wrong.
|
|
1412
|
+
//
|
|
1413
|
+
// ⚠ AND NOTHING RELEVANT INJECTS NOTHING — `memory` is the empty string, with the reason saying
|
|
1414
|
+
// why, rather than a section that fabricates relevance the plane does not have.
|
|
1415
|
+
const requirements = (() => {
|
|
1416
|
+
try {
|
|
1417
|
+
return readFileSync(join(resolved.runDir, '00-requirements.md'), 'utf8')
|
|
1418
|
+
} catch {
|
|
1419
|
+
// A run with no requirements yet has a weaker query, not an error: the paths still count.
|
|
1420
|
+
return ''
|
|
1421
|
+
}
|
|
1422
|
+
})()
|
|
1423
|
+
// `phase` IS the artifact file name (`getNextLegalPhase` returns one of PHASE_SEQUENCE), so the gate
|
|
1424
|
+
// lookup needs no mapping — and the artifact's own text is what says whether it is still owed.
|
|
1425
|
+
const pending = pendingGateFor(phase, (() => {
|
|
1426
|
+
try {
|
|
1427
|
+
return readFileSync(join(resolved.runDir, phase), 'utf8')
|
|
1428
|
+
} catch {
|
|
1429
|
+
return null
|
|
1430
|
+
}
|
|
1431
|
+
})())
|
|
1432
|
+
const selection = selectMemory(root, {
|
|
1433
|
+
query: requirements.slice(0, 4000),
|
|
1434
|
+
// FU-4: the run's OWN changed paths, computed when the caller supplies none — so T29's path
|
|
1435
|
+
// weighting is fed by a real run rather than only by tests. `[]` from a non-git root is fine: the
|
|
1436
|
+
// query still ranks, and a memory hint must never be why a phase call fails.
|
|
1437
|
+
files: files ?? changedPaths(root),
|
|
1438
|
+
// P2: the phase in play, so an entry declaring it applies here outranks general guidance.
|
|
1439
|
+
// P3b: the counters, read ONCE here and handed to the ranking — the book is evidence about retrieval,
|
|
1440
|
+
// and where it lives is the caller's business, not the ranker's.
|
|
1441
|
+
feedback: readFeedback(root),
|
|
1442
|
+
phase,
|
|
1443
|
+
})
|
|
1444
|
+
// P3b: record what this phase was shown, so the loop has evidence to settle at closeout.
|
|
1445
|
+
recordInjection(resolved.runDir, selection.shards.map((shard) => ({
|
|
1446
|
+
source: shard.entry.source,
|
|
1447
|
+
title: shard.entry.title,
|
|
1448
|
+
score: shard.score,
|
|
1449
|
+
})), phase)
|
|
1450
|
+
// ⚠ AND RECORD THAT THE READ HAPPENED, EVEN WHEN IT RETURNED NOTHING — the fact the phase-0 write gate
|
|
1451
|
+
// (`memory-read` in `policy-globs.ts`) is keyed on. `recordInjection` above can only write a row when a
|
|
1452
|
+
// shard was selected, so on an EMPTY plane it writes nothing at all and a gate reading it would refuse
|
|
1453
|
+
// forever in a fresh workspace. The receipt is written on every phase entry, `injected: false` included:
|
|
1454
|
+
// the requirement is that memory was READ before the run's requirements were authored, not that the
|
|
1455
|
+
// plane had something to say. This is the ONE place a read becomes durable, because this function is the
|
|
1456
|
+
// one place a phase entry reads memory (see the T29 note above).
|
|
1457
|
+
recordMemoryRead(resolved.runDir, phase, {
|
|
1458
|
+
injected: selection.injected,
|
|
1459
|
+
shards: selection.shards.length,
|
|
1460
|
+
reason: selection.reason,
|
|
1461
|
+
})
|
|
1462
|
+
return {
|
|
1463
|
+
runId: resolved.runId,
|
|
1464
|
+
phase,
|
|
1465
|
+
...phaseRulesFor(phase),
|
|
1466
|
+
// ⚠ `'phase'` IS PASSED BECAUSE THIS IS NOT A REVIEW BUNDLE. The renderer's wording used to be the
|
|
1467
|
+
// reviewer's (*"relevant to this REVIEW"*) and this payload inherited it, so the sentence named a
|
|
1468
|
+
// reader this text never reaches. The context is the ONE thing that differs between the two callers
|
|
1469
|
+
// (see `MemoryRenderContext`), so the renderer stays single.
|
|
1470
|
+
memory: selection.injected ? renderMemorySection(selection.shards.map((shard) => shard.entry), 'phase') : '',
|
|
1471
|
+
memoryReason: selection.reason,
|
|
1472
|
+
// FU-7 — THE PHASE-ENTRY CALL POINTS. Phase 03 owes a `TDD Mode` decision and phase 05 a
|
|
1473
|
+
// `QA Execution Mode` one; both are surfaced HERE, at the entry the tool already makes, so a
|
|
1474
|
+
// caller does not have to know the workflow's gate vocabulary to be asked the right question.
|
|
1475
|
+
// Omitted entirely once the artifact carries the marker (that IS the once-gate).
|
|
1476
|
+
...(pending === null ? {} : { ask: { gate: pending, ...buildAskQuestion(pending), artifact: GATE_DEFAULT_ARTIFACT[pending] } }),
|
|
1477
|
+
}
|
|
1478
|
+
}
|
|
1479
|
+
|
|
1480
|
+
/**
|
|
1481
|
+
* Scaffold a run directory with FULL per-phase templates (no-op if exists).
|
|
1482
|
+
* 00-requirements.md + 00-worktree.md are byte-identical to canonical
|
|
1483
|
+
* recursive-init.py (incl. git-context prefill); later phases carry every
|
|
1484
|
+
* required section (get_artifact_required_sections) + TODO + FAIL gates.
|
|
1485
|
+
* Also scaffolds addenda/subagents/router-prompts/evidence dirs. Returns the
|
|
1486
|
+
* run dir + created artifacts.
|
|
1487
|
+
*
|
|
1488
|
+
* When `opts.createWorktree` is true, a linked worktree is first created at
|
|
1489
|
+
* `.worktrees/<runId>/` and the run is scaffolded INSIDE it (per the
|
|
1490
|
+
* "all subsequent phases execute in worktree context" rule). The worktree
|
|
1491
|
+
* branch defaults to `recursive/<runId>` and is cut from `opts.baseBranch`
|
|
1492
|
+
* (default: current HEAD branch of the root checkout).
|
|
1493
|
+
*/
|
|
1494
|
+
async initRun(runId: string, agent?: { session?: { header?: { cwd?: string } } } | null, opts?: { createWorktree?: boolean; baseBranch?: string }): Promise<{ runDir: string; runId: string; created: string[]; existing: string[]; worktree?: CreateWorktreeResult }> {
|
|
1495
|
+
const root = await this.resolveRootFor(agent)
|
|
1496
|
+
if (!root) throw new Error('cannot resolve workspace control-plane root for this session')
|
|
1497
|
+
let scaffoldRoot = root
|
|
1498
|
+
let worktree: CreateWorktreeResult | undefined
|
|
1499
|
+
if (opts?.createWorktree) {
|
|
1500
|
+
// T10: a linked-worktree create is the operation the item names as the "hung" one — git
|
|
1501
|
+
// can take a long time on a large repo — and like the linter it CANNOT stop mid-flight,
|
|
1502
|
+
// because `createLinkedWorktree` is synchronous. What the job buys is therefore the same
|
|
1503
|
+
// and is stated plainly: the board SEES it running and the caller can stop WAITING.
|
|
1504
|
+
//
|
|
1505
|
+
// The failure path is unchanged: `createLinkedWorktree` reports `{ ok: false, error }`
|
|
1506
|
+
// rather than throwing, so a refused create still throws exactly what it threw before.
|
|
1507
|
+
const created = await runTracked(this.jobs, {
|
|
1508
|
+
kind: 'worktree',
|
|
1509
|
+
label: 'worktree ' + runId,
|
|
1510
|
+
run: async ({ report, signal }) => {
|
|
1511
|
+
report('creating ' + runId)
|
|
1512
|
+
if (signal.aborted) throw new Error('worktree create cancelled before it started')
|
|
1513
|
+
return createLinkedWorktree({ repoRoot: root, runId, baseBranch: opts.baseBranch })
|
|
1514
|
+
},
|
|
1515
|
+
})
|
|
1516
|
+
if (created.status !== 'completed' || created.value === undefined) {
|
|
1517
|
+
throw new Error(created.detail ?? created.error ?? 'worktree create did not complete')
|
|
1518
|
+
}
|
|
1519
|
+
worktree = created.value
|
|
1520
|
+
if (!worktree.ok) throw new Error(worktree.error ?? 'worktree create failed')
|
|
1521
|
+
scaffoldRoot = worktree.worktreeDir
|
|
1522
|
+
// T39: recorded into the SCAFFOLD root, because a worktree run's layer lives INSIDE the
|
|
1523
|
+
// worktree — recording it against the main checkout would file it where no run exists.
|
|
1524
|
+
// A create that FAILED records nothing here: there is no run layer to record into, and
|
|
1525
|
+
// the failure already reaches the caller as the thrown error above.
|
|
1526
|
+
recordJobRun(scaffoldRoot, runId, { kind: 'worktree', label: 'worktree ' + runId, result: created })
|
|
1527
|
+
}
|
|
1528
|
+
const runDir = join(scaffoldRoot, '.recursive', 'run', runId)
|
|
1529
|
+
mkdirSync(runDir, { recursive: true })
|
|
1530
|
+
const created: string[] = []
|
|
1531
|
+
const existing: string[] = []
|
|
1532
|
+
|
|
1533
|
+
// Scaffold dirs (canonical recursive-init sequence).
|
|
1534
|
+
for (const dir of RUN_SCAFFOLD_DIRS) {
|
|
1535
|
+
const p = join(runDir, dir)
|
|
1536
|
+
if (existsSync(p)) { existing.push(dir + '/'); continue }
|
|
1537
|
+
mkdirSync(p, { recursive: true })
|
|
1538
|
+
created.push(dir + '/')
|
|
1539
|
+
}
|
|
1540
|
+
|
|
1541
|
+
// Git context for the Phase 0 diff-basis prefill (canonical parity). When a
|
|
1542
|
+
// worktree was created, the git context + Phase 0 record the WORKTREE.
|
|
1543
|
+
const { context: gitContext, error: prefillError } = detectGitContext(scaffoldRoot)
|
|
1544
|
+
|
|
1545
|
+
// Phase 0 templates: byte-identical to canonical recursive-init.py.
|
|
1546
|
+
const phase0: Array<[string, string]> = [
|
|
1547
|
+
['00-requirements.md', requirementsContent(runId, 'feature', '')],
|
|
1548
|
+
['00-worktree.md', worktreeContent(runId, scaffoldRoot, gitContext, prefillError)],
|
|
1549
|
+
]
|
|
1550
|
+
for (const [file, content] of phase0) {
|
|
1551
|
+
const path = join(runDir, file)
|
|
1552
|
+
if (existsSync(path)) { existing.push(file); continue }
|
|
1553
|
+
writeFileSync(path, content, 'utf8')
|
|
1554
|
+
created.push(file)
|
|
1555
|
+
}
|
|
1556
|
+
|
|
1557
|
+
// Later phases: full required-sections scaffold.
|
|
1558
|
+
const laterPhases = [
|
|
1559
|
+
'01-as-is.md',
|
|
1560
|
+
'01.5-root-cause.md',
|
|
1561
|
+
'02-to-be-plan.md',
|
|
1562
|
+
'03-implementation-summary.md',
|
|
1563
|
+
'03.5-code-review.md',
|
|
1564
|
+
'04-test-summary.md',
|
|
1565
|
+
'05-manual-qa.md',
|
|
1566
|
+
'06-decisions-update.md',
|
|
1567
|
+
'07-state-update.md',
|
|
1568
|
+
'08-memory-impact.md',
|
|
1569
|
+
]
|
|
1570
|
+
for (const file of laterPhases) {
|
|
1571
|
+
const path = join(runDir, file)
|
|
1572
|
+
if (existsSync(path)) { existing.push(file); continue }
|
|
1573
|
+
writeFileSync(path, laterPhaseContent(runId, file), 'utf8')
|
|
1574
|
+
created.push(file)
|
|
1575
|
+
}
|
|
1576
|
+
|
|
1577
|
+
const result: { runDir: string; runId: string; created: string[]; existing: string[]; worktree?: CreateWorktreeResult; runStartApproval: { approved: boolean; artifact: string; reason: string; gate: string } } = { runDir, runId, created, existing, runStartApproval: { ...this.readRunStartApproval(scaffoldRoot, runId), gate: RUN_START_GATE_ID } }
|
|
1578
|
+
if (worktree) result.worktree = worktree
|
|
1579
|
+
// ⚠ PHASE 0 — SCAFFOLDING A RUN MUST NOT START IT. This used to arm a durable run goal right here,
|
|
1580
|
+
// and arming is what makes the harness drive autonomous rounds: asking for a run spec was enough to
|
|
1581
|
+
// start an unattended run. The rule is that phase 0 requires EXPLICIT approval to start a run and
|
|
1582
|
+
// goal, so init now ONLY SCAFFOLDS and REPORTS what is owed. `result.runStartApproval` is that
|
|
1583
|
+
// report, and it is the pointer the model needs: the run stays inert until `recursive_ask` answers
|
|
1584
|
+
// the `run-start` gate (which re-reads the approval and arms the goal through `armRunGoalIfApproved`).
|
|
1585
|
+
// The spec is not forbidden — the whole run directory was just written. What is withheld is the goal.
|
|
1586
|
+
return result
|
|
1587
|
+
}
|
|
1588
|
+
|
|
1589
|
+
/**
|
|
1590
|
+
* PHASE 0 — THE APPROVAL ACT: record the human's `Start run` decision and arm the run's goal.
|
|
1591
|
+
*
|
|
1592
|
+
* ⚠ THE ONLY PATH THAT STARTS A RUN. It exists as one method rather than as "write a line, then
|
|
1593
|
+
* project the goal" at the tool, because those two steps must not be separable: an approval recorded
|
|
1594
|
+
* without the arm (or an arm without the record) is exactly the half-state that made this defect hard
|
|
1595
|
+
* to see. `tests/run-start-approval.spec.ts` drives both halves through this one call.
|
|
1596
|
+
*
|
|
1597
|
+
* The approval line goes into the run's own Phase 0 artifact, so it is durable, citable, and survives
|
|
1598
|
+
* the session — and so a reader of the run can answer "was this run started, and by what?" without the
|
|
1599
|
+
* transcript. `answer` is validated against the gate's own labels before it reaches here.
|
|
1600
|
+
*/
|
|
1601
|
+
approveRunStart(root: string, runId: string, agent?: { session?: { header?: { cwd?: string } } } | null, answer: string = RUN_START_APPROVE): { ok: boolean; reason: string; path: string; replaced: boolean; goal: SyncResult } {
|
|
1602
|
+
if (root.trim() === '' || runId.trim() === '') {
|
|
1603
|
+
return { ok: false, reason: 'a run start needs a workspace root and a run id', path: '', replaced: false, goal: { ok: false, reason: 'no run to start' } }
|
|
1604
|
+
}
|
|
1605
|
+
const path = runStartArtifactPath(root, runId)
|
|
1606
|
+
// The record is written through the same in-place marker write every other gate uses, so a changed
|
|
1607
|
+
// mind REPLACES its line instead of leaving two answers to one question.
|
|
1608
|
+
const written = this.recordAskAnswer(root, runId, RUN_START_ARTIFACT, '- ' + RUN_START_MARKER + ': ' + answer)
|
|
1609
|
+
const approval = this.readRunStartApproval(root, runId)
|
|
1610
|
+
if (!approval.approved) {
|
|
1611
|
+
// A `Hold` (or anything else) is recorded as the decision it is and STARTS NOTHING. The goal is
|
|
1612
|
+
// not merely paused: an unstarted run has no goal at all (see goals-projection.ts branch 3).
|
|
1613
|
+
return { ok: false, reason: approval.reason, path: written.path, replaced: written.replaced, goal: { ok: false, reason: RUN_START_NOT_APPROVED } }
|
|
1614
|
+
}
|
|
1615
|
+
const goal = this.armRunGoalIfApproved(agent, root, runId, 'active')
|
|
1616
|
+
return { ok: true, reason: '', path: written.path, replaced: written.replaced, goal }
|
|
1617
|
+
}
|
|
1618
|
+
|
|
1619
|
+
/**
|
|
1620
|
+
* Create a linked worktree for a run under the given workspace root. The
|
|
1621
|
+
* worktree branch defaults to `recursive/<runId>` and is cut from the given
|
|
1622
|
+
* base branch (default: the current HEAD branch of the root checkout).
|
|
1623
|
+
* Refuses to create over an existing run directory. Workspace-scoped.
|
|
1624
|
+
*/
|
|
1625
|
+
createRunWorktree(root: string, runId: string, baseBranch?: string): CreateWorktreeResult {
|
|
1626
|
+
return createLinkedWorktree({ repoRoot: root, runId, baseBranch })
|
|
1627
|
+
}
|
|
1628
|
+
|
|
1629
|
+
/**
|
|
1630
|
+
* Promote a branch up the dev/stage/main chain (fast-forward). Workspace-scoped.
|
|
1631
|
+
*/
|
|
1632
|
+
promoteRunBranch(root: string, fromBranch: string, toBranch: string): PromoteBranchResult {
|
|
1633
|
+
return promoteBranch({ repoRoot: root, fromBranch, toBranch })
|
|
1634
|
+
}
|
|
1635
|
+
|
|
1636
|
+
/**
|
|
1637
|
+
* Worktree + branch status for a workspace root: the linked worktrees,
|
|
1638
|
+
* which branch each is on, and the current checkout's base/upstream context.
|
|
1639
|
+
*/
|
|
1640
|
+
worktreeStatus(root: string): Record<string, unknown> {
|
|
1641
|
+
const facts = gitFacts(root)
|
|
1642
|
+
const worktrees = listWorktrees(root)
|
|
1643
|
+
return {
|
|
1644
|
+
root,
|
|
1645
|
+
isWorktree: facts.isWorktree,
|
|
1646
|
+
branch: facts.branch,
|
|
1647
|
+
upstreamBranch: facts.upstreamBranch,
|
|
1648
|
+
worktrees,
|
|
1649
|
+
}
|
|
1650
|
+
}
|
|
1651
|
+
|
|
1652
|
+
/**
|
|
1653
|
+
* Lock a DRAFT artifact (or reopen a LOCKED one). Validates prerequisites;
|
|
1654
|
+
* writes Status/LockedAt/LockHash + receipt. Returns the lock result.
|
|
1655
|
+
*/
|
|
1656
|
+
async lockArtifact(runId: string, artifact: string, reopen = false, agent?: { session?: { header?: { cwd?: string } } } | null): Promise<LockArtifactResult> {
|
|
1657
|
+
const root = await this.resolveRootFor(agent)
|
|
1658
|
+
if (!root) throw new Error('cannot resolve workspace control-plane root for this session')
|
|
1659
|
+
const runDir = join(root, '.recursive', 'run', runId)
|
|
1660
|
+
const artifactPath = join(runDir, artifact)
|
|
1661
|
+
if (reopen) {
|
|
1662
|
+
return this.reopenArtifact(root, runDir, runId, artifact, artifactPath, agent)
|
|
1663
|
+
}
|
|
1664
|
+
if (!existsSync(artifactPath)) throw new Error('Artifact not found: ' + artifact)
|
|
1665
|
+
const status = getLockStatus(artifactPath)
|
|
1666
|
+
if (status === 'LOCKED') throw new Error('Artifact already LOCKED: ' + artifact)
|
|
1667
|
+
// B2: the pre-step gate reads transition intent from the session log; the
|
|
1668
|
+
// durable commit below is what the live fs route folds. No phase-intent event.
|
|
1669
|
+
const blockers = getPrerequisiteBlockers(runDir, artifact)
|
|
1670
|
+
if (blockers.length > 0) {
|
|
1671
|
+
// T1 (goals projection): a gate-block becomes a durable, UI-visible goal
|
|
1672
|
+
// block rather than a one-line advisory. Best-effort before the throw.
|
|
1673
|
+
const message = 'monotonic lock-order: ' + blockers.map(b => b.artifact + ' (' + b.status + ')').join(', ')
|
|
1674
|
+
try { this.blockRunToGoal(agent, runId, { code: 'prerequisite-blockers', message }) } catch { /* best-effort */ }
|
|
1675
|
+
throw new Error('Prerequisite blockers: ' + blockers.map(b => b.artifact + ' (' + b.status + ')').join(', '))
|
|
1676
|
+
}
|
|
1677
|
+
// T18 — QUIESCENCE. A lock is only sound at a point where nothing is in
|
|
1678
|
+
// flight, so a run with an unresolved delegation refuses to lock and names
|
|
1679
|
+
// what is blocking. Checked AFTER the prerequisite gate deliberately: the
|
|
1680
|
+
// monotonic lock-order rule is the canonical one that the parity goldens and
|
|
1681
|
+
// every existing test know, and a run that is both out of order AND has a
|
|
1682
|
+
// delegation open still reports ordering first, exactly as before this item.
|
|
1683
|
+
const inFlight = pendingWork(runDir)
|
|
1684
|
+
if (inFlight.length > 0) {
|
|
1685
|
+
throw new Error(toolError('PENDING_WORK', inFlight.map((p) => p.detail).join('; ')))
|
|
1686
|
+
}
|
|
1687
|
+
// T40 - THE PHASE-8 MEMORY GATE. Phase 8 exists to promote this run's durable memory, and until
|
|
1688
|
+
// this gate the step was a TODO line the agent ticked for itself: a live workspace held three
|
|
1689
|
+
// completed runs and a .recursive/memory/ tree still holding only bootstrap placeholders, with
|
|
1690
|
+
// run 03's artifact declaring the step done against ANOTHER plugin's store. The requirement is
|
|
1691
|
+
// now a fact about files: a path under the plane, declared by the artifact, whose own text
|
|
1692
|
+
// carries THIS run's Source-Runs. Placed AFTER quiescence and BEFORE the lint gate: lint stays
|
|
1693
|
+
// last (its documented invariant), lock order stays first, and a below-standard artifact is
|
|
1694
|
+
// reported with the one thing it must still do. Mode-independent, like every other refusal in
|
|
1695
|
+
// this chain - lockArtifact has no mode input and the lock chain has never been advisory-gated.
|
|
1696
|
+
const memoryRefusal = phase8MemoryLockRefusal(root, runId, artifact)
|
|
1697
|
+
if (memoryRefusal !== null) {
|
|
1698
|
+
try { this.blockRunToGoal(agent, runId, { code: 'phase8-memory-missing', message: memoryRefusal }) } catch { /* best-effort, like the prerequisite gate */ }
|
|
1699
|
+
throw new Error(memoryRefusal)
|
|
1700
|
+
}
|
|
1701
|
+
// README §4.1 — THE STANDARD GATE. `recursive_lock` promises that it "refuses
|
|
1702
|
+
// if the artifact does not meet the standard", and until this gate existed it
|
|
1703
|
+
// checked existence, re-lock, lock ORDER and quiescence and never consulted the
|
|
1704
|
+
// linter, so a 14-FAIL artifact locked cleanly and its receipt certified work no
|
|
1705
|
+
// check had accepted. The linter is the authority on the standard, so the same
|
|
1706
|
+
// entry point the `recursive_lint` tool calls is consulted here.
|
|
1707
|
+
//
|
|
1708
|
+
// PLACED LAST among the refusals, immediately before the LOCKED-fields mutation:
|
|
1709
|
+
// every cheaper refusal keeps its existing precedence, and in particular LOCK
|
|
1710
|
+
// ORDER STAYS FIRST — a run that is both out of order AND below standard still
|
|
1711
|
+
// reports ordering, exactly as it did before this gate. The already-LOCKED check
|
|
1712
|
+
// and the `reopen` branch precede this point too, so nothing previously locked is
|
|
1713
|
+
// disturbed and reopen does not suddenly demand a standard it never had.
|
|
1714
|
+
//
|
|
1715
|
+
// A plain `Error`, in the same style as the refusals above, rather than a new
|
|
1716
|
+
// `toolError` code: no registry entry describes "below the phase standard", and
|
|
1717
|
+
// inventing one would add a code `tests/errors.spec.ts` has to be taught, for a
|
|
1718
|
+
// refusal whose remedy is the FAIL list it already carries.
|
|
1719
|
+
//
|
|
1720
|
+
// CONSERVATIVE WHEN THE LINT CANNOT RUN: a lint whose job was killed or failed
|
|
1721
|
+
// returns `passed: false` with the reason in `errors`, so an artifact that could
|
|
1722
|
+
// not be checked is refused rather than waved through — "not measured" is not
|
|
1723
|
+
// "meets the standard".
|
|
1724
|
+
const lint = await this.lintArtifact(runId, artifact, agent)
|
|
1725
|
+
if (!lint.passed) {
|
|
1726
|
+
throw new Error(
|
|
1727
|
+
'Artifact ' + artifact + ' does not meet the phase standard, so it was not locked: ' + lint.errors.join('; '),
|
|
1728
|
+
)
|
|
1729
|
+
}
|
|
1730
|
+
let content = readFileSync(artifactPath, 'utf8')
|
|
1731
|
+
const lockedAt = new Date().toISOString().replace(/\.\d{3}Z$/, 'Z')
|
|
1732
|
+
content = setOrInsertField(content, 'Status', 'LOCKED', ['Phase'])
|
|
1733
|
+
content = setOrInsertField(content, 'LockedAt', lockedAt, ['Status'])
|
|
1734
|
+
const provisional = setOrInsertField(content, 'LockHash', '0'.repeat(64), ['LockedAt', 'Status'])
|
|
1735
|
+
const lockHash = lockHashFromContent(provisional)
|
|
1736
|
+
content = setOrInsertField(content, 'LockHash', lockHash, ['LockedAt', 'Status'])
|
|
1737
|
+
writeFileSync(artifactPath, content, 'utf8')
|
|
1738
|
+
const receipt = writeReceipt(runDir, artifact, artifactPath)
|
|
1739
|
+
// B2: the commit is durable — the live fs route folds it on the next GET.
|
|
1740
|
+
return {
|
|
1741
|
+
artifact,
|
|
1742
|
+
runId,
|
|
1743
|
+
status: 'LOCKED',
|
|
1744
|
+
lockedAt,
|
|
1745
|
+
lockHash,
|
|
1746
|
+
receipt,
|
|
1747
|
+
blockers: [],
|
|
1748
|
+
}
|
|
1749
|
+
}
|
|
1750
|
+
|
|
1751
|
+
/** Reopen a LOCKED artifact to DRAFT (delete LockedAt/LockHash, invalidate downstream receipts). */
|
|
1752
|
+
private reopenArtifact(root: string, runDir: string, runId: string, artifact: string, artifactPath: string, agent?: { session?: { header?: { cwd?: string } } } | null): LockArtifactResult {
|
|
1753
|
+
if (!existsSync(artifactPath)) throw new Error('Artifact not found: ' + artifact)
|
|
1754
|
+
let content = readFileSync(artifactPath, 'utf8')
|
|
1755
|
+
|
|
1756
|
+
// T19 — DETERMINISTIC OPERATION IDENTITY. Reopen is the one genuinely DESTRUCTIVE
|
|
1757
|
+
// operation here: it strips the lock fields and invalidates every downstream
|
|
1758
|
+
// receipt, so running it twice by accident destroys evidence. A retry after a
|
|
1759
|
+
// partial failure was previously indistinguishable from a fresh request.
|
|
1760
|
+
//
|
|
1761
|
+
// The id covers the INPUTS plus the STATE BEING REOPENED, and that third
|
|
1762
|
+
// component is what makes the rule correct rather than merely present:
|
|
1763
|
+
// - a retry after a FAILED reopen still sees the same state, because a failed
|
|
1764
|
+
// reopen leaves the artifact locked — so the retry IS recognised, which is
|
|
1765
|
+
// the case the item exists for;
|
|
1766
|
+
// - a DELIBERATE second reopen after a REPAIR sees a different state, so it is
|
|
1767
|
+
// a different operation and runs — the legitimate workflow stays intact.
|
|
1768
|
+
// Keying on `{runId, artifact}` alone would have made the second reopen a no-op
|
|
1769
|
+
// and quietly broken reopen→fix→lock→reopen.
|
|
1770
|
+
//
|
|
1771
|
+
// The state is the artifact's BODY — its content with the lock fields normalised
|
|
1772
|
+
// out — and NOT its `LockHash`. That distinction is a bug fix, found by T37's
|
|
1773
|
+
// flake hunt: `LockHash` covers `LockedAt`, which is wall-clock truncated to the
|
|
1774
|
+
// SECOND, so two locks of byte-identical content hash differently whenever the
|
|
1775
|
+
// clock crosses a second boundary. Keying on it made this guard fire or not
|
|
1776
|
+
// depending on timing — it failed 6 of 12 measured runs. The body is stable
|
|
1777
|
+
// across a re-lock of identical content and differs as soon as the author
|
|
1778
|
+
// changes anything that matters.
|
|
1779
|
+
const bodyAtEntry = lockHashFromContent(
|
|
1780
|
+
content.replace(/^[ \t]*Status:.*$/m, '').replace(/^[ \t]*LockedAt:.*\n?/m, ''),
|
|
1781
|
+
)
|
|
1782
|
+
const operation = operationId({ act: 'reopen', input: { runId, artifact, body: bodyAtEntry } })
|
|
1783
|
+
if (findOperation(runDir, operation)?.outcome === 'applied') {
|
|
1784
|
+
// This exact operation already completed on this exact state. Reported as a
|
|
1785
|
+
// recognised repeat rather than executed again — a second strip-and-invalidate
|
|
1786
|
+
// would destroy whatever the first one left behind.
|
|
1787
|
+
throw new Error(
|
|
1788
|
+
'reopen ' + artifact + ' is a recognised repeat of an operation already applied (operation ' +
|
|
1789
|
+
operation + '); refusing to reopen it a second time',
|
|
1790
|
+
)
|
|
1791
|
+
}
|
|
1792
|
+
|
|
1793
|
+
content = content.replace(/^[ \t]*Status:.*$/m, 'Status: `DRAFT`')
|
|
1794
|
+
content = content.replace(/^[ \t]*LockedAt:.*\n?/m, '')
|
|
1795
|
+
content = content.replace(/^[ \t]*LockHash:.*\n?/m, '')
|
|
1796
|
+
writeFileSync(artifactPath, content, 'utf8')
|
|
1797
|
+
// Record the attempt AFTER it succeeded, best-effort: a failed index write must
|
|
1798
|
+
// never change the outcome of an operation the caller already performed.
|
|
1799
|
+
recordOperation(runDir, { id: operation, act: 'reopen', at: new Date().toISOString().replace(/\.\d{3}Z$/, 'Z'), outcome: 'applied' })
|
|
1800
|
+
const hadReceipt = invalidateReceipt(runDir, artifact)
|
|
1801
|
+
const stale = getStaleDownstreamPhases(runDir, artifact)
|
|
1802
|
+
for (const entry of stale) invalidateReceipt(runDir, entry.artifact)
|
|
1803
|
+
// B2: reopen reverts to DRAFT — the live fs route folds the reverted state.
|
|
1804
|
+
// T1 (goals projection): re-arm the durable run goal (reopen un-blocks).
|
|
1805
|
+
// ⚠ PHASE 0: only for a run that WAS started — the approval is read from the run's own Phase 0
|
|
1806
|
+
// artifact here, so reopening an unapproved run cannot arm the goal init deliberately withheld.
|
|
1807
|
+
try { this.resumeRunToGoal(agent, runId, this.readRunStartApproval(root, runId).approved) } catch { /* best-effort */ }
|
|
1808
|
+
return {
|
|
1809
|
+
artifact,
|
|
1810
|
+
runId,
|
|
1811
|
+
status: 'DRAFT',
|
|
1812
|
+
lockedAt: null,
|
|
1813
|
+
lockHash: null,
|
|
1814
|
+
blockers: hadReceipt || stale.length > 0 ? ['Downstream receipts invalidated'] : [],
|
|
1815
|
+
}
|
|
1816
|
+
}
|
|
1817
|
+
|
|
1818
|
+
/**
|
|
1819
|
+
* Lint an artifact with FULL canonical parity: runs the in-process ts-lint
|
|
1820
|
+
* port (ts-lint.ts `lintRun`) against the whole run and filters the verdict
|
|
1821
|
+
* lines to the requested artifact. `errors` = FAIL lines, `warnings` =
|
|
1822
|
+
* WARN lines, `passed` = no FAIL (WARN-only passes, matching the canonical
|
|
1823
|
+
* non-strict verdict).
|
|
1824
|
+
*/
|
|
1825
|
+
async lintArtifact(runId: string, artifact?: string, agent?: { session?: { header?: { cwd?: string } } } | null): Promise<LintArtifactResult> {
|
|
1826
|
+
const root = await this.resolveRootFor(agent)
|
|
1827
|
+
if (!root) throw new Error('cannot resolve workspace control-plane root for this session')
|
|
1828
|
+
const runDir = join(root, '.recursive', 'run', runId)
|
|
1829
|
+
const target = artifact ?? getNextLegalPhase(runDir) ?? '00-requirements.md'
|
|
1830
|
+
const artifactPath = join(runDir, target)
|
|
1831
|
+
if (!existsSync(artifactPath)) {
|
|
1832
|
+
return { artifact: target, runId, errors: ['Artifact not found'], warnings: [], passed: false }
|
|
1833
|
+
}
|
|
1834
|
+
const linted = await runTracked(this.jobs, {
|
|
1835
|
+
kind: 'lint',
|
|
1836
|
+
// The label is what the board shows, so it names the run AND the artifact.
|
|
1837
|
+
label: 'lint ' + runId + ' ' + target,
|
|
1838
|
+
// NO TIMEOUT HERE, deliberately: a deadline would silently convert a slow-but-working
|
|
1839
|
+
// lint into a failure, which is a policy change rather than visibility. The job gives
|
|
1840
|
+
// the board a running entry and a kill switch, which is what the item asks for;
|
|
1841
|
+
// `runTracked`'s `timeoutMs` stays available to a caller who wants a deadline.
|
|
1842
|
+
run: async ({ report, signal }) => {
|
|
1843
|
+
report('linting ' + target)
|
|
1844
|
+
// A synchronous in-process linter cannot be interrupted mid-call, so the signal is
|
|
1845
|
+
// honoured at the BOUNDARY. Stated plainly rather than implied: a kill settles the
|
|
1846
|
+
// JOB — the board and the caller stop waiting — but a hung `lintRun` call itself
|
|
1847
|
+
// runs to completion in this process, because JavaScript cannot pre-empt it.
|
|
1848
|
+
if (signal.aborted) throw new Error('lint cancelled before it started')
|
|
1849
|
+
return lintRun(root, runId)
|
|
1850
|
+
},
|
|
1851
|
+
})
|
|
1852
|
+
// T39: record the run in the RUN LAYER, where the run id is known — see job-log.ts for why
|
|
1853
|
+
// this is a file rather than an event subscription (an event carries a job id, not the run
|
|
1854
|
+
// it belongs to, and parsing run ids back out of labels is a mapping that breaks silently).
|
|
1855
|
+
recordJobRun(root, runId, { kind: 'lint', label: 'lint ' + runId + ' ' + target, result: linted })
|
|
1856
|
+
if (linted.status !== 'completed' || linted.value === undefined) {
|
|
1857
|
+
// A killed or failed job is reported through the EXISTING error path, so the tool's
|
|
1858
|
+
// payload shape does not change and every existing assertion still holds.
|
|
1859
|
+
return {
|
|
1860
|
+
artifact: target,
|
|
1861
|
+
runId,
|
|
1862
|
+
errors: [linted.detail ?? linted.error ?? 'lint did not complete'],
|
|
1863
|
+
warnings: [],
|
|
1864
|
+
passed: false,
|
|
1865
|
+
}
|
|
1866
|
+
}
|
|
1867
|
+
const stdout = [...linted.value.errors, ...linted.value.warnings].join('\n')
|
|
1868
|
+
return parseLintOutput(stdout, target, runId)
|
|
1869
|
+
}
|
|
1870
|
+
|
|
1871
|
+
/** Legacy structural fallback (kept for type reference; lintArtifact uses lintRun). */
|
|
1872
|
+
async lintArtifactStructuralFallback(runId: string, artifact: string, agent?: { session?: { header?: { cwd?: string } } } | null): Promise<LintArtifactResult> {
|
|
1873
|
+
const root = await this.resolveRootFor(agent)
|
|
1874
|
+
if (!root) throw new Error('cannot resolve workspace control-plane root for this session')
|
|
1875
|
+
const runDir = join(root, '.recursive', 'run', runId)
|
|
1876
|
+
const target = artifact ?? getNextLegalPhase(runDir) ?? '00-requirements.md'
|
|
1877
|
+
const artifactPath = join(runDir, target)
|
|
1878
|
+
if (!existsSync(artifactPath)) {
|
|
1879
|
+
return { artifact: target, runId, errors: ['Artifact not found'], warnings: [], passed: false }
|
|
1880
|
+
}
|
|
1881
|
+
// Structural fallback (no vendored script available).
|
|
1882
|
+
const content = readFileSync(artifactPath, 'utf8')
|
|
1883
|
+
const errors: string[] = []
|
|
1884
|
+
const warnings: string[] = []
|
|
1885
|
+
const status = getLockStatus(artifactPath)
|
|
1886
|
+
if (status === 'MISSING') errors.push('File missing')
|
|
1887
|
+
else if (status === 'DRAFT') warnings.push('Artifact is DRAFT (not locked)')
|
|
1888
|
+
else if (status === 'STALE_LOCK') errors.push('LockHash mismatch or missing lock fields')
|
|
1889
|
+
if (!/^[ \t]*## TODO[ \t]*$/m.test(content)) errors.push('Missing ## TODO section')
|
|
1890
|
+
for (const gate of ['Coverage', 'Approval']) {
|
|
1891
|
+
if (!new RegExp('^[ \t]*' + gate + ':\s*(PASS|FAIL)\s*$', 'm').test(content)) warnings.push('Missing ' + gate + ' gate')
|
|
1892
|
+
}
|
|
1893
|
+
return { artifact: target, runId, errors, warnings, passed: errors.length === 0 }
|
|
1894
|
+
}
|
|
1895
|
+
|
|
1896
|
+
/** Phase C R4: Layer 2 tool guard decision (caller of the transition set). */
|
|
1897
|
+
guardTool(exec: ToolExecLike, root: string, runId: string, config?: EnforcementConfig): ToolGuardDecision {
|
|
1898
|
+
const mode = (config ?? this.enforcementConfig).toolGuards
|
|
1899
|
+
return evaluateToolGuard(exec, root, runId, mode)
|
|
1900
|
+
}
|
|
1901
|
+
|
|
1902
|
+
/** Phase C R8: fs/observed tamper detection. */
|
|
1903
|
+
detectTamper(targetPath: string, root: string, runId: string) {
|
|
1904
|
+
return detectTamper(targetPath, root, runId)
|
|
1905
|
+
}
|
|
1906
|
+
|
|
1907
|
+
/** Phase C R5: render the current-phase policy contract. */
|
|
1908
|
+
renderPolicy(root: string, runId: string, folded: RecursivePhaseState | null): string {
|
|
1909
|
+
return renderRecursivePolicy({ worktreeRoot: root, runId, folded, config: this.enforcementConfig } as PolicyContext)
|
|
1910
|
+
}
|
|
1911
|
+
|
|
1912
|
+
/** Phase C R6: couple a gate-block to the goal service (graceful no-op). */
|
|
1913
|
+
coupleGateBlockToGoal(goalService: unknown, agent: unknown, ref: unknown, reason: { code: string; message: string }): boolean {
|
|
1914
|
+
return coupleGateBlockToGoal(goalService as never, agent, ref, reason)
|
|
1915
|
+
}
|
|
1916
|
+
|
|
1917
|
+
/** Phase C R7: resolve the enforcement config (strict|advisory, default strict). */
|
|
1918
|
+
get enforcementConfig(): EnforcementConfig {
|
|
1919
|
+
return this._enforcementConfig ?? DEFAULT_ENFORCEMENT
|
|
1920
|
+
}
|
|
1921
|
+
|
|
1922
|
+
setEnforcementConfig(config: unknown): EnforcementConfig {
|
|
1923
|
+
this._enforcementConfig = resolveEnforcementConfig(config)
|
|
1924
|
+
return this._enforcementConfig
|
|
1925
|
+
}
|
|
1926
|
+
}
|
|
1927
|
+
|
|
1928
|
+
/** set_or_insert_field: replace first field occurrence, else insert after the last listed after-field. */
|
|
1929
|
+
function setOrInsertField(content: string, fieldName: string, value: string, afterFields: string[]): string {
|
|
1930
|
+
const line = fieldName + ': `' + value + '`'
|
|
1931
|
+
const fieldRe = new RegExp('^[ \\t]*(?:[-*][ \\t]+)?' + escapeRegExp(fieldName) + ':\\s*.*$', 'm')
|
|
1932
|
+
if (fieldRe.test(content)) {
|
|
1933
|
+
return content.replace(fieldRe, line)
|
|
1934
|
+
}
|
|
1935
|
+
const lines = content.replace(/\r\n/g, '\n').replace(/\r/g, '\n').split('\n')
|
|
1936
|
+
let insertAt = 0
|
|
1937
|
+
for (const afterField of afterFields) {
|
|
1938
|
+
const afterRe = new RegExp('^[ \\t]*(?:[-*][ \\t]+)?' + escapeRegExp(afterField) + ':\\s*.*$', 'm')
|
|
1939
|
+
for (let i = 0; i < lines.length; i++) {
|
|
1940
|
+
if (afterRe.test(lines[i])) insertAt = Math.max(insertAt, i + 1)
|
|
1941
|
+
}
|
|
1942
|
+
}
|
|
1943
|
+
lines.splice(insertAt, 0, line)
|
|
1944
|
+
return lines.join('\n')
|
|
1945
|
+
}
|
|
1946
|
+
|
|
1947
|
+
function escapeRegExp(s: string): string {
|
|
1948
|
+
return s.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')
|
|
1949
|
+
}
|
|
1950
|
+
|
|
1951
|
+
/**
|
|
1952
|
+
* Parse the vendored lint-recursive-run.py stdout into a LintArtifactResult.
|
|
1953
|
+
* [FAIL] lines -> errors, [WARN] lines -> warnings. When the run directory has
|
|
1954
|
+
* multiple artifacts, FAIL/WARN lines carry the artifact path; we keep only the
|
|
1955
|
+
* lines for the requested artifact (or all when the path is ambiguous).
|
|
1956
|
+
*/
|
|
1957
|
+
function parseLintOutput(stdout: string, target: string, runId: string): LintArtifactResult {
|
|
1958
|
+
const errors: string[] = []
|
|
1959
|
+
const warnings: string[] = []
|
|
1960
|
+
const failRe = /^\[FAIL\]\s+(.*)$/gm
|
|
1961
|
+
const warnRe = /^\[WARN\]\s+(.*)$/gm
|
|
1962
|
+
const targetSuffix = target.replace(/^.*[\\/]/, '')
|
|
1963
|
+
const isForTarget = (line: string) => {
|
|
1964
|
+
const pathMatch = line.match(/[A-Za-z0-9_.-]+\.md/)
|
|
1965
|
+
return !pathMatch || pathMatch[0] === targetSuffix
|
|
1966
|
+
}
|
|
1967
|
+
let m: RegExpExecArray | null
|
|
1968
|
+
while ((m = failRe.exec(stdout)) !== null) {
|
|
1969
|
+
if (isForTarget(m[1])) errors.push(m[1].trim())
|
|
1970
|
+
}
|
|
1971
|
+
while ((m = warnRe.exec(stdout)) !== null) {
|
|
1972
|
+
if (isForTarget(m[1])) warnings.push(m[1].trim())
|
|
1973
|
+
}
|
|
1974
|
+
// If the target artifact appears nowhere in the output but the run was linted,
|
|
1975
|
+
// keep the parsed errors/warnings (canonical script lints the whole run).
|
|
1976
|
+
return { artifact: target, runId, errors, warnings, passed: errors.length === 0 }
|
|
1977
|
+
}
|
|
1978
|
+
|