@try-works/dsh-recursive-mode 0.3.0 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +959 -0
- package/lib/client.js +9 -2
- package/lib/closeout-report.d.ts +113 -0
- package/lib/closeout-standards.d.ts +35 -0
- package/lib/closeout.d.ts +12 -0
- package/lib/commands.d.ts +1 -1
- package/lib/config.d.ts +202 -0
- package/lib/delegation.d.ts +123 -3
- package/lib/enforcement.d.ts +90 -1
- package/lib/errors.d.ts +168 -0
- package/lib/git-context.d.ts +17 -0
- package/lib/guard-log.d.ts +39 -0
- package/lib/handoff.d.ts +29 -0
- package/lib/hooks.d.ts +103 -0
- package/lib/identity.d.ts +61 -0
- package/lib/index.d.ts +33 -12
- package/lib/index.js +10017 -3969
- package/lib/job-log.d.ts +34 -0
- package/lib/jobs-runner.d.ts +105 -0
- package/lib/json-safe.d.ts +33 -0
- package/lib/lock.d.ts +42 -0
- package/lib/memory-feedback.d.ts +52 -0
- package/lib/memory-select.d.ts +78 -0
- package/lib/memory.d.ts +137 -0
- package/lib/model-inventory.d.ts +106 -0
- package/lib/phase-graph.d.ts +111 -0
- package/lib/phase-rules.d.ts +67 -8
- package/lib/plan-gate.d.ts +68 -0
- package/lib/policy-globs.d.ts +222 -0
- package/lib/policy-write.d.ts +42 -0
- package/lib/policy.d.ts +39 -0
- package/lib/recursive_ask.tool.d.ts +88 -0
- package/lib/recursive_closeout.tool.d.ts +1 -1
- package/lib/recursive_delegate.tool.d.ts +22 -0
- package/lib/recursive_preview.tool.d.ts +48 -0
- package/lib/recursive_review.tool.d.ts +28 -0
- package/lib/result-cap.d.ts +70 -0
- package/lib/review-round.d.ts +82 -0
- package/lib/review.d.ts +9 -0
- package/lib/role-route.d.ts +122 -0
- package/lib/router.d.ts +90 -5
- package/lib/runtime.d.ts +252 -12
- package/lib/settlement.d.ts +132 -0
- package/lib/skills-phase.d.ts +71 -0
- package/lib/skills.d.ts +70 -0
- package/lib/status.d.ts +53 -1
- package/lib/teams-loop.d.ts +91 -2
- package/lib/training.d.ts +211 -0
- package/lib/ts-lint.d.ts +15 -0
- package/lib/types.d.ts +48 -0
- package/lib/workflow-audit.d.ts +207 -0
- package/package.json +31 -31
- package/preset/recursive.patch.yml +312 -0
- package/scripts/e2e-run.mjs +51 -0
- package/scripts/link-dsh.mjs +233 -0
- package/scripts/live/fake-llm.mjs +150 -0
- package/scripts/live-session-plugin.mjs +179 -0
- package/scripts/live-session-stock.mjs +106 -0
- package/scripts/live-session.mjs +139 -0
- package/skills/recursive-mode/SKILL.md +66 -0
- package/src/client/derive.ts +18 -2
- package/src/closeout-report.ts +274 -0
- package/src/closeout-standards.ts +102 -0
- package/src/closeout.ts +39 -2
- package/src/commands.ts +116 -4
- package/src/config.ts +113 -0
- package/src/delegation.ts +336 -18
- package/src/enforcement.ts +262 -72
- package/src/errors.ts +197 -0
- package/src/git-context.ts +33 -2
- package/src/guard-log.ts +134 -0
- package/src/handoff.ts +62 -0
- package/src/hooks.ts +316 -0
- package/src/identity.ts +230 -0
- package/src/index.ts +394 -20
- package/src/job-log.ts +112 -0
- package/src/jobs-runner.ts +222 -0
- package/src/json-safe.ts +75 -0
- package/src/lock.ts +153 -16
- package/src/memory-feedback.ts +185 -0
- package/src/memory-select.ts +187 -0
- package/src/memory.ts +309 -0
- package/src/model-inventory.ts +196 -0
- package/src/phase-graph.ts +191 -0
- package/src/phase-rules.ts +236 -0
- package/src/plan-gate.ts +111 -0
- package/src/policy-globs.ts +636 -0
- package/src/policy-write.ts +210 -0
- package/src/policy.ts +70 -5
- package/src/recursive_ask.tool.ts +276 -0
- package/src/recursive_audit_team.tool.ts +7 -3
- package/src/recursive_closeout.tool.ts +36 -35
- package/src/recursive_delegate.tool.ts +194 -0
- package/src/recursive_init.tool.ts +4 -3
- package/src/recursive_lint.tool.ts +81 -6
- package/src/recursive_lock.tool.ts +21 -4
- package/src/recursive_phase.tool.ts +3 -2
- package/src/recursive_preview.tool.ts +142 -0
- package/src/recursive_review.tool.ts +190 -0
- package/src/recursive_scratch.tool.ts +5 -4
- package/src/recursive_status.tool.ts +3 -2
- package/src/recursive_worktree.tool.ts +6 -5
- package/src/result-cap.ts +130 -0
- package/src/review-round.ts +335 -0
- package/src/review.ts +17 -3
- package/src/role-route.ts +230 -0
- package/src/router.ts +128 -2
- package/src/runtime.ts +968 -39
- package/src/settlement.ts +355 -0
- package/src/skills-phase.ts +143 -0
- package/src/skills.ts +151 -0
- package/src/snapshot.ts +39 -8
- package/src/status.ts +209 -4
- package/src/teams-loop.ts +223 -9
- package/src/training.ts +565 -0
- package/src/ts-lint.ts +38 -4
- package/src/types.ts +51 -0
- package/src/workflow-audit.ts +288 -0
- package/scripts/install-preset.cmd +0 -7
- package/scripts/install-preset.js +0 -101
package/lib/client.js
CHANGED
|
@@ -153,6 +153,10 @@ window.__ModuleLoader__.load({
|
|
|
153
153
|
if (best === null) return "0";
|
|
154
154
|
return laneOf(best + "-x") ?? "0";
|
|
155
155
|
}
|
|
156
|
+
/** T21: the single derived positions present on a card's rows, in phase order. */
|
|
157
|
+
function rowPositions(card) {
|
|
158
|
+
return Object.values(card.phases).map((row) => row.position).filter((p) => p !== void 0);
|
|
159
|
+
}
|
|
156
160
|
/** Derive §11.4 presentation facts from one card (no fs, no session). */
|
|
157
161
|
function cardFacts(card) {
|
|
158
162
|
const groups = /* @__PURE__ */ new Set();
|
|
@@ -184,10 +188,13 @@ window.__ModuleLoader__.load({
|
|
|
184
188
|
currentPhase = key;
|
|
185
189
|
}
|
|
186
190
|
}
|
|
187
|
-
const tampered = Object.keys(card.tampers).length > 0;
|
|
191
|
+
const tampered = Object.keys(card.tampers).length > 0 || rowPositions(card).includes("tampered");
|
|
188
192
|
const allMandatoryLocked = MANDATORY_GROUPS.every((g) => {
|
|
189
193
|
const row = Object.entries(card.phases).find(([key]) => phaseGroupOf(key) === g);
|
|
190
|
-
|
|
194
|
+
if (row === void 0) return false;
|
|
195
|
+
const phase = row[1];
|
|
196
|
+
if (phase.position !== void 0) return phase.position === "locked";
|
|
197
|
+
return phase.status === "LOCKED";
|
|
191
198
|
});
|
|
192
199
|
const lockValidity = tampered ? "tampered" : allMandatoryLocked ? "locked" : "in-progress";
|
|
193
200
|
return {
|
|
@@ -0,0 +1,113 @@
|
|
|
1
|
+
/** The phase → artifact map. A LOOKUP, not a template: nothing here is ever rendered into a file. */
|
|
2
|
+
export declare const CLOSEOUT_PHASE_FILES: Record<string, {
|
|
3
|
+
file: string;
|
|
4
|
+
label: string;
|
|
5
|
+
}>;
|
|
6
|
+
/** One thing the agent must add or fix before the artifact can lock. */
|
|
7
|
+
export interface CloseoutFinding {
|
|
8
|
+
kind: 'missing-artifact' | 'missing-section' | 'gate-not-passing';
|
|
9
|
+
/** Directory-ready: what is wrong, in the words the agent should act on. */
|
|
10
|
+
detail: string;
|
|
11
|
+
}
|
|
12
|
+
export interface CloseoutReport {
|
|
13
|
+
phase: string;
|
|
14
|
+
artifact: string;
|
|
15
|
+
label: string;
|
|
16
|
+
exists: boolean;
|
|
17
|
+
/** `LOCKED`, `DRAFT`, `STALE_LOCK`, `MISSING` — from the same reader the lock tool uses. */
|
|
18
|
+
status: string;
|
|
19
|
+
/** Empty means the artifact fits the standard; the caller tells the agent only when it is not. */
|
|
20
|
+
findings: CloseoutFinding[];
|
|
21
|
+
/**
|
|
22
|
+
* Advisory only: earlier artifacts that are not yet LOCKED. **Not a refusal** — the reference treats these
|
|
23
|
+
* as warnings (*"hard enforcement happens at lock time"*), and a closeout that refuses here cannot do the
|
|
24
|
+
* one thing it exists for.
|
|
25
|
+
*/
|
|
26
|
+
prerequisites: Array<{
|
|
27
|
+
artifact: string;
|
|
28
|
+
status: string;
|
|
29
|
+
}>;
|
|
30
|
+
/** The rules text to hand the agent, straight from the rules module. */
|
|
31
|
+
guidance: string[];
|
|
32
|
+
/** Addenda found anywhere in the run tree — cited as evidence, never written to. */
|
|
33
|
+
addenda: string[];
|
|
34
|
+
}
|
|
35
|
+
export interface CloseoutReportOptions {
|
|
36
|
+
workflowProfile?: string;
|
|
37
|
+
/** Include the advisory prerequisite list. Default true. */
|
|
38
|
+
checkPrerequisites?: boolean;
|
|
39
|
+
}
|
|
40
|
+
/**
|
|
41
|
+
* Examine one phase artifact and report what it is missing. **Reads only.**
|
|
42
|
+
*
|
|
43
|
+
* An absent artifact is a finding rather than an error: the closeout's job is to say what should be there,
|
|
44
|
+
* and "not written yet" is the most useful thing it can tell an agent at phase entry.
|
|
45
|
+
*/
|
|
46
|
+
export declare function closeoutReport(runDir: string, phase: string, options?: CloseoutReportOptions): CloseoutReport;
|
|
47
|
+
/** Anything in the run's artifact sequence that this report does not cover, for a whole-run sweep. */
|
|
48
|
+
export declare function uncoveredArtifacts(): string[];
|
|
49
|
+
/**
|
|
50
|
+
* Where a closeout receipt lives: **its own file, beside the lock receipts, never over the artifact.**
|
|
51
|
+
*
|
|
52
|
+
* ⚠ FOLLOWS `lock.ts`'s `receiptPath`, which is the convention this plugin already has: a receipt is a JSON
|
|
53
|
+
* file under `<runDir>/locks/` named after the artifact stem. The closeout receipt deliberately does NOT
|
|
54
|
+
* reuse `<stem>.receipt.json` — that name belongs to the LOCK receipt, which carries a hash and is read by
|
|
55
|
+
* `getLockStatus`. A distinct suffix keeps them together without a collision.
|
|
56
|
+
*/
|
|
57
|
+
export declare function closeoutReceiptPath(runDir: string, phase: string): string;
|
|
58
|
+
/**
|
|
59
|
+
* Record the report as a closeout receipt. **The only write in this module, and it is never the artifact.**
|
|
60
|
+
*
|
|
61
|
+
* ⚠ WHY A RECEIPT AT ALL, in the user's words: a closeout *"should potentially create a close out receipt,
|
|
62
|
+
* that is ok and valuable, as long as it doesnt overwrite the phase docs."* So the durable trace of "the
|
|
63
|
+
* closeout examined this phase" is a file of its own, and the phase document is read and reported on but
|
|
64
|
+
* never touched.
|
|
65
|
+
*
|
|
66
|
+
* The JSON is a plain projection of the report — no timestamp, so the receipt is a pure function of the run
|
|
67
|
+
* state and two calls on the same run produce identical bytes.
|
|
68
|
+
*/
|
|
69
|
+
export declare function writeCloseoutReceipt(runDir: string, phase: string, options?: CloseoutReportOptions): {
|
|
70
|
+
path: string;
|
|
71
|
+
report: CloseoutReport;
|
|
72
|
+
runReceipt: {
|
|
73
|
+
path: string;
|
|
74
|
+
report: RunCloseoutReport;
|
|
75
|
+
} | null;
|
|
76
|
+
};
|
|
77
|
+
/** Where the RUN-level closeout receipt lives: the run root, one per run. */
|
|
78
|
+
export declare function runCloseoutReceiptPath(runDir: string): string;
|
|
79
|
+
/** One artifact's line in the run-level receipt. */
|
|
80
|
+
export interface RunArtifactState {
|
|
81
|
+
artifact: string;
|
|
82
|
+
exists: boolean;
|
|
83
|
+
/** `LOCKED`, `DRAFT`, `STALE_LOCK` — or `ABSENT` when the artifact was never written. */
|
|
84
|
+
status: string;
|
|
85
|
+
/** Empty when this artifact fits the standard. */
|
|
86
|
+
findings: CloseoutFinding[];
|
|
87
|
+
}
|
|
88
|
+
export interface RunCloseoutReport {
|
|
89
|
+
/** Every artifact in the run's sequence, in order — 00 through 08, not only the closeout phases. */
|
|
90
|
+
artifacts: RunArtifactState[];
|
|
91
|
+
/** How many of them fit the standard, for a one-line summary. */
|
|
92
|
+
conforming: number;
|
|
93
|
+
}
|
|
94
|
+
/**
|
|
95
|
+
* The RUN-level report: every artifact in the run, not just the phases the closeout reports on.
|
|
96
|
+
*
|
|
97
|
+
* ⚠ WHY IT SPANS 00–08: the user's point, and it is the right one — a receipt *"represents what was done in
|
|
98
|
+
* the run"*, so it must include the early artifacts. The earlier correction still holds and is not undone by
|
|
99
|
+
* this: the closeout must never WRITE those documents (`recursive_init` owns them). It READS them, holds them
|
|
100
|
+
* to the same standard, and records their state. Reading a broad set and writing a narrow one is exactly the
|
|
101
|
+
* distinction that was missing.
|
|
102
|
+
*/
|
|
103
|
+
export declare function runCloseoutReport(runDir: string, options?: CloseoutReportOptions): RunCloseoutReport;
|
|
104
|
+
/**
|
|
105
|
+
* Record the run-level receipt. **Its own file at the run root**, never a phase document.
|
|
106
|
+
*
|
|
107
|
+
* The JSON is a pure projection of the run state — no timestamp — so two calls on the same run produce
|
|
108
|
+
* identical bytes, which is what makes it comparable between runs.
|
|
109
|
+
*/
|
|
110
|
+
export declare function writeRunCloseoutReceipt(runDir: string, options?: CloseoutReportOptions): {
|
|
111
|
+
path: string;
|
|
112
|
+
report: RunCloseoutReport;
|
|
113
|
+
};
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
export interface StandardViolation {
|
|
2
|
+
/** Run-relative artifact name, or the sequence entry that was absent. */
|
|
3
|
+
artifact: string;
|
|
4
|
+
kind: 'missing-artifact' | 'missing-section' | 'gate-not-passing' | 'not-locked';
|
|
5
|
+
/** What is wrong, spelled so a caller can print it without re-deriving anything. */
|
|
6
|
+
detail: string;
|
|
7
|
+
}
|
|
8
|
+
export interface StandardsReport {
|
|
9
|
+
/** How many artifacts from the sequence were present and examined. */
|
|
10
|
+
checked: number;
|
|
11
|
+
/** Artifacts in the sequence that do not exist yet — informational, not necessarily a failure. */
|
|
12
|
+
absent: string[];
|
|
13
|
+
violations: StandardViolation[];
|
|
14
|
+
}
|
|
15
|
+
export interface StandardsOptions {
|
|
16
|
+
/** Workflow profile whose rules apply; defaults to the current one, as every caller does. */
|
|
17
|
+
workflowProfile?: string;
|
|
18
|
+
/**
|
|
19
|
+
* Report artifacts that are present but not LOCKED. **OFF by default**: a closeout runs mid-run and a DRAFT
|
|
20
|
+
* artifact is the normal state, so treating it as a violation would make the check useless exactly when it
|
|
21
|
+
* is most wanted. Callers that need an all-locked assertion ask for it.
|
|
22
|
+
*/
|
|
23
|
+
requireLocked?: boolean;
|
|
24
|
+
}
|
|
25
|
+
/**
|
|
26
|
+
* Examine every artifact in the run against the phase rules.
|
|
27
|
+
*
|
|
28
|
+
* Reports, per present artifact: required headings that are missing, gate lines that are missing or FAIL, and
|
|
29
|
+
* (only when `requireLocked`) a status that is not LOCKED. Absent artifacts are listed separately — during a
|
|
30
|
+
* closeout most of the sequence legitimately does not exist yet, and conflating "not written yet" with
|
|
31
|
+
* "written wrong" would bury the findings that matter.
|
|
32
|
+
*/
|
|
33
|
+
export declare function verifyRunStandards(runDir: string, options?: StandardsOptions): StandardsReport;
|
|
34
|
+
/** One line per violation, for a tool result or a CLI print. Empty when the run fits the standard. */
|
|
35
|
+
export declare function formatStandardsReport(report: StandardsReport): string[];
|
package/lib/closeout.d.ts
CHANGED
|
@@ -10,6 +10,12 @@ export interface CloseoutResult {
|
|
|
10
10
|
file: string;
|
|
11
11
|
created: string[];
|
|
12
12
|
existing: string[];
|
|
13
|
+
/**
|
|
14
|
+
* Every addendum found anywhere in the run tree, as run-relative paths — and cited in the scaffolded
|
|
15
|
+
* receipt. Optional on the TYPE so a caller that builds its own result literal still compiles; the
|
|
16
|
+
* closeout itself always populates it.
|
|
17
|
+
*/
|
|
18
|
+
addenda?: string[];
|
|
13
19
|
}
|
|
14
20
|
export interface CloseoutOptions {
|
|
15
21
|
/** Enforce prerequisite-lock gating (default true). */
|
|
@@ -19,5 +25,11 @@ export interface CloseoutOptions {
|
|
|
19
25
|
* Create or update a closeout receipt stub for the given phase.
|
|
20
26
|
* Returns the file + created/existing lists. When strict, refuses phases whose
|
|
21
27
|
* prerequisite artifacts are not all LOCKED (reuses lock.ts chain validation).
|
|
28
|
+
*
|
|
29
|
+
* ⚠ ADDENDA ARE PART OF THE CLOSEOUT WORK, and they are CITED, never written. An addendum closes a gap in
|
|
30
|
+
* an artifact that is already locked, which is precisely the kind of thing a receipt must account for — and
|
|
31
|
+
* they are not always in `addenda/`: the real ones sit beside the artifact they close, in the run ROOT. So
|
|
32
|
+
* the receipt lists every addendum in the run tree (see {@link getRunTreeAddenda}), and says so explicitly
|
|
33
|
+
* when there are none, because "we looked and found none" and "we never looked" must not read alike.
|
|
22
34
|
*/
|
|
23
35
|
export declare function closeoutPhase(runDir: string, phase: string, opts?: CloseoutOptions): CloseoutResult;
|
package/lib/commands.d.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import type { RecursiveRuntime } from './runtime.ts';
|
|
2
|
-
export type RecursiveVerb = 'status' | 'spec' | 'worktree' | 'init' | 'lock' | 'qa' | 'closeout' | 'addendum' | 'review' | 'scratch' | 'bootstrap' | 'list' | 'help';
|
|
2
|
+
export type RecursiveVerb = 'status' | 'spec' | 'worktree' | 'init' | 'lock' | 'qa' | 'closeout' | 'addendum' | 'review' | 'scratch' | 'memory' | 'model' | 'bootstrap' | 'list' | 'help';
|
|
3
3
|
export declare const PRESET_VERBS: RecursiveVerb[];
|
|
4
4
|
export declare const GLOBAL_VERBS: RecursiveVerb[];
|
|
5
5
|
export declare const ALL_VERBS: RecursiveVerb[];
|
package/lib/config.d.ts
ADDED
|
@@ -0,0 +1,202 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* T7 — the plugin's Config: the `recursive` settings namespace.
|
|
3
|
+
*
|
|
4
|
+
* HOW THE SETTINGS SERVICE ACTUALLY SEES THIS, measured against
|
|
5
|
+
* `@deepseek-ai/dsh-settings`'s own types rather than assumed: `describe()` "read[s]
|
|
6
|
+
* active plugin schemas and their live values", and `update(ns, patch, expectedRevision)`
|
|
7
|
+
* "merge[s] editable fields into an entry's config". There is **no `register(...)` call
|
|
8
|
+
* for a plugin to make** — the item's `registerEnforcementSettings(settings)` wording
|
|
9
|
+
* describes a seam that does not exist. **Declaring this schema IS the registration:**
|
|
10
|
+
* the Loader owns the profile entry, the service projects the schema into a form, and an
|
|
11
|
+
* edit re-applies the plugin with the new config. That re-application IS the hot reload,
|
|
12
|
+
* which is why there is no watcher here to keep in sync.
|
|
13
|
+
*
|
|
14
|
+
* VALIDATION IS NOT DUPLICATED. The schema is what the UI reads and what gives the form
|
|
15
|
+
* its shape; `resolveEnforcementConfig` remains the STRICT validator (unknown keys
|
|
16
|
+
* rejected, budgets required to be positive integers, fail-loud at load). A schema that
|
|
17
|
+
* silently coerced a bad value would be a second, weaker contract beside the real one.
|
|
18
|
+
*/
|
|
19
|
+
import z from '@deepseek-ai/schemastery';
|
|
20
|
+
export interface RecursiveModeBudgets {
|
|
21
|
+
maxAuditRounds?: number;
|
|
22
|
+
maxRepairAttempts?: number;
|
|
23
|
+
maxDelegationDepth?: number;
|
|
24
|
+
maxChildrenPerPhase?: number;
|
|
25
|
+
maxResultBytes?: number;
|
|
26
|
+
}
|
|
27
|
+
export interface RecursiveModeEnforcement {
|
|
28
|
+
preStep?: 'strict' | 'advisory';
|
|
29
|
+
toolGuards?: 'strict' | 'advisory';
|
|
30
|
+
tamper?: 'strict' | 'advisory';
|
|
31
|
+
budgets?: RecursiveModeBudgets;
|
|
32
|
+
}
|
|
33
|
+
export interface RecursiveModeConfig {
|
|
34
|
+
/** R4 shell split: expose ONLY the client bundle, registering nothing server-side. */
|
|
35
|
+
shellOnly?: boolean;
|
|
36
|
+
repoRoot?: string;
|
|
37
|
+
/** T7/T28: the enforcement modes and the budgets, editable live. */
|
|
38
|
+
enforcement?: RecursiveModeEnforcement;
|
|
39
|
+
/** T7: overrides for the workspace router file — see the schema comment on defaults. */
|
|
40
|
+
router?: {
|
|
41
|
+
defaults?: Record<string, string | number | boolean | undefined>;
|
|
42
|
+
};
|
|
43
|
+
}
|
|
44
|
+
/**
|
|
45
|
+
* The schema the settings service discovers. Field descriptions are the form's help
|
|
46
|
+
* text, so they are written for the person toggling them, not for the reader of this file.
|
|
47
|
+
*/
|
|
48
|
+
export declare const Config: z<Schemastery.ObjectS<NoInfer<{
|
|
49
|
+
shellOnly: z<boolean, boolean, "defined">;
|
|
50
|
+
repoRoot: z<string, string, "plain">;
|
|
51
|
+
enforcement: z<Schemastery.ObjectS<NoInfer<{
|
|
52
|
+
preStep: z<"strict" | "advisory", "strict" | "advisory", "defined">;
|
|
53
|
+
toolGuards: z<"strict" | "advisory", "strict" | "advisory", "defined">;
|
|
54
|
+
tamper: z<"strict" | "advisory", "strict" | "advisory", "defined">;
|
|
55
|
+
budgets: z<Schemastery.ObjectS<NoInfer<{
|
|
56
|
+
maxAuditRounds: z<number, number, "defined">;
|
|
57
|
+
maxRepairAttempts: z<number, number, "defined">;
|
|
58
|
+
maxDelegationDepth: z<number, number, "defined">;
|
|
59
|
+
maxChildrenPerPhase: z<number, number, "defined">;
|
|
60
|
+
maxResultBytes: z<number, number, "defined">;
|
|
61
|
+
}>>, Schemastery.ObjectT<NoInfer<{
|
|
62
|
+
maxAuditRounds: z<number, number, "defined">;
|
|
63
|
+
maxRepairAttempts: z<number, number, "defined">;
|
|
64
|
+
maxDelegationDepth: z<number, number, "defined">;
|
|
65
|
+
maxChildrenPerPhase: z<number, number, "defined">;
|
|
66
|
+
maxResultBytes: z<number, number, "defined">;
|
|
67
|
+
}>>, "plain">;
|
|
68
|
+
}>>, Schemastery.ObjectT<NoInfer<{
|
|
69
|
+
preStep: z<"strict" | "advisory", "strict" | "advisory", "defined">;
|
|
70
|
+
toolGuards: z<"strict" | "advisory", "strict" | "advisory", "defined">;
|
|
71
|
+
tamper: z<"strict" | "advisory", "strict" | "advisory", "defined">;
|
|
72
|
+
budgets: z<Schemastery.ObjectS<NoInfer<{
|
|
73
|
+
maxAuditRounds: z<number, number, "defined">;
|
|
74
|
+
maxRepairAttempts: z<number, number, "defined">;
|
|
75
|
+
maxDelegationDepth: z<number, number, "defined">;
|
|
76
|
+
maxChildrenPerPhase: z<number, number, "defined">;
|
|
77
|
+
maxResultBytes: z<number, number, "defined">;
|
|
78
|
+
}>>, Schemastery.ObjectT<NoInfer<{
|
|
79
|
+
maxAuditRounds: z<number, number, "defined">;
|
|
80
|
+
maxRepairAttempts: z<number, number, "defined">;
|
|
81
|
+
maxDelegationDepth: z<number, number, "defined">;
|
|
82
|
+
maxChildrenPerPhase: z<number, number, "defined">;
|
|
83
|
+
maxResultBytes: z<number, number, "defined">;
|
|
84
|
+
}>>, "plain">;
|
|
85
|
+
}>>, "plain">;
|
|
86
|
+
/**
|
|
87
|
+
* T7 — the ROUTER overrides. Deliberately NO defaults on these fields: `.default()`
|
|
88
|
+
* would make every field present, and a present field overrides the workspace's
|
|
89
|
+
* `recursive-router.json` — so defaulting them would silently shadow the declarative
|
|
90
|
+
* file forever. Absent means "defer to the file"; present means "override it".
|
|
91
|
+
*/
|
|
92
|
+
router: z<Schemastery.ObjectS<NoInfer<{
|
|
93
|
+
defaults: z<Schemastery.ObjectS<NoInfer<{
|
|
94
|
+
when_role_unconfigured: z<"ask" | "fallback-local", "ask" | "fallback-local", "plain">;
|
|
95
|
+
when_cli_unavailable: z<"ask" | "fallback-local", "ask" | "fallback-local", "plain">;
|
|
96
|
+
when_model_unknown: z<"ask" | "fallback-local", "ask" | "fallback-local", "plain">;
|
|
97
|
+
allow_auto_assign_if_single_cli: z<boolean, boolean, "plain">;
|
|
98
|
+
probe_timeout_ms: z<number, number, "plain">;
|
|
99
|
+
invoke_timeout_ms: z<number, number, "plain">;
|
|
100
|
+
}>>, Schemastery.ObjectT<NoInfer<{
|
|
101
|
+
when_role_unconfigured: z<"ask" | "fallback-local", "ask" | "fallback-local", "plain">;
|
|
102
|
+
when_cli_unavailable: z<"ask" | "fallback-local", "ask" | "fallback-local", "plain">;
|
|
103
|
+
when_model_unknown: z<"ask" | "fallback-local", "ask" | "fallback-local", "plain">;
|
|
104
|
+
allow_auto_assign_if_single_cli: z<boolean, boolean, "plain">;
|
|
105
|
+
probe_timeout_ms: z<number, number, "plain">;
|
|
106
|
+
invoke_timeout_ms: z<number, number, "plain">;
|
|
107
|
+
}>>, "plain">;
|
|
108
|
+
}>>, Schemastery.ObjectT<NoInfer<{
|
|
109
|
+
defaults: z<Schemastery.ObjectS<NoInfer<{
|
|
110
|
+
when_role_unconfigured: z<"ask" | "fallback-local", "ask" | "fallback-local", "plain">;
|
|
111
|
+
when_cli_unavailable: z<"ask" | "fallback-local", "ask" | "fallback-local", "plain">;
|
|
112
|
+
when_model_unknown: z<"ask" | "fallback-local", "ask" | "fallback-local", "plain">;
|
|
113
|
+
allow_auto_assign_if_single_cli: z<boolean, boolean, "plain">;
|
|
114
|
+
probe_timeout_ms: z<number, number, "plain">;
|
|
115
|
+
invoke_timeout_ms: z<number, number, "plain">;
|
|
116
|
+
}>>, Schemastery.ObjectT<NoInfer<{
|
|
117
|
+
when_role_unconfigured: z<"ask" | "fallback-local", "ask" | "fallback-local", "plain">;
|
|
118
|
+
when_cli_unavailable: z<"ask" | "fallback-local", "ask" | "fallback-local", "plain">;
|
|
119
|
+
when_model_unknown: z<"ask" | "fallback-local", "ask" | "fallback-local", "plain">;
|
|
120
|
+
allow_auto_assign_if_single_cli: z<boolean, boolean, "plain">;
|
|
121
|
+
probe_timeout_ms: z<number, number, "plain">;
|
|
122
|
+
invoke_timeout_ms: z<number, number, "plain">;
|
|
123
|
+
}>>, "plain">;
|
|
124
|
+
}>>, "plain">;
|
|
125
|
+
}>>, Schemastery.ObjectT<NoInfer<{
|
|
126
|
+
shellOnly: z<boolean, boolean, "defined">;
|
|
127
|
+
repoRoot: z<string, string, "plain">;
|
|
128
|
+
enforcement: z<Schemastery.ObjectS<NoInfer<{
|
|
129
|
+
preStep: z<"strict" | "advisory", "strict" | "advisory", "defined">;
|
|
130
|
+
toolGuards: z<"strict" | "advisory", "strict" | "advisory", "defined">;
|
|
131
|
+
tamper: z<"strict" | "advisory", "strict" | "advisory", "defined">;
|
|
132
|
+
budgets: z<Schemastery.ObjectS<NoInfer<{
|
|
133
|
+
maxAuditRounds: z<number, number, "defined">;
|
|
134
|
+
maxRepairAttempts: z<number, number, "defined">;
|
|
135
|
+
maxDelegationDepth: z<number, number, "defined">;
|
|
136
|
+
maxChildrenPerPhase: z<number, number, "defined">;
|
|
137
|
+
maxResultBytes: z<number, number, "defined">;
|
|
138
|
+
}>>, Schemastery.ObjectT<NoInfer<{
|
|
139
|
+
maxAuditRounds: z<number, number, "defined">;
|
|
140
|
+
maxRepairAttempts: z<number, number, "defined">;
|
|
141
|
+
maxDelegationDepth: z<number, number, "defined">;
|
|
142
|
+
maxChildrenPerPhase: z<number, number, "defined">;
|
|
143
|
+
maxResultBytes: z<number, number, "defined">;
|
|
144
|
+
}>>, "plain">;
|
|
145
|
+
}>>, Schemastery.ObjectT<NoInfer<{
|
|
146
|
+
preStep: z<"strict" | "advisory", "strict" | "advisory", "defined">;
|
|
147
|
+
toolGuards: z<"strict" | "advisory", "strict" | "advisory", "defined">;
|
|
148
|
+
tamper: z<"strict" | "advisory", "strict" | "advisory", "defined">;
|
|
149
|
+
budgets: z<Schemastery.ObjectS<NoInfer<{
|
|
150
|
+
maxAuditRounds: z<number, number, "defined">;
|
|
151
|
+
maxRepairAttempts: z<number, number, "defined">;
|
|
152
|
+
maxDelegationDepth: z<number, number, "defined">;
|
|
153
|
+
maxChildrenPerPhase: z<number, number, "defined">;
|
|
154
|
+
maxResultBytes: z<number, number, "defined">;
|
|
155
|
+
}>>, Schemastery.ObjectT<NoInfer<{
|
|
156
|
+
maxAuditRounds: z<number, number, "defined">;
|
|
157
|
+
maxRepairAttempts: z<number, number, "defined">;
|
|
158
|
+
maxDelegationDepth: z<number, number, "defined">;
|
|
159
|
+
maxChildrenPerPhase: z<number, number, "defined">;
|
|
160
|
+
maxResultBytes: z<number, number, "defined">;
|
|
161
|
+
}>>, "plain">;
|
|
162
|
+
}>>, "plain">;
|
|
163
|
+
/**
|
|
164
|
+
* T7 — the ROUTER overrides. Deliberately NO defaults on these fields: `.default()`
|
|
165
|
+
* would make every field present, and a present field overrides the workspace's
|
|
166
|
+
* `recursive-router.json` — so defaulting them would silently shadow the declarative
|
|
167
|
+
* file forever. Absent means "defer to the file"; present means "override it".
|
|
168
|
+
*/
|
|
169
|
+
router: z<Schemastery.ObjectS<NoInfer<{
|
|
170
|
+
defaults: z<Schemastery.ObjectS<NoInfer<{
|
|
171
|
+
when_role_unconfigured: z<"ask" | "fallback-local", "ask" | "fallback-local", "plain">;
|
|
172
|
+
when_cli_unavailable: z<"ask" | "fallback-local", "ask" | "fallback-local", "plain">;
|
|
173
|
+
when_model_unknown: z<"ask" | "fallback-local", "ask" | "fallback-local", "plain">;
|
|
174
|
+
allow_auto_assign_if_single_cli: z<boolean, boolean, "plain">;
|
|
175
|
+
probe_timeout_ms: z<number, number, "plain">;
|
|
176
|
+
invoke_timeout_ms: z<number, number, "plain">;
|
|
177
|
+
}>>, Schemastery.ObjectT<NoInfer<{
|
|
178
|
+
when_role_unconfigured: z<"ask" | "fallback-local", "ask" | "fallback-local", "plain">;
|
|
179
|
+
when_cli_unavailable: z<"ask" | "fallback-local", "ask" | "fallback-local", "plain">;
|
|
180
|
+
when_model_unknown: z<"ask" | "fallback-local", "ask" | "fallback-local", "plain">;
|
|
181
|
+
allow_auto_assign_if_single_cli: z<boolean, boolean, "plain">;
|
|
182
|
+
probe_timeout_ms: z<number, number, "plain">;
|
|
183
|
+
invoke_timeout_ms: z<number, number, "plain">;
|
|
184
|
+
}>>, "plain">;
|
|
185
|
+
}>>, Schemastery.ObjectT<NoInfer<{
|
|
186
|
+
defaults: z<Schemastery.ObjectS<NoInfer<{
|
|
187
|
+
when_role_unconfigured: z<"ask" | "fallback-local", "ask" | "fallback-local", "plain">;
|
|
188
|
+
when_cli_unavailable: z<"ask" | "fallback-local", "ask" | "fallback-local", "plain">;
|
|
189
|
+
when_model_unknown: z<"ask" | "fallback-local", "ask" | "fallback-local", "plain">;
|
|
190
|
+
allow_auto_assign_if_single_cli: z<boolean, boolean, "plain">;
|
|
191
|
+
probe_timeout_ms: z<number, number, "plain">;
|
|
192
|
+
invoke_timeout_ms: z<number, number, "plain">;
|
|
193
|
+
}>>, Schemastery.ObjectT<NoInfer<{
|
|
194
|
+
when_role_unconfigured: z<"ask" | "fallback-local", "ask" | "fallback-local", "plain">;
|
|
195
|
+
when_cli_unavailable: z<"ask" | "fallback-local", "ask" | "fallback-local", "plain">;
|
|
196
|
+
when_model_unknown: z<"ask" | "fallback-local", "ask" | "fallback-local", "plain">;
|
|
197
|
+
allow_auto_assign_if_single_cli: z<boolean, boolean, "plain">;
|
|
198
|
+
probe_timeout_ms: z<number, number, "plain">;
|
|
199
|
+
invoke_timeout_ms: z<number, number, "plain">;
|
|
200
|
+
}>>, "plain">;
|
|
201
|
+
}>>, "plain">;
|
|
202
|
+
}>>, "plain">;
|
package/lib/delegation.d.ts
CHANGED
|
@@ -57,6 +57,18 @@ export interface SubagentStartRequestLike {
|
|
|
57
57
|
persona?: string;
|
|
58
58
|
parent?: unknown;
|
|
59
59
|
signal?: unknown;
|
|
60
|
+
/**
|
|
61
|
+
* T9 — the child's provider/model overrides.
|
|
62
|
+
*
|
|
63
|
+
* ⚠ `SubagentStartRequest.agentOptions` is only valid for a provider that DECLARES
|
|
64
|
+
* `capabilities.agentOptions`; the harness REJECTS a start that sends it otherwise. Passing
|
|
65
|
+
* it unconditionally would therefore BREAK delegations on providers that do not support it,
|
|
66
|
+
* which is why the caller gates on the capability rather than on the model being non-null.
|
|
67
|
+
*/
|
|
68
|
+
agentOptions?: {
|
|
69
|
+
model?: string;
|
|
70
|
+
provider?: string;
|
|
71
|
+
};
|
|
60
72
|
}
|
|
61
73
|
export interface SubagentResultLike {
|
|
62
74
|
output?: string;
|
|
@@ -155,13 +167,77 @@ export interface ContinuableDelegationLike {
|
|
|
155
167
|
accepted: boolean;
|
|
156
168
|
/** True when the fallback one-shot `delegate()` was used (no continuable seam). */
|
|
157
169
|
fellBackToOneShot?: boolean;
|
|
170
|
+
/**
|
|
171
|
+
* True when the round ended because NO settlement has landed yet — the caller's
|
|
172
|
+
* signal to resume on a later turn with the SAME `childId`, not a failure. The
|
|
173
|
+
* harness offers no parent-side await-settlement promise, so this is the honest
|
|
174
|
+
* report of "the child is still working".
|
|
175
|
+
*/
|
|
176
|
+
parked?: boolean;
|
|
158
177
|
}
|
|
159
178
|
/** Verdict vocabulary shared by T3/T4 (matches the delegated review schema). */
|
|
160
179
|
export type DelegationVerdict = 'APPROVE' | 'REVISE' | 'REJECT';
|
|
180
|
+
/**
|
|
181
|
+
* T28 — how much delegation depth is left for a child of a parent at `parentDepth`.
|
|
182
|
+
*
|
|
183
|
+
* The configured maximum is a ceiling for the WHOLE recursion, not a fresh allowance
|
|
184
|
+
* at every level. Passing the configured maximum down unchanged at each level is how
|
|
185
|
+
* a "depth 3" budget silently permits 3^depth children, which bounds nothing.
|
|
186
|
+
*
|
|
187
|
+
* A caller may ask for LESS than what remains (`requested`) and never for more:
|
|
188
|
+
* `min(remaining, requested)`. A parent already at or past the cap yields **0** —
|
|
189
|
+
* "delegate no further" — never a negative that some downstream comparison could read
|
|
190
|
+
* as permission.
|
|
191
|
+
*/
|
|
192
|
+
export declare function remainingDepthFor(budgets: {
|
|
193
|
+
maxDelegationDepth: number;
|
|
194
|
+
}, parentDepth: number, requested?: number): number;
|
|
161
195
|
/** Read the verdict from a review-schema structured result (pure). */
|
|
162
196
|
export declare function readVerdictFromStructured(result: SubagentResultLike): DelegationVerdict;
|
|
163
197
|
/** Read the repair instruction from a review-schema structured result (pure). */
|
|
164
198
|
export declare function readRepairFromStructured(result: SubagentResultLike): string;
|
|
199
|
+
/** What a delegated child's `reply.md` said, as far as the plugin can tell. */
|
|
200
|
+
export interface ReplyVerdict {
|
|
201
|
+
/** The verdict the reply STATES, or null when it states none. */
|
|
202
|
+
verdict: DelegationVerdict | null;
|
|
203
|
+
/** Finding titles, when the reply carried review-schema JSON. */
|
|
204
|
+
findings: string[];
|
|
205
|
+
/** Why no verdict was read, for the repair instruction. Null when one was. */
|
|
206
|
+
problem: string | null;
|
|
207
|
+
}
|
|
208
|
+
/**
|
|
209
|
+
* Read a verdict out of a child's `reply.md`, FAILING CLOSED.
|
|
210
|
+
*
|
|
211
|
+
* WHY THIS EXISTS RATHER THAN `readVerdictFromStructured`. The settlement's closing
|
|
212
|
+
* text is free-form — a child may report prose, a fenced JSON block, or a bare
|
|
213
|
+
* field line — and the structured reader's fallback for "no verdict" is
|
|
214
|
+
* `APPROVE`. That default is defensible where the caller re-evaluates the result,
|
|
215
|
+
* but it is the wrong default for a REVIEW ROUND: a child that answered with prose,
|
|
216
|
+
* or answered the wrong question, or wrote nothing parseable, must never be read as
|
|
217
|
+
* having approved the work. Verification that fails open is not verification.
|
|
218
|
+
*
|
|
219
|
+
* So this reader accepts exactly three things — review-schema JSON (fenced or
|
|
220
|
+
* bare), or an explicit `Verdict:` field — and reports `verdict: null` plus a
|
|
221
|
+
* `problem` for anything else. The caller turns that into a REVISE with a repair
|
|
222
|
+
* instruction that says what was wrong, so an unreadable reply costs a round rather
|
|
223
|
+
* than a false approval.
|
|
224
|
+
*/
|
|
225
|
+
export declare function parseReplyVerdict(replyText: string): ReplyVerdict;
|
|
226
|
+
/**
|
|
227
|
+
* The verdict for one round, from the child's reply text, FAILING CLOSED: an
|
|
228
|
+
* unreadable reply becomes `REVISE`, never `APPROVE`.
|
|
229
|
+
*/
|
|
230
|
+
export declare function readVerdictFromReply(replyText: string): DelegationVerdict;
|
|
231
|
+
/**
|
|
232
|
+
* The repair instruction for a round whose reply did not approve.
|
|
233
|
+
*
|
|
234
|
+
* Findings drive it when the reply carried them (and are the ONLY source of
|
|
235
|
+
* instruction text — a child cannot inject instructions, since only the titles
|
|
236
|
+
* travel). When there is nothing to quote, the instruction states the contract
|
|
237
|
+
* violation instead of asking vaguely for "improvement", because a repair request
|
|
238
|
+
* that does not say what was wrong cannot be acted on.
|
|
239
|
+
*/
|
|
240
|
+
export declare function readRepairFromReply(replyText: string): string;
|
|
165
241
|
/**
|
|
166
242
|
* Run a multi-round delegated task on ONE durable continuable child:
|
|
167
243
|
* 1. `startContinuable` (initial prompt) — `start()` is never called.
|
|
@@ -183,10 +259,22 @@ export declare function delegateContinuable(input: {
|
|
|
183
259
|
toolFilter?: unknown;
|
|
184
260
|
maxDepth?: number;
|
|
185
261
|
childId?: ContinuableChildId;
|
|
262
|
+
/**
|
|
263
|
+
* T36: RESUME an existing durable child instead of starting one. The turn-shaped
|
|
264
|
+
* caller passes the childId from a previous `parked` round, which is what makes
|
|
265
|
+
* the loop resumable across turns — `startContinuable` is not called, so a parked
|
|
266
|
+
* round does not create a second child.
|
|
267
|
+
*/
|
|
268
|
+
resumeChild?: ContinuableChildId;
|
|
186
269
|
maxRounds?: number;
|
|
187
270
|
readVerdict?: (result: SubagentResultLike) => DelegationVerdict;
|
|
188
271
|
readRepair?: (result: SubagentResultLike) => string | undefined;
|
|
189
272
|
awaitRoundResult?: (childId: ContinuableChildId, messageId: ContinuableMessageId) => Promise<SubagentResultLike | null>;
|
|
273
|
+
/** T9: the child's provider/model overrides, forwarded onto the start request verbatim. */
|
|
274
|
+
agentOptions?: {
|
|
275
|
+
model?: string;
|
|
276
|
+
provider?: string;
|
|
277
|
+
};
|
|
190
278
|
}): Promise<ContinuableDelegationLike>;
|
|
191
279
|
/**
|
|
192
280
|
* T4 kill switch: interrupt one live continuable child's current turn. Admission
|
|
@@ -243,11 +331,27 @@ export interface ActionRecordInput {
|
|
|
243
331
|
findings?: string[];
|
|
244
332
|
success: boolean;
|
|
245
333
|
stopReason?: string;
|
|
334
|
+
/**
|
|
335
|
+
* ⚠ FU-9 — WHY IT FAILED, when it did. `success` is a boolean, so a record could say `Status: failed` and
|
|
336
|
+
* nothing else: a delegation that FAILED and a delegation that NEVER HAPPENED read identically, which is what
|
|
337
|
+
* let me conclude for three rounds that the host was not scheduling children. The caller ALREADY passed
|
|
338
|
+
* `stopReason`, and the live record said `n/a` — because there was no result to take a stop reason from.
|
|
339
|
+
*/
|
|
340
|
+
failure?: string;
|
|
246
341
|
}
|
|
247
342
|
/**
|
|
248
|
-
* Write a durable action record under subagents/
|
|
249
|
-
* (
|
|
250
|
-
*
|
|
343
|
+
* Write a durable action record under subagents/ in the shape this repo's own
|
|
344
|
+
* linter accepts (ts-lint.ts lintSubagentActionRecordFile — every top-level .md
|
|
345
|
+
* under a run's subagents/ is linted as one):
|
|
346
|
+
* - the literal title `# Subagent Action Record`;
|
|
347
|
+
* - `Run ID` and `Timestamp` in ## Metadata (Run ID must equal the run dir name);
|
|
348
|
+
* - `Current Artifact`, `Artifact Content Hash` (the artifact's LF-normalized
|
|
349
|
+
* sha256, derived from the artifact itself — no extra caller input), `Diff
|
|
350
|
+
* Basis`, `Review Bundle`, and the NAMED fields `Upstream Artifacts` /
|
|
351
|
+
* `Code Refs` strictly inside ## Inputs Provided. The linter resolves each of
|
|
352
|
+
* those through the heading body, so a field under another heading is not
|
|
353
|
+
* found at all.
|
|
354
|
+
* A success:false attempt is written with a failed status and is NOT accepted.
|
|
251
355
|
*/
|
|
252
356
|
export declare function writeActionRecord(input: ActionRecordInput): string;
|
|
253
357
|
/**
|
|
@@ -258,3 +362,19 @@ export declare function evaluateDelegationResult(result: SubagentResultLike): {
|
|
|
258
362
|
accepted: boolean;
|
|
259
363
|
reason: string;
|
|
260
364
|
};
|
|
365
|
+
/**
|
|
366
|
+
* T8 — the child's CLAIMED references, read from wherever the delegation put them.
|
|
367
|
+
*
|
|
368
|
+
* The review output schema requires `references`, so a reviewer states which files back its
|
|
369
|
+
* verdict — and NOTHING read that field: `evaluateDelegationResult` looks only at
|
|
370
|
+
* `success`/`stopReason`, so a review citing files that do not exist was indistinguishable
|
|
371
|
+
* from one citing real evidence. This reads the claims so they can be checked.
|
|
372
|
+
*
|
|
373
|
+
* Both carriers are tried, because a delegation may return structured output or the raw
|
|
374
|
+
* JSON text: `structured` first (the native path), then `output` parsed as JSON. A result
|
|
375
|
+
* that carries neither yields `[]` — "no claims" — which the caller treats as NOTHING TO
|
|
376
|
+
* CHECK rather than as a pass, so an unparseable result can never be mistaken for a
|
|
377
|
+
* verified one. Malformed entries are dropped rather than thrown on: this reads a model's
|
|
378
|
+
* output, which is untrusted.
|
|
379
|
+
*/
|
|
380
|
+
export declare function referencesFromResult(result: SubagentResultLike): Reference[];
|