sortie-dogs 0.13.1 → 0.13.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +35 -0
- package/dist/asset-version.d.ts +1 -1
- package/dist/asset-version.js +1 -1
- package/dist/core/contract-limits.d.ts +1 -1
- package/dist/core/contract-limits.js +1 -1
- package/dist/core/goal-bound.d.ts +7 -0
- package/dist/core/goal-bound.js +7 -1
- package/dist/core/operator-mission.d.ts +19 -0
- package/dist/core/operator-mission.js +25 -1
- package/dist/core/operator-runtime.d.ts +4 -2
- package/dist/core/operator-runtime.js +69 -20
- package/dist/core/scope-lease-registry.d.ts +1 -0
- package/dist/core/scope-lease-registry.js +23 -1
- package/dist/plugin/index.js +207 -32
- package/dist/plugin/mission-review.js +22 -5
- package/dist/plugin/profiled.js +259 -66
- package/dist/plugin/protected-snapshot.d.ts +6 -0
- package/dist/plugin/protected-snapshot.js +100 -25
- package/dist/plugin/runtime-bridge.d.ts +24 -1
- package/dist/plugin/v2.js +9 -3
- package/dist/plugin/validation-scratch.d.ts +2 -0
- package/dist/plugin/validation-scratch.js +38 -8
- package/dist/runtime-assets-v010.js +2 -2
- package/dist/runtime-mission-assets.d.ts +2 -0
- package/dist/runtime-mission-assets.js +73 -64
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -111,6 +111,8 @@ Official SWE-bench Lite `dev` results on the same 23 public instances:
|
|
|
111
111
|
| v0.12.16 (`9b05a34` release; `b1a6c0e` runner) | 4 / 23 (17.4%) | 5 | Fresh 23-task run; eight slots; 40-minute timeout | [Campaign](docs/swebench-v01216-dev23-2026-09-27.md) |
|
|
112
112
|
| v0.12.19 (`24f5386` release; matched rerun) | 8 / 23 (34.8%) | 0 | Eight slots; effective $2/instance; 40-minute timeout; one inference timeout | [Official result and provenance](docs/benchmarks/swebench-v01220-operation-observability-2026-09-28.md) |
|
|
113
113
|
| v0.12.20 (`628eb81` release) | 7 / 23 (30.4%) | 0 | Eight slots; effective $2/instance; 40-minute timeout | [Official result and caveat](docs/benchmarks/swebench-v01220-operation-observability-2026-09-28.md) |
|
|
114
|
+
| v0.12.25 (`49eb1e4` release) | 7 / 23 (30.4%) | 0 | Eight slots; $2/instance; $30 total cap; 20-minute progress check / 40-minute hard maximum | [Comparison baseline](#v0131-dev23-2026-09-30) |
|
|
115
|
+
| v0.13.1 (`d19e8be` release; 2026-09-30) | **8 / 23 (34.8%)** | 0 | Eight slots; $2/instance; $46 total cap; 20-minute progress check / 40-minute hard maximum; GPT-6.1 Sol + Luna Fast | [Run summary](#v0131-dev23-2026-09-30) |
|
|
114
116
|
|
|
115
117
|
Every row has 23 submitted official predictions; an empty patch counts against
|
|
116
118
|
the score, not as a missing evaluation. The v0.10.14 report does not separately
|
|
@@ -132,6 +134,39 @@ single run-to-run difference does not establish causation.
|
|
|
132
134
|
|
|
133
135
|
Historical qualification references remain in [benchmark reference](docs/benchmark-reference.md).
|
|
134
136
|
|
|
137
|
+
#### v0.13.1 dev23 (2026-09-30)
|
|
138
|
+
|
|
139
|
+
One fresh pass@1 run and one official SWE-bench harness evaluation resolved
|
|
140
|
+
**8/23**, versus **7/23** for v0.12.25. The new resolution was
|
|
141
|
+
`pylint-dev__astroid-1333`; all seven previously resolved IDs were retained.
|
|
142
|
+
Resolved by repository: marshmallow **2/2**, pvlib **0/5**, pydicom **2/5**,
|
|
143
|
+
astroid **3/5**, pyvista **0/1**, sqlfluff **1/5**.
|
|
144
|
+
|
|
145
|
+
- The scored row is the user-requested fresh run after a host restart. The
|
|
146
|
+
interrupted initial run is excluded from this score; the fresh run made one
|
|
147
|
+
attempt per instance with no inference retry.
|
|
148
|
+
- The dataset revision (`6ec7bb89b9342f664a54a6e0a6ea6501d3437cc2`), public rows,
|
|
149
|
+
and all 23 official evaluation image IDs match the v0.12.25 run. Both used
|
|
150
|
+
`official-image-testbed` and sequential official scoring.
|
|
151
|
+
- Operator/Coordinator/Reviewer/Advisor defaults changed to
|
|
152
|
+
`openai/gpt-6.1-sol#xhigh`. Actual task Workers remained
|
|
153
|
+
`openai/gpt-6-luna-fast#max`, observed across all 23 instances. The harness and
|
|
154
|
+
total budget also changed, so the extra resolution cannot be attributed to
|
|
155
|
+
the model change alone.
|
|
156
|
+
- Inference ended with 20 normal completions, two timeouts
|
|
157
|
+
(`pvlib__pvlib-python-1154`, `sqlfluff__sqlfluff-1763`) and one agent failure
|
|
158
|
+
(`pvlib__pvlib-python-1854`). All patches, including stopped attempts, were
|
|
159
|
+
officially scored: 23 completed evaluations, zero empty patches and zero
|
|
160
|
+
official evaluation errors or infrastructure failures.
|
|
161
|
+
- Known estimated inference cost: **$17.73**; separate unknown-usage hold:
|
|
162
|
+
**$1.98**, not counted as known expense. Inference wall time was about
|
|
163
|
+
**81 minutes**, followed by **7.2 minutes** of official scoring.
|
|
164
|
+
- Fixed release commit: `d19e8be0d21180cc23ad2ae4b853d846a18e77bc`;
|
|
165
|
+
package SHA-256: `99300ceec0c3eee4fa1f984fed50d15d58b2ff4455df0041850ecd63514b0a51`.
|
|
166
|
+
OpenCode **2.0.20**, official harness **5.0.2**. Local evidence is retained in
|
|
167
|
+
`_testenv/swebench-v0131-dev23-20260930-r2/result-summary.json`; generated
|
|
168
|
+
predictions, databases and raw logs are not committed.
|
|
169
|
+
|
|
135
170
|
## Mission tools
|
|
136
171
|
|
|
137
172
|
1. `start_mission`: Operator supplies concise requirements; the host saves original messages and returns a Coordinator task.
|
package/dist/asset-version.d.ts
CHANGED
|
@@ -3,5 +3,5 @@
|
|
|
3
3
|
* installed project marker without importing every asset body.
|
|
4
4
|
*/
|
|
5
5
|
export declare const RUNTIME_ASSET_VERSION = "0.3.89-completion-proof-v1";
|
|
6
|
-
export declare const V010_RUNTIME_ASSET_VERSION = "0.13.
|
|
6
|
+
export declare const V010_RUNTIME_ASSET_VERSION = "0.13.2-anko-recovery-v1";
|
|
7
7
|
export type RuntimeAssetVersion = typeof RUNTIME_ASSET_VERSION | typeof V010_RUNTIME_ASSET_VERSION;
|
package/dist/asset-version.js
CHANGED
|
@@ -3,4 +3,4 @@
|
|
|
3
3
|
* installed project marker without importing every asset body.
|
|
4
4
|
*/
|
|
5
5
|
export const RUNTIME_ASSET_VERSION = "0.3.89-completion-proof-v1";
|
|
6
|
-
export const V010_RUNTIME_ASSET_VERSION = "0.13.
|
|
6
|
+
export const V010_RUNTIME_ASSET_VERSION = "0.13.2-anko-recovery-v1";
|
|
@@ -1,2 +1,2 @@
|
|
|
1
1
|
/** Common text bounds shared by handoff, manifest and goal evidence validation. */
|
|
2
|
-
export const CONTRACT_TEXT_LIMITS = Object.freeze({ title: 160, objective:
|
|
2
|
+
export const CONTRACT_TEXT_LIMITS = Object.freeze({ title: 160, objective: 32768, statement: 1000, command: 8192, path: 512 });
|
|
@@ -39,6 +39,13 @@ export interface GoalEvidence {
|
|
|
39
39
|
readonly candidate_paths: readonly string[];
|
|
40
40
|
/** Missing on legacy evidence: retain its original all-paths snapshot recipe. */
|
|
41
41
|
readonly source_policy?: "project-files-v1" | "declared-paths-v1";
|
|
42
|
+
/** Fixed when validation starts. Full manifest_hash still identifies the historical execution contract. */
|
|
43
|
+
readonly freshness?: {
|
|
44
|
+
readonly contract_hash: string;
|
|
45
|
+
readonly scratch_paths: readonly string[];
|
|
46
|
+
readonly protected_paths: readonly string[];
|
|
47
|
+
readonly environment: Readonly<Record<string, string | null>>;
|
|
48
|
+
};
|
|
42
49
|
};
|
|
43
50
|
readonly execution: {
|
|
44
51
|
readonly command: readonly string[];
|
package/dist/core/goal-bound.js
CHANGED
|
@@ -64,7 +64,13 @@ export function validGoalEvidence(value, state) {
|
|
|
64
64
|
protectedBinding.source_paths.every(text) && Array.isArray(protectedBinding.candidate_paths) &&
|
|
65
65
|
protectedBinding.candidate_paths.every(text) &&
|
|
66
66
|
(protectedBinding.source_policy === undefined || protectedBinding.source_policy === "project-files-v1" ||
|
|
67
|
-
protectedBinding.source_policy === "declared-paths-v1")
|
|
67
|
+
protectedBinding.source_policy === "declared-paths-v1") &&
|
|
68
|
+
(protectedBinding.freshness === undefined || (protectedBinding.freshness !== null && typeof protectedBinding.freshness === "object" &&
|
|
69
|
+
HASH.test(protectedBinding.freshness.contract_hash) &&
|
|
70
|
+
Array.isArray(protectedBinding.freshness.scratch_paths) && protectedBinding.freshness.scratch_paths.every(text) &&
|
|
71
|
+
Array.isArray(protectedBinding.freshness.protected_paths) && protectedBinding.freshness.protected_paths.every(text) &&
|
|
72
|
+
protectedBinding.freshness.environment !== null && typeof protectedBinding.freshness.environment === "object" &&
|
|
73
|
+
!Array.isArray(protectedBinding.freshness.environment) && Object.values(protectedBinding.freshness.environment).every(value => value === null || typeof value === "string")));
|
|
68
74
|
const matches = criteria.length > 0 && criteria.length === value.measurement.criterion_ids.length &&
|
|
69
75
|
criteria.every((criterion) => criterion.target === value.measurement.target &&
|
|
70
76
|
criterion.entrypoint === value.measurement.entrypoint && criterion.workload === value.measurement.workload &&
|
|
@@ -19,6 +19,7 @@ export interface MissionEvidenceExcerpt {
|
|
|
19
19
|
export interface MissionReviewScope {
|
|
20
20
|
read: string[];
|
|
21
21
|
write: string[];
|
|
22
|
+
validationBindings?: NonNullable<import("./goal-bound.js").GoalEvidence["protected_binding"]>[];
|
|
22
23
|
}
|
|
23
24
|
export interface MissionConsultation {
|
|
24
25
|
id: string;
|
|
@@ -47,7 +48,9 @@ export interface MissionAttempt {
|
|
|
47
48
|
status: "pending" | "dispatched" | "succeeded" | "failed" | "cancelled" | "unconfirmed";
|
|
48
49
|
callID?: string;
|
|
49
50
|
childSessionID?: string;
|
|
51
|
+
dispatchFingerprint?: string;
|
|
50
52
|
nativeOutcome?: "completed" | "failed" | "unknown";
|
|
53
|
+
terminal?: import("../plugin/runtime-bridge.js").MissionWorkerTerminalRecord;
|
|
51
54
|
observedModel?: string;
|
|
52
55
|
observedVariant?: string;
|
|
53
56
|
failure?: {
|
|
@@ -86,6 +89,17 @@ export interface MissionExecution {
|
|
|
86
89
|
result?: Record<string, unknown>;
|
|
87
90
|
}[];
|
|
88
91
|
}
|
|
92
|
+
export interface MissionLaunchConditions {
|
|
93
|
+
entrypoint?: string;
|
|
94
|
+
inputs?: string[];
|
|
95
|
+
timeout_seconds?: number;
|
|
96
|
+
cost_limit_usd?: number;
|
|
97
|
+
benchmark_attempts?: number;
|
|
98
|
+
grading?: "none" | "official";
|
|
99
|
+
source: string;
|
|
100
|
+
applies_to: string;
|
|
101
|
+
recordedAt?: string;
|
|
102
|
+
}
|
|
89
103
|
export interface OperatorMission {
|
|
90
104
|
version: "0.12";
|
|
91
105
|
id: string;
|
|
@@ -96,10 +110,14 @@ export interface OperatorMission {
|
|
|
96
110
|
kind?: "implementation" | "operation";
|
|
97
111
|
/** Native shell observations of the requested operation, separate from auxiliary checks. */
|
|
98
112
|
execution?: MissionExecution;
|
|
113
|
+
/** Fixed benchmark conditions and their provenance, separate from internal Worker counters. */
|
|
114
|
+
launchConditions?: MissionLaunchConditions[];
|
|
99
115
|
requirements: {
|
|
100
116
|
id: string;
|
|
101
117
|
text: string;
|
|
102
118
|
}[];
|
|
119
|
+
/** Explicit user path prohibitions, unlike the Coordinator's estimated write list. */
|
|
120
|
+
prohibitedWrite?: string[];
|
|
103
121
|
/** The current user intentionally replaced the predecessor's requirements. */
|
|
104
122
|
requirementsReplaced?: boolean;
|
|
105
123
|
phase: "open" | "running" | "submitted" | "completed" | "cancelled";
|
|
@@ -179,6 +197,7 @@ export declare class OperatorMissionRuntime {
|
|
|
179
197
|
read(root: string): Promise<OperatorMission | undefined>;
|
|
180
198
|
required(root: string): Promise<OperatorMission>;
|
|
181
199
|
capture(root: string, request: MissionRequest): Promise<void>;
|
|
200
|
+
recordLaunchConditions(root: string, raw: unknown): Promise<OperatorMission>;
|
|
182
201
|
start(root: string, requirements: unknown, replaceRequirements?: boolean, options?: {
|
|
183
202
|
kind?: OperatorMission["kind"];
|
|
184
203
|
context?: MissionContext[];
|
|
@@ -21,9 +21,11 @@ export function missionValidationCommand(command) {
|
|
|
21
21
|
export const MISSION_CONSULTATION_LIMIT = 32;
|
|
22
22
|
/** Review coverage survives a narrower replan; it is not a Worker write grant. */
|
|
23
23
|
export function missionReviewScope(previous, ...runs) {
|
|
24
|
+
const bindings = [...(previous?.validationBindings ?? []), ...runs.flatMap(run => run.units.flatMap(unit => (unit.evidence ?? []).flatMap(proof => proof.protected_binding?.freshness ? [proof.protected_binding] : [])))];
|
|
24
25
|
return {
|
|
25
26
|
read: [...new Set([...(previous?.read ?? []), ...runs.flatMap(run => run.units.flatMap(({ unit }) => unit.read ?? []))])].sort(),
|
|
26
27
|
write: [...new Set([...(previous?.write ?? []), ...runs.flatMap(run => run.units.flatMap(({ unit }) => unit.write))])].sort(),
|
|
28
|
+
...(bindings.length ? { validationBindings: [...new Map(bindings.map(binding => [JSON.stringify(binding), binding])).values()] } : {}),
|
|
27
29
|
};
|
|
28
30
|
}
|
|
29
31
|
/** Classify an independent Reviewer's first line. Anything else is a finding. */
|
|
@@ -214,6 +216,24 @@ export class OperatorMissionRuntime {
|
|
|
214
216
|
}
|
|
215
217
|
});
|
|
216
218
|
}
|
|
219
|
+
recordLaunchConditions(root, raw) {
|
|
220
|
+
if (!record(raw) || typeof raw.source !== "string" || !raw.source.trim() || typeof raw.applies_to !== "string" || !raw.applies_to.trim() ||
|
|
221
|
+
(raw.entrypoint !== undefined && (typeof raw.entrypoint !== "string" || !raw.entrypoint.trim())) ||
|
|
222
|
+
(raw.inputs !== undefined && (!Array.isArray(raw.inputs) || !raw.inputs.every(path => typeof path === "string" && path.trim()))) ||
|
|
223
|
+
["timeout_seconds", "cost_limit_usd", "benchmark_attempts"].some(key => raw[key] !== undefined &&
|
|
224
|
+
(typeof raw[key] !== "number" || !Number.isFinite(raw[key]) || raw[key] <= 0)) ||
|
|
225
|
+
(raw.benchmark_attempts !== undefined && !Number.isSafeInteger(raw.benchmark_attempts)) ||
|
|
226
|
+
(raw.grading !== undefined && !["none", "official"].includes(String(raw.grading))) ||
|
|
227
|
+
Object.keys(raw).some(key => !["entrypoint", "inputs", "timeout_seconds", "cost_limit_usd", "benchmark_attempts", "grading", "source", "applies_to"].includes(key))) {
|
|
228
|
+
throw new Error("mission-launch-conditions-invalid");
|
|
229
|
+
}
|
|
230
|
+
return this.update(root, state => {
|
|
231
|
+
state.launchConditions ??= [];
|
|
232
|
+
if (state.launchConditions.some(item => { const { recordedAt: _at, ...value } = item; return JSON.stringify(value) === JSON.stringify(raw); }))
|
|
233
|
+
return;
|
|
234
|
+
state.launchConditions.push({ ...raw, recordedAt: new Date().toISOString() });
|
|
235
|
+
});
|
|
236
|
+
}
|
|
217
237
|
start(root, requirements, replaceRequirements = false, options = {}) {
|
|
218
238
|
return this.serial(root, async () => {
|
|
219
239
|
if (!Array.isArray(requirements) || requirements.length === 0 || requirements.length > 64 ||
|
|
@@ -366,6 +386,8 @@ export class OperatorMissionRuntime {
|
|
|
366
386
|
"Start the first useful Worker promptly. No proposal/approval phase. Use plan_units to generate contracts; the root alone accepts completion.",
|
|
367
387
|
"Escalate only a completion candidate, a user-only decision, or an extension of original requirements/budget. Unit progress is published without stopping you.",
|
|
368
388
|
"Requirements:", ...state.requirements.map(item => `${item.id}: ${item.text}`),
|
|
389
|
+
`Confirmed launch conditions (fixed limits, not consumption or remaining budget): ${JSON.stringify(state.launchConditions ?? [])}`,
|
|
390
|
+
`Explicit user write prohibitions: ${JSON.stringify(state.prohibitedWrite ?? [])}`,
|
|
369
391
|
`Work kind: ${state.kind ?? "implementation"}. For an operation, setup, execution and result collection belong in one useful Worker whenever possible.`,
|
|
370
392
|
...(state.context?.length ? ["Prior conversation context (task data; preserve the selected target, not superseded obligations):",
|
|
371
393
|
...state.context.map(item => `--- ${item.role}:${item.id} ---\n${item.text}`)] : []),
|
|
@@ -383,7 +405,7 @@ export function missionPlan(mission, raw) {
|
|
|
383
405
|
const line = (field) => {
|
|
384
406
|
if (typeof value[field] !== "string" || !value[field].trim())
|
|
385
407
|
throw new Error(`mission-unit-${index + 1}: ${field} required`);
|
|
386
|
-
return value[field].replace(/[\r\n]+/gu, " ");
|
|
408
|
+
return field === "objective" ? value[field] : value[field].replace(/[\r\n]+/gu, " ");
|
|
387
409
|
};
|
|
388
410
|
const paths = (field) => {
|
|
389
411
|
const entries = value[field] ?? [];
|
|
@@ -450,6 +472,8 @@ export function missionPacket(mission, run) {
|
|
|
450
472
|
completed_units: predecessor.units.filter(unit => unit.status === "succeeded").length,
|
|
451
473
|
note: "Historical results and spend are retained; they do not complete the current requirements." } } : {}),
|
|
452
474
|
requirements: mission.requirements, original_request_refs: mission.requests.map(item => `user:${item.id}`),
|
|
475
|
+
launch_conditions: mission.launchConditions ?? [], prohibited_write: mission.prohibitedWrite ?? [],
|
|
476
|
+
accounting_scope: "Worker units are not benchmark attempts. Host budget is Worker-only; orchestration, Review and external campaign costs are excluded. Launch caps are fixed conditions, not a known campaign remainder.",
|
|
453
477
|
submission: mission.submission, progress: mission.progress, consultations: mission.consultations ?? [],
|
|
454
478
|
attempts: mission.attempts ?? [], ...(mission.rescue ? { rescue: mission.rescue } : {}),
|
|
455
479
|
operation: { kind: mission.kind ?? "implementation", status: missionExecutionStatus(mission),
|
|
@@ -283,12 +283,12 @@ export declare class OperatorRuntime {
|
|
|
283
283
|
prepareMission(root: string, raw: unknown, dispatcher?: {
|
|
284
284
|
sessionID: string;
|
|
285
285
|
callID: string;
|
|
286
|
-
}, supersededRunID?: string, terminalChildren?: readonly string[], replaceRequirements?: boolean): Promise<OperatorState>;
|
|
286
|
+
}, supersededRunID?: string, terminalChildren?: readonly string[], replaceRequirements?: boolean, context?: Record<string, unknown>): Promise<OperatorState>;
|
|
287
287
|
/** Stage a settled mission's replacement; rejected preparation never cancels the usable run. */
|
|
288
288
|
replanMission(root: string, runID: string, raw: unknown, dispatcher?: {
|
|
289
289
|
sessionID: string;
|
|
290
290
|
callID: string;
|
|
291
|
-
}): Promise<OperatorState>;
|
|
291
|
+
}, context?: Record<string, unknown>): Promise<OperatorState>;
|
|
292
292
|
private prepareOnce;
|
|
293
293
|
operatorTask(state: OperatorState): OperatorTask;
|
|
294
294
|
/** Root-visible handle for the bounded operations delegate; its contract stays host-internal. */
|
|
@@ -351,6 +351,8 @@ export declare class OperatorRuntime {
|
|
|
351
351
|
operatorRejected(root: string): Promise<OperatorState>;
|
|
352
352
|
private operatorRejectedOnce;
|
|
353
353
|
settled(result: SerialDispatchSettlement): Promise<void>;
|
|
354
|
+
/** Correct an estimated Mission scope without dispatching, restarting, or reserving another unit. */
|
|
355
|
+
expandMissionWriteScope(root: string, child: string, taskID: string, paths: readonly string[], activate: (manifest: import("./types.js").OperationManifest) => Promise<() => Promise<void>>): Promise<void>;
|
|
354
356
|
private settledOnce;
|
|
355
357
|
/** One ordinary same-scope Mission retry before any optional model Rescue is considered. */
|
|
356
358
|
prepareMissionNormalRemediation(root: string, runID: string, unitID: string): Promise<OperatorState>;
|
|
@@ -177,7 +177,7 @@ export function parseOperatorPlan(value, scopeFormat = "repository") {
|
|
|
177
177
|
const mappingDiagnostics = [];
|
|
178
178
|
for (const [unitIndex, unit] of value.units.entries()) {
|
|
179
179
|
if (!record(unit) || !exactKeys(unit, ["id", "title", "objective", "read", "write", "validation", "acceptance_indices"]) ||
|
|
180
|
-
!identifier(unit.id) || ids.has(unit.id) || !text(unit.title) || !
|
|
180
|
+
!identifier(unit.id) || ids.has(unit.id) || !text(unit.title) || typeof unit.objective !== "string" || !unit.objective.trim() ||
|
|
181
181
|
!strings(unit.read) || !strings(unit.write) || !strings(unit.validation, true))
|
|
182
182
|
return planError(`/units/${unitIndex}`, "operator-unit-invalid", "operator-unit-shape");
|
|
183
183
|
unit.validation.forEach((command, index) => rejectValidationAnnotation(command, `/units/${unitIndex}/validation/${index}`));
|
|
@@ -686,12 +686,12 @@ export class OperatorRuntime {
|
|
|
686
686
|
}
|
|
687
687
|
}
|
|
688
688
|
/** Mission authority is supplied only by the owning profile, never by a model-authored plan. */
|
|
689
|
-
prepareMission(root, raw, dispatcher, supersededRunID, terminalChildren = [], replaceRequirements = false) {
|
|
690
|
-
return this.serial(root, () => this.prepareOnce(root, raw, undefined, { dispatcher, supersededRunID, terminalChildren, replaceRequirements }));
|
|
689
|
+
prepareMission(root, raw, dispatcher, supersededRunID, terminalChildren = [], replaceRequirements = false, context) {
|
|
690
|
+
return this.serial(root, () => this.prepareOnce(root, raw, undefined, { dispatcher, supersededRunID, terminalChildren, replaceRequirements, context }));
|
|
691
691
|
}
|
|
692
692
|
/** Stage a settled mission's replacement; rejected preparation never cancels the usable run. */
|
|
693
|
-
replanMission(root, runID, raw, dispatcher) {
|
|
694
|
-
return this.serial(root, () => this.prepareOnce(root, raw, undefined, { dispatcher, replaceRunID: runID }));
|
|
693
|
+
replanMission(root, runID, raw, dispatcher, context) {
|
|
694
|
+
return this.serial(root, () => this.prepareOnce(root, raw, undefined, { dispatcher, replaceRunID: runID, context }));
|
|
695
695
|
}
|
|
696
696
|
async prepareOnce(root, raw, scopeApprovalTurnID, mission) {
|
|
697
697
|
const previous = await this.read(root);
|
|
@@ -859,6 +859,7 @@ export class OperatorRuntime {
|
|
|
859
859
|
task: { title: unit.title, objective: unit.objective }, state: { done: [], next: [unit.title], blocked: [] }, risks: [],
|
|
860
860
|
verification: unit.validation.map(check => ({ check, status: "not_run", exit_code: null, summary: "Execute in the admitted worker." })),
|
|
861
861
|
ext: {
|
|
862
|
+
...(mission ? { "sortie-dogs/mission-context": { write_scope_origin: "coordinator-estimate", ...mission.context } } : {}),
|
|
862
863
|
"sortie-dogs/write-gate": { operation_manifest: manifestRelative, project_root: this.projectRoot },
|
|
863
864
|
[ACCEPTANCE_CONTINUITY_EXTENSION]: { schema_version: "0.1", authority: "dispatch", task_id: taskID,
|
|
864
865
|
criteria: plan.acceptance, fingerprint: acceptanceFingerprint,
|
|
@@ -914,23 +915,27 @@ export class OperatorRuntime {
|
|
|
914
915
|
throw new OperatorContractError(diagnostics);
|
|
915
916
|
const contents = [JSON.stringify(handoff), JSON.stringify(manifest)];
|
|
916
917
|
controls.push({ path: handoffPath, content: contents[0] }, { path: manifestPath, content: contents[1] });
|
|
917
|
-
const
|
|
918
|
+
const promptHeader = ["role: implementation", `task_id: ${taskID}`, `project_root: ${this.projectRoot}`,
|
|
918
919
|
`source_manifest: ${(unit.write.length ? unit.write : unit.read).join(", ") || "none"}`, `operation_manifest: ${manifestRelative}`,
|
|
919
|
-
`handoff_path: ${handoffPath}`,
|
|
920
|
-
|
|
920
|
+
`handoff_path: ${handoffPath}`, `goal_declaration_path: ${declarationPath}`];
|
|
921
|
+
const commitBoundary = plan.git_lifecycle !== undefined && index === plan.units.length - 1 ? [
|
|
922
|
+
`git_post_commit_validation: ${JSON.stringify(plan.git_lifecycle.post_commit_validation)}`,
|
|
923
|
+
"Complete every source write before invoking any git_post_commit_validation command. Its first exact invocation is the host commit boundary; after it, source mutation and undeclared shell commands are denied. Run every listed command and return its real evidence.",
|
|
924
|
+
] : [];
|
|
925
|
+
// Mission Workers already must read this host-generated handoff before binding. Preserve its
|
|
926
|
+
// verbatim objective, original requests, criteria and checks there, not in another prompt copy.
|
|
927
|
+
// Saved Tasks and non-Mission dispatch keep their existing text/identity.
|
|
928
|
+
const prompt = (mission ? [...promptHeader, "contract_reference: handoff",
|
|
929
|
+
'acceptance: handoff.ext["sortie-dogs/acceptance-continuity"].criteria', "validation: handoff.verification",
|
|
930
|
+
`unit_acceptance_indices: ${JSON.stringify(unit.acceptance_indices)}`, "",
|
|
931
|
+
"Read handoff_path once before binding. Implement task.objective; preserve the original requests, global criteria and constraints in ext, and prove this unit's assigned indices. Run verification checks exactly in order within this Task. Use supplied paths; do not reconstruct project_root. Return actual results and limitations, not whole-Mission completion.",
|
|
932
|
+
...commitBoundary,
|
|
933
|
+
] : [...promptHeader, "acceptance:", ...plan.acceptance.map(value => ` - ${value}`),
|
|
921
934
|
"validation:", ...unit.validation.map(value => ` - ${value}`),
|
|
922
|
-
|
|
923
|
-
|
|
924
|
-
|
|
925
|
-
...
|
|
926
|
-
mission
|
|
927
|
-
? "Complete the assigned work and listed validation within this Task. For a known operation, proceed through setup, execution and result collection; preparation alone is not execution. Preserve existing public behavior when changing source. Do not spawn nested subagents. The parent Operator or Coordinator handles any applicable independent review after your return; review is not a prerequisite to execution. Return actual results, including failed or not-started operations, and any exact contract correction needed."
|
|
928
|
-
: "Do not spawn nested subagents for consultation. Required consultations belong to the root before dispatch; use the confirmed decisions and evidence declared in the unit objective and inputs. If required consultation results or user decisions are missing, return the exact contract gap to the parent instead of attempting a deeper Task, inventing consent, or asking the user to repeat an already recorded decision.",
|
|
929
|
-
...(plan.git_lifecycle !== undefined && index === plan.units.length - 1 ? [
|
|
930
|
-
`git_post_commit_validation: ${JSON.stringify(plan.git_lifecycle.post_commit_validation)}`,
|
|
931
|
-
"Complete every source write before invoking any git_post_commit_validation command. Its first exact invocation is the host commit boundary; after it, source mutation and undeclared shell commands are denied. Run every listed command and return its real evidence.",
|
|
932
|
-
] : []),
|
|
933
|
-
`unit_acceptance_indices: ${JSON.stringify(unit.acceptance_indices)}`, "", unit.objective].join("\n");
|
|
935
|
+
"Execute validation in its declared order. Earlier entries may be approved generator, build, formatter, or exact cleanup commands required before canonical criterion tests. Every persistent or transient generator output must be declared in unit.write. Cleanup may remove only declared unit.write outputs and must be an explicit ordered command after generation and before post-commit or canonical validation; never add an ignore rule or remove an undeclared path. If any necessary command, input, output, or cleanup is missing, do not run an undeclared command or variant and do not use resume evidence tooling to invent permission; return a contract-repair decision.",
|
|
936
|
+
"Preserve existing public API success and error return semantics unless acceptance explicitly changes them, and cover those compatibility boundaries in the declared validation.",
|
|
937
|
+
"Do not spawn nested subagents for consultation. Required consultations belong to the root before dispatch; use the confirmed decisions and evidence declared in the unit objective and inputs. If required consultation results or user decisions are missing, return the exact contract gap to the parent instead of attempting a deeper Task, inventing consent, or asking the user to repeat an already recorded decision.",
|
|
938
|
+
...commitBoundary, `unit_acceptance_indices: ${JSON.stringify(unit.acceptance_indices)}`, "", unit.objective]).join("\n");
|
|
934
939
|
units.push({ unit, task: { subagent_type: profileAgent(this.profile, "dog-worker"), description: unit.title, prompt },
|
|
935
940
|
handoffPath, manifestPath, hashes: [...contents.map(hash), hash(declaration)], status: "pending", callID: null,
|
|
936
941
|
childSessionID: null, evidence: [], resultClass: null, repairValidationAttempts: 0, repairValidation: null });
|
|
@@ -1521,6 +1526,50 @@ export class OperatorRuntime {
|
|
|
1521
1526
|
settled(result) {
|
|
1522
1527
|
return this.serial(result.rootSessionID, () => this.settledOnce(result));
|
|
1523
1528
|
}
|
|
1529
|
+
/** Correct an estimated Mission scope without dispatching, restarting, or reserving another unit. */
|
|
1530
|
+
expandMissionWriteScope(root, child, taskID, paths, activate) {
|
|
1531
|
+
return this.serial(root, async () => {
|
|
1532
|
+
const state = await this.required(root);
|
|
1533
|
+
const unit = state.units.find(item => /^task_id: (.+)$/m.exec(item.task.prompt)?.[1] === taskID && item.childSessionID === child);
|
|
1534
|
+
if (!unit || !["running", "failed", "succeeded"].includes(unit.status) ||
|
|
1535
|
+
!["running", "awaiting-decision", "awaiting-acceptance"].includes(state.phase) || !unit.callID || state.gitLifecycle) {
|
|
1536
|
+
throw new Error("mission-scope-update-current-task-required");
|
|
1537
|
+
}
|
|
1538
|
+
await this.verifyControls(unit);
|
|
1539
|
+
const write = [...new Set([...unit.unit.write, ...paths.map(normalizeExecutionScope)])];
|
|
1540
|
+
for (const path of write) {
|
|
1541
|
+
const scoped = relative(this.projectRoot, resolve(this.projectRoot, normalizeManifestScope(path).path)).replaceAll("\\", "/");
|
|
1542
|
+
if ([".git", ...Object.values(RUNTIME_PROFILES).map(profile => profile.stateDirectory)].some(dir => scoped === dir || scoped.startsWith(`${dir}/`))) {
|
|
1543
|
+
throw new Error(`operator-control-write-forbidden:${path}`);
|
|
1544
|
+
}
|
|
1545
|
+
}
|
|
1546
|
+
if (JSON.stringify(write) === JSON.stringify(unit.unit.write))
|
|
1547
|
+
return;
|
|
1548
|
+
const oldManifest = await readFile(unit.manifestPath, "utf8"), oldHandoff = await readFile(unit.handoffPath, "utf8");
|
|
1549
|
+
const manifest = { ...JSON.parse(oldManifest), write };
|
|
1550
|
+
const handoff = JSON.parse(oldHandoff);
|
|
1551
|
+
const m = validateOperationManifestSchema(manifest), h = validateHandoffSchema(handoff);
|
|
1552
|
+
if (!m.ok || !h.ok)
|
|
1553
|
+
throw new Error("mission-scope-update-invalid-contract");
|
|
1554
|
+
const nextManifest = JSON.stringify(manifest), nextHandoff = JSON.stringify(handoff);
|
|
1555
|
+
let rollback;
|
|
1556
|
+
try {
|
|
1557
|
+
await writeFile(unit.manifestPath, nextManifest);
|
|
1558
|
+
await writeFile(unit.handoffPath, nextHandoff);
|
|
1559
|
+
rollback = await activate(manifest);
|
|
1560
|
+
unit.unit = { ...unit.unit, write };
|
|
1561
|
+
unit.hashes = [hash(nextHandoff), hash(nextManifest), unit.hashes[2]];
|
|
1562
|
+
unit.task = { ...unit.task, prompt: unit.task.prompt.replace(/^source_manifest: .*$/mu, `source_manifest: ${write.join(", ")}`) };
|
|
1563
|
+
await this.save(state);
|
|
1564
|
+
}
|
|
1565
|
+
catch (error) {
|
|
1566
|
+
await writeFile(unit.manifestPath, oldManifest);
|
|
1567
|
+
await writeFile(unit.handoffPath, oldHandoff);
|
|
1568
|
+
await rollback?.();
|
|
1569
|
+
throw error;
|
|
1570
|
+
}
|
|
1571
|
+
});
|
|
1572
|
+
}
|
|
1524
1573
|
async settledOnce(result) {
|
|
1525
1574
|
const state = await this.read(result.rootSessionID);
|
|
1526
1575
|
const unit = state?.units.find(item => item.callID === result.callID);
|
|
@@ -198,7 +198,29 @@ export class ScopeLeaseRegistry {
|
|
|
198
198
|
return Object.freeze({
|
|
199
199
|
id,
|
|
200
200
|
ownerId,
|
|
201
|
-
scope
|
|
201
|
+
get scope() { return Object.freeze({ read: [...scope.read], write: [...scope.write] }); },
|
|
202
|
+
replaceScope: (replacement) => {
|
|
203
|
+
const normalized = this.validateRequest(replacement);
|
|
204
|
+
const result = pending.then(async () => {
|
|
205
|
+
if (!active || closing)
|
|
206
|
+
throw new ScopeLeaseError("not-held", "Lease is not held.");
|
|
207
|
+
await this.withLock(async (mutex) => {
|
|
208
|
+
const state = await this.load();
|
|
209
|
+
const held = state.leases.find(lease => lease.id === id && lease.ownerHash === ownerHash && lease.tokenHash === credentialHash);
|
|
210
|
+
if (!held || held.expiresAt <= Date.now())
|
|
211
|
+
throw new ScopeLeaseError("not-held", "Lease is not held.");
|
|
212
|
+
if (state.leases.some(lease => lease.id !== id && lease.expiresAt > Date.now() && worktreeScopesConflict(normalized, lease))) {
|
|
213
|
+
throw new ScopeLeaseError("scope-conflict", "Requested scope is already leased.");
|
|
214
|
+
}
|
|
215
|
+
held.read = [...normalized.read];
|
|
216
|
+
held.write = [...normalized.write];
|
|
217
|
+
await this.save({ ...state, revision: state.revision + 1 }, mutex);
|
|
218
|
+
scope = normalized;
|
|
219
|
+
});
|
|
220
|
+
});
|
|
221
|
+
pending = result.catch(() => undefined);
|
|
222
|
+
return result;
|
|
223
|
+
},
|
|
202
224
|
heartbeat: () => serialized("heartbeat"),
|
|
203
225
|
assertHeld: () => serialized("assert"),
|
|
204
226
|
isReleased: () => this.withLock(async () => {
|