sortie-dogs 0.13.1 → 0.13.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -111,6 +111,8 @@ Official SWE-bench Lite `dev` results on the same 23 public instances:
111
111
  | v0.12.16 (`9b05a34` release; `b1a6c0e` runner) | 4 / 23 (17.4%) | 5 | Fresh 23-task run; eight slots; 40-minute timeout | [Campaign](docs/swebench-v01216-dev23-2026-09-27.md) |
112
112
  | v0.12.19 (`24f5386` release; matched rerun) | 8 / 23 (34.8%) | 0 | Eight slots; effective $2/instance; 40-minute timeout; one inference timeout | [Official result and provenance](docs/benchmarks/swebench-v01220-operation-observability-2026-09-28.md) |
113
113
  | v0.12.20 (`628eb81` release) | 7 / 23 (30.4%) | 0 | Eight slots; effective $2/instance; 40-minute timeout | [Official result and caveat](docs/benchmarks/swebench-v01220-operation-observability-2026-09-28.md) |
114
+ | v0.12.25 (`49eb1e4` release) | 7 / 23 (30.4%) | 0 | Eight slots; $2/instance; $30 total cap; 20-minute progress check / 40-minute hard maximum | [Comparison baseline](#v0131-dev23-2026-09-30) |
115
+ | v0.13.1 (`d19e8be` release; 2026-09-30) | **8 / 23 (34.8%)** | 0 | Eight slots; $2/instance; $46 total cap; 20-minute progress check / 40-minute hard maximum; GPT-6.1 Sol + Luna Fast | [Run summary](#v0131-dev23-2026-09-30) |
114
116
 
115
117
  Every row has 23 submitted official predictions; an empty patch counts against
116
118
  the score, not as a missing evaluation. The v0.10.14 report does not separately
@@ -132,6 +134,39 @@ single run-to-run difference does not establish causation.
132
134
 
133
135
  Historical qualification references remain in [benchmark reference](docs/benchmark-reference.md).
134
136
 
137
+ #### v0.13.1 dev23 (2026-09-30)
138
+
139
+ One fresh pass@1 run and one official SWE-bench harness evaluation resolved
140
+ **8/23**, versus **7/23** for v0.12.25. The new resolution was
141
+ `pylint-dev__astroid-1333`; all seven previously resolved IDs were retained.
142
+ Resolved by repository: marshmallow **2/2**, pvlib **0/5**, pydicom **2/5**,
143
+ astroid **3/5**, pyvista **0/1**, sqlfluff **1/5**.
144
+
145
+ - The scored row is the user-requested fresh run after a host restart. The
146
+ interrupted initial run is excluded from this score; the fresh run made one
147
+ attempt per instance with no inference retry.
148
+ - The dataset revision (`6ec7bb89b9342f664a54a6e0a6ea6501d3437cc2`), public rows,
149
+ and all 23 official evaluation image IDs match the v0.12.25 run. Both used
150
+ `official-image-testbed` and sequential official scoring.
151
+ - Operator/Coordinator/Reviewer/Advisor defaults changed to
152
+ `openai/gpt-6.1-sol#xhigh`. Actual task Workers remained
153
+ `openai/gpt-6-luna-fast#max`, observed across all 23 instances. The harness and
154
+ total budget also changed, so the extra resolution cannot be attributed to
155
+ the model change alone.
156
+ - Inference ended with 20 normal completions, two timeouts
157
+ (`pvlib__pvlib-python-1154`, `sqlfluff__sqlfluff-1763`) and one agent failure
158
+ (`pvlib__pvlib-python-1854`). All patches, including stopped attempts, were
159
+ officially scored: 23 completed evaluations, zero empty patches and zero
160
+ official evaluation errors or infrastructure failures.
161
+ - Known estimated inference cost: **$17.73**; separate unknown-usage hold:
162
+ **$1.98**, not counted as known expense. Inference wall time was about
163
+ **81 minutes**, followed by **7.2 minutes** of official scoring.
164
+ - Fixed release commit: `d19e8be0d21180cc23ad2ae4b853d846a18e77bc`;
165
+ package SHA-256: `99300ceec0c3eee4fa1f984fed50d15d58b2ff4455df0041850ecd63514b0a51`.
166
+ OpenCode **2.0.20**, official harness **5.0.2**. Local evidence is retained in
167
+ `_testenv/swebench-v0131-dev23-20260930-r2/result-summary.json`; generated
168
+ predictions, databases and raw logs are not committed.
169
+
135
170
  ## Mission tools
136
171
 
137
172
  1. `start_mission`: Operator supplies concise requirements; the host saves original messages and returns a Coordinator task.
@@ -3,5 +3,5 @@
3
3
  * installed project marker without importing every asset body.
4
4
  */
5
5
  export declare const RUNTIME_ASSET_VERSION = "0.3.89-completion-proof-v1";
6
- export declare const V010_RUNTIME_ASSET_VERSION = "0.13.1-quality-first-review-v1";
6
+ export declare const V010_RUNTIME_ASSET_VERSION = "0.13.2-anko-recovery-v1";
7
7
  export type RuntimeAssetVersion = typeof RUNTIME_ASSET_VERSION | typeof V010_RUNTIME_ASSET_VERSION;
@@ -3,4 +3,4 @@
3
3
  * installed project marker without importing every asset body.
4
4
  */
5
5
  export const RUNTIME_ASSET_VERSION = "0.3.89-completion-proof-v1";
6
- export const V010_RUNTIME_ASSET_VERSION = "0.13.1-quality-first-review-v1";
6
+ export const V010_RUNTIME_ASSET_VERSION = "0.13.2-anko-recovery-v1";
@@ -1,7 +1,7 @@
1
1
  /** Common text bounds shared by handoff, manifest and goal evidence validation. */
2
2
  export declare const CONTRACT_TEXT_LIMITS: Readonly<{
3
3
  title: 160;
4
- objective: 2000;
4
+ objective: 32768;
5
5
  statement: 1000;
6
6
  command: 8192;
7
7
  path: 512;
@@ -1,2 +1,2 @@
1
1
  /** Common text bounds shared by handoff, manifest and goal evidence validation. */
2
- export const CONTRACT_TEXT_LIMITS = Object.freeze({ title: 160, objective: 2000, statement: 1000, command: 8192, path: 512 });
2
+ export const CONTRACT_TEXT_LIMITS = Object.freeze({ title: 160, objective: 32768, statement: 1000, command: 8192, path: 512 });
@@ -39,6 +39,13 @@ export interface GoalEvidence {
39
39
  readonly candidate_paths: readonly string[];
40
40
  /** Missing on legacy evidence: retain its original all-paths snapshot recipe. */
41
41
  readonly source_policy?: "project-files-v1" | "declared-paths-v1";
42
+ /** Fixed when validation starts. Full manifest_hash still identifies the historical execution contract. */
43
+ readonly freshness?: {
44
+ readonly contract_hash: string;
45
+ readonly scratch_paths: readonly string[];
46
+ readonly protected_paths: readonly string[];
47
+ readonly environment: Readonly<Record<string, string | null>>;
48
+ };
42
49
  };
43
50
  readonly execution: {
44
51
  readonly command: readonly string[];
@@ -64,7 +64,13 @@ export function validGoalEvidence(value, state) {
64
64
  protectedBinding.source_paths.every(text) && Array.isArray(protectedBinding.candidate_paths) &&
65
65
  protectedBinding.candidate_paths.every(text) &&
66
66
  (protectedBinding.source_policy === undefined || protectedBinding.source_policy === "project-files-v1" ||
67
- protectedBinding.source_policy === "declared-paths-v1");
67
+ protectedBinding.source_policy === "declared-paths-v1") &&
68
+ (protectedBinding.freshness === undefined || (protectedBinding.freshness !== null && typeof protectedBinding.freshness === "object" &&
69
+ HASH.test(protectedBinding.freshness.contract_hash) &&
70
+ Array.isArray(protectedBinding.freshness.scratch_paths) && protectedBinding.freshness.scratch_paths.every(text) &&
71
+ Array.isArray(protectedBinding.freshness.protected_paths) && protectedBinding.freshness.protected_paths.every(text) &&
72
+ protectedBinding.freshness.environment !== null && typeof protectedBinding.freshness.environment === "object" &&
73
+ !Array.isArray(protectedBinding.freshness.environment) && Object.values(protectedBinding.freshness.environment).every(value => value === null || typeof value === "string")));
68
74
  const matches = criteria.length > 0 && criteria.length === value.measurement.criterion_ids.length &&
69
75
  criteria.every((criterion) => criterion.target === value.measurement.target &&
70
76
  criterion.entrypoint === value.measurement.entrypoint && criterion.workload === value.measurement.workload &&
@@ -19,6 +19,7 @@ export interface MissionEvidenceExcerpt {
19
19
  export interface MissionReviewScope {
20
20
  read: string[];
21
21
  write: string[];
22
+ validationBindings?: NonNullable<import("./goal-bound.js").GoalEvidence["protected_binding"]>[];
22
23
  }
23
24
  export interface MissionConsultation {
24
25
  id: string;
@@ -47,7 +48,9 @@ export interface MissionAttempt {
47
48
  status: "pending" | "dispatched" | "succeeded" | "failed" | "cancelled" | "unconfirmed";
48
49
  callID?: string;
49
50
  childSessionID?: string;
51
+ dispatchFingerprint?: string;
50
52
  nativeOutcome?: "completed" | "failed" | "unknown";
53
+ terminal?: import("../plugin/runtime-bridge.js").MissionWorkerTerminalRecord;
51
54
  observedModel?: string;
52
55
  observedVariant?: string;
53
56
  failure?: {
@@ -86,6 +89,17 @@ export interface MissionExecution {
86
89
  result?: Record<string, unknown>;
87
90
  }[];
88
91
  }
92
+ export interface MissionLaunchConditions {
93
+ entrypoint?: string;
94
+ inputs?: string[];
95
+ timeout_seconds?: number;
96
+ cost_limit_usd?: number;
97
+ benchmark_attempts?: number;
98
+ grading?: "none" | "official";
99
+ source: string;
100
+ applies_to: string;
101
+ recordedAt?: string;
102
+ }
89
103
  export interface OperatorMission {
90
104
  version: "0.12";
91
105
  id: string;
@@ -96,10 +110,14 @@ export interface OperatorMission {
96
110
  kind?: "implementation" | "operation";
97
111
  /** Native shell observations of the requested operation, separate from auxiliary checks. */
98
112
  execution?: MissionExecution;
113
+ /** Fixed benchmark conditions and their provenance, separate from internal Worker counters. */
114
+ launchConditions?: MissionLaunchConditions[];
99
115
  requirements: {
100
116
  id: string;
101
117
  text: string;
102
118
  }[];
119
+ /** Explicit user path prohibitions, unlike the Coordinator's estimated write list. */
120
+ prohibitedWrite?: string[];
103
121
  /** The current user intentionally replaced the predecessor's requirements. */
104
122
  requirementsReplaced?: boolean;
105
123
  phase: "open" | "running" | "submitted" | "completed" | "cancelled";
@@ -179,6 +197,7 @@ export declare class OperatorMissionRuntime {
179
197
  read(root: string): Promise<OperatorMission | undefined>;
180
198
  required(root: string): Promise<OperatorMission>;
181
199
  capture(root: string, request: MissionRequest): Promise<void>;
200
+ recordLaunchConditions(root: string, raw: unknown): Promise<OperatorMission>;
182
201
  start(root: string, requirements: unknown, replaceRequirements?: boolean, options?: {
183
202
  kind?: OperatorMission["kind"];
184
203
  context?: MissionContext[];
@@ -21,9 +21,11 @@ export function missionValidationCommand(command) {
21
21
  export const MISSION_CONSULTATION_LIMIT = 32;
22
22
  /** Review coverage survives a narrower replan; it is not a Worker write grant. */
23
23
  export function missionReviewScope(previous, ...runs) {
24
+ const bindings = [...(previous?.validationBindings ?? []), ...runs.flatMap(run => run.units.flatMap(unit => (unit.evidence ?? []).flatMap(proof => proof.protected_binding?.freshness ? [proof.protected_binding] : [])))];
24
25
  return {
25
26
  read: [...new Set([...(previous?.read ?? []), ...runs.flatMap(run => run.units.flatMap(({ unit }) => unit.read ?? []))])].sort(),
26
27
  write: [...new Set([...(previous?.write ?? []), ...runs.flatMap(run => run.units.flatMap(({ unit }) => unit.write))])].sort(),
28
+ ...(bindings.length ? { validationBindings: [...new Map(bindings.map(binding => [JSON.stringify(binding), binding])).values()] } : {}),
27
29
  };
28
30
  }
29
31
  /** Classify an independent Reviewer's first line. Anything else is a finding. */
@@ -214,6 +216,24 @@ export class OperatorMissionRuntime {
214
216
  }
215
217
  });
216
218
  }
219
+ recordLaunchConditions(root, raw) {
220
+ if (!record(raw) || typeof raw.source !== "string" || !raw.source.trim() || typeof raw.applies_to !== "string" || !raw.applies_to.trim() ||
221
+ (raw.entrypoint !== undefined && (typeof raw.entrypoint !== "string" || !raw.entrypoint.trim())) ||
222
+ (raw.inputs !== undefined && (!Array.isArray(raw.inputs) || !raw.inputs.every(path => typeof path === "string" && path.trim()))) ||
223
+ ["timeout_seconds", "cost_limit_usd", "benchmark_attempts"].some(key => raw[key] !== undefined &&
224
+ (typeof raw[key] !== "number" || !Number.isFinite(raw[key]) || raw[key] <= 0)) ||
225
+ (raw.benchmark_attempts !== undefined && !Number.isSafeInteger(raw.benchmark_attempts)) ||
226
+ (raw.grading !== undefined && !["none", "official"].includes(String(raw.grading))) ||
227
+ Object.keys(raw).some(key => !["entrypoint", "inputs", "timeout_seconds", "cost_limit_usd", "benchmark_attempts", "grading", "source", "applies_to"].includes(key))) {
228
+ throw new Error("mission-launch-conditions-invalid");
229
+ }
230
+ return this.update(root, state => {
231
+ state.launchConditions ??= [];
232
+ if (state.launchConditions.some(item => { const { recordedAt: _at, ...value } = item; return JSON.stringify(value) === JSON.stringify(raw); }))
233
+ return;
234
+ state.launchConditions.push({ ...raw, recordedAt: new Date().toISOString() });
235
+ });
236
+ }
217
237
  start(root, requirements, replaceRequirements = false, options = {}) {
218
238
  return this.serial(root, async () => {
219
239
  if (!Array.isArray(requirements) || requirements.length === 0 || requirements.length > 64 ||
@@ -366,6 +386,8 @@ export class OperatorMissionRuntime {
366
386
  "Start the first useful Worker promptly. No proposal/approval phase. Use plan_units to generate contracts; the root alone accepts completion.",
367
387
  "Escalate only a completion candidate, a user-only decision, or an extension of original requirements/budget. Unit progress is published without stopping you.",
368
388
  "Requirements:", ...state.requirements.map(item => `${item.id}: ${item.text}`),
389
+ `Confirmed launch conditions (fixed limits, not consumption or remaining budget): ${JSON.stringify(state.launchConditions ?? [])}`,
390
+ `Explicit user write prohibitions: ${JSON.stringify(state.prohibitedWrite ?? [])}`,
369
391
  `Work kind: ${state.kind ?? "implementation"}. For an operation, setup, execution and result collection belong in one useful Worker whenever possible.`,
370
392
  ...(state.context?.length ? ["Prior conversation context (task data; preserve the selected target, not superseded obligations):",
371
393
  ...state.context.map(item => `--- ${item.role}:${item.id} ---\n${item.text}`)] : []),
@@ -383,7 +405,7 @@ export function missionPlan(mission, raw) {
383
405
  const line = (field) => {
384
406
  if (typeof value[field] !== "string" || !value[field].trim())
385
407
  throw new Error(`mission-unit-${index + 1}: ${field} required`);
386
- return value[field].replace(/[\r\n]+/gu, " ");
408
+ return field === "objective" ? value[field] : value[field].replace(/[\r\n]+/gu, " ");
387
409
  };
388
410
  const paths = (field) => {
389
411
  const entries = value[field] ?? [];
@@ -450,6 +472,8 @@ export function missionPacket(mission, run) {
450
472
  completed_units: predecessor.units.filter(unit => unit.status === "succeeded").length,
451
473
  note: "Historical results and spend are retained; they do not complete the current requirements." } } : {}),
452
474
  requirements: mission.requirements, original_request_refs: mission.requests.map(item => `user:${item.id}`),
475
+ launch_conditions: mission.launchConditions ?? [], prohibited_write: mission.prohibitedWrite ?? [],
476
+ accounting_scope: "Worker units are not benchmark attempts. Host budget is Worker-only; orchestration, Review and external campaign costs are excluded. Launch caps are fixed conditions, not a known campaign remainder.",
453
477
  submission: mission.submission, progress: mission.progress, consultations: mission.consultations ?? [],
454
478
  attempts: mission.attempts ?? [], ...(mission.rescue ? { rescue: mission.rescue } : {}),
455
479
  operation: { kind: mission.kind ?? "implementation", status: missionExecutionStatus(mission),
@@ -283,12 +283,12 @@ export declare class OperatorRuntime {
283
283
  prepareMission(root: string, raw: unknown, dispatcher?: {
284
284
  sessionID: string;
285
285
  callID: string;
286
- }, supersededRunID?: string, terminalChildren?: readonly string[], replaceRequirements?: boolean): Promise<OperatorState>;
286
+ }, supersededRunID?: string, terminalChildren?: readonly string[], replaceRequirements?: boolean, context?: Record<string, unknown>): Promise<OperatorState>;
287
287
  /** Stage a settled mission's replacement; rejected preparation never cancels the usable run. */
288
288
  replanMission(root: string, runID: string, raw: unknown, dispatcher?: {
289
289
  sessionID: string;
290
290
  callID: string;
291
- }): Promise<OperatorState>;
291
+ }, context?: Record<string, unknown>): Promise<OperatorState>;
292
292
  private prepareOnce;
293
293
  operatorTask(state: OperatorState): OperatorTask;
294
294
  /** Root-visible handle for the bounded operations delegate; its contract stays host-internal. */
@@ -351,6 +351,8 @@ export declare class OperatorRuntime {
351
351
  operatorRejected(root: string): Promise<OperatorState>;
352
352
  private operatorRejectedOnce;
353
353
  settled(result: SerialDispatchSettlement): Promise<void>;
354
+ /** Correct an estimated Mission scope without dispatching, restarting, or reserving another unit. */
355
+ expandMissionWriteScope(root: string, child: string, taskID: string, paths: readonly string[], activate: (manifest: import("./types.js").OperationManifest) => Promise<() => Promise<void>>): Promise<void>;
354
356
  private settledOnce;
355
357
  /** One ordinary same-scope Mission retry before any optional model Rescue is considered. */
356
358
  prepareMissionNormalRemediation(root: string, runID: string, unitID: string): Promise<OperatorState>;
@@ -177,7 +177,7 @@ export function parseOperatorPlan(value, scopeFormat = "repository") {
177
177
  const mappingDiagnostics = [];
178
178
  for (const [unitIndex, unit] of value.units.entries()) {
179
179
  if (!record(unit) || !exactKeys(unit, ["id", "title", "objective", "read", "write", "validation", "acceptance_indices"]) ||
180
- !identifier(unit.id) || ids.has(unit.id) || !text(unit.title) || !text(unit.objective) ||
180
+ !identifier(unit.id) || ids.has(unit.id) || !text(unit.title) || typeof unit.objective !== "string" || !unit.objective.trim() ||
181
181
  !strings(unit.read) || !strings(unit.write) || !strings(unit.validation, true))
182
182
  return planError(`/units/${unitIndex}`, "operator-unit-invalid", "operator-unit-shape");
183
183
  unit.validation.forEach((command, index) => rejectValidationAnnotation(command, `/units/${unitIndex}/validation/${index}`));
@@ -686,12 +686,12 @@ export class OperatorRuntime {
686
686
  }
687
687
  }
688
688
  /** Mission authority is supplied only by the owning profile, never by a model-authored plan. */
689
- prepareMission(root, raw, dispatcher, supersededRunID, terminalChildren = [], replaceRequirements = false) {
690
- return this.serial(root, () => this.prepareOnce(root, raw, undefined, { dispatcher, supersededRunID, terminalChildren, replaceRequirements }));
689
+ prepareMission(root, raw, dispatcher, supersededRunID, terminalChildren = [], replaceRequirements = false, context) {
690
+ return this.serial(root, () => this.prepareOnce(root, raw, undefined, { dispatcher, supersededRunID, terminalChildren, replaceRequirements, context }));
691
691
  }
692
692
  /** Stage a settled mission's replacement; rejected preparation never cancels the usable run. */
693
- replanMission(root, runID, raw, dispatcher) {
694
- return this.serial(root, () => this.prepareOnce(root, raw, undefined, { dispatcher, replaceRunID: runID }));
693
+ replanMission(root, runID, raw, dispatcher, context) {
694
+ return this.serial(root, () => this.prepareOnce(root, raw, undefined, { dispatcher, replaceRunID: runID, context }));
695
695
  }
696
696
  async prepareOnce(root, raw, scopeApprovalTurnID, mission) {
697
697
  const previous = await this.read(root);
@@ -859,6 +859,7 @@ export class OperatorRuntime {
859
859
  task: { title: unit.title, objective: unit.objective }, state: { done: [], next: [unit.title], blocked: [] }, risks: [],
860
860
  verification: unit.validation.map(check => ({ check, status: "not_run", exit_code: null, summary: "Execute in the admitted worker." })),
861
861
  ext: {
862
+ ...(mission ? { "sortie-dogs/mission-context": { write_scope_origin: "coordinator-estimate", ...mission.context } } : {}),
862
863
  "sortie-dogs/write-gate": { operation_manifest: manifestRelative, project_root: this.projectRoot },
863
864
  [ACCEPTANCE_CONTINUITY_EXTENSION]: { schema_version: "0.1", authority: "dispatch", task_id: taskID,
864
865
  criteria: plan.acceptance, fingerprint: acceptanceFingerprint,
@@ -914,23 +915,27 @@ export class OperatorRuntime {
914
915
  throw new OperatorContractError(diagnostics);
915
916
  const contents = [JSON.stringify(handoff), JSON.stringify(manifest)];
916
917
  controls.push({ path: handoffPath, content: contents[0] }, { path: manifestPath, content: contents[1] });
917
- const prompt = ["role: implementation", `task_id: ${taskID}`, `project_root: ${this.projectRoot}`,
918
+ const promptHeader = ["role: implementation", `task_id: ${taskID}`, `project_root: ${this.projectRoot}`,
918
919
  `source_manifest: ${(unit.write.length ? unit.write : unit.read).join(", ") || "none"}`, `operation_manifest: ${manifestRelative}`,
919
- `handoff_path: ${handoffPath}`,
920
- `goal_declaration_path: ${declarationPath}`, "acceptance:", ...plan.acceptance.map(value => ` - ${value}`),
920
+ `handoff_path: ${handoffPath}`, `goal_declaration_path: ${declarationPath}`];
921
+ const commitBoundary = plan.git_lifecycle !== undefined && index === plan.units.length - 1 ? [
922
+ `git_post_commit_validation: ${JSON.stringify(plan.git_lifecycle.post_commit_validation)}`,
923
+ "Complete every source write before invoking any git_post_commit_validation command. Its first exact invocation is the host commit boundary; after it, source mutation and undeclared shell commands are denied. Run every listed command and return its real evidence.",
924
+ ] : [];
925
+ // Mission Workers already must read this host-generated handoff before binding. Preserve its
926
+ // verbatim objective, original requests, criteria and checks there, not in another prompt copy.
927
+ // Saved Tasks and non-Mission dispatch keep their existing text/identity.
928
+ const prompt = (mission ? [...promptHeader, "contract_reference: handoff",
929
+ 'acceptance: handoff.ext["sortie-dogs/acceptance-continuity"].criteria', "validation: handoff.verification",
930
+ `unit_acceptance_indices: ${JSON.stringify(unit.acceptance_indices)}`, "",
931
+ "Read handoff_path once before binding. Implement task.objective; preserve the original requests, global criteria and constraints in ext, and prove this unit's assigned indices. Run verification checks exactly in order within this Task. Use supplied paths; do not reconstruct project_root. Return actual results and limitations, not whole-Mission completion.",
932
+ ...commitBoundary,
933
+ ] : [...promptHeader, "acceptance:", ...plan.acceptance.map(value => ` - ${value}`),
921
934
  "validation:", ...unit.validation.map(value => ` - ${value}`),
922
- ...(mission ? ["Read handoff_path and other repository files using their project-relative paths as supplied. The native working directory is project_root; do not prepend or reconstruct its absolute path for read/search/shell. Copy project_root only when binding the write gate. Preserve explicitly declared external paths."] : []),
923
- mission ? "Read-only investigation commands are unrestricted. Use shell to reproduce and diagnose without asking for command registration. Keep all writes, including generated/transient outputs and cleanup, inside unit.write. Run formal validation exactly as listed, in order and in separate calls, so the host records its real result. If a write scope or formal check must change, return the precise change to your parent Operator or Coordinator; it can extend/redeclare immediately within the original requirements. Diagnostic success is not formal acceptance evidence."
924
- : "Execute validation in its declared order. Earlier entries may be approved generator, build, formatter, or exact cleanup commands required before canonical criterion tests. Every persistent or transient generator output must be declared in unit.write. Cleanup may remove only declared unit.write outputs and must be an explicit ordered command after generation and before post-commit or canonical validation; never add an ignore rule or remove an undeclared path. If any necessary command, input, output, or cleanup is missing, do not run an undeclared command or variant and do not use resume evidence tooling to invent permission; return a contract-repair decision.",
925
- ...(mission ? [] : ["Preserve existing public API success and error return semantics unless acceptance explicitly changes them, and cover those compatibility boundaries in the declared validation."]),
926
- mission
927
- ? "Complete the assigned work and listed validation within this Task. For a known operation, proceed through setup, execution and result collection; preparation alone is not execution. Preserve existing public behavior when changing source. Do not spawn nested subagents. The parent Operator or Coordinator handles any applicable independent review after your return; review is not a prerequisite to execution. Return actual results, including failed or not-started operations, and any exact contract correction needed."
928
- : "Do not spawn nested subagents for consultation. Required consultations belong to the root before dispatch; use the confirmed decisions and evidence declared in the unit objective and inputs. If required consultation results or user decisions are missing, return the exact contract gap to the parent instead of attempting a deeper Task, inventing consent, or asking the user to repeat an already recorded decision.",
929
- ...(plan.git_lifecycle !== undefined && index === plan.units.length - 1 ? [
930
- `git_post_commit_validation: ${JSON.stringify(plan.git_lifecycle.post_commit_validation)}`,
931
- "Complete every source write before invoking any git_post_commit_validation command. Its first exact invocation is the host commit boundary; after it, source mutation and undeclared shell commands are denied. Run every listed command and return its real evidence.",
932
- ] : []),
933
- `unit_acceptance_indices: ${JSON.stringify(unit.acceptance_indices)}`, "", unit.objective].join("\n");
935
+ "Execute validation in its declared order. Earlier entries may be approved generator, build, formatter, or exact cleanup commands required before canonical criterion tests. Every persistent or transient generator output must be declared in unit.write. Cleanup may remove only declared unit.write outputs and must be an explicit ordered command after generation and before post-commit or canonical validation; never add an ignore rule or remove an undeclared path. If any necessary command, input, output, or cleanup is missing, do not run an undeclared command or variant and do not use resume evidence tooling to invent permission; return a contract-repair decision.",
936
+ "Preserve existing public API success and error return semantics unless acceptance explicitly changes them, and cover those compatibility boundaries in the declared validation.",
937
+ "Do not spawn nested subagents for consultation. Required consultations belong to the root before dispatch; use the confirmed decisions and evidence declared in the unit objective and inputs. If required consultation results or user decisions are missing, return the exact contract gap to the parent instead of attempting a deeper Task, inventing consent, or asking the user to repeat an already recorded decision.",
938
+ ...commitBoundary, `unit_acceptance_indices: ${JSON.stringify(unit.acceptance_indices)}`, "", unit.objective]).join("\n");
934
939
  units.push({ unit, task: { subagent_type: profileAgent(this.profile, "dog-worker"), description: unit.title, prompt },
935
940
  handoffPath, manifestPath, hashes: [...contents.map(hash), hash(declaration)], status: "pending", callID: null,
936
941
  childSessionID: null, evidence: [], resultClass: null, repairValidationAttempts: 0, repairValidation: null });
@@ -1521,6 +1526,50 @@ export class OperatorRuntime {
1521
1526
  settled(result) {
1522
1527
  return this.serial(result.rootSessionID, () => this.settledOnce(result));
1523
1528
  }
1529
+ /** Correct an estimated Mission scope without dispatching, restarting, or reserving another unit. */
1530
+ expandMissionWriteScope(root, child, taskID, paths, activate) {
1531
+ return this.serial(root, async () => {
1532
+ const state = await this.required(root);
1533
+ const unit = state.units.find(item => /^task_id: (.+)$/m.exec(item.task.prompt)?.[1] === taskID && item.childSessionID === child);
1534
+ if (!unit || !["running", "failed", "succeeded"].includes(unit.status) ||
1535
+ !["running", "awaiting-decision", "awaiting-acceptance"].includes(state.phase) || !unit.callID || state.gitLifecycle) {
1536
+ throw new Error("mission-scope-update-current-task-required");
1537
+ }
1538
+ await this.verifyControls(unit);
1539
+ const write = [...new Set([...unit.unit.write, ...paths.map(normalizeExecutionScope)])];
1540
+ for (const path of write) {
1541
+ const scoped = relative(this.projectRoot, resolve(this.projectRoot, normalizeManifestScope(path).path)).replaceAll("\\", "/");
1542
+ if ([".git", ...Object.values(RUNTIME_PROFILES).map(profile => profile.stateDirectory)].some(dir => scoped === dir || scoped.startsWith(`${dir}/`))) {
1543
+ throw new Error(`operator-control-write-forbidden:${path}`);
1544
+ }
1545
+ }
1546
+ if (JSON.stringify(write) === JSON.stringify(unit.unit.write))
1547
+ return;
1548
+ const oldManifest = await readFile(unit.manifestPath, "utf8"), oldHandoff = await readFile(unit.handoffPath, "utf8");
1549
+ const manifest = { ...JSON.parse(oldManifest), write };
1550
+ const handoff = JSON.parse(oldHandoff);
1551
+ const m = validateOperationManifestSchema(manifest), h = validateHandoffSchema(handoff);
1552
+ if (!m.ok || !h.ok)
1553
+ throw new Error("mission-scope-update-invalid-contract");
1554
+ const nextManifest = JSON.stringify(manifest), nextHandoff = JSON.stringify(handoff);
1555
+ let rollback;
1556
+ try {
1557
+ await writeFile(unit.manifestPath, nextManifest);
1558
+ await writeFile(unit.handoffPath, nextHandoff);
1559
+ rollback = await activate(manifest);
1560
+ unit.unit = { ...unit.unit, write };
1561
+ unit.hashes = [hash(nextHandoff), hash(nextManifest), unit.hashes[2]];
1562
+ unit.task = { ...unit.task, prompt: unit.task.prompt.replace(/^source_manifest: .*$/mu, `source_manifest: ${write.join(", ")}`) };
1563
+ await this.save(state);
1564
+ }
1565
+ catch (error) {
1566
+ await writeFile(unit.manifestPath, oldManifest);
1567
+ await writeFile(unit.handoffPath, oldHandoff);
1568
+ await rollback?.();
1569
+ throw error;
1570
+ }
1571
+ });
1572
+ }
1524
1573
  async settledOnce(result) {
1525
1574
  const state = await this.read(result.rootSessionID);
1526
1575
  const unit = state?.units.find(item => item.callID === result.callID);
@@ -23,6 +23,7 @@ export interface ScopeLease {
23
23
  assertHeld(): Promise<void>;
24
24
  isReleased(): Promise<boolean>;
25
25
  release(): Promise<void>;
26
+ replaceScope(scope: WorktreeScope): Promise<void>;
26
27
  abandon(): Promise<void>;
27
28
  close(): void;
28
29
  }
@@ -198,7 +198,29 @@ export class ScopeLeaseRegistry {
198
198
  return Object.freeze({
199
199
  id,
200
200
  ownerId,
201
- scope: Object.freeze({ read: [...scope.read], write: [...scope.write] }),
201
+ get scope() { return Object.freeze({ read: [...scope.read], write: [...scope.write] }); },
202
+ replaceScope: (replacement) => {
203
+ const normalized = this.validateRequest(replacement);
204
+ const result = pending.then(async () => {
205
+ if (!active || closing)
206
+ throw new ScopeLeaseError("not-held", "Lease is not held.");
207
+ await this.withLock(async (mutex) => {
208
+ const state = await this.load();
209
+ const held = state.leases.find(lease => lease.id === id && lease.ownerHash === ownerHash && lease.tokenHash === credentialHash);
210
+ if (!held || held.expiresAt <= Date.now())
211
+ throw new ScopeLeaseError("not-held", "Lease is not held.");
212
+ if (state.leases.some(lease => lease.id !== id && lease.expiresAt > Date.now() && worktreeScopesConflict(normalized, lease))) {
213
+ throw new ScopeLeaseError("scope-conflict", "Requested scope is already leased.");
214
+ }
215
+ held.read = [...normalized.read];
216
+ held.write = [...normalized.write];
217
+ await this.save({ ...state, revision: state.revision + 1 }, mutex);
218
+ scope = normalized;
219
+ });
220
+ });
221
+ pending = result.catch(() => undefined);
222
+ return result;
223
+ },
202
224
  heartbeat: () => serialized("heartbeat"),
203
225
  assertHeld: () => serialized("assert"),
204
226
  isReleased: () => this.withLock(async () => {