sortie-dogs 0.13.0 → 0.13.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. package/README.md +53 -6
  2. package/dist/asset-version.d.ts +1 -1
  3. package/dist/asset-version.js +1 -1
  4. package/dist/core/contract-limits.d.ts +1 -1
  5. package/dist/core/contract-limits.js +1 -1
  6. package/dist/core/goal-bound.d.ts +7 -0
  7. package/dist/core/goal-bound.js +7 -1
  8. package/dist/core/initialize.d.ts +1 -1
  9. package/dist/core/initialize.js +1 -1
  10. package/dist/core/operator-mission.d.ts +24 -9
  11. package/dist/core/operator-mission.js +48 -52
  12. package/dist/core/operator-runtime.d.ts +23 -2
  13. package/dist/core/operator-runtime.js +98 -31
  14. package/dist/core/scope-lease-registry.d.ts +1 -0
  15. package/dist/core/scope-lease-registry.js +23 -1
  16. package/dist/plugin/adaptive-remediation-host.d.ts +1 -1
  17. package/dist/plugin/adaptive-remediation-host.js +1 -1
  18. package/dist/plugin/index.js +207 -32
  19. package/dist/plugin/mission-review.js +58 -14
  20. package/dist/plugin/model-cost.d.ts +3 -3
  21. package/dist/plugin/model-cost.js +10 -6
  22. package/dist/plugin/model-routing.d.ts +7 -7
  23. package/dist/plugin/model-routing.js +7 -12
  24. package/dist/plugin/profiled.js +325 -93
  25. package/dist/plugin/protected-snapshot.d.ts +6 -0
  26. package/dist/plugin/protected-snapshot.js +100 -25
  27. package/dist/plugin/run-metrics.d.ts +1 -0
  28. package/dist/plugin/run-metrics.js +5 -27
  29. package/dist/plugin/runtime-bridge.d.ts +24 -1
  30. package/dist/plugin/v2.js +13 -7
  31. package/dist/plugin/validation-scratch.d.ts +2 -0
  32. package/dist/plugin/validation-scratch.js +38 -8
  33. package/dist/runtime-assets-v010.js +8 -32
  34. package/dist/runtime-assets.d.ts +2 -2
  35. package/dist/runtime-assets.js +2 -2
  36. package/dist/runtime-mission-assets.d.ts +6 -1
  37. package/dist/runtime-mission-assets.js +175 -94
  38. package/package.json +2 -1
package/README.md CHANGED
@@ -1,5 +1,9 @@
1
1
  # Sortie-dogs
2
2
 
3
+ <p align="center">
4
+ <img src="docs/assets/sortie-dogs-logo.png" alt="Sortie-dogs logo" width="640">
5
+ </p>
6
+
3
7
  **A goal-preserving, adaptive execution harness for OpenCode that optimizes cost,
4
8
  time, and proof without taking your setup over.**
5
9
 
@@ -28,6 +32,14 @@ implementation, validation, review, and model routing.
28
32
  Guides: [日本語](docs/guide-ja.md) · [简体中文](docs/guide-zh-CN.md) ·
29
33
  [Testing](docs/testing.md) · [CLI testing](docs/cli-testing.md)
30
34
 
35
+ ## SWE-bench Lite: 170/300 (56.67%)
36
+
37
+ The fixed **Sortie-dogs v0.12.24** harness resolved **170 of 300 SWE-bench Lite test issues** in one pass@1 campaign, with 9 empty patches and no official evaluation errors. Every instance has a frozen prediction and an inference-time trajectory. The task Workers ran `openai/gpt-6-luna-fast#max`; operator, coordinator and review roles ran `openai/gpt-6-sol#xhigh`. This is a system result, **not** a Luna-only model comparison or a Verified/full SWE-bench score.
38
+
39
+ [Technical report and per-repository results](docs/benchmarks/swebench-lite-v01224-test300-2026-09-29.md) · [Public predictions, logs and trajectories](https://github.com/zufall-upon/sortie-dogs-swebench-lite-20260929)
40
+
41
+ The single official 300-instance report and frozen predictions are hash-bound in the report. Confirmed inference expense was **$162.99**; a separate **$34.60** of usage has unknown pricing and is held against the campaign cap, **not** counted as known expense. Leaderboard registration and maintainer acceptance are separate from this official local evaluation.
42
+
31
43
  > **Beta:** v0.12.2 builds on the v0.10.23 execution engine. Runtime behavior,
32
44
  > configuration, and generated assets may still change before 1.0.
33
45
 
@@ -99,6 +111,8 @@ Official SWE-bench Lite `dev` results on the same 23 public instances:
99
111
  | v0.12.16 (`9b05a34` release; `b1a6c0e` runner) | 4 / 23 (17.4%) | 5 | Fresh 23-task run; eight slots; 40-minute timeout | [Campaign](docs/swebench-v01216-dev23-2026-09-27.md) |
100
112
  | v0.12.19 (`24f5386` release; matched rerun) | 8 / 23 (34.8%) | 0 | Eight slots; effective $2/instance; 40-minute timeout; one inference timeout | [Official result and provenance](docs/benchmarks/swebench-v01220-operation-observability-2026-09-28.md) |
101
113
  | v0.12.20 (`628eb81` release) | 7 / 23 (30.4%) | 0 | Eight slots; effective $2/instance; 40-minute timeout | [Official result and caveat](docs/benchmarks/swebench-v01220-operation-observability-2026-09-28.md) |
114
+ | v0.12.25 (`49eb1e4` release) | 7 / 23 (30.4%) | 0 | Eight slots; $2/instance; $30 total cap; 20-minute progress check / 40-minute hard maximum | [Comparison baseline](#v0131-dev23-2026-09-30) |
115
+ | v0.13.1 (`d19e8be` release; 2026-09-30) | **8 / 23 (34.8%)** | 0 | Eight slots; $2/instance; $46 total cap; 20-minute progress check / 40-minute hard maximum; GPT-6.1 Sol + Luna Fast | [Run summary](#v0131-dev23-2026-09-30) |
102
116
 
103
117
  Every row has 23 submitted official predictions; an empty patch counts against
104
118
  the score, not as a missing evaluation. The v0.10.14 report does not separately
@@ -120,6 +134,39 @@ single run-to-run difference does not establish causation.
120
134
 
121
135
  Historical qualification references remain in [benchmark reference](docs/benchmark-reference.md).
122
136
 
137
+ #### v0.13.1 dev23 (2026-09-30)
138
+
139
+ One fresh pass@1 run and one official SWE-bench harness evaluation resolved
140
+ **8/23**, versus **7/23** for v0.12.25. The new resolution was
141
+ `pylint-dev__astroid-1333`; all seven previously resolved IDs were retained.
142
+ Resolved by repository: marshmallow **2/2**, pvlib **0/5**, pydicom **2/5**,
143
+ astroid **3/5**, pyvista **0/1**, sqlfluff **1/5**.
144
+
145
+ - The scored row is the user-requested fresh run after a host restart. The
146
+ interrupted initial run is excluded from this score; the fresh run made one
147
+ attempt per instance with no inference retry.
148
+ - The dataset revision (`6ec7bb89b9342f664a54a6e0a6ea6501d3437cc2`), public rows,
149
+ and all 23 official evaluation image IDs match the v0.12.25 run. Both used
150
+ `official-image-testbed` and sequential official scoring.
151
+ - Operator/Coordinator/Reviewer/Advisor defaults changed to
152
+ `openai/gpt-6.1-sol#xhigh`. Actual task Workers remained
153
+ `openai/gpt-6-luna-fast#max`, observed across all 23 instances. The harness and
154
+ total budget also changed, so the extra resolution cannot be attributed to
155
+ the model change alone.
156
+ - Inference ended with 20 normal completions, two timeouts
157
+ (`pvlib__pvlib-python-1154`, `sqlfluff__sqlfluff-1763`) and one agent failure
158
+ (`pvlib__pvlib-python-1854`). All patches, including stopped attempts, were
159
+ officially scored: 23 completed evaluations, zero empty patches and zero
160
+ official evaluation errors or infrastructure failures.
161
+ - Known estimated inference cost: **$17.73**; separate unknown-usage hold:
162
+ **$1.98**, not counted as known expense. Inference wall time was about
163
+ **81 minutes**, followed by **7.2 minutes** of official scoring.
164
+ - Fixed release commit: `d19e8be0d21180cc23ad2ae4b853d846a18e77bc`;
165
+ package SHA-256: `99300ceec0c3eee4fa1f984fed50d15d58b2ff4455df0041850ecd63514b0a51`.
166
+ OpenCode **2.0.20**, official harness **5.0.2**. Local evidence is retained in
167
+ `_testenv/swebench-v0131-dev23-20260930-r2/result-summary.json`; generated
168
+ predictions, databases and raw logs are not committed.
169
+
123
170
  ## Mission tools
124
171
 
125
172
  1. `start_mission`: Operator supplies concise requirements; the host saves original messages and returns a Coordinator task.
@@ -202,7 +249,7 @@ not invent, probe, or translate variant names.
202
249
  - `freeTierFallbackModels`: ordered global last-resort model IDs. Default:
203
250
  `opencode/deepseek-v4-flash-free`; `[]` disables this fallback.
204
251
  - `dedicatedWorkerModel`: canonical stable serial target, default
205
- `openai/gpt-6-sol` / `medium`. The v0.10 profile also supplies its explicit
252
+ `openai/gpt-6.1-sol` / `medium`. The v0.10 profile also supplies its explicit
206
253
  role routes below; do not infer the v0.10 worker route from this stable setting.
207
254
  - `consultation.strategy`: fixed advisor identity, optional `required`, and
208
255
  positive `maxCallsPerCandidate`; default one call and not required.
@@ -239,14 +286,14 @@ or explicit risk. Workers own static, targeted, and related checks; the root own
239
286
  canonical and full-suite checks. An unchanged candidate, command, and environment
240
287
  reuse the same evidence instead of spending the validation budget again.
241
288
 
242
- ### Default v0.12 routes
289
+ ### Default routes
243
290
 
244
- - `dog-operator`: `openai/gpt-6-sol` / `xhigh`
245
- - `dogs-coordinator`: `openai/gpt-6-sol` / `xhigh`
291
+ - `dog-operator`: `openai/gpt-6.1-sol` / `xhigh`
292
+ - `dogs-coordinator`: `openai/gpt-6.1-sol` / `xhigh`
246
293
  - `dog-worker-v010`: `openai/gpt-6-luna-fast` / `max`
247
294
  - `dog-scout-v010`: `openai/gpt-6-luna-fast` / `max`
248
- - `dog-reviewer-v010`: `openai/gpt-6-sol` / `xhigh`
249
- - `dog-advisor-v010`: `openai/gpt-6-sol` / `xhigh`
295
+ - `dog-reviewer-v010`: `openai/gpt-6.1-sol` / `xhigh`
296
+ - `dog-advisor-v010`: `openai/gpt-6.1-sol` / `xhigh`
250
297
 
251
298
  An explicit model and variant selected in OpenCode remains authoritative for that
252
299
  session. Child role defaults fill absent native settings and may be overridden by
@@ -3,5 +3,5 @@
3
3
  * installed project marker without importing every asset body.
4
4
  */
5
5
  export declare const RUNTIME_ASSET_VERSION = "0.3.89-completion-proof-v1";
6
- export declare const V010_RUNTIME_ASSET_VERSION = "0.13.0-fast-first-v1";
6
+ export declare const V010_RUNTIME_ASSET_VERSION = "0.13.2-anko-recovery-v1";
7
7
  export type RuntimeAssetVersion = typeof RUNTIME_ASSET_VERSION | typeof V010_RUNTIME_ASSET_VERSION;
@@ -3,4 +3,4 @@
3
3
  * installed project marker without importing every asset body.
4
4
  */
5
5
  export const RUNTIME_ASSET_VERSION = "0.3.89-completion-proof-v1";
6
- export const V010_RUNTIME_ASSET_VERSION = "0.13.0-fast-first-v1";
6
+ export const V010_RUNTIME_ASSET_VERSION = "0.13.2-anko-recovery-v1";
@@ -1,7 +1,7 @@
1
1
  /** Common text bounds shared by handoff, manifest and goal evidence validation. */
2
2
  export declare const CONTRACT_TEXT_LIMITS: Readonly<{
3
3
  title: 160;
4
- objective: 2000;
4
+ objective: 32768;
5
5
  statement: 1000;
6
6
  command: 8192;
7
7
  path: 512;
@@ -1,2 +1,2 @@
1
1
  /** Common text bounds shared by handoff, manifest and goal evidence validation. */
2
- export const CONTRACT_TEXT_LIMITS = Object.freeze({ title: 160, objective: 2000, statement: 1000, command: 8192, path: 512 });
2
+ export const CONTRACT_TEXT_LIMITS = Object.freeze({ title: 160, objective: 32768, statement: 1000, command: 8192, path: 512 });
@@ -39,6 +39,13 @@ export interface GoalEvidence {
39
39
  readonly candidate_paths: readonly string[];
40
40
  /** Missing on legacy evidence: retain its original all-paths snapshot recipe. */
41
41
  readonly source_policy?: "project-files-v1" | "declared-paths-v1";
42
+ /** Fixed when validation starts. Full manifest_hash still identifies the historical execution contract. */
43
+ readonly freshness?: {
44
+ readonly contract_hash: string;
45
+ readonly scratch_paths: readonly string[];
46
+ readonly protected_paths: readonly string[];
47
+ readonly environment: Readonly<Record<string, string | null>>;
48
+ };
42
49
  };
43
50
  readonly execution: {
44
51
  readonly command: readonly string[];
@@ -64,7 +64,13 @@ export function validGoalEvidence(value, state) {
64
64
  protectedBinding.source_paths.every(text) && Array.isArray(protectedBinding.candidate_paths) &&
65
65
  protectedBinding.candidate_paths.every(text) &&
66
66
  (protectedBinding.source_policy === undefined || protectedBinding.source_policy === "project-files-v1" ||
67
- protectedBinding.source_policy === "declared-paths-v1");
67
+ protectedBinding.source_policy === "declared-paths-v1") &&
68
+ (protectedBinding.freshness === undefined || (protectedBinding.freshness !== null && typeof protectedBinding.freshness === "object" &&
69
+ HASH.test(protectedBinding.freshness.contract_hash) &&
70
+ Array.isArray(protectedBinding.freshness.scratch_paths) && protectedBinding.freshness.scratch_paths.every(text) &&
71
+ Array.isArray(protectedBinding.freshness.protected_paths) && protectedBinding.freshness.protected_paths.every(text) &&
72
+ protectedBinding.freshness.environment !== null && typeof protectedBinding.freshness.environment === "object" &&
73
+ !Array.isArray(protectedBinding.freshness.environment) && Object.values(protectedBinding.freshness.environment).every(value => value === null || typeof value === "string")));
68
74
  const matches = criteria.length > 0 && criteria.length === value.measurement.criterion_ids.length &&
69
75
  criteria.every((criterion) => criterion.target === value.measurement.target &&
70
76
  criterion.entrypoint === value.measurement.entrypoint && criterion.workload === value.measurement.workload &&
@@ -11,7 +11,7 @@ export declare class ProjectInitializationError extends Error {
11
11
  readonly code: ProjectInitializationErrorCode;
12
12
  constructor(code: ProjectInitializationErrorCode, message: string, options?: ErrorOptions);
13
13
  }
14
- export declare const V010_COORDINATOR_MODEL = "openai/gpt-6-sol#xhigh";
14
+ export declare const V010_COORDINATOR_MODEL = "openai/gpt-6.1-sol#xhigh";
15
15
  /**
16
16
  * Report a user config that routes the v0.10 Coordinator away from its packaged model. Measured Luna
17
17
  * Coordinators repeated plans and review dispatches until timeout, so init surfaces it without editing it.
@@ -542,7 +542,7 @@ async function initializeRoot(requestedRoot, layout, installAssets = runtimeAsse
542
542
  preservedLegacyPaths,
543
543
  };
544
544
  }
545
- export const V010_COORDINATOR_MODEL = "openai/gpt-6-sol#xhigh";
545
+ export const V010_COORDINATOR_MODEL = "openai/gpt-6.1-sol#xhigh";
546
546
  /**
547
547
  * Report a user config that routes the v0.10 Coordinator away from its packaged model. Measured Luna
548
548
  * Coordinators repeated plans and review dispatches until timeout, so init surfaces it without editing it.
@@ -19,6 +19,7 @@ export interface MissionEvidenceExcerpt {
19
19
  export interface MissionReviewScope {
20
20
  read: string[];
21
21
  write: string[];
22
+ validationBindings?: NonNullable<import("./goal-bound.js").GoalEvidence["protected_binding"]>[];
22
23
  }
23
24
  export interface MissionConsultation {
24
25
  id: string;
@@ -47,7 +48,9 @@ export interface MissionAttempt {
47
48
  status: "pending" | "dispatched" | "succeeded" | "failed" | "cancelled" | "unconfirmed";
48
49
  callID?: string;
49
50
  childSessionID?: string;
51
+ dispatchFingerprint?: string;
50
52
  nativeOutcome?: "completed" | "failed" | "unknown";
53
+ terminal?: import("../plugin/runtime-bridge.js").MissionWorkerTerminalRecord;
51
54
  observedModel?: string;
52
55
  observedVariant?: string;
53
56
  failure?: {
@@ -80,11 +83,23 @@ export interface MissionExecution {
80
83
  startedAt: string;
81
84
  completedAt?: string;
82
85
  exit?: number;
83
- status?: "completed" | "error";
86
+ status?: "running" | "completed" | "error";
87
+ shellID?: string;
84
88
  outcome?: "not-started" | "execution-failed" | "executed";
85
89
  result?: Record<string, unknown>;
86
90
  }[];
87
91
  }
92
+ export interface MissionLaunchConditions {
93
+ entrypoint?: string;
94
+ inputs?: string[];
95
+ timeout_seconds?: number;
96
+ cost_limit_usd?: number;
97
+ benchmark_attempts?: number;
98
+ grading?: "none" | "official";
99
+ source: string;
100
+ applies_to: string;
101
+ recordedAt?: string;
102
+ }
88
103
  export interface OperatorMission {
89
104
  version: "0.12";
90
105
  id: string;
@@ -95,10 +110,14 @@ export interface OperatorMission {
95
110
  kind?: "implementation" | "operation";
96
111
  /** Native shell observations of the requested operation, separate from auxiliary checks. */
97
112
  execution?: MissionExecution;
113
+ /** Fixed benchmark conditions and their provenance, separate from internal Worker counters. */
114
+ launchConditions?: MissionLaunchConditions[];
98
115
  requirements: {
99
116
  id: string;
100
117
  text: string;
101
118
  }[];
119
+ /** Explicit user path prohibitions, unlike the Coordinator's estimated write list. */
120
+ prohibitedWrite?: string[];
102
121
  /** The current user intentionally replaced the predecessor's requirements. */
103
122
  requirementsReplaced?: boolean;
104
123
  phase: "open" | "running" | "submitted" | "completed" | "cancelled";
@@ -141,15 +160,10 @@ export interface OperatorMission {
141
160
  child?: string;
142
161
  /** Completed, independent initial review for this mission, not merely an inherited child ID. */
143
162
  initialPrompt?: string;
144
- /** Reviews on this mission that found only missing evidence; bounded by MISSION_EVIDENCE_GAP_REVIEW_LIMIT. */
163
+ /** Observed evidence-only reviews; reporting only, never an acceptance threshold. */
145
164
  evidenceGapReviews?: number;
146
165
  };
147
166
  }
148
- /**
149
- * Evidence-only findings do not establish a defect. Each extra round costs a full Reviewer pass and often a
150
- * Worker unit, so after this many evidence-only reviews the candidate may be submitted with the gaps listed.
151
- */
152
- export declare const MISSION_EVIDENCE_GAP_REVIEW_LIMIT = 2;
153
167
  export declare const MISSION_CONSULTATION_LIMIT = 32;
154
168
  /** Review coverage survives a narrower replan; it is not a Worker write grant. */
155
169
  export declare function missionReviewScope(previous: MissionReviewScope | undefined, ...runs: OperatorState[]): MissionReviewScope;
@@ -183,6 +197,7 @@ export declare class OperatorMissionRuntime {
183
197
  read(root: string): Promise<OperatorMission | undefined>;
184
198
  required(root: string): Promise<OperatorMission>;
185
199
  capture(root: string, request: MissionRequest): Promise<void>;
200
+ recordLaunchConditions(root: string, raw: unknown): Promise<OperatorMission>;
186
201
  start(root: string, requirements: unknown, replaceRequirements?: boolean, options?: {
187
202
  kind?: OperatorMission["kind"];
188
203
  context?: MissionContext[];
@@ -217,5 +232,5 @@ export declare function missionPlan(mission: OperatorMission, raw: unknown): Ope
217
232
  export declare function missionPacket(mission: OperatorMission, run?: OperatorState): Record<string, unknown>;
218
233
  /** Models forward a short capability, never recopy the host's evidence hashes and source packet. */
219
234
  export declare function missionReviewTask(mission: OperatorMission): OperatorTask;
220
- /** Project grouped R-ID traces, including consecutive ranges, into the existing per-criterion mapping. */
221
- export declare function missionReviewTraces(mission: OperatorMission, raw: unknown): string[];
235
+ /** Optional implementation notes supplement host-observed source/checks; prose is not a coverage gate. */
236
+ export declare function missionReviewTraces(_mission: OperatorMission, raw: unknown): string[];
@@ -18,17 +18,14 @@ export function missionValidationCommand(command) {
18
18
  const expression = `exec(${JSON.stringify(heredoc[4])})`;
19
19
  return `${heredoc[1]} -c '${expression.replaceAll("'", "'\\''")}'`;
20
20
  }
21
- /**
22
- * Evidence-only findings do not establish a defect. Each extra round costs a full Reviewer pass and often a
23
- * Worker unit, so after this many evidence-only reviews the candidate may be submitted with the gaps listed.
24
- */
25
- export const MISSION_EVIDENCE_GAP_REVIEW_LIMIT = 2;
26
21
  export const MISSION_CONSULTATION_LIMIT = 32;
27
22
  /** Review coverage survives a narrower replan; it is not a Worker write grant. */
28
23
  export function missionReviewScope(previous, ...runs) {
24
+ const bindings = [...(previous?.validationBindings ?? []), ...runs.flatMap(run => run.units.flatMap(unit => (unit.evidence ?? []).flatMap(proof => proof.protected_binding?.freshness ? [proof.protected_binding] : [])))];
29
25
  return {
30
26
  read: [...new Set([...(previous?.read ?? []), ...runs.flatMap(run => run.units.flatMap(({ unit }) => unit.read ?? []))])].sort(),
31
27
  write: [...new Set([...(previous?.write ?? []), ...runs.flatMap(run => run.units.flatMap(({ unit }) => unit.write))])].sort(),
28
+ ...(bindings.length ? { validationBindings: [...new Map(bindings.map(binding => [JSON.stringify(binding), binding])).values()] } : {}),
32
29
  };
33
30
  }
34
31
  /** Classify an independent Reviewer's first line. Anything else is a finding. */
@@ -38,7 +35,7 @@ export function missionReviewVerdict(text) {
38
35
  /** Whether the recorded review permits submission and acceptance of the current candidate. */
39
36
  export function missionReviewAccepted(review) {
40
37
  return review.verdict === "PASS" || review.verdict === "skipped-low-risk" ||
41
- (review.verdict === "evidence-gaps" && (review.evidenceGapReviews ?? 0) >= MISSION_EVIDENCE_GAP_REVIEW_LIMIT);
38
+ review.verdict === "evidence-gaps";
42
39
  }
43
40
  export function missionExecutionStatus(mission) {
44
41
  if (mission.kind !== "operation")
@@ -219,6 +216,24 @@ export class OperatorMissionRuntime {
219
216
  }
220
217
  });
221
218
  }
219
+ recordLaunchConditions(root, raw) {
220
+ if (!record(raw) || typeof raw.source !== "string" || !raw.source.trim() || typeof raw.applies_to !== "string" || !raw.applies_to.trim() ||
221
+ (raw.entrypoint !== undefined && (typeof raw.entrypoint !== "string" || !raw.entrypoint.trim())) ||
222
+ (raw.inputs !== undefined && (!Array.isArray(raw.inputs) || !raw.inputs.every(path => typeof path === "string" && path.trim()))) ||
223
+ ["timeout_seconds", "cost_limit_usd", "benchmark_attempts"].some(key => raw[key] !== undefined &&
224
+ (typeof raw[key] !== "number" || !Number.isFinite(raw[key]) || raw[key] <= 0)) ||
225
+ (raw.benchmark_attempts !== undefined && !Number.isSafeInteger(raw.benchmark_attempts)) ||
226
+ (raw.grading !== undefined && !["none", "official"].includes(String(raw.grading))) ||
227
+ Object.keys(raw).some(key => !["entrypoint", "inputs", "timeout_seconds", "cost_limit_usd", "benchmark_attempts", "grading", "source", "applies_to"].includes(key))) {
228
+ throw new Error("mission-launch-conditions-invalid");
229
+ }
230
+ return this.update(root, state => {
231
+ state.launchConditions ??= [];
232
+ if (state.launchConditions.some(item => { const { recordedAt: _at, ...value } = item; return JSON.stringify(value) === JSON.stringify(raw); }))
233
+ return;
234
+ state.launchConditions.push({ ...raw, recordedAt: new Date().toISOString() });
235
+ });
236
+ }
222
237
  start(root, requirements, replaceRequirements = false, options = {}) {
223
238
  return this.serial(root, async () => {
224
239
  if (!Array.isArray(requirements) || requirements.length === 0 || requirements.length > 64 ||
@@ -371,6 +386,8 @@ export class OperatorMissionRuntime {
371
386
  "Start the first useful Worker promptly. No proposal/approval phase. Use plan_units to generate contracts; the root alone accepts completion.",
372
387
  "Escalate only a completion candidate, a user-only decision, or an extension of original requirements/budget. Unit progress is published without stopping you.",
373
388
  "Requirements:", ...state.requirements.map(item => `${item.id}: ${item.text}`),
389
+ `Confirmed launch conditions (fixed limits, not consumption or remaining budget): ${JSON.stringify(state.launchConditions ?? [])}`,
390
+ `Explicit user write prohibitions: ${JSON.stringify(state.prohibitedWrite ?? [])}`,
374
391
  `Work kind: ${state.kind ?? "implementation"}. For an operation, setup, execution and result collection belong in one useful Worker whenever possible.`,
375
392
  ...(state.context?.length ? ["Prior conversation context (task data; preserve the selected target, not superseded obligations):",
376
393
  ...state.context.map(item => `--- ${item.role}:${item.id} ---\n${item.text}`)] : []),
@@ -388,7 +405,7 @@ export function missionPlan(mission, raw) {
388
405
  const line = (field) => {
389
406
  if (typeof value[field] !== "string" || !value[field].trim())
390
407
  throw new Error(`mission-unit-${index + 1}: ${field} required`);
391
- return value[field].replace(/[\r\n]+/gu, " ");
408
+ return field === "objective" ? value[field] : value[field].replace(/[\r\n]+/gu, " ");
392
409
  };
393
410
  const paths = (field) => {
394
411
  const entries = value[field] ?? [];
@@ -455,6 +472,8 @@ export function missionPacket(mission, run) {
455
472
  completed_units: predecessor.units.filter(unit => unit.status === "succeeded").length,
456
473
  note: "Historical results and spend are retained; they do not complete the current requirements." } } : {}),
457
474
  requirements: mission.requirements, original_request_refs: mission.requests.map(item => `user:${item.id}`),
475
+ launch_conditions: mission.launchConditions ?? [], prohibited_write: mission.prohibitedWrite ?? [],
476
+ accounting_scope: "Worker units are not benchmark attempts. Host budget is Worker-only; orchestration, Review and external campaign costs are excluded. Launch caps are fixed conditions, not a known campaign remainder.",
458
477
  submission: mission.submission, progress: mission.progress, consultations: mission.consultations ?? [],
459
478
  attempts: mission.attempts ?? [], ...(mission.rescue ? { rescue: mission.rescue } : {}),
460
479
  operation: { kind: mission.kind ?? "implementation", status: missionExecutionStatus(mission),
@@ -469,7 +488,7 @@ export function missionPacket(mission, run) {
469
488
  run_id: mission.review.runID, current_run: currentReview,
470
489
  source_fingerprint: mission.review.source, reviewer_session_id: mission.review.child ?? null,
471
490
  result: mission.review.result ?? null, evidence_gap_reviews: mission.review.evidenceGapReviews ?? 0,
472
- evidence_gap_review_limit: MISSION_EVIDENCE_GAP_REVIEW_LIMIT, accepted: reviewAccepted,
491
+ evidence_gaps_advisory: true, accepted: reviewAccepted,
473
492
  passed: currentReview && mission.review.verdict === "PASS", permits_submission: reviewAccepted && operationComplete } : null,
474
493
  ...(run ? { run_id: run.runID, status: run.phase, decision: run.decision,
475
494
  units: run.units.map(unit => ({ id: unit.unit.id, title: unit.unit.title, status: unit.status,
@@ -482,21 +501,21 @@ export function missionPacket(mission, run) {
482
501
  next_action: mission.phase === "completed" ? "Mission completed. Report the accepted result and retained review gaps; no further dispatch or completion call is needed."
483
502
  : mission.phase === "submitted" && mission.submission?.status === "ready"
484
503
  ? "Operator: compare the submitted candidate with the original requirements and actual evidence, then complete_mission if satisfied. Report remaining evidence gaps; they are not a review PASS."
485
- : run?.phase === "awaiting-decision" ? (run.units.some(unit => unit.dispatchDenial)
486
- ? "Coordinator: inspect units[].dispatch_denial before changing the plan. Correct only its diagnosed cause; do not repeat an unchanged refused Task or replan for a host-state mismatch. Report an unresolved runtime mismatch with the loaded runtime identity; preserve requirements and cumulative spend."
487
- : run.units.some(unit => unit.status === "failed" && unit.resultClass === "acceptance" && unit.failure?.outcome === "fail" && !unit.normalRemediationUsed)
488
- ? "Coordinator: one exact declared validation failed. Reconcile the returned Worker and cumulative budget, then call retry_mission_unit once for that unit before replanning. This preserves the same scope and acceptance."
489
- : run.units.some(unit => unit.status === "failed" && unit.resultClass === "acceptance" && unit.failure?.outcome === "fail" && unit.normalRemediationUsed && !unit.terminalRescue)
490
- ? "Coordinator: the one ordinary remediation failed the declared validation again. If its native terminal, writer release, exact candidate and cumulative budget permit, call rescue_mission_unit once; inspect and report a non_rescue reason, then continue ordinary correction within budget. Rescue does not bypass validation, review or root acceptance."
491
- : "Coordinator: correct the cause and call plan_units with the remaining work and all requirements; budget is cumulative.")
492
- : run?.phase === "awaiting-acceptance" && !operationComplete
493
- ? `Requested operation is ${operationStatus}. Continue the actual operation or report its blocker; auxiliary checks and review disposition cannot complete it.`
494
- : run?.phase === "awaiting-acceptance" ? (currentReview && mission.review?.verdict === "evidence-gaps" && !reviewAccepted
495
- ? "Coordinator: the Reviewer found only missing evidence. Supply focused original-file excerpts through review_mission evidence and traces; do not re-implement. At the evidence-gap limit, review ends with the gaps listed, not with execution proof."
496
- : reviewAccepted
497
- ? "Coordinator: review permits submission. Submit the candidate with any remaining gaps; do not repeat passed validation or review. Operator performs final acceptance."
504
+ : mission.kind === "operation" && operationStatus === "running"
505
+ ? "The declared operation is already running. Inspect its native shell/progress; do not start another Worker or run. Wait for a terminal result, or report the existing run as blocked if its completion cannot be observed."
506
+ : run?.phase === "awaiting-decision" ? (run.units.some(unit => unit.dispatchDenial)
507
+ ? "Coordinator: inspect units[].dispatch_denial before changing the plan. Correct only its diagnosed cause; do not repeat an unchanged refused Task or replan for a host-state mismatch. Report an unresolved runtime mismatch with the loaded runtime identity; preserve requirements and cumulative spend."
508
+ : run.units.some(unit => unit.status === "failed" && unit.resultClass === "acceptance" && unit.failure?.outcome === "fail" && !unit.normalRemediationUsed)
509
+ ? "Coordinator: one exact declared validation failed. Reconcile the returned Worker and cumulative budget, then call retry_mission_unit once for that unit before replanning. This preserves the same scope and acceptance."
510
+ : run.units.some(unit => unit.status === "failed" && unit.resultClass === "acceptance" && unit.failure?.outcome === "fail" && unit.normalRemediationUsed && !unit.terminalRescue)
511
+ ? "Coordinator: the one ordinary remediation failed the declared validation again. If its native terminal, writer release, exact candidate and cumulative budget permit, call rescue_mission_unit once; inspect and report a non_rescue reason, then continue ordinary correction within budget. Rescue does not bypass validation, review or root acceptance."
512
+ : "Coordinator: correct the cause and call plan_units with the remaining work and all requirements; budget is cumulative.")
513
+ : run?.phase === "awaiting-acceptance" && !operationComplete
514
+ ? `Requested operation is ${operationStatus}. Continue the actual operation or report its blocker; auxiliary checks and review disposition cannot complete it.`
515
+ : run?.phase === "awaiting-acceptance" ? (reviewAccepted
516
+ ? "Coordinator: review permits submission. Retain advisory review notes; do not repeat passed validation or review for evidence formatting. Operator performs final acceptance against the original requirements."
498
517
  : "Coordinator: address recorded findings or obtain the required independent review, then submit_mission. Operator compares all requirements with source/evidence before complete_mission.")
499
- : "Coordinator: continue the next useful unit within original requirements. Return only a completion candidate, user-only decision, or scope/budget extension." };
518
+ : "Coordinator: continue the next useful unit within original requirements. Return only a completion candidate, user-only decision, or scope/budget extension." };
500
519
  }
501
520
  /** Models forward a short capability, never recopy the host's evidence hashes and source packet. */
502
521
  export function missionReviewTask(mission) {
@@ -506,35 +525,12 @@ export function missionReviewTask(mission) {
506
525
  return { ...review.task, prompt: `${MISSION_REVIEW_REFERENCE}${JSON.stringify({ r: mission.root,
507
526
  m: mission.id, n: review.runID, h: digest(review.task.prompt) })}` };
508
527
  }
509
- /** Project grouped R-ID traces, including consecutive ranges, into the existing per-criterion mapping. */
510
- export function missionReviewTraces(mission, raw) {
511
- if (!Array.isArray(raw) || raw.length === 0 || !raw.every(item => typeof item === "string" && item.trim())) {
512
- throw new Error("mission-review-traces: supply concise implementation/test traces");
528
+ /** Optional implementation notes supplement host-observed source/checks; prose is not a coverage gate. */
529
+ export function missionReviewTraces(_mission, raw) {
530
+ if (raw === undefined)
531
+ return [];
532
+ if (!Array.isArray(raw) || !raw.every(item => typeof item === "string")) {
533
+ throw new Error("mission-review-traces: optional notes must be strings");
513
534
  }
514
- const grouped = raw.map(text => ({ text,
515
- ids: [...text.matchAll(/(?:^|[.;]\s+|[\r\n])\s*((?:R\d+\s*[-/,]?\s*)+):/gu)]
516
- .flatMap(match => [...match[1].matchAll(/R(\d+)(?:\s*-\s*R(\d+))?/gu)]
517
- .flatMap(([, first, last]) => {
518
- const firstID = `R${first}`;
519
- if (last === undefined)
520
- return [firstID];
521
- const lastID = `R${last}`;
522
- const start = mission.requirements.findIndex(item => item.id === firstID);
523
- const end = mission.requirements.findIndex(item => item.id === lastID);
524
- if (start < 0 || end < 0)
525
- return [firstID, lastID];
526
- if (start > end)
527
- throw new Error(`mission-review-traces: descending range ${firstID}-${lastID}`);
528
- return mission.requirements.slice(start, end + 1).map(item => item.id);
529
- })) }));
530
- if (grouped.every(item => item.ids.length === 0) && raw.length === mission.requirements.length)
531
- return raw;
532
- const unknown = grouped.flatMap(item => item.ids).filter(id => !mission.requirements.some(requirement => requirement.id === id));
533
- const missing = mission.requirements.filter(requirement => !grouped.some(item => item.ids.includes(requirement.id)));
534
- if (unknown.length || missing.length)
535
- throw new Error(`mission-review-traces: name the existing R IDs; ${[
536
- ...(missing.length ? [`missing ${missing.map(item => item.id).join(", ")}`] : []),
537
- ...(unknown.length ? [`unknown ${unknown.join(", ")}`] : []),
538
- ].join("; ")}`);
539
- return mission.requirements.map(requirement => grouped.filter(item => item.ids.includes(requirement.id)).map(item => item.text).join("\n"));
535
+ return raw.map(text => text.trim()).filter(Boolean);
540
536
  }
@@ -227,6 +227,22 @@ export interface OperatorState {
227
227
  decision: string | null;
228
228
  receipt: GoalTerminalReceipt | null;
229
229
  }
230
+ export interface OperatorProgress {
231
+ readonly profile: string;
232
+ readonly view: "progress";
233
+ readonly run_id: string;
234
+ readonly stage: OperatorPhase;
235
+ readonly decision: string | null;
236
+ readonly current_unit: {
237
+ readonly id: string;
238
+ readonly title: string;
239
+ readonly status: UnitState["status"];
240
+ } | null;
241
+ readonly completed_units: number;
242
+ readonly total_units: number;
243
+ readonly budget_remaining_units: number | null;
244
+ readonly next_action: string | null;
245
+ }
230
246
  /** A later mission may replace an unexecuted cancellation, but must retain settled work's acceptance. */
231
247
  export declare function cancelledMissionRetainsAcceptance(state: OperatorState): boolean;
232
248
  export declare function operatorGitPathAuthorized(path: string, scopes: readonly string[], platform?: NodeJS.Platform): boolean;
@@ -267,12 +283,12 @@ export declare class OperatorRuntime {
267
283
  prepareMission(root: string, raw: unknown, dispatcher?: {
268
284
  sessionID: string;
269
285
  callID: string;
270
- }, supersededRunID?: string, terminalChildren?: readonly string[], replaceRequirements?: boolean): Promise<OperatorState>;
286
+ }, supersededRunID?: string, terminalChildren?: readonly string[], replaceRequirements?: boolean, context?: Record<string, unknown>): Promise<OperatorState>;
271
287
  /** Stage a settled mission's replacement; rejected preparation never cancels the usable run. */
272
288
  replanMission(root: string, runID: string, raw: unknown, dispatcher?: {
273
289
  sessionID: string;
274
290
  callID: string;
275
- }): Promise<OperatorState>;
291
+ }, context?: Record<string, unknown>): Promise<OperatorState>;
276
292
  private prepareOnce;
277
293
  operatorTask(state: OperatorState): OperatorTask;
278
294
  /** Root-visible handle for the bounded operations delegate; its contract stays host-internal. */
@@ -335,6 +351,8 @@ export declare class OperatorRuntime {
335
351
  operatorRejected(root: string): Promise<OperatorState>;
336
352
  private operatorRejectedOnce;
337
353
  settled(result: SerialDispatchSettlement): Promise<void>;
354
+ /** Correct an estimated Mission scope without dispatching, restarting, or reserving another unit. */
355
+ expandMissionWriteScope(root: string, child: string, taskID: string, paths: readonly string[], activate: (manifest: import("./types.js").OperationManifest) => Promise<() => Promise<void>>): Promise<void>;
338
356
  private settledOnce;
339
357
  /** One ordinary same-scope Mission retry before any optional model Rescue is considered. */
340
358
  prepareMissionNormalRemediation(root: string, runID: string, unitID: string): Promise<OperatorState>;
@@ -371,10 +389,13 @@ export declare class OperatorRuntime {
371
389
  draftStatus(root: string): Promise<unknown | undefined>;
372
390
  completionGoalFingerprint(state: OperatorState): Promise<string>;
373
391
  verifyContinuityControls(state: OperatorState): Promise<void>;
392
+ /** Compact, read-only projection of the same durable run and decision as the complete packet. */
393
+ progress(state: OperatorState, budgetRemainingUnits: number | null, nextAction?: string | null): OperatorProgress;
374
394
  packet(state: OperatorState): unknown;
375
395
  /** Reconstruct the admitted child's instructions without trusting a lossy conversation summary. */
376
396
  workerContext(root: string, child: string): Promise<string | undefined>;
377
397
  continuationCheckpoint(root: string): Promise<string | undefined>;
398
+ private continuationNextAction;
378
399
  private controlReference;
379
400
  private verifyControls;
380
401
  /** Follow only the host's authorized parent-link repair; keep the durable control pin in sync. */