sortie-dogs 0.13.1 → 0.13.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. package/README.md +52 -0
  2. package/dist/asset-version.d.ts +1 -1
  3. package/dist/asset-version.js +1 -1
  4. package/dist/core/contract-limits.d.ts +6 -1
  5. package/dist/core/contract-limits.js +3 -1
  6. package/dist/core/goal-bound.d.ts +7 -0
  7. package/dist/core/goal-bound.js +7 -1
  8. package/dist/core/operator-mission.d.ts +72 -4
  9. package/dist/core/operator-mission.js +165 -29
  10. package/dist/core/operator-runtime.d.ts +30 -2
  11. package/dist/core/operator-runtime.js +200 -29
  12. package/dist/core/runtime-profile.js +3 -0
  13. package/dist/core/scope-lease-registry.d.ts +1 -0
  14. package/dist/core/scope-lease-registry.js +23 -1
  15. package/dist/core/validate-schema.js +0 -1
  16. package/dist/core/validation-budget.d.ts +5 -0
  17. package/dist/core/validation-budget.js +5 -2
  18. package/dist/plugin/gate.d.ts +5 -1
  19. package/dist/plugin/gate.js +119 -22
  20. package/dist/plugin/index.d.ts +17 -0
  21. package/dist/plugin/index.js +462 -77
  22. package/dist/plugin/mission-review.d.ts +103 -2
  23. package/dist/plugin/mission-review.js +227 -38
  24. package/dist/plugin/native-background.d.ts +40 -0
  25. package/dist/plugin/native-background.js +211 -0
  26. package/dist/plugin/native-contract-read.d.ts +9 -0
  27. package/dist/plugin/native-contract-read.js +90 -0
  28. package/dist/plugin/profiled.js +885 -199
  29. package/dist/plugin/protected-snapshot.d.ts +10 -0
  30. package/dist/plugin/protected-snapshot.js +132 -27
  31. package/dist/plugin/run-metrics.d.ts +1 -1
  32. package/dist/plugin/run-metrics.js +3 -2
  33. package/dist/plugin/runtime-bridge.d.ts +74 -3
  34. package/dist/plugin/v2.d.ts +40 -20
  35. package/dist/plugin/v2.js +347 -58
  36. package/dist/plugin/validation-scratch.d.ts +2 -0
  37. package/dist/plugin/validation-scratch.js +38 -8
  38. package/dist/runtime-assets-v010.js +16 -8
  39. package/dist/runtime-mission-assets.d.ts +5 -1
  40. package/dist/runtime-mission-assets.js +127 -115
  41. package/package.json +1 -1
package/README.md CHANGED
@@ -14,6 +14,9 @@ implementation, validation, review, and model routing.
14
14
  continuation, remediation, and restart.
15
15
  - **Adaptive execution**: small work stays small; additional agents and stronger
16
16
  models are used only when task shape or risk justifies them.
17
+ - **Clear Worker instructions**: give Luna a concise goal, explicit constraints
18
+ and completion criteria; keep orchestration bookkeeping in the harness.
19
+ See the [instruction design principles](docs/worker-instruction-design.md).
17
20
  - **Coexistence**: Sortie activates only when selected and preserves normal
18
21
  OpenCode agents, settings, and user-owned files.
19
22
  - **Cost, time, and proof**: the objective is a verified result at the lowest
@@ -78,6 +81,20 @@ internal children and must not be selected as task entry points.
78
81
  model routing. A new session alone does not reload an updated
79
82
  plugin process, so restart OpenCode after installation or upgrade.
80
83
 
84
+ ## v0.13.3 runtime updates
85
+
86
+ The current release integrates native background Mission execution, concise authoritative Worker
87
+ handoffs, and same-context Reviewer corrections from PRs #148/#149 and their Ubuntu remediation.
88
+ After finding defects, the original Reviewer can correct and explicitly self-recheck in the same
89
+ native session, retaining formal checks, current-source evidence, Git delivery and cumulative budget.
90
+ Author self-recheck is recorded as non-independent; a different Reviewer is conditional on a concrete
91
+ reachable residual Major risk. Unresolved Major or Medium findings still block acceptance.
92
+
93
+ The installed native correction fixture succeeded, but the original Anko measurements remain
94
+ unaccepted, including the latest 25-minute run. This release does not claim general speedup,
95
+ original-task completion or a new SWE-bench score. See the [release notes](docs/release-v0.13.3.md)
96
+ and [retained implementation and measurement history](docs/anko-pr149-ubuntu-handoff.md).
97
+
81
98
  ## v0.12.0 workflow
82
99
 
83
100
  v0.12.0 keeps Operator → Coordinator → Worker, with independent review:
@@ -111,6 +128,8 @@ Official SWE-bench Lite `dev` results on the same 23 public instances:
111
128
  | v0.12.16 (`9b05a34` release; `b1a6c0e` runner) | 4 / 23 (17.4%) | 5 | Fresh 23-task run; eight slots; 40-minute timeout | [Campaign](docs/swebench-v01216-dev23-2026-09-27.md) |
112
129
  | v0.12.19 (`24f5386` release; matched rerun) | 8 / 23 (34.8%) | 0 | Eight slots; effective $2/instance; 40-minute timeout; one inference timeout | [Official result and provenance](docs/benchmarks/swebench-v01220-operation-observability-2026-09-28.md) |
113
130
  | v0.12.20 (`628eb81` release) | 7 / 23 (30.4%) | 0 | Eight slots; effective $2/instance; 40-minute timeout | [Official result and caveat](docs/benchmarks/swebench-v01220-operation-observability-2026-09-28.md) |
131
+ | v0.12.25 (`49eb1e4` release) | 7 / 23 (30.4%) | 0 | Eight slots; $2/instance; $30 total cap; 20-minute progress check / 40-minute hard maximum | [Comparison baseline](#v0131-dev23-2026-09-30) |
132
+ | v0.13.1 (`d19e8be` release; 2026-09-30) | **8 / 23 (34.8%)** | 0 | Eight slots; $2/instance; $46 total cap; 20-minute progress check / 40-minute hard maximum; GPT-6.1 Sol + Luna Fast | [Run summary](#v0131-dev23-2026-09-30) |
114
133
 
115
134
  Every row has 23 submitted official predictions; an empty patch counts against
116
135
  the score, not as a missing evaluation. The v0.10.14 report does not separately
@@ -132,6 +151,39 @@ single run-to-run difference does not establish causation.
132
151
 
133
152
  Historical qualification references remain in [benchmark reference](docs/benchmark-reference.md).
134
153
 
154
+ #### v0.13.1 dev23 (2026-09-30)
155
+
156
+ One fresh pass@1 run and one official SWE-bench harness evaluation resolved
157
+ **8/23**, versus **7/23** for v0.12.25. The new resolution was
158
+ `pylint-dev__astroid-1333`; all seven previously resolved IDs were retained.
159
+ Resolved by repository: marshmallow **2/2**, pvlib **0/5**, pydicom **2/5**,
160
+ astroid **3/5**, pyvista **0/1**, sqlfluff **1/5**.
161
+
162
+ - The scored row is the user-requested fresh run after a host restart. The
163
+ interrupted initial run is excluded from this score; the fresh run made one
164
+ attempt per instance with no inference retry.
165
+ - The dataset revision (`6ec7bb89b9342f664a54a6e0a6ea6501d3437cc2`), public rows,
166
+ and all 23 official evaluation image IDs match the v0.12.25 run. Both used
167
+ `official-image-testbed` and sequential official scoring.
168
+ - Operator/Coordinator/Reviewer/Advisor defaults changed to
169
+ `openai/gpt-6.1-sol#xhigh`. Actual task Workers remained
170
+ `openai/gpt-6-luna-fast#max`, observed across all 23 instances. The harness and
171
+ total budget also changed, so the extra resolution cannot be attributed to
172
+ the model change alone.
173
+ - Inference ended with 20 normal completions, two timeouts
174
+ (`pvlib__pvlib-python-1154`, `sqlfluff__sqlfluff-1763`) and one agent failure
175
+ (`pvlib__pvlib-python-1854`). All patches, including stopped attempts, were
176
+ officially scored: 23 completed evaluations, zero empty patches and zero
177
+ official evaluation errors or infrastructure failures.
178
+ - Known estimated inference cost: **$17.73**; separate unknown-usage hold:
179
+ **$1.98**, not counted as known expense. Inference wall time was about
180
+ **81 minutes**, followed by **7.2 minutes** of official scoring.
181
+ - Fixed release commit: `d19e8be0d21180cc23ad2ae4b853d846a18e77bc`;
182
+ package SHA-256: `99300ceec0c3eee4fa1f984fed50d15d58b2ff4455df0041850ecd63514b0a51`.
183
+ OpenCode **2.0.20**, official harness **5.0.2**. Local evidence is retained in
184
+ `_testenv/swebench-v0131-dev23-20260930-r2/result-summary.json`; generated
185
+ predictions, databases and raw logs are not committed.
186
+
135
187
  ## Mission tools
136
188
 
137
189
  1. `start_mission`: Operator supplies concise requirements; the host saves original messages and returns a Coordinator task.
@@ -3,5 +3,5 @@
3
3
  * installed project marker without importing every asset body.
4
4
  */
5
5
  export declare const RUNTIME_ASSET_VERSION = "0.3.89-completion-proof-v1";
6
- export declare const V010_RUNTIME_ASSET_VERSION = "0.13.1-quality-first-review-v1";
6
+ export declare const V010_RUNTIME_ASSET_VERSION = "0.13.3-reviewer-context-v1";
7
7
  export type RuntimeAssetVersion = typeof RUNTIME_ASSET_VERSION | typeof V010_RUNTIME_ASSET_VERSION;
@@ -3,4 +3,4 @@
3
3
  * installed project marker without importing every asset body.
4
4
  */
5
5
  export const RUNTIME_ASSET_VERSION = "0.3.89-completion-proof-v1";
6
- export const V010_RUNTIME_ASSET_VERSION = "0.13.1-quality-first-review-v1";
6
+ export const V010_RUNTIME_ASSET_VERSION = "0.13.3-reviewer-context-v1";
@@ -1,8 +1,13 @@
1
1
  /** Common text bounds shared by handoff, manifest and goal evidence validation. */
2
2
  export declare const CONTRACT_TEXT_LIMITS: Readonly<{
3
3
  title: 160;
4
- objective: 2000;
4
+ objective: 32768;
5
5
  statement: 1000;
6
6
  command: 8192;
7
7
  path: 512;
8
8
  }>;
9
+ /** New Mission task generation only; persisted/legacy contracts retain their original bounds. */
10
+ export declare const MISSION_OBJECTIVE_LIMITS: Readonly<{
11
+ target: 2000;
12
+ maximum: 3000;
13
+ }>;
@@ -1,2 +1,4 @@
1
1
  /** Common text bounds shared by handoff, manifest and goal evidence validation. */
2
- export const CONTRACT_TEXT_LIMITS = Object.freeze({ title: 160, objective: 2000, statement: 1000, command: 8192, path: 512 });
2
+ export const CONTRACT_TEXT_LIMITS = Object.freeze({ title: 160, objective: 32768, statement: 1000, command: 8192, path: 512 });
3
+ /** New Mission task generation only; persisted/legacy contracts retain their original bounds. */
4
+ export const MISSION_OBJECTIVE_LIMITS = Object.freeze({ target: 2000, maximum: 3000 });
@@ -39,6 +39,13 @@ export interface GoalEvidence {
39
39
  readonly candidate_paths: readonly string[];
40
40
  /** Missing on legacy evidence: retain its original all-paths snapshot recipe. */
41
41
  readonly source_policy?: "project-files-v1" | "declared-paths-v1";
42
+ /** Fixed when validation starts. Full manifest_hash still identifies the historical execution contract. */
43
+ readonly freshness?: {
44
+ readonly contract_hash: string;
45
+ readonly scratch_paths: readonly string[];
46
+ readonly protected_paths: readonly string[];
47
+ readonly environment: Readonly<Record<string, string | null>>;
48
+ };
42
49
  };
43
50
  readonly execution: {
44
51
  readonly command: readonly string[];
@@ -64,7 +64,13 @@ export function validGoalEvidence(value, state) {
64
64
  protectedBinding.source_paths.every(text) && Array.isArray(protectedBinding.candidate_paths) &&
65
65
  protectedBinding.candidate_paths.every(text) &&
66
66
  (protectedBinding.source_policy === undefined || protectedBinding.source_policy === "project-files-v1" ||
67
- protectedBinding.source_policy === "declared-paths-v1");
67
+ protectedBinding.source_policy === "declared-paths-v1") &&
68
+ (protectedBinding.freshness === undefined || (protectedBinding.freshness !== null && typeof protectedBinding.freshness === "object" &&
69
+ HASH.test(protectedBinding.freshness.contract_hash) &&
70
+ Array.isArray(protectedBinding.freshness.scratch_paths) && protectedBinding.freshness.scratch_paths.every(text) &&
71
+ Array.isArray(protectedBinding.freshness.protected_paths) && protectedBinding.freshness.protected_paths.every(text) &&
72
+ protectedBinding.freshness.environment !== null && typeof protectedBinding.freshness.environment === "object" &&
73
+ !Array.isArray(protectedBinding.freshness.environment) && Object.values(protectedBinding.freshness.environment).every(value => value === null || typeof value === "string")));
68
74
  const matches = criteria.length > 0 && criteria.length === value.measurement.criterion_ids.length &&
69
75
  criteria.every((criterion) => criterion.target === value.measurement.target &&
70
76
  criterion.entrypoint === value.measurement.entrypoint && criterion.workload === value.measurement.workload &&
@@ -1,4 +1,4 @@
1
- import { type OperatorPlan, type OperatorState, type OperatorTask } from "./operator-runtime.js";
1
+ import { type OperatorPlan, type OperatorState, type OperatorTask, type OperatorRuntime } from "./operator-runtime.js";
2
2
  import { type RuntimeProfile } from "./runtime-profile.js";
3
3
  export declare const MISSION_REFERENCE = "SORTIE_MISSION_REF ";
4
4
  export declare const MISSION_REVIEW_REFERENCE = "SORTIE_MISSION_REVIEW_REF ";
@@ -19,6 +19,7 @@ export interface MissionEvidenceExcerpt {
19
19
  export interface MissionReviewScope {
20
20
  read: string[];
21
21
  write: string[];
22
+ validationBindings?: NonNullable<import("./goal-bound.js").GoalEvidence["protected_binding"]>[];
22
23
  }
23
24
  export interface MissionConsultation {
24
25
  id: string;
@@ -43,11 +44,13 @@ export interface MissionAttempt {
43
44
  predecessorAttemptID?: string | null;
44
45
  /** Fingerprint of the settled scoped candidate against which a later Rescue is proposed. */
45
46
  candidateID?: string;
46
- kind: "implementation" | "normal_remediation" | "astra_rescue";
47
+ kind: "implementation" | "normal_remediation" | "astra_rescue" | "reviewer_correction";
47
48
  status: "pending" | "dispatched" | "succeeded" | "failed" | "cancelled" | "unconfirmed";
48
49
  callID?: string;
49
50
  childSessionID?: string;
51
+ dispatchFingerprint?: string;
50
52
  nativeOutcome?: "completed" | "failed" | "unknown";
53
+ terminal?: import("../plugin/runtime-bridge.js").MissionWorkerTerminalRecord;
51
54
  observedModel?: string;
52
55
  observedVariant?: string;
53
56
  failure?: {
@@ -86,6 +89,35 @@ export interface MissionExecution {
86
89
  result?: Record<string, unknown>;
87
90
  }[];
88
91
  }
92
+ export interface MissionLaunchConditions {
93
+ entrypoint?: string;
94
+ inputs?: string[];
95
+ timeout_seconds?: number;
96
+ cost_limit_usd?: number;
97
+ benchmark_attempts?: number;
98
+ grading?: "none" | "official";
99
+ source: string;
100
+ applies_to: string;
101
+ recordedAt?: string;
102
+ }
103
+ /** Native terminal self-recheck by the correction author, explicitly not independent approval. */
104
+ export interface MissionSelfRecheck {
105
+ runID: string;
106
+ source: string;
107
+ /** Candidate source/check identity, excluding optional excerpt presentation. */
108
+ candidateSource?: string;
109
+ author: string;
110
+ callID: string;
111
+ promptID: string;
112
+ messageID: string;
113
+ nativeOutcome: "completed";
114
+ result: string;
115
+ unresolvedFindings: string[];
116
+ residualMajor?: {
117
+ reachable_path: string;
118
+ consequence: string;
119
+ };
120
+ }
89
121
  export interface OperatorMission {
90
122
  version: "0.12";
91
123
  id: string;
@@ -93,13 +125,23 @@ export interface OperatorMission {
93
125
  requests: MissionRequest[];
94
126
  /** Prior public conversation context, not additional immutable requirements. */
95
127
  context?: MissionContext[];
128
+ /** Explicit continue deliveries, keyed by the original real turn. */
129
+ steering?: {
130
+ requestID: string;
131
+ child: string;
132
+ status: "pending" | "queued";
133
+ }[];
96
134
  kind?: "implementation" | "operation";
97
135
  /** Native shell observations of the requested operation, separate from auxiliary checks. */
98
136
  execution?: MissionExecution;
137
+ /** Fixed benchmark conditions and their provenance, separate from internal Worker counters. */
138
+ launchConditions?: MissionLaunchConditions[];
99
139
  requirements: {
100
140
  id: string;
101
141
  text: string;
102
142
  }[];
143
+ /** Explicit user path prohibitions, unlike the Coordinator's estimated write list. */
144
+ prohibitedWrite?: string[];
103
145
  /** The current user intentionally replaced the predecessor's requirements. */
104
146
  requirementsReplaced?: boolean;
105
147
  phase: "open" | "running" | "submitted" | "completed" | "cancelled";
@@ -134,17 +176,36 @@ export interface OperatorMission {
134
176
  runID: string;
135
177
  risk: string[];
136
178
  source: string;
179
+ candidateSource?: string;
137
180
  task: OperatorTask | null;
181
+ callID?: string;
138
182
  evidence?: MissionEvidenceExcerpt[];
139
183
  requestFingerprint?: string;
140
- verdict: "pending" | "PASS" | "findings" | "evidence-gaps" | "skipped-low-risk";
184
+ verdict: "pending" | "PASS" | "findings" | "evidence-gaps" | "skipped-low-risk" | "self-rechecked";
141
185
  result?: string;
142
186
  child?: string;
187
+ mode?: "independent" | "self-recheck";
188
+ admittedAt?: number;
189
+ promptID?: string;
190
+ selfRecheck?: MissionSelfRecheck;
143
191
  /** Completed, independent initial review for this mission, not merely an inherited child ID. */
144
192
  initialPrompt?: string;
145
193
  /** Observed evidence-only reviews; reporting only, never an acceptance threshold. */
146
194
  evidenceGapReviews?: number;
147
195
  };
196
+ /** Retained independently of later review generations; authors never become independent reviewers. */
197
+ corrections?: {
198
+ author: string;
199
+ reviewIdentity: string;
200
+ priorRunID: string;
201
+ runID: string;
202
+ priorSource: string;
203
+ findings: string;
204
+ initialPrompt: string;
205
+ baseline?: string;
206
+ status: "prepared" | "running" | "ready" | "failed" | "cancelled";
207
+ selfRecheck?: MissionSelfRecheck;
208
+ }[];
148
209
  }
149
210
  export declare const MISSION_CONSULTATION_LIMIT = 32;
150
211
  /** Review coverage survives a narrower replan; it is not a Worker write grant. */
@@ -153,6 +214,8 @@ export declare function missionReviewScope(previous: MissionReviewScope | undefi
153
214
  export declare function missionReviewVerdict(text: string): "PASS" | "evidence-gaps" | "findings";
154
215
  /** Whether the recorded review permits submission and acceptance of the current candidate. */
155
216
  export declare function missionReviewAccepted(review: NonNullable<OperatorMission["review"]>): boolean;
217
+ /** A short native report, not a tag/hash-based second-review policy or an approval checklist. */
218
+ export declare function missionSelfRecheckReport(text: string, source: string, hostBound?: boolean): Pick<MissionSelfRecheck, "unresolvedFindings" | "residualMajor"> | undefined;
156
219
  export declare function missionExecutionStatus(mission: OperatorMission): "not-required" | "not-started" | "running" | "execution-failed" | "executed";
157
220
  /** Observe the command's own terminal summary, never Worker/Reviewer prose. Scores are result data,
158
221
  * not process success. Unstructured commands retain their native exit semantics. */
@@ -169,6 +232,7 @@ export declare class OperatorMissionRuntime {
169
232
  private readonly writes;
170
233
  constructor(projectRoot: string, profile: RuntimeProfile);
171
234
  private file;
235
+ correctionReference(root: string, reviewIdentity: string): string;
172
236
  private load;
173
237
  /** Recover non-replacement continuations only; an explicit replacement must link the current cancelled run. */
174
238
  private loadMission;
@@ -179,6 +243,7 @@ export declare class OperatorMissionRuntime {
179
243
  read(root: string): Promise<OperatorMission | undefined>;
180
244
  required(root: string): Promise<OperatorMission>;
181
245
  capture(root: string, request: MissionRequest): Promise<void>;
246
+ recordLaunchConditions(root: string, raw: unknown): Promise<OperatorMission>;
182
247
  start(root: string, requirements: unknown, replaceRequirements?: boolean, options?: {
183
248
  kind?: OperatorMission["kind"];
184
249
  context?: MissionContext[];
@@ -209,7 +274,10 @@ export declare class OperatorMissionRuntime {
209
274
  brief(state: OperatorMission): string;
210
275
  }
211
276
  /** The model supplies only useful unit facts; IDs, proof projection and control documents are generated here. */
212
- export declare function missionPlan(mission: OperatorMission, raw: unknown): OperatorPlan;
277
+ export declare function missionPlan(mission: OperatorMission, raw: unknown, projectRoot?: string): OperatorPlan;
278
+ /** Compact provenance, not a new acceptance verdict or a substitute for original-request comparison. */
279
+ export declare function missionAcceptanceSummary(mission: OperatorMission, run: OperatorState | undefined, operators: OperatorRuntime, observe?: (validation: readonly string[], child: string | null, notBefore?: number) => Promise<unknown>): Promise<Record<string, unknown>>;
280
+ export declare function missionReviewIndependent(mission: OperatorMission, child: string | undefined): boolean;
213
281
  export declare function missionPacket(mission: OperatorMission, run?: OperatorState): Record<string, unknown>;
214
282
  /** Models forward a short capability, never recopy the host's evidence hashes and source packet. */
215
283
  export declare function missionReviewTask(mission: OperatorMission): OperatorTask;