sortie-dogs 0.13.2 → 0.13.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +17 -0
- package/dist/asset-version.d.ts +1 -1
- package/dist/asset-version.js +1 -1
- package/dist/core/contract-limits.d.ts +5 -0
- package/dist/core/contract-limits.js +2 -0
- package/dist/core/operator-mission.d.ts +53 -4
- package/dist/core/operator-mission.js +141 -29
- package/dist/core/operator-runtime.d.ts +26 -0
- package/dist/core/operator-runtime.js +136 -14
- package/dist/core/runtime-profile.js +3 -0
- package/dist/core/validate-schema.js +0 -1
- package/dist/core/validation-budget.d.ts +5 -0
- package/dist/core/validation-budget.js +5 -2
- package/dist/plugin/gate.d.ts +5 -1
- package/dist/plugin/gate.js +119 -22
- package/dist/plugin/index.d.ts +17 -0
- package/dist/plugin/index.js +260 -50
- package/dist/plugin/mission-review.d.ts +103 -2
- package/dist/plugin/mission-review.js +208 -36
- package/dist/plugin/native-background.d.ts +40 -0
- package/dist/plugin/native-background.js +211 -0
- package/dist/plugin/native-contract-read.d.ts +9 -0
- package/dist/plugin/native-contract-read.js +90 -0
- package/dist/plugin/profiled.js +640 -147
- package/dist/plugin/protected-snapshot.d.ts +4 -0
- package/dist/plugin/protected-snapshot.js +38 -8
- package/dist/plugin/run-metrics.d.ts +1 -1
- package/dist/plugin/run-metrics.js +3 -2
- package/dist/plugin/runtime-bridge.d.ts +50 -2
- package/dist/plugin/v2.d.ts +40 -20
- package/dist/plugin/v2.js +338 -55
- package/dist/runtime-assets-v010.js +14 -6
- package/dist/runtime-mission-assets.d.ts +4 -2
- package/dist/runtime-mission-assets.js +119 -116
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -14,6 +14,9 @@ implementation, validation, review, and model routing.
|
|
|
14
14
|
continuation, remediation, and restart.
|
|
15
15
|
- **Adaptive execution**: small work stays small; additional agents and stronger
|
|
16
16
|
models are used only when task shape or risk justifies them.
|
|
17
|
+
- **Clear Worker instructions**: give Luna a concise goal, explicit constraints
|
|
18
|
+
and completion criteria; keep orchestration bookkeeping in the harness.
|
|
19
|
+
See the [instruction design principles](docs/worker-instruction-design.md).
|
|
17
20
|
- **Coexistence**: Sortie activates only when selected and preserves normal
|
|
18
21
|
OpenCode agents, settings, and user-owned files.
|
|
19
22
|
- **Cost, time, and proof**: the objective is a verified result at the lowest
|
|
@@ -78,6 +81,20 @@ internal children and must not be selected as task entry points.
|
|
|
78
81
|
model routing. A new session alone does not reload an updated
|
|
79
82
|
plugin process, so restart OpenCode after installation or upgrade.
|
|
80
83
|
|
|
84
|
+
## v0.13.3 runtime updates
|
|
85
|
+
|
|
86
|
+
The current release integrates native background Mission execution, concise authoritative Worker
|
|
87
|
+
handoffs, and same-context Reviewer corrections from PRs #148/#149 and their Ubuntu remediation.
|
|
88
|
+
After finding defects, the original Reviewer can correct and explicitly self-recheck in the same
|
|
89
|
+
native session, retaining formal checks, current-source evidence, Git delivery and cumulative budget.
|
|
90
|
+
Author self-recheck is recorded as non-independent; a different Reviewer is conditional on a concrete
|
|
91
|
+
reachable residual Major risk. Unresolved Major or Medium findings still block acceptance.
|
|
92
|
+
|
|
93
|
+
The installed native correction fixture succeeded, but the original Anko measurements remain
|
|
94
|
+
unaccepted, including the latest 25-minute run. This release does not claim general speedup,
|
|
95
|
+
original-task completion or a new SWE-bench score. See the [release notes](docs/release-v0.13.3.md)
|
|
96
|
+
and [retained implementation and measurement history](docs/anko-pr149-ubuntu-handoff.md).
|
|
97
|
+
|
|
81
98
|
## v0.12.0 workflow
|
|
82
99
|
|
|
83
100
|
v0.12.0 keeps Operator → Coordinator → Worker, with independent review:
|
package/dist/asset-version.d.ts
CHANGED
|
@@ -3,5 +3,5 @@
|
|
|
3
3
|
* installed project marker without importing every asset body.
|
|
4
4
|
*/
|
|
5
5
|
export declare const RUNTIME_ASSET_VERSION = "0.3.89-completion-proof-v1";
|
|
6
|
-
export declare const V010_RUNTIME_ASSET_VERSION = "0.13.
|
|
6
|
+
export declare const V010_RUNTIME_ASSET_VERSION = "0.13.3-reviewer-context-v1";
|
|
7
7
|
export type RuntimeAssetVersion = typeof RUNTIME_ASSET_VERSION | typeof V010_RUNTIME_ASSET_VERSION;
|
package/dist/asset-version.js
CHANGED
|
@@ -3,4 +3,4 @@
|
|
|
3
3
|
* installed project marker without importing every asset body.
|
|
4
4
|
*/
|
|
5
5
|
export const RUNTIME_ASSET_VERSION = "0.3.89-completion-proof-v1";
|
|
6
|
-
export const V010_RUNTIME_ASSET_VERSION = "0.13.
|
|
6
|
+
export const V010_RUNTIME_ASSET_VERSION = "0.13.3-reviewer-context-v1";
|
|
@@ -6,3 +6,8 @@ export declare const CONTRACT_TEXT_LIMITS: Readonly<{
|
|
|
6
6
|
command: 8192;
|
|
7
7
|
path: 512;
|
|
8
8
|
}>;
|
|
9
|
+
/** New Mission task generation only; persisted/legacy contracts retain their original bounds. */
|
|
10
|
+
export declare const MISSION_OBJECTIVE_LIMITS: Readonly<{
|
|
11
|
+
target: 2000;
|
|
12
|
+
maximum: 3000;
|
|
13
|
+
}>;
|
|
@@ -1,2 +1,4 @@
|
|
|
1
1
|
/** Common text bounds shared by handoff, manifest and goal evidence validation. */
|
|
2
2
|
export const CONTRACT_TEXT_LIMITS = Object.freeze({ title: 160, objective: 32768, statement: 1000, command: 8192, path: 512 });
|
|
3
|
+
/** New Mission task generation only; persisted/legacy contracts retain their original bounds. */
|
|
4
|
+
export const MISSION_OBJECTIVE_LIMITS = Object.freeze({ target: 2000, maximum: 3000 });
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { type OperatorPlan, type OperatorState, type OperatorTask } from "./operator-runtime.js";
|
|
1
|
+
import { type OperatorPlan, type OperatorState, type OperatorTask, type OperatorRuntime } from "./operator-runtime.js";
|
|
2
2
|
import { type RuntimeProfile } from "./runtime-profile.js";
|
|
3
3
|
export declare const MISSION_REFERENCE = "SORTIE_MISSION_REF ";
|
|
4
4
|
export declare const MISSION_REVIEW_REFERENCE = "SORTIE_MISSION_REVIEW_REF ";
|
|
@@ -44,7 +44,7 @@ export interface MissionAttempt {
|
|
|
44
44
|
predecessorAttemptID?: string | null;
|
|
45
45
|
/** Fingerprint of the settled scoped candidate against which a later Rescue is proposed. */
|
|
46
46
|
candidateID?: string;
|
|
47
|
-
kind: "implementation" | "normal_remediation" | "astra_rescue";
|
|
47
|
+
kind: "implementation" | "normal_remediation" | "astra_rescue" | "reviewer_correction";
|
|
48
48
|
status: "pending" | "dispatched" | "succeeded" | "failed" | "cancelled" | "unconfirmed";
|
|
49
49
|
callID?: string;
|
|
50
50
|
childSessionID?: string;
|
|
@@ -100,6 +100,24 @@ export interface MissionLaunchConditions {
|
|
|
100
100
|
applies_to: string;
|
|
101
101
|
recordedAt?: string;
|
|
102
102
|
}
|
|
103
|
+
/** Native terminal self-recheck by the correction author, explicitly not independent approval. */
|
|
104
|
+
export interface MissionSelfRecheck {
|
|
105
|
+
runID: string;
|
|
106
|
+
source: string;
|
|
107
|
+
/** Candidate source/check identity, excluding optional excerpt presentation. */
|
|
108
|
+
candidateSource?: string;
|
|
109
|
+
author: string;
|
|
110
|
+
callID: string;
|
|
111
|
+
promptID: string;
|
|
112
|
+
messageID: string;
|
|
113
|
+
nativeOutcome: "completed";
|
|
114
|
+
result: string;
|
|
115
|
+
unresolvedFindings: string[];
|
|
116
|
+
residualMajor?: {
|
|
117
|
+
reachable_path: string;
|
|
118
|
+
consequence: string;
|
|
119
|
+
};
|
|
120
|
+
}
|
|
103
121
|
export interface OperatorMission {
|
|
104
122
|
version: "0.12";
|
|
105
123
|
id: string;
|
|
@@ -107,6 +125,12 @@ export interface OperatorMission {
|
|
|
107
125
|
requests: MissionRequest[];
|
|
108
126
|
/** Prior public conversation context, not additional immutable requirements. */
|
|
109
127
|
context?: MissionContext[];
|
|
128
|
+
/** Explicit continue deliveries, keyed by the original real turn. */
|
|
129
|
+
steering?: {
|
|
130
|
+
requestID: string;
|
|
131
|
+
child: string;
|
|
132
|
+
status: "pending" | "queued";
|
|
133
|
+
}[];
|
|
110
134
|
kind?: "implementation" | "operation";
|
|
111
135
|
/** Native shell observations of the requested operation, separate from auxiliary checks. */
|
|
112
136
|
execution?: MissionExecution;
|
|
@@ -152,17 +176,36 @@ export interface OperatorMission {
|
|
|
152
176
|
runID: string;
|
|
153
177
|
risk: string[];
|
|
154
178
|
source: string;
|
|
179
|
+
candidateSource?: string;
|
|
155
180
|
task: OperatorTask | null;
|
|
181
|
+
callID?: string;
|
|
156
182
|
evidence?: MissionEvidenceExcerpt[];
|
|
157
183
|
requestFingerprint?: string;
|
|
158
|
-
verdict: "pending" | "PASS" | "findings" | "evidence-gaps" | "skipped-low-risk";
|
|
184
|
+
verdict: "pending" | "PASS" | "findings" | "evidence-gaps" | "skipped-low-risk" | "self-rechecked";
|
|
159
185
|
result?: string;
|
|
160
186
|
child?: string;
|
|
187
|
+
mode?: "independent" | "self-recheck";
|
|
188
|
+
admittedAt?: number;
|
|
189
|
+
promptID?: string;
|
|
190
|
+
selfRecheck?: MissionSelfRecheck;
|
|
161
191
|
/** Completed, independent initial review for this mission, not merely an inherited child ID. */
|
|
162
192
|
initialPrompt?: string;
|
|
163
193
|
/** Observed evidence-only reviews; reporting only, never an acceptance threshold. */
|
|
164
194
|
evidenceGapReviews?: number;
|
|
165
195
|
};
|
|
196
|
+
/** Retained independently of later review generations; authors never become independent reviewers. */
|
|
197
|
+
corrections?: {
|
|
198
|
+
author: string;
|
|
199
|
+
reviewIdentity: string;
|
|
200
|
+
priorRunID: string;
|
|
201
|
+
runID: string;
|
|
202
|
+
priorSource: string;
|
|
203
|
+
findings: string;
|
|
204
|
+
initialPrompt: string;
|
|
205
|
+
baseline?: string;
|
|
206
|
+
status: "prepared" | "running" | "ready" | "failed" | "cancelled";
|
|
207
|
+
selfRecheck?: MissionSelfRecheck;
|
|
208
|
+
}[];
|
|
166
209
|
}
|
|
167
210
|
export declare const MISSION_CONSULTATION_LIMIT = 32;
|
|
168
211
|
/** Review coverage survives a narrower replan; it is not a Worker write grant. */
|
|
@@ -171,6 +214,8 @@ export declare function missionReviewScope(previous: MissionReviewScope | undefi
|
|
|
171
214
|
export declare function missionReviewVerdict(text: string): "PASS" | "evidence-gaps" | "findings";
|
|
172
215
|
/** Whether the recorded review permits submission and acceptance of the current candidate. */
|
|
173
216
|
export declare function missionReviewAccepted(review: NonNullable<OperatorMission["review"]>): boolean;
|
|
217
|
+
/** A short native report, not a tag/hash-based second-review policy or an approval checklist. */
|
|
218
|
+
export declare function missionSelfRecheckReport(text: string, source: string, hostBound?: boolean): Pick<MissionSelfRecheck, "unresolvedFindings" | "residualMajor"> | undefined;
|
|
174
219
|
export declare function missionExecutionStatus(mission: OperatorMission): "not-required" | "not-started" | "running" | "execution-failed" | "executed";
|
|
175
220
|
/** Observe the command's own terminal summary, never Worker/Reviewer prose. Scores are result data,
|
|
176
221
|
* not process success. Unstructured commands retain their native exit semantics. */
|
|
@@ -187,6 +232,7 @@ export declare class OperatorMissionRuntime {
|
|
|
187
232
|
private readonly writes;
|
|
188
233
|
constructor(projectRoot: string, profile: RuntimeProfile);
|
|
189
234
|
private file;
|
|
235
|
+
correctionReference(root: string, reviewIdentity: string): string;
|
|
190
236
|
private load;
|
|
191
237
|
/** Recover non-replacement continuations only; an explicit replacement must link the current cancelled run. */
|
|
192
238
|
private loadMission;
|
|
@@ -228,7 +274,10 @@ export declare class OperatorMissionRuntime {
|
|
|
228
274
|
brief(state: OperatorMission): string;
|
|
229
275
|
}
|
|
230
276
|
/** The model supplies only useful unit facts; IDs, proof projection and control documents are generated here. */
|
|
231
|
-
export declare function missionPlan(mission: OperatorMission, raw: unknown): OperatorPlan;
|
|
277
|
+
export declare function missionPlan(mission: OperatorMission, raw: unknown, projectRoot?: string): OperatorPlan;
|
|
278
|
+
/** Compact provenance, not a new acceptance verdict or a substitute for original-request comparison. */
|
|
279
|
+
export declare function missionAcceptanceSummary(mission: OperatorMission, run: OperatorState | undefined, operators: OperatorRuntime, observe?: (validation: readonly string[], child: string | null, notBefore?: number) => Promise<unknown>): Promise<Record<string, unknown>>;
|
|
280
|
+
export declare function missionReviewIndependent(mission: OperatorMission, child: string | undefined): boolean;
|
|
232
281
|
export declare function missionPacket(mission: OperatorMission, run?: OperatorState): Record<string, unknown>;
|
|
233
282
|
/** Models forward a short capability, never recopy the host's evidence hashes and source packet. */
|
|
234
283
|
export declare function missionReviewTask(mission: OperatorMission): OperatorTask;
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { createHash, randomUUID } from "node:crypto";
|
|
2
2
|
import { mkdir, readFile, readdir, rename, rm, writeFile } from "node:fs/promises";
|
|
3
|
-
import { join } from "node:path";
|
|
3
|
+
import { join, resolve } from "node:path";
|
|
4
4
|
import { normalizeExecutionScope } from "./path.js";
|
|
5
5
|
import { parseOperatorPlan } from "./operator-runtime.js";
|
|
6
6
|
import { profileAgent } from "./runtime-profile.js";
|
|
@@ -34,9 +34,39 @@ export function missionReviewVerdict(text) {
|
|
|
34
34
|
}
|
|
35
35
|
/** Whether the recorded review permits submission and acceptance of the current candidate. */
|
|
36
36
|
export function missionReviewAccepted(review) {
|
|
37
|
+
if (review.verdict === "self-rechecked") {
|
|
38
|
+
const checked = review.selfRecheck;
|
|
39
|
+
return review.mode === "self-recheck" && !!checked && checked.runID === review.runID &&
|
|
40
|
+
checked.source === review.source && checked.author === review.child && checked.callID === review.callID &&
|
|
41
|
+
checked.promptID === review.promptID && !!checked.messageID && checked.nativeOutcome === "completed" &&
|
|
42
|
+
checked.unresolvedFindings.length === 0 && !checked.residualMajor;
|
|
43
|
+
}
|
|
44
|
+
if (review.mode === "self-recheck")
|
|
45
|
+
return false;
|
|
37
46
|
return review.verdict === "PASS" || review.verdict === "skipped-low-risk" ||
|
|
38
47
|
review.verdict === "evidence-gaps";
|
|
39
48
|
}
|
|
49
|
+
/** A short native report, not a tag/hash-based second-review policy or an approval checklist. */
|
|
50
|
+
export function missionSelfRecheckReport(text, source, hostBound = false) {
|
|
51
|
+
if (!/^\s*SELF_RECHECKED(?:\s|$)/u.test(text))
|
|
52
|
+
return undefined;
|
|
53
|
+
const line = /^self_recheck: (.+)$/mu.exec(text)?.[1];
|
|
54
|
+
try {
|
|
55
|
+
const report = JSON.parse(line ?? "");
|
|
56
|
+
if (!record(report) || (report.candidate !== source && !(hostBound && report.candidate === "current-validated")) || !Array.isArray(report.unresolved_findings) ||
|
|
57
|
+
!report.unresolved_findings.every(item => typeof item === "string" && item.trim()) ||
|
|
58
|
+
!(report.residual_major === null || record(report.residual_major) &&
|
|
59
|
+
typeof report.residual_major.reachable_path === "string" && report.residual_major.reachable_path.trim() &&
|
|
60
|
+
typeof report.residual_major.consequence === "string" && report.residual_major.consequence.trim()))
|
|
61
|
+
return undefined;
|
|
62
|
+
return { unresolvedFindings: report.unresolved_findings,
|
|
63
|
+
...(record(report.residual_major) ? { residualMajor: { reachable_path: report.residual_major.reachable_path,
|
|
64
|
+
consequence: report.residual_major.consequence } } : {}) };
|
|
65
|
+
}
|
|
66
|
+
catch {
|
|
67
|
+
return undefined;
|
|
68
|
+
}
|
|
69
|
+
}
|
|
40
70
|
export function missionExecutionStatus(mission) {
|
|
41
71
|
if (mission.kind !== "operation")
|
|
42
72
|
return "not-required";
|
|
@@ -100,6 +130,9 @@ export class OperatorMissionRuntime {
|
|
|
100
130
|
file(root, suffix = "") {
|
|
101
131
|
return join(this.projectRoot, this.profile.stateDirectory, "missions", `${digest(root)}${suffix}.json`);
|
|
102
132
|
}
|
|
133
|
+
correctionReference(root, reviewIdentity) {
|
|
134
|
+
return JSON.stringify({ path: this.file(root), field: "corrections[]", review_identity: reviewIdentity });
|
|
135
|
+
}
|
|
103
136
|
async load(file) {
|
|
104
137
|
try {
|
|
105
138
|
return JSON.parse(await readFile(file, "utf8"));
|
|
@@ -209,11 +242,6 @@ export class OperatorMissionRuntime {
|
|
|
209
242
|
await this.serial(root, async () => {
|
|
210
243
|
// Captured before prompt rewriting, including exact whitespace. Never ask a model to recopy it.
|
|
211
244
|
await this.save(this.file(root, ".request"), request);
|
|
212
|
-
const state = await this.loadMission(root);
|
|
213
|
-
if (state && !["completed", "cancelled"].includes(state.phase) && !state.requests.some(item => item.id === request.id)) {
|
|
214
|
-
state.requests.push(request);
|
|
215
|
-
await this.save(this.file(root), state);
|
|
216
|
-
}
|
|
217
245
|
});
|
|
218
246
|
}
|
|
219
247
|
recordLaunchConditions(root, raw) {
|
|
@@ -249,6 +277,9 @@ export class OperatorMissionRuntime {
|
|
|
249
277
|
throw new Error("mission-requirements-preserved: keep the existing ordered requirements and append user additions");
|
|
250
278
|
}
|
|
251
279
|
previous.requirements = requirements.map((text, index) => ({ id: `R${index + 1}`, text }));
|
|
280
|
+
// Only an explicit Mission action adopts the latest real turn; chat capture alone never does.
|
|
281
|
+
if (!previous.requests.some(item => item.id === request.id))
|
|
282
|
+
previous.requests.push(request);
|
|
252
283
|
if (options.kind === "operation")
|
|
253
284
|
previous.kind = "operation";
|
|
254
285
|
await this.save(this.file(root), previous);
|
|
@@ -395,7 +426,7 @@ export class OperatorMissionRuntime {
|
|
|
395
426
|
}
|
|
396
427
|
}
|
|
397
428
|
/** The model supplies only useful unit facts; IDs, proof projection and control documents are generated here. */
|
|
398
|
-
export function missionPlan(mission, raw) {
|
|
429
|
+
export function missionPlan(mission, raw, projectRoot) {
|
|
399
430
|
if (!Array.isArray(raw) || raw.length === 0 || raw.length > 32)
|
|
400
431
|
throw new Error("mission-units: declare 1..32 units");
|
|
401
432
|
const acceptance = mission.requirements.map(item => item.text);
|
|
@@ -411,7 +442,12 @@ export function missionPlan(mission, raw) {
|
|
|
411
442
|
const entries = value[field] ?? [];
|
|
412
443
|
if (!Array.isArray(entries) || !entries.every(item => typeof item === "string"))
|
|
413
444
|
throw new Error(`mission-unit-${index + 1}: ${field} must be paths`);
|
|
414
|
-
return [...new Set(entries.map(item =>
|
|
445
|
+
return [...new Set(entries.map(item => {
|
|
446
|
+
// A model's repository-root read means the current project, not an invalid empty path.
|
|
447
|
+
// Resolve only this read shorthand at the host boundary; saved plans and write scopes stay exact.
|
|
448
|
+
const rootRead = field === "read" && [".", "./", ".\\", "./**", ".\\**"].includes(item);
|
|
449
|
+
return normalizeExecutionScope(rootRead && projectRoot ? `${resolve(projectRoot).replaceAll("\\", "/")}/**` : item);
|
|
450
|
+
}))];
|
|
415
451
|
};
|
|
416
452
|
// A sole unit owns the whole request. This schedules work; it does not prove acceptance.
|
|
417
453
|
const ids = value.requirement_ids ?? (raw.length === 1 || mission.requirements.length === 1 ? mission.requirements.map(item => item.id) : []);
|
|
@@ -423,9 +459,8 @@ export function missionPlan(mission, raw) {
|
|
|
423
459
|
throw new Error(`mission-unit-${index + 1}: validation must contain exact commands; final command proves the unit`);
|
|
424
460
|
}
|
|
425
461
|
const validation = value.validation.map(missionValidationCommand);
|
|
426
|
-
// Retain last occurrences so removing a redundant check preserves the final proof command.
|
|
427
462
|
return { id: `unit-${index + 1}`, title: line("title"), objective: line("objective"), read: paths("read"), write: paths("write"),
|
|
428
|
-
validation
|
|
463
|
+
validation,
|
|
429
464
|
acceptance_indices: [...new Set(ids.map(id => mission.requirements.findIndex(item => item.id === id)))] };
|
|
430
465
|
});
|
|
431
466
|
// The serial engine counts new proof milestones. Units sharing the same final suite are one
|
|
@@ -441,7 +476,7 @@ export function missionPlan(mission, raw) {
|
|
|
441
476
|
same.read = [...new Set([...same.read, ...unit.read])];
|
|
442
477
|
same.write = [...new Set([...same.write, ...unit.write])];
|
|
443
478
|
same.acceptance_indices = [...new Set([...same.acceptance_indices, ...unit.acceptance_indices])];
|
|
444
|
-
same.validation = [...
|
|
479
|
+
same.validation = [...same.validation, ...unit.validation];
|
|
445
480
|
}
|
|
446
481
|
const uncovered = mission.requirements.filter((_, i) => !units.some(unit => unit.acceptance_indices.includes(i)));
|
|
447
482
|
if (uncovered.length)
|
|
@@ -459,6 +494,73 @@ export function missionPlan(mission, raw) {
|
|
|
459
494
|
entrypoint: (unit.write[0] ?? unit.read[0] ?? unit.id).slice(0, 512), validation_command: unit.validation.at(-1) })) },
|
|
460
495
|
units }, "execution");
|
|
461
496
|
}
|
|
497
|
+
/** Compact provenance, not a new acceptance verdict or a substitute for original-request comparison. */
|
|
498
|
+
export async function missionAcceptanceSummary(mission, run, operators, observe) {
|
|
499
|
+
const current = run?.runID === mission.runID ? run : undefined;
|
|
500
|
+
const history = current ? await operators.acceptanceHistory(current, [...new Set((mission.attempts ?? []).map(item => item.runID))])
|
|
501
|
+
: { status: "unavailable", runs: [], reason: "current-mission-run-unavailable" };
|
|
502
|
+
const acceptedAnchor = (unit) => current?.priorAcceptedUnits.find(item => item.handoffPath === unit.handoffPath &&
|
|
503
|
+
item.handoffHash === unit.hashes[0] && item.taskID === unit.task.prompt.match(/^task_id: (.+)$/mu)?.[1]);
|
|
504
|
+
const missingAnchors = current?.priorAcceptedUnits.filter(anchor => !history.runs.some(item => item.state.units.some(unit => unit.status === "succeeded" && acceptedAnchor(unit)?.taskID === anchor.taskID))) ?? [];
|
|
505
|
+
const validation = (state, path, historical) => state.units.flatMap(unit => {
|
|
506
|
+
const anchor = historical ? acceptedAnchor(unit) : undefined;
|
|
507
|
+
if (historical && (!anchor || unit.status !== "succeeded"))
|
|
508
|
+
return [];
|
|
509
|
+
return (unit.evidence ?? []).map(proof => ({ run_id: state.runID, unit_id: unit.unit.id,
|
|
510
|
+
task_id: unit.task.prompt.match(/^task_id: (.+)$/mu)?.[1] ?? null,
|
|
511
|
+
worker_session_id: unit.childSessionID, state_archive_path: path, handoff_path: unit.handoffPath,
|
|
512
|
+
...(anchor ? { accepted_anchor: anchor } : {}), evidence_id: proof.evidence_id,
|
|
513
|
+
command: proof.execution.command, exit: proof.execution.exit_code, outcome: proof.execution.outcome,
|
|
514
|
+
started_at: proof.execution.started_at, ended_at: proof.execution.ended_at,
|
|
515
|
+
identity: proof.identity,
|
|
516
|
+
...(proof.protected_binding ? { binding_hashes: { manifest_hash: proof.protected_binding.manifest_hash,
|
|
517
|
+
...(proof.protected_binding.freshness ? { contract_hash: proof.protected_binding.freshness.contract_hash } : {}) },
|
|
518
|
+
operation_manifest_path: proof.protected_binding.manifest_path } : {}),
|
|
519
|
+
details_ref: { path: path ?? operators.statePath(state.rootSessionID),
|
|
520
|
+
unit_id: unit.unit.id, evidence_id: proof.evidence_id,
|
|
521
|
+
omitted: "protected_binding path arrays, project root and environment remain in this exact persisted evidence record" },
|
|
522
|
+
proof_scope: proof.proof_scope,
|
|
523
|
+
applicability: historical ? "historical-reference; current applicability not established by this projection"
|
|
524
|
+
: "current-run record; consult completion readiness for current protected identity" }));
|
|
525
|
+
});
|
|
526
|
+
const nativeValidation = await Promise.all([
|
|
527
|
+
...(current ? [{ path: null, state: current, historical: false }] : []),
|
|
528
|
+
...history.runs.map(item => ({ ...item, historical: true })),
|
|
529
|
+
].flatMap(item => item.state.units.filter(unit => !item.historical ||
|
|
530
|
+
(unit.status === "succeeded" && acceptedAnchor(unit))).map(async (unit) => {
|
|
531
|
+
const provenance = { run_id: item.state.runID, unit_id: unit.unit.id, worker_session_id: unit.childSessionID,
|
|
532
|
+
state_archive_path: item.path, handoff_path: unit.handoffPath, historical: item.historical,
|
|
533
|
+
authority: "native declared-command observations; not additional formal evidence or current freshness" };
|
|
534
|
+
try {
|
|
535
|
+
if (!observe || !unit.childSessionID)
|
|
536
|
+
throw new Error("native-worker-history-unavailable");
|
|
537
|
+
return { ...provenance, status: "available", observations: await observe(unit.unit.validation, unit.childSessionID, unit.reviewerCorrection ? Date.parse(unit.reviewerCorrection.admittedAt ?? item.state.createdAt) : undefined) };
|
|
538
|
+
}
|
|
539
|
+
catch (error) {
|
|
540
|
+
return { ...provenance, status: "unavailable", reason: error instanceof Error ? error.message : String(error) };
|
|
541
|
+
}
|
|
542
|
+
})));
|
|
543
|
+
return { original_requests: mission.requests, requirements: mission.requirements,
|
|
544
|
+
history: { status: missingAnchors.length ? "unavailable" : history.status,
|
|
545
|
+
...(history.reason ? { reason: history.reason } : missingAnchors.length ? { reason: "accepted-handoff-anchor-unavailable" } : {}),
|
|
546
|
+
...(missingAnchors.length ? { unavailable_anchors: missingAnchors } : {}),
|
|
547
|
+
selection: "same-mission Worker run IDs, parent lineage and exact accepted handoff anchors" },
|
|
548
|
+
formal_validation: [...(current ? validation(current, null, false) : []),
|
|
549
|
+
...history.runs.flatMap(item => validation(item.state, item.path, true))],
|
|
550
|
+
native_declared_validation: nativeValidation,
|
|
551
|
+
independent_review: mission.review ? { run_id: mission.review.runID, current_run: mission.review.runID === current?.runID,
|
|
552
|
+
independent: missionReviewIndependent(mission, mission.review.child), mode: mission.review.mode ?? "independent",
|
|
553
|
+
reviewer_session_id: mission.review.child ?? null, source_fingerprint: mission.review.source,
|
|
554
|
+
verdict: mission.review.verdict, result: mission.review.result ?? null, self_recheck: mission.review.selfRecheck ?? null,
|
|
555
|
+
freshness: "not established by run ID; existing source comparison remains required" } : null,
|
|
556
|
+
delivery: { submission: mission.submission, git_lifecycle: current?.gitLifecycle ?? null,
|
|
557
|
+
observation_source: "persisted operator Git lifecycle and formal validation records; no new Git inspection",
|
|
558
|
+
clean: "not independently observed by this projection" },
|
|
559
|
+
interpretation: "Compare original requests with the submitted candidate and actual evidence. Historical PASS is not current PASS. Inspect concrete gaps, not routine archive searches or full source rereads. Existing completion and Review guards still apply." };
|
|
560
|
+
}
|
|
561
|
+
export function missionReviewIndependent(mission, child) {
|
|
562
|
+
return child !== undefined && !(mission.corrections ?? []).some(correction => correction.author === child);
|
|
563
|
+
}
|
|
462
564
|
export function missionPacket(mission, run) {
|
|
463
565
|
const predecessor = run && mission.runID !== run.runID && (mission.supersededRunID === run.runID || run.phase === "cancelled") ? run : undefined;
|
|
464
566
|
if (predecessor)
|
|
@@ -473,9 +575,9 @@ export function missionPacket(mission, run) {
|
|
|
473
575
|
note: "Historical results and spend are retained; they do not complete the current requirements." } } : {}),
|
|
474
576
|
requirements: mission.requirements, original_request_refs: mission.requests.map(item => `user:${item.id}`),
|
|
475
577
|
launch_conditions: mission.launchConditions ?? [], prohibited_write: mission.prohibitedWrite ?? [],
|
|
476
|
-
accounting_scope: "
|
|
578
|
+
accounting_scope: "Implementation units (Worker or scoped Reviewer correction) are not benchmark attempts. Host budget uses the existing Worker-unit ledger; orchestration, read-only Review and external campaign costs are excluded. Launch caps are fixed conditions, not a known campaign remainder.",
|
|
477
579
|
submission: mission.submission, progress: mission.progress, consultations: mission.consultations ?? [],
|
|
478
|
-
attempts: mission.attempts ?? [], ...(mission.rescue ? { rescue: mission.rescue } : {}),
|
|
580
|
+
attempts: mission.attempts ?? [], corrections: (mission.corrections ?? []).map(({ findings: _findings, initialPrompt: _prompt, ...item }) => item), ...(mission.rescue ? { rescue: mission.rescue } : {}),
|
|
479
581
|
operation: { kind: mission.kind ?? "implementation", status: missionExecutionStatus(mission),
|
|
480
582
|
...(mission.execution ? { ...mission.execution } : {}) },
|
|
481
583
|
execution_summary: { completed_units: run?.units.filter(unit => unit.status === "succeeded").length ?? 0,
|
|
@@ -485,6 +587,8 @@ export function missionPacket(mission, run) {
|
|
|
485
587
|
historical_failed_attempts: mission.progress.filter(unit => unit.status === "failed").length,
|
|
486
588
|
accepted: mission.phase === "completed" },
|
|
487
589
|
review: mission.review ? { risk_tags: mission.review.risk, verdict: mission.review.verdict,
|
|
590
|
+
independent: missionReviewIndependent(mission, mission.review.child), mode: mission.review.mode ?? "independent",
|
|
591
|
+
self_recheck: mission.review.selfRecheck ?? null,
|
|
488
592
|
run_id: mission.review.runID, current_run: currentReview,
|
|
489
593
|
source_fingerprint: mission.review.source, reviewer_session_id: mission.review.child ?? null,
|
|
490
594
|
result: mission.review.result ?? null, evidence_gap_reviews: mission.review.evidenceGapReviews ?? 0,
|
|
@@ -500,22 +604,30 @@ export function missionPacket(mission, run) {
|
|
|
500
604
|
...(unit.dispatchDenial ? { dispatch_denial: unit.dispatchDenial } : {}) })) } : {}),
|
|
501
605
|
next_action: mission.phase === "completed" ? "Mission completed. Report the accepted result and retained review gaps; no further dispatch or completion call is needed."
|
|
502
606
|
: mission.phase === "submitted" && mission.submission?.status === "ready"
|
|
503
|
-
? "Operator: compare the submitted candidate with the original
|
|
504
|
-
:
|
|
505
|
-
?
|
|
506
|
-
|
|
507
|
-
|
|
508
|
-
|
|
509
|
-
|
|
510
|
-
|
|
511
|
-
|
|
512
|
-
|
|
513
|
-
|
|
514
|
-
|
|
515
|
-
|
|
516
|
-
|
|
517
|
-
|
|
518
|
-
|
|
607
|
+
? "Operator: use acceptance_summary to compare the submitted candidate with the verbatim original requests and actual evidence, inspect concrete gaps only, then complete_mission if satisfied. Do not routinely search archives or reread all source. Historical PASS is not current PASS. Report remaining evidence gaps; they are not a review PASS."
|
|
608
|
+
: run?.phase === "awaiting-acceptance" && mission.corrections?.some(item => item.runID === run.runID && item.status === "ready") && !reviewAccepted
|
|
609
|
+
? mission.corrections.find(item => item.runID === run.runID)?.selfRecheck?.unresolvedFindings.length
|
|
610
|
+
? "Known Major/Medium findings remain after self-recheck. Call repair_review for the SAME original correction owner; do not dispatch another Reviewer or accept unresolved Medium. Retain original requirements and cumulative spend."
|
|
611
|
+
: mission.corrections.find(item => item.runID === run.runID)?.selfRecheck?.residualMajor
|
|
612
|
+
? "A concrete reachable Major risk remains after native author self-recheck. Call review_mission for a DIFFERENT Reviewer of the correction, retained findings and relevant impact; dispatch the exact returned Task if not already active. No fresh Worker or unchanged validation is required."
|
|
613
|
+
: "Correction ready; call review_mission for the SAME native author to explicitly self-recheck the retained findings, relevant impact and original requirements. Only a concrete reachable Major risk remaining after self-recheck requires a DIFFERENT Reviewer. Known Major/Medium defects still require correction. Do not restart unchanged checks or investigation; CORRECTION_READY alone is not acceptance."
|
|
614
|
+
: mission.kind === "operation" && operationStatus === "running"
|
|
615
|
+
? "The declared operation is already running. Inspect its native shell/progress; do not start another Worker or run. Wait for a terminal result, or report the existing run as blocked if its completion cannot be observed."
|
|
616
|
+
: run?.phase === "awaiting-decision" && mission.corrections?.some(item => item.runID === run.runID && item.status === "failed")
|
|
617
|
+
? "Correction validation or native execution failed. Call repair_review for a new admission in the SAME original Reviewer's native session; fix its own regression, rerun inherited checks and retain the commit/clean boundary and cumulative spend. Do not use ordinary retry, Rescue or a fresh Worker replan. Failure is never acceptance."
|
|
618
|
+
: run?.phase === "awaiting-decision" ? (run.units.some(unit => unit.dispatchDenial)
|
|
619
|
+
? "Coordinator: inspect units[].dispatch_denial before changing the plan. Correct only its diagnosed cause; do not repeat an unchanged refused Task or replan for a host-state mismatch. Report an unresolved runtime mismatch with the loaded runtime identity; preserve requirements and cumulative spend."
|
|
620
|
+
: run.units.some(unit => unit.status === "failed" && unit.resultClass === "acceptance" && unit.failure?.outcome === "fail" && !unit.normalRemediationUsed)
|
|
621
|
+
? "Coordinator: one exact declared validation failed. Reconcile the returned Worker and cumulative budget, then call retry_mission_unit once for that unit before replanning. This preserves the same scope and acceptance."
|
|
622
|
+
: run.units.some(unit => unit.status === "failed" && unit.resultClass === "acceptance" && unit.failure?.outcome === "fail" && unit.normalRemediationUsed && !unit.terminalRescue)
|
|
623
|
+
? "Coordinator: the one ordinary remediation failed the declared validation again. If its native terminal, writer release, exact candidate and cumulative budget permit, call rescue_mission_unit once; inspect and report a non_rescue reason, then continue ordinary correction within budget. Rescue does not bypass validation, review or root acceptance."
|
|
624
|
+
: "Coordinator: correct the cause and call plan_units with the remaining work and all requirements; budget is cumulative.")
|
|
625
|
+
: run?.phase === "awaiting-acceptance" && !operationComplete
|
|
626
|
+
? `Requested operation is ${operationStatus}. Continue the actual operation or report its blocker; auxiliary checks and review disposition cannot complete it.`
|
|
627
|
+
: run?.phase === "awaiting-acceptance" ? (reviewAccepted
|
|
628
|
+
? "Coordinator: review permits submission. Retain advisory review notes; do not repeat passed validation or review for evidence formatting. Operator performs final acceptance against the original requirements."
|
|
629
|
+
: "Coordinator: correct known Major/Medium findings, then obtain explicit native author self-recheck (different Reviewer only for residual concrete Major risk), or initial independent review, then submit_mission. Operator compares all requirements with source/evidence before complete_mission.")
|
|
630
|
+
: "Coordinator: continue the next useful unit within original requirements. Return only a completion candidate, user-only decision, or scope/budget extension." };
|
|
519
631
|
}
|
|
520
632
|
/** Models forward a short capability, never recopy the host's evidence hashes and source packet. */
|
|
521
633
|
export function missionReviewTask(mission) {
|
|
@@ -113,6 +113,15 @@ interface UnitState {
|
|
|
113
113
|
status: "pending" | "running" | "succeeded" | "failed" | "cancelled";
|
|
114
114
|
callID: string | null;
|
|
115
115
|
childSessionID: string | null;
|
|
116
|
+
/** A scoped implementation continuation in the original native Reviewer, never a review PASS. */
|
|
117
|
+
reviewerCorrection?: {
|
|
118
|
+
author: string;
|
|
119
|
+
reviewIdentity: string;
|
|
120
|
+
writeUnion: readonly string[];
|
|
121
|
+
admittedAt?: string;
|
|
122
|
+
promptID?: string;
|
|
123
|
+
checks?: import("../plugin/runtime-bridge.js").ReviewerCorrectionCheck[];
|
|
124
|
+
};
|
|
116
125
|
evidence: readonly GoalEvidence[];
|
|
117
126
|
resultClass: string | null;
|
|
118
127
|
failure?: SerialDispatchSettlement["failure"];
|
|
@@ -257,6 +266,17 @@ export declare class OperatorRuntime {
|
|
|
257
266
|
readonly projectRoot: string;
|
|
258
267
|
constructor(projectRoot: string, profile: RuntimeProfile, gitPath?: string);
|
|
259
268
|
private file;
|
|
269
|
+
/** Exact persisted source for read-only status projections. */
|
|
270
|
+
statePath(root: string): string;
|
|
271
|
+
/** Read-only lineage projection. Never imports archived evidence into current acceptance. */
|
|
272
|
+
acceptanceHistory(state: OperatorState, missionRunIDs: readonly string[]): Promise<{
|
|
273
|
+
status: "available" | "unavailable";
|
|
274
|
+
runs: {
|
|
275
|
+
path: string;
|
|
276
|
+
state: OperatorState;
|
|
277
|
+
}[];
|
|
278
|
+
reason?: string;
|
|
279
|
+
}>;
|
|
260
280
|
read(root: string): Promise<OperatorState | undefined>;
|
|
261
281
|
private save;
|
|
262
282
|
private serial;
|
|
@@ -289,6 +309,12 @@ export declare class OperatorRuntime {
|
|
|
289
309
|
sessionID: string;
|
|
290
310
|
callID: string;
|
|
291
311
|
}, context?: Record<string, unknown>): Promise<OperatorState>;
|
|
312
|
+
prepareReviewerCorrection(root: string, runID: string, raw: unknown, author: string, reviewIdentity: string, writeUnion: readonly string[], dispatcher?: {
|
|
313
|
+
sessionID: string;
|
|
314
|
+
callID: string;
|
|
315
|
+
}, context?: Record<string, unknown>): Promise<OperatorState>;
|
|
316
|
+
recordReviewerCorrectionCheck(root: string, taskID: string, check: import("../plugin/runtime-bridge.js").ReviewerCorrectionCheck): Promise<void>;
|
|
317
|
+
bindReviewerCorrectionPrompt(root: string, child: string, callID: string, promptID: string): Promise<void>;
|
|
292
318
|
private prepareOnce;
|
|
293
319
|
operatorTask(state: OperatorState): OperatorTask;
|
|
294
320
|
/** Root-visible handle for the bounded operations delegate; its contract stays host-internal. */
|