sortie-dogs 0.12.20 → 0.12.21
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +7 -0
- package/dist/asset-version.d.ts +1 -1
- package/dist/asset-version.js +1 -1
- package/dist/plugin/index.js +21 -7
- package/dist/plugin/profiled.js +26 -5
- package/dist/plugin/protected-snapshot.d.ts +3 -0
- package/dist/plugin/protected-snapshot.js +18 -0
- package/dist/runtime-mission-assets.js +11 -0
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -97,6 +97,8 @@ Official SWE-bench Lite `dev` results on the same 23 public instances:
|
|
|
97
97
|
| v0.12.8 (`84ccdf1` main adapter) | 6 / 23 (26.1%) | 2 | Fresh 23-task run; four slots; 30-minute timeout | [Main-integrated run](docs/swebench-main-84ccdf1-dev23-2026-09-26.md) |
|
|
98
98
|
| v0.12.15 (`0c9690d`) | 5 / 23 (21.7%) | 3 | Fresh 23-task ext4 retry; four slots; 40-minute timeout | [Campaign](docs/swebench-v01215-dev23-2026-09-27.md) |
|
|
99
99
|
| v0.12.16 (`9b05a34` release; `b1a6c0e` runner) | 4 / 23 (17.4%) | 5 | Fresh 23-task run; eight slots; 40-minute timeout | [Campaign](docs/swebench-v01216-dev23-2026-09-27.md) |
|
|
100
|
+
| v0.12.19 (`24f5386` release; matched rerun) | 8 / 23 (34.8%) | 0 | Eight slots; effective $2/instance; 40-minute timeout; one inference timeout | [Official result and provenance](docs/benchmarks/swebench-v01220-operation-observability-2026-09-28.md) |
|
|
101
|
+
| v0.12.20 (`628eb81` release) | 7 / 23 (30.4%) | 0 | Eight slots; effective $2/instance; 40-minute timeout | [Official result and caveat](docs/benchmarks/swebench-v01220-operation-observability-2026-09-28.md) |
|
|
100
102
|
|
|
101
103
|
Every row has 23 submitted official predictions; an empty patch counts against
|
|
102
104
|
the score, not as a missing evaluation. The v0.10.14 report does not separately
|
|
@@ -110,6 +112,11 @@ estimated $15.75 in model cost and a 15.2-minute median agent runtime.
|
|
|
110
112
|
The v0.12.16 run is the first eight-slot inference score in this table;
|
|
111
113
|
five runners timed out, and this does not establish a model-quality regression
|
|
112
114
|
against runs with different concurrency and conditions.
|
|
115
|
+
The v0.12.19 row is the corrected run with an effective $2 per-instance cap;
|
|
116
|
+
an earlier v0.12.19 run scored 7/23 but had no effective per-instance cap and
|
|
117
|
+
is not a same-condition comparison. The v0.12.20 run lost the
|
|
118
|
+
`sqlfluff__sqlfluff-2419` resolution relative to the corrected run; this
|
|
119
|
+
single run-to-run difference does not establish causation.
|
|
113
120
|
|
|
114
121
|
Historical qualification references remain in [benchmark reference](docs/benchmark-reference.md).
|
|
115
122
|
|
package/dist/asset-version.d.ts
CHANGED
|
@@ -3,5 +3,5 @@
|
|
|
3
3
|
* installed project marker without importing every asset body.
|
|
4
4
|
*/
|
|
5
5
|
export declare const RUNTIME_ASSET_VERSION = "0.3.89-completion-proof-v1";
|
|
6
|
-
export declare const V010_RUNTIME_ASSET_VERSION = "0.12.
|
|
6
|
+
export declare const V010_RUNTIME_ASSET_VERSION = "0.12.21-validation-recovery-v1";
|
|
7
7
|
export type RuntimeAssetVersion = typeof RUNTIME_ASSET_VERSION | typeof V010_RUNTIME_ASSET_VERSION;
|
package/dist/asset-version.js
CHANGED
|
@@ -3,4 +3,4 @@
|
|
|
3
3
|
* installed project marker without importing every asset body.
|
|
4
4
|
*/
|
|
5
5
|
export const RUNTIME_ASSET_VERSION = "0.3.89-completion-proof-v1";
|
|
6
|
-
export const V010_RUNTIME_ASSET_VERSION = "0.12.
|
|
6
|
+
export const V010_RUNTIME_ASSET_VERSION = "0.12.21-validation-recovery-v1";
|
package/dist/plugin/index.js
CHANGED
|
@@ -51,7 +51,7 @@ import { collectRunMetrics, createSortieResult, createGoalReport, insertRunMetri
|
|
|
51
51
|
import { STABLE_RUNTIME_PROFILE } from "../core/runtime-profile.js";
|
|
52
52
|
import { evidenceFromObservedExecution } from "../core/observed-goal-evidence.js";
|
|
53
53
|
import { receiptBoundTerminalText } from "./receipt-presentation.js";
|
|
54
|
-
import { operationInputSnapshot, protectedSnapshot, refreshProtectedSnapshot } from "./protected-snapshot.js";
|
|
54
|
+
import { operationInputSnapshot, protectedSnapshot, refreshProtectedSnapshot, validationInputSnapshot } from "./protected-snapshot.js";
|
|
55
55
|
import { settledUnitUsage } from "./unit-usage.js";
|
|
56
56
|
import { goalCompletionReadiness } from "./goal-completion.js";
|
|
57
57
|
const INPUT_LIMITS = { config: 64 * 1024, manifest: 512 * 1024, handoff: 2 * 1024 * 1024, parallel: 512 * 1024 };
|
|
@@ -1481,11 +1481,14 @@ export const SortieDogsPlugin = async (input, options) => {
|
|
|
1481
1481
|
const operationInputs = mission?.kind === "operation" && mission.execution?.commands.includes(rawCommand) &&
|
|
1482
1482
|
resolve(input.directory, typeof args?.workdir === "string" ? args.workdir : ".") === mission.execution.directory
|
|
1483
1483
|
? await operationInputSnapshot(authorization.projectRoot, snapshot.binding).catch(() => undefined) : undefined;
|
|
1484
|
+
const validationInputs = validation !== undefined && operationInputs === undefined
|
|
1485
|
+
? await validationInputSnapshot(authorization.projectRoot, snapshot.binding).catch(() => undefined) : undefined;
|
|
1484
1486
|
hostGoalExecutions.set(toolInput.callID, { root, projectRoot: authorization.projectRoot,
|
|
1485
1487
|
sessionID: toolInput.sessionID,
|
|
1486
1488
|
callID: toolInput.callID, tool: toolInput.tool, command: [rawCommand], startedAt: new Date().toISOString(),
|
|
1487
1489
|
binding: snapshot.binding, source: snapshot.source, candidate: snapshot.candidate,
|
|
1488
1490
|
...(operationInputs === undefined ? {} : { operationInputs }),
|
|
1491
|
+
...(validationInputs === undefined ? {} : { validationInputs }),
|
|
1489
1492
|
owner: validation?.request.owner ?? "worker", validation,
|
|
1490
1493
|
...(reusedEvidence === undefined ? {} : { reusedEvidence }) });
|
|
1491
1494
|
}
|
|
@@ -1513,11 +1516,13 @@ export const SortieDogsPlugin = async (input, options) => {
|
|
|
1513
1516
|
? new Date(timing.start).toISOString() : execution.startedAt;
|
|
1514
1517
|
const endedAt = typeof timing?.end === "number" && Number.isFinite(timing.end)
|
|
1515
1518
|
? new Date(timing.end).toISOString() : new Date().toISOString();
|
|
1516
|
-
//
|
|
1517
|
-
//
|
|
1518
|
-
const fresh = refreshed !== undefined && (execution.operationInputs
|
|
1519
|
-
?
|
|
1520
|
-
:
|
|
1519
|
+
// Declared outputs can change during a successful validation. Read inputs must not;
|
|
1520
|
+
// the evidence still binds to the complete post-command candidate and source.
|
|
1521
|
+
const fresh = refreshed !== undefined && (execution.operationInputs !== undefined
|
|
1522
|
+
? await operationInputSnapshot(execution.projectRoot, execution.binding).catch(() => undefined) === execution.operationInputs
|
|
1523
|
+
: execution.validationInputs !== undefined
|
|
1524
|
+
? await validationInputSnapshot(execution.projectRoot, execution.binding).catch(() => undefined) === execution.validationInputs
|
|
1525
|
+
: refreshed.source === execution.source);
|
|
1521
1526
|
const immutableRef = outcome === undefined || execution.reusedEvidence !== undefined ? undefined : goalFingerprint({ root: execution.root,
|
|
1522
1527
|
child_session_id: execution.sessionID, call_id: execution.callID, command: execution.command,
|
|
1523
1528
|
started_at: observedStartedAt, ended_at: endedAt, exit_code: exitCode ?? null, outcome,
|
|
@@ -7528,8 +7533,17 @@ export const SortieDogsPlugin = async (input, options) => {
|
|
|
7528
7533
|
return { status: "unproven", reservation_id: reservation.reservation_id, reason: "owned-native-lineage-unavailable" };
|
|
7529
7534
|
}
|
|
7530
7535
|
const sessionAPI = input.client.session;
|
|
7536
|
+
// V2's session.context is bounded. The aborted parent Task can be days older than
|
|
7537
|
+
// that window; use the owning service's paginated history when the adapter offers it.
|
|
7538
|
+
const historyAPI = sessionAPI;
|
|
7539
|
+
const readHistory = historyAPI.reviewMessages ?? sessionAPI.messages;
|
|
7531
7540
|
const messages = async (id) => {
|
|
7532
|
-
const
|
|
7541
|
+
const request = { path: { id }, query: { directory: input.directory } };
|
|
7542
|
+
const response = await readHistory.call(sessionAPI, request).catch(error => {
|
|
7543
|
+
if (readHistory === sessionAPI.messages)
|
|
7544
|
+
throw error;
|
|
7545
|
+
return sessionAPI.messages(request);
|
|
7546
|
+
});
|
|
7533
7547
|
return isRecord(response) && Array.isArray(response.data) ? response.data.filter(isRecord) : [];
|
|
7534
7548
|
};
|
|
7535
7549
|
const children = async (id) => {
|
package/dist/plugin/profiled.js
CHANGED
|
@@ -1697,7 +1697,18 @@ export function createProfiledPlugin(profile, assetVersion) {
|
|
|
1697
1697
|
if (!reason?.trim())
|
|
1698
1698
|
throw new Error("mission-replan-reason-required: name the observed correction or write-scope extension");
|
|
1699
1699
|
}
|
|
1700
|
-
|
|
1700
|
+
// A cancelled V2 delegate may leave its Worker Task running after the parent Task aborts.
|
|
1701
|
+
// Reconcile that exact native orphan before checking the cumulative budget or replacing
|
|
1702
|
+
// the run; an unproven orphan remains reserved and the normal terminal check still blocks.
|
|
1703
|
+
const cancelledPredecessor = previous?.phase === "cancelled" &&
|
|
1704
|
+
(mission.supersededRunID === previous.runID ||
|
|
1705
|
+
(mission.supersededRunID === undefined && ["explicit-cancellation", "agent-changed"].includes(previous.decision ?? "") &&
|
|
1706
|
+
previous.units.some(unit => (previous.decision === "agent-changed" || unit.status === "cancelled") && unit.childSessionID !== null)));
|
|
1707
|
+
let budget = await control.currentBudget(root);
|
|
1708
|
+
if (cancelledPredecessor && budget?.reserved_units) {
|
|
1709
|
+
await control.reconcileAbortedOperatorOrphan(root);
|
|
1710
|
+
budget = await control.currentBudget(root);
|
|
1711
|
+
}
|
|
1701
1712
|
if (budget && budget.remaining_units < plan.units.length && !same)
|
|
1702
1713
|
throw new Error(`mission-budget-exhausted: plan needs ${plan.units.length} units; ${budget.remaining_units} remain. ` +
|
|
1703
1714
|
"The current run is retained. Correct the plan within the remaining budget, or report a necessary cumulative extension to Operator.");
|
|
@@ -1708,10 +1719,6 @@ export function createProfiledPlugin(profile, assetVersion) {
|
|
|
1708
1719
|
}
|
|
1709
1720
|
// Cancellation marks the durable units before interrupting their native sessions. A cancelled
|
|
1710
1721
|
// status alone therefore cannot prove the old Worker stopped or its reservation settled.
|
|
1711
|
-
const cancelledPredecessor = previous?.phase === "cancelled" &&
|
|
1712
|
-
(mission.supersededRunID === previous.runID ||
|
|
1713
|
-
(mission.supersededRunID === undefined && ["explicit-cancellation", "agent-changed"].includes(previous.decision ?? "") &&
|
|
1714
|
-
previous.units.some(unit => (previous.decision === "agent-changed" || unit.status === "cancelled") && unit.childSessionID !== null)));
|
|
1715
1722
|
const terminalChildren = cancelledPredecessor
|
|
1716
1723
|
? await terminalCancelledMissionChildren(profile, root, previous, budget, {
|
|
1717
1724
|
get: async (id) => {
|
|
@@ -2350,6 +2357,20 @@ export function createProfiledPlugin(profile, assetVersion) {
|
|
|
2350
2357
|
}
|
|
2351
2358
|
}
|
|
2352
2359
|
}
|
|
2360
|
+
const operationArgs = record(mapped.args) ? mapped.args : {};
|
|
2361
|
+
// A decorated launch would spend on the real operation but leave the native mission observation empty.
|
|
2362
|
+
// Ask the same Worker to correct its shell input before that expensive side effect.
|
|
2363
|
+
if (who.role === "dog-worker" && ["bash", "shell"].includes(request.tool.toLowerCase()) && mission?.execution &&
|
|
2364
|
+
typeof operationArgs.command === "string" &&
|
|
2365
|
+
resolve(input.directory, typeof operationArgs.workdir === "string" ? operationArgs.workdir : ".") === mission.execution.directory) {
|
|
2366
|
+
const actual = normalizeCommand(operationArgs.command);
|
|
2367
|
+
const decorated = mission.execution.commands.some(declared => actual.startsWith(declared) &&
|
|
2368
|
+
/^\s*(?:\||\d?>|&&?|;)/u.test(actual.slice(declared.length)));
|
|
2369
|
+
if (decorated)
|
|
2370
|
+
throw new Error("mission-operation-command-not-observed: run the declared operation command exactly; " +
|
|
2371
|
+
"do not append a pipe, tee, redirection or chained command. Capture its tool output or write a separate result file afterward. " +
|
|
2372
|
+
"No new plan or approval is needed; if the run already started, inspect its existing state instead of starting another one.");
|
|
2373
|
+
}
|
|
2353
2374
|
await core["tool.execute.before"]?.({ ...translate(request, false),
|
|
2354
2375
|
...(delegated ? { sessionID: root, agent: "dog-coordinator" } : {}) }, mapped);
|
|
2355
2376
|
if (who.role === "dog-worker") {
|
|
@@ -5,6 +5,9 @@ export declare function isRuntimeControlPath(path: string): boolean;
|
|
|
5
5
|
/** An existing operation may create its declared outputs. Compare its other inputs during execution;
|
|
6
6
|
* ordinary evidence still pins the full post-operation source and outputs for later acceptance. */
|
|
7
7
|
export declare function operationInputSnapshot(projectRoot: string, binding: Binding): Promise<string | undefined>;
|
|
8
|
+
/** A validation may populate write-only caches. Keep its declared read inputs stable
|
|
9
|
+
* while binding the resulting candidate (including those outputs) after the command. */
|
|
10
|
+
export declare function validationInputSnapshot(projectRoot: string, binding: Binding): Promise<string | undefined>;
|
|
8
11
|
export declare function protectedSnapshot(authorization: {
|
|
9
12
|
manifestPath: string;
|
|
10
13
|
manifestHash: string;
|
|
@@ -86,6 +86,24 @@ export async function operationInputSnapshot(projectRoot, binding) {
|
|
|
86
86
|
? await declaredScopeDigest(projectRoot, paths, hash, true, outputs)
|
|
87
87
|
: await protectedScopeDigest(projectRoot, paths, hash, binding.source_policy, outputs);
|
|
88
88
|
}
|
|
89
|
+
/** A validation may populate write-only caches. Keep its declared read inputs stable
|
|
90
|
+
* while binding the resulting candidate (including those outputs) after the command. */
|
|
91
|
+
export async function validationInputSnapshot(projectRoot, binding) {
|
|
92
|
+
const manifestSource = await readFile(resolve(projectRoot, binding.manifest_path)).catch(() => undefined);
|
|
93
|
+
if (!manifestSource || `sha256:${createHash("sha256").update(manifestSource).digest("hex")}` !== binding.manifest_hash)
|
|
94
|
+
return undefined;
|
|
95
|
+
const manifest = JSON.parse(manifestSource.toString("utf8"));
|
|
96
|
+
if (!manifest.read.length)
|
|
97
|
+
return undefined; // Keep the original full-source check when no inputs were declared.
|
|
98
|
+
const paths = manifest.read.map(entry => {
|
|
99
|
+
const path = normalizeManifestScope(entry);
|
|
100
|
+
return path.kind === "relative" ? resolve(projectRoot, path.path) : resolve(path.path);
|
|
101
|
+
});
|
|
102
|
+
const hash = binding.manifest_hash.slice("sha256:".length);
|
|
103
|
+
return binding.source_policy === "declared-paths-v1"
|
|
104
|
+
? declaredScopeDigest(projectRoot, paths, hash, true)
|
|
105
|
+
: protectedScopeDigest(projectRoot, paths, hash, binding.source_policy);
|
|
106
|
+
}
|
|
89
107
|
export async function protectedSnapshot(authorization) {
|
|
90
108
|
const manifestSource = await readFile(authorization.manifestPath).catch(() => undefined);
|
|
91
109
|
if (manifestSource === undefined)
|
|
@@ -13,6 +13,13 @@ For tag observation use git tag --points-at <commit>. For HTTPS downloads use cu
|
|
|
13
13
|
--write-out '%{http_code} %{size_download}\\n' may print metadata to stdout. Declare all actual outputs.
|
|
14
14
|
Reuse pinned artifacts and successful checks when the request permits and the inputs are unchanged.
|
|
15
15
|
Do not repeat candidate discovery, dependency setup or validation merely because a Worker changed.
|
|
16
|
+
Before launching a detached operation, verify supported options and budget from the CLI/preflight;
|
|
17
|
+
a preview is not the live run. Check the actual state after launch, before relying on its limits.
|
|
18
|
+
Run each declared execution command as the exact native shell input; do not append a tee pipeline,
|
|
19
|
+
redirection or wrapper that was not declared. The host observes that command, not a nearby script or result file.
|
|
20
|
+
Save its output separately when needed. A successful launch only proves the process started;
|
|
21
|
+
use that same run's terminal state and official result for completion, never launch it again
|
|
22
|
+
to repair a missing observation.
|
|
16
23
|
`;
|
|
17
24
|
function controls(profile, names) {
|
|
18
25
|
return names.map(name => ` ${profile.toolPrefix}${name}: true`).join("\n");
|
|
@@ -88,6 +95,8 @@ results to guide the following units. Do not invent preparation units or plan-ap
|
|
|
88
95
|
For operations, plan_units.execution names the actual run/grade commands and working directory. Keep
|
|
89
96
|
setup, execution and result collection in the same Worker. The host records native execution; NO_START
|
|
90
97
|
or setup success cannot complete the operation. Reward/score zero is a result, not failure to execute.
|
|
98
|
+
Do not turn a chosen preflight step into a user requirement that the live run's state exists before launch.
|
|
99
|
+
Observe supported flags and budget before launch, then observe the real state immediately after launch.
|
|
91
100
|
Use requirement_ids when splitting multiple requirements across units; a single unit inherits all requirements
|
|
92
101
|
when they are omitted. These are related requirements,
|
|
93
102
|
not claims that a command proves every semantic obligation. Compare the final result yourself.
|
|
@@ -255,6 +264,8 @@ it with sharper traces and evidence: [{path, offset, limit}] from the existing o
|
|
|
255
264
|
Declared external input/output excerpts remain available. The host caps evidence-only reviews; at its limit,
|
|
256
265
|
review is closed with gaps, but ready still requires the requested operation/result to be complete.
|
|
257
266
|
Running an existing procedure alone is not a public-logic source change; use the low-risk skip where applicable.
|
|
267
|
+
Evaluating an unchanged published package is not a release or source edit: use the native execution,
|
|
268
|
+
result and hash records rather than adding an independent source-review round solely for its label.
|
|
258
269
|
Preserve candidate lineage and independence; your own opinion or Worker PASS is not independent review.
|
|
259
270
|
|
|
260
271
|
Call submit_mission only for: ready (complete candidate with evidence/review), needs-decision (only the user
|