sortie-dogs 0.13.3 → 0.13.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +136 -59
- package/dist/asset-version.d.ts +1 -1
- package/dist/asset-version.js +1 -1
- package/dist/core/contract-limits.d.ts +2 -1
- package/dist/core/contract-limits.js +2 -1
- package/dist/core/operator-mission.d.ts +17 -1
- package/dist/core/operator-mission.js +11 -6
- package/dist/core/operator-runtime.d.ts +10 -0
- package/dist/core/operator-runtime.js +56 -7
- package/dist/plugin/gate.d.ts +2 -0
- package/dist/plugin/gate.js +14 -4
- package/dist/plugin/index.js +95 -10
- package/dist/plugin/mission-review.d.ts +9 -0
- package/dist/plugin/mission-review.js +17 -0
- package/dist/plugin/native-contract-read.js +11 -3
- package/dist/plugin/profiled.js +325 -64
- package/dist/plugin/protected-snapshot.js +13 -5
- package/dist/plugin/receipt-presentation.d.ts +2 -0
- package/dist/plugin/receipt-presentation.js +48 -0
- package/dist/plugin/runtime-bridge.d.ts +17 -0
- package/dist/plugin/v2.js +14 -5
- package/dist/plugin/validation-scratch.d.ts +2 -2
- package/dist/plugin/validation-scratch.js +29 -17
- package/dist/runtime-mission-assets.d.ts +2 -2
- package/dist/runtime-mission-assets.js +95 -43
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -35,6 +35,11 @@ implementation, validation, review, and model routing.
|
|
|
35
35
|
Guides: [日本語](docs/guide-ja.md) · [简体中文](docs/guide-zh-CN.md) ·
|
|
36
36
|
[Testing](docs/testing.md) · [CLI testing](docs/cli-testing.md)
|
|
37
37
|
|
|
38
|
+
**Current release: [v0.13.4](https://github.com/zufall-upon/Sortie-dogs/releases/tag/v0.13.4)**
|
|
39
|
+
([release notes](docs/release-v0.13.4.md)). The default Mission runtime retains the `v010`
|
|
40
|
+
profile, command and configuration names for compatibility; these names do not mean v0.10 is installed.
|
|
41
|
+
The current asset marker is `0.13.4-reviewer-continuous-v1`.
|
|
42
|
+
|
|
38
43
|
## SWE-bench Lite: 170/300 (56.67%)
|
|
39
44
|
|
|
40
45
|
The fixed **Sortie-dogs v0.12.24** harness resolved **170 of 300 SWE-bench Lite test issues** in one pass@1 campaign, with 9 empty patches and no official evaluation errors. Every instance has a frozen prediction and an inference-time trajectory. The task Workers ran `openai/gpt-6-luna-fast#max`; operator, coordinator and review roles ran `openai/gpt-6-sol#xhigh`. This is a system result, **not** a Luna-only model comparison or a Verified/full SWE-bench score.
|
|
@@ -43,22 +48,25 @@ The fixed **Sortie-dogs v0.12.24** harness resolved **170 of 300 SWE-bench Lite
|
|
|
43
48
|
|
|
44
49
|
The single official 300-instance report and frozen predictions are hash-bound in the report. Confirmed inference expense was **$162.99**; a separate **$34.60** of usage has unknown pricing and is held against the campaign cap, **not** counted as known expense. Leaderboard registration and maintainer acceptance are separate from this official local evaluation.
|
|
45
50
|
|
|
46
|
-
|
|
51
|
+
Historical scores below belong to their fixed candidates, not v0.13.4. SWE-bench is a separate,
|
|
52
|
+
optional measurement rather than a mandatory release gate.
|
|
53
|
+
|
|
54
|
+
> **Beta:** v0.13.x is still stabilizing. Runtime behavior,
|
|
47
55
|
> configuration, and generated assets may still change before 1.0.
|
|
48
56
|
|
|
49
57
|
## Quick start
|
|
50
58
|
|
|
51
|
-
Requirements: Node.js 22.6 or newer, npm, and OpenCode.
|
|
59
|
+
Requirements: Node.js 22.6 or newer, npm, and OpenCode V2.
|
|
52
60
|
|
|
53
61
|
Run these commands in the target project:
|
|
54
62
|
|
|
55
63
|
```sh
|
|
56
|
-
npm install --save-dev sortie-dogs
|
|
64
|
+
npm install --save-dev sortie-dogs@latest
|
|
57
65
|
npx sortie-dogs init .
|
|
58
66
|
```
|
|
59
67
|
|
|
60
|
-
|
|
61
|
-
|
|
68
|
+
`init` defaults to the `v010` Mission profile and registers the OpenCode V2 plugin in
|
|
69
|
+
`.opencode/opencode.json(c)`, preserving unrelated settings. It sets subagent depth to at least two:
|
|
62
70
|
|
|
63
71
|
```json
|
|
64
72
|
{
|
|
@@ -67,51 +75,72 @@ two-level subagent depth to `.opencode/opencode.json(c)`, preserving existing se
|
|
|
67
75
|
}
|
|
68
76
|
```
|
|
69
77
|
|
|
70
|
-
|
|
78
|
+
If an existing local bridge already imports `sortie-dogs/server`, `init` reuses it instead of adding
|
|
79
|
+
a duplicate package entry. A larger existing subagent depth is retained.
|
|
80
|
+
|
|
81
|
+
Completely restart OpenCode, then run:
|
|
71
82
|
|
|
72
83
|
```text
|
|
73
84
|
/sortie-v010 <task>
|
|
74
85
|
```
|
|
75
86
|
|
|
76
87
|
Selecting `dog-operator` directly starts the same workflow. `dog-operator` is the
|
|
77
|
-
|
|
88
|
+
user-facing entry point for the Mission profile. `dogs-coordinator` and every `*-v010` role are
|
|
78
89
|
internal children and must not be selected as task entry points.
|
|
79
90
|
|
|
80
|
-
`init` installs runtime assets and merges the required OpenCode settings; the
|
|
81
|
-
|
|
82
|
-
|
|
91
|
+
`init` installs runtime assets and merges the required OpenCode settings; the package entry or
|
|
92
|
+
existing local bridge loads enforcement and model routing. OpenCode can reload watched configuration,
|
|
93
|
+
but replacing an installed dependency may require a full restart. A new chat session alone does not
|
|
94
|
+
prove the newly installed plugin is loaded.
|
|
95
|
+
|
|
96
|
+
## v0.13.4 runtime updates
|
|
83
97
|
|
|
84
|
-
|
|
98
|
+
PR #152 restores same-session Coordinator/Operator implementation and formal validation through
|
|
99
|
+
`plan_units(executor="self")`, `start_direct_unit` and `finish_direct_unit`. Known single-unit work can
|
|
100
|
+
combine Mission start and planning; Luna Fast/max Worker routing remains the default. Full generated
|
|
101
|
+
contracts remain visible when an explicit Read line range covers the file. Native background
|
|
102
|
+
responsiveness remains: a launch acknowledgement or idle root is not Mission completion.
|
|
85
103
|
|
|
86
|
-
The
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
reachable residual Major risk. Unresolved Major or Medium findings still block acceptance.
|
|
104
|
+
The initial independent Reviewer can investigate, correct, formally validate, deliver and self-recheck
|
|
105
|
+
continuously in its original native Task. Later Major/Medium findings accumulate in that same correction
|
|
106
|
+
context. Author self-recheck remains `self-rechecked`, `independent=false`, never independent `PASS`.
|
|
107
|
+
A different Reviewer is conditional on concrete residual Major risk; unresolved Major/Medium findings
|
|
108
|
+
still block acceptance. Operator owns final comparison and receipt.
|
|
92
109
|
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
110
|
+
Inherited compiler scratch no longer falsely invalidates broad-scope formal proof. Host-observed Git
|
|
111
|
+
delivery, caller-setting review and test-composition guidance reduce avoidable detours. Saved host
|
|
112
|
+
completion cards remain in tool history/UI, while outgoing V2 model/compaction context omits only their
|
|
113
|
+
presentation body, retaining receipt and evidence identities.
|
|
97
114
|
|
|
98
|
-
|
|
115
|
+
Two runs of the same fixed pre-release v8 package completed the original Anko task with official local
|
|
116
|
+
score 1 (F2P 9/9, P2P 94/94) and selected public probes 9/9. Times were 26m49s and 25m55s, costs
|
|
117
|
+
$1.47908048 and $1.54610816; each saved more than seven minutes versus the recorded v5 sample.
|
|
118
|
+
These limited same-task observations do not establish general speedup or a new SWE-bench score.
|
|
119
|
+
Release preflight, full tests, fixed-commit Windows CI and native Worker-start receipts are retained in
|
|
120
|
+
`_testenv/releases/0.13.4/`; startup/model identity is not task completion. See the
|
|
121
|
+
[release notes](docs/release-v0.13.4.md) and [quality-loop evidence](docs/nightly-quality-loop-20261002.md).
|
|
99
122
|
|
|
100
|
-
|
|
123
|
+
## Mission workflow
|
|
124
|
+
|
|
125
|
+
Use Operator → Worker when one useful unit and its meaningful formal check are known;
|
|
126
|
+
use Operator → Coordinator → Worker for actual discovery or decomposition:
|
|
101
127
|
|
|
102
128
|
- `dog-operator` states a few requirements/negative constraints and owns user decisions and final acceptance.
|
|
103
129
|
The host saves the original user message verbatim.
|
|
104
130
|
- Hidden `dogs-coordinator` owns investigation, unit declarations, Worker/Scout/Advisor/Reviewer dispatch,
|
|
105
131
|
in-request write-scope extensions, and corrections. It can read/search and run confirmation shell commands;
|
|
106
|
-
|
|
132
|
+
it can implement and formally validate directly in its own session, or delegate a unit to Worker.
|
|
107
133
|
- `dog-worker-v010` implements a host-generated unit within its file/directory write scopes.
|
|
108
134
|
Investigation commands need no pre-registration; formal checks retain real host-recorded results.
|
|
109
|
-
- High-risk changes require an independent Reviewer. Low-risk skips
|
|
110
|
-
|
|
135
|
+
- High-risk changes require an initial independent Reviewer with read/search access. Low-risk skips
|
|
136
|
+
are explicit and recorded. Reviewer-owned corrections follow the self-recheck policy above.
|
|
137
|
+
- Fast-lane can include high-risk single-unit work; it never implies a review skip.
|
|
138
|
+
- Investigation, edits, formal checks and requested Git delivery stay in the same implementing child.
|
|
139
|
+
Explicit user ordering is retained; no routine plan-approval or commit-only handoff is needed.
|
|
111
140
|
- Unit progress appears on the running Task without stopping Coordinator or prompting Operator.
|
|
112
141
|
|
|
113
|
-
The
|
|
114
|
-
parallel integration path are not exposed in this profile. More agents are not a
|
|
142
|
+
The `v010` Mission profile is serial by design; background responsiveness does not add parallel writers.
|
|
143
|
+
The stable profile's Luna fabric and parallel integration path are not exposed in this profile. More agents are not a
|
|
115
144
|
goal; preserving quality while reducing unnecessary expensive work is.
|
|
116
145
|
|
|
117
146
|
### SWE-bench evaluation
|
|
@@ -186,32 +215,51 @@ astroid **3/5**, pyvista **0/1**, sqlfluff **1/5**.
|
|
|
186
215
|
|
|
187
216
|
## Mission tools
|
|
188
217
|
|
|
189
|
-
1. `start_mission`: Operator supplies concise requirements; the host saves original messages
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
|
|
218
|
+
1. `start_mission`: Operator supplies concise requirements; the host saves original messages. A known single unit
|
|
219
|
+
can include `unit` to combine start/planning and return its configured Worker task.
|
|
220
|
+
2. `plan_units`: Operator or Coordinator supplies title, objective, file/directory scopes and formal checks. The host generates
|
|
221
|
+
IDs, handoff, manifest, proof mapping and the ready Worker task. `executor="self"` keeps execution in the
|
|
222
|
+
same controller session; `start_direct_unit` / `finish_direct_unit` retain observed formal-check freshness.
|
|
223
|
+
3. `operator_next`: advance serial units. `expand_unit` reconciles required in-request outputs while
|
|
224
|
+
preserving the same Task. A reasoned `plan_units` correction or `retry_mission_unit` handles ordinary
|
|
225
|
+
unit recovery under the original requirements and cumulative budget.
|
|
194
226
|
4. `review_mission`: generate the independent review packet from source, requirements and observed checks;
|
|
195
227
|
dispatch its Reviewer task for high-risk changes or record a low-risk skip.
|
|
196
|
-
5. `
|
|
197
|
-
|
|
228
|
+
5. `repair_review`: record findings and continue correction in the running original Reviewer Task, or resume
|
|
229
|
+
that same native session. Later findings accumulate; run inherited checks/requested Git delivery and
|
|
230
|
+
explicitly report `SELF_RECHECKED` in that same Task. Legacy
|
|
231
|
+
`CORRECTION_READY` alone requires a same-author read-only fallback through `review_mission`.
|
|
232
|
+
6. `submit_mission`: Coordinator returns a completion candidate, user-only decision, or proven external/scope/budget blocker.
|
|
233
|
+
7. `complete_mission`: Operator compares the original request, source and evidence, then explicitly accepts.
|
|
198
234
|
Only a succeeded receipt authorizes DONE and the measured 🐾 return report.
|
|
199
235
|
|
|
200
236
|
All tool names use the `sortie_v010_` prefix. Prior proposal/plan-repair tools remain in the compatibility
|
|
201
|
-
implementation but are hidden from the normal
|
|
202
|
-
|
|
237
|
+
implementation but are hidden from the normal Mission tool list. Operator/Coordinator handle in-request
|
|
238
|
+
path reconciliation without a new user approval. Changes beyond the original requirements or cumulative
|
|
239
|
+
budget return to Operator/user. `EVIDENCE_GAPS` is an advisory limitation, not Review `PASS` or an
|
|
240
|
+
automatic extra review; failed or missing required checks still prevent acceptance.
|
|
203
241
|
|
|
204
242
|
Durable profile state and hash-bound task references support restart and
|
|
205
243
|
compaction recovery without reconstructing criteria from summary prose. Stale,
|
|
206
|
-
foreign-root, or changed references are rejected.
|
|
207
|
-
|
|
208
|
-
|
|
244
|
+
foreign-root, or changed references are rejected. Requested `git add <paths>` and `git commit -m ...`
|
|
245
|
+
use the actual source write scope, not a fabricated `.git/**` scope. An optional host-managed Git
|
|
246
|
+
lifecycle also retains its branch, commit and post-commit boundaries. Neither mode grants arbitrary
|
|
247
|
+
Git, force push, release or publication authority.
|
|
248
|
+
|
|
249
|
+
### Progress and acceptance evidence
|
|
250
|
+
|
|
251
|
+
`sortie_v010_operator_status` keeps original requests, formal command/exit/timing observations,
|
|
252
|
+
review disposition and recorded delivery in a compact Mission view. `{ "view": "progress" }` exposes
|
|
253
|
+
the current unit, completed/total units, host budget and next action; `{ "view": "full" }` or
|
|
254
|
+
`details_ref` provides full snapshot diagnostics. Unknown clean state or a failed commit is not delivery
|
|
255
|
+
success. Reading progress does not dispatch, retry or accept work; use native completion notifications
|
|
256
|
+
instead of polling. Existing status reconciliation can recover a missed child terminal event.
|
|
209
257
|
|
|
210
258
|
## Configuration
|
|
211
259
|
|
|
212
260
|
### Profile files and precedence
|
|
213
261
|
|
|
214
|
-
The default package entry is the
|
|
262
|
+
The default package entry is the `v010` Mission profile:
|
|
215
263
|
|
|
216
264
|
- Command: `/sortie-v010`
|
|
217
265
|
- Primary agent: `dog-operator`
|
|
@@ -221,9 +269,11 @@ The default package entry is the v0.10 profile:
|
|
|
221
269
|
- Runtime state: `.sortie-dogs-v010/`
|
|
222
270
|
- Installed asset marker: `.opencode/sortie-dogs-v010.version`
|
|
223
271
|
|
|
272
|
+
For a global install, the marker is `<OpenCode config root>/sortie-dogs-v010.version`.
|
|
273
|
+
|
|
224
274
|
Precedence is built-in defaults, global file, project file, environment JSON,
|
|
225
275
|
then plugin factory options. Unknown properties or invalid types are rejected.
|
|
226
|
-
Use external
|
|
276
|
+
Use external `v010` role names such as `dog-operator`, `dogs-coordinator`, and
|
|
227
277
|
`dog-reviewer-v010` in `modelRouting`; do not also declare their stable aliases.
|
|
228
278
|
|
|
229
279
|
Example `.opencode/sortie-dogs-v010.json`:
|
|
@@ -266,8 +316,8 @@ not invent, probe, or translate variant names.
|
|
|
266
316
|
- `freeTierFallbackModels`: ordered global last-resort model IDs. Default:
|
|
267
317
|
`opencode/deepseek-v4-flash-free`; `[]` disables this fallback.
|
|
268
318
|
- `dedicatedWorkerModel`: canonical stable serial target, default
|
|
269
|
-
`openai/gpt-6.1-sol` / `medium`. The
|
|
270
|
-
role routes below; do not infer
|
|
319
|
+
`openai/gpt-6.1-sol` / `medium`. The Mission profile supplies its explicit
|
|
320
|
+
role routes below; do not infer its Worker route from this stable setting.
|
|
271
321
|
- `consultation.strategy`: fixed advisor identity, optional `required`, and
|
|
272
322
|
positive `maxCallsPerCandidate`; default one call and not required.
|
|
273
323
|
- `consultation.sourceReview`: risk-based review with `maxCallsPerCandidate`
|
|
@@ -281,27 +331,35 @@ not invent, probe, or translate variant names.
|
|
|
281
331
|
- `continuation.summarizeModel`: optional explicit compaction model; omission
|
|
282
332
|
reuses the latest observed root model.
|
|
283
333
|
- `validationProfile`: `fast`, `balanced`, or `assurance`; default `balanced`.
|
|
284
|
-
- `reflection`:
|
|
285
|
-
|
|
286
|
-
|
|
334
|
+
- `reflection`: enabled by default for `run`, `project` and `global` layers, with at most three
|
|
335
|
+
entries / 500 estimated tokens injected. Root Operator can use `sortie_v010_reflection` to retain
|
|
336
|
+
verified process causes/preventions for later turns and sessions; this is not model training.
|
|
337
|
+
Storage and managed blocks are separate from stable. Set `reflection.enabled` to `false` to disable.
|
|
287
338
|
|
|
288
|
-
The
|
|
339
|
+
The Mission host owns handoff and manifest controls under
|
|
289
340
|
`.sortie-dogs-v010/contracts/`. Do not create a legacy root
|
|
290
341
|
`operation-manifest.json` for this profile and do not edit generated controls.
|
|
291
342
|
Delete `.sortie-dogs-v010/` only when no Sortie run is active.
|
|
292
343
|
|
|
293
344
|
### Validation policy
|
|
294
345
|
|
|
295
|
-
`validationProfile` chooses non-canonical depth:
|
|
346
|
+
`validationProfile` chooses supplementary non-canonical depth:
|
|
296
347
|
|
|
297
348
|
- `fast`: static checks
|
|
298
349
|
- `balanced`: targeted checks
|
|
299
350
|
- `assurance`: related checks
|
|
300
351
|
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
|
|
352
|
+
It does not replace meaningful declared formal checks or user/project-required broad validation.
|
|
353
|
+
Batch related edits, run focused checks, then execute every declared formal check in order on the
|
|
354
|
+
stable candidate. The implementing Worker or admitted correcting Reviewer runs those commands;
|
|
355
|
+
canonical/full-suite `owner=coordinator` is evidence accounting, not a requirement for root execution.
|
|
356
|
+
|
|
357
|
+
Keep required broad checks for the final integrated candidate. Reuse valid unchanged evidence only
|
|
358
|
+
when the contract permits; identity includes candidate, command, environment, scope and owner.
|
|
359
|
+
Required repeated occurrences retain their own execution identities and cannot be skipped as duplicates.
|
|
360
|
+
Native commands, actual working directory, exit, duration and saved source bindings establish freshness.
|
|
361
|
+
Diagnostics do not substitute for formal proof. Repeat checks when changes, failures or freshness require
|
|
362
|
+
it, rather than solely because a Worker changed or documentation was edited.
|
|
305
363
|
|
|
306
364
|
### Default routes
|
|
307
365
|
|
|
@@ -332,25 +390,40 @@ export { SortieDogsPlugin } from "sortie-dogs/plugin/stable";
|
|
|
332
390
|
|
|
333
391
|
The stable profile uses `/sortie`, `dog-coordinator`,
|
|
334
392
|
`.opencode/sortie-dogs.json`, `SORTIE_DOGS_CONFIG`, and `.sortie-dogs/`. Do not
|
|
335
|
-
register stable and
|
|
393
|
+
register stable and `v010` from the same package installation path in one host.
|
|
336
394
|
|
|
337
395
|
## Global availability
|
|
338
396
|
|
|
339
|
-
Project-local installation is recommended. To expose
|
|
397
|
+
Project-local installation is recommended. To expose the current Mission assets globally:
|
|
398
|
+
|
|
399
|
+
```sh
|
|
400
|
+
npm install --global sortie-dogs@0.13.4
|
|
401
|
+
sortie-dogs init --global --profile v010
|
|
402
|
+
```
|
|
403
|
+
|
|
404
|
+
Global initialization registers the package or reuses an existing local V2 bridge, and sets subagent
|
|
405
|
+
depth to at least two, preserving unrelated settings and the default agent.
|
|
406
|
+
|
|
407
|
+
An existing `<OpenCode config root>/plugins/sortie-dogs/index.js` bridge importing `sortie-dogs/server`
|
|
408
|
+
can resolve a **separate dependency** under that config root. Updating npm-global alone does not update
|
|
409
|
+
it. For that layout, also install the same release at the actual config root, then rerun global init:
|
|
340
410
|
|
|
341
411
|
```sh
|
|
342
|
-
npm install --
|
|
412
|
+
npm install --prefix "$HOME/.config/opencode" sortie-dogs@0.13.4
|
|
343
413
|
sortie-dogs init --global --profile v010
|
|
344
414
|
```
|
|
345
415
|
|
|
346
|
-
|
|
347
|
-
|
|
416
|
+
The command shows the default config root; use your actual root if overridden. For a configured npm
|
|
417
|
+
package entry, OpenCode V2 also provides `opencode plugin list` / `opencode plugin update`; exact
|
|
418
|
+
version pins require an explicit version change. Completely restart OpenCode after updating, then
|
|
419
|
+
check the installed package, asset marker and loaded plugin version.
|
|
348
420
|
|
|
349
421
|
## Updates and removal
|
|
350
422
|
|
|
351
|
-
|
|
423
|
+
For project-local updates, replace the dependency, rerun initialization and completely restart OpenCode:
|
|
352
424
|
|
|
353
425
|
```sh
|
|
426
|
+
npm install --save-dev sortie-dogs@latest
|
|
354
427
|
npx sortie-dogs init .
|
|
355
428
|
```
|
|
356
429
|
|
|
@@ -358,6 +431,10 @@ npx sortie-dogs init .
|
|
|
358
431
|
version, preserves user configuration, and stops safely on unknown ownership or
|
|
359
432
|
conflicting files.
|
|
360
433
|
|
|
434
|
+
Align any exact version pin or separate bridge dependency with the intended release too. An installed
|
|
435
|
+
marker of `0.13.4-reviewer-continuous-v1` identifies the assets; it does not prove an already-running
|
|
436
|
+
OpenCode process has reloaded the plugin.
|
|
437
|
+
|
|
361
438
|
There is no supported uninstall command. Remove the npm dependency separately,
|
|
362
439
|
then follow the [safe manual removal guide](docs/uninstall.md). Delete only known
|
|
363
440
|
Sortie-owned paths; never remove the whole `.opencode` directory or use broad
|
package/dist/asset-version.d.ts
CHANGED
|
@@ -3,5 +3,5 @@
|
|
|
3
3
|
* installed project marker without importing every asset body.
|
|
4
4
|
*/
|
|
5
5
|
export declare const RUNTIME_ASSET_VERSION = "0.3.89-completion-proof-v1";
|
|
6
|
-
export declare const V010_RUNTIME_ASSET_VERSION = "0.13.
|
|
6
|
+
export declare const V010_RUNTIME_ASSET_VERSION = "0.13.4-reviewer-continuous-v1";
|
|
7
7
|
export type RuntimeAssetVersion = typeof RUNTIME_ASSET_VERSION | typeof V010_RUNTIME_ASSET_VERSION;
|
package/dist/asset-version.js
CHANGED
|
@@ -3,4 +3,4 @@
|
|
|
3
3
|
* installed project marker without importing every asset body.
|
|
4
4
|
*/
|
|
5
5
|
export const RUNTIME_ASSET_VERSION = "0.3.89-completion-proof-v1";
|
|
6
|
-
export const V010_RUNTIME_ASSET_VERSION = "0.13.
|
|
6
|
+
export const V010_RUNTIME_ASSET_VERSION = "0.13.4-reviewer-continuous-v1";
|
|
@@ -6,7 +6,8 @@ export declare const CONTRACT_TEXT_LIMITS: Readonly<{
|
|
|
6
6
|
command: 8192;
|
|
7
7
|
path: 512;
|
|
8
8
|
}>;
|
|
9
|
-
/** New Mission task generation only; persisted/legacy contracts retain their original bounds.
|
|
9
|
+
/** New Mission task generation only; persisted/legacy contracts retain their original bounds.
|
|
10
|
+
* Only target is authoring guidance. maximum is internal headroom, never a new target. */
|
|
10
11
|
export declare const MISSION_OBJECTIVE_LIMITS: Readonly<{
|
|
11
12
|
target: 2000;
|
|
12
13
|
maximum: 3000;
|
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
/** Common text bounds shared by handoff, manifest and goal evidence validation. */
|
|
2
2
|
export const CONTRACT_TEXT_LIMITS = Object.freeze({ title: 160, objective: 32768, statement: 1000, command: 8192, path: 512 });
|
|
3
|
-
/** New Mission task generation only; persisted/legacy contracts retain their original bounds.
|
|
3
|
+
/** New Mission task generation only; persisted/legacy contracts retain their original bounds.
|
|
4
|
+
* Only target is authoring guidance. maximum is internal headroom, never a new target. */
|
|
4
5
|
export const MISSION_OBJECTIVE_LIMITS = Object.freeze({ target: 2000, maximum: 3000 });
|
|
@@ -44,7 +44,7 @@ export interface MissionAttempt {
|
|
|
44
44
|
predecessorAttemptID?: string | null;
|
|
45
45
|
/** Fingerprint of the settled scoped candidate against which a later Rescue is proposed. */
|
|
46
46
|
candidateID?: string;
|
|
47
|
-
kind: "implementation" | "normal_remediation" | "astra_rescue" | "reviewer_correction";
|
|
47
|
+
kind: "implementation" | "normal_remediation" | "astra_rescue" | "reviewer_correction" | "direct_execution";
|
|
48
48
|
status: "pending" | "dispatched" | "succeeded" | "failed" | "cancelled" | "unconfirmed";
|
|
49
49
|
callID?: string;
|
|
50
50
|
childSessionID?: string;
|
|
@@ -151,6 +151,15 @@ export interface OperatorMission {
|
|
|
151
151
|
runID: string | null;
|
|
152
152
|
/** Git HEAD before this mission's first implementation unit, retained across replans and commits. */
|
|
153
153
|
reviewBaseline?: string;
|
|
154
|
+
/** A dated host Git observation, never an inference from a Worker report or lifecycle plan. */
|
|
155
|
+
deliveryObservation?: {
|
|
156
|
+
run_id: string;
|
|
157
|
+
observed_at: string;
|
|
158
|
+
head: string | null;
|
|
159
|
+
branch: string | null;
|
|
160
|
+
clean: boolean;
|
|
161
|
+
source: string;
|
|
162
|
+
};
|
|
154
163
|
/** Cumulative declared review inputs/outputs, including units that failed after writing source. */
|
|
155
164
|
reviewScope?: MissionReviewScope;
|
|
156
165
|
/** Optional Advisor/Scout decisions and native consultation outcomes; never an admission gate. */
|
|
@@ -203,6 +212,13 @@ export interface OperatorMission {
|
|
|
203
212
|
findings: string;
|
|
204
213
|
initialPrompt: string;
|
|
205
214
|
baseline?: string;
|
|
215
|
+
/** Actual still-running initial Review; direct correction is not another child terminal. */
|
|
216
|
+
inlineReview?: {
|
|
217
|
+
callID: string;
|
|
218
|
+
promptID: string;
|
|
219
|
+
admittedAt: number;
|
|
220
|
+
reviewIdentity: string;
|
|
221
|
+
};
|
|
206
222
|
status: "prepared" | "running" | "ready" | "failed" | "cancelled";
|
|
207
223
|
selfRecheck?: MissionSelfRecheck;
|
|
208
224
|
}[];
|
|
@@ -413,8 +413,8 @@ export class OperatorMissionRuntime {
|
|
|
413
413
|
}
|
|
414
414
|
brief(state) {
|
|
415
415
|
return [`mission_id: ${state.id}`, `project_root: ${this.projectRoot}`, "Use the user's language below for all replies and Task titles.",
|
|
416
|
-
"Own investigation,
|
|
417
|
-
"
|
|
416
|
+
"Own investigation, implementation, formal validation, corrections and Worker/Scout/Advisor/independent Reviewer dispatch.",
|
|
417
|
+
"Use plan_units with executor=self for direct work in this session, or delegate promptly when useful. finish_direct_unit records native checks without a Worker handoff. No proposal/approval phase; root alone accepts completion.",
|
|
418
418
|
"Escalate only a completion candidate, a user-only decision, or an extension of original requirements/budget. Unit progress is published without stopping you.",
|
|
419
419
|
"Requirements:", ...state.requirements.map(item => `${item.id}: ${item.text}`),
|
|
420
420
|
`Confirmed launch conditions (fixed limits, not consumption or remaining budget): ${JSON.stringify(state.launchConditions ?? [])}`,
|
|
@@ -508,7 +508,9 @@ export async function missionAcceptanceSummary(mission, run, operators, observe)
|
|
|
508
508
|
return [];
|
|
509
509
|
return (unit.evidence ?? []).map(proof => ({ run_id: state.runID, unit_id: unit.unit.id,
|
|
510
510
|
task_id: unit.task.prompt.match(/^task_id: (.+)$/mu)?.[1] ?? null,
|
|
511
|
-
worker_session_id: unit.
|
|
511
|
+
worker_session_id: unit.directExecution ? null : unit.childSessionID,
|
|
512
|
+
...(unit.directExecution ? { execution_mode: "direct", executor_session_id: unit.directExecution.actor } : {}),
|
|
513
|
+
state_archive_path: path, handoff_path: unit.handoffPath,
|
|
512
514
|
...(anchor ? { accepted_anchor: anchor } : {}), evidence_id: proof.evidence_id,
|
|
513
515
|
command: proof.execution.command, exit: proof.execution.exit_code, outcome: proof.execution.outcome,
|
|
514
516
|
started_at: proof.execution.started_at, ended_at: proof.execution.ended_at,
|
|
@@ -534,7 +536,8 @@ export async function missionAcceptanceSummary(mission, run, operators, observe)
|
|
|
534
536
|
try {
|
|
535
537
|
if (!observe || !unit.childSessionID)
|
|
536
538
|
throw new Error("native-worker-history-unavailable");
|
|
537
|
-
return { ...provenance, status: "available", observations: await observe(unit.unit.validation, unit.childSessionID, unit.
|
|
539
|
+
return { ...provenance, status: "available", observations: await observe(unit.unit.validation, unit.childSessionID, unit.directExecution ? Date.parse(unit.directExecution.startedAt) :
|
|
540
|
+
unit.reviewerCorrection ? Date.parse(unit.reviewerCorrection.admittedAt ?? item.state.createdAt) : undefined) };
|
|
538
541
|
}
|
|
539
542
|
catch (error) {
|
|
540
543
|
return { ...provenance, status: "unavailable", reason: error instanceof Error ? error.message : String(error) };
|
|
@@ -554,8 +557,9 @@ export async function missionAcceptanceSummary(mission, run, operators, observe)
|
|
|
554
557
|
verdict: mission.review.verdict, result: mission.review.result ?? null, self_recheck: mission.review.selfRecheck ?? null,
|
|
555
558
|
freshness: "not established by run ID; existing source comparison remains required" } : null,
|
|
556
559
|
delivery: { submission: mission.submission, git_lifecycle: current?.gitLifecycle ?? null,
|
|
560
|
+
observation: mission.deliveryObservation && mission.deliveryObservation.run_id === current?.runID ? mission.deliveryObservation : null,
|
|
557
561
|
observation_source: "persisted operator Git lifecycle and formal validation records; no new Git inspection",
|
|
558
|
-
clean: "not independently observed by this projection" },
|
|
562
|
+
clean: mission.deliveryObservation && mission.deliveryObservation.run_id === current?.runID ? mission.deliveryObservation.clean : "not independently observed by this projection" },
|
|
559
563
|
interpretation: "Compare original requests with the submitted candidate and actual evidence. Historical PASS is not current PASS. Inspect concrete gaps, not routine archive searches or full source rereads. Existing completion and Review guards still apply." };
|
|
560
564
|
}
|
|
561
565
|
export function missionReviewIndependent(mission, child) {
|
|
@@ -575,7 +579,8 @@ export function missionPacket(mission, run) {
|
|
|
575
579
|
note: "Historical results and spend are retained; they do not complete the current requirements." } } : {}),
|
|
576
580
|
requirements: mission.requirements, original_request_refs: mission.requests.map(item => `user:${item.id}`),
|
|
577
581
|
launch_conditions: mission.launchConditions ?? [], prohibited_write: mission.prohibitedWrite ?? [],
|
|
578
|
-
accounting_scope: "Implementation units (Worker or scoped Reviewer correction) are not benchmark attempts. Host budget uses the existing
|
|
582
|
+
accounting_scope: "Implementation units (Worker, direct controller execution or scoped Reviewer correction) are not benchmark attempts. Host budget uses the existing unit ledger; orchestration, read-only Review and external campaign costs are excluded. Launch caps are fixed conditions, not a known campaign remainder.",
|
|
583
|
+
delivery_observation: mission.deliveryObservation?.run_id === (run?.runID ?? mission.runID) ? mission.deliveryObservation : null,
|
|
579
584
|
submission: mission.submission, progress: mission.progress, consultations: mission.consultations ?? [],
|
|
580
585
|
attempts: mission.attempts ?? [], corrections: (mission.corrections ?? []).map(({ findings: _findings, initialPrompt: _prompt, ...item }) => item), ...(mission.rescue ? { rescue: mission.rescue } : {}),
|
|
581
586
|
operation: { kind: mission.kind ?? "implementation", status: missionExecutionStatus(mission),
|
|
@@ -123,6 +123,13 @@ interface UnitState {
|
|
|
123
123
|
checks?: import("../plugin/runtime-bridge.js").ReviewerCorrectionCheck[];
|
|
124
124
|
};
|
|
125
125
|
evidence: readonly GoalEvidence[];
|
|
126
|
+
/** Implementation in the existing controller session, without a native child Task. */
|
|
127
|
+
directExecution?: {
|
|
128
|
+
actor: string;
|
|
129
|
+
startedAt: string;
|
|
130
|
+
finishedAt?: string;
|
|
131
|
+
checks: import("../plugin/runtime-bridge.js").ReviewerCorrectionCheck[];
|
|
132
|
+
};
|
|
126
133
|
resultClass: string | null;
|
|
127
134
|
failure?: SerialDispatchSettlement["failure"];
|
|
128
135
|
normalRemediationUsed?: boolean;
|
|
@@ -345,6 +352,9 @@ export declare class OperatorRuntime {
|
|
|
345
352
|
next(root: string, actor: string): Promise<unknown>;
|
|
346
353
|
private nextOnce;
|
|
347
354
|
admitWorker(root: string, actor: string, callID: string, args: unknown): Promise<OperatorTask>;
|
|
355
|
+
admitDirect(root: string, actor: string, callID: string): Promise<OperatorState>;
|
|
356
|
+
/** Continue an admitted independent Review in-place; no second native Task or prompt. */
|
|
357
|
+
admitReviewerDirect(root: string, author: string, callID: string, promptID: string): Promise<OperatorState>;
|
|
348
358
|
private admitWorkerOnce;
|
|
349
359
|
rejectedAdmission(root: string, callID: string, reason?: string): Promise<void>;
|
|
350
360
|
rejectDispatch(root: string, callID: string, decision?: string): Promise<void>;
|
|
@@ -741,10 +741,11 @@ export class OperatorRuntime {
|
|
|
741
741
|
return this.serial(root, async () => {
|
|
742
742
|
const state = await this.required(root);
|
|
743
743
|
const unit = state.units.find(item => /^task_id: (.+)$/mu.exec(item.task.prompt)?.[1] === taskID);
|
|
744
|
-
|
|
745
|
-
|
|
744
|
+
const execution = unit?.reviewerCorrection ?? unit?.directExecution;
|
|
745
|
+
if (!unit || !execution || unit.callID !== check.dispatchCallID || unit.childSessionID !== check.childSessionID ||
|
|
746
|
+
(unit.reviewerCorrection?.author ?? unit.directExecution?.actor) !== check.childSessionID)
|
|
746
747
|
throw new Error("mission-review-correction-check-owner-mismatch");
|
|
747
|
-
const checks =
|
|
748
|
+
const checks = execution.checks ??= [];
|
|
748
749
|
if (checks.some(item => item.callID === check.callID))
|
|
749
750
|
return;
|
|
750
751
|
checks.push(structuredClone(check));
|
|
@@ -838,7 +839,7 @@ export class OperatorRuntime {
|
|
|
838
839
|
throw new Error("mission-superseded-run-mismatch");
|
|
839
840
|
const terminalChildren = mission?.terminalChildren ?? [];
|
|
840
841
|
const retainAcceptance = previous !== undefined && cancelledMissionRetainsAcceptance(previous);
|
|
841
|
-
const predecessorChildren = previous?.units.flatMap(unit => unit.childSessionID !== null
|
|
842
|
+
const predecessorChildren = previous?.units.flatMap(unit => !unit.directExecution && unit.childSessionID !== null
|
|
842
843
|
? [unit.childSessionID] : []) ?? [];
|
|
843
844
|
if (superseding && previous && (!replacingFailedAcceptance && !["explicit-cancellation", "agent-changed"].includes(previous.decision ?? "") || previous.gitLifecycle !== null ||
|
|
844
845
|
previous.repairResidualPaths.length > 0 || previous.contractRepair !== null ||
|
|
@@ -866,7 +867,7 @@ export class OperatorRuntime {
|
|
|
866
867
|
const parent = replanning ? { ...previous, phase: "cancelled", decision: "mission-replan" }
|
|
867
868
|
: (!superseding || (retainAcceptance && !mission?.replaceRequirements)) && previous?.phase === "cancelled" ? previous : undefined;
|
|
868
869
|
if (parent && ["explicit-cancellation", "agent-changed"].includes(parent.decision ?? "")) {
|
|
869
|
-
const cancelled = parent.units.flatMap(unit => (parent.decision === "agent-changed" || unit.status === "cancelled") && unit.childSessionID !== null
|
|
870
|
+
const cancelled = parent.units.flatMap(unit => !unit.directExecution && (parent.decision === "agent-changed" || unit.status === "cancelled") && unit.childSessionID !== null
|
|
870
871
|
? [unit.childSessionID] : []);
|
|
871
872
|
if (cancelled.length > 0 && (new Set(terminalChildren).size !== terminalChildren.length ||
|
|
872
873
|
cancelled.some(id => !terminalChildren.includes(id)))) {
|
|
@@ -1337,6 +1338,48 @@ export class OperatorRuntime {
|
|
|
1337
1338
|
admitWorker(root, actor, callID, args) {
|
|
1338
1339
|
return this.serial(root, () => this.admitWorkerOnce(root, actor, callID, args));
|
|
1339
1340
|
}
|
|
1341
|
+
admitDirect(root, actor, callID) {
|
|
1342
|
+
return this.serial(root, async () => {
|
|
1343
|
+
const state = await this.required(root);
|
|
1344
|
+
if (actor !== (state.operatorSessionID ?? root) || !["prepared", "running"].includes(state.phase)) {
|
|
1345
|
+
throw new Error("operator-direct-owner-mismatch");
|
|
1346
|
+
}
|
|
1347
|
+
const unit = state.units.find(item => item.status !== "succeeded");
|
|
1348
|
+
if (!unit || unit.status !== "pending" || unit.reviewerCorrection || unit.repairValidation || unit.terminalRescue) {
|
|
1349
|
+
throw new Error("operator-direct-unit-unavailable");
|
|
1350
|
+
}
|
|
1351
|
+
await this.verifyControls(unit);
|
|
1352
|
+
unit.directExecution = { actor, startedAt: new Date().toISOString(), checks: [] };
|
|
1353
|
+
unit.childSessionID = actor;
|
|
1354
|
+
unit.callID = callID;
|
|
1355
|
+
unit.status = "running";
|
|
1356
|
+
state.phase = "running";
|
|
1357
|
+
await this.save(state);
|
|
1358
|
+
return state;
|
|
1359
|
+
});
|
|
1360
|
+
}
|
|
1361
|
+
/** Continue an admitted independent Review in-place; no second native Task or prompt. */
|
|
1362
|
+
admitReviewerDirect(root, author, callID, promptID) {
|
|
1363
|
+
return this.serial(root, async () => {
|
|
1364
|
+
const state = await this.required(root);
|
|
1365
|
+
const unit = state.units.find(item => item.status !== "succeeded");
|
|
1366
|
+
if (state.phase !== "prepared" || !unit || unit.status !== "pending" ||
|
|
1367
|
+
unit.reviewerCorrection?.author !== author || !promptID || unit.repairValidation || unit.terminalRescue) {
|
|
1368
|
+
throw new Error("mission-review-direct-unit-unavailable");
|
|
1369
|
+
}
|
|
1370
|
+
await this.verifyControls(unit);
|
|
1371
|
+
const startedAt = new Date().toISOString();
|
|
1372
|
+
unit.directExecution = { actor: author, startedAt, checks: [] };
|
|
1373
|
+
unit.reviewerCorrection.admittedAt = startedAt;
|
|
1374
|
+
unit.reviewerCorrection.promptID = promptID;
|
|
1375
|
+
unit.childSessionID = author;
|
|
1376
|
+
unit.callID = callID;
|
|
1377
|
+
unit.status = "running";
|
|
1378
|
+
state.phase = "running";
|
|
1379
|
+
await this.save(state);
|
|
1380
|
+
return state;
|
|
1381
|
+
});
|
|
1382
|
+
}
|
|
1340
1383
|
async admitWorkerOnce(root, actor, callID, args) {
|
|
1341
1384
|
const state = await this.required(root);
|
|
1342
1385
|
if (actor !== state.operatorSessionID && !(actor === root && state.units.length === 1 && state.operatorSessionID === null))
|
|
@@ -1709,6 +1752,8 @@ export class OperatorRuntime {
|
|
|
1709
1752
|
else
|
|
1710
1753
|
delete unit.failure;
|
|
1711
1754
|
unit.status = result.disposition;
|
|
1755
|
+
if (unit.directExecution)
|
|
1756
|
+
unit.directExecution.finishedAt = new Date().toISOString();
|
|
1712
1757
|
if (unit.terminalRescue) {
|
|
1713
1758
|
unit.terminalRescue.status = "settled";
|
|
1714
1759
|
unit.terminalRescue.disposition = result.disposition;
|
|
@@ -1873,10 +1918,14 @@ export class OperatorRuntime {
|
|
|
1873
1918
|
else if (state.decision !== ACCEPTANCE_REMEDIATION_DECISION)
|
|
1874
1919
|
state.decision = reason;
|
|
1875
1920
|
for (const unit of state.units)
|
|
1876
|
-
if (unit.status === "running")
|
|
1921
|
+
if (unit.status === "running") {
|
|
1877
1922
|
unit.status = "cancelled";
|
|
1923
|
+
if (unit.directExecution)
|
|
1924
|
+
unit.directExecution.finishedAt = new Date().toISOString();
|
|
1925
|
+
}
|
|
1878
1926
|
await this.save(state);
|
|
1879
|
-
return [state.operatorSessionID, ...state.units.
|
|
1927
|
+
return [state.operatorSessionID, ...state.units.filter(unit => !unit.directExecution).map(unit => unit.childSessionID)]
|
|
1928
|
+
.filter((id) => id !== null && id !== root);
|
|
1880
1929
|
}
|
|
1881
1930
|
terminal(root, receipt) {
|
|
1882
1931
|
return this.serial(root, () => this.terminalOnce(root, receipt));
|
package/dist/plugin/gate.d.ts
CHANGED
|
@@ -36,6 +36,8 @@ interface Extraction {
|
|
|
36
36
|
gitCommit?: boolean;
|
|
37
37
|
gitCommitAdds?: string[][];
|
|
38
38
|
gitMutation?: boolean;
|
|
39
|
+
/** Ambiguity in the Git mutation itself, not another native shell segment. */
|
|
40
|
+
gitAmbiguous?: boolean;
|
|
39
41
|
remoteMutation?: boolean;
|
|
40
42
|
issue?: CommandIssue;
|
|
41
43
|
}
|