@sjawhar/opencode-legion-envoy 3.17.0 → 3.18.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/src/server.js +64 -3
- package/package.json +1 -1
- package/skills/AGENTS.md +1 -1
- package/skills/envoy/SKILL.md +1 -1
- package/skills/legion-retro/SKILL.md +7 -1
- package/skills/legion-worker/SKILL.md +1 -1
- package/skills/legion-worker/references/merge-gate.md +11 -8
- package/skills/legion-worker/references/pr-body.md +33 -5
- package/skills/thermonuclear-deep-review/SKILL.md +26 -0
package/dist/src/server.js
CHANGED
|
@@ -14363,6 +14363,8 @@ var EnvelopeSchema = exports_external.object({
|
|
|
14363
14363
|
// ../contracts/src/handoff-schema.ts
|
|
14364
14364
|
var HANDOFF_SCHEMA_VERSION = 1;
|
|
14365
14365
|
var HANDOFF_PHASES = ["architect", "plan", "implement", "test", "review"];
|
|
14366
|
+
var PLAN_REVIEW_MAX_ROUNDS = 3;
|
|
14367
|
+
var PLAN_REVIEW_VERDICTS = ["approved", "rejected", "failed"];
|
|
14366
14368
|
var isoTimestamp = exports_external.string().regex(/^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}/);
|
|
14367
14369
|
var handoffPhase = exports_external.enum(HANDOFF_PHASES);
|
|
14368
14370
|
var nonEmpty = exports_external.string().trim().min(1);
|
|
@@ -14399,6 +14401,16 @@ var requiredSkillsSchema = exports_external.object({
|
|
|
14399
14401
|
test: exports_external.array(exports_external.string()).optional(),
|
|
14400
14402
|
review: exports_external.array(exports_external.string()).optional()
|
|
14401
14403
|
}).passthrough().optional();
|
|
14404
|
+
var gapAnalysisSchema = exports_external.object({
|
|
14405
|
+
findings: exports_external.array(exports_external.object({ finding: exports_external.string(), answer: exports_external.string() }).passthrough()).optional(),
|
|
14406
|
+
error: exports_external.string().optional()
|
|
14407
|
+
}).passthrough().optional();
|
|
14408
|
+
var planReviewSchema = exports_external.object({
|
|
14409
|
+
verdict: exports_external.enum(PLAN_REVIEW_VERDICTS),
|
|
14410
|
+
rounds: exports_external.number(),
|
|
14411
|
+
remainingIssues: exports_external.array(exports_external.object({ issue: exports_external.string(), evidence: exports_external.string() }).passthrough()).optional(),
|
|
14412
|
+
error: exports_external.string().optional()
|
|
14413
|
+
}).passthrough().optional();
|
|
14402
14414
|
var planSchema = baseHandoffSchema.extend({
|
|
14403
14415
|
phase: exports_external.literal("plan"),
|
|
14404
14416
|
taskCount: exports_external.number().optional(),
|
|
@@ -14406,7 +14418,9 @@ var planSchema = baseHandoffSchema.extend({
|
|
|
14406
14418
|
routingHints: routingHintsSchema,
|
|
14407
14419
|
concerns: exports_external.array(exports_external.string()).optional(),
|
|
14408
14420
|
workflowRecommendation: exports_external.string().optional(),
|
|
14409
|
-
requiredSkills: requiredSkillsSchema
|
|
14421
|
+
requiredSkills: requiredSkillsSchema,
|
|
14422
|
+
gapAnalysis: gapAnalysisSchema,
|
|
14423
|
+
planReview: planReviewSchema
|
|
14410
14424
|
});
|
|
14411
14425
|
var implementSchema = baseHandoffSchema.extend({
|
|
14412
14426
|
phase: exports_external.literal("implement"),
|
|
@@ -14446,13 +14460,60 @@ var reviewSchema = baseHandoffSchema.extend({
|
|
|
14446
14460
|
verdict: exports_external.enum(["approved", "changes_requested"]).optional(),
|
|
14447
14461
|
keyFindings: exports_external.array(exports_external.object({ severity: exports_external.string(), file: exports_external.string(), description: exports_external.string() }).passthrough()).optional()
|
|
14448
14462
|
});
|
|
14449
|
-
var nonEmptySkillList = exports_external.array(
|
|
14463
|
+
var nonEmptySkillList = exports_external.array(nonEmpty).min(1);
|
|
14464
|
+
var recorded = (shape, whatToRecord) => exports_external.object(shape, {
|
|
14465
|
+
error: (issue2) => issue2.input === undefined ? `missing \u2014 record ${whatToRecord}` : undefined
|
|
14466
|
+
}).passthrough();
|
|
14467
|
+
var gapAnalysisWriteSchema = recorded({
|
|
14468
|
+
findings: exports_external.array(exports_external.object({ finding: nonEmpty, answer: nonEmpty }).passthrough()).optional(),
|
|
14469
|
+
error: nonEmpty.optional()
|
|
14470
|
+
}, "the gap analyst's `findings`, each with how the plan answers it (`[]` when it found none), or its failed call's `error`").refine((analysis) => analysis.findings === undefined !== (analysis.error === undefined), {
|
|
14471
|
+
message: "record exactly one of `findings` or the failed call's `error`"
|
|
14472
|
+
});
|
|
14473
|
+
var planReviewWriteSchema = recorded({
|
|
14474
|
+
verdict: exports_external.enum(PLAN_REVIEW_VERDICTS),
|
|
14475
|
+
rounds: exports_external.number().int().min(1).max(PLAN_REVIEW_MAX_ROUNDS),
|
|
14476
|
+
remainingIssues: exports_external.array(exports_external.object({ issue: nonEmpty, evidence: nonEmpty }).passthrough()).optional(),
|
|
14477
|
+
error: nonEmpty.optional()
|
|
14478
|
+
}, "the plan review's `verdict` and `rounds`, with `remainingIssues` when it was rejected or `error` when a review's call failed").superRefine((review, ctx) => {
|
|
14479
|
+
const remaining = review.remainingIssues?.length ?? 0;
|
|
14480
|
+
if (review.verdict === "rejected" && remaining === 0) {
|
|
14481
|
+
ctx.addIssue({
|
|
14482
|
+
code: "custom",
|
|
14483
|
+
path: ["remainingIssues"],
|
|
14484
|
+
message: "a rejected review records the blocking issues its last round named"
|
|
14485
|
+
});
|
|
14486
|
+
}
|
|
14487
|
+
if (review.verdict === "rejected" && review.rounds < PLAN_REVIEW_MAX_ROUNDS) {
|
|
14488
|
+
ctx.addIssue({
|
|
14489
|
+
code: "custom",
|
|
14490
|
+
path: ["rounds"],
|
|
14491
|
+
message: `a review still rejecting after ${review.rounds} of ${PLAN_REVIEW_MAX_ROUNDS} rounds is revised and reviewed again, not recorded`
|
|
14492
|
+
});
|
|
14493
|
+
}
|
|
14494
|
+
if (review.verdict === "approved" && remaining > 0) {
|
|
14495
|
+
ctx.addIssue({
|
|
14496
|
+
code: "custom",
|
|
14497
|
+
path: ["remainingIssues"],
|
|
14498
|
+
message: "an approved review leaves no blocking issue standing"
|
|
14499
|
+
});
|
|
14500
|
+
}
|
|
14501
|
+
if (review.verdict === "failed" !== (review.error !== undefined)) {
|
|
14502
|
+
ctx.addIssue({
|
|
14503
|
+
code: "custom",
|
|
14504
|
+
path: ["error"],
|
|
14505
|
+
message: "a failed review records its call's error, and only a failed review does"
|
|
14506
|
+
});
|
|
14507
|
+
}
|
|
14508
|
+
});
|
|
14450
14509
|
var planWriteSchema = planSchema.extend({
|
|
14451
14510
|
requiredSkills: exports_external.object({
|
|
14452
14511
|
implement: nonEmptySkillList,
|
|
14453
14512
|
test: nonEmptySkillList,
|
|
14454
14513
|
review: nonEmptySkillList
|
|
14455
|
-
}).passthrough()
|
|
14514
|
+
}).passthrough(),
|
|
14515
|
+
gapAnalysis: gapAnalysisWriteSchema,
|
|
14516
|
+
planReview: planReviewWriteSchema
|
|
14456
14517
|
});
|
|
14457
14518
|
var phaseHandoffSchema = exports_external.discriminatedUnion("phase", [
|
|
14458
14519
|
architectSchema,
|
package/package.json
CHANGED
package/skills/AGENTS.md
CHANGED
|
@@ -16,7 +16,7 @@ event intake, process lifecycle, credentials, and role delivery.
|
|
|
16
16
|
| `legion-retro/` | the implementer, at retro | the pre-merge retrospective and its Dispatch message |
|
|
17
17
|
| `legion-worker/` | planner, implementer, tester, reviewer, merger | the phase contracts: handoffs, GitHub identity, PR body and READY discipline, the merge-gate order |
|
|
18
18
|
| `thermonuclear-code-quality/` | the `thermonuclear-code-quality` agent | the maintainability rubric of the reviewer's pair |
|
|
19
|
-
| `thermonuclear-deep-review/` | the `thermonuclear-deep-review` agent
|
|
19
|
+
| `thermonuclear-deep-review/` | the `thermonuclear-deep-review` agent, and the reviewer (the Security Guidelines) | the correctness rubric of the reviewer's pair, with its tagged, diff-triggered Security Guidelines and the attack on the PR body's claims |
|
|
20
20
|
|
|
21
21
|
The owning skill above is where each contract is defined; a role prompt that needs a contract from its own seat points there or restates only its own step. This file lists and does not restate.
|
|
22
22
|
A Legion prompt (a skill here, a role prompt, or an agent definition in `packages/pi-envoy/agents/`) names a task agent only as `task(agent="<name>")` and a skill it tells the model to load only as `skill://<name>`. Those are the two forms the Go daemon's boot gate and `legion probe-image` resolve through Oh My Pi, refusing by name one it cannot find; a dispatch or a load written any other way goes unchecked. An agent or skill Legion's prompts name is shipped here or in `packages/pi-envoy/agents/`, unless Oh My Pi bundles it.
|
package/skills/envoy/SKILL.md
CHANGED
|
@@ -45,7 +45,7 @@ Envoy renders an annotated delivery before its source summary and complete paylo
|
|
|
45
45
|
|
|
46
46
|
```text
|
|
47
47
|
envoy:
|
|
48
|
-
to: you (
|
|
48
|
+
to: you (ses_example_recipient)
|
|
49
49
|
from: 01a0bbbb-cccc-7ddd-eeee-0123456789ab (Reviewer)
|
|
50
50
|
at: "2026-09-07T04:41:12Z"
|
|
51
51
|
id: agent-message-2
|
|
@@ -65,7 +65,13 @@ before step 3. The design gate is not a substitute for review and retro.
|
|
|
65
65
|
## Durable outputs
|
|
66
66
|
|
|
67
67
|
Write the integrated learning as one or more discoverable documents under `docs/solutions/`.
|
|
68
|
-
Organize by reusable topic rather than by pull request.
|
|
68
|
+
Organize by reusable topic rather than by pull request. Search `docs/solutions/` for the topic
|
|
69
|
+
first: when a document already states the rule, update it in place (sharpen the rule, add this
|
|
70
|
+
issue and pull request to `related_issues`) rather than writing a sibling; when the new learning
|
|
71
|
+
replaces an old document, set the old one's `status: superseded` and add
|
|
72
|
+
`superseded_by: docs/solutions/<path>.md`. Open each document with the rule in a few imperative
|
|
73
|
+
lines; the incident that taught it goes in an Evidence section below, never in the rule. Each
|
|
74
|
+
document uses this front matter:
|
|
69
75
|
|
|
70
76
|
```yaml
|
|
71
77
|
---
|
|
@@ -287,7 +287,7 @@ rather than creating a replacement bookmark or PR.
|
|
|
287
287
|
|
|
288
288
|
## PR body, review, and the merge gate
|
|
289
289
|
|
|
290
|
-
The implementer writes the pull request body
|
|
290
|
+
The implementer writes the pull request body from the template when it opens the pull request,
|
|
291
291
|
and every later phase edits its own lines of the live body rather than replacing it. Each proof
|
|
292
292
|
(the implementer's `E2E (implementer)` line and `proof` array, the tester's `E2E (tester)` line
|
|
293
293
|
and `proof` array) is the changed behaviour exercised on a production-like surface, recorded as
|
|
@@ -71,14 +71,17 @@ completion leaves the issue in reviewing until you finish.
|
|
|
71
71
|
First `cd -- "$LEGION_WORKSPACE" && jj -R "$LEGION_WORKSPACE" git fetch && jj -R
|
|
72
72
|
"$LEGION_WORKSPACE" diff --from <approved-sha> --to <tip-sha> --summary`, whose output is quoted
|
|
73
73
|
in READY (an empty output is quoted as `no file changes above the approved head`); then the same
|
|
74
|
-
with `'~docs/solutions'` appended, which must print nothing. The
|
|
75
|
-
`READY #<n> at <current sha> (approved at <approved sha>) for <KEY> (<pr url>)`
|
|
76
|
-
`packages/pi-envoy/roles/merger.md` defines),
|
|
77
|
-
`
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
74
|
+
with `'~docs/solutions'` appended, which must print nothing. *The READY packet*: the merger
|
|
75
|
+
always posts `READY #<n> at <current sha> (approved at <approved sha>) for <KEY> (<pr url>)`
|
|
76
|
+
(the shape `packages/pi-envoy/roles/merger.md` defines), then the PR body's `Outcome:` line and
|
|
77
|
+
its `Not proven / risk:` value — every bullet under that label joined with `; ` on the one
|
|
78
|
+
READY line, or `none` — quoted from the `## For the reviewer` block at that same head (or one
|
|
79
|
+
line saying the body carries no brief — the packet still publishes), then the `--summary`
|
|
80
|
+
output and the PR body's gate facts, as a `dispatch_message` on the issue. When the `Legion
|
|
81
|
+
addressing` line names a merge queue, it also publishes the same packet there with
|
|
82
|
+
`envoy_publish`; a 404 means the Dispatch message remains the durable notice and the merger
|
|
83
|
+
stays idle. The READY packet names both the implementer's and tester's `E2E` lines; a missing
|
|
84
|
+
one is reported to the architect instead of published. Legion never merges.
|
|
82
85
|
|
|
83
86
|
## After the human merge
|
|
84
87
|
|
|
@@ -4,12 +4,24 @@ Part of `skill://legion-worker`. Read it before you write or edit any line of th
|
|
|
4
4
|
body, put a `proof` array in a handoff, verify another phase's proof, or run the simplify pass.
|
|
5
5
|
Every path it cites is in sjawhar/legion.
|
|
6
6
|
|
|
7
|
-
## The
|
|
7
|
+
## The pull request body template
|
|
8
8
|
|
|
9
|
-
The implementer writes the PR body
|
|
9
|
+
The implementer writes the PR body from this template from the moment the PR opens, and every
|
|
10
10
|
later phase keeps it current rather than replacing it:
|
|
11
11
|
|
|
12
12
|
```
|
|
13
|
+
## For the reviewer
|
|
14
|
+
|
|
15
|
+
**Outcome:** <one sentence a user of this repository would recognise: what someone can now do, or what stops going wrong>
|
|
16
|
+
**Why:** <the problem, one or two sentences, ending with the Dispatch key in parentheses — the key only, never a URL>
|
|
17
|
+
**Change:**
|
|
18
|
+
- <two to five bullets, each one behaviour a user or operator meets, never a file name>
|
|
19
|
+
**Look at first:** <one to three `path:line` places where a wrong decision would hurt> (the reviewer writes this line)
|
|
20
|
+
**Proven by:** <the `E2E (implementer)` line's surface and run, one line>
|
|
21
|
+
**Not proven / risk:**
|
|
22
|
+
- <one line per claim recorded as unproven before READY>, or the single word `none` (the reviewer writes this line; `none` is invalid while such a claim stands)
|
|
23
|
+
**Size:** <files changed, +added/−removed>
|
|
24
|
+
|
|
13
25
|
## Verification
|
|
14
26
|
|
|
15
27
|
**CI:** `Tests` run <run-id> — jobs lint, typecheck, test all success at <head-sha>; `PR Title` run <run-id> — job pr-title success at <head-sha>.
|
|
@@ -28,10 +40,10 @@ left open <thread URL> — newest reply by <login> is not its opener's or the Le
|
|
|
28
40
|
<verdict>. (omitted entirely on a docs-only PR — there is no code for either pass, so neither runs)
|
|
29
41
|
|
|
30
42
|
**E2E (implementer):** <surface> — ran `<command or run id>`, observed <result>, at head <sha>.
|
|
31
|
-
Negative control: <deliberately broken input> → <refusal or failure observed>.
|
|
43
|
+
Negative control: <deliberately broken input or call> → <refusal or failure observed>.
|
|
32
44
|
|
|
33
45
|
**E2E (tester):** <surface> — ran `<command or run id>`, observed <result>, at head <sha>.
|
|
34
|
-
Negative control: <deliberately broken input> → <refusal or failure observed>.
|
|
46
|
+
Negative control: <deliberately broken input or call> → <refusal or failure observed>.
|
|
35
47
|
Verified the implementer's proof by <re-running its command | driving the same surface independently>.
|
|
36
48
|
|
|
37
49
|
**Production:** <what was checked in production, how, what was observed> — merge commit <sha>.
|
|
@@ -42,11 +54,27 @@ Verified the implementer's proof by <re-running its command | driving the same s
|
|
|
42
54
|
**Chain:** stacked on <base bookmark> frozen at <sha> / not stacked.
|
|
43
55
|
```
|
|
44
56
|
|
|
57
|
+
## The brief for the human
|
|
58
|
+
|
|
59
|
+
`## For the reviewer` is written for the person who merges; `## Verification` below it stays the
|
|
60
|
+
ledger the reviewer and merger check against GitHub. The implementer writes `Outcome`, `Why`,
|
|
61
|
+
`Change`, `Proven by`, and `Size` when the pull request opens, and keeps them true after every
|
|
62
|
+
push; `Outcome` is a sentence a user of the repository would recognise, never "fix bug" or a file
|
|
63
|
+
name, and `Why` ends with the Dispatch key, never a URL. The reviewer writes `Look at first` and
|
|
64
|
+
`Not proven / risk` at each round, into the live body (`legion gh -- api
|
|
65
|
+
repos/{owner}/{repo}/pulls/{number} --jq .body`, edit, then `--method PATCH ... -F body=@body.md`);
|
|
66
|
+
`Not proven / risk` copies every claim recorded as unproven before READY — the tester's
|
|
67
|
+
`failures`, the reviewer's own review, any proof-check comment already on the pull request —
|
|
68
|
+
word for word, and `none` is a finding while one stands. The merger quotes `Outcome` and
|
|
69
|
+
`Not proven / risk` from the body at the published head in the READY packet
|
|
70
|
+
(*The READY packet* in `skill://legion-worker/references/merge-gate.md`); a stale `Outcome` that no
|
|
71
|
+
longer describes the diff is a finding against the implementer, not a line the merger rewrites.
|
|
72
|
+
|
|
45
73
|
## What a proof is
|
|
46
74
|
|
|
47
75
|
**A proof** is the changed behaviour exercised on the surface a user reaches it through, recorded
|
|
48
76
|
as the exact command or run id, what was observed, the head SHA, and one negative control —
|
|
49
|
-
a deliberately broken input and the refusal or failure observed. The surface is
|
|
77
|
+
a deliberately broken input or call and the refusal or failure observed. The surface is
|
|
50
78
|
**production-like** — the repository's real-process test harness and fixtures, a sandbox
|
|
51
79
|
repository, a real browser, a devN stack, staging, or a local stack with real migrations, one that
|
|
52
80
|
has the resource the change touches — and each `E2E` line carries a **link** to that run,
|
|
@@ -56,6 +56,32 @@ The codebase might gate features behind feature flags or internal-only checks. D
|
|
|
56
56
|
## Intended Breakage Guidelines
|
|
57
57
|
If a high-risk effect is an intentional, well-constrained change, do not report it as a defect. Report it when the scope or consequences appear unclear, including when a safeguard or feature gate is removed.
|
|
58
58
|
|
|
59
|
+
## Claims in the PR body
|
|
60
|
+
|
|
61
|
+
A safety or correctness claim written in a PR body is a claim like any other, and the only reader who catches a wrong one is the reader told to ATTACK it. Verifying reviewers read the code against the claim and pass; that is what they are for. Attack the body's claims, not only its diff.
|
|
62
|
+
|
|
63
|
+
Two shapes to attack first:
|
|
64
|
+
|
|
65
|
+
- **Neutralization ORDER, not coverage.** Any pipeline that sanitizes and then edits can create what it sanitized. The test is not "did it neutralize everything" but "can any later pass CREATE what was being neutralized".
|
|
66
|
+
- **A severity resting on a third party's formatting is a dependency, not a mitigation.** Rate it as the bet it is, and fix rather than disclose.
|
|
67
|
+
|
|
68
|
+
## Security Guidelines
|
|
69
|
+
|
|
70
|
+
For each row whose surface the diff touches, answer with a file:line citation. End your report with one `Security:` line: each touched row's tag and its answer with file:line, or `Security: no sensitive surface in this diff.` when the diff touches none.
|
|
71
|
+
|
|
72
|
+
| Tag | If the diff touches… | Answer, with file:line |
|
|
73
|
+
| --- | --- | --- |
|
|
74
|
+
| `authz` | an authorization or refusal check, or a new route, command, or tool | who may call it, who may not, where the diff enforces that, and what the unauthorized caller gets |
|
|
75
|
+
| `secret` | a token, grant, secret file, credential helper, or its lifetime | what widened: who can read it, for how long, in which process |
|
|
76
|
+
| `untrusted-input` | a subprocess, argv, path, template, or query built from text an outside party controls (an issue body, a PR comment, a webhook payload, model output) | the boundary that neutralizes it, and whether any later pass can re-create what was neutralized (order, not coverage) |
|
|
77
|
+
| `prompt` | a prompt that embeds untrusted text into an agent's instructions | what delimits the untrusted region, and what the agent may do if it obeys that text (OWASP LLM01 prompt injection, LLM06 excessive agency) |
|
|
78
|
+
| `supply-chain` | a dependency, lockfile, base image, or GitHub Action | the version, the pin (a digest or a SHA checked against its tag), and the permissions the workflow runs with |
|
|
79
|
+
| `sandbox` | a sandbox or pod manifest, a capability, or a network policy | which isolation property changed, and against whom |
|
|
80
|
+
| `agent-def` | an agent definition, skill, role prompt, `.omp/` config, or `AGENTS.md` | whether this diff can steer its own reviewers, and why this edit is trustworthy anyway |
|
|
81
|
+
| `transport` | TLS/certificate verification, a signature, HMAC, or randomness source, or an unbounded read/write sized by untrusted input | what changed, the check or bound it relies on, and the ASVS V11/V12 identifier it maps to |
|
|
82
|
+
|
|
83
|
+
Cite an OWASP ASVS v5.0.0 identifier where one applies (`v5.0.0-1.2.5` style). A security finding with no stated exploit path is not a finding: call it hardening and rank it Minor. A security finding that states an exploit path is always its own finding at its real priority, never folded into hardening. Report at most two hardening items, ranked, and fold the rest into one hardening paragraph. Start each security finding with its row's tag, `Security[<tag>]:`, so it can be told from the others and counted by row.
|
|
84
|
+
|
|
59
85
|
## Over-reporting Guidelines
|
|
60
86
|
If you report issues as High priority when they are not in fact high priority / meaningful issues, devs will lose trust in you and stop listening to you over time.
|
|
61
87
|
Never misreport priority or importance. Trace issues end to end and report only what the evidence supports.
|