@sjawhar/opencode-legion-envoy 2.0.2 → 2.0.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/src/server.js +3 -0
- package/package.json +1 -1
- package/skills/AGENTS.md +4 -0
- package/skills/ce-simplify-code/LICENSE +21 -0
- package/skills/ce-simplify-code/SKILL.md +64 -0
- package/skills/ce-simplify-code/references/personas/code-quality-reviewer.md +17 -0
- package/skills/ce-simplify-code/references/personas/code-reuse-reviewer.md +7 -0
- package/skills/ce-simplify-code/references/personas/efficiency-reviewer.md +11 -0
- package/skills/dispatch/SKILL.md +28 -7
- package/skills/legion-architect/SKILL.md +5 -5
- package/skills/legion-controller/SKILL.md +1 -1
- package/skills/legion-worker/SKILL.md +13 -14
- package/skills/thermonuclear-code-quality/SKILL.md +192 -0
- package/skills/thermonuclear-deep-review/SKILL.md +72 -0
package/dist/src/server.js
CHANGED
|
@@ -13561,6 +13561,7 @@ function serviceSubjectLabel(subject) {
|
|
|
13561
13561
|
}
|
|
13562
13562
|
var ASK_TURNS = ["human", "agent"];
|
|
13563
13563
|
var DELIVERY_CAPABILITIES = ["aside", "btw", "steer"];
|
|
13564
|
+
var DELIVERY_DUPLICATE_WINDOW_MS = 72 * 60 * 60 * 1000;
|
|
13564
13565
|
var DispatchEventSchema = exports_external.object({
|
|
13565
13566
|
issue_key: exports_external.string().nullable(),
|
|
13566
13567
|
artifact_id: exports_external.string().nullish(),
|
|
@@ -13648,6 +13649,7 @@ var CommentDeliverySchema = exports_external.object({
|
|
|
13648
13649
|
delivery: exports_external.enum(DELIVERY_CAPABILITIES),
|
|
13649
13650
|
session_id: exports_external.string().nullable(),
|
|
13650
13651
|
envelope_id: exports_external.string().nullable(),
|
|
13652
|
+
duplicate: exports_external.boolean().optional(),
|
|
13651
13653
|
state: exports_external.enum(["pending", "sent", "failed"]),
|
|
13652
13654
|
error: exports_external.string().nullable(),
|
|
13653
13655
|
resolve_error: exports_external.string().nullable(),
|
|
@@ -13733,6 +13735,7 @@ var MessageDeliveryEventPayloadSchema = exports_external.object({
|
|
|
13733
13735
|
target: exports_external.string().optional(),
|
|
13734
13736
|
title: exports_external.string().optional(),
|
|
13735
13737
|
state: exports_external.enum(["sent", "failed"]).optional(),
|
|
13738
|
+
duplicate: exports_external.boolean().optional(),
|
|
13736
13739
|
error: exports_external.string().optional()
|
|
13737
13740
|
});
|
|
13738
13741
|
var ChildStatusEventPayloadSchema = exports_external.object({
|
package/package.json
CHANGED
package/skills/AGENTS.md
CHANGED
|
@@ -6,6 +6,7 @@ event intake, process lifecycle, credentials, and role delivery.
|
|
|
6
6
|
|
|
7
7
|
| Skill | Who reads it | What it owns |
|
|
8
8
|
| --- | --- | --- |
|
|
9
|
+
| `ce-simplify-code/` | the implementer, once per pull request | the behaviour-preserving simplify pass before the reviewer's final pass (Legion's copy of the MIT-licensed Compound Engineering skill; `LICENSE` beside it) |
|
|
9
10
|
| `dispatch/` | every role, and any session writing to Dispatch | specs, asks, comments, artifacts, and messages on native Dispatch |
|
|
10
11
|
| `envoy/` | every role | subscriptions, agent-to-agent messages, and topic formats |
|
|
11
12
|
| `legion-architect/` | root and sub-architects | tree ownership, decomposition, waves, gates, integration, sign-off |
|
|
@@ -13,8 +14,11 @@ event intake, process lifecycle, credentials, and role delivery.
|
|
|
13
14
|
| `legion-oracle/` | any role doing research | repository-grounded research |
|
|
14
15
|
| `legion-retro/` | the implementer, at retro | the pre-merge retrospective and its Dispatch message |
|
|
15
16
|
| `legion-worker/` | planner, implementer, tester, reviewer, merger | the phase contracts: handoffs, GitHub identity, PR body and READY discipline, the merge-gate order |
|
|
17
|
+
| `thermonuclear-code-quality/` | the `thermonuclear-code-quality` agent | the maintainability rubric of the reviewer's pair |
|
|
18
|
+
| `thermonuclear-deep-review/` | the `thermonuclear-deep-review` agent | the security and correctness rubric of the reviewer's pair |
|
|
16
19
|
|
|
17
20
|
The owning skill above is where each contract is defined; a role prompt that needs a contract from its own seat points there or restates only its own step. This file lists and does not restate.
|
|
21
|
+
A Legion prompt (a skill here, a role prompt, or an agent definition in `packages/pi-envoy/agents/`) names a task agent only as `task(agent="<name>")` and a skill it tells the model to load only as `skill://<name>`. Those are the two forms the Go daemon's boot gate and `legion probe-image` resolve through Oh My Pi, refusing by name one it cannot find; a dispatch or a load written any other way goes unchecked. An agent or skill Legion's prompts name is shipped here or in `packages/pi-envoy/agents/`, unless Oh My Pi bundles it.
|
|
18
22
|
The text a worker boots with (its role prompt) lives in `packages/pi-envoy/roles/` and is
|
|
19
23
|
composed per role in `packages/daemon/src/daemon/processes.ts`; `packages/pi-envoy/roles/roles.test.ts`
|
|
20
24
|
holds the structural rules for those parts.
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2025 Every
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: ce-simplify-code
|
|
3
|
+
description: "Simplify settled, recently changed code for clarity, reuse, quality, and efficiency while preserving behavior. Use after implementation and before review."
|
|
4
|
+
argument-hint: "[blank to simplify current branch changes, or describe what to simplify]"
|
|
5
|
+
---
|
|
6
|
+
|
|
7
|
+
Simplify recently changed code for clarity, reuse, quality, and efficiency while preserving exact behavior. Prioritize readable, explicit code over compact code — fewer lines is not the goal.
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
## Step 1: Identify scope
|
|
11
|
+
|
|
12
|
+
Resolve the simplification scope in this order:
|
|
13
|
+
|
|
14
|
+
1. **User-named scope** is authoritative; do not widen it.
|
|
15
|
+
2. **Otherwise, in git**, use the current branch versus its base. Without a usable base, use staged and unstaged changes (`git diff HEAD`).
|
|
16
|
+
3. **Outside git or without a diff**, use files the user named or that were edited earlier in the conversation.
|
|
17
|
+
|
|
18
|
+
If none of the above produces a non-empty scope, stop and ask the user what to simplify rather than guessing. Use the host's blocking question tool already in the current tool list (match by capability, not by a host-specific name). Presence in the current tool list is proof the tool exists; never call a user-facing question tool to discover whether it exists. If a matching tool is listed but unloaded, use the host's tool-discovery primitive to load that capability — do not search for another host's tool name. Fall back to numbered options on the host's user-visible chat surface only when no such tool is in the list or a real question call errors. Never silently skip the question.
|
|
19
|
+
|
|
20
|
+
**Preflight.** If the scope has no substantive human-authored code — only documentation, generated or vendored files, dependencies or lockfiles, or mechanical churn — report that there is nothing to simplify and stop without reviewers. For mixed scopes, retain only the code. This is a kind gate, never a size gate: explicit small scopes still run, and callers own any size or cost threshold.
|
|
21
|
+
|
|
22
|
+
When the platform's task-tracking capability is available, show the review, apply, and verification outcomes without creating one task per reviewer. Otherwise continue without simulating a task list in chat.
|
|
23
|
+
|
|
24
|
+
## Step 2: Launch 3 review agents in parallel
|
|
25
|
+
|
|
26
|
+
Dispatch three generic subagents — code-reuse, code-quality, and efficiency reviewers — via the platform's subagent primitive (`Agent`/`Task` in Claude Code, `spawn_agent` in Codex) where available; otherwise run the reviews inline or serially. For each reviewer, read its prompt asset from this skill's directory and pass the **full file content** as the subagent's prompt, together with the resolved scope (the full diff or file set) so it has complete context:
|
|
27
|
+
|
|
28
|
+
- `references/personas/code-reuse-reviewer.md`
|
|
29
|
+
- `references/personas/code-quality-reviewer.md`
|
|
30
|
+
- `references/personas/efficiency-reviewer.md`
|
|
31
|
+
|
|
32
|
+
Do not paraphrase these rubrics from memory — read each file and pass it verbatim, or the reviewer loses the gating rules that keep the pass behavior-preserving.
|
|
33
|
+
|
|
34
|
+
**Bounded dispatch.** Queue the three reviewers and launch only as many as the harness accepts at once; treat a concurrency/active-agent-limit error as backpressure (leave the reviewer queued and retry after a slot frees), not as reviewer failure. If a dispatch fails for a reason that survives correcting the invocation, run that reviewer's pass inline in the parent context using the same prompt asset, and disclose the substitution in one line.
|
|
35
|
+
|
|
36
|
+
**Agent.** In Oh My Pi, dispatch each persona as the bundled reviewer, `task(agent="reviewer")`. Which model that agent runs on is the operator's configuration.
|
|
37
|
+
|
|
38
|
+
**Permission mode.** Omit the `mode` parameter on the dispatch call so the user's configured permission settings apply.
|
|
39
|
+
|
|
40
|
+
## Step 3: Fix issues
|
|
41
|
+
|
|
42
|
+
Proceed only after all three review outcomes are complete, whether returned by subagents or produced inline. Apply worthwhile findings directly; record false positives and low-value findings as skipped without asking the user.
|
|
43
|
+
|
|
44
|
+
Inspect beyond the resolved scope when needed to evaluate a finding, but edit only that scope and its necessary import/export seams. For a user-named file or directory scope, those seams must also be inside it; skip any fix that would edit outside the mutation boundary.
|
|
45
|
+
|
|
46
|
+
Each fix must preserve outputs, errors, side effects, and ordering. If that cannot be established, skip it.
|
|
47
|
+
|
|
48
|
+
An interface or data shape that existed only in an earlier iteration of the current unshipped scope is not protected behavior once you verify it has no deployed, persisted, public, external, dependent-branch, or in-repo caller outside the resolved scope. Remove that compatibility path only when every required caller update fits the existing mutation boundary; otherwise preserve it.
|
|
49
|
+
|
|
50
|
+
**Never simplify away a safety check.** Preserve trust-boundary validation, data-loss protection, security checks, and accessibility affordances. Skip any finding that would thin or remove one.
|
|
51
|
+
|
|
52
|
+
**Honor caller-passed structure pins.** A plan path passed with the structure-pin constraint is context, not scope. Preserve its `session-settled:` Key Technical Decisions, including deliberate duplication or separation.
|
|
53
|
+
|
|
54
|
+
## Step 4: Verify behavior is preserved
|
|
55
|
+
|
|
56
|
+
Run project-wide typecheck and lint. Run tests matched to blast radius: scoped tests for local changes, broader tests for shared or wide-reach changes, and the full suite when the runner cannot scope tests.
|
|
57
|
+
|
|
58
|
+
Report failures with the check name and relevant output. Fix simplification-caused failures or revert the responsible change; never relax assertions, weaken types, or skip tests.
|
|
59
|
+
|
|
60
|
+
If no test suite, lint, or typecheck is configured, state that explicitly in the summary; do not silently skip verification.
|
|
61
|
+
|
|
62
|
+
## Step 5: Summarize
|
|
63
|
+
|
|
64
|
+
Summarize what was already sound and what improved. Report applied counts by reuse, quality, and efficiency; skipped count; and check outcomes. Name the agent (or model) that produced each reviewer's findings, so the reader can see which model reviewed. If nothing changed, say so. Do not use net lines removed as the success metric.
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
You are the **Code Quality Reviewer**. You receive recently changed code as a diff or resolved file set. Find hacky patterns, while preserving exact behavior. Review for:
|
|
2
|
+
|
|
3
|
+
1. **Redundant state**: state that duplicates existing state, cached values that could be derived, observers/effects that could be direct calls
|
|
4
|
+
2. **Parameter sprawl**: adding new parameters to a function instead of generalizing or restructuring existing ones
|
|
5
|
+
3. **Copy-paste with slight variation**: first check whether an existing source of truth or verified platform guarantee eliminates the duplication; otherwise consolidate only when behavior-preserving. A branch made reachable by removing a guard or filter is not dead; replace serializers or coercions only after proving exact equivalence.
|
|
6
|
+
4. **Leaky abstractions**: exposing internal details that should be encapsulated, or breaking existing abstraction boundaries
|
|
7
|
+
5. **Stringly-typed code**: using raw strings where constants, enums (string unions), or branded types already exist in the codebase
|
|
8
|
+
6. **Unnecessary wrapper elements (framework-gated)**: in component-tree UI frameworks only, flag wrappers with no layout or behavioral role; skip elsewhere
|
|
9
|
+
7. **Nested conditionals**: ternary, if/else, or switch nesting 3+ levels deep
|
|
10
|
+
8. **Unnecessary comments**: flag comments that restate the code, narrate changes, or preserve task history; keep non-obvious constraints and invariants
|
|
11
|
+
9. **Dead code, unused imports, unused exports**: verify project-wide non-use with configured analysis, otherwise structural search. Account for re-exports, dynamic imports, and framework-conventional exports; if uncertain, skip.
|
|
12
|
+
10. **Context-dependent vocabulary**: rename conversation- or iteration-bound and inconsistent terms toward established codebase vocabulary; preserve precise domain terms
|
|
13
|
+
11. **Pre-release compatibility scaffolding**: remove forms superseded entirely within the current branch only after verifying they were never deployed, persisted, public, external, or consumed by a dependent branch; if uncertain, skip
|
|
14
|
+
|
|
15
|
+
**Balance.** Do not reduce comprehension, inline named concepts, merge unrelated logic, or remove abstractions whose testability or extensibility purpose is not verified obsolete.
|
|
16
|
+
|
|
17
|
+
Return each finding as: location (`file:line`), the issue, and the concrete fix. If there is nothing to flag, say so explicitly.
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
You are the **Code Reuse Reviewer**. You receive recently changed code as a diff or resolved file set. Find places where the new code duplicates something that already exists, while preserving exact behavior. For each change:
|
|
2
|
+
|
|
3
|
+
1. **Existing utilities and helpers**: search for behavior-equivalent symbols that replace new functions or inline logic; name the symbol to use
|
|
4
|
+
2. **Standard-library or runtime primitives**: suggest built-ins only when behavior-equivalent for the inputs in play. Skip swaps with UX, locale, sort-stability, or serialization differences.
|
|
5
|
+
3. **Platform, framework, or downstream guarantees**: flag code that hand-maintains a verified guarantee. Name the provider and resulting simplification. Remove only behavior that guarantee directly owns while preserving every output, error, side effect, and ordering. Keep value transformations before downstream projection. Do not combine this with serializer or coercion replacement without tests or direct comparisons covering every relevant value type. Newly reachable branches are not dead code.
|
|
6
|
+
|
|
7
|
+
Return each finding as: location (`file:line`), the duplication or missed reuse, and the existing utility or built-in to use instead. If there is nothing to flag, say so explicitly.
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
You are the **Efficiency Reviewer**. You receive recently changed code as a diff or resolved file set. Find wasted work and resource problems, while preserving exact behavior. Review for:
|
|
2
|
+
|
|
3
|
+
1. **Unnecessary work**: redundant computations, repeated file reads, duplicate network/API calls, N+1 patterns
|
|
4
|
+
2. **Missed concurrency**: independent operations run sequentially when they could run in parallel
|
|
5
|
+
3. **Hot-path bloat**: new blocking work added to startup or per-request/per-render hot paths
|
|
6
|
+
4. **Recurring no-op updates**: guard polling, event, and reducer updates; verify wrappers preserve the platform's no-change signal, such as a same-reference return
|
|
7
|
+
5. **Unnecessary existence checks**: pre-checking file/resource existence before operating (TOCTOU anti-pattern) — operate directly and handle the error
|
|
8
|
+
6. **Memory**: unbounded data structures, missing cleanup, event listener leaks
|
|
9
|
+
7. **Overly broad operations**: reading entire files when only a portion is needed, loading all items when filtering for one
|
|
10
|
+
|
|
11
|
+
Return each finding as: location (`file:line`), the inefficiency, and the concrete fix. If there is nothing to flag, say so explicitly.
|
package/skills/dispatch/SKILL.md
CHANGED
|
@@ -92,8 +92,9 @@ A spec has two readers: the human who decides reads the **Summary** and **New si
|
|
|
92
92
|
|
|
93
93
|
- The spec is the issue's one primary document. Extend it in place — a new version that keeps the
|
|
94
94
|
human's own text — never a second "spec" artifact beside it.
|
|
95
|
-
- No hedging ("might", "could consider"). No TBD, TODO, or placeholders: an open item is
|
|
96
|
-
|
|
95
|
+
- No hedging ("might", "could consider"). No TBD, TODO, or placeholders: an open item is an ask
|
|
96
|
+
block, a technical decision your lane makes and records as a Requirement, or, for a contract
|
|
97
|
+
between two lanes or a halt condition, a question for the platform PO (see
|
|
97
98
|
[Before you ask](#before-you-ask) under Asking).
|
|
98
99
|
- Keep each section to one screen; work that exceeds one screen per section is two specs.
|
|
99
100
|
- Update the spec as decisions land: the spec is the record, comments are the discussion.
|
|
@@ -281,9 +282,13 @@ eval_id? What's an R4 model header? What exactly is the question or uncertainty
|
|
|
281
282
|
production import). Every `dispatch_ask` passes three gates first:
|
|
282
283
|
|
|
283
284
|
1. **Does it need his authority, taste, or risk appetite?** This is the bar for a decision
|
|
284
|
-
written as an `:::ask` block in context ([Decision blocks](#decision-blocks)).
|
|
285
|
-
|
|
286
|
-
the
|
|
285
|
+
written as an `:::ask` block in context ([Decision blocks](#decision-blocks)). Technical
|
|
286
|
+
decisions inside your outcome do not: schema shapes, table layouts, field names, and migration
|
|
287
|
+
internals are your lane's to decide and record in the spec. Two things still go to the
|
|
288
|
+
platform PO over Envoy: a contract between two lanes, and a halt condition (a change to IAM,
|
|
289
|
+
deletion or exposure of production data, anything that reaches a customer). The PO takes those
|
|
290
|
+
to Sami as a Dispatch ask; you do not open one yourself, even as a permission ask under gate 2
|
|
291
|
+
(Sami, 2026-09-25, AGENTC-34 §12).
|
|
287
292
|
2. **Is there genuine uncertainty?** If not, it is a plan you execute. The one legitimate ask
|
|
288
293
|
without uncertainty is permission for an action only a human can authorise — a production
|
|
289
294
|
write, an external send, a console action — and then the question is that action in one
|
|
@@ -381,11 +386,13 @@ It returns the imported commit, or the recorded error when the model was rejecte
|
|
|
381
386
|
|
|
382
387
|
Before saying you are waiting for human input, call `dispatch_open_asks`. With no arguments it lists this session's active asks across open issues and project documents, including whether the human or agent owes the next reply. With `dispatch_open_asks({ project })` it lists every open ask in that project — on its issues and on its documents, whoever authored them — which is how you audit what a whole project is waiting on rather than just your own asks.
|
|
383
388
|
|
|
384
|
-
**Unsettled product shape needs a decision before implementation.** When a page, navigation entry, table key, customer-scoping rule, or persisted sidecar would set product shape that Sami has not already settled, send a one-line ask before the first implementation commit. A
|
|
389
|
+
**Unsettled product shape needs a decision before implementation.** When a page, navigation entry, table key, customer-scoping rule, or persisted sidecar would set product shape that Sami has not already settled, send a one-line ask before the first implementation commit. A lane's schema decision or a platform-PO contract ruling does not settle product shape. This does not turn a user-specified decision or routine implementation into an approval request; it is inferred from AGENTC-186's 2026-09-16 retro (platform PO, 2026-09-17).
|
|
385
390
|
|
|
386
391
|
**Anything that needs the human is an ask, or it does not exist.** An approval, a credential,
|
|
387
392
|
a setting only they can change, a review click, a conflict between two of their own rules - if
|
|
388
|
-
your work waits on it, open a `dispatch_ask` the moment you know, the action as the question.
|
|
393
|
+
your work waits on it, open a `dispatch_ask` the moment you know, the action as the question. The
|
|
394
|
+
exception is a halt condition from [Before you ask](#before-you-ask) gate 1, which goes to the
|
|
395
|
+
platform PO over Envoy instead.
|
|
389
396
|
Never write it into a spec, a comment reply, a message, or a
|
|
390
397
|
pull-request body: nothing in those paths reaches the human's Inbox, and a human who is not
|
|
391
398
|
reading your document does not know they are the blocker. Before asking, try to remove the
|
|
@@ -606,6 +613,20 @@ survive moves and retyping.
|
|
|
606
613
|
block id, keeps a typed block's body, and uses `attributes` for client-owned typed attributes. Use it when
|
|
607
614
|
an existing paragraph is the question that should become a decision.
|
|
608
615
|
|
|
616
|
+
### A document that is reloading
|
|
617
|
+
|
|
618
|
+
These calls can answer `DOC_SERVICE_UNAVAILABLE` (HTTP 503), because each writes a document inside its
|
|
619
|
+
transaction: `dispatch_doc_edit`; `dispatch_ask` and `dispatch_comment` on a quote; a `dispatch_comment` reply
|
|
620
|
+
in a thread whose first comment is anchored; `dispatch_suggest`; `dispatch_resolve_comment` on an anchored
|
|
621
|
+
comment; `dispatch_artifact` replacing a document that already exists; and, on an ask that lives in a `:::ask`
|
|
622
|
+
block, `dispatch_edit_ask` and `dispatch_resolve_ask`, which write that block. Creating an issue with a spec,
|
|
623
|
+
uploading a new document, `dispatch_request_approval` and `dispatch_message` never answer it, and neither do
|
|
624
|
+
`dispatch_edit_ask` and `dispatch_resolve_ask` on an ask that has no block. It means that document's live room
|
|
625
|
+
failed and is reloading from its durable copy, so the server refused rather than wait for it; your call wrote
|
|
626
|
+
nothing and the document is intact. Nothing retries it for you: the Dispatch client hands a 503 straight back.
|
|
627
|
+
Wait a few seconds and make the same call again. A second refusal in a row is worth telling your human about,
|
|
628
|
+
with the document's reference.
|
|
629
|
+
|
|
609
630
|
## Typed blocks
|
|
610
631
|
|
|
611
632
|
The server declares typed document blocks at `GET /api/v1/schema/blocks`. Write one only with the
|
|
@@ -80,7 +80,7 @@ exercise a criterion end to end, building that path is a child issue of this tre
|
|
|
80
80
|
answers the tree's asks); under a personal token it goes to that token's owner, whom
|
|
81
81
|
`dispatch_whoami` names. Set it only when a human told you a specific person owns that child.
|
|
82
82
|
|
|
83
|
-
Specifications written into Dispatch follow
|
|
83
|
+
Specifications written into Dispatch follow `skill://dispatch`'s [Writing a spec](../dispatch/SKILL.md#writing-a-spec).
|
|
84
84
|
Wave releases, child closures, and your own status are visible from the issue tree and the
|
|
85
85
|
handoffs; do not narrate them into the spec or a `dispatch_message`. A blocker only Sami can
|
|
86
86
|
clear is a `dispatch_ask`.
|
|
@@ -213,7 +213,7 @@ legion({
|
|
|
213
213
|
op: "spawn_worker",
|
|
214
214
|
issue: "LEGION-40",
|
|
215
215
|
role: "implementer",
|
|
216
|
-
task: "
|
|
216
|
+
task: "Load skill://legion-retro and run it now. Capture durable learnings and post the retro message on the Dispatch issue with dispatch_message; do not create a .legion handoff file."
|
|
217
217
|
})
|
|
218
218
|
```
|
|
219
219
|
|
|
@@ -233,7 +233,7 @@ Preserve this order exactly:
|
|
|
233
233
|
|
|
234
234
|
1. tester green and review cycles complete;
|
|
235
235
|
2. on a clean review, `spawn_worker` the implementer once more to push only the `.legion/`
|
|
236
|
-
deletion (the
|
|
236
|
+
deletion (only the implementer pushes the issue branch), then the reviewer approves that
|
|
237
237
|
head. The deletion must land before that approval, which is head-pinned. An implementer
|
|
238
238
|
completion advances the status only from `in_progress` to `testing`; this push, like retro
|
|
239
239
|
later, leaves the status where it is, so you set nothing by hand — on its `phase-complete`
|
|
@@ -256,8 +256,8 @@ Preserve this order exactly:
|
|
|
256
256
|
with options for its outcomes, and the issue waits for it.
|
|
257
257
|
|
|
258
258
|
What returns the tree to review: a changed diff — a commit above the approved head that
|
|
259
|
-
touches anything outside `docs/solutions/`, or a rebase whose fingerprint
|
|
260
|
-
skill's unchanged-diff check) differs from the approved head's. What does not: retro's
|
|
259
|
+
touches anything outside `docs/solutions/`, or a rebase whose fingerprint
|
|
260
|
+
(`skill://legion-worker`'s unchanged-diff check) differs from the approved head's. What does not: retro's
|
|
261
261
|
`docs/solutions/` commit, and a rebase forced by a GitHub-reported conflict whose fingerprint
|
|
262
262
|
is unchanged. For that rebase the order is: the implementer rebases and posts the before/after
|
|
263
263
|
fingerprints; the tester re-runs the bare gates only; the reviewer confirms and approves the new
|
|
@@ -99,7 +99,7 @@ quoted here.
|
|
|
99
99
|
- **Controller state is disposable.** Do not reconstruct or preserve local controller
|
|
100
100
|
bookkeeping between turns.
|
|
101
101
|
- **Write for a human.** Every `dispatch_comment`, `dispatch_message`, and `dispatch_ask` you
|
|
102
|
-
post follows
|
|
102
|
+
post follows `skill://dispatch`'s "Writing for the human" rules: plain sentences, every
|
|
103
103
|
identifier expanded on first use, no coined shorthand. A triage note that reads like a log
|
|
104
104
|
line is not a triage note.
|
|
105
105
|
|
|
@@ -143,7 +143,7 @@ other tree paused.
|
|
|
143
143
|
|
|
144
144
|
## Phase work
|
|
145
145
|
|
|
146
|
-
Specifications written into Dispatch follow
|
|
146
|
+
Specifications written into Dispatch follow `skill://dispatch`'s [Writing a spec](../dispatch/SKILL.md#writing-a-spec).
|
|
147
147
|
|
|
148
148
|
Follow the repository's normal engineering workflow and the assigned issue's acceptance
|
|
149
149
|
criteria. Your phase's own charter and the predecessor handoffs you read define the phase
|
|
@@ -207,7 +207,7 @@ Append this exact structured footer to **every** pull-request comment and review
|
|
|
207
207
|
phase posts on GitHub. It preserves session provenance on the artifact itself so work stays
|
|
208
208
|
attributable to the session that produced it. Dispatch comments carry session provenance
|
|
209
209
|
natively through their own `actor`/`origin` fields; this footer is for GitHub PR artifacts and
|
|
210
|
-
for the retro's Dispatch message (`
|
|
210
|
+
for the retro's Dispatch message (`skill://legion-retro`):
|
|
211
211
|
|
|
212
212
|
```html
|
|
213
213
|
<!-- legion: {"session":"<session-id>","phase":"<phase>"} -->
|
|
@@ -314,10 +314,9 @@ this proof.
|
|
|
314
314
|
`Accepted: not a defect — <reason>`, or `Still open: <what remains>`; nothing else is an
|
|
315
315
|
acceptance, and nobody replies after an `Accepted:` (any later reply that is not itself an
|
|
316
316
|
`Accepted:` — the opener's own follow-up included — leaves the thread open, since the command
|
|
317
|
-
reads only the newest comment). The review App can reply on a thread but
|
|
318
|
-
|
|
319
|
-
|
|
320
|
-
(`packages/daemon/src/daemon/AGENTS.md`, GitHub Apps) — so the
|
|
317
|
+
reads only the newest comment). The review App can reply on a thread, but GitHub refuses it
|
|
318
|
+
`resolveReviewThread` (its token reads `viewerCanResolve: false`), and the workflow gives
|
|
319
|
+
pushing to the implementer alone (`packages/daemon/src/daemon/AGENTS.md`, GitHub Apps) — so the
|
|
321
320
|
**implementer** runs `legion threads resolve --pr <number> --repo <owner>/<repo>` before every
|
|
322
321
|
push that answers a review (the corrective push and the final `.legion/` deletion push) and
|
|
323
322
|
pastes its output into the `Threads` section. The command resolves each unresolved thread
|
|
@@ -372,7 +371,7 @@ this proof.
|
|
|
372
371
|
`E2E` line's head to the new SHA with
|
|
373
372
|
`rebase re-check <old-sha> → <new-sha>: fingerprint unchanged, bare gates only`; the
|
|
374
373
|
real-surface verification is not repeated. Different: a full test round.
|
|
375
|
-
- **The implementer runs `ce-simplify-code` once per pull request, after the last review round
|
|
374
|
+
- **The implementer runs `skill://ce-simplify-code` once per pull request, after the last review round
|
|
376
375
|
closes and before the reviewer's final pass, when the diff touches runtime code; a docs-only
|
|
377
376
|
diff gets none.** It is scoped to the pull request's own diff, at the head where the last review
|
|
378
377
|
round closed: nothing applied leaves that head final; applied → the applied head is the final
|
|
@@ -400,7 +399,7 @@ this proof.
|
|
|
400
399
|
Legion footer), and a `comments[]` array of `{path, line, side, body}`, one entry per
|
|
401
400
|
finding — never one `pr review` call per finding (each submission fires a `pr-review` wake).
|
|
402
401
|
Then return the issue to the architect; when clean, have the architect send the implementer
|
|
403
|
-
back to push the `.legion/` deletion (the
|
|
402
|
+
back to push the `.legion/` deletion (only the implementer pushes the issue branch), then review **that** head
|
|
404
403
|
and approve it by name. After a conflict-forced rebase, compute the fingerprint at the
|
|
405
404
|
`commit_id` of your last submitted review and at the new head. Equal and that review was
|
|
406
405
|
`APPROVE`: submit one more `APPROVE` naming the new head by SHA, its body naming both SHAs
|
|
@@ -538,7 +537,8 @@ cd -- "$LEGION_WORKSPACE" && \
|
|
|
538
537
|
|
|
539
538
|
**Only the implementer pushes the issue branch.** It acts as the code-writing App
|
|
540
539
|
(`legion-implementer[bot]`, `appRoleForLegionRole` in `packages/daemon/src/daemon/github-apps.ts`),
|
|
541
|
-
the one
|
|
540
|
+
the one role the workflow lets push (the review App's installation holds `contents: write` too, but
|
|
541
|
+
no role acting as it pushes; the merger acts as the implement App but pushes nothing: it
|
|
542
542
|
verifies and publishes READY). If you are the implementer, advance the issue bookmark and push it
|
|
543
543
|
with the provisioned credential helper. `--bookmark` also publishes the locally provisioned
|
|
544
544
|
bookmark on its first push — a bookmark not yet tracking a remote one is tracked automatically:
|
|
@@ -550,12 +550,11 @@ cd -- "$LEGION_WORKSPACE" && \
|
|
|
550
550
|
```
|
|
551
551
|
|
|
552
552
|
Every other role — planner, tester, reviewer, architects — acts as the review App
|
|
553
|
-
(`legion-reviewer[bot]`)
|
|
553
|
+
(`legion-reviewer[bot]`) and never pushes: the `split` above is your last step, and the commit
|
|
554
554
|
rides the implementer's next push (the corrective push after a review, or the final `.legion/`
|
|
555
|
-
deletion).
|
|
556
|
-
`
|
|
557
|
-
|
|
558
|
-
or retry.
|
|
555
|
+
deletion). GitHub does not stop a push from one of those roles: the review App's installation
|
|
556
|
+
holds `contents: write`, so the push would succeed. The rule is the workflow's, and nothing but
|
|
557
|
+
the rule enforces it.
|
|
559
558
|
|
|
560
559
|
Do not report phase completion until the write, existence check, and handoff commit succeed —
|
|
561
560
|
and, for the implementer, until the push has too. This is the committed copy the next phase
|
|
@@ -0,0 +1,192 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: thermonuclear-code-quality
|
|
3
|
+
description: Run a strict maintainability review for abstraction quality, oversized files, and ad-hoc branching growth. Use for a thermonuclear code-quality review or a deep maintainability audit.
|
|
4
|
+
disable-model-invocation: true
|
|
5
|
+
---
|
|
6
|
+
|
|
7
|
+
# Thermonuclear Code Quality
|
|
8
|
+
|
|
9
|
+
Use this skill for an unusually strict review focused on implementation quality, maintainability, abstraction quality, and codebase health.
|
|
10
|
+
|
|
11
|
+
Look for structural simplifications that preserve behavior while making the implementation smaller, more direct, and easier to maintain. Do not stop at local cleanup when a coherent simplification is available.
|
|
12
|
+
|
|
13
|
+
## Core Prompt
|
|
14
|
+
|
|
15
|
+
Start from this baseline:
|
|
16
|
+
|
|
17
|
+
> Perform a deep code quality audit of the current branch's changes.
|
|
18
|
+
> Rethink how to structure / implement the changes to meaningfully improve code quality without impacting behavior.
|
|
19
|
+
> Work to improve abstractions, modularity, reduce Spaghetti code, improve succinctness and legibility.
|
|
20
|
+
> Restructure when there is a clear path to a simpler implementation.
|
|
21
|
+
> Be thorough and rigorous. Verify claims before reporting them.
|
|
22
|
+
|
|
23
|
+
## Non-Negotiable Additional Standards
|
|
24
|
+
|
|
25
|
+
Apply the baseline prompt above, plus these explicit review rules:
|
|
26
|
+
|
|
27
|
+
0. **Be ambitious about structural simplification.**
|
|
28
|
+
- Do not stop at "this could be a bit cleaner."
|
|
29
|
+
- Look for opportunities to reframe the change so that whole branches, helpers, modes, conditionals, or layers disappear entirely.
|
|
30
|
+
- Prefer the solution that makes the code feel inevitable in hindsight.
|
|
31
|
+
- Assume there is often a "code judo" move available: a re-organization that uses the existing architecture more effectively and makes the change dramatically simpler and more elegant.
|
|
32
|
+
- If you see a path to delete complexity rather than rearrange it, push hard for that path.
|
|
33
|
+
|
|
34
|
+
1. **Do not let a PR push a file from under 1k lines to over 1k lines without a very strong reason.**
|
|
35
|
+
- Treat this as a strong code-quality smell by default.
|
|
36
|
+
- Prefer extracting helpers, subcomponents, modules, or local abstractions instead of letting a file sprawl past 1000 lines.
|
|
37
|
+
- If the diff crosses that threshold, explicitly ask whether the code should be decomposed first.
|
|
38
|
+
- Depart from this only if there is a compelling structural reason and the resulting file remains well organized.
|
|
39
|
+
|
|
40
|
+
2. **Do not allow random spaghetti growth in existing code.**
|
|
41
|
+
- Be highly suspicious of new ad-hoc conditionals, scattered special cases, or one-off branches inserted into unrelated flows.
|
|
42
|
+
- If a change adds "weird if statements in random places", treat that as a design problem, not a stylistic nit.
|
|
43
|
+
- Prefer pushing the logic into a dedicated abstraction, helper, state machine, policy object, or separate module instead of tangling an existing path.
|
|
44
|
+
- Call out changes that make the surrounding code harder to reason about, even if they technically work.
|
|
45
|
+
|
|
46
|
+
3. **Bias toward cleaning the design, not just accepting working code.**
|
|
47
|
+
- If behavior can stay the same while the structure becomes meaningfully cleaner, push for the cleaner version.
|
|
48
|
+
- Do not rubber-stamp "it works" implementations that leave the codebase messier.
|
|
49
|
+
- Strongly prefer simplifications that remove moving pieces altogether over refactors that merely spread the same complexity around.
|
|
50
|
+
|
|
51
|
+
4. **Prefer direct, boring, maintainable code over hacky or magical code.**
|
|
52
|
+
- Treat brittle, ad-hoc, or "magic" behavior as a code-quality problem.
|
|
53
|
+
- Be skeptical of generic mechanisms that hide simple data-shape assumptions.
|
|
54
|
+
- Flag thin abstractions, identity wrappers, or pass-through helpers that add indirection without buying clarity.
|
|
55
|
+
|
|
56
|
+
5. **Push hard on type and boundary cleanliness when they affect maintainability.**
|
|
57
|
+
- Question unnecessary optionality, `unknown`, `any`, or cast-heavy code when a clearer type boundary could exist.
|
|
58
|
+
- Prefer explicit typed models or shared contracts over loosely-shaped ad-hoc objects.
|
|
59
|
+
- If a branch relies on silent fallback to paper over an unclear invariant, ask whether the boundary should be made explicit instead.
|
|
60
|
+
|
|
61
|
+
6. **Keep logic in the canonical layer and reuse existing helpers.**
|
|
62
|
+
- Call out feature logic leaking into shared paths or implementation details leaking through APIs.
|
|
63
|
+
- Prefer existing canonical utilities/helpers over bespoke one-offs.
|
|
64
|
+
- Push code toward the right package, service, or module instead of normalizing architectural drift.
|
|
65
|
+
|
|
66
|
+
7. **Treat unnecessary sequential orchestration and non-atomic updates as design smells when the cleaner structure is obvious.**
|
|
67
|
+
- If independent work is serialized for no good reason, ask whether the flow should run in parallel instead.
|
|
68
|
+
- If related updates can leave state half-applied, push for a more atomic structure.
|
|
69
|
+
- Do not over-index on micro-optimizations, but do flag avoidable orchestration complexity that makes the implementation more brittle.
|
|
70
|
+
|
|
71
|
+
## Primary Review Questions
|
|
72
|
+
|
|
73
|
+
For every meaningful change, ask:
|
|
74
|
+
|
|
75
|
+
- Is there a "code judo" move that would make this dramatically simpler?
|
|
76
|
+
- Can this change be reframed so fewer concepts, branches, or helper layers are needed?
|
|
77
|
+
- Does this improve or worsen the local architecture?
|
|
78
|
+
- Did the diff add branching complexity where a better abstraction should exist?
|
|
79
|
+
- Did a previously cohesive module become more coupled, more stateful, or harder to scan?
|
|
80
|
+
- Is this logic living in the right file and layer?
|
|
81
|
+
- Did this change enlarge a file or component past a healthy size boundary?
|
|
82
|
+
- Are there repeated conditionals that signal a missing model or missing helper?
|
|
83
|
+
- Is the implementation direct and legible, or does it rely on special cases and incidental control flow?
|
|
84
|
+
- Is this abstraction actually earning its keep, or is it just a wrapper?
|
|
85
|
+
- Did the diff introduce casts, optionality, or ad-hoc object shapes that obscure the real invariant?
|
|
86
|
+
- Is this logic living in the canonical layer, or did the diff leak details across a boundary?
|
|
87
|
+
- Is this orchestration more sequential or less atomic than it needs to be?
|
|
88
|
+
|
|
89
|
+
## What to Flag Aggressively
|
|
90
|
+
|
|
91
|
+
Escalate findings when you see:
|
|
92
|
+
|
|
93
|
+
- A complicated implementation where a cleaner reframing could delete whole categories of complexity.
|
|
94
|
+
- Refactors that move code around but fail to reduce the number of concepts a reader must hold in their head.
|
|
95
|
+
- A file crossing 1000 lines due to the PR, especially if the new code could be split out.
|
|
96
|
+
- New conditionals bolted onto unrelated code paths.
|
|
97
|
+
- One-off booleans, nullable modes, or flags that complicate existing control flow.
|
|
98
|
+
- Feature-specific logic leaking into general-purpose modules.
|
|
99
|
+
- Generic "magic" handling that hides simple structure and makes the code harder to reason about.
|
|
100
|
+
- Thin wrappers or identity abstractions that add indirection without simplifying anything.
|
|
101
|
+
- Unnecessary casts, `any`, `unknown`, or optional params that muddy the real contract.
|
|
102
|
+
- Copy-pasted logic instead of extracted helpers.
|
|
103
|
+
- Narrow edge-case handling implemented in the middle of an already busy function.
|
|
104
|
+
- Refactors that technically pass tests but make the code less modular or less readable.
|
|
105
|
+
- "Temporary" branching that is likely to become permanent debt.
|
|
106
|
+
- Bespoke helpers where the codebase already has a canonical utility for the job.
|
|
107
|
+
- Logic added in the wrong layer/package when it should live somewhere more central.
|
|
108
|
+
- Sequential async flow where independent work could use simpler parallel execution.
|
|
109
|
+
- Partial-update logic that leaves state less atomic than necessary.
|
|
110
|
+
|
|
111
|
+
## Preferred Remedies
|
|
112
|
+
|
|
113
|
+
When you identify a code-quality problem, prefer suggestions like:
|
|
114
|
+
|
|
115
|
+
- Delete a whole layer of indirection rather than polishing it.
|
|
116
|
+
- Reframe the state model so conditionals disappear instead of getting centralized.
|
|
117
|
+
- Change the ownership boundary so the feature becomes a natural extension of an existing abstraction.
|
|
118
|
+
- Turn special-case logic into a simpler default flow with fewer exceptions.
|
|
119
|
+
- Extract a helper or pure function.
|
|
120
|
+
- Split a large file into smaller focused modules.
|
|
121
|
+
- Move feature-specific logic behind a dedicated abstraction.
|
|
122
|
+
- Replace condition chains with a typed model or explicit dispatcher.
|
|
123
|
+
- Separate orchestration from business logic.
|
|
124
|
+
- Collapse duplicate branches into a single clearer flow.
|
|
125
|
+
- Delete wrappers that do not meaningfully clarify the API.
|
|
126
|
+
- Reuse the existing canonical helper instead of introducing a near-duplicate.
|
|
127
|
+
- Make type boundaries more explicit so the control flow gets simpler.
|
|
128
|
+
- Move the logic to the package/module/layer that already owns the concept.
|
|
129
|
+
- Parallelize independent work when that also simplifies the orchestration.
|
|
130
|
+
- Restructure related updates into a more atomic flow when partial state would be harder to reason about.
|
|
131
|
+
|
|
132
|
+
Do not be satisfied with "maybe rename this" feedback when the real issue is structural.
|
|
133
|
+
Do not be satisfied with a merely cleaner version of the same messy idea if there is a plausible path to a much simpler idea.
|
|
134
|
+
|
|
135
|
+
## Review Tone
|
|
136
|
+
|
|
137
|
+
Be direct, serious, and demanding about quality.
|
|
138
|
+
Do not be rude, but do not soften major maintainability issues into mild suggestions.
|
|
139
|
+
If the code makes the codebase messier, say so.
|
|
140
|
+
If the implementation missed a substantial simplification, state it.
|
|
141
|
+
|
|
142
|
+
Good phrases:
|
|
143
|
+
|
|
144
|
+
- `this pushes the file past 1k lines. can we decompose this first?`
|
|
145
|
+
- `this adds another special-case branch into an already busy flow. can we move this behind its own abstraction?`
|
|
146
|
+
- `this works, but it makes the surrounding code more spaghetti. let's keep the behavior and restructure the implementation.`
|
|
147
|
+
- `this feels like feature logic leaking into a shared path. can we isolate it?`
|
|
148
|
+
- `this abstraction seems unnecessary. can we just keep the direct flow?`
|
|
149
|
+
- `why does this need a cast / optional here? can we make the boundary more explicit instead?`
|
|
150
|
+
- `this looks like a bespoke helper for something we already have elsewhere. can we reuse the canonical one?`
|
|
151
|
+
- `i think there's a code-judo move here that makes this much simpler. can we reframe this so these branches disappear?`
|
|
152
|
+
- `this refactor moves complexity around, but doesn't really delete it. is there a way to make the model itself simpler?`
|
|
153
|
+
|
|
154
|
+
## Output Expectations
|
|
155
|
+
|
|
156
|
+
Prioritize findings in this order:
|
|
157
|
+
|
|
158
|
+
1. Structural code-quality regressions
|
|
159
|
+
2. Missed opportunities for dramatic simplification / code-judo restructuring
|
|
160
|
+
3. Spaghetti / branching complexity increases
|
|
161
|
+
4. Boundary / abstraction / type-contract problems that make the code harder to reason about
|
|
162
|
+
5. File-size and decomposition concerns
|
|
163
|
+
6. Modularity and abstraction issues
|
|
164
|
+
7. Legibility and maintainability concerns
|
|
165
|
+
|
|
166
|
+
Do not flood the review with low-value nits if there are larger structural issues.
|
|
167
|
+
Prefer a smaller number of high-conviction comments over a long list of cosmetic notes.
|
|
168
|
+
|
|
169
|
+
## Approval Bar
|
|
170
|
+
|
|
171
|
+
Do not approve merely because behavior seems correct.
|
|
172
|
+
The bar for approval is:
|
|
173
|
+
|
|
174
|
+
- no clear structural regression
|
|
175
|
+
- no obvious missed opportunity to make the implementation dramatically simpler when such a path is visible
|
|
176
|
+
- no unjustified file-size explosion
|
|
177
|
+
- no obvious spaghetti-growth from special-case branching
|
|
178
|
+
- no hacky or magical abstraction that makes the code harder to reason about
|
|
179
|
+
- no unnecessary wrapper/cast/optionality churn obscuring the real design
|
|
180
|
+
- no clear architecture-boundary leak or avoidable canonical-helper duplication
|
|
181
|
+
- no missed opportunity for an obvious decomposition that would materially improve maintainability
|
|
182
|
+
|
|
183
|
+
Treat these as presumptive blockers unless the author can justify them with evidence:
|
|
184
|
+
|
|
185
|
+
- the PR preserves a lot of incidental complexity when there is a plausible code-judo move that would delete it
|
|
186
|
+
- the PR pushes a file from below 1000 lines to above 1000 lines
|
|
187
|
+
- the PR adds ad-hoc branching that makes an existing flow more tangled
|
|
188
|
+
- the PR solves a local problem by scattering feature checks across shared code
|
|
189
|
+
- the PR adds an unnecessary abstraction, wrapper, or cast-heavy contract that makes the design more indirect
|
|
190
|
+
- the PR duplicates an existing helper or puts logic in the wrong layer when there is a clear canonical home
|
|
191
|
+
|
|
192
|
+
If those conditions are not met, leave explicit, actionable feedback and push for a cleaner decomposition.
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: thermonuclear-deep-review
|
|
3
|
+
description: Comprehensive security and correctness audit of a branch's changes. Use for thermonuclear or deep-review requests, or branch and PR diff audits focused on bugs, breaking changes, security issues, developer-experience regressions, and feature-gate leaks.
|
|
4
|
+
disable-model-invocation: true
|
|
5
|
+
---
|
|
6
|
+
|
|
7
|
+
# Thermonuclear Deep Review
|
|
8
|
+
|
|
9
|
+
Use this skill for a comprehensive security and correctness audit of a checked-out branch.
|
|
10
|
+
|
|
11
|
+
## Prompt
|
|
12
|
+
|
|
13
|
+
You are a security reviewer performing a comprehensive review of a checked-out branch. Audit its changes for bugs, regressions, and security vulnerabilities. Be rigorous, careful, and evidence-led.
|
|
14
|
+
|
|
15
|
+
# Scope
|
|
16
|
+
ONLY report issues related to code that is being ADDED or MODIFIED in this PR.
|
|
17
|
+
Focus on changes in the diff.
|
|
18
|
+
DO NOT report vulnerabilities in existing code that is not being changed.
|
|
19
|
+
EXCEPT where that unchanged code CONSUMES behavior the diff alters — see Downstream Consumer
|
|
20
|
+
Guidelines. An unchanged consumer that silently breaks IS a defect in this diff, and it will
|
|
21
|
+
never appear in the diff itself.
|
|
22
|
+
|
|
23
|
+
# Guidelines
|
|
24
|
+
|
|
25
|
+
## Breaking Functionality Guidelines
|
|
26
|
+
This is a complex codebase with many cross-package and module dependencies. Trace possible side effects of the changes through their callers and contracts.
|
|
27
|
+
|
|
28
|
+
## Downstream Consumer Guidelines
|
|
29
|
+
A diff cannot show you what depends on the behavior it changes, and "trace the callers" does not
|
|
30
|
+
cover it: nothing *calls* a log line, but a monitor predicate consumes it. For EVERY behavior the
|
|
31
|
+
diff alters — a log level, a status code, an exception type, a metric or field name, a payload
|
|
32
|
+
shape, an identity or uniqueness key, a timing/retry characteristic, a default — name its
|
|
33
|
+
downstream consumers and show each one still works. Consumers routinely live outside the diff,
|
|
34
|
+
outside the package, and outside the repository:
|
|
35
|
+
- alert/monitor predicates (a monitor keyed on `status:error` silently stops firing when a line is
|
|
36
|
+
downgraded to WARNING; the monitor appears nowhere in the diff that disables it)
|
|
37
|
+
- log-level contracts, structured-log field names, log-derived metric filters
|
|
38
|
+
- status codes and error-type strings written to request logs, webhooks, or audit records
|
|
39
|
+
- identity/uniqueness keys that a persistence layer arbitrates on (a fresh id for an existing
|
|
40
|
+
logical row collides against a constraint the diff never mentions)
|
|
41
|
+
- dashboards, SLOs, saved queries, downstream parsers
|
|
42
|
+
If you cannot locate a consumer, say so explicitly rather than assuming none exists.
|
|
43
|
+
A change that removes its own alarm is the highest-severity finding of this class, because it also
|
|
44
|
+
removes the signal that would have caught it.
|
|
45
|
+
|
|
46
|
+
## Breaking Devex Guidelines
|
|
47
|
+
It can be easy to break developers' ability to run / build the code locally. You MUST catch changes that will impact users' developer experience. Some examples (not exhaustive):
|
|
48
|
+
- Modifying how secrets are read / where they are read from
|
|
49
|
+
- Updating environment variable names / adding environment variables
|
|
50
|
+
- Remapping ports / networking
|
|
51
|
+
- Adding scripts that must be run for certain functionality to continue working. Broadly speaking these are changes that will modify the way developers currently run / build the code. This does not include changes that introduce new alternative ways to run/build things. Adding dependencies with package managers does not count as a devex breaking change, unless it requires the user to do some very new thing that is not part of their normal development workflow, like manually installing software off of a website / App Store.
|
|
52
|
+
|
|
53
|
+
## Feature Leak Guidelines
|
|
54
|
+
The codebase might gate features behind feature flags or internal-only checks. Do not allow a gated feature to leak. These leaks can be subtle, so trace the relevant gate and its callers.
|
|
55
|
+
|
|
56
|
+
## Intended Breakage Guidelines
|
|
57
|
+
If a high-risk effect is an intentional, well-constrained change, do not report it as a defect. Report it when the scope or consequences appear unclear, including when a safeguard or feature gate is removed.
|
|
58
|
+
|
|
59
|
+
## Over-reporting Guidelines
|
|
60
|
+
If you report issues as High priority when they are not in fact high priority / meaningful issues, devs will lose trust in you and stop listening to you over time.
|
|
61
|
+
Never misreport priority or importance. Trace issues end to end and report only what the evidence supports.
|
|
62
|
+
|
|
63
|
+
# Final Response
|
|
64
|
+
IF you have medium-to-high priority / risk findings, and there is a PR for this branch, then check the PR/MR discussion using gh/glab cli to see if there are comments from BugBot or others present.
|
|
65
|
+
If so, take their findings into account. If they found issues you missed, evaluate them to determine if they are valid and include them in your report. If they found some of the same issues you did, see if there is anything from their findings that are worth incorporating into your response.
|
|
66
|
+
Flag issues found by BugBot or others in the PR/MR discussion that you include in your report.
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
# Critical Rules
|
|
70
|
+
- NEVER present issues with unfinished research. E.g. Never say something like, "The client has issue X, but if handled in the backend then this is ok." if you have access to the backend code and can check for yourself.
|
|
71
|
+
- Wait to check PR discussion until after the independent audit so fresh evidence drives the review.
|
|
72
|
+
- Be rigorous, careful, and specific about what the evidence supports.
|