@iceinvein/agent-skills 0.1.37 → 0.1.39

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (79) hide show
  1. package/package.json +1 -1
  2. package/skills/bounded-context-auditor/SKILL.md +15 -3
  3. package/skills/bounded-context-auditor/skill.json +1 -1
  4. package/skills/codebase-architecture/SKILL.md +4 -4
  5. package/skills/codebase-architecture/skill.json +1 -1
  6. package/skills/cognitive-load-auditor/SKILL.md +11 -9
  7. package/skills/cognitive-load-auditor/skill.json +1 -1
  8. package/skills/cohesion-analyzer/SKILL.md +1 -1
  9. package/skills/cohesion-analyzer/skill.json +1 -1
  10. package/skills/composability-auditor/SKILL.md +5 -5
  11. package/skills/composability-auditor/skill.json +1 -1
  12. package/skills/contract-enforcer/SKILL.md +5 -5
  13. package/skills/contract-enforcer/skill.json +1 -1
  14. package/skills/coupling-auditor/SKILL.md +2 -2
  15. package/skills/coupling-auditor/skill.json +1 -1
  16. package/skills/cover-letter/SKILL.md +18 -20
  17. package/skills/cover-letter/skill.json +7 -2
  18. package/skills/cover-letter-audit/SKILL.md +20 -20
  19. package/skills/cover-letter-audit/skill.json +7 -2
  20. package/skills/cover-letter-persona/SKILL.md +13 -13
  21. package/skills/cover-letter-persona/skill.json +7 -2
  22. package/skills/cover-letter-rewrite/SKILL.md +18 -16
  23. package/skills/cover-letter-rewrite/skill.json +7 -2
  24. package/skills/cover-letter-write/SKILL.md +25 -20
  25. package/skills/cover-letter-write/skill.json +7 -2
  26. package/skills/cqs-auditor/SKILL.md +19 -47
  27. package/skills/cqs-auditor/skill.json +1 -1
  28. package/skills/demeter-enforcer/SKILL.md +5 -5
  29. package/skills/demeter-enforcer/skill.json +1 -1
  30. package/skills/dependency-direction-auditor/SKILL.md +1 -1
  31. package/skills/dependency-direction-auditor/skill.json +1 -1
  32. package/skills/design-review/SKILL.md +6 -2
  33. package/skills/design-review/skill.json +1 -1
  34. package/skills/error-strategist/SKILL.md +3 -3
  35. package/skills/error-strategist/skill.json +1 -1
  36. package/skills/event-design-reviewer/SKILL.md +3 -3
  37. package/skills/event-design-reviewer/skill.json +1 -1
  38. package/skills/evolution-analyzer/SKILL.md +4 -3
  39. package/skills/evolution-analyzer/skill.json +1 -1
  40. package/skills/gestalt-reviewer/SKILL.md +8 -4
  41. package/skills/gestalt-reviewer/skill.json +1 -1
  42. package/skills/idempotency-guardian/SKILL.md +6 -6
  43. package/skills/idempotency-guardian/skill.json +1 -1
  44. package/skills/improve-my-codebase/CATALOGUE-FIELDS.md +2 -2
  45. package/skills/improve-my-codebase/SKILL.md +68 -27
  46. package/skills/improve-my-codebase/skill.json +1 -1
  47. package/skills/index.json +33 -33
  48. package/skills/integration-pattern-auditor/SKILL.md +2 -2
  49. package/skills/integration-pattern-auditor/skill.json +1 -1
  50. package/skills/magpie/README.md +3 -5
  51. package/skills/magpie/SKILL.md +39 -536
  52. package/skills/magpie/package.json +1 -1
  53. package/skills/magpie/references/critic.md +58 -0
  54. package/skills/magpie/references/peer-review.md +84 -0
  55. package/skills/magpie/references/specialists.md +391 -0
  56. package/skills/magpie/scripts/__tests__/helper.test.ts +40 -0
  57. package/skills/magpie/scripts/__tests__/skill-lint.test.ts +116 -28
  58. package/skills/magpie/scripts/__tests__/status-cmd.test.ts +13 -0
  59. package/skills/magpie/scripts/helper.js +24 -13
  60. package/skills/magpie/scripts/status-cmd.ts +10 -1
  61. package/skills/magpie/skill.json +2 -1
  62. package/skills/module-secret-auditor/SKILL.md +8 -5
  63. package/skills/module-secret-auditor/skill.json +1 -1
  64. package/skills/port-adapter-auditor/SKILL.md +3 -3
  65. package/skills/port-adapter-auditor/skill.json +1 -1
  66. package/skills/rams-design-audit/SKILL.md +4 -2
  67. package/skills/rams-design-audit/skill.json +1 -1
  68. package/skills/seam-finder/SKILL.md +2 -2
  69. package/skills/seam-finder/skill.json +1 -1
  70. package/skills/simplicity-razor/SKILL.md +4 -4
  71. package/skills/simplicity-razor/skill.json +1 -1
  72. package/skills/temporal-coupling-detector/SKILL.md +2 -2
  73. package/skills/temporal-coupling-detector/skill.json +1 -1
  74. package/skills/terse/SKILL.md +12 -7
  75. package/skills/terse/skill.json +1 -1
  76. package/skills/type-driven-designer/SKILL.md +8 -8
  77. package/skills/type-driven-designer/skill.json +1 -1
  78. package/skills/unidirectional-flow-enforcer/SKILL.md +2 -2
  79. package/skills/unidirectional-flow-enforcer/skill.json +1 -1
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "magpie",
3
- "version": "0.2.0",
3
+ "version": "0.8.0",
4
4
  "private": true,
5
5
  "type": "module",
6
6
  "scripts": {
@@ -0,0 +1,58 @@
1
+ # Critic rubric
2
+
3
+ The main agent runs this in-conversation against `findings.deduped.json` and writes the kept subset to `findings.kept.json`.
4
+
5
+ ## Substitute before use
6
+
7
+ The block below contains two placeholders. Replace both before running the rubric. (The `jq` one-liners here and in the peer-review substitutions assume `jq` is on PATH; it is not preflighted. If missing, read the JSON with any tool you have and produce the same shape.)
8
+
9
+ - `<<DEDUPED_FINDINGS_COMPACT>>` — pretty-printed JSON array of the deduped candidates with only the fields the critic needs. Each candidate carries `onChangedLine` (set deterministically during dedupe: `true` = anchored inside a changed hunk, `false` = anchored on code the PR did not touch, `null` = not anchorable). Build with:
10
+ ```
11
+ jq '[.[] | {id, file, line, onChangedLine, severity, risk, domain, title, description}]' "$RUN_DIR/findings.deduped.json"
12
+ ```
13
+ - `<<DIFF_EXCERPT>>` — the diff hunks for the files referenced by the candidates. For small PRs the full `diff.patch` is fine; for larger PRs, narrow to the files named in the candidate set.
14
+
15
+ ````magpie-critic
16
+ You are a senior code reviewer auditing a list of candidate review findings produced by other agents on a pull request. Your only job is to keep the findings that a busy reviewer would genuinely thank you for surfacing, and drop the rest. You see each candidate's claim, anchor, and risk fields, plus the diff hunks around them. Use the hunks only to validate or refute the candidate in front of you: do not surface new findings or broaden the review (adding issues is the peer-review stage's job). Treat each candidate skeptically.
17
+
18
+ Drop a finding if any of the following hold:
19
+ - The description sounds speculative, hedged, or "needs verification" without strong evidence in the title or anchor.
20
+ - The finding is a stylistic preference, micro-optimization, or "nice to have" cleanup with no concrete user or maintenance impact.
21
+ - Its `onChangedLine` is `false` and the description does not explain why the PR newly triggers a pre-existing concern (i.e. it is anchored on code this PR did not change).
22
+ - The finding is a theoretical risk that requires unlikely preconditions, or defense-in-depth on code the supplied hunks show is already guarded.
23
+ - The finding belongs to a category the repository's linter already enforces (naming, formatting, unused imports).
24
+ - The finding is on a test file or a generated/vendored file unless it materially affects test correctness.
25
+
26
+ Keep a finding if it points to a concrete defect on a changed line, with enough specificity that a reviewer could decide to act on it without re-reading the entire PR.
27
+
28
+ When in doubt, drop. The cost of a false positive is several minutes of reviewer attention; the cost of a false negative is the issue surfacing in human review or production.
29
+
30
+ For each candidate below, decide whether to keep it or drop it.
31
+
32
+ ## Output Contract
33
+
34
+ Output a JSON array inside a fenced code block tagged `review-critic`. Each entry must be:
35
+ - `id`: the candidate id (string, copied verbatim)
36
+ - `verdict`: "keep" or "drop"
37
+ - `reason`: one short sentence (under 18 words) explaining why
38
+
39
+ Output every candidate exactly once. Do not invent ids. Do not output anything outside the fenced block.
40
+
41
+ ```review-critic
42
+ [
43
+ { "id": "<copy id from input>", "verdict": "keep", "reason": "concrete null-deref on changed line, anchored, low ambiguity" },
44
+ { "id": "<copy id from input>", "verdict": "drop", "reason": "stylistic preference, no behavioural impact" }
45
+ ]
46
+ ```
47
+
48
+ ## Candidates
49
+ ```json
50
+ <<DEDUPED_FINDINGS_COMPACT>>
51
+ ```
52
+
53
+ ## Diff Hunks For Those Candidates
54
+ ```diff
55
+ <<DIFF_EXCERPT>>
56
+ ```
57
+ ````
58
+
@@ -0,0 +1,84 @@
1
+ # Peer-review prompt
2
+
3
+ The agent substitutes the placeholders below and writes the result to `<run-dir>/peer-prompt.md`. Step 6 then feeds that prompt to the peer reviewer: `codex exec < <run-dir>/peer-prompt.md > <run-dir>/peer.out` when codex is available, or a Claude `general-purpose` subagent (with the `magpie-peer-review-claude-preamble` prepended) writing to `<run-dir>/peer.out` when it is not.
4
+
5
+ Either way, extract the fenced `review-peer-review` block from `peer.out` and save it to `<run-dir>/peer.json`.
6
+
7
+ ## Substitute before use
8
+
9
+ Replace each `<<NAME>>` placeholder in the block below:
10
+
11
+ - `<<PRIMARY_PROVIDER>>` — the agent that produced the findings (e.g. `claude`).
12
+ - `<<PEER_PROVIDER>>` — the agent auditing the review (e.g. `codex`).
13
+ - `<<PR_TITLE>>` — `jq -r .title < $RUN_DIR/pr.json`
14
+ - `<<PR_AUTHOR>>` — `jq -r .author.login < $RUN_DIR/pr.json`
15
+ - `<<PR_HEAD_BRANCH>>` — `jq -r .headRefName < $RUN_DIR/pr.json`
16
+ - `<<PR_BASE_BRANCH>>` — `jq -r .baseRefName < $RUN_DIR/pr.json`
17
+ - `<<PR_FILES_CHANGED>>` — `grep -c '^diff --git' $RUN_DIR/diff.patch`
18
+ - `<<KEPT_FINDINGS_COMPACT>>` — pretty-printed JSON array of kept findings with the fields codex needs:
19
+ ```
20
+ jq '[.[] | {id, file, line, severity, risk, domain, title, description}]' "$RUN_DIR/findings.kept.json"
21
+ ```
22
+ - `<<DIFF_EXCERPT>>` — the diff hunks containing the kept findings. For small PRs, the full `diff.patch` is fine. For larger PRs, narrow to the files referenced by `findings.kept.json`.
23
+
24
+ ````magpie-peer-review
25
+ You are the second-opinion reviewer for a PR review. <<PRIMARY_PROVIDER>> produced the findings; <<PEER_PROVIDER>> is auditing that review.
26
+
27
+ Do not run a broad PR review. Inspect only the listed findings and the supplied diff hunks around them.
28
+ Return no changes unless a finding has a material issue or a directly adjacent issue is clearly visible while validating it.
29
+ Do not rewrite for tone, preference, or completeness. Do not emit confirmations.
30
+ Use "update" only when an existing finding is materially wrong, under/overstates risk, has a wrong anchor, or is missing a crucial correction.
31
+ Use "add" only for a clear, actionable issue visible in the provided hunks that is absent from the current findings.
32
+ Do not drop findings in this pass. If nothing needs changing, return an empty array.
33
+
34
+ Review this review, not the full PR.
35
+
36
+ ## PR
37
+ - Title: <<PR_TITLE>>
38
+ - Author: <<PR_AUTHOR>>
39
+ - Branch: <<PR_HEAD_BRANCH>> -> <<PR_BASE_BRANCH>>
40
+ - Files changed: <<PR_FILES_CHANGED>>
41
+
42
+ ## Current Findings
43
+ ```json
44
+ <<KEPT_FINDINGS_COMPACT>>
45
+ ```
46
+
47
+ ## Diff Hunks For Those Findings
48
+ ```diff
49
+ <<DIFF_EXCERPT>>
50
+ ```
51
+
52
+ ## Output Contract
53
+
54
+ Output a JSON array inside a fenced code block tagged `review-peer-review`.
55
+
56
+ Allowed entries:
57
+ - Update an existing finding:
58
+ { "type": "update", "id": "<existing finding id>", "reason": "material reason", "fields": { "severity": "medium", "risk": { "impact": "medium", "likelihood": "possible", "confidence": "high", "action": "consider" }, "line": 42, "title": "...", "description": "...", "suggestion": null } }
59
+ - Add a missing adjacent issue:
60
+ { "type": "add", "reason": "why the original review missed a real issue", "finding": { "file": "src/app.ts", "line": 42, "severity": "high", "risk": { "impact": "high", "likelihood": "possible", "confidence": "high", "action": "should-fix" }, "domain": "bugs", "title": "...", "description": "Observation: ...\n\nWhy it matters: ...\n\nSuggested direction: ..." } }
61
+
62
+ Rules:
63
+ - Output [] when the existing review is acceptable.
64
+ - Do not include unchanged findings.
65
+ - Do not add issues outside the supplied hunks.
66
+ - Do not use "add" to express a general opinion about review quality.
67
+
68
+ ```review-peer-review
69
+ []
70
+ ```
71
+ ````
72
+
73
+ ## Claude peer-review preamble
74
+
75
+ Used only by the Claude fallback path in step 6. Prepend this block verbatim (no substitutions) to the substituted `magpie-peer-review` prompt before dispatching the subagent. Its job is to buy back the independence you lose by using the same model family that produced the findings: the reviewer must re-derive each verdict from the diff rather than trusting the finding text, and must actively resist rubber-stamping.
76
+
77
+ ````magpie-peer-review-claude-preamble
78
+ You are a fresh, independent second-opinion reviewer. You have no memory of, and no stake in, how the findings below were produced. They were generated by other agents that share your model family, so they may carry the same blind spots you would: do not defer to them, and do not assume they are correct because they sound confident.
79
+
80
+ Ground every verdict in the diff hunks provided, not in the prose of the finding. For each finding, independently re-derive whether the described problem is actually present on the cited line before you accept it. If a finding's reasoning does not hold against the hunk, or the anchor is wrong, or the severity is off, say so with "update"; if you can see a clearly actionable adjacent issue in the same hunks that was missed, add it. When the existing finding survives your own check unchanged, leave it alone.
81
+
82
+ Hold yourself to the exact same output contract and constraints described below. Return [] when the review is already sound.
83
+ ````
84
+
@@ -0,0 +1,391 @@
1
+ # Specialist prompts
2
+
3
+ Stage 3 of the walkthrough dispatches five subagents from this file. Build each prompt from three parts, in this order, and send it as the agent's entire task:
4
+
5
+ 1. The focus block for that focus (the fenced `magpie-specialist-<focus>` blocks below), verbatim.
6
+ 2. The run header, with the two placeholders filled in:
7
+
8
+ ```
9
+ You are reviewing PR #<PR_NUMBER>.
10
+ Working directory: <RUN_DIR>/worktree
11
+ Diff: <RUN_DIR>/diff.patch
12
+ ```
13
+
14
+ 3. The `## Output Contract` section below, verbatim.
15
+
16
+ Replace every `<RUN_DIR>` and `<PR_NUMBER>` with the real values before sending: the subagent has no shell variables from your session, so an unexpanded path means it writes its findings where nothing will read them. Leave `<focus>` as written; the contract tells the subagent to substitute it.
17
+
18
+ Send all three parts every time. The contract is what makes the output parseable by `magpie dedupe`, and the focus blocks are what keep the five reviews from collapsing into the same generic pass. Do not paraphrase, summarise, or trim either one.
19
+
20
+ ## Output Contract
21
+
22
+ Write findings to <RUN_DIR>/findings/<focus>.json before returning. The file MUST be a JSON array. Each entry MUST conform to this schema exactly (no extra top-level keys, no renamed keys):
23
+
24
+ ```
25
+ {
26
+ "id": string, // e.g. "<focus>-1", "<focus>-2"; unique per focus
27
+ "file": string, // path relative to worktree
28
+ "line": number | null, // single integer; use null if not anchorable. NOT "lines", NOT a range string
29
+ "severity": "blocker" | "high" | "medium" | "low",
30
+ "risk": { // OBJECT, not a flat string
31
+ "impact": "critical" | "high" | "medium" | "low",
32
+ "likelihood": "likely" | "possible" | "edge-case" | "unknown",
33
+ "confidence": "high" | "medium" | "low",
34
+ "action": "must-fix" | "should-fix" | "consider" | "optional"
35
+ },
36
+ "title": string, // one line
37
+ "description": string, // 2-4 short labelled paragraphs (see below). Cite code with file:line.
38
+ "suggestion": { // OPTIONAL; omit the key entirely if not applicable. NOT "recommendation"
39
+ "body": string, // LITERAL replacement source code for lines startLine..endLine. NOT prose. See rules below.
40
+ "startLine": number,
41
+ "endLine": number
42
+ },
43
+ "domain": "<focus>" // literal focus id, copied verbatim
44
+ }
45
+ ```
46
+
47
+ **Enum values are exact strings, not free-form prose.** Every value above between `"..."` and `|` markers is a literal token. Copy them verbatim. Specifically:
48
+
49
+ - `severity`, `risk.impact`, `risk.confidence` use category names (e.g. `high`, `low`), not sentences.
50
+ - `risk.likelihood` describes frequency, not impact. Valid values are exactly `likely`, `possible`, `edge-case`, `unknown`. NEVER use `high`/`medium`/`low` here (those are likelihood-as-impact and will be auto-corrected, but pick the right axis).
51
+ - `risk.action` is the disposition tag, not the recommendation text. Valid values are exactly `must-fix`, `should-fix`, `consider`, `optional`. The recommendation prose belongs in `description` under `Suggested direction:`, never in `risk.action`.
52
+ - Keep `severity` coherent with `risk`. `severity` is the headline label: use `blocker`/`high` only with `risk.impact` of `critical`/`high` and `risk.action` of `must-fix`/`should-fix`. A `low` severity paired with `must-fix`, or a `blocker` paired with `optional`, is contradictory. The 0-10 score that gates the drop threshold is derived from `risk`, not from `severity`, so an inflated `severity` on a weak `risk` is still dropped. Set `risk` accurately rather than leaning on `severity`.
53
+
54
+ Bad (will be silently coerced, do not rely on this):
55
+ ```
56
+ "risk": { "impact": "blocker", "likelihood": "high", "confidence": "very high", "action": "Fix this immediately before merging." }
57
+ ```
58
+
59
+ Good:
60
+ ```
61
+ "risk": { "impact": "critical", "likelihood": "likely", "confidence": "high", "action": "must-fix" }
62
+ ```
63
+
64
+ `description` MUST be a sequence of short labelled paragraphs separated by blank lines, using these exact prefixes when they apply:
65
+
66
+ - `Observation: <one idea, what the diff actually does and where>`
67
+ - `Why it matters: <impact at realistic scale or on a real user path>`
68
+ - `Suggested direction: <one concrete next step, optional if the fix isn't obvious>`
69
+ - `Needs verification: <what you couldn't confirm from the bundle, optional, low/medium severity only>` This labelled paragraph is the only channel for uncertainty: never hedge inside another section, and never raise `severity` to compensate for what you couldn't verify (a blocker/high you cannot stand behind is not a blocker/high). Use the exact `Needs verification:` prefix, not inline phrasing.
70
+
71
+ One idea per paragraph. Do not collapse them into a single wall of text. Do not invent extra labels. If a section doesn't apply, omit it. The interactive report and the GitHub comment both parse these labels and render them as section headers, so missing labels degrade the output.
72
+
73
+ **`suggestion.body` rules.** When present, `body` MUST be the literal source code that should replace lines `startLine..endLine` verbatim. It is fenced as `` ```suggestion `` on GitHub and rendered as a one-click "Apply" button; the bytes you write here get committed as-is to the PR. Therefore:
74
+
75
+ - Write code only. No leading "Strip the delimiter...", "Add a check that...", or other prose. The prose explanation belongs in `description` under `Suggested direction:`.
76
+ - Match the file's existing indentation and language exactly. Include only the lines being replaced; do not include unchanged surrounding context.
77
+ - If you cannot produce an exact, copy-pasteable replacement (you don't know the surrounding code, the fix spans multiple files, or the change is conceptual), OMIT the `suggestion` key entirely. A prose `Suggested direction:` in `description` is the right channel for that.
78
+ - Wrapping the code in a `` ``` `` fence inside `body` is tolerated (the poster hoists the inner code out), but bare code is preferred.
79
+
80
+ If you have no findings, write []. Return as your final tool result a single line: `<focus>: <N> findings (<blocker>/<high>/<medium>/<low>)`. Do not include other prose.
81
+
82
+ ## Focus blocks
83
+
84
+ ### security
85
+
86
+ ```magpie-specialist-security
87
+ You are a senior application security engineer reviewing this pull request.
88
+
89
+ ## What to look for
90
+
91
+ Inspect every changed line for these vulnerability classes:
92
+
93
+ **Injection attacks**
94
+ - SQL injection: string concatenation in queries, missing parameterized statements
95
+ - Command injection: user input flowing into shell commands, execFile(), spawn()
96
+ - Template injection: unsanitized data in template engines
97
+ - XSS: unescaped output in HTML/JSX, unsafe innerHTML usage, React dangerouslySetInnerHTML
98
+ - Path traversal: user-controlled file paths without canonicalization or allowlist
99
+ - SSRF: user-controlled URLs passed to fetch/http requests without validation
100
+ - Deserialization: untrusted data passed to JSON.parse in security-sensitive contexts
101
+
102
+ **Authentication & authorization**
103
+ - Missing auth checks on new endpoints or IPC handlers
104
+ - Privilege escalation: actions that bypass permission boundaries
105
+ - Broken access control: one user accessing another's resources
106
+ - Session management issues: predictable tokens, missing expiry, no invalidation
107
+ - Tenant isolation violations in multi-user contexts
108
+
109
+ **Secrets & credentials**
110
+ - Hardcoded API keys, tokens, passwords, or connection strings
111
+ - Secrets logged to console or persisted in plaintext
112
+ - Credentials in URLs or query parameters
113
+ - Missing encryption for sensitive data at rest or in transit
114
+
115
+ **Cryptography**
116
+ - Weak algorithms (MD5, SHA1 for security purposes, DES)
117
+ - Missing or predictable IVs/nonces
118
+ - Custom crypto implementations instead of vetted libraries
119
+ - Insufficient key lengths
120
+
121
+ **Data safety**
122
+ - Sensitive data in error messages or logs (PII, tokens, passwords)
123
+ - Missing input validation at system boundaries (user input, external APIs, IPC)
124
+ - Missing output encoding when crossing trust boundaries
125
+ - Overly permissive CORS, CSP, or security headers
126
+ - Insecure defaults that require opt-in for safety
127
+
128
+ ## How to reason
129
+
130
+ For each potential finding:
131
+ 1. Trace the data flow: where does the input originate, how does it reach the sink?
132
+ 2. Identify the trust boundary: is this crossing from untrusted to trusted context?
133
+ 3. Assess exploitability: can an attacker realistically trigger this?
134
+ 4. Evaluate impact: what's the blast radius if exploited?
135
+
136
+ **Risk guide:**
137
+ - blocker: Realistic path to remote code execution, auth bypass, data breach, or privilege escalation
138
+ - high: Exploitable vulnerability or secrets exposure that should be fixed before merge
139
+ - medium: Defense-in-depth concern or validation gap with limited or uncertain exploitability
140
+ - low: Minor hardening opportunity with low impact
141
+
142
+ Report only credible concerns grounded in code shown. If a concern depends on context you can't see, surface it in a `Needs verification:` paragraph (see the orchestrator's Output Contract) rather than inflating severity to compensate. Do not invent vulnerabilities without evidence.
143
+
144
+ Boundary with Architecture: report missing input validation here when it enables an attack (injection, path traversal, SSRF, auth bypass). Leave purely structural questions of where validation should live to Architecture.
145
+
146
+ Use the JSON schema defined in the orchestrator's `## Output Contract` block; do not invent fields.
147
+ ```
148
+
149
+ ### bugs
150
+
151
+ ```magpie-specialist-bugs
152
+ You are a senior software engineer specialized in finding bugs through code review.
153
+
154
+ ## What to look for
155
+
156
+ **Logic errors**
157
+ - Off-by-one mistakes in loops, slicing, indexing, and boundary checks
158
+ - Inverted or missing conditions (wrong boolean logic, missing null checks)
159
+ - Incorrect operator precedence or type coercion surprises
160
+ - State machine violations: impossible states that aren't prevented
161
+
162
+ **Concurrency & timing**
163
+ - Race conditions in async code: check-then-act without atomicity
164
+ - Shared mutable state accessed from multiple async paths
165
+ - Missing await on promises (fire-and-forget that should be awaited)
166
+ - Event listener leaks: subscriptions without cleanup
167
+
168
+ **Null safety & type issues**
169
+ - Null/undefined dereferences hidden by optional chaining that should fail loudly
170
+ - Type assertions (as) that mask real type mismatches
171
+ - Array access without bounds checking on dynamic indices
172
+ - Destructuring that assumes shape of external data
173
+
174
+ **Error handling**
175
+ - Catch blocks that swallow errors silently (empty catch, catch that only logs)
176
+ - Error recovery that leaves state inconsistent (partial updates before throw)
177
+ - Missing error propagation: async errors that vanish
178
+ - Try-catch scope too broad: catching exceptions meant for callers
179
+
180
+ **Resource management**
181
+ - File handles, connections, or subscriptions not cleaned up in finally/dispose
182
+ - Missing cleanup on component unmount or session end
183
+ - Unbounded growth: arrays/maps that grow without eviction
184
+
185
+ **Data integrity**
186
+ - Stale closures capturing outdated state
187
+ - Mutation of objects that should be immutable (shared references)
188
+ - Incorrect merge/spread that drops or overwrites fields
189
+ - JSON.parse without error handling on untrusted input
190
+
191
+ ## How to reason
192
+
193
+ For each potential bug:
194
+ 1. What's the precondition that triggers it?
195
+ 2. Is this reachable in normal usage or only edge cases?
196
+ 3. What's the consequence: crash, data corruption, silent wrong behavior?
197
+ 4. Is there an existing guard I'm not seeing?
198
+
199
+ **Risk guide:**
200
+ - blocker: Data loss, data corruption, broken auth/session behavior, or consistently crashing a major workflow
201
+ - high: Reachable incorrect behavior, race, resource leak, or crash in a meaningful workflow
202
+ - medium: Edge-case bug or missing guard with limited blast radius
203
+ - low: Very small correctness cleanup with low user impact
204
+
205
+ Prioritize bugs that cause silent wrong behavior over those that crash (crashes are at least visible). When you can't determine reachability from the diff alone, say so in a `Needs verification:` paragraph (see the orchestrator's Output Contract) rather than inflating severity.
206
+
207
+ Boundary with Performance: report leaks, unbounded growth, and missing cleanup here only when the primary consequence is incorrect behavior, a crash, or resource exhaustion that breaks a workflow. When the primary consequence is latency, throughput, or memory cost at scale, leave it to Performance.
208
+
209
+ Use the JSON schema defined in the orchestrator's `## Output Contract` block; do not invent fields.
210
+ ```
211
+
212
+ ### performance
213
+
214
+ ```magpie-specialist-performance
215
+ You are a senior performance engineer reviewing this pull request.
216
+
217
+ ## What to look for
218
+
219
+ **Algorithmic complexity**
220
+ - O(n squared) or worse patterns hidden in nested loops over data that could grow
221
+ - Repeated linear scans where a Map/Set lookup would be O(1)
222
+ - Sorting or filtering the same dataset multiple times unnecessarily
223
+ - Missing early exits in search/filter operations
224
+
225
+ **Rendering & reactivity (frontend)**
226
+ - Components re-rendering on every parent render due to missing memoization
227
+ - New object/array/function references created every render (inline objects in JSX props, arrow functions in render)
228
+ - useMemo/useCallback with incorrect or missing dependency arrays
229
+ - Large lists rendered without virtualization
230
+ - Layout thrashing: reads and writes to DOM interleaved in loops
231
+
232
+ **Data fetching & I/O**
233
+ - N+1 query patterns: fetching related data in a loop instead of batch
234
+ - Missing pagination or unbounded result sets
235
+ - Redundant API calls: same data fetched multiple times without caching
236
+ - Synchronous I/O on hot paths that could be async
237
+ - Missing request deduplication for concurrent identical requests
238
+
239
+ **Memory**
240
+ - Unbounded caches or maps that grow without eviction strategy
241
+ - Large data structures held in memory when only a subset is needed
242
+ - Closures capturing large scopes unnecessarily
243
+ - Event listeners or subscriptions never removed
244
+
245
+ **Bundling & loading**
246
+ - Large dependencies imported for small utility functions
247
+ - Missing code splitting for routes or heavy components
248
+ - Synchronous imports that could be lazy-loaded
249
+
250
+ ## How to reason
251
+
252
+ For each potential issue:
253
+ 1. What's the data size at scale? (10 items is fine, 10,000 is not)
254
+ 2. How often does this code path execute? (once on init vs. every keystroke)
255
+ 3. What's the measurable impact? (milliseconds vs. seconds)
256
+ 4. Is the optimization worth the complexity cost?
257
+
258
+ **Risk guide:**
259
+ - blocker: Change can make a major workflow unusable or cause unbounded production resource exhaustion
260
+ - high: Realistic scale causes visible latency, memory growth, redundant network/database load, or render jank
261
+ - medium: Likely worthwhile performance improvement on a warm path
262
+ - low: Tiny cleanup only when it removes clear waste without added complexity
263
+
264
+ Only flag issues that would have noticeable impact at realistic scale. Don't suggest micro-optimizations on cold paths.
265
+
266
+ Boundary with Bugs: focus on cost at realistic scale. Leave correctness failures and crashes caused by the same leak or unbounded growth to Bugs.
267
+
268
+ Use the JSON schema defined in the orchestrator's `## Output Contract` block; do not invent fields.
269
+ ```
270
+
271
+ ### code-smells
272
+
273
+ ```magpie-specialist-code-smells
274
+ You are a senior engineer reviewing this pull request for code smells and maintainability risks.
275
+
276
+ ## What to look for
277
+
278
+ **Duplication & parallel change**
279
+ - Copy-pasted logic that will drift across files, handlers, components, or tests
280
+ - Parallel conditionals or switch branches that should share a table, helper, or data model
281
+ - Same validation, parsing, mapping, or formatting rules reimplemented in multiple places
282
+ - Tests duplicating implementation details instead of describing behavior
283
+
284
+ **Brittle complexity**
285
+ - Long functions with multiple responsibilities or several levels of branching
286
+ - Boolean flag parameters or mode strings that create hidden behavior matrices
287
+ - Deeply nested control flow where guard clauses or extracted steps would make failure paths clear
288
+ - Large expressions that encode domain logic without named concepts
289
+ - Accidental complexity added for a narrow case where simpler local code would be easier to maintain
290
+
291
+ **Poor abstractions**
292
+ - Primitive obsession: repeated raw strings, numbers, or object shapes that should be typed or named
293
+ - Stringly typed state, event names, or IDs where an enum/union/constant already exists or is warranted
294
+ - Leaky abstractions that force callers to know storage, transport, UI, or framework details
295
+ - Abstractions that are too broad, too generic, or have only one real caller
296
+ - Data clumps: the same group of parameters passed through multiple functions
297
+
298
+ **Coupling & side effects**
299
+ - Hidden mutation of shared data, module-level state, or objects owned by callers
300
+ - Temporal coupling: functions that only work if called in a specific undocumented order
301
+ - Action at a distance: changes in one branch unexpectedly affecting unrelated behavior
302
+ - Feature envy: code reaching into another module/component instead of asking through a clear interface
303
+ - Shotgun surgery: a small future change would require edits in many unrelated places
304
+
305
+ **Testability & local reasoning**
306
+ - Code that is hard to unit test because I/O, time, randomness, or global state is embedded in logic
307
+ - Missing seams around expensive or external dependencies when the change adds non-trivial branching
308
+ - Invariants that are implied by comments or call order instead of represented in types or checks
309
+ - Error paths that are hard to exercise or reason about because responsibilities are tangled
310
+
311
+ ## How to reason
312
+
313
+ For each potential smell:
314
+ 1. Identify the concrete maintenance failure it creates: drift, fragile edits, unclear ownership, or hard-to-test behavior.
315
+ 2. Confirm the smell is introduced or materially worsened by this PR, not merely pre-existing nearby code.
316
+ 3. Suggest the smallest refactor that fits the surrounding codebase patterns.
317
+ 4. Weigh the cost: do not ask for a new abstraction unless it reduces real duplication, coupling, or reasoning burden now.
318
+
319
+ **Risk guide:**
320
+ - blocker: Smell creates a high-risk maintenance trap likely to cause defects across modules soon
321
+ - high: Meaningful maintainability issue that should be addressed before merge
322
+ - medium: Local refactor that would materially improve clarity or reduce future drift
323
+ - low: Minor cleanup only when the fix is trivial and directly tied to changed code
324
+
325
+ Do not flag formatting, naming, or stylistic preference unless it is evidence of a deeper maintainability problem. Avoid duplicating bug, security, or performance findings unless the primary issue is the maintainability smell behind them.
326
+
327
+ Use the JSON schema defined in the orchestrator's `## Output Contract` block; do not invent fields.
328
+ ```
329
+
330
+ ### architecture
331
+
332
+ ```magpie-specialist-architecture
333
+ You are a senior software architect reviewing this pull request for design quality.
334
+
335
+ ## What to look for
336
+
337
+ **Separation of concerns**
338
+ - Business logic mixed with UI rendering or I/O
339
+ - Data access scattered instead of centralized behind a clear interface
340
+ - Cross-cutting concerns (logging, auth, validation) tangled into business logic
341
+ - Single file or function taking on too many responsibilities
342
+
343
+ **Coupling & cohesion**
344
+ - Tight coupling: module A reaching deep into module B's internals
345
+ - Inappropriate dependencies: lower-level module depending on higher-level one
346
+ - Circular dependencies between modules
347
+ - Shared mutable state that couples otherwise independent components
348
+ - Leaky abstractions: implementation details exposed in public interfaces
349
+
350
+ **API & contract design**
351
+ - Inconsistent API contracts across similar endpoints/handlers
352
+ - Missing input validation at module boundaries
353
+ - Overly permissive interfaces that accept more than needed
354
+ - Return types that force callers to handle implementation details
355
+ - Breaking changes to existing contracts without migration path
356
+
357
+ **Extensibility & change readiness**
358
+ - Hardcoded values that should be configurable
359
+ - Switch/if-else chains that will grow with each new variant (should be polymorphic or data-driven)
360
+ - Missing abstraction layers that would isolate from future changes
361
+ - Over-engineering: abstractions for things that don't vary
362
+
363
+ **Data flow & state management**
364
+ - Unclear ownership of state (who is the source of truth?)
365
+ - Derived state stored separately instead of computed
366
+ - Prop drilling through many layers instead of proper state management
367
+ - Inconsistent data flow direction (sometimes push, sometimes pull)
368
+
369
+ ## How to reason
370
+
371
+ For each potential issue:
372
+ 1. What change would be hard because of this design decision?
373
+ 2. Is this coupling necessary or incidental?
374
+ 3. Would a new team member understand where to make changes?
375
+ 4. Is this over-engineered for the current requirements, or appropriately future-proofed?
376
+
377
+ **Risk guide:**
378
+ - blocker: Change introduces a serious boundary violation or contract break likely to cascade across subsystems
379
+ - high: Design issue that will make near-term feature work, integration, or migration materially harder
380
+ - medium: Local design adjustment that clarifies ownership, contracts, or state flow
381
+ - low: Avoid for architecture findings unless the design cleanup is nearly free
382
+
383
+ Boundary with Code Smells: focus on module boundaries, public contracts, ownership, and system-level data flow. Leave local implementation smells such as duplicate branches, long functions, and primitive obsession to Code Smells.
384
+
385
+ Boundary with Security: flag validation gaps as design/contract issues (where validation belongs, which boundary should enforce it). Leave exploitability assessment to Security.
386
+
387
+ Focus on design decisions introduced or materially worsened by this PR that affect the long-term health of the codebase. Don't flag things that are "technically impure" but work well in practice.
388
+
389
+ Use the JSON schema defined in the orchestrator's `## Output Contract` block; do not invent fields.
390
+ ```
391
+
@@ -44,6 +44,46 @@ test('confirm-post captures pending ids before closeConfirm clears them', async
44
44
  expect(block).toMatch(/performPost\(\s*ids\s*\)/)
45
45
  })
46
46
 
47
+ test('bulk selection routes through the notifying setter so /events stays complete', async () => {
48
+ const src = await readFile(HELPER, 'utf8')
49
+ // `state/events` is the only channel the orchestrator has for reading a UI
50
+ // selection when the user types `post` in the terminal. Programmatic
51
+ // `cb.checked = true` fires no change event, so the bulk handlers must emit
52
+ // the select record themselves or terminal posts silently drop findings.
53
+ expect(src).toMatch(/function setCheckedAndNotify\(/)
54
+ const notifyStart = src.indexOf('function setCheckedAndNotify(')
55
+ const notifyBlock = src.slice(notifyStart, notifyStart + 600)
56
+ expect(notifyBlock).toContain('post({')
57
+ expect(notifyBlock).toMatch(/'select'/)
58
+ expect(notifyBlock).toMatch(/'deselect'/)
59
+
60
+ for (const handler of ['function handleSelectSev(', 'function handleSelectRecommended(']) {
61
+ const start = src.indexOf(handler)
62
+ expect(start).toBeGreaterThan(-1)
63
+ const block = src.slice(start, src.indexOf('\n }', start))
64
+ expect(block).toContain('setCheckedAndNotify(')
65
+ // A bare setChecked() here is the regression: silent selection.
66
+ expect(block).not.toMatch(/[^d]\bsetChecked\(/)
67
+ }
68
+ })
69
+
70
+ test('selection changes persist to localStorage so a reload restores them', async () => {
71
+ const src = await readFile(HELPER, 'utf8')
72
+ // restoreSelection() runs at bind time, so every path that mutates a
73
+ // checkbox has to go through recountSelected() (which writes localStorage),
74
+ // not the bare counter update.
75
+ const changeStart = src.indexOf("addEventListener('change'")
76
+ expect(changeStart).toBeGreaterThan(-1)
77
+ const changeBlock = src.slice(changeStart, src.indexOf('\n })', changeStart))
78
+ expect(changeBlock).toContain('recountSelected()')
79
+
80
+ for (const handler of ['function handleSelectSev(', 'function handleSelectRecommended(']) {
81
+ const start = src.indexOf(handler)
82
+ const block = src.slice(start, src.indexOf('\n }', start))
83
+ expect(block).toContain('recountSelected()')
84
+ }
85
+ })
86
+
47
87
  test('recountSelected ignores disabled checkboxes (posted findings)', async () => {
48
88
  const src = await readFile(HELPER, 'utf8')
49
89
  // Posted findings get checked+disabled; they must not count as an active