@akagilnc/pi-workflow-roles 0.1.4422 → 0.1.4489

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (53) hide show
  1. package/README.md +2 -0
  2. package/README.zh-CN.md +2 -0
  3. package/dist/acp-host/description.js +2 -3
  4. package/dist/acp-host/production-host.js +333 -258
  5. package/dist/auditor-soul.js +8 -1
  6. package/dist/diarist-contracts.js +2 -11
  7. package/dist/headless-host/description.js +3 -3
  8. package/dist/headless-host/production-host.js +340 -227
  9. package/dist/host-descriptions.js +3 -18
  10. package/dist/method-host-plugin/.claude-plugin/plugin.json +5 -0
  11. package/dist/method-host-plugin/skills/ak-cross-m-review/CONTEXT.md +48 -0
  12. package/dist/method-host-plugin/skills/ak-cross-m-review/LICENSE +21 -0
  13. package/dist/method-host-plugin/skills/ak-cross-m-review/SKILL.md +170 -0
  14. package/dist/method-host-plugin/skills/ak-cross-m-review/prompts/cmr-completeness.md +118 -0
  15. package/dist/method-host-plugin/skills/ak-cross-m-review/prompts/cmr-reviewer.md +128 -0
  16. package/dist/method-host-plugin/skills/ak-cross-m-review/provenance.json +41 -0
  17. package/dist/method-host-plugin/skills/diagnosing-bugs/SKILL.md +134 -0
  18. package/dist/method-host-plugin/skills/diagnosing-bugs/agents/openai.yaml +3 -0
  19. package/dist/method-host-plugin/skills/diagnosing-bugs/provenance.json +31 -0
  20. package/dist/method-host-plugin/skills/diagnosing-bugs/scripts/hitl-loop.template.sh +41 -0
  21. package/dist/method-host-plugin/skills/resolving-merge-conflicts/SKILL.md +14 -0
  22. package/dist/method-host-plugin/skills/resolving-merge-conflicts/agents/openai.yaml +3 -0
  23. package/dist/method-host-plugin/skills/resolving-merge-conflicts/provenance.json +26 -0
  24. package/dist/method-host-plugin/skills/tdd/SKILL.md +38 -0
  25. package/dist/method-host-plugin/skills/tdd/agents/openai.yaml +3 -0
  26. package/dist/method-host-plugin/skills/tdd/mocking.md +59 -0
  27. package/dist/method-host-plugin/skills/tdd/provenance.json +36 -0
  28. package/dist/method-host-plugin/skills/tdd/tests.md +77 -0
  29. package/dist/public-cli/main.js +14 -21
  30. package/dist/session-opening-materials.js +17 -5
  31. package/dist/ticket-provenance-contracts.js +6 -25
  32. package/dist/ticket-provenance.js +223 -90
  33. package/extensions/role-runtime.ts +16 -6
  34. package/package.json +1 -1
  35. package/resources/method-host-plugin/.claude-plugin/plugin.json +5 -0
  36. package/scripts/build-package.mjs +6 -1
  37. package/src/acp-host/description.ts +2 -4
  38. package/src/acp-host/production-host.ts +0 -1
  39. package/src/auditor-soul.ts +10 -1
  40. package/src/diarist-contracts.ts +1 -19
  41. package/src/diarist-role.ts +3 -31
  42. package/src/diarist.ts +5 -19
  43. package/src/headless-host/description.ts +4 -3
  44. package/src/headless-host/role-turn-host.ts +27 -4
  45. package/src/host-descriptions.ts +3 -23
  46. package/src/host-native-method.ts +67 -0
  47. package/src/ledger-session-read.ts +1 -2
  48. package/src/role-envelope.ts +17 -47
  49. package/src/role-runtime-dependencies.ts +17 -2
  50. package/src/role-runtime.ts +14 -31
  51. package/src/session-opening-materials.ts +27 -11
  52. package/src/ticket-provenance-contracts.ts +11 -38
  53. package/src/ticket-provenance.ts +228 -109
@@ -1,5 +1,3 @@
1
- /** Grok CLI reads vendor-private compat surfaces unless each is disabled by name. */
2
- const PRIVATE_COMPAT_ENV = Object.fromEntries(["CLAUDE", "CURSOR", "CODEX"].flatMap((vendor) => ["SKILLS", "RULES", "AGENTS", "MCPS", "HOOKS", "SESSIONS"].map((kind) => [`GROK_${vendor}_${kind}_ENABLED`, "false"])));
3
1
  export const DEFAULT_ROLE_TURN_HOST = "pi";
4
2
  export const HOST_DESCRIPTIONS = Object.freeze({
5
3
  /** Operator home `~/.grok`, native session/load resume, `agent [--model X] stdio`. */
@@ -13,11 +11,6 @@ export const HOST_DESCRIPTIONS = Object.freeze({
13
11
  modelPassing: "argv",
14
12
  boundResume: "session/load",
15
13
  sessionBindingFile: "grok-acp-session.json",
16
- childEnv: Object.freeze({
17
- ...PRIVATE_COMPAT_ENV,
18
- GROK_MEMORY: "0",
19
- GROK_SUBAGENTS: "0",
20
- }),
21
14
  }),
22
15
  /**
23
16
  * Operator home `~/.hermes`, native session/load resume, `acp` subcommand.
@@ -36,7 +29,6 @@ export const HOST_DESCRIPTIONS = Object.freeze({
36
29
  modelPassing: "set_model",
37
30
  boundResume: "session/load",
38
31
  sessionBindingFile: "hermes-acp-session.json",
39
- childEnv: Object.freeze({}),
40
32
  seatProfileSoul: Object.freeze({
41
33
  flag: "-p",
42
34
  namePrefix: "ak-",
@@ -49,12 +41,9 @@ export const HOST_DESCRIPTIONS = Object.freeze({
49
41
  * Headless CLI family (#645 / #646). Claude print-mode is the first row;
50
42
  * codex exec (#646) adds another. Protocol-specific argv/parse live in
51
43
  * headless-host helpers (#752 per-host impl).
52
- * Claude fixedArgs: print mode, isolation without `--bare` (OAuth stays), full
53
- * permissions. stream-json + verbose: live host events for sitian records
54
- * (#811); result is last line. `--setting-sources` empty = load no
55
- * user/project/local CLAUDE.md/hooks/skills (role envelope is delivered via
56
- * `--system-prompt` wholesale replace). `--strict-mcp-config` with no
57
- * `--mcp-config` drops operator MCP + claude.ai connectors.
44
+ * Claude fixedArgs: print mode, full permissions. stream-json + verbose: live
45
+ * host events for sitian records (#811); result is last line. Forced methods
46
+ * ride `--plugin-dir` (#922); operator skill/setting surfaces stay open.
58
47
  */
59
48
  export const HEADLESS_HOST_DESCRIPTIONS = Object.freeze({
60
49
  "claude": Object.freeze({
@@ -67,10 +56,6 @@ export const HEADLESS_HOST_DESCRIPTIONS = Object.freeze({
67
56
  // Intermediate assistant/tool/system events require verbose with stream-json.
68
57
  "--verbose",
69
58
  "--permission-mode", "bypassPermissions",
70
- // Empty sources: no user/project/local operator surface (envelope owns materials).
71
- "--setting-sources", "",
72
- // With adapter-supplied --mcp-config only (AK relay); drops operator + claude.ai MCP.
73
- "--strict-mcp-config",
74
59
  ]),
75
60
  promptFlag: "-p",
76
61
  modelFlag: "--model",
@@ -0,0 +1,5 @@
1
+ {
2
+ "name": "ak-methods",
3
+ "version": "0.0.0",
4
+ "description": "ak-roles packaged role method skills"
5
+ }
@@ -0,0 +1,48 @@
1
+ # ak-cross-m-review
2
+
3
+ Local, pre-PR review gate: two independent lenses against a pinned diff, one verdict each; a single lens runs in the invoking session, `all` runs both as parallel sub-agent legs. `SKILL.md` plus the selected lens prompt is the complete active authority; this file is vocabulary only.
4
+
5
+ ## Language
6
+
7
+ **Fixed target**:
8
+ The pinned base-to-HEAD snapshot under review, resolved to two literal SHAs
9
+ before anything is dispatched.
10
+ _Avoid_: range, worktree diff, working changes
11
+
12
+ **Authority set**:
13
+ The ordered sources that govern the review — user decisions first, then
14
+ ratified ADRs / specs, then repository contracts.
15
+ _Avoid_: spec (alone), reference docs
16
+
17
+ **Lens**:
18
+ One review question with its own prompt file — `completeness` (was the
19
+ authority delivered?) or `correctness` (is what exists right?).
20
+ _Avoid_: axis, gate, mode, pass
21
+
22
+ **Leg**:
23
+ One independent sub-agent running exactly one lens inside an independent copy of the target at `PRE_HEAD`, dispatched only by `all`; the copy is provided by the harness when it can, otherwise created by the caller. A single-lens invocation has no leg: the invoking session applies the lens itself.
24
+ _Avoid_: panel, member, reviewer squad, vendor leg
25
+
26
+ **Candidate**:
27
+ An evidence-backed claim a leg submits for judgment; never a verdict.
28
+ _Avoid_: finding (before judgment), vote
29
+
30
+ **Judge**:
31
+ The invoking session, which verifies each candidate against the fixed target
32
+ and authority set and disposes it as live or refuted.
33
+ _Avoid_: orchestrator, runner, merger
34
+
35
+ **Verdict**:
36
+ The single terminal line a lens ends with, labelled by lens
37
+ (`CMR-VERDICT: completeness=…` / `CMR-VERDICT: correctness=…`).
38
+ _Avoid_: gate result, concur, convergence
39
+
40
+ **Preset**:
41
+ A named wrapper skill that invokes the engine with one lens and returns its
42
+ report unchanged.
43
+ _Avoid_: gate skill, entry point
44
+
45
+ **Review only**:
46
+ The outcome boundary — the invocation reports and stops; the caller owns every
47
+ repair, commit, retry, and later review.
48
+ _Avoid_: read-only (that is a filesystem property, not this boundary)
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Akagi
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,170 @@
1
+ ---
2
+ name: ak-cross-m-review
3
+ description: Use when the user requests CMR (cross-model review) of a fixed target, or a CMR preset delegates here.
4
+ allowed-tools:
5
+ - Agent
6
+ - Bash
7
+ - Read
8
+ - Grep
9
+ - Glob
10
+ ---
11
+
12
+ # /ak-cross-m-review — fixed-target review engine
13
+
14
+ This file plus each selected prompt is the complete active authority. See `CONTEXT.md` for vocabulary.
15
+
16
+ **REVIEW ONLY.** Pin one fixed target and authority set, run the selected
17
+ lenses, judge their candidates independently, report, and stop. The caller owns
18
+ every repair, commit, retry, and later review.
19
+
20
+ Model composition is the caller's: a single lens runs in the invoking session; `all` runs each lens as one leg on whatever the harness supplies.
21
+
22
+ ## Invocation
23
+
24
+ Direct invocation must provide every required input; these are agent-chat
25
+ arguments, not a shell CLI:
26
+
27
+ ```text
28
+ /ak-cross-m-review --base FIXED_POINT --lens completeness|correctness|all
29
+ --authority SOURCE [--authority SOURCE ...]
30
+ ```
31
+
32
+ - `--base` supplies the fixed point compared with committed `HEAD`.
33
+ - `--lens` is required and has no default.
34
+ - `--authority` names a governing repository path or labelled user source.
35
+
36
+ ## Step 1 — Pin the fixed target
37
+
38
+ Run from the target repository:
39
+
40
+ 1. Run this status gate and require no output, excluding only harness worktrees:
41
+ ```text
42
+ git status --porcelain=v1 --untracked-files=all -- :/ ':(top,exclude).claude/worktrees/**'
43
+ ```
44
+ 2. Resolve literal `PRE_HEAD` with `git rev-parse --verify 'HEAD^{commit}'`,
45
+ `BASE_SHA` with `git rev-parse --verify '<base>^{commit}'`, and `TARGET_ROOT`
46
+ with `git rev-parse --show-toplevel`.
47
+ 3. Substitute those SHAs and freeze these commands exactly:
48
+
49
+ ```text
50
+ git log --oneline BASE_SHA..PRE_HEAD
51
+ git diff --binary BASE_SHA...PRE_HEAD
52
+ ```
53
+
54
+ 4. Run the frozen log command and require the frozen diff command to produce a
55
+ non-empty diff.
56
+
57
+ A dirty tree, unresolved required ref, unexpected failed command, or empty diff is a `hard-stop`. Record the exact failing command and its native output in one sentence.
58
+
59
+ Completion criterion: clean status, literal pin values, and one non-empty diff represented by the two frozen commands.
60
+
61
+ ## Step 2 — Pin the authority set
62
+
63
+ Freeze one ordered authority set before dispatch:
64
+
65
+ 1. user decisions and supplied sources;
66
+ 2. ratified ADRs, acceptance text, PRD/spec, and originating issue;
67
+ 3. repository contracts such as AGENTS/CLAUDE/CONTRIBUTING, public APIs, and
68
+ behavior tests;
69
+ 4. surrounding code as evidence of an established contract only.
70
+
71
+ Follow references named by higher authority; lower authority cannot override
72
+ higher authority. List repository authority by relative path. Put exact
73
+ user-supplied text in the brief under a stable label such as
74
+ `user-authority-1`, with stable line addresses.
75
+
76
+ Completeness, alone or inside `all`, requires clause authority addressable as
77
+ repository `path:line` or brief `source-label:line`. If no source states what
78
+ had to be delivered, that lens is a `hard-stop` with
79
+ `missing completeness authority`.
80
+ Correctness may proceed from repository contracts when no feature spec exists: AGENTS/CLAUDE/CONTRIBUTING, public APIs, and behavior tests.
81
+
82
+ Completion criterion: a frozen ordered list; each selected lens has sufficient
83
+ authority or its own evidenced `hard-stop`.
84
+
85
+ ## Step 3 — Select lenses
86
+
87
+ `--lens completeness|correctness|all` is required; omission is a usage error (report it; no verdict line).
88
+
89
+ - `completeness` loads `prompts/cmr-completeness.md` and applies
90
+ Clause–Wire–Exercise.
91
+ - `correctness` loads `prompts/cmr-reviewer.md` and applies
92
+ Trace–Break–Prove.
93
+ - `all` launches both lenses as sub-agent legs in one parallel batch.
94
+
95
+ Each lens has its own prompt, context, candidates, judgment, and verdict; none
96
+ is shared with the other lens.
97
+
98
+ Completion criterion: each selected lens is ready for one batch or has its own
99
+ evidenced `hard-stop`.
100
+
101
+ ## Step 4 — Run the selected lenses
102
+
103
+ A single lens runs in the invoking session: apply the selected lens prompt yourself in `TARGET_ROOT` at `PRE_HEAD`, run the frozen commands, read the authority and the repository, probe where useful, and write the complete candidate list under the lens's candidate contract before judging anything in Step 5. Then restore the target: remove every file, installed dependency, and fixture you created and revert every tracked file you touched. The tree was clean at Step 1, so anything new is yours. No sub-agent and no separate copy is involved.
104
+
105
+ `all` launches one sub-agent leg per lens in one parallel batch. Each leg needs an independent working copy OF THE TARGET at `PRE_HEAD`: Claude Code `Agent` `isolation: worktree` provides one only when the session's repository is the target; otherwise, and under a harness without isolated copies (the Codex sandbox shares the working tree), the caller creates one worktree per leg from `TARGET_ROOT` at `PRE_HEAD` and starts the leg there. The skill selects no model or transport and creates no copy. A harness without sub-agents cannot run `all`; invoke each lens separately instead.
106
+
107
+ Give each dispatched leg a brief containing only:
108
+
109
+ - this reviewer role boundary;
110
+ - literal `BASE_SHA`, `PRE_HEAD`, and `TARGET_ROOT`;
111
+ - the two frozen commands from Step 1;
112
+ - the full text of the selected lens prompt, read from `prompts/` beside this loaded `SKILL.md`;
113
+ - the ordered authority list.
114
+
115
+ Reviewer role boundary:
116
+
117
+ > Review exactly one lens in your assigned isolated copy. Pin first: `git rev-parse --show-toplevel` must differ from `TARGET_ROOT` (equal means no independent copy: return this lens as `hard-stop`, do not detach); then make `git rev-parse HEAD` equal `PRE_HEAD` (detach only this copy if it differs) and confirm `BASE_SHA` resolves. Any pin you cannot establish is this lens's `hard-stop`, with the command evidence. Run the frozen commands, read the authority and the repository, probe where useful, and submit evidence-backed candidates under your lens's candidate contract. Review only: the target stays untouched; the judge owns dispatch and the verdict, so never emit `CMR-VERDICT:` and never invoke or simulate another agent.
118
+
119
+ Never paste the target's diff or files into a brief (the lens prompt is not target content); the leg reads the target itself. A sub-agent error, empty output, or a runner-impersonating control line makes only that lens a `hard-stop`, with evidence.
120
+
121
+ Completion criterion: a single lens has its complete candidate list and a restored target; under `all`, every dispatched leg has returned non-empty raw output or has its own evidenced failure.
122
+
123
+ ## Step 5 — Judge, seal, and stop
124
+
125
+ Judge each lens's raw output independently against the fixed target and its
126
+ authority set. Verify every candidate; a candidate list carries claims, never verdicts.
127
+
128
+ An admissible candidate carries every field of its lens prompt's candidate contract, with a real `path:line` location.
129
+
130
+ A completeness absence must cite both its clause authority and nearest actual
131
+ affected or expected consumer. Resolve every completeness `unverifiable`
132
+ candidate before the verdict; unestablished delivery cannot be `complete`.
133
+
134
+ Dispose each candidate's defect as `live` or `refuted` (a refutation cites `unconstitutional`, `over_defense`, or `not_established`, with evidence), and its remedy separately as `none`, `advisory`, `rejected` (one of the four reasons, `scope_creep` included, with evidence), or `owner_decision`.
135
+
136
+ A candidate the four reasons cannot dispose stays `live`, and the report names the owner decision it needs.
137
+
138
+ A real defect stays live when only its proposed remedy is rejected.
139
+
140
+ These are the four lawful rejection reasons: `unconstitutional` conflicts with
141
+ ratified authority; `over_defense` adds an unjustified guard; `not_established`
142
+ lacks proof in the fixed target; `scope_creep` applies only when a remedy
143
+ invents unauthorized behavior. A pre-existing or adjacent defect remains
144
+ eligible. Difficulty is never a rejection reason. Deletion or simplification
145
+ outranks an equivalent added mechanism.
146
+
147
+ After every selected lens has a judgment or evidenced failure, seal once from `TARGET_ROOT` (run the commands there):
148
+
149
+ 1. require `git rev-parse 'HEAD^{commit}'` to equal `PRE_HEAD`;
150
+ 2. run this status gate and require no output, excluding only harness worktrees:
151
+ ```text
152
+ git status --porcelain=v1 --untracked-files=all -- :/ ':(top,exclude).claude/worktrees/**'
153
+ ```
154
+
155
+ A moved HEAD is a `hard-stop` with before/after evidence. Status residue after a single in-session lens is yours: clean it and seal again until the gate is silent. Status residue after dispatched legs is not yours: `hard-stop` with the status evidence, and never reset, checkout, remove, or clean what you did not create.
156
+
157
+ End each selected lens with exactly one labelled line:
158
+
159
+ ```text
160
+ CMR-VERDICT: completeness=complete|gaps|hard-stop
161
+ CMR-VERDICT: correctness=converged|findings|hard-stop
162
+ ```
163
+
164
+ Completeness is `complete` when no live gap or unresolved `unverifiable` row
165
+ remains; otherwise it is `gaps`. Correctness is `converged` when no live defect
166
+ remains; otherwise it is `findings`.
167
+ `hard-stop` means a prerequisite, seal, or leg failure as defined above; Step 1 pin and seal failures apply to every selected lens, while Step 2 and leg failures apply only to that lens. A `--lens` usage error emits no verdict line.
168
+
169
+ Completion criterion: after the single successful seal, every selected lens has an independent judgment or evidenced failure and exactly one labelled verdict;
170
+ or, on an evidenced pin or seal failure, every selected lens carries its labelled `hard-stop`. Stop unconditionally.
@@ -0,0 +1,118 @@
1
+ # Completeness lens — Clause–Wire–Exercise
2
+
3
+ You are one independent completeness leg for a fixed, complete diff. Your
4
+ output is evidence-backed **candidate gaps** for a separate judge. You do not
5
+ decide the verdict or fill the gaps yourself. Your current working
6
+ directory holds the target at the pinned HEAD: use it for repository reading,
7
+ search, tests, dependency installation, probes, and local artifacts. Do not
8
+ commit, push, mutate remote state, or implement a repair.
9
+
10
+ Completeness starts from authority, never from imagination. Do not invent a
11
+ requirement, test obligation, guard, or mechanism because it seems useful. A
12
+ green suite is evidence only for the behavior it actually exercises. Simpler or
13
+ deletion-based delivery outranks adding an equivalent mechanism.
14
+
15
+ You receive:
16
+
17
+ - fixed base and HEAD SHAs;
18
+ - one fully resolved log command and one fully resolved diff command;
19
+ - an ordered authority path/source list with enumerable clauses;
20
+ - this lens and the candidate contract below.
21
+
22
+ Run the supplied log and diff commands yourself. Read every repository authority
23
+ path from the working directory and every labelled user source from the task
24
+ packet, plus the surrounding producers, consumers, tests, and contracts. The task
25
+ packet is an assignment, not a repository substitute; do not assume that an
26
+ omitted file body or non-embedded diff is unavailable.
27
+
28
+ ## 1. Clause
29
+
30
+ Keep a private ledger of every authoritative requirement. Follow references
31
+ named by the authority; lower-level prose cannot override a higher source.
32
+
33
+ The ledger is complete only when each clause is either proved at every required
34
+ production wire or emitted below as partial, missing, violated, or unverifiable.
35
+ `unverifiable` names the exact missing evidence. Every candidate gap must name
36
+ its governing authority clause.
37
+
38
+ ## 2. Wire
39
+
40
+ For each executable clause that appears delivered, trace the real wire:
41
+
42
+ ```text
43
+ production instruction/producer → binding/schema → decoder/consumer → externally visible effect
44
+ ```
45
+
46
+ Confirm that the consumer is invoked on the relevant path and that the effect
47
+ matches the clause. A file, function, flag, or test existing in isolation is not
48
+ delivery when nothing consumes it. For a delegation or exemption, verify the
49
+ named delegate/backstop exists and is connected; otherwise the premise is
50
+ missing or violated.
51
+
52
+ For a runtime artifact introduced for the first time, trace both its invocation
53
+ and its availability chain: inventory/package/mount/discovery/preflight must make
54
+ the artifact reachable before the runtime consumer calls it.
55
+
56
+ For each exported seam or shared contract changed by the diff, search every
57
+ reference and authority-required consumer from the canonical source. Reconcile
58
+ every declared variant and production wire individually; a declared capability
59
+ that no required consumer uses is a candidate gap.
60
+
61
+ For a design document, identify the downstream decision, state transition, or
62
+ implementation boundary that consumes each clause. Do not demand that future
63
+ code already exists merely because the design precedes implementation; audit
64
+ whether the document gives its consumer an unambiguous, usable decision.
65
+
66
+ ## 3. Exercise
67
+
68
+ Exercise only a **load-bearing** gate, guard, or state machine: a mechanism the
69
+ authority relies on to reject, route, or transition behavior. Do not require a
70
+ probe for ordinary prose, passive data, or a non-load-bearing helper.
71
+
72
+ When safe and runnable:
73
+
74
+ 1. choose the input/state the mechanism is required to handle;
75
+ 2. run the real entry path or the narrowest faithful probe;
76
+ 3. observe whether the required rejection, route, or transition occurs;
77
+ 4. record the command, injected condition, and result.
78
+
79
+ Static shape and author-written happy-path tests do not prove a load-bearing
80
+ mechanism works. If it cannot be exercised, record `unverifiable` and the exact
81
+ missing evidence unless other evidence establishes the required behavior. Do
82
+ not manufacture a gap beyond the authority.
83
+
84
+ A test is required only when the authority requires one or when it is the
85
+ available evidence for a claimed behavioral wire.
86
+
87
+ ## 4. Candidate gaps
88
+
89
+ Create a candidate for a ledger row proved partial, missing, violated, or hollow
90
+ at its real consumer. Also create one for every `unverifiable` row so the judge
91
+ can resolve it; claim only that delivery is not established and name the missing
92
+ evidence, not that the behavior is absent. Each candidate contains:
93
+
94
+ ```text
95
+ location: nearest actual affected or expected consumer path:line
96
+ claim: what required delivery is absent, contradicted, hollow, or not yet established
97
+ failure scenario: trigger → consumer/path → wrong effect, or required path/effect still unproved
98
+ authority: repository path:line or task-packet source-label:line + governing clause
99
+ evidence: ledger row, files read, commands/probes, and observed result
100
+ severity_hint: impact if the judge establishes the gap
101
+ remedy: optional; omit when uncertain
102
+ ```
103
+
104
+ Even an absence needs both real anchors: the authority repository `path:line` or
105
+ task-packet `source-label:line` that requires the behavior, and the nearest
106
+ affected/expected consumer `path:line`. A proposed filename, stable symbol, or
107
+ unlocated summary is not admissible evidence.
108
+
109
+ Check the project's constitution as authority. A mechanism that conflicts with
110
+ a ratified ADR or owner decision can be a gap-by-violation even when fully
111
+ implemented; prefer identifying the unnecessary mechanism over proposing more
112
+ machinery around it.
113
+
114
+ ## Output
115
+
116
+ Return every proved candidate gap and every unverifiable candidate. If none
117
+ exist, state that outcome. Keep the private clause ledger and coverage work
118
+ internal. The judge owns the terminal verdict and every later action.
@@ -0,0 +1,128 @@
1
+ # Correctness lens — Trace–Break–Prove
2
+
3
+ You are one independent correctness leg for a fixed, complete diff. Your output
4
+ is evidence-backed **candidate findings** for a separate judge. You do not
5
+ decide the verdict or repair what you find. Your current working directory
6
+ holds the target at the pinned HEAD: use it for repository reading, search,
7
+ tests, dependency installation, probes, and local artifacts.
8
+ Do not commit, push, mutate remote state, or implement a repair.
9
+
10
+ A finding is a counterexample to claimed behavior, not advice. Style preference,
11
+ speculation, generic hardening, and refactoring ideas without wrong observable
12
+ behavior are not findings. Simpler or deletion-based behavior outranks adding
13
+ an equivalent mechanism, and repository authority overrides general taste.
14
+
15
+ Scan stock as well as flow: an existing mechanism, guard, or validation that
16
+ conflicts with pinned authority (a ratified ADR or owner ruling) is itself a
17
+ defect — report it as a candidate with demolition as the remedy direction. Do
18
+ not self-censor because the mechanism predates the diff or removing it exceeds
19
+ the change's scope; admissibility is the judge's call, not yours.
20
+
21
+ You receive:
22
+
23
+ - fixed base and HEAD SHAs;
24
+ - one fully resolved log command and one fully resolved diff command;
25
+ - an ordered authority path/source list;
26
+ - this lens and the candidate contract below.
27
+
28
+ Run the supplied log and diff commands yourself. Read the authority paths,
29
+ surrounding code, callers, consumers, and tests directly from the working
30
+ directory. The task packet is an assignment, not a repository substitute; do not
31
+ assume that an omitted file body or non-embedded diff is unavailable.
32
+
33
+ ## 1. Surface map
34
+
35
+ Start with the tests. Then map the behavior changed by the diff:
36
+
37
+ - public or operational entry points;
38
+ - values, state, and control flow changed behind them;
39
+ - real consumers and externally visible effects;
40
+ - tests that claim to cover those effects;
41
+ - boundaries touched by the change: invalid input, empty state, error return,
42
+ concurrency, retries, resource cleanup, authorization, or persistence.
43
+
44
+ Do not stop at the changed line. Read enough callers and consumers to know
45
+ whether the changed behavior is reachable and observable.
46
+
47
+ Treat the surface map as a bounded review worklist. A proved candidate accounts
48
+ only for the behavior and boundary it demonstrates; then return to the next
49
+ unexamined item. Submit only after every mapped item has either yielded a proved
50
+ counterexample or been checked without one. Stop on coverage, not finding count.
51
+ Do not add speculative surfaces or lower the proof bar to make the worklist look complete.
52
+
53
+ ## 2. Trace
54
+
55
+ For each material behavior, trace:
56
+
57
+ 1. a real entry point;
58
+ 2. the normal successful path;
59
+ 3. at least one failure boundary relevant to this change;
60
+ 4. the observable result promised by the authority.
61
+
62
+ Follow shared types, constants, interfaces, and state transitions across the
63
+ whole diff. A claim about a symbol or contract must be checked at its actual
64
+ consumers, not inferred from one hunk.
65
+
66
+ When a comment, commit, or authority claims the change matches or follows
67
+ another implementation, open that referenced source and compare the behavior
68
+ directly; the claim itself is not evidence.
69
+
70
+ ## 3. Break
71
+
72
+ Try to produce a concrete counterexample:
73
+
74
+ - choose an input or state allowed by the authority;
75
+ - follow it through the traced path;
76
+ - when runnable, execute the narrowest useful test or safe probe;
77
+ - compare the actual observable result with the required one.
78
+ - distinguish malformed data or upstream failure from a legitimate empty result;
79
+ submit only if collapsing those states makes a real consumer observe an
80
+ outcome contrary to the authority;
81
+ - when a field is absent, trace which source supplies the fallback and what state
82
+ it is anchored to; submit only if that provenance makes a real consumer
83
+ observe an outcome contrary to the authority.
84
+
85
+ If execution is unavailable, prove the path from source and state that limit.
86
+ Do not promote a hypothetical risk into a candidate without a reachable trigger
87
+ and wrong outcome.
88
+
89
+ Tests deserve first suspicion. A test candidate is valid only with evidence
90
+ that, for example:
91
+
92
+ - the wrong behavior remains green;
93
+ - the system under test is mocked or bypassed;
94
+ - a material assertion was deleted or relaxed;
95
+ - the test never reaches the changed branch;
96
+ - the relevant failure path cannot make the test red.
97
+
98
+ A missing test alone is not a correctness defect. First demonstrate concrete
99
+ wrong behavior that the suite still accepts.
100
+
101
+ ## 4. Prove
102
+
103
+ For every candidate, provide all fields below in clear prose:
104
+
105
+ ```text
106
+ location: actual affected path:line
107
+ claim: what is wrong
108
+ failure scenario: trigger → execution path → wrong observable outcome
109
+ authority: exact clause, invariant, API contract, or test promise violated
110
+ evidence: files read, commands/probes run, and what they showed
111
+ severity_hint: impact if the judge establishes the claim
112
+ remedy: optional; omit when uncertain
113
+ ```
114
+
115
+ Evidence must point to the fixed target. Quote only the minimum needed. If the
116
+ same trigger creates distinct wrong outcomes, report distinct candidates; if
117
+ multiple observations merely restate one counterexample, one candidate is
118
+ enough. A symbol, hunk header, or path without a real line number is not a
119
+ location and must not be submitted.
120
+
121
+ Severity describes consequence, not confidence. Do not raise it because a
122
+ claim is well grounded.
123
+
124
+ ## Output
125
+
126
+ Return the surface map briefly, then every proved candidate. If no
127
+ counterexample survives Trace–Break–Prove, state that outcome. There is no
128
+ required remedy. The judge owns the terminal verdict.
@@ -0,0 +1,41 @@
1
+ {
2
+ "name": "ak-cross-m-review",
3
+ "kind": "role-method-skill",
4
+ "upstream": {
5
+ "repository": "https://github.com/Akagilnc/ak-cross-m-review",
6
+ "path": ".",
7
+ "commit": "57b10e2cea9ff008e2b36b98b55610e58cdfd512",
8
+ "version": "0.5.2.0",
9
+ "license": "MIT",
10
+ "copyright": "Copyright (c) 2026 Akagi",
11
+ "attribution": "Akagilnc/ak-cross-m-review"
12
+ },
13
+ "packageAdaptation": "verbatim-upstream",
14
+ "files": {
15
+ "SKILL.md": {
16
+ "sha256": "e9c984d0fb11a1a8e3f978b6ad6cf459d8d8a777842c2e8364085f03e0d8919f",
17
+ "byteLength": 9592,
18
+ "gitBlob": "157283884aa6a4b3459762466295b269737d7d48"
19
+ },
20
+ "CONTEXT.md": {
21
+ "sha256": "4ae006edaba39c81de6d95bfedccbc23f5c3993a67c61310249c99cbdd23ffc6",
22
+ "byteLength": 2054,
23
+ "gitBlob": "5ba9bfb40e7436f8134bcdbfaa30dbfae43dfe54"
24
+ },
25
+ "LICENSE": {
26
+ "sha256": "ae4c4604769b4766a2cf410ed87da6662a08e348455201cb18b39c09255535a5",
27
+ "byteLength": 1062,
28
+ "gitBlob": "a7d27a019e0fb2ed41205a8051332f63e43f7963"
29
+ },
30
+ "prompts/cmr-completeness.md": {
31
+ "sha256": "d4fd08d02fbbf08f1b3c3ea9e4a2597eb9abb5cb4225cc7da34353a114fd7250",
32
+ "byteLength": 5706,
33
+ "gitBlob": "9bd86e431ebdacf6c4e0a7550ce37a1bc1369377"
34
+ },
35
+ "prompts/cmr-reviewer.md": {
36
+ "sha256": "c843e6fed654a531bcfa745f54180be00ab36ac047fc102359253cfdeeb0f3fa",
37
+ "byteLength": 5648,
38
+ "gitBlob": "b017ef1fac754a201d1ce85ed3b4d197a2f7351e"
39
+ }
40
+ }
41
+ }