axstack 0.20.21 → 0.20.23

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -99,7 +99,7 @@ upgrades, conflicts, and uninstalling.
99
99
  inline or from an explicitly scoped backlog without touching active, manual,
100
100
  uncertain, user-owned, dirty, unpushed, or useful unmerged work.
101
101
 
102
- Choose an explicit role preset:
102
+ Choose an explicit role preset (26 stable role IDs in each):
103
103
  [mixed](profiles/presets/mixed.json),
104
104
  [codex-only](profiles/presets/codex-only.json), or
105
105
  [claude-only](profiles/presets/claude-only.json).
package/bin/axstack.js CHANGED
@@ -335,6 +335,7 @@ async function main() {
335
335
  } else {
336
336
  if (summary.added.length) console.log(`added: ${summary.added.join(', ')}`);
337
337
  if (summary.updated.length) console.log(`updated: ${summary.updated.join(', ')}`);
338
+ if (summary.addedRoleIds?.length) console.log(`added role IDs: ${summary.addedRoleIds.join(', ')}`);
338
339
  if (summary.removed.length) console.log(`removed: ${summary.removed.join(', ')}`);
339
340
  if (summary.unchanged.length) console.log(`unchanged: ${summary.unchanged.join(', ')}`);
340
341
  }
@@ -71,7 +71,7 @@ profiles/presets/codex-only.json
71
71
  profiles/presets/claude-only.json
72
72
  ```
73
73
 
74
- Each has exactly `{ "version": 1, "roles": [...] }` with the same 24 stable
74
+ Each has exactly `{ "version": 1, "roles": [...] }` with the same 26 stable
75
75
  role IDs. Installation writes `<skills-dir>/axstack/roles.json` as
76
76
  `{ "version": 1, "preset": "<selected preset>", "roles": [...] }` and records
77
77
  its ownership hash like every other installed skill asset. There is no second
@@ -140,10 +140,10 @@ The complete bundle is validated before writes:
140
140
  the filename's selected identity supplied by the caller, and the same role-ID
141
141
  set as its peers;
142
142
  - every role has valid preserved fields, while the mixed checker,
143
- `axstack-research-web-google`, and `axstack-research-x` launch-by-agent-id
143
+ `axstack-research-web-google`, `axstack-research-x`, and both arena candidate launch-by-agent-id
144
144
  routes explicitly permit `model: null`;
145
145
  in each single-provider preset, the unavailable adviser and its matching arena
146
- judge seat explicitly permit `model: null`, as do both cross-provider research routes;
146
+ judge seat explicitly permit `model: null`, as do both cross-provider research routes and both arena candidate seats;
147
147
  - obsolete runtime configuration flags fail before mutation with migration
148
148
  guidance.
149
149
 
@@ -171,7 +171,7 @@ to rewrite them.
171
171
  ## Role behavior after installation
172
172
 
173
173
  The runtime reads `roles.json` from the installed shared root `skills/axstack/`.
174
- A new run records the selected preset plus all 24 role rows. An active run keeps
174
+ A new run records the selected preset plus all 26 role rows. An active run keeps
175
175
  that snapshot after a later preset install unless the user explicitly changes
176
176
  it and accepts the resulting evidence invalidation.
177
177
 
package/docs/workflows.md CHANGED
@@ -48,7 +48,7 @@ only affected work.
48
48
 
49
49
  Installation requires one explicit canonical preset. The three bundle files
50
50
  under `profiles/presets/` each contain exactly
51
- `{ "version": 1, "roles": [...] }` and the same 24 stable IDs.
51
+ `{ "version": 1, "roles": [...] }` and the same 26 stable IDs.
52
52
 
53
53
  The current chat drives on whatever model runs it; no preset carries a driver
54
54
  role.
@@ -122,9 +122,9 @@ session and evidence remain valid.
122
122
  - `axstack-align` maps facts and dependencies, asks prioritized questions, and
123
123
  consults Astra and Fable independently with the same bounded evidence and
124
124
  question. It synthesizes disagreements and reuses unchanged receipts. For a
125
- hard-to-reverse design choice it runs one arena round instead: Astra and
126
- Fable each author a candidate, `axstack-arena-judge-astra` and
127
- `axstack-arena-judge-fable` score both against the driver's rubric, and the
125
+ hard-to-reverse design choice it runs one arena round instead: Astra,
126
+ Fable, Grok, and Antigravity each author a candidate, `axstack-arena-judge-astra` and
127
+ `axstack-arena-judge-fable` score every candidate against the driver's rubric, and the
128
128
  driver picks a base, grafts the losers' strong ideas, and presents the
129
129
  synthesis as the recommendation; the note lands as `Decisions` rows in the
130
130
  run record.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "axstack",
3
- "version": "0.20.21",
3
+ "version": "0.20.23",
4
4
  "description": "Axstack installer and setup CLI: installs owned chat skills and role data, configures supported harness settings, and checks Orca capabilities.",
5
5
  "keywords": [
6
6
  "claude-code",
@@ -206,7 +206,7 @@
206
206
  "model": null,
207
207
  "modeId": "bypassPermissions",
208
208
  "thinkingOptionId": "xhigh",
209
- "notes": "Required Arena judge (Astra seat). Read-only cross-judge for an arena-grade Align decision: receives the rubric and both candidates by label, scores each criterion, and recommends a base with rationale. Never authors a candidate, never cross-reads the other judge, never mutates. Disagreement between judges is surfaced by the driver, never averaged. Intentionally absent in the claude-only preset; the arena-grade decision holds without substitution."
209
+ "notes": "Required Arena judge (Astra seat). Read-only cross-judge for an arena-grade Align decision: receives the rubric and every candidate by label, scores each criterion, and recommends a base with rationale. Never authors a candidate, never cross-reads the other judge, never mutates. Disagreement between judges is surfaced by the driver, never averaged. Intentionally absent in the claude-only preset; the arena-grade decision holds without substitution."
210
210
  },
211
211
  {
212
212
  "id": "axstack-arena-judge-fable",
@@ -215,7 +215,25 @@
215
215
  "model": "claude-fable-5-1",
216
216
  "modeId": "bypassPermissions",
217
217
  "thinkingOptionId": "xhigh",
218
- "notes": "Arena judge (Fable seat). Read-only cross-judge for an arena-grade Align decision: receives the rubric and both candidates by label, scores each criterion, and recommends a base with rationale. Never authors a candidate, never cross-reads the other judge, never mutates. Disagreement between judges is surfaced by the driver, never averaged."
218
+ "notes": "Arena judge (Fable seat). Read-only cross-judge for an arena-grade Align decision: receives the rubric and every candidate by label, scores each criterion, and recommends a base with rationale. Never authors a candidate, never cross-reads the other judge, never mutates. Disagreement between judges is surfaced by the driver, never averaged."
219
+ },
220
+ {
221
+ "id": "axstack-arena-candidate-grok",
222
+ "name": "Axstack arena candidate Grok (unavailable)",
223
+ "provider": "claude",
224
+ "model": null,
225
+ "modeId": "bypassPermissions",
226
+ "thinkingOptionId": "high",
227
+ "notes": "Intentional single-provider absence in the claude-only preset; Grok candidate unavailable. Arena holds without substitution."
228
+ },
229
+ {
230
+ "id": "axstack-arena-candidate-antigravity",
231
+ "name": "Axstack arena candidate Antigravity (unavailable)",
232
+ "provider": "claude",
233
+ "model": null,
234
+ "modeId": "bypassPermissions",
235
+ "thinkingOptionId": "high",
236
+ "notes": "Intentional single-provider absence in the claude-only preset; Antigravity candidate unavailable. Arena holds without substitution."
219
237
  }
220
238
  ]
221
239
  }
@@ -206,7 +206,7 @@
206
206
  "model": "gpt-6-astra",
207
207
  "modeId": "full-access",
208
208
  "thinkingOptionId": "xhigh",
209
- "notes": "Arena judge (Astra seat). Read-only cross-judge for an arena-grade Align decision: receives the rubric and both candidates by label, scores each criterion, and recommends a base with rationale. Never authors a candidate, never cross-reads the other judge, never mutates. Disagreement between judges is surfaced by the driver, never averaged."
209
+ "notes": "Arena judge (Astra seat). Read-only cross-judge for an arena-grade Align decision: receives the rubric and every candidate by label, scores each criterion, and recommends a base with rationale. Never authors a candidate, never cross-reads the other judge, never mutates. Disagreement between judges is surfaced by the driver, never averaged."
210
210
  },
211
211
  {
212
212
  "id": "axstack-arena-judge-fable",
@@ -215,7 +215,25 @@
215
215
  "model": null,
216
216
  "modeId": "full-access",
217
217
  "thinkingOptionId": "xhigh",
218
- "notes": "Required Arena judge (Fable seat). Read-only cross-judge for an arena-grade Align decision: receives the rubric and both candidates by label, scores each criterion, and recommends a base with rationale. Never authors a candidate, never cross-reads the other judge, never mutates. Disagreement between judges is surfaced by the driver, never averaged. Intentionally absent in the codex-only preset; the arena-grade decision holds without substitution."
218
+ "notes": "Required Arena judge (Fable seat). Read-only cross-judge for an arena-grade Align decision: receives the rubric and every candidate by label, scores each criterion, and recommends a base with rationale. Never authors a candidate, never cross-reads the other judge, never mutates. Disagreement between judges is surfaced by the driver, never averaged. Intentionally absent in the codex-only preset; the arena-grade decision holds without substitution."
219
+ },
220
+ {
221
+ "id": "axstack-arena-candidate-grok",
222
+ "name": "Axstack arena candidate Grok (unavailable)",
223
+ "provider": "codex",
224
+ "model": null,
225
+ "modeId": "full-access",
226
+ "thinkingOptionId": "high",
227
+ "notes": "Intentional single-provider absence in the codex-only preset; Grok candidate unavailable. Arena holds without substitution."
228
+ },
229
+ {
230
+ "id": "axstack-arena-candidate-antigravity",
231
+ "name": "Axstack arena candidate Antigravity (unavailable)",
232
+ "provider": "codex",
233
+ "model": null,
234
+ "modeId": "full-access",
235
+ "thinkingOptionId": "high",
236
+ "notes": "Intentional single-provider absence in the codex-only preset; Antigravity candidate unavailable. Arena holds without substitution."
219
237
  }
220
238
  ]
221
239
  }
@@ -206,7 +206,7 @@
206
206
  "model": "gpt-6-astra",
207
207
  "modeId": "full-access",
208
208
  "thinkingOptionId": "xhigh",
209
- "notes": "Arena judge (Astra seat). Read-only cross-judge for an arena-grade Align decision: receives the rubric and both candidates by label, scores each criterion, and recommends a base with rationale. Never authors a candidate, never cross-reads the other judge, never mutates. Disagreement between judges is surfaced by the driver, never averaged."
209
+ "notes": "Arena judge (Astra seat). Read-only cross-judge for an arena-grade Align decision: receives the rubric and every candidate by label, scores each criterion, and recommends a base with rationale. Never authors a candidate, never cross-reads the other judge, never mutates. Disagreement between judges is surfaced by the driver, never averaged."
210
210
  },
211
211
  {
212
212
  "id": "axstack-arena-judge-fable",
@@ -215,7 +215,25 @@
215
215
  "model": "claude-fable-5-1",
216
216
  "modeId": "bypassPermissions",
217
217
  "thinkingOptionId": "xhigh",
218
- "notes": "Arena judge (Fable seat). Read-only cross-judge for an arena-grade Align decision: receives the rubric and both candidates by label, scores each criterion, and recommends a base with rationale. Never authors a candidate, never cross-reads the other judge, never mutates. Disagreement between judges is surfaced by the driver, never averaged."
218
+ "notes": "Arena judge (Fable seat). Read-only cross-judge for an arena-grade Align decision: receives the rubric and every candidate by label, scores each criterion, and recommends a base with rationale. Never authors a candidate, never cross-reads the other judge, never mutates. Disagreement between judges is surfaced by the driver, never averaged."
219
+ },
220
+ {
221
+ "id": "axstack-arena-candidate-grok",
222
+ "name": "Axstack arena candidate Grok",
223
+ "provider": "grok",
224
+ "model": null,
225
+ "modeId": "full-access",
226
+ "thinkingOptionId": "high",
227
+ "notes": "Independent Grok Rung 2 design candidate. Launch by agent id grok; model selected by the TUI default. Receives the same brief without cross-reading; returns design, rationale, and rejected alternatives."
228
+ },
229
+ {
230
+ "id": "axstack-arena-candidate-antigravity",
231
+ "name": "Axstack arena candidate Antigravity",
232
+ "provider": "antigravity",
233
+ "model": null,
234
+ "modeId": "full-access",
235
+ "thinkingOptionId": "high",
236
+ "notes": "Independent Antigravity Rung 2 design candidate. Launch by agent id antigravity; model selected by the TUI default. Receives the same brief without cross-reading; returns design, rationale, and rejected alternatives."
219
237
  }
220
238
  ]
221
239
  }
@@ -0,0 +1,81 @@
1
+ # Design lens
2
+
3
+ In Align, set the rung from researched facts, not a user choice. Carry the
4
+ sketch through the scope identity.
5
+
6
+ ## Ladder
7
+
8
+ - **Rung 0 — skip.** The change stays inside one module's existing interface,
9
+ ownership, data flow, and failure guarantees. Ask no design questions and
10
+ carry no sketch. “Might be small” is not a reason to skip.
11
+ - **Rung 1 — design questions.** Any change that fails Rung 0. Typical triggers:
12
+ crossing a module boundary; adding a module, service, interface, or external
13
+ dependency; changing an interface or failure guarantee; changing data
14
+ ownership, schema, wire format, or persisted format. A design question does
15
+ not reclassify work: Rung 1 can stay small. An unsettled material design
16
+ question still makes routing reassess size.
17
+ - **Rung 2 — arena.** A Rung 1 design that also meets the existing ADR test:
18
+ a meaningful, hard-to-reverse, non-obvious trade-off. Use Align's all-family
19
+ arena for that question.
20
+
21
+ There is no numeric threshold, file-count gate, or class-count gate.
22
+
23
+ ## Questions in order
24
+
25
+ Ask only unresolved areas in numbered `Qn` rounds of one to three. Give each a
26
+ recommendation, reason, and trade-off; use Align's adviser critique. Design and
27
+ arena questions share its unchanged budget: 20 normally, a justified extension
28
+ to 35, then opted-in refinement of at most five.
29
+
30
+ 1. **Scope:** What exists, the minimum change, and what is explicitly excluded?
31
+ 2. **Caller first:** Write realistic usage before drawing the shape.
32
+ 3. **Shape:** Which boundaries, module depth, ownership, invariants, seams,
33
+ and adapters serve that usage?
34
+ 4. **Flow and failure:** Trace the data flow. For each materially new path,
35
+ name one realistic production failure and its observable behavior.
36
+ 5. **Reversibility:** Name compatibility, migration, rejected alternatives, and
37
+ what we accept for what benefit.
38
+
39
+ ## Vocabulary and red flags
40
+
41
+ A **deep module** hides substantial behavior behind a small interface. A
42
+ **shallow module** exposes most of its machinery. Treat an **interface as test
43
+ surface**: test caller behavior and observable failures. A **seam** permits a
44
+ second implementation; one adapter is hypothetical, two make it real. Prefer
45
+ **locality** and explicit **ownership** of state and invariants. The **deletion
46
+ test** asks what contract would be lost if a layer vanished. Choose **boring by
47
+ default** and **reversible over clever**.
48
+
49
+ Red flags—shallow module, information leakage, temporal decomposition, and
50
+ pass-through layers—are prompts for evidence, not automatic defects. Ask what
51
+ crosses each boundary and whether a layer can be removed.
52
+
53
+ ## Sketch
54
+
55
+ For Rung 1 or 2, put this block in the spec's `Design` section or the returned
56
+ small-change intent. `Binding` names commitments; other lines may illustrate.
57
+
58
+ ```text
59
+ Usage: <call site or command as the caller writes it>
60
+ Shape: <signatures or a <=10-line ASCII/Mermaid diagram>
61
+ Binding: <which lines are committed; the rest is illustrative>
62
+ Flow + failure: <path -> one realistic failure -> observable behavior>
63
+ We accept: <X> for <Y>
64
+ Rejected: <alternative -> why>
65
+ Open: <question?>
66
+ ```
67
+
68
+ Keep settled choices in `Decisions` rows. `CONTEXT.md` stays a glossary; ADRs
69
+ keep their existing test. Add no design document or phase.
70
+
71
+ ## Arena rubric candidates
72
+
73
+ For each arena, the driver selects three to six and tailors the rubric criteria
74
+ to the question. Score candidate designs on evidence, not a generic checklist:
75
+
76
+ - Is `Usage` realistic and simple for the caller?
77
+ - Are module boundaries deep enough and ownership unambiguous?
78
+ - Is behavior local with a useful test surface?
79
+ - Are materially new failure paths observable and recoverable?
80
+ - Are compatibility, migration, and deletion costs explicit?
81
+ - Does the trade-off justify complexity against rejected options?
@@ -38,7 +38,7 @@ Read `roles.json` from the installed shared root `skills/axstack/`. The installe
38
38
  shape is `{ "version": 1, "preset": "<name>", "roles": [...] }`. Bundled
39
39
  profiles are setup inputs shaped as
40
40
  `{ "version": 1, "roles": [...] }`. A new run records the selected preset and
41
- all 24 role rows once. An active run keeps the exact snapshot until the user
41
+ all 26 role rows once. An active run keeps the exact snapshot until the user
42
42
  explicitly changes it.
43
43
 
44
44
  Select the requested role by stable ID. A missing or null model holds only that role;
@@ -5,19 +5,19 @@ Driver entry sweep follows [Workspace hygiene](workspace-hygiene.md).
5
5
 
6
6
  ## Role routing
7
7
 
8
- Presets: `mixed`, `codex-only`, `claude-only`. For a new run, read
8
+ Presets: `mixed`, `codex-only`, `claude-only`. For new runs, use
9
9
  `profiles.preset` from `.axstack-manifest.json` at the actually loaded
10
- skills root, or an explicit user selection recorded in the run record. Proceed
11
- only with exactly one unambiguous preset; missing or contradictory sources are
10
+ skills root, or an explicit user selection recorded in the run record. Use one
11
+ unambiguous preset; missing or contradictory sources are
12
12
  a setup gap: hold. Never infer from live profiles or `list_profiles`, harness,
13
13
  tools, credentials, quota, subscription, or default to `mixed`.
14
14
 
15
- At run start, snapshot all 24 role IDs with provider/model/mode/effort; absent
15
+ At run start, snapshot all 26 role IDs with provider/model/mode/effort; absent
16
16
  or unconfigured roles are recorded explicitly; invent no provider default.
17
- Such a role holds only that role's work, not the run. A role installed or changed later
18
- must not silently enter the snapshot; adding it needs an explicit user
19
- decision. Live profiles are authoritative at snapshot time and for availability;
20
- bundled presets are setup inputs, not runtime proof.
17
+ Such a role holds only that role's work. A role installed or changed later must not
18
+ silently enter the snapshot; adding it needs an explicit user decision. Live profiles
19
+ are authoritative at snapshot time and for availability; bundled presets are setup
20
+ inputs, not runtime proof.
21
21
 
22
22
  Preset changes apply to new runs only; an active run keeps its snapshot.
23
23
  Changing it or replacing a session needs an explicit user decision and
@@ -25,8 +25,6 @@ revalidation. Unavailable models, unsupported efforts, missing roles, and
25
25
  incompatible overrides hold only affected work; no automatic fallback, quota
26
26
  routing, subscription inference, or silent provider/model/effort substitution.
27
27
 
28
- Role IDs:
29
-
30
28
  - Chat drives (no role ID); `axstack-owner` owns one PR and
31
29
  `axstack-author` its sole writer.
32
30
  - `axstack-reviewer-primary` and `axstack-reviewer-secondary` are the ordered
@@ -38,10 +36,11 @@ Role IDs:
38
36
  | `mixed` | Claude / Opus (`claude/claude-opus-5-5`) | `axstack-reviewer-primary` (`codex/gpt-6-sol` medium) |
39
37
  | `codex-only` | Codex / Sol (`codex/gpt-6-sol`) | `axstack-reviewer-secondary` (`codex/gpt-6-luna` xhigh) |
40
38
  | `claude-only` | Claude / Opus (`claude/claude-opus-5-5`) | `axstack-reviewer-secondary` (`claude/claude-sonnet-5` xhigh) |
41
- - `axstack-advisor-astra` and `axstack-advisor-fable` advise independently
42
- and author align arena candidates; `axstack-arena-judge-astra` and
43
- `axstack-arena-judge-fable` judge them. `axstack-auditor` audits;
44
- `axstack-checker` reports discrepancies.
39
+ - `axstack-advisor-astra`/`axstack-advisor-fable` advise independently and
40
+ author arena candidates; `axstack-arena-candidate-grok`/
41
+ `axstack-arena-candidate-antigravity` add families.
42
+ `axstack-arena-judge-astra`/`axstack-arena-judge-fable` judge them.
43
+ `axstack-auditor` audits; `axstack-checker` reports discrepancies.
45
44
  - `axstack-explainer`/`axstack-explainer-review`: explain/review.
46
45
  `axstack-monitor`: standalone watch never sends; chat-run watch: bounded
47
46
  internal reports to its Run and original driver.
@@ -51,17 +50,16 @@ Provenance is matched on provider/model ID; effort never maps. Missing table-row
51
50
  provenance is unsupported and `INCOMPLETE`; report it and ask the user. Never
52
51
  infer from slot, driver, owner, or provider. Author and owner never review.
53
52
 
54
- The `axstack-implement` loop requires `mixed`; single-provider presets hold at
55
- step (3) for user routing, with no substitution or same-provider review.
53
+ `axstack-implement` requires `mixed`; single-provider presets hold at
54
+ step (3) for user routing: no substitution or same-provider review.
56
55
 
57
56
  ## Direct routes (no spec ceremony)
58
57
 
59
58
  - One bounded research question -> `axstack-research`: verify primary sources
60
- and code, return a cited note with limitations. Fan out only distinct
61
- questions.
59
+ and code; return a cited note with limitations. Fan out only distinct questions.
62
60
  - Understanding a system, change, or implementation gap -> `axstack-explain`:
63
- current/intended behavior, evidence dimensions, and bounded gaps from project
64
- docs and rendered behavior. "What could this break" follows
61
+ current/intended behavior, evidence dimensions, bounded gaps from project docs and
62
+ rendered behavior. "What could this break" follows
65
63
  [Blast radius](blast-radius.md). Publication needs separate authority.
66
64
  - A bug, failing test, regression, or wrong behavior, red loop wanted ->
67
65
  `axstack-debug`: diagnose, escalate via adviser-directed investigators, hand
@@ -69,11 +67,11 @@ step (3) for user routing, with no substitution or same-provider review.
69
67
  - Codebase-quality or refactor discovery -> `axstack-improve`: inspect bounded
70
68
  scope, rank evidenced candidates, report only; no spec, tickets, or source
71
69
  edits.
72
- - Accepted worker/Task/Run completion or bounded backlog request -> invoke
73
- `axstack-cleanup` inline in the driver; never dispatch it.
70
+ - Accepted worker/Task/Run completion or bounded backlog request -> driver invokes
71
+ `axstack-cleanup` inline; never dispatch it.
74
72
  - Preparation completion, watch expiry, resume, or reconciliation -> the
75
- [lifecycle](lifecycle.md#native-handoff-and-resume): reconcile the run
76
- record, keep its owner, launch no native handoff.
73
+ [lifecycle](lifecycle.md#native-handoff-and-resume): reconcile run record,
74
+ keep owner, launch no native handoff.
77
75
  - Explicit user-requested ownership transfer -> the same lifecycle section.
78
76
  Load the [Orca runtime boundary](orca-runtime.md), follow the runtime-owned
79
77
  handoff guide, and require explicit recipient acceptance before ownership
@@ -81,10 +79,10 @@ step (3) for user routing, with no substitution or same-provider review.
81
79
  - Colleague PR review -> `axstack-review`, peer mode.
82
80
  - Codebase review -> `axstack-review` codebase mode, report only.
83
81
  - A status question about an own open PR or stack ("check now", "what's left",
84
- "are we done", or "is it approved") -> `axstack-watch` in observation-only
82
+ "are we done", or "is it approved") -> `axstack-watch` observation-only
85
83
  mode. Explicit "address", "patch", or "fix" grants authorized maintenance.
86
- - Chat-run PR watch -> `axstack-watch`: original driver; verified run PRs
87
- and explicit adoptions only.
84
+ - Chat-run PR watch -> `axstack-watch`: original driver; verified run PRs and
85
+ explicit adoptions only.
88
86
  - Other own PR work -> `axstack-review` authored mode or `axstack-watch`
89
87
  adoption.
90
88
 
@@ -102,13 +100,13 @@ reason in the run record, or in the brief for tiny direct work.
102
100
  - **Small:** clear, bounded one-PR work. The driver captures the named
103
101
  **small-change intent** from the current request or user-chosen existing
104
102
  issue plus explicit acceptance checks and exclusions, snapshots it once, and
105
- proceeds. No earlier snapshot, spec, ticket ceremony, or second approval is
106
- required; do not route to `axstack-align` solely because that snapshot is
107
- not yet written. Strict TDD, mode-specific review, model, risk, and human-merge
103
+ proceeds. No prior snapshot, spec, tickets, or second approval is required; do not route
104
+ to `axstack-align` solely because the snapshot is not yet written. Strict TDD,
105
+ mode-specific review, model, risk, and human-merge
108
106
  contracts still apply.
109
- - **Unclear:** clarify the uncertainty through `axstack-align` or one bounded
110
- question, then classify it as small or substantial; a small ambiguity does
111
- not force substantial-work paperwork.
107
+ - **Unclear:** clarify via `axstack-align` or a bounded question, then
108
+ classify small or substantial; it does not force substantial-work paperwork.
109
+ [Design lens](design-lens.md) Rung 1 is Unclear; use `axstack-align`.
112
110
 
113
111
  Reassess size when growth adds an additional PR, a new execution dependency
114
112
  that materially expands scope, an unsettled material design question, or a
@@ -17,6 +17,17 @@ Load before acting:
17
17
 
18
18
  This preserves the required contracts -> lifecycle -> audit load edge.
19
19
 
20
+ ## Design the shape
21
+
22
+ Set the rung from researched facts; never ask the user to choose it. A change
23
+ inside one module's existing interface, ownership, data flow, and failure
24
+ guarantees is Rung 0: no design questions or sketch. Otherwise load the
25
+ [design lens ladder](../axstack/references/design-lens.md) for Rung 1 or 2
26
+ and settle only unresolved areas in its order within the existing budget. Carry a
27
+ Rung 1 or 2 sketch in the substantial spec's `Design` section or the returned
28
+ small-change intent. A design question alone does not make small work
29
+ substantial; apply routing's existing size reassessment rule.
30
+
20
31
  ## Settle the frontier
21
32
 
22
33
  1. **Research and map dependencies.** Inspect the available code, docs, and
@@ -70,29 +81,30 @@ hold Align; safe fact work may continue without substitution.
70
81
 
71
82
  ## Arena for hard-to-reverse design choices
72
83
 
73
- Critique of one draft anchors every reader to that draft's shape. When a
74
- question is arena-grade, the same test as for an ADR (a meaningful,
84
+ Critique of one draft anchors every reader to that draft's shape. Rung 2 designs
85
+ alone enter the arena: they meet the same test as for an ADR (a meaningful,
75
86
  hard-to-reverse, non-obvious trade-off: architecture, module boundaries, data
76
- model, migration strategy), replace the critique round for that question with
87
+ model, migration strategy). Replace the critique round for that question with
77
88
  one arena round. Small or routine questions never enter the arena.
78
89
 
79
90
  1. **Frame.** The driver writes the brief (the artifact, its constraints, the
80
91
  settled decisions it must respect) and three to six gradeable rubric
81
92
  criteria. Candidates receive only the brief; the rubric is for judging.
82
- 2. **Fan out.** `axstack-advisor-astra` and `axstack-advisor-fable` each
83
- independently produce one candidate design plus a short rationale naming
84
- the alternatives considered and rejected, from the same bounded evidence and
85
- question, without cross-reading. The driver authors no candidate.
86
- 3. **Cross-judge.** After both candidates are complete, `axstack-arena-judge-astra`
87
- and `axstack-arena-judge-fable` each independently score every candidate
88
- per criterion from the rubric and candidates by label, and recommend a base
89
- with a reason. Judges never author, never cross-read each other.
93
+ 2. **Fan out.** Produce one candidate per configured family independently from the same brief,
94
+ without cross-reading: `axstack-advisor-astra`, `axstack-advisor-fable`,
95
+ `axstack-arena-candidate-grok`, and `axstack-arena-candidate-antigravity`.
96
+ Each gives a design, rationale, and rejected alternatives. The driver authors no candidate.
97
+ 3. **Cross-judge.** After every candidate completes, give the judges anonymized,
98
+ relabeled candidates against the rubric.
99
+ `axstack-arena-judge-astra` and `axstack-arena-judge-fable` each independently
100
+ score every candidate against the driver-tailored rubric per criterion and
101
+ recommend a base with a reason. Judges never author, never cross-read each other.
90
102
  4. **Pick.** The driver reads every candidate end to end and scores per
91
103
  criterion, not on holistic feel, then compares with both judges. Agreement
92
104
  confirms the base. Disagreement between judges or with the driver means one
93
- reading is biased or the rubric was ambiguous: re-read both rationales and
105
+ reading is biased or the rubric was ambiguous: re-read the rationales and
94
106
  decide with a stated reason; never average verdicts or fabricate consensus.
95
- 5. **Graft.** Walk the losing candidate once more for the one or two ideas
107
+ 5. **Graft.** Walk the losing candidates once more for the one or two ideas
96
108
  worth porting and fold them into the base by hand so the result stays
97
109
  coherent under one mental model. Convergence on the same shape is a strong
98
110
  agreement signal: adopt the consensus shape, no graft. Wide divergence
@@ -106,9 +118,11 @@ Record the synthesis note (base, grafts and their source candidate, rejections,
106
118
  dropouts, both judge verdicts) as `Decisions` rows in the
107
119
  [run record](../axstack/references/run-record.md). Load
108
120
  [Orca runtime](../axstack/references/orca-runtime.md) immediately before the
109
- first candidate or judge dispatch. If either adviser or judge seat is
110
- unavailable, hold that question without substitution; unaffected fact work
111
- and questions continue.
121
+ first candidate or judge dispatch. If any configured candidate or judge seat is
122
+ unavailable at launch or returns a failed receipt, hold that question without
123
+ substitution, record the gap, and ask: the user decides whether to proceed without it.
124
+ For an uncertain dispatch, reconcile natively; it is never treated as absent.
125
+ Unaffected fact work and questions continue.
112
126
 
113
127
  ## Bound the interview
114
128
 
@@ -84,8 +84,17 @@ Size alone never requires user approval.
84
84
  Use the normal behavior path unless the accepted improvement scope is
85
85
  explicitly marked **structure-preserving**. The author never chooses that tag.
86
86
 
87
+ Only when the scope identity carries a sketch, copy it into the author brief
88
+ under the [design lens](../axstack/references/design-lens.md).
89
+
87
90
  ### Normal behavior path
88
91
 
92
+ When that sketch exists, make the first red check target its `Usage` line.
93
+ The structure-preserving path stays as is.
94
+ If a repeated workaround or unnamed boundary conflicts with the sketch, the
95
+ author stops and returns a sketch conflict. The driver reopens only the
96
+ affected decision through Align under the existing material-revision rule.
97
+
89
98
  Choose a behavior from the accepted scope, including its failure behavior or a
90
99
  real integration boundary. Test it through an observable interface rather than
91
100
  restating source text or mirroring the intended implementation. Execute the
@@ -42,7 +42,9 @@ specialization materially helps; create no new profile.
42
42
  Produce a small ranked candidate set. For each candidate include:
43
43
 
44
44
  1. Source evidence and the scoped problem.
45
- 2. Current and proposed shape.
45
+ 2. Current and proposed shape. Only when the scope identity carries a sketch,
46
+ use the [design lens](../axstack/references/design-lens.md) vocabulary and
47
+ red flags and return candidates in sketch form.
46
48
  3. Concrete benefit and tradeoffs.
47
49
  4. Behavior to preserve and test approach.
48
50
  5. Uncertainty and recommendation strength.
@@ -243,7 +243,10 @@ This section applies to peer and authored PR modes.
243
243
  is safe because of on the evidence ladder; below "ran it" is unproven.
244
244
  4. Requirements, acceptance, and user behavior.
245
245
  5. Architecture and solution design, including SOLID and credible simpler
246
- alternatives.
246
+ alternatives. Only when the scope identity carries a sketch, compare
247
+ the architecture with the [design lens](../axstack/references/design-lens.md)
248
+ sketch and red flags. A deviation from a `Binding` line without an
249
+ accepted spec revision is a finding.
247
250
  6. Simplicity and maintainability: KISS, YAGNI, and cyclomatic complexity
248
251
  where measurement is useful. Never invent a metric or demand an
249
252
  abstraction merely to satisfy a principle.
@@ -35,7 +35,10 @@ and the lifecycle's [audit skill](../axstack-audit/SKILL.md) hook.
35
35
  Missing access preserves the GitHub selection and stops the phase without
36
36
  mutation or fallback. Markdown mode skips external access preflight.
37
37
  3. **Draft with decision evidence.** Write observable acceptance criteria
38
- and explicit exclusions in the selected store. First record the driver's
38
+ and explicit exclusions in the selected store. Only when the scope identity
39
+ carries a sketch, include the [design lens](../axstack/references/design-lens.md)
40
+ sketch in the approved revision's `Design` section and its `Usage` line in
41
+ acceptance. First record the driver's
39
42
  independent assessment, then load
40
43
  [Orca runtime](../axstack/references/orca-runtime.md) before dispatching the
41
44
  configured `axstack-advisor-astra` and `axstack-advisor-fable` independently,
@@ -41,8 +41,12 @@ an actual checker dispatch, not for ordinary mapping or state reconciliation.
41
41
  issues represent user-visible capabilities; one capability may span several
42
42
  tasks and PRs. GitHub capability issues link the approved spec issue and its
43
43
  SHA-256 body digest. Keep detailed execution breakdowns in the repository.
44
- For every capability, derive acceptance checks from the pinned spec and
45
- identify internal tasks, dependencies, PR ownership, and worktrees. For each task the driver records
44
+ For every capability, derive acceptance checks from the pinned spec. Only
45
+ when its scope identity carries a sketch, follow the
46
+ [design lens](../axstack/references/design-lens.md) sketch's modules, put
47
+ its named failure in capability acceptance, and treat a task spanning a
48
+ sketch boundary as a split signal. Identify internal tasks, dependencies,
49
+ PR ownership, and worktrees. For each task the driver records
46
50
  one theme and a coarse size estimate from the ownership, interface, and
47
51
  dependency map. A task estimated in the exception band is assessed for a
48
52
  split at mapping time and split where a green, atomic, reviewable split
package/src/installer.js CHANGED
@@ -626,6 +626,16 @@ export async function installBundle({
626
626
  if (rel in ownedFiles) installedHashes[rel] = ownedFiles[rel];
627
627
  }
628
628
  }
629
+ const previousRoles = desired.find(({ rel }) => rel === 'axstack/roles.json')?.current;
630
+ if (previousRoles && summary.updated.some((rel) => rel.startsWith('axstack/roles.json'))) {
631
+ try {
632
+ const old = JSON.parse(previousRoles.toString());
633
+ if (Array.isArray(old.roles)) {
634
+ const oldIds = new Set(old.roles.map((role) => role?.id));
635
+ summary.addedRoleIds = bundle.bundleRoles.filter((role) => !oldIds.has(role.id)).map((role) => role.id);
636
+ }
637
+ } catch { /* No reliable role-ID diff for malformed prior bytes. */ }
638
+ }
629
639
 
630
640
  // Stale manifest entries (owned files the bundle no longer ships):
631
641
  // the phase-1 plan validated every stale destination read-only
package/src/roles.js CHANGED
@@ -66,8 +66,10 @@ export function assessRoleReadiness(roles, preset) {
66
66
  (preset === 'mixed' && role.id === 'axstack-checker' && role.provider === 'antigravity') ||
67
67
  (preset === 'mixed' && role.id === 'axstack-research-web-google' && role.provider === 'antigravity') ||
68
68
  (preset === 'mixed' && role.id === 'axstack-research-x' && role.provider === 'grok') ||
69
- (preset === 'codex-only' && ['axstack-advisor-fable', 'axstack-arena-judge-fable', 'axstack-research-web-google', 'axstack-research-x'].includes(role.id)) ||
70
- (preset === 'claude-only' && ['axstack-advisor-astra', 'axstack-arena-judge-astra', 'axstack-research-web-google', 'axstack-research-x'].includes(role.id))
69
+ (preset === 'mixed' && role.id === 'axstack-arena-candidate-grok' && role.provider === 'grok') ||
70
+ (preset === 'mixed' && role.id === 'axstack-arena-candidate-antigravity' && role.provider === 'antigravity') ||
71
+ (preset === 'codex-only' && ['axstack-advisor-fable', 'axstack-arena-judge-fable', 'axstack-research-web-google', 'axstack-research-x', 'axstack-arena-candidate-grok', 'axstack-arena-candidate-antigravity'].includes(role.id)) ||
72
+ (preset === 'claude-only' && ['axstack-advisor-astra', 'axstack-arena-judge-astra', 'axstack-research-web-google', 'axstack-research-x', 'axstack-arena-candidate-grok', 'axstack-arena-candidate-antigravity'].includes(role.id))
71
73
  );
72
74
  for (const role of roles) {
73
75
  if (!bounds.has(role.provider)) {