axstack 0.20.21 → 0.20.23
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/bin/axstack.js +1 -0
- package/docs/installation.md +4 -4
- package/docs/workflows.md +4 -4
- package/package.json +1 -1
- package/profiles/presets/claude-only.json +20 -2
- package/profiles/presets/codex-only.json +20 -2
- package/profiles/presets/mixed.json +20 -2
- package/skills/axstack/references/design-lens.md +81 -0
- package/skills/axstack/references/orca-runtime.md +1 -1
- package/skills/axstack/references/routing.md +31 -33
- package/skills/axstack-align/SKILL.md +30 -16
- package/skills/axstack-implement/SKILL.md +9 -0
- package/skills/axstack-improve/SKILL.md +3 -1
- package/skills/axstack-review/SKILL.md +4 -1
- package/skills/axstack-spec/SKILL.md +4 -1
- package/skills/axstack-tickets/SKILL.md +6 -2
- package/src/installer.js +10 -0
- package/src/roles.js +4 -2
package/README.md
CHANGED
|
@@ -99,7 +99,7 @@ upgrades, conflicts, and uninstalling.
|
|
|
99
99
|
inline or from an explicitly scoped backlog without touching active, manual,
|
|
100
100
|
uncertain, user-owned, dirty, unpushed, or useful unmerged work.
|
|
101
101
|
|
|
102
|
-
Choose an explicit role preset:
|
|
102
|
+
Choose an explicit role preset (26 stable role IDs in each):
|
|
103
103
|
[mixed](profiles/presets/mixed.json),
|
|
104
104
|
[codex-only](profiles/presets/codex-only.json), or
|
|
105
105
|
[claude-only](profiles/presets/claude-only.json).
|
package/bin/axstack.js
CHANGED
|
@@ -335,6 +335,7 @@ async function main() {
|
|
|
335
335
|
} else {
|
|
336
336
|
if (summary.added.length) console.log(`added: ${summary.added.join(', ')}`);
|
|
337
337
|
if (summary.updated.length) console.log(`updated: ${summary.updated.join(', ')}`);
|
|
338
|
+
if (summary.addedRoleIds?.length) console.log(`added role IDs: ${summary.addedRoleIds.join(', ')}`);
|
|
338
339
|
if (summary.removed.length) console.log(`removed: ${summary.removed.join(', ')}`);
|
|
339
340
|
if (summary.unchanged.length) console.log(`unchanged: ${summary.unchanged.join(', ')}`);
|
|
340
341
|
}
|
package/docs/installation.md
CHANGED
|
@@ -71,7 +71,7 @@ profiles/presets/codex-only.json
|
|
|
71
71
|
profiles/presets/claude-only.json
|
|
72
72
|
```
|
|
73
73
|
|
|
74
|
-
Each has exactly `{ "version": 1, "roles": [...] }` with the same
|
|
74
|
+
Each has exactly `{ "version": 1, "roles": [...] }` with the same 26 stable
|
|
75
75
|
role IDs. Installation writes `<skills-dir>/axstack/roles.json` as
|
|
76
76
|
`{ "version": 1, "preset": "<selected preset>", "roles": [...] }` and records
|
|
77
77
|
its ownership hash like every other installed skill asset. There is no second
|
|
@@ -140,10 +140,10 @@ The complete bundle is validated before writes:
|
|
|
140
140
|
the filename's selected identity supplied by the caller, and the same role-ID
|
|
141
141
|
set as its peers;
|
|
142
142
|
- every role has valid preserved fields, while the mixed checker,
|
|
143
|
-
`axstack-research-web-google`,
|
|
143
|
+
`axstack-research-web-google`, `axstack-research-x`, and both arena candidate launch-by-agent-id
|
|
144
144
|
routes explicitly permit `model: null`;
|
|
145
145
|
in each single-provider preset, the unavailable adviser and its matching arena
|
|
146
|
-
judge seat explicitly permit `model: null`, as do both cross-provider research routes;
|
|
146
|
+
judge seat explicitly permit `model: null`, as do both cross-provider research routes and both arena candidate seats;
|
|
147
147
|
- obsolete runtime configuration flags fail before mutation with migration
|
|
148
148
|
guidance.
|
|
149
149
|
|
|
@@ -171,7 +171,7 @@ to rewrite them.
|
|
|
171
171
|
## Role behavior after installation
|
|
172
172
|
|
|
173
173
|
The runtime reads `roles.json` from the installed shared root `skills/axstack/`.
|
|
174
|
-
A new run records the selected preset plus all
|
|
174
|
+
A new run records the selected preset plus all 26 role rows. An active run keeps
|
|
175
175
|
that snapshot after a later preset install unless the user explicitly changes
|
|
176
176
|
it and accepts the resulting evidence invalidation.
|
|
177
177
|
|
package/docs/workflows.md
CHANGED
|
@@ -48,7 +48,7 @@ only affected work.
|
|
|
48
48
|
|
|
49
49
|
Installation requires one explicit canonical preset. The three bundle files
|
|
50
50
|
under `profiles/presets/` each contain exactly
|
|
51
|
-
`{ "version": 1, "roles": [...] }` and the same
|
|
51
|
+
`{ "version": 1, "roles": [...] }` and the same 26 stable IDs.
|
|
52
52
|
|
|
53
53
|
The current chat drives on whatever model runs it; no preset carries a driver
|
|
54
54
|
role.
|
|
@@ -122,9 +122,9 @@ session and evidence remain valid.
|
|
|
122
122
|
- `axstack-align` maps facts and dependencies, asks prioritized questions, and
|
|
123
123
|
consults Astra and Fable independently with the same bounded evidence and
|
|
124
124
|
question. It synthesizes disagreements and reuses unchanged receipts. For a
|
|
125
|
-
hard-to-reverse design choice it runs one arena round instead: Astra
|
|
126
|
-
Fable each author a candidate, `axstack-arena-judge-astra` and
|
|
127
|
-
`axstack-arena-judge-fable` score
|
|
125
|
+
hard-to-reverse design choice it runs one arena round instead: Astra,
|
|
126
|
+
Fable, Grok, and Antigravity each author a candidate, `axstack-arena-judge-astra` and
|
|
127
|
+
`axstack-arena-judge-fable` score every candidate against the driver's rubric, and the
|
|
128
128
|
driver picks a base, grafts the losers' strong ideas, and presents the
|
|
129
129
|
synthesis as the recommendation; the note lands as `Decisions` rows in the
|
|
130
130
|
run record.
|
package/package.json
CHANGED
|
@@ -206,7 +206,7 @@
|
|
|
206
206
|
"model": null,
|
|
207
207
|
"modeId": "bypassPermissions",
|
|
208
208
|
"thinkingOptionId": "xhigh",
|
|
209
|
-
"notes": "Required Arena judge (Astra seat). Read-only cross-judge for an arena-grade Align decision: receives the rubric and
|
|
209
|
+
"notes": "Required Arena judge (Astra seat). Read-only cross-judge for an arena-grade Align decision: receives the rubric and every candidate by label, scores each criterion, and recommends a base with rationale. Never authors a candidate, never cross-reads the other judge, never mutates. Disagreement between judges is surfaced by the driver, never averaged. Intentionally absent in the claude-only preset; the arena-grade decision holds without substitution."
|
|
210
210
|
},
|
|
211
211
|
{
|
|
212
212
|
"id": "axstack-arena-judge-fable",
|
|
@@ -215,7 +215,25 @@
|
|
|
215
215
|
"model": "claude-fable-5-1",
|
|
216
216
|
"modeId": "bypassPermissions",
|
|
217
217
|
"thinkingOptionId": "xhigh",
|
|
218
|
-
"notes": "Arena judge (Fable seat). Read-only cross-judge for an arena-grade Align decision: receives the rubric and
|
|
218
|
+
"notes": "Arena judge (Fable seat). Read-only cross-judge for an arena-grade Align decision: receives the rubric and every candidate by label, scores each criterion, and recommends a base with rationale. Never authors a candidate, never cross-reads the other judge, never mutates. Disagreement between judges is surfaced by the driver, never averaged."
|
|
219
|
+
},
|
|
220
|
+
{
|
|
221
|
+
"id": "axstack-arena-candidate-grok",
|
|
222
|
+
"name": "Axstack arena candidate Grok (unavailable)",
|
|
223
|
+
"provider": "claude",
|
|
224
|
+
"model": null,
|
|
225
|
+
"modeId": "bypassPermissions",
|
|
226
|
+
"thinkingOptionId": "high",
|
|
227
|
+
"notes": "Intentional single-provider absence in the claude-only preset; Grok candidate unavailable. Arena holds without substitution."
|
|
228
|
+
},
|
|
229
|
+
{
|
|
230
|
+
"id": "axstack-arena-candidate-antigravity",
|
|
231
|
+
"name": "Axstack arena candidate Antigravity (unavailable)",
|
|
232
|
+
"provider": "claude",
|
|
233
|
+
"model": null,
|
|
234
|
+
"modeId": "bypassPermissions",
|
|
235
|
+
"thinkingOptionId": "high",
|
|
236
|
+
"notes": "Intentional single-provider absence in the claude-only preset; Antigravity candidate unavailable. Arena holds without substitution."
|
|
219
237
|
}
|
|
220
238
|
]
|
|
221
239
|
}
|
|
@@ -206,7 +206,7 @@
|
|
|
206
206
|
"model": "gpt-6-astra",
|
|
207
207
|
"modeId": "full-access",
|
|
208
208
|
"thinkingOptionId": "xhigh",
|
|
209
|
-
"notes": "Arena judge (Astra seat). Read-only cross-judge for an arena-grade Align decision: receives the rubric and
|
|
209
|
+
"notes": "Arena judge (Astra seat). Read-only cross-judge for an arena-grade Align decision: receives the rubric and every candidate by label, scores each criterion, and recommends a base with rationale. Never authors a candidate, never cross-reads the other judge, never mutates. Disagreement between judges is surfaced by the driver, never averaged."
|
|
210
210
|
},
|
|
211
211
|
{
|
|
212
212
|
"id": "axstack-arena-judge-fable",
|
|
@@ -215,7 +215,25 @@
|
|
|
215
215
|
"model": null,
|
|
216
216
|
"modeId": "full-access",
|
|
217
217
|
"thinkingOptionId": "xhigh",
|
|
218
|
-
"notes": "Required Arena judge (Fable seat). Read-only cross-judge for an arena-grade Align decision: receives the rubric and
|
|
218
|
+
"notes": "Required Arena judge (Fable seat). Read-only cross-judge for an arena-grade Align decision: receives the rubric and every candidate by label, scores each criterion, and recommends a base with rationale. Never authors a candidate, never cross-reads the other judge, never mutates. Disagreement between judges is surfaced by the driver, never averaged. Intentionally absent in the codex-only preset; the arena-grade decision holds without substitution."
|
|
219
|
+
},
|
|
220
|
+
{
|
|
221
|
+
"id": "axstack-arena-candidate-grok",
|
|
222
|
+
"name": "Axstack arena candidate Grok (unavailable)",
|
|
223
|
+
"provider": "codex",
|
|
224
|
+
"model": null,
|
|
225
|
+
"modeId": "full-access",
|
|
226
|
+
"thinkingOptionId": "high",
|
|
227
|
+
"notes": "Intentional single-provider absence in the codex-only preset; Grok candidate unavailable. Arena holds without substitution."
|
|
228
|
+
},
|
|
229
|
+
{
|
|
230
|
+
"id": "axstack-arena-candidate-antigravity",
|
|
231
|
+
"name": "Axstack arena candidate Antigravity (unavailable)",
|
|
232
|
+
"provider": "codex",
|
|
233
|
+
"model": null,
|
|
234
|
+
"modeId": "full-access",
|
|
235
|
+
"thinkingOptionId": "high",
|
|
236
|
+
"notes": "Intentional single-provider absence in the codex-only preset; Antigravity candidate unavailable. Arena holds without substitution."
|
|
219
237
|
}
|
|
220
238
|
]
|
|
221
239
|
}
|
|
@@ -206,7 +206,7 @@
|
|
|
206
206
|
"model": "gpt-6-astra",
|
|
207
207
|
"modeId": "full-access",
|
|
208
208
|
"thinkingOptionId": "xhigh",
|
|
209
|
-
"notes": "Arena judge (Astra seat). Read-only cross-judge for an arena-grade Align decision: receives the rubric and
|
|
209
|
+
"notes": "Arena judge (Astra seat). Read-only cross-judge for an arena-grade Align decision: receives the rubric and every candidate by label, scores each criterion, and recommends a base with rationale. Never authors a candidate, never cross-reads the other judge, never mutates. Disagreement between judges is surfaced by the driver, never averaged."
|
|
210
210
|
},
|
|
211
211
|
{
|
|
212
212
|
"id": "axstack-arena-judge-fable",
|
|
@@ -215,7 +215,25 @@
|
|
|
215
215
|
"model": "claude-fable-5-1",
|
|
216
216
|
"modeId": "bypassPermissions",
|
|
217
217
|
"thinkingOptionId": "xhigh",
|
|
218
|
-
"notes": "Arena judge (Fable seat). Read-only cross-judge for an arena-grade Align decision: receives the rubric and
|
|
218
|
+
"notes": "Arena judge (Fable seat). Read-only cross-judge for an arena-grade Align decision: receives the rubric and every candidate by label, scores each criterion, and recommends a base with rationale. Never authors a candidate, never cross-reads the other judge, never mutates. Disagreement between judges is surfaced by the driver, never averaged."
|
|
219
|
+
},
|
|
220
|
+
{
|
|
221
|
+
"id": "axstack-arena-candidate-grok",
|
|
222
|
+
"name": "Axstack arena candidate Grok",
|
|
223
|
+
"provider": "grok",
|
|
224
|
+
"model": null,
|
|
225
|
+
"modeId": "full-access",
|
|
226
|
+
"thinkingOptionId": "high",
|
|
227
|
+
"notes": "Independent Grok Rung 2 design candidate. Launch by agent id grok; model selected by the TUI default. Receives the same brief without cross-reading; returns design, rationale, and rejected alternatives."
|
|
228
|
+
},
|
|
229
|
+
{
|
|
230
|
+
"id": "axstack-arena-candidate-antigravity",
|
|
231
|
+
"name": "Axstack arena candidate Antigravity",
|
|
232
|
+
"provider": "antigravity",
|
|
233
|
+
"model": null,
|
|
234
|
+
"modeId": "full-access",
|
|
235
|
+
"thinkingOptionId": "high",
|
|
236
|
+
"notes": "Independent Antigravity Rung 2 design candidate. Launch by agent id antigravity; model selected by the TUI default. Receives the same brief without cross-reading; returns design, rationale, and rejected alternatives."
|
|
219
237
|
}
|
|
220
238
|
]
|
|
221
239
|
}
|
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
# Design lens
|
|
2
|
+
|
|
3
|
+
In Align, set the rung from researched facts, not a user choice. Carry the
|
|
4
|
+
sketch through the scope identity.
|
|
5
|
+
|
|
6
|
+
## Ladder
|
|
7
|
+
|
|
8
|
+
- **Rung 0 — skip.** The change stays inside one module's existing interface,
|
|
9
|
+
ownership, data flow, and failure guarantees. Ask no design questions and
|
|
10
|
+
carry no sketch. “Might be small” is not a reason to skip.
|
|
11
|
+
- **Rung 1 — design questions.** Any change that fails Rung 0. Typical triggers:
|
|
12
|
+
crossing a module boundary; adding a module, service, interface, or external
|
|
13
|
+
dependency; changing an interface or failure guarantee; changing data
|
|
14
|
+
ownership, schema, wire format, or persisted format. A design question does
|
|
15
|
+
not reclassify work: Rung 1 can stay small. An unsettled material design
|
|
16
|
+
question still makes routing reassess size.
|
|
17
|
+
- **Rung 2 — arena.** A Rung 1 design that also meets the existing ADR test:
|
|
18
|
+
a meaningful, hard-to-reverse, non-obvious trade-off. Use Align's all-family
|
|
19
|
+
arena for that question.
|
|
20
|
+
|
|
21
|
+
There is no numeric threshold, file-count gate, or class-count gate.
|
|
22
|
+
|
|
23
|
+
## Questions in order
|
|
24
|
+
|
|
25
|
+
Ask only unresolved areas in numbered `Qn` rounds of one to three. Give each a
|
|
26
|
+
recommendation, reason, and trade-off; use Align's adviser critique. Design and
|
|
27
|
+
arena questions share its unchanged budget: 20 normally, a justified extension
|
|
28
|
+
to 35, then opted-in refinement of at most five.
|
|
29
|
+
|
|
30
|
+
1. **Scope:** What exists, the minimum change, and what is explicitly excluded?
|
|
31
|
+
2. **Caller first:** Write realistic usage before drawing the shape.
|
|
32
|
+
3. **Shape:** Which boundaries, module depth, ownership, invariants, seams,
|
|
33
|
+
and adapters serve that usage?
|
|
34
|
+
4. **Flow and failure:** Trace the data flow. For each materially new path,
|
|
35
|
+
name one realistic production failure and its observable behavior.
|
|
36
|
+
5. **Reversibility:** Name compatibility, migration, rejected alternatives, and
|
|
37
|
+
what we accept for what benefit.
|
|
38
|
+
|
|
39
|
+
## Vocabulary and red flags
|
|
40
|
+
|
|
41
|
+
A **deep module** hides substantial behavior behind a small interface. A
|
|
42
|
+
**shallow module** exposes most of its machinery. Treat an **interface as test
|
|
43
|
+
surface**: test caller behavior and observable failures. A **seam** permits a
|
|
44
|
+
second implementation; one adapter is hypothetical, two make it real. Prefer
|
|
45
|
+
**locality** and explicit **ownership** of state and invariants. The **deletion
|
|
46
|
+
test** asks what contract would be lost if a layer vanished. Choose **boring by
|
|
47
|
+
default** and **reversible over clever**.
|
|
48
|
+
|
|
49
|
+
Red flags—shallow module, information leakage, temporal decomposition, and
|
|
50
|
+
pass-through layers—are prompts for evidence, not automatic defects. Ask what
|
|
51
|
+
crosses each boundary and whether a layer can be removed.
|
|
52
|
+
|
|
53
|
+
## Sketch
|
|
54
|
+
|
|
55
|
+
For Rung 1 or 2, put this block in the spec's `Design` section or the returned
|
|
56
|
+
small-change intent. `Binding` names commitments; other lines may illustrate.
|
|
57
|
+
|
|
58
|
+
```text
|
|
59
|
+
Usage: <call site or command as the caller writes it>
|
|
60
|
+
Shape: <signatures or a <=10-line ASCII/Mermaid diagram>
|
|
61
|
+
Binding: <which lines are committed; the rest is illustrative>
|
|
62
|
+
Flow + failure: <path -> one realistic failure -> observable behavior>
|
|
63
|
+
We accept: <X> for <Y>
|
|
64
|
+
Rejected: <alternative -> why>
|
|
65
|
+
Open: <question?>
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
Keep settled choices in `Decisions` rows. `CONTEXT.md` stays a glossary; ADRs
|
|
69
|
+
keep their existing test. Add no design document or phase.
|
|
70
|
+
|
|
71
|
+
## Arena rubric candidates
|
|
72
|
+
|
|
73
|
+
For each arena, the driver selects three to six and tailors the rubric criteria
|
|
74
|
+
to the question. Score candidate designs on evidence, not a generic checklist:
|
|
75
|
+
|
|
76
|
+
- Is `Usage` realistic and simple for the caller?
|
|
77
|
+
- Are module boundaries deep enough and ownership unambiguous?
|
|
78
|
+
- Is behavior local with a useful test surface?
|
|
79
|
+
- Are materially new failure paths observable and recoverable?
|
|
80
|
+
- Are compatibility, migration, and deletion costs explicit?
|
|
81
|
+
- Does the trade-off justify complexity against rejected options?
|
|
@@ -38,7 +38,7 @@ Read `roles.json` from the installed shared root `skills/axstack/`. The installe
|
|
|
38
38
|
shape is `{ "version": 1, "preset": "<name>", "roles": [...] }`. Bundled
|
|
39
39
|
profiles are setup inputs shaped as
|
|
40
40
|
`{ "version": 1, "roles": [...] }`. A new run records the selected preset and
|
|
41
|
-
all
|
|
41
|
+
all 26 role rows once. An active run keeps the exact snapshot until the user
|
|
42
42
|
explicitly changes it.
|
|
43
43
|
|
|
44
44
|
Select the requested role by stable ID. A missing or null model holds only that role;
|
|
@@ -5,19 +5,19 @@ Driver entry sweep follows [Workspace hygiene](workspace-hygiene.md).
|
|
|
5
5
|
|
|
6
6
|
## Role routing
|
|
7
7
|
|
|
8
|
-
Presets: `mixed`, `codex-only`, `claude-only`. For
|
|
8
|
+
Presets: `mixed`, `codex-only`, `claude-only`. For new runs, use
|
|
9
9
|
`profiles.preset` from `.axstack-manifest.json` at the actually loaded
|
|
10
|
-
skills root, or an explicit user selection recorded in the run record.
|
|
11
|
-
|
|
10
|
+
skills root, or an explicit user selection recorded in the run record. Use one
|
|
11
|
+
unambiguous preset; missing or contradictory sources are
|
|
12
12
|
a setup gap: hold. Never infer from live profiles or `list_profiles`, harness,
|
|
13
13
|
tools, credentials, quota, subscription, or default to `mixed`.
|
|
14
14
|
|
|
15
|
-
At run start, snapshot all
|
|
15
|
+
At run start, snapshot all 26 role IDs with provider/model/mode/effort; absent
|
|
16
16
|
or unconfigured roles are recorded explicitly; invent no provider default.
|
|
17
|
-
Such a role holds only that role's work
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
17
|
+
Such a role holds only that role's work. A role installed or changed later must not
|
|
18
|
+
silently enter the snapshot; adding it needs an explicit user decision. Live profiles
|
|
19
|
+
are authoritative at snapshot time and for availability; bundled presets are setup
|
|
20
|
+
inputs, not runtime proof.
|
|
21
21
|
|
|
22
22
|
Preset changes apply to new runs only; an active run keeps its snapshot.
|
|
23
23
|
Changing it or replacing a session needs an explicit user decision and
|
|
@@ -25,8 +25,6 @@ revalidation. Unavailable models, unsupported efforts, missing roles, and
|
|
|
25
25
|
incompatible overrides hold only affected work; no automatic fallback, quota
|
|
26
26
|
routing, subscription inference, or silent provider/model/effort substitution.
|
|
27
27
|
|
|
28
|
-
Role IDs:
|
|
29
|
-
|
|
30
28
|
- Chat drives (no role ID); `axstack-owner` owns one PR and
|
|
31
29
|
`axstack-author` its sole writer.
|
|
32
30
|
- `axstack-reviewer-primary` and `axstack-reviewer-secondary` are the ordered
|
|
@@ -38,10 +36,11 @@ Role IDs:
|
|
|
38
36
|
| `mixed` | Claude / Opus (`claude/claude-opus-5-5`) | `axstack-reviewer-primary` (`codex/gpt-6-sol` medium) |
|
|
39
37
|
| `codex-only` | Codex / Sol (`codex/gpt-6-sol`) | `axstack-reviewer-secondary` (`codex/gpt-6-luna` xhigh) |
|
|
40
38
|
| `claude-only` | Claude / Opus (`claude/claude-opus-5-5`) | `axstack-reviewer-secondary` (`claude/claude-sonnet-5` xhigh) |
|
|
41
|
-
- `axstack-advisor-astra
|
|
42
|
-
|
|
43
|
-
`axstack-arena-
|
|
44
|
-
`axstack-
|
|
39
|
+
- `axstack-advisor-astra`/`axstack-advisor-fable` advise independently and
|
|
40
|
+
author arena candidates; `axstack-arena-candidate-grok`/
|
|
41
|
+
`axstack-arena-candidate-antigravity` add families.
|
|
42
|
+
`axstack-arena-judge-astra`/`axstack-arena-judge-fable` judge them.
|
|
43
|
+
`axstack-auditor` audits; `axstack-checker` reports discrepancies.
|
|
45
44
|
- `axstack-explainer`/`axstack-explainer-review`: explain/review.
|
|
46
45
|
`axstack-monitor`: standalone watch never sends; chat-run watch: bounded
|
|
47
46
|
internal reports to its Run and original driver.
|
|
@@ -51,17 +50,16 @@ Provenance is matched on provider/model ID; effort never maps. Missing table-row
|
|
|
51
50
|
provenance is unsupported and `INCOMPLETE`; report it and ask the user. Never
|
|
52
51
|
infer from slot, driver, owner, or provider. Author and owner never review.
|
|
53
52
|
|
|
54
|
-
|
|
55
|
-
step (3) for user routing
|
|
53
|
+
`axstack-implement` requires `mixed`; single-provider presets hold at
|
|
54
|
+
step (3) for user routing: no substitution or same-provider review.
|
|
56
55
|
|
|
57
56
|
## Direct routes (no spec ceremony)
|
|
58
57
|
|
|
59
58
|
- One bounded research question -> `axstack-research`: verify primary sources
|
|
60
|
-
and code
|
|
61
|
-
questions.
|
|
59
|
+
and code; return a cited note with limitations. Fan out only distinct questions.
|
|
62
60
|
- Understanding a system, change, or implementation gap -> `axstack-explain`:
|
|
63
|
-
current/intended behavior, evidence dimensions,
|
|
64
|
-
|
|
61
|
+
current/intended behavior, evidence dimensions, bounded gaps from project docs and
|
|
62
|
+
rendered behavior. "What could this break" follows
|
|
65
63
|
[Blast radius](blast-radius.md). Publication needs separate authority.
|
|
66
64
|
- A bug, failing test, regression, or wrong behavior, red loop wanted ->
|
|
67
65
|
`axstack-debug`: diagnose, escalate via adviser-directed investigators, hand
|
|
@@ -69,11 +67,11 @@ step (3) for user routing, with no substitution or same-provider review.
|
|
|
69
67
|
- Codebase-quality or refactor discovery -> `axstack-improve`: inspect bounded
|
|
70
68
|
scope, rank evidenced candidates, report only; no spec, tickets, or source
|
|
71
69
|
edits.
|
|
72
|
-
- Accepted worker/Task/Run completion or bounded backlog request ->
|
|
73
|
-
`axstack-cleanup` inline
|
|
70
|
+
- Accepted worker/Task/Run completion or bounded backlog request -> driver invokes
|
|
71
|
+
`axstack-cleanup` inline; never dispatch it.
|
|
74
72
|
- Preparation completion, watch expiry, resume, or reconciliation -> the
|
|
75
|
-
[lifecycle](lifecycle.md#native-handoff-and-resume): reconcile
|
|
76
|
-
|
|
73
|
+
[lifecycle](lifecycle.md#native-handoff-and-resume): reconcile run record,
|
|
74
|
+
keep owner, launch no native handoff.
|
|
77
75
|
- Explicit user-requested ownership transfer -> the same lifecycle section.
|
|
78
76
|
Load the [Orca runtime boundary](orca-runtime.md), follow the runtime-owned
|
|
79
77
|
handoff guide, and require explicit recipient acceptance before ownership
|
|
@@ -81,10 +79,10 @@ step (3) for user routing, with no substitution or same-provider review.
|
|
|
81
79
|
- Colleague PR review -> `axstack-review`, peer mode.
|
|
82
80
|
- Codebase review -> `axstack-review` codebase mode, report only.
|
|
83
81
|
- A status question about an own open PR or stack ("check now", "what's left",
|
|
84
|
-
"are we done", or "is it approved") -> `axstack-watch`
|
|
82
|
+
"are we done", or "is it approved") -> `axstack-watch` observation-only
|
|
85
83
|
mode. Explicit "address", "patch", or "fix" grants authorized maintenance.
|
|
86
|
-
- Chat-run PR watch -> `axstack-watch`: original driver; verified run PRs
|
|
87
|
-
|
|
84
|
+
- Chat-run PR watch -> `axstack-watch`: original driver; verified run PRs and
|
|
85
|
+
explicit adoptions only.
|
|
88
86
|
- Other own PR work -> `axstack-review` authored mode or `axstack-watch`
|
|
89
87
|
adoption.
|
|
90
88
|
|
|
@@ -102,13 +100,13 @@ reason in the run record, or in the brief for tiny direct work.
|
|
|
102
100
|
- **Small:** clear, bounded one-PR work. The driver captures the named
|
|
103
101
|
**small-change intent** from the current request or user-chosen existing
|
|
104
102
|
issue plus explicit acceptance checks and exclusions, snapshots it once, and
|
|
105
|
-
proceeds. No
|
|
106
|
-
|
|
107
|
-
|
|
103
|
+
proceeds. No prior snapshot, spec, tickets, or second approval is required; do not route
|
|
104
|
+
to `axstack-align` solely because the snapshot is not yet written. Strict TDD,
|
|
105
|
+
mode-specific review, model, risk, and human-merge
|
|
108
106
|
contracts still apply.
|
|
109
|
-
- **Unclear:** clarify
|
|
110
|
-
|
|
111
|
-
|
|
107
|
+
- **Unclear:** clarify via `axstack-align` or a bounded question, then
|
|
108
|
+
classify small or substantial; it does not force substantial-work paperwork.
|
|
109
|
+
[Design lens](design-lens.md) Rung 1 is Unclear; use `axstack-align`.
|
|
112
110
|
|
|
113
111
|
Reassess size when growth adds an additional PR, a new execution dependency
|
|
114
112
|
that materially expands scope, an unsettled material design question, or a
|
|
@@ -17,6 +17,17 @@ Load before acting:
|
|
|
17
17
|
|
|
18
18
|
This preserves the required contracts -> lifecycle -> audit load edge.
|
|
19
19
|
|
|
20
|
+
## Design the shape
|
|
21
|
+
|
|
22
|
+
Set the rung from researched facts; never ask the user to choose it. A change
|
|
23
|
+
inside one module's existing interface, ownership, data flow, and failure
|
|
24
|
+
guarantees is Rung 0: no design questions or sketch. Otherwise load the
|
|
25
|
+
[design lens ladder](../axstack/references/design-lens.md) for Rung 1 or 2
|
|
26
|
+
and settle only unresolved areas in its order within the existing budget. Carry a
|
|
27
|
+
Rung 1 or 2 sketch in the substantial spec's `Design` section or the returned
|
|
28
|
+
small-change intent. A design question alone does not make small work
|
|
29
|
+
substantial; apply routing's existing size reassessment rule.
|
|
30
|
+
|
|
20
31
|
## Settle the frontier
|
|
21
32
|
|
|
22
33
|
1. **Research and map dependencies.** Inspect the available code, docs, and
|
|
@@ -70,29 +81,30 @@ hold Align; safe fact work may continue without substitution.
|
|
|
70
81
|
|
|
71
82
|
## Arena for hard-to-reverse design choices
|
|
72
83
|
|
|
73
|
-
Critique of one draft anchors every reader to that draft's shape.
|
|
74
|
-
|
|
84
|
+
Critique of one draft anchors every reader to that draft's shape. Rung 2 designs
|
|
85
|
+
alone enter the arena: they meet the same test as for an ADR (a meaningful,
|
|
75
86
|
hard-to-reverse, non-obvious trade-off: architecture, module boundaries, data
|
|
76
|
-
model, migration strategy)
|
|
87
|
+
model, migration strategy). Replace the critique round for that question with
|
|
77
88
|
one arena round. Small or routine questions never enter the arena.
|
|
78
89
|
|
|
79
90
|
1. **Frame.** The driver writes the brief (the artifact, its constraints, the
|
|
80
91
|
settled decisions it must respect) and three to six gradeable rubric
|
|
81
92
|
criteria. Candidates receive only the brief; the rubric is for judging.
|
|
82
|
-
2. **Fan out.**
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
3. **Cross-judge.** After
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
93
|
+
2. **Fan out.** Produce one candidate per configured family independently from the same brief,
|
|
94
|
+
without cross-reading: `axstack-advisor-astra`, `axstack-advisor-fable`,
|
|
95
|
+
`axstack-arena-candidate-grok`, and `axstack-arena-candidate-antigravity`.
|
|
96
|
+
Each gives a design, rationale, and rejected alternatives. The driver authors no candidate.
|
|
97
|
+
3. **Cross-judge.** After every candidate completes, give the judges anonymized,
|
|
98
|
+
relabeled candidates against the rubric.
|
|
99
|
+
`axstack-arena-judge-astra` and `axstack-arena-judge-fable` each independently
|
|
100
|
+
score every candidate against the driver-tailored rubric per criterion and
|
|
101
|
+
recommend a base with a reason. Judges never author, never cross-read each other.
|
|
90
102
|
4. **Pick.** The driver reads every candidate end to end and scores per
|
|
91
103
|
criterion, not on holistic feel, then compares with both judges. Agreement
|
|
92
104
|
confirms the base. Disagreement between judges or with the driver means one
|
|
93
|
-
reading is biased or the rubric was ambiguous: re-read
|
|
105
|
+
reading is biased or the rubric was ambiguous: re-read the rationales and
|
|
94
106
|
decide with a stated reason; never average verdicts or fabricate consensus.
|
|
95
|
-
5. **Graft.** Walk the losing
|
|
107
|
+
5. **Graft.** Walk the losing candidates once more for the one or two ideas
|
|
96
108
|
worth porting and fold them into the base by hand so the result stays
|
|
97
109
|
coherent under one mental model. Convergence on the same shape is a strong
|
|
98
110
|
agreement signal: adopt the consensus shape, no graft. Wide divergence
|
|
@@ -106,9 +118,11 @@ Record the synthesis note (base, grafts and their source candidate, rejections,
|
|
|
106
118
|
dropouts, both judge verdicts) as `Decisions` rows in the
|
|
107
119
|
[run record](../axstack/references/run-record.md). Load
|
|
108
120
|
[Orca runtime](../axstack/references/orca-runtime.md) immediately before the
|
|
109
|
-
first candidate or judge dispatch. If
|
|
110
|
-
unavailable, hold that question without
|
|
111
|
-
and
|
|
121
|
+
first candidate or judge dispatch. If any configured candidate or judge seat is
|
|
122
|
+
unavailable at launch or returns a failed receipt, hold that question without
|
|
123
|
+
substitution, record the gap, and ask: the user decides whether to proceed without it.
|
|
124
|
+
For an uncertain dispatch, reconcile natively; it is never treated as absent.
|
|
125
|
+
Unaffected fact work and questions continue.
|
|
112
126
|
|
|
113
127
|
## Bound the interview
|
|
114
128
|
|
|
@@ -84,8 +84,17 @@ Size alone never requires user approval.
|
|
|
84
84
|
Use the normal behavior path unless the accepted improvement scope is
|
|
85
85
|
explicitly marked **structure-preserving**. The author never chooses that tag.
|
|
86
86
|
|
|
87
|
+
Only when the scope identity carries a sketch, copy it into the author brief
|
|
88
|
+
under the [design lens](../axstack/references/design-lens.md).
|
|
89
|
+
|
|
87
90
|
### Normal behavior path
|
|
88
91
|
|
|
92
|
+
When that sketch exists, make the first red check target its `Usage` line.
|
|
93
|
+
The structure-preserving path stays as is.
|
|
94
|
+
If a repeated workaround or unnamed boundary conflicts with the sketch, the
|
|
95
|
+
author stops and returns a sketch conflict. The driver reopens only the
|
|
96
|
+
affected decision through Align under the existing material-revision rule.
|
|
97
|
+
|
|
89
98
|
Choose a behavior from the accepted scope, including its failure behavior or a
|
|
90
99
|
real integration boundary. Test it through an observable interface rather than
|
|
91
100
|
restating source text or mirroring the intended implementation. Execute the
|
|
@@ -42,7 +42,9 @@ specialization materially helps; create no new profile.
|
|
|
42
42
|
Produce a small ranked candidate set. For each candidate include:
|
|
43
43
|
|
|
44
44
|
1. Source evidence and the scoped problem.
|
|
45
|
-
2. Current and proposed shape.
|
|
45
|
+
2. Current and proposed shape. Only when the scope identity carries a sketch,
|
|
46
|
+
use the [design lens](../axstack/references/design-lens.md) vocabulary and
|
|
47
|
+
red flags and return candidates in sketch form.
|
|
46
48
|
3. Concrete benefit and tradeoffs.
|
|
47
49
|
4. Behavior to preserve and test approach.
|
|
48
50
|
5. Uncertainty and recommendation strength.
|
|
@@ -243,7 +243,10 @@ This section applies to peer and authored PR modes.
|
|
|
243
243
|
is safe because of on the evidence ladder; below "ran it" is unproven.
|
|
244
244
|
4. Requirements, acceptance, and user behavior.
|
|
245
245
|
5. Architecture and solution design, including SOLID and credible simpler
|
|
246
|
-
alternatives.
|
|
246
|
+
alternatives. Only when the scope identity carries a sketch, compare
|
|
247
|
+
the architecture with the [design lens](../axstack/references/design-lens.md)
|
|
248
|
+
sketch and red flags. A deviation from a `Binding` line without an
|
|
249
|
+
accepted spec revision is a finding.
|
|
247
250
|
6. Simplicity and maintainability: KISS, YAGNI, and cyclomatic complexity
|
|
248
251
|
where measurement is useful. Never invent a metric or demand an
|
|
249
252
|
abstraction merely to satisfy a principle.
|
|
@@ -35,7 +35,10 @@ and the lifecycle's [audit skill](../axstack-audit/SKILL.md) hook.
|
|
|
35
35
|
Missing access preserves the GitHub selection and stops the phase without
|
|
36
36
|
mutation or fallback. Markdown mode skips external access preflight.
|
|
37
37
|
3. **Draft with decision evidence.** Write observable acceptance criteria
|
|
38
|
-
and explicit exclusions in the selected store.
|
|
38
|
+
and explicit exclusions in the selected store. Only when the scope identity
|
|
39
|
+
carries a sketch, include the [design lens](../axstack/references/design-lens.md)
|
|
40
|
+
sketch in the approved revision's `Design` section and its `Usage` line in
|
|
41
|
+
acceptance. First record the driver's
|
|
39
42
|
independent assessment, then load
|
|
40
43
|
[Orca runtime](../axstack/references/orca-runtime.md) before dispatching the
|
|
41
44
|
configured `axstack-advisor-astra` and `axstack-advisor-fable` independently,
|
|
@@ -41,8 +41,12 @@ an actual checker dispatch, not for ordinary mapping or state reconciliation.
|
|
|
41
41
|
issues represent user-visible capabilities; one capability may span several
|
|
42
42
|
tasks and PRs. GitHub capability issues link the approved spec issue and its
|
|
43
43
|
SHA-256 body digest. Keep detailed execution breakdowns in the repository.
|
|
44
|
-
For every capability, derive acceptance checks from the pinned spec
|
|
45
|
-
|
|
44
|
+
For every capability, derive acceptance checks from the pinned spec. Only
|
|
45
|
+
when its scope identity carries a sketch, follow the
|
|
46
|
+
[design lens](../axstack/references/design-lens.md) sketch's modules, put
|
|
47
|
+
its named failure in capability acceptance, and treat a task spanning a
|
|
48
|
+
sketch boundary as a split signal. Identify internal tasks, dependencies,
|
|
49
|
+
PR ownership, and worktrees. For each task the driver records
|
|
46
50
|
one theme and a coarse size estimate from the ownership, interface, and
|
|
47
51
|
dependency map. A task estimated in the exception band is assessed for a
|
|
48
52
|
split at mapping time and split where a green, atomic, reviewable split
|
package/src/installer.js
CHANGED
|
@@ -626,6 +626,16 @@ export async function installBundle({
|
|
|
626
626
|
if (rel in ownedFiles) installedHashes[rel] = ownedFiles[rel];
|
|
627
627
|
}
|
|
628
628
|
}
|
|
629
|
+
const previousRoles = desired.find(({ rel }) => rel === 'axstack/roles.json')?.current;
|
|
630
|
+
if (previousRoles && summary.updated.some((rel) => rel.startsWith('axstack/roles.json'))) {
|
|
631
|
+
try {
|
|
632
|
+
const old = JSON.parse(previousRoles.toString());
|
|
633
|
+
if (Array.isArray(old.roles)) {
|
|
634
|
+
const oldIds = new Set(old.roles.map((role) => role?.id));
|
|
635
|
+
summary.addedRoleIds = bundle.bundleRoles.filter((role) => !oldIds.has(role.id)).map((role) => role.id);
|
|
636
|
+
}
|
|
637
|
+
} catch { /* No reliable role-ID diff for malformed prior bytes. */ }
|
|
638
|
+
}
|
|
629
639
|
|
|
630
640
|
// Stale manifest entries (owned files the bundle no longer ships):
|
|
631
641
|
// the phase-1 plan validated every stale destination read-only
|
package/src/roles.js
CHANGED
|
@@ -66,8 +66,10 @@ export function assessRoleReadiness(roles, preset) {
|
|
|
66
66
|
(preset === 'mixed' && role.id === 'axstack-checker' && role.provider === 'antigravity') ||
|
|
67
67
|
(preset === 'mixed' && role.id === 'axstack-research-web-google' && role.provider === 'antigravity') ||
|
|
68
68
|
(preset === 'mixed' && role.id === 'axstack-research-x' && role.provider === 'grok') ||
|
|
69
|
-
(preset === '
|
|
70
|
-
(preset === '
|
|
69
|
+
(preset === 'mixed' && role.id === 'axstack-arena-candidate-grok' && role.provider === 'grok') ||
|
|
70
|
+
(preset === 'mixed' && role.id === 'axstack-arena-candidate-antigravity' && role.provider === 'antigravity') ||
|
|
71
|
+
(preset === 'codex-only' && ['axstack-advisor-fable', 'axstack-arena-judge-fable', 'axstack-research-web-google', 'axstack-research-x', 'axstack-arena-candidate-grok', 'axstack-arena-candidate-antigravity'].includes(role.id)) ||
|
|
72
|
+
(preset === 'claude-only' && ['axstack-advisor-astra', 'axstack-arena-judge-astra', 'axstack-research-web-google', 'axstack-research-x', 'axstack-arena-candidate-grok', 'axstack-arena-candidate-antigravity'].includes(role.id))
|
|
71
73
|
);
|
|
72
74
|
for (const role of roles) {
|
|
73
75
|
if (!bounds.has(role.provider)) {
|