axstack 0.20.22 → 0.20.24
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/bin/axstack.js +2 -0
- package/docs/installation.md +12 -9
- package/docs/workflows.md +16 -12
- package/package.json +1 -1
- package/profiles/presets/claude-only.json +36 -9
- package/profiles/presets/codex-only.json +48 -21
- package/profiles/presets/mixed.json +44 -17
- package/skills/axstack/references/contracts.md +13 -4
- package/skills/axstack/references/design-lens.md +1 -1
- package/skills/axstack/references/orca-runtime.md +2 -2
- package/skills/axstack/references/routing.md +26 -28
- package/skills/axstack-align/SKILL.md +31 -23
- package/skills/axstack-audit/SKILL.md +7 -3
- package/skills/axstack-audit/references/record.md +3 -2
- package/skills/axstack-debug/SKILL.md +4 -3
- package/skills/axstack-review/SKILL.md +1 -1
- package/skills/axstack-spec/SKILL.md +3 -2
- package/src/installer.js +14 -0
- package/src/roles.js +5 -3
package/README.md
CHANGED
|
@@ -99,7 +99,7 @@ upgrades, conflicts, and uninstalling.
|
|
|
99
99
|
inline or from an explicitly scoped backlog without touching active, manual,
|
|
100
100
|
uncertain, user-owned, dirty, unpushed, or useful unmerged work.
|
|
101
101
|
|
|
102
|
-
Choose an explicit role preset:
|
|
102
|
+
Choose an explicit role preset (27 stable role IDs in each):
|
|
103
103
|
[mixed](profiles/presets/mixed.json),
|
|
104
104
|
[codex-only](profiles/presets/codex-only.json), or
|
|
105
105
|
[claude-only](profiles/presets/claude-only.json).
|
package/bin/axstack.js
CHANGED
|
@@ -335,6 +335,8 @@ async function main() {
|
|
|
335
335
|
} else {
|
|
336
336
|
if (summary.added.length) console.log(`added: ${summary.added.join(', ')}`);
|
|
337
337
|
if (summary.updated.length) console.log(`updated: ${summary.updated.join(', ')}`);
|
|
338
|
+
if (summary.addedRoleIds?.length) console.log(`added role IDs: ${summary.addedRoleIds.join(', ')}`);
|
|
339
|
+
if (summary.removedRoleIds?.length) console.log(`removed role IDs: ${summary.removedRoleIds.join(', ')}`);
|
|
338
340
|
if (summary.removed.length) console.log(`removed: ${summary.removed.join(', ')}`);
|
|
339
341
|
if (summary.unchanged.length) console.log(`unchanged: ${summary.unchanged.join(', ')}`);
|
|
340
342
|
}
|
package/docs/installation.md
CHANGED
|
@@ -71,7 +71,7 @@ profiles/presets/codex-only.json
|
|
|
71
71
|
profiles/presets/claude-only.json
|
|
72
72
|
```
|
|
73
73
|
|
|
74
|
-
Each has exactly `{ "version": 1, "roles": [...] }` with the same
|
|
74
|
+
Each has exactly `{ "version": 1, "roles": [...] }` with the same 27 stable
|
|
75
75
|
role IDs. Installation writes `<skills-dir>/axstack/roles.json` as
|
|
76
76
|
`{ "version": 1, "preset": "<selected preset>", "roles": [...] }` and records
|
|
77
77
|
its ownership hash like every other installed skill asset. There is no second
|
|
@@ -140,10 +140,11 @@ The complete bundle is validated before writes:
|
|
|
140
140
|
the filename's selected identity supplied by the caller, and the same role-ID
|
|
141
141
|
set as its peers;
|
|
142
142
|
- every role has valid preserved fields, while the mixed checker,
|
|
143
|
-
`axstack-research-web-google`,
|
|
143
|
+
`axstack-research-web-google`, `axstack-research-x`, and both arena candidate launch-by-agent-id
|
|
144
144
|
routes explicitly permit `model: null`;
|
|
145
|
-
in each single-provider preset, the unavailable adviser and
|
|
146
|
-
|
|
145
|
+
in each single-provider preset, the unavailable adviser and round-2 seat
|
|
146
|
+
explicitly permit `model: null`, as do both cross-provider research routes
|
|
147
|
+
and both arena candidate seats;
|
|
147
148
|
- obsolete runtime configuration flags fail before mutation with migration
|
|
148
149
|
guidance.
|
|
149
150
|
|
|
@@ -171,7 +172,7 @@ to rewrite them.
|
|
|
171
172
|
## Role behavior after installation
|
|
172
173
|
|
|
173
174
|
The runtime reads `roles.json` from the installed shared root `skills/axstack/`.
|
|
174
|
-
A new run records the selected preset plus all
|
|
175
|
+
A new run records the selected preset plus all 27 role rows. An active run keeps
|
|
175
176
|
that snapshot after a later preset install unless the user explicitly changes
|
|
176
177
|
it and accepts the resulting evidence invalidation.
|
|
177
178
|
|
|
@@ -181,10 +182,12 @@ The mixed checker and `axstack-research-web-google` have provider
|
|
|
181
182
|
their notes authorize launch by agent ID, and the run record snapshots the model
|
|
182
183
|
reported by the TUI. The single-provider presets configure the checker and keep
|
|
183
184
|
both cross-provider research routes as intentional absences. Their
|
|
184
|
-
unavailable adviser and
|
|
185
|
-
|
|
186
|
-
Align and Spec still hold until both Astra and
|
|
187
|
-
receipts
|
|
185
|
+
unavailable adviser and round-2 seat remain explicit same-provider
|
|
186
|
+
`model: null` roles, which do not make installation unready;
|
|
187
|
+
Align and Spec still hold until both Astra and Opus can return independent
|
|
188
|
+
receipts. For an arena-grade Align question, round 1 needs Opus; round 2, if
|
|
189
|
+
invoked, needs escalation Fable and Astra; a required seat that is unavailable holds that
|
|
190
|
+
round. The current chat drives on whatever
|
|
188
191
|
model runs it; no preset carries a driver role. Every other missing, invalid, unsupported, or unavailable role value holds only
|
|
189
192
|
the affected work. There is no model substitution, subscription inference, or
|
|
190
193
|
quota routing.
|
package/docs/workflows.md
CHANGED
|
@@ -48,16 +48,16 @@ only affected work.
|
|
|
48
48
|
|
|
49
49
|
Installation requires one explicit canonical preset. The three bundle files
|
|
50
50
|
under `profiles/presets/` each contain exactly
|
|
51
|
-
`{ "version": 1, "roles": [...] }` and the same
|
|
51
|
+
`{ "version": 1, "roles": [...] }` and the same 27 stable IDs.
|
|
52
52
|
|
|
53
53
|
The current chat drives on whatever model runs it; no preset carries a driver
|
|
54
54
|
role.
|
|
55
55
|
|
|
56
|
-
| Preset | Author | Ordered peer reviewers | Astra /
|
|
56
|
+
| Preset | Author | Ordered peer reviewers | Astra / Opus advisers | Auditor |
|
|
57
57
|
| --- | --- | --- | --- | --- |
|
|
58
|
-
| `mixed` | Sol high | Sol
|
|
59
|
-
| `codex-only` | Sol high | Sol
|
|
60
|
-
| `claude-only` | Opus medium | Opus medium; Sonnet xhigh | unavailable /
|
|
58
|
+
| `mixed` | Sol high | Sol high; Opus medium | Astra high / Opus xhigh | Luna xhigh |
|
|
59
|
+
| `codex-only` | Sol high | Sol high; Luna xhigh | Astra high / unavailable | Luna xhigh |
|
|
60
|
+
| `claude-only` | Opus medium | Opus medium; Sonnet xhigh | unavailable / Opus xhigh | Sonnet xhigh |
|
|
61
61
|
|
|
62
62
|
The installed `<skills-dir>/axstack/roles.json` adds the selected preset name:
|
|
63
63
|
`{ "version": 1, "preset": "<name>", "roles": [...] }`. The runtime reads it
|
|
@@ -120,14 +120,18 @@ session and evidence remain valid.
|
|
|
120
120
|
## Phases
|
|
121
121
|
|
|
122
122
|
- `axstack-align` maps facts and dependencies, asks prioritized questions, and
|
|
123
|
-
consults Astra and
|
|
123
|
+
consults Astra and Opus independently with the same bounded evidence and
|
|
124
124
|
question. It synthesizes disagreements and reuses unchanged receipts. For a
|
|
125
|
-
hard-to-reverse design choice it runs
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
125
|
+
hard-to-reverse design choice it runs an arena instead: Astra, Opus, Grok,
|
|
126
|
+
and Antigravity each author a candidate. `axstack-arena-judge-opus` scores
|
|
127
|
+
them in round 1; the driver compares its own pick with that verdict. If they
|
|
128
|
+
disagree on the base or the user rejects the round-1 synthesis,
|
|
129
|
+
`axstack-escalation-fable` and `axstack-arena-judge-astra` independently
|
|
130
|
+
score the same anonymized candidates and rubric in round 2. The driver
|
|
131
|
+
picks a base, grafts strong ideas, and records judge verdicts per round in
|
|
132
|
+
the `Decisions` rows without averaging. Fable escalation uses a fresh
|
|
133
|
+
session for round 2, high-stakes agreement, or the bounded trigger in
|
|
134
|
+
[Standing contracts](../skills/axstack/references/contracts.md).
|
|
131
135
|
- `axstack-spec` writes observable acceptance, exclusions, decisions, and one
|
|
132
136
|
user-approved revision baseline. Linear is the default authoritative store;
|
|
133
137
|
GitHub Issues and repository Markdown are explicit alternatives. A GitHub
|
package/package.json
CHANGED
|
@@ -11,13 +11,13 @@
|
|
|
11
11
|
"notes": "Required independent Astra adviser for Align, Spec, and unresolved consequential decisions; unavailable in claude-only. The explicit null holds Align and Spec without provider substitution."
|
|
12
12
|
},
|
|
13
13
|
{
|
|
14
|
-
"id": "axstack-advisor-
|
|
15
|
-
"name": "Axstack
|
|
14
|
+
"id": "axstack-advisor-opus",
|
|
15
|
+
"name": "Axstack Opus adviser",
|
|
16
16
|
"provider": "claude",
|
|
17
|
-
"model": "claude-
|
|
17
|
+
"model": "claude-opus-5-5",
|
|
18
18
|
"modeId": "bypassPermissions",
|
|
19
|
-
"thinkingOptionId": "
|
|
20
|
-
"notes": "Independent
|
|
19
|
+
"thinkingOptionId": "xhigh",
|
|
20
|
+
"notes": "Independent Opus adviser for Align, Spec, and debug L1; authors the Claude arena candidate. Same bounded evidence and question as Astra; reuse only unchanged receipts."
|
|
21
21
|
},
|
|
22
22
|
{
|
|
23
23
|
"id": "axstack-owner",
|
|
@@ -206,16 +206,43 @@
|
|
|
206
206
|
"model": null,
|
|
207
207
|
"modeId": "bypassPermissions",
|
|
208
208
|
"thinkingOptionId": "xhigh",
|
|
209
|
-
"notes": "Required Arena judge (Astra seat). Read-only cross-judge for an arena-grade Align decision: receives the rubric and
|
|
209
|
+
"notes": "Required Arena judge (Astra seat). Read-only cross-judge for an arena-grade Align decision: receives the rubric and every candidate by label, scores each criterion, and recommends a base with rationale. Never authors a candidate, never cross-reads the other judge, never mutates. Disagreement between judges is surfaced by the driver, never averaged. Intentionally absent in the claude-only preset; the arena-grade decision holds without substitution."
|
|
210
210
|
},
|
|
211
211
|
{
|
|
212
|
-
"id": "axstack-
|
|
213
|
-
"name": "Axstack
|
|
212
|
+
"id": "axstack-escalation-fable",
|
|
213
|
+
"name": "Axstack escalation Fable",
|
|
214
214
|
"provider": "claude",
|
|
215
215
|
"model": "claude-fable-5-1",
|
|
216
216
|
"modeId": "bypassPermissions",
|
|
217
217
|
"thinkingOptionId": "xhigh",
|
|
218
|
-
"notes": "
|
|
218
|
+
"notes": "Fable escalation seat for arena round 2, high-stakes plain AGREE, or the bounded escalation trigger. Fresh session per use; never reuses adviser or candidate context. Read-only round 2 judge: scores every candidate by label; never authors a candidate."
|
|
219
|
+
},
|
|
220
|
+
{
|
|
221
|
+
"id": "axstack-arena-judge-opus",
|
|
222
|
+
"name": "Axstack arena judge Opus",
|
|
223
|
+
"provider": "claude",
|
|
224
|
+
"model": "claude-opus-5-5",
|
|
225
|
+
"modeId": "bypassPermissions",
|
|
226
|
+
"thinkingOptionId": "xhigh",
|
|
227
|
+
"notes": "Read-only round 1 arena judge. Receives the rubric and every candidate by label, scores each criterion, and recommends a base with rationale. Never authors a candidate, never cross-reads another judge, never mutates; the driver compares its verdict without averaging."
|
|
228
|
+
},
|
|
229
|
+
{
|
|
230
|
+
"id": "axstack-arena-candidate-grok",
|
|
231
|
+
"name": "Axstack arena candidate Grok (unavailable)",
|
|
232
|
+
"provider": "claude",
|
|
233
|
+
"model": null,
|
|
234
|
+
"modeId": "bypassPermissions",
|
|
235
|
+
"thinkingOptionId": "high",
|
|
236
|
+
"notes": "Intentional single-provider absence in the claude-only preset; Grok candidate unavailable. Arena holds without substitution."
|
|
237
|
+
},
|
|
238
|
+
{
|
|
239
|
+
"id": "axstack-arena-candidate-antigravity",
|
|
240
|
+
"name": "Axstack arena candidate Antigravity (unavailable)",
|
|
241
|
+
"provider": "claude",
|
|
242
|
+
"model": null,
|
|
243
|
+
"modeId": "bypassPermissions",
|
|
244
|
+
"thinkingOptionId": "high",
|
|
245
|
+
"notes": "Intentional single-provider absence in the claude-only preset; Antigravity candidate unavailable. Arena holds without substitution."
|
|
219
246
|
}
|
|
220
247
|
]
|
|
221
248
|
}
|
|
@@ -8,16 +8,16 @@
|
|
|
8
8
|
"model": "gpt-6-astra",
|
|
9
9
|
"modeId": "full-access",
|
|
10
10
|
"thinkingOptionId": "high",
|
|
11
|
-
"notes": "Independent Astra adviser for Align, Spec, and unresolved consequential decisions. Receives the same bounded evidence and question as
|
|
11
|
+
"notes": "Independent Astra adviser for Align, Spec, and unresolved consequential decisions. Receives the same bounded evidence and question as Opus; reuse only unchanged receipts."
|
|
12
12
|
},
|
|
13
13
|
{
|
|
14
|
-
"id": "axstack-advisor-
|
|
15
|
-
"name": "Axstack
|
|
14
|
+
"id": "axstack-advisor-opus",
|
|
15
|
+
"name": "Axstack Opus adviser (unavailable)",
|
|
16
16
|
"provider": "codex",
|
|
17
17
|
"model": null,
|
|
18
18
|
"modeId": "full-access",
|
|
19
|
-
"thinkingOptionId": "
|
|
20
|
-
"notes": "
|
|
19
|
+
"thinkingOptionId": "xhigh",
|
|
20
|
+
"notes": "Opus adviser intentionally absent in codex-only; the explicit null holds Align and Spec without substitution."
|
|
21
21
|
},
|
|
22
22
|
{
|
|
23
23
|
"id": "axstack-owner",
|
|
@@ -43,8 +43,8 @@
|
|
|
43
43
|
"provider": "codex",
|
|
44
44
|
"model": "gpt-6-sol",
|
|
45
45
|
"modeId": "full-access",
|
|
46
|
-
"thinkingOptionId": "
|
|
47
|
-
"notes": "Primary reviewer in the ordered codex-only peer pair: Sol
|
|
46
|
+
"thinkingOptionId": "high",
|
|
47
|
+
"notes": "Primary reviewer in the ordered codex-only peer pair: Sol high followed by Luna xhigh. Never author or owner; review exact SHA and base across all six angles and acceptance. Peer first pass stays isolated."
|
|
48
48
|
},
|
|
49
49
|
{
|
|
50
50
|
"id": "axstack-reviewer-secondary",
|
|
@@ -53,7 +53,7 @@
|
|
|
53
53
|
"model": "gpt-6-luna",
|
|
54
54
|
"modeId": "full-access",
|
|
55
55
|
"thinkingOptionId": "xhigh",
|
|
56
|
-
"notes": "Secondary reviewer in the ordered codex-only peer pair: Sol
|
|
56
|
+
"notes": "Secondary reviewer in the ordered codex-only peer pair: Sol high followed by Luna xhigh. Eligible authored reviewer for a Sol-authored candidate. Never author or owner; review exact SHA and base across all six angles and acceptance. Peer first pass stays isolated."
|
|
57
57
|
},
|
|
58
58
|
{
|
|
59
59
|
"id": "axstack-checker",
|
|
@@ -79,7 +79,7 @@
|
|
|
79
79
|
"provider": "codex",
|
|
80
80
|
"model": "gpt-6-sol",
|
|
81
81
|
"modeId": "full-access",
|
|
82
|
-
"thinkingOptionId": "
|
|
82
|
+
"thinkingOptionId": "high",
|
|
83
83
|
"notes": "Research code investigator: verifies behavior against inspected code and executable evidence. Validate configured availability at launch; hold affected work without fallback."
|
|
84
84
|
},
|
|
85
85
|
{
|
|
@@ -133,7 +133,7 @@
|
|
|
133
133
|
"provider": "codex",
|
|
134
134
|
"model": "gpt-6-sol",
|
|
135
135
|
"modeId": "full-access",
|
|
136
|
-
"thinkingOptionId": "
|
|
136
|
+
"thinkingOptionId": "high",
|
|
137
137
|
"notes": "Codebase mapper: explores repository structure and interfaces for research and handoff context. Validate configured availability at launch; hold affected work without fallback."
|
|
138
138
|
},
|
|
139
139
|
{
|
|
@@ -142,7 +142,7 @@
|
|
|
142
142
|
"provider": "codex",
|
|
143
143
|
"model": "gpt-6-sol",
|
|
144
144
|
"modeId": "full-access",
|
|
145
|
-
"thinkingOptionId": "
|
|
145
|
+
"thinkingOptionId": "high",
|
|
146
146
|
"notes": "Execution explorer: runs bounded checks of runtime behavior where authorized. Validate configured availability at launch; hold affected work without fallback."
|
|
147
147
|
},
|
|
148
148
|
{
|
|
@@ -169,8 +169,8 @@
|
|
|
169
169
|
"provider": "codex",
|
|
170
170
|
"model": "gpt-6-sol",
|
|
171
171
|
"modeId": "full-access",
|
|
172
|
-
"thinkingOptionId": "
|
|
173
|
-
"notes": "Debug investigator seat 1. Dispatched only by axstack-debug at L1 with the shared evidence packet and one distinct brief; never reads another investigator's output. Works in its own disposable worktree at the pinned revision plus the recorded dirty patch; may instrument there for probes; never commits, pushes, publishes, or creates children. Returns one receipt per brief. Independence comes from brief isolation, not model diversity. This preset repeats gpt-6-sol at
|
|
172
|
+
"thinkingOptionId": "high",
|
|
173
|
+
"notes": "Debug investigator seat 1. Dispatched only by axstack-debug at L1 with the shared evidence packet and one distinct brief; never reads another investigator's output. Works in its own disposable worktree at the pinned revision plus the recorded dirty patch; may instrument there for probes; never commits, pushes, publishes, or creates children. Returns one receipt per brief. Independence comes from brief isolation, not model diversity. This preset repeats gpt-6-sol at high effort because it has fewer model families."
|
|
174
174
|
},
|
|
175
175
|
{
|
|
176
176
|
"id": "axstack-debug-investigator-2",
|
|
@@ -178,8 +178,8 @@
|
|
|
178
178
|
"provider": "codex",
|
|
179
179
|
"model": "gpt-6-sol",
|
|
180
180
|
"modeId": "full-access",
|
|
181
|
-
"thinkingOptionId": "
|
|
182
|
-
"notes": "Debug investigator seat 2. Dispatched only by axstack-debug at L1 with the shared evidence packet and one distinct brief; never reads another investigator's output. Works in its own disposable worktree at the pinned revision plus the recorded dirty patch; may instrument there for probes; never commits, pushes, publishes, or creates children. Returns one receipt per brief. Independence comes from brief isolation, not model diversity. This preset repeats gpt-6-sol at
|
|
181
|
+
"thinkingOptionId": "high",
|
|
182
|
+
"notes": "Debug investigator seat 2. Dispatched only by axstack-debug at L1 with the shared evidence packet and one distinct brief; never reads another investigator's output. Works in its own disposable worktree at the pinned revision plus the recorded dirty patch; may instrument there for probes; never commits, pushes, publishes, or creates children. Returns one receipt per brief. Independence comes from brief isolation, not model diversity. This preset repeats gpt-6-sol at high effort because it has fewer model families."
|
|
183
183
|
},
|
|
184
184
|
{
|
|
185
185
|
"id": "axstack-debug-investigator-3",
|
|
@@ -196,8 +196,8 @@
|
|
|
196
196
|
"provider": "codex",
|
|
197
197
|
"model": "gpt-6-sol",
|
|
198
198
|
"modeId": "full-access",
|
|
199
|
-
"thinkingOptionId": "
|
|
200
|
-
"notes": "Debug investigator seat 4. Dispatched only by axstack-debug at L1 with the shared evidence packet and one distinct brief; never reads another investigator's output. Works in its own disposable worktree at the pinned revision plus the recorded dirty patch; may instrument there for probes; never commits, pushes, publishes, or creates children. Returns one receipt per brief. Independence comes from brief isolation, not model diversity. This preset repeats gpt-6-sol at
|
|
199
|
+
"thinkingOptionId": "high",
|
|
200
|
+
"notes": "Debug investigator seat 4. Dispatched only by axstack-debug at L1 with the shared evidence packet and one distinct brief; never reads another investigator's output. Works in its own disposable worktree at the pinned revision plus the recorded dirty patch; may instrument there for probes; never commits, pushes, publishes, or creates children. Returns one receipt per brief. Independence comes from brief isolation, not model diversity. This preset repeats gpt-6-sol at high effort because it has fewer model families."
|
|
201
201
|
},
|
|
202
202
|
{
|
|
203
203
|
"id": "axstack-arena-judge-astra",
|
|
@@ -206,16 +206,43 @@
|
|
|
206
206
|
"model": "gpt-6-astra",
|
|
207
207
|
"modeId": "full-access",
|
|
208
208
|
"thinkingOptionId": "xhigh",
|
|
209
|
-
"notes": "Arena judge (Astra seat). Read-only cross-judge for an arena-grade Align decision: receives the rubric and
|
|
209
|
+
"notes": "Arena judge (Astra seat). Read-only cross-judge for an arena-grade Align decision: receives the rubric and every candidate by label, scores each criterion, and recommends a base with rationale. Never authors a candidate, never cross-reads the other judge, never mutates. Disagreement between judges is surfaced by the driver, never averaged."
|
|
210
|
+
},
|
|
211
|
+
{
|
|
212
|
+
"id": "axstack-escalation-fable",
|
|
213
|
+
"name": "Axstack escalation Fable (unavailable)",
|
|
214
|
+
"provider": "codex",
|
|
215
|
+
"model": null,
|
|
216
|
+
"modeId": "full-access",
|
|
217
|
+
"thinkingOptionId": "xhigh",
|
|
218
|
+
"notes": "Fable escalation intentionally absent in codex-only; a required use holds without substitution. Read-only round 2 judge; never authors a candidate."
|
|
210
219
|
},
|
|
211
220
|
{
|
|
212
|
-
"id": "axstack-arena-judge-
|
|
213
|
-
"name": "Axstack arena judge
|
|
221
|
+
"id": "axstack-arena-judge-opus",
|
|
222
|
+
"name": "Axstack arena judge Opus",
|
|
214
223
|
"provider": "codex",
|
|
215
224
|
"model": null,
|
|
216
225
|
"modeId": "full-access",
|
|
217
226
|
"thinkingOptionId": "xhigh",
|
|
218
|
-
"notes": "
|
|
227
|
+
"notes": "Read-only round 1 arena judge. Receives the rubric and every candidate by label, scores each criterion, and recommends a base with rationale. Never authors a candidate, never cross-reads another judge, never mutates; the driver compares its verdict without averaging. Intentionally absent in the codex-only preset; round 1 holds without substitution."
|
|
228
|
+
},
|
|
229
|
+
{
|
|
230
|
+
"id": "axstack-arena-candidate-grok",
|
|
231
|
+
"name": "Axstack arena candidate Grok (unavailable)",
|
|
232
|
+
"provider": "codex",
|
|
233
|
+
"model": null,
|
|
234
|
+
"modeId": "full-access",
|
|
235
|
+
"thinkingOptionId": "high",
|
|
236
|
+
"notes": "Intentional single-provider absence in the codex-only preset; Grok candidate unavailable. Arena holds without substitution."
|
|
237
|
+
},
|
|
238
|
+
{
|
|
239
|
+
"id": "axstack-arena-candidate-antigravity",
|
|
240
|
+
"name": "Axstack arena candidate Antigravity (unavailable)",
|
|
241
|
+
"provider": "codex",
|
|
242
|
+
"model": null,
|
|
243
|
+
"modeId": "full-access",
|
|
244
|
+
"thinkingOptionId": "high",
|
|
245
|
+
"notes": "Intentional single-provider absence in the codex-only preset; Antigravity candidate unavailable. Arena holds without substitution."
|
|
219
246
|
}
|
|
220
247
|
]
|
|
221
248
|
}
|
|
@@ -8,16 +8,16 @@
|
|
|
8
8
|
"model": "gpt-6-astra",
|
|
9
9
|
"modeId": "full-access",
|
|
10
10
|
"thinkingOptionId": "high",
|
|
11
|
-
"notes": "Independent Astra adviser for Align, Spec, and unresolved consequential decisions. Receives the same bounded evidence and question as
|
|
11
|
+
"notes": "Independent Astra adviser for Align, Spec, and unresolved consequential decisions. Receives the same bounded evidence and question as Opus; reuse only unchanged receipts."
|
|
12
12
|
},
|
|
13
13
|
{
|
|
14
|
-
"id": "axstack-advisor-
|
|
15
|
-
"name": "Axstack
|
|
14
|
+
"id": "axstack-advisor-opus",
|
|
15
|
+
"name": "Axstack Opus adviser",
|
|
16
16
|
"provider": "claude",
|
|
17
|
-
"model": "claude-
|
|
17
|
+
"model": "claude-opus-5-5",
|
|
18
18
|
"modeId": "bypassPermissions",
|
|
19
|
-
"thinkingOptionId": "
|
|
20
|
-
"notes": "Independent
|
|
19
|
+
"thinkingOptionId": "xhigh",
|
|
20
|
+
"notes": "Independent Opus adviser for Align, Spec, and debug L1; authors the Claude arena candidate. Same bounded evidence and question as Astra; reuse only unchanged receipts."
|
|
21
21
|
},
|
|
22
22
|
{
|
|
23
23
|
"id": "axstack-owner",
|
|
@@ -43,8 +43,8 @@
|
|
|
43
43
|
"provider": "codex",
|
|
44
44
|
"model": "gpt-6-sol",
|
|
45
45
|
"modeId": "full-access",
|
|
46
|
-
"thinkingOptionId": "
|
|
47
|
-
"notes": "Primary reviewer in the ordered mixed peer pair: Sol
|
|
46
|
+
"thinkingOptionId": "high",
|
|
47
|
+
"notes": "Primary reviewer in the ordered mixed peer pair: Sol high followed by Opus medium. Eligible authored reviewer for an Opus-authored candidate. Never author or owner; review exact SHA and base across all six angles and acceptance. Peer first pass stays isolated."
|
|
48
48
|
},
|
|
49
49
|
{
|
|
50
50
|
"id": "axstack-reviewer-secondary",
|
|
@@ -53,7 +53,7 @@
|
|
|
53
53
|
"model": "claude-opus-5-5",
|
|
54
54
|
"modeId": "bypassPermissions",
|
|
55
55
|
"thinkingOptionId": "medium",
|
|
56
|
-
"notes": "Secondary reviewer in the ordered mixed peer pair: Sol
|
|
56
|
+
"notes": "Secondary reviewer in the ordered mixed peer pair: Sol high followed by Opus medium. Eligible authored reviewer for a Sol-authored candidate. Never author or owner; review exact SHA and base across all six angles and acceptance. Peer first pass stays isolated."
|
|
57
57
|
},
|
|
58
58
|
{
|
|
59
59
|
"id": "axstack-checker",
|
|
@@ -79,7 +79,7 @@
|
|
|
79
79
|
"provider": "codex",
|
|
80
80
|
"model": "gpt-6-sol",
|
|
81
81
|
"modeId": "full-access",
|
|
82
|
-
"thinkingOptionId": "
|
|
82
|
+
"thinkingOptionId": "high",
|
|
83
83
|
"notes": "Research code investigator: verifies behavior against inspected code and executable evidence. Validate configured availability at launch; hold affected work without fallback."
|
|
84
84
|
},
|
|
85
85
|
{
|
|
@@ -142,7 +142,7 @@
|
|
|
142
142
|
"provider": "codex",
|
|
143
143
|
"model": "gpt-6-sol",
|
|
144
144
|
"modeId": "full-access",
|
|
145
|
-
"thinkingOptionId": "
|
|
145
|
+
"thinkingOptionId": "high",
|
|
146
146
|
"notes": "Execution explorer: runs bounded checks of runtime behavior where authorized. Validate configured availability at launch; hold affected work without fallback."
|
|
147
147
|
},
|
|
148
148
|
{
|
|
@@ -178,7 +178,7 @@
|
|
|
178
178
|
"provider": "codex",
|
|
179
179
|
"model": "gpt-6-sol",
|
|
180
180
|
"modeId": "full-access",
|
|
181
|
-
"thinkingOptionId": "
|
|
181
|
+
"thinkingOptionId": "high",
|
|
182
182
|
"notes": "Debug investigator seat 2. Dispatched only by axstack-debug at L1 with the shared evidence packet and one distinct brief; never reads another investigator's output. Works in its own disposable worktree at the pinned revision plus the recorded dirty patch; may instrument there for probes; never commits, pushes, publishes, or creates children. Returns one receipt per brief. Independence comes from brief isolation, not model diversity."
|
|
183
183
|
},
|
|
184
184
|
{
|
|
@@ -196,7 +196,7 @@
|
|
|
196
196
|
"provider": "codex",
|
|
197
197
|
"model": "gpt-6-sol",
|
|
198
198
|
"modeId": "full-access",
|
|
199
|
-
"thinkingOptionId": "
|
|
199
|
+
"thinkingOptionId": "high",
|
|
200
200
|
"notes": "Debug investigator seat 4. Dispatched only by axstack-debug at L1 with the shared evidence packet and one distinct brief; never reads another investigator's output. Works in its own disposable worktree at the pinned revision plus the recorded dirty patch; may instrument there for probes; never commits, pushes, publishes, or creates children. Returns one receipt per brief. Independence comes from brief isolation, not model diversity."
|
|
201
201
|
},
|
|
202
202
|
{
|
|
@@ -206,16 +206,43 @@
|
|
|
206
206
|
"model": "gpt-6-astra",
|
|
207
207
|
"modeId": "full-access",
|
|
208
208
|
"thinkingOptionId": "xhigh",
|
|
209
|
-
"notes": "Arena judge (Astra seat). Read-only cross-judge for an arena-grade Align decision: receives the rubric and
|
|
209
|
+
"notes": "Arena judge (Astra seat). Read-only cross-judge for an arena-grade Align decision: receives the rubric and every candidate by label, scores each criterion, and recommends a base with rationale. Never authors a candidate, never cross-reads the other judge, never mutates. Disagreement between judges is surfaced by the driver, never averaged."
|
|
210
210
|
},
|
|
211
211
|
{
|
|
212
|
-
"id": "axstack-
|
|
213
|
-
"name": "Axstack
|
|
212
|
+
"id": "axstack-escalation-fable",
|
|
213
|
+
"name": "Axstack escalation Fable",
|
|
214
214
|
"provider": "claude",
|
|
215
215
|
"model": "claude-fable-5-1",
|
|
216
216
|
"modeId": "bypassPermissions",
|
|
217
217
|
"thinkingOptionId": "xhigh",
|
|
218
|
-
"notes": "
|
|
218
|
+
"notes": "Fable escalation seat for arena round 2, high-stakes plain AGREE, or the bounded escalation trigger. Fresh session per use; never reuses adviser or candidate context. Read-only round 2 judge: scores every candidate by label; never authors a candidate."
|
|
219
|
+
},
|
|
220
|
+
{
|
|
221
|
+
"id": "axstack-arena-judge-opus",
|
|
222
|
+
"name": "Axstack arena judge Opus",
|
|
223
|
+
"provider": "claude",
|
|
224
|
+
"model": "claude-opus-5-5",
|
|
225
|
+
"modeId": "bypassPermissions",
|
|
226
|
+
"thinkingOptionId": "xhigh",
|
|
227
|
+
"notes": "Read-only round 1 arena judge. Receives the rubric and every candidate by label, scores each criterion, and recommends a base with rationale. Never authors a candidate, never cross-reads another judge, never mutates; the driver compares its verdict without averaging."
|
|
228
|
+
},
|
|
229
|
+
{
|
|
230
|
+
"id": "axstack-arena-candidate-grok",
|
|
231
|
+
"name": "Axstack arena candidate Grok",
|
|
232
|
+
"provider": "grok",
|
|
233
|
+
"model": null,
|
|
234
|
+
"modeId": "full-access",
|
|
235
|
+
"thinkingOptionId": "high",
|
|
236
|
+
"notes": "Independent Grok Rung 2 design candidate. Launch by agent id grok; model selected by the TUI default. Receives the same brief without cross-reading; returns design, rationale, and rejected alternatives."
|
|
237
|
+
},
|
|
238
|
+
{
|
|
239
|
+
"id": "axstack-arena-candidate-antigravity",
|
|
240
|
+
"name": "Axstack arena candidate Antigravity",
|
|
241
|
+
"provider": "antigravity",
|
|
242
|
+
"model": null,
|
|
243
|
+
"modeId": "full-access",
|
|
244
|
+
"thinkingOptionId": "high",
|
|
245
|
+
"notes": "Independent Antigravity Rung 2 design candidate. Launch by agent id antigravity; model selected by the TUI default. Receives the same brief without cross-reading; returns design, rationale, and rejected alternatives."
|
|
219
246
|
}
|
|
220
247
|
]
|
|
221
248
|
}
|
|
@@ -48,7 +48,7 @@ The current chat is the driver, whatever model runs it; there is no driver
|
|
|
48
48
|
profile. Record the driver's provider and model in the run record.
|
|
49
49
|
|
|
50
50
|
For Align and Spec, the driver forms an independent assessment first, then
|
|
51
|
-
consults `axstack-advisor-astra` and `axstack-advisor-
|
|
51
|
+
consults `axstack-advisor-astra` and `axstack-advisor-opus` independently with
|
|
52
52
|
the same bounded evidence and question. The one exception is an arena-grade
|
|
53
53
|
Align question: there the driver frames the brief and rubric, the advisers
|
|
54
54
|
author candidates, and the driver assesses only after the candidates and judge
|
|
@@ -62,11 +62,11 @@ For `axstack-debug`, ordinary diagnosis consults the preset's configured
|
|
|
62
62
|
adviser roles (both in `mixed`; the one configured adviser in a single-provider
|
|
63
63
|
preset, recording the other as an intentional absence). The high-stakes and
|
|
64
64
|
serious-risk contracts override that rule whenever their conditions arise. A
|
|
65
|
-
configured but unavailable adviser holds debug L1
|
|
65
|
+
configured but unavailable adviser holds debug L1; an unavailable escalation seat holds L2 without substitution.
|
|
66
66
|
Reuse a debug receipt while its evidence packet is unchanged.
|
|
67
67
|
|
|
68
|
-
High-stakes decisions require
|
|
69
|
-
accepted assessment. Resolve disagreement with bounded checks; silence and an
|
|
68
|
+
High-stakes decisions require `axstack-advisor-astra` and
|
|
69
|
+
`axstack-escalation-fable` plain AGREE plus the driver's accepted assessment. Resolve disagreement with bounded checks; silence and an
|
|
70
70
|
unavailable model do not authorize fallback. Ordinary work uses the configured
|
|
71
71
|
author and reviewer roles selected by review mode and the routing snapshot. The
|
|
72
72
|
existing mixed high-stakes route keeps its Opus high author and Sol high
|
|
@@ -76,6 +76,15 @@ its effort and do not add a redundant reviewer. No single-provider high-stakes
|
|
|
76
76
|
mapping is defined: pause for an explicit user decision rather than borrowing
|
|
77
77
|
another preset or inventing a route. There is no silent fallback.
|
|
78
78
|
|
|
79
|
+
Use `axstack-escalation-fable` only for arena round 2, high-stakes decisions,
|
|
80
|
+
or when two attempts at the same named goal fail a stated acceptance check
|
|
81
|
+
(including debug L2 or a spec decision rejected twice). It also applies when
|
|
82
|
+
Astra and Opus still contradict after one reconciliation with no safe
|
|
83
|
+
discriminating check. A bare "stuck" claim needs pointers to two failed attempts;
|
|
84
|
+
setup slips and new user requirements do not count. Each use starts a fresh
|
|
85
|
+
session that never reuses an adviser or candidate context. Escalation never
|
|
86
|
+
resets existing holds or attempt budgets.
|
|
87
|
+
|
|
79
88
|
## Serious risk
|
|
80
89
|
|
|
81
90
|
Raise credible serious security, downtime, data-loss, or major-design risk
|
|
@@ -15,7 +15,7 @@ sketch through the scope identity.
|
|
|
15
15
|
not reclassify work: Rung 1 can stay small. An unsettled material design
|
|
16
16
|
question still makes routing reassess size.
|
|
17
17
|
- **Rung 2 — arena.** A Rung 1 design that also meets the existing ADR test:
|
|
18
|
-
a meaningful, hard-to-reverse, non-obvious trade-off. Use Align's
|
|
18
|
+
a meaningful, hard-to-reverse, non-obvious trade-off. Use Align's all-family
|
|
19
19
|
arena for that question.
|
|
20
20
|
|
|
21
21
|
There is no numeric threshold, file-count gate, or class-count gate.
|
|
@@ -38,7 +38,7 @@ Read `roles.json` from the installed shared root `skills/axstack/`. The installe
|
|
|
38
38
|
shape is `{ "version": 1, "preset": "<name>", "roles": [...] }`. Bundled
|
|
39
39
|
profiles are setup inputs shaped as
|
|
40
40
|
`{ "version": 1, "roles": [...] }`. A new run records the selected preset and
|
|
41
|
-
all
|
|
41
|
+
all 27 role rows once. An active run keeps the exact snapshot until the user
|
|
42
42
|
explicitly changes it.
|
|
43
43
|
|
|
44
44
|
Select the requested role by stable ID. A missing or null model holds only that role;
|
|
@@ -50,7 +50,7 @@ permission fields are conservative intent, not proof of effective permission
|
|
|
50
50
|
parity or a security boundary. Requested settings, input acceptance, effective
|
|
51
51
|
settings, and completed work are separate evidence. An unsupported or
|
|
52
52
|
unavailable value holds affected work for the user's decision without fallback.
|
|
53
|
-
The single-provider preset's null adviser
|
|
53
|
+
The single-provider preset's null adviser and round-2 seat are intentional installation data, not
|
|
54
54
|
readiness failure; because Align and Spec require both adviser receipts, either
|
|
55
55
|
null adviser still holds those phases. The current chat is the driver and has
|
|
56
56
|
no role row in any preset.
|
|
@@ -5,15 +5,14 @@ Driver entry sweep follows [Workspace hygiene](workspace-hygiene.md).
|
|
|
5
5
|
|
|
6
6
|
## Role routing
|
|
7
7
|
|
|
8
|
-
Presets: `mixed`, `codex-only`, `claude-only`. For
|
|
8
|
+
Presets: `mixed`, `codex-only`, `claude-only`. For new runs, use
|
|
9
9
|
`profiles.preset` from `.axstack-manifest.json` at the actually loaded
|
|
10
|
-
skills root, or an explicit user selection
|
|
11
|
-
only with exactly one unambiguous preset; missing or contradictory sources are
|
|
10
|
+
skills root, or an explicit user selection in the run record. Missing or contradictory sources are
|
|
12
11
|
a setup gap: hold. Never infer from live profiles or `list_profiles`, harness,
|
|
13
12
|
tools, credentials, quota, subscription, or default to `mixed`.
|
|
14
13
|
|
|
15
|
-
At
|
|
16
|
-
or unconfigured roles are recorded explicitly;
|
|
14
|
+
At start, snapshot all 27 role IDs with provider/model/mode/effort; absent
|
|
15
|
+
or unconfigured roles are recorded explicitly; never default.
|
|
17
16
|
Such a role holds only that role's work. A role installed or changed later must not
|
|
18
17
|
silently enter the snapshot; adding it needs an explicit user decision. Live profiles
|
|
19
18
|
are authoritative at snapshot time and for availability; bundled presets are setup
|
|
@@ -21,11 +20,9 @@ inputs, not runtime proof.
|
|
|
21
20
|
|
|
22
21
|
Preset changes apply to new runs only; an active run keeps its snapshot.
|
|
23
22
|
Changing it or replacing a session needs an explicit user decision and
|
|
24
|
-
revalidation. Unavailable models,
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
Role IDs:
|
|
23
|
+
revalidation. Unavailable models, efforts, roles, or overrides hold only affected
|
|
24
|
+
work; no automatic fallback, quota routing, subscription inference, or silent
|
|
25
|
+
provider/model/effort substitution.
|
|
29
26
|
|
|
30
27
|
- Chat drives (no role ID); `axstack-owner` owns one PR and
|
|
31
28
|
`axstack-author` its sole writer.
|
|
@@ -35,13 +32,15 @@ Role IDs:
|
|
|
35
32
|
| Preset | Author | Reviewer (model/effort) |
|
|
36
33
|
| --- | --- | --- |
|
|
37
34
|
| `mixed` | Codex / Sol (`codex/gpt-6-sol`) | `axstack-reviewer-secondary` (`claude/claude-opus-5-5` medium) |
|
|
38
|
-
| `mixed` | Claude / Opus (`claude/claude-opus-5-5`) | `axstack-reviewer-primary` (`codex/gpt-6-sol`
|
|
35
|
+
| `mixed` | Claude / Opus (`claude/claude-opus-5-5`) | `axstack-reviewer-primary` (`codex/gpt-6-sol` high) |
|
|
39
36
|
| `codex-only` | Codex / Sol (`codex/gpt-6-sol`) | `axstack-reviewer-secondary` (`codex/gpt-6-luna` xhigh) |
|
|
40
37
|
| `claude-only` | Claude / Opus (`claude/claude-opus-5-5`) | `axstack-reviewer-secondary` (`claude/claude-sonnet-5` xhigh) |
|
|
41
|
-
- `axstack-advisor-astra
|
|
42
|
-
|
|
43
|
-
`axstack-arena-
|
|
44
|
-
`axstack-
|
|
38
|
+
- `axstack-advisor-astra`/`axstack-advisor-opus` advise and author candidates;
|
|
39
|
+
`axstack-arena-candidate-grok`/
|
|
40
|
+
`axstack-arena-candidate-antigravity` add families.
|
|
41
|
+
`axstack-arena-judge-opus` judges round 1; `axstack-escalation-fable`/`axstack-arena-judge-astra` judge round 2.
|
|
42
|
+
High-stakes/trigger: fresh [contract](contracts.md) session.
|
|
43
|
+
`axstack-auditor` audits; `axstack-checker` reports discrepancies.
|
|
45
44
|
- `axstack-explainer`/`axstack-explainer-review`: explain/review.
|
|
46
45
|
`axstack-monitor`: standalone watch never sends; chat-run watch: bounded
|
|
47
46
|
internal reports to its Run and original driver.
|
|
@@ -51,17 +50,16 @@ Provenance is matched on provider/model ID; effort never maps. Missing table-row
|
|
|
51
50
|
provenance is unsupported and `INCOMPLETE`; report it and ask the user. Never
|
|
52
51
|
infer from slot, driver, owner, or provider. Author and owner never review.
|
|
53
52
|
|
|
54
|
-
|
|
55
|
-
step (3) for user routing
|
|
53
|
+
`axstack-implement` requires `mixed`; single-provider presets hold at
|
|
54
|
+
step (3) for user routing: no substitution or same-provider review.
|
|
56
55
|
|
|
57
56
|
## Direct routes (no spec ceremony)
|
|
58
57
|
|
|
59
58
|
- One bounded research question -> `axstack-research`: verify primary sources
|
|
60
|
-
and code
|
|
61
|
-
questions.
|
|
59
|
+
and code; return a cited note with limitations. Fan out only distinct questions.
|
|
62
60
|
- Understanding a system, change, or implementation gap -> `axstack-explain`:
|
|
63
|
-
current/intended behavior, evidence dimensions,
|
|
64
|
-
|
|
61
|
+
current/intended behavior, evidence dimensions, bounded gaps from project docs and
|
|
62
|
+
rendered behavior. "What could this break" follows
|
|
65
63
|
[Blast radius](blast-radius.md). Publication needs separate authority.
|
|
66
64
|
- A bug, failing test, regression, or wrong behavior, red loop wanted ->
|
|
67
65
|
`axstack-debug`: diagnose, escalate via adviser-directed investigators, hand
|
|
@@ -69,11 +67,11 @@ step (3) for user routing, with no substitution or same-provider review.
|
|
|
69
67
|
- Codebase-quality or refactor discovery -> `axstack-improve`: inspect bounded
|
|
70
68
|
scope, rank evidenced candidates, report only; no spec, tickets, or source
|
|
71
69
|
edits.
|
|
72
|
-
- Accepted worker/Task/Run completion or bounded backlog request ->
|
|
73
|
-
`axstack-cleanup` inline
|
|
70
|
+
- Accepted worker/Task/Run completion or bounded backlog request -> driver invokes
|
|
71
|
+
`axstack-cleanup` inline; never dispatch it.
|
|
74
72
|
- Preparation completion, watch expiry, resume, or reconciliation -> the
|
|
75
|
-
[lifecycle](lifecycle.md#native-handoff-and-resume): reconcile
|
|
76
|
-
|
|
73
|
+
[lifecycle](lifecycle.md#native-handoff-and-resume): reconcile run record,
|
|
74
|
+
keep owner, launch no native handoff.
|
|
77
75
|
- Explicit user-requested ownership transfer -> the same lifecycle section.
|
|
78
76
|
Load the [Orca runtime boundary](orca-runtime.md), follow the runtime-owned
|
|
79
77
|
handoff guide, and require explicit recipient acceptance before ownership
|
|
@@ -81,10 +79,10 @@ step (3) for user routing, with no substitution or same-provider review.
|
|
|
81
79
|
- Colleague PR review -> `axstack-review`, peer mode.
|
|
82
80
|
- Codebase review -> `axstack-review` codebase mode, report only.
|
|
83
81
|
- A status question about an own open PR or stack ("check now", "what's left",
|
|
84
|
-
"are we done", or "is it approved") -> `axstack-watch`
|
|
82
|
+
"are we done", or "is it approved") -> `axstack-watch` observation-only
|
|
85
83
|
mode. Explicit "address", "patch", or "fix" grants authorized maintenance.
|
|
86
|
-
- Chat-run PR watch -> `axstack-watch`: original driver; verified run PRs
|
|
87
|
-
|
|
84
|
+
- Chat-run PR watch -> `axstack-watch`: original driver; verified run PRs and
|
|
85
|
+
explicit adoptions only.
|
|
88
86
|
- Other own PR work -> `axstack-review` authored mode or `axstack-watch`
|
|
89
87
|
adoption.
|
|
90
88
|
|
|
@@ -65,7 +65,7 @@ user round, the driver independently drafts the prioritized frontier and
|
|
|
65
65
|
recommendations, except for an arena-grade question (below), where the driver
|
|
66
66
|
writes the brief and rubric but drafts no recommendation until the candidates
|
|
67
67
|
and judge verdicts return, so nothing anchors them. Then consult `axstack-advisor-astra` and
|
|
68
|
-
`axstack-advisor-
|
|
68
|
+
`axstack-advisor-opus` independently, without cross-reading, using the same
|
|
69
69
|
bounded evidence and question. Each adviser challenges assumptions, edges,
|
|
70
70
|
omissions, and alternatives; the driver synthesizes disagreements and accepts
|
|
71
71
|
or rejects each material point with a reason. Use one focused reply when
|
|
@@ -81,45 +81,53 @@ hold Align; safe fact work may continue without substitution.
|
|
|
81
81
|
|
|
82
82
|
## Arena for hard-to-reverse design choices
|
|
83
83
|
|
|
84
|
-
Critique of one draft anchors every reader to that draft's shape.
|
|
85
|
-
|
|
84
|
+
Critique of one draft anchors every reader to that draft's shape. Rung 2 designs
|
|
85
|
+
alone enter the arena: they meet the same test as for an ADR (a meaningful,
|
|
86
86
|
hard-to-reverse, non-obvious trade-off: architecture, module boundaries, data
|
|
87
|
-
model, migration strategy)
|
|
88
|
-
|
|
87
|
+
model, migration strategy). Replace the critique round for that question with
|
|
88
|
+
an arena. Small or routine questions never enter the arena.
|
|
89
89
|
|
|
90
90
|
1. **Frame.** The driver writes the brief (the artifact, its constraints, the
|
|
91
91
|
settled decisions it must respect) and three to six gradeable rubric
|
|
92
92
|
criteria. Candidates receive only the brief; the rubric is for judging.
|
|
93
|
-
2. **Fan out.**
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
3. **Cross-judge.** After
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
93
|
+
2. **Fan out.** Produce one candidate per configured family independently from the same brief,
|
|
94
|
+
without cross-reading: `axstack-advisor-astra`, `axstack-advisor-opus`,
|
|
95
|
+
`axstack-arena-candidate-grok`, and `axstack-arena-candidate-antigravity`.
|
|
96
|
+
Each gives a design, rationale, and rejected alternatives. The driver authors no candidate.
|
|
97
|
+
3. **Cross-judge.** After every candidate completes, give round 1 judge
|
|
98
|
+
`axstack-arena-judge-opus` the anonymized, relabeled candidates and rubric to
|
|
99
|
+
score every candidate per criterion and recommend a base with a reason.
|
|
100
|
+
The driver compares its own pick with the Opus verdict. Only if the driver
|
|
101
|
+
and the Opus judge disagree on the base, or the user rejects the round-1
|
|
102
|
+
synthesis,
|
|
103
|
+
run round 2 with fresh sessions: `axstack-escalation-fable` and `axstack-arena-judge-astra`
|
|
104
|
+
independently score the same anonymized candidates and rubric. Judges never
|
|
105
|
+
author, never cross-read each other, and never average verdicts. After round-2
|
|
106
|
+
verdicts return, the driver re-picks in step 4 and re-presents in step 6.
|
|
101
107
|
4. **Pick.** The driver reads every candidate end to end and scores per
|
|
102
|
-
criterion, not on holistic feel, then compares with
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
5. **Graft.** Walk the losing
|
|
108
|
+
criterion, not on holistic feel, then compares with the judge verdicts from
|
|
109
|
+
each completed round. Agreement confirms the base. On disagreement, re-read
|
|
110
|
+
the rationales and decide with a stated reason; never average verdicts or
|
|
111
|
+
fabricate consensus.
|
|
112
|
+
5. **Graft.** Walk the losing candidates once more for the one or two ideas
|
|
107
113
|
worth porting and fold them into the base by hand so the result stays
|
|
108
114
|
coherent under one mental model. Convergence on the same shape is a strong
|
|
109
115
|
agreement signal: adopt the consensus shape, no graft. Wide divergence
|
|
110
116
|
means the frame was under-specified: reframe and rerun once, never
|
|
111
117
|
average.
|
|
112
118
|
6. **Present.** The synthesized design is the recommendation in the next
|
|
113
|
-
`Qn`, with its trade-off, judge verdicts, and what was grafted or rejected.
|
|
119
|
+
`Qn`, with its trade-off, judge verdicts per round, and what was grafted or rejected.
|
|
114
120
|
The user still decides; spec approval remains the one human checkpoint.
|
|
115
121
|
|
|
116
122
|
Record the synthesis note (base, grafts and their source candidate, rejections,
|
|
117
|
-
dropouts,
|
|
123
|
+
dropouts, judge verdicts per round) as `Decisions` rows in the
|
|
118
124
|
[run record](../axstack/references/run-record.md). Load
|
|
119
125
|
[Orca runtime](../axstack/references/orca-runtime.md) immediately before the
|
|
120
|
-
first candidate or judge dispatch. If
|
|
121
|
-
|
|
122
|
-
and
|
|
126
|
+
first candidate or judge dispatch. If any configured candidate or judge seat
|
|
127
|
+
required for that round is unavailable at launch or returns a failed receipt,
|
|
128
|
+
hold that question without substitution, record the gap, and ask: the user decides whether to proceed without it.
|
|
129
|
+
For an uncertain dispatch, reconcile natively; it is never treated as absent.
|
|
130
|
+
Unaffected fact work and questions continue.
|
|
123
131
|
|
|
124
132
|
## Bound the interview
|
|
125
133
|
|
|
@@ -47,7 +47,8 @@ Use actual records, never memory:
|
|
|
47
47
|
|
|
48
48
|
- the approved spec, or the accepted peer, research, or maintenance scope;
|
|
49
49
|
- the decision log and configured `axstack-advisor-astra` and
|
|
50
|
-
`axstack-advisor-
|
|
50
|
+
`axstack-advisor-opus` receipts and fresh `axstack-escalation-fable`
|
|
51
|
+
receipts when triggered;
|
|
51
52
|
- exact git revisions;
|
|
52
53
|
- test and review evidence; and
|
|
53
54
|
- the run execution record at its recorded `progress.md` path.
|
|
@@ -71,7 +72,10 @@ counts with denominators plus the evidence behind the count:
|
|
|
71
72
|
- Planned steps completed and deviated, each deviation with why and approval.
|
|
72
73
|
- Configured adviser coverage across Align, Spec creation, Spec revision,
|
|
73
74
|
solution design, and unresolved consequential decisions, including
|
|
74
|
-
independent same-question receipts
|
|
75
|
+
independent same-question receipts and disagreement synthesis. Measure
|
|
76
|
+
Astra/Opus routine coverage and Astra/Fable high-stakes plain AGREE coverage
|
|
77
|
+
over eligible uses; record escalation triggers, evidence pointers, and whether
|
|
78
|
+
the outcome changed.
|
|
75
79
|
- Debug evidence where `axstack-debug` ran: rung reached, loop command, fix
|
|
76
80
|
attempts with why each failed, adviser and investigator receipts, and
|
|
77
81
|
isolation evidence (pinned worktree, preserved probe artifacts).
|
|
@@ -95,7 +99,7 @@ counts with denominators plus the evidence behind the count:
|
|
|
95
99
|
record. Record `UNKNOWN` when a receipt lacks the measurement.
|
|
96
100
|
This is evidence, not a score to game. Audit treats routine shape choices as
|
|
97
101
|
autonomous driver decisions; size alone never requires user approval.
|
|
98
|
-
-
|
|
102
|
+
- API-dollar cost by model when provider receipts are available, UNKNOWN otherwise.
|
|
99
103
|
|
|
100
104
|
Record `UNKNOWN` where evidence is absent. Never count missing evidence as a
|
|
101
105
|
pass or collapse gaps into a vanity score. Prefer parallelism for genuinely
|
|
@@ -9,7 +9,8 @@ Run: <run identity + scope/authority + audit mode (end-of-run | checkpoint)>
|
|
|
9
9
|
Baseline: <approved spec rev | peer mode | research mode | maintenance scope>
|
|
10
10
|
Acceptance: <passed / failed / unverified + test + SHA traces>
|
|
11
11
|
Steps: <completed / deviated + why + approval per deviation>
|
|
12
|
-
Advisers: <Astra/
|
|
12
|
+
Advisers: <Astra/Opus configured coverage / eligible uses + same-question receipts; Astra/Fable high-stakes AGREE coverage / eligible uses>
|
|
13
|
+
Decisions: <escalation trigger + evidence pointers + outcome changed yes/no, or n/a>
|
|
13
14
|
Debug: <rung reached + loop command + fix attempts + adviser and investigator receipts + isolation evidence | n/a>
|
|
14
15
|
TDD: <applicable evidence path: normal real red-green | accepted structure-preserving old revision green before edits + same checks new revision green; absent proof: noncompliance | unavailable records: UNKNOWN with reason>
|
|
15
16
|
Review: <exact-rev independent review status + unresolved findings>
|
|
@@ -17,7 +18,7 @@ Rework: <cycles + causes>
|
|
|
17
18
|
Interventions: <avoidable user interventions, or unsupported by records>
|
|
18
19
|
Parallelism: <identified vs dispatched + dependency/writer isolation>
|
|
19
20
|
Shape: <PRs within band / total PRs + rationale-band cohesion rationale + exception-band full driver exception record; missing measurement: UNKNOWN>
|
|
20
|
-
Cost: <model
|
|
21
|
+
Cost: <API dollars by model when measured, or UNKNOWN with reason>
|
|
21
22
|
Judgment: <execution outcome vs procedural adherence vs measurement coverage>
|
|
22
23
|
Proposals: <bounded hypothesized changes with regression-first plan, or none>
|
|
23
24
|
Learning candidates: <each candidate's statement + scope + evidence/revision pointers + target instruction surfaces + contradiction/uncertainty + disposition; explicit already-covered no-op or none>
|
|
@@ -91,7 +91,7 @@ before any further probe or attempt.
|
|
|
91
91
|
| L0 | entry | driver alone | phases 1–7; at most one fix attempt through `axstack-implement` |
|
|
92
92
|
| L1 | the L0 fix attempt failed; or phases 1–3 complete plus at least one discriminating probe (or a recorded reason no safe probe exists) and no hypothesis ranks | the preset's configured advisers, independently, same evidence packet | ranked hypotheses and one investigator brief per hypothesis |
|
|
93
93
|
| L1 fan-out | driver merges the adviser plans | `axstack-debug-investigator-1..4`, identical packet, distinct briefs, no cross-reading | one receipt per brief |
|
|
94
|
-
| L2 | the
|
|
94
|
+
| L2 | two failed attempts at the same named goal and stated acceptance check (second failure overall); or each fix reveals a new symptom elsewhere | `axstack-advisor-astra` and `axstack-escalation-fable` (fresh session) on architecture, then the user | wrong-architecture finding, bounded refactor proposal, or one bounded next diagnostic action; the user decides before any third attempt |
|
|
95
95
|
|
|
96
96
|
L1 is inadmissible without a red loop. Ordinary diagnosis is not a high-stakes
|
|
97
97
|
decision: in `mixed` both advisers are consulted and both receipts are
|
|
@@ -100,8 +100,9 @@ records the other as an intentional absence. Whenever a run exposes a
|
|
|
100
100
|
high-stakes architecture choice, serious security, downtime, or data-loss
|
|
101
101
|
risk, the standing high-stakes and serious-risk contracts override this rule,
|
|
102
102
|
including the single-provider high-stakes hold. An adviser configured but
|
|
103
|
-
unavailable at launch holds L1
|
|
104
|
-
|
|
103
|
+
unavailable at launch holds L1 without substitution; a null or unavailable L2
|
|
104
|
+
seat holds L2 without substitution. L0 continues.
|
|
105
|
+
Reuse an L1 adviser receipt while the packet is unchanged; a changed packet needs
|
|
105
106
|
a fresh receipt.
|
|
106
107
|
|
|
107
108
|
**Plan merge.** The driver de-duplicates the advisers' hypotheses, ranks the
|
|
@@ -207,7 +207,7 @@ This section applies to peer and authored PR modes.
|
|
|
207
207
|
| Preset | Actual author provider/model | Reviewer role (configured model/effort) |
|
|
208
208
|
| --- | --- | --- |
|
|
209
209
|
| `mixed` | Codex / Sol (`codex/gpt-6-sol`) | `axstack-reviewer-secondary` (`claude/claude-opus-5-5` medium) |
|
|
210
|
-
| `mixed` | Claude / Opus (`claude/claude-opus-5-5`) | `axstack-reviewer-primary` (`codex/gpt-6-sol`
|
|
210
|
+
| `mixed` | Claude / Opus (`claude/claude-opus-5-5`) | `axstack-reviewer-primary` (`codex/gpt-6-sol` high) |
|
|
211
211
|
| `codex-only` | Codex / Sol (`codex/gpt-6-sol`) | `axstack-reviewer-secondary` (`codex/gpt-6-luna` xhigh) |
|
|
212
212
|
| `claude-only` | Claude / Opus (`claude/claude-opus-5-5`) | `axstack-reviewer-secondary` (`claude/claude-sonnet-5` xhigh) |
|
|
213
213
|
|
|
@@ -41,7 +41,7 @@ and the lifecycle's [audit skill](../axstack-audit/SKILL.md) hook.
|
|
|
41
41
|
acceptance. First record the driver's
|
|
42
42
|
independent assessment, then load
|
|
43
43
|
[Orca runtime](../axstack/references/orca-runtime.md) before dispatching the
|
|
44
|
-
configured `axstack-advisor-astra` and `axstack-advisor-
|
|
44
|
+
configured `axstack-advisor-astra` and `axstack-advisor-opus` independently,
|
|
45
45
|
without cross-reading, with the same bounded evidence and question. The
|
|
46
46
|
driver synthesizes disagreements. Cache both receipts with the draft and
|
|
47
47
|
reuse unchanged receipts only while their evidence, scope, and question
|
|
@@ -50,7 +50,8 @@ and the lifecycle's [audit skill](../axstack-audit/SKILL.md) hook.
|
|
|
50
50
|
criteria, exclusions, and both adviser receipts or the reported hold.
|
|
51
51
|
4. **Obtain the specification checkpoint.** The driver owns the draft and the
|
|
52
52
|
user approves it; adviser input cannot grant approval. High-stakes decisions
|
|
53
|
-
require
|
|
53
|
+
require `axstack-advisor-astra` and a fresh `axstack-escalation-fable`
|
|
54
|
+
session to return plain AGREE. Present one
|
|
54
55
|
reviewable, identified revision for this checkpoint. Its user approval
|
|
55
56
|
creates the execution baseline.
|
|
56
57
|
5. **Snapshot the baseline.** Record the approved revision identity and a
|
package/src/installer.js
CHANGED
|
@@ -626,6 +626,20 @@ export async function installBundle({
|
|
|
626
626
|
if (rel in ownedFiles) installedHashes[rel] = ownedFiles[rel];
|
|
627
627
|
}
|
|
628
628
|
}
|
|
629
|
+
const previousRoles = desired.find(({ rel }) => rel === 'axstack/roles.json')?.current;
|
|
630
|
+
if (previousRoles && summary.updated.some((rel) => rel.startsWith('axstack/roles.json'))) {
|
|
631
|
+
try {
|
|
632
|
+
const old = JSON.parse(previousRoles.toString());
|
|
633
|
+
if (Array.isArray(old.roles)) {
|
|
634
|
+
const oldIds = new Set(old.roles.map((role) => role?.id));
|
|
635
|
+
const newIds = new Set(bundle.bundleRoles.map((role) => role.id));
|
|
636
|
+
const addedRoleIds = bundle.bundleRoles.filter((role) => !oldIds.has(role.id)).map((role) => role.id);
|
|
637
|
+
const removedRoleIds = old.roles.filter((role) => role?.id && !newIds.has(role.id)).map((role) => role.id);
|
|
638
|
+
summary.addedRoleIds = addedRoleIds;
|
|
639
|
+
summary.removedRoleIds = removedRoleIds;
|
|
640
|
+
}
|
|
641
|
+
} catch { /* No reliable role-ID diff for malformed prior bytes. */ }
|
|
642
|
+
}
|
|
629
643
|
|
|
630
644
|
// Stale manifest entries (owned files the bundle no longer ships):
|
|
631
645
|
// the phase-1 plan validated every stale destination read-only
|
package/src/roles.js
CHANGED
|
@@ -7,7 +7,7 @@ const PROVIDER_BOUNDS = Object.freeze({
|
|
|
7
7
|
const AUTHORED_ROUTES = Object.freeze({
|
|
8
8
|
mixed: {
|
|
9
9
|
'codex/gpt-6-sol': ['axstack-reviewer-secondary', 'claude/claude-opus-5-5', 'medium'],
|
|
10
|
-
'claude/claude-opus-5-5': ['axstack-reviewer-primary', 'codex/gpt-6-sol', '
|
|
10
|
+
'claude/claude-opus-5-5': ['axstack-reviewer-primary', 'codex/gpt-6-sol', 'high'],
|
|
11
11
|
},
|
|
12
12
|
'codex-only': {
|
|
13
13
|
'codex/gpt-6-sol': ['axstack-reviewer-secondary', 'codex/gpt-6-luna', 'xhigh'],
|
|
@@ -66,8 +66,10 @@ export function assessRoleReadiness(roles, preset) {
|
|
|
66
66
|
(preset === 'mixed' && role.id === 'axstack-checker' && role.provider === 'antigravity') ||
|
|
67
67
|
(preset === 'mixed' && role.id === 'axstack-research-web-google' && role.provider === 'antigravity') ||
|
|
68
68
|
(preset === 'mixed' && role.id === 'axstack-research-x' && role.provider === 'grok') ||
|
|
69
|
-
(preset === '
|
|
70
|
-
(preset === '
|
|
69
|
+
(preset === 'mixed' && role.id === 'axstack-arena-candidate-grok' && role.provider === 'grok') ||
|
|
70
|
+
(preset === 'mixed' && role.id === 'axstack-arena-candidate-antigravity' && role.provider === 'antigravity') ||
|
|
71
|
+
(preset === 'codex-only' && ['axstack-advisor-opus', 'axstack-escalation-fable', 'axstack-arena-judge-opus', 'axstack-research-web-google', 'axstack-research-x', 'axstack-arena-candidate-grok', 'axstack-arena-candidate-antigravity'].includes(role.id)) ||
|
|
72
|
+
(preset === 'claude-only' && ['axstack-advisor-astra', 'axstack-arena-judge-astra', 'axstack-research-web-google', 'axstack-research-x', 'axstack-arena-candidate-grok', 'axstack-arena-candidate-antigravity'].includes(role.id))
|
|
71
73
|
);
|
|
72
74
|
for (const role of roles) {
|
|
73
75
|
if (!bounds.has(role.provider)) {
|